Compare commits
134
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b9c97ffac9 | ||
|
|
b9ebc872ad | ||
|
|
c57daaf861 | ||
|
|
43699b46db | ||
|
|
854320ac3f | ||
|
|
ad4d256d8f | ||
|
|
3ed8bd7be9 | ||
|
|
96aa8176cd | ||
|
|
5cbfa893a5 | ||
|
|
2961beb8fe | ||
|
|
b14bacbfc6 | ||
|
|
6e47730501 | ||
|
|
ac403b9cd3 | ||
|
|
7791ed74f2 | ||
|
|
6abce99b49 | ||
|
|
b76d0acec7 | ||
|
|
580032056e | ||
|
|
2755e41ff3 | ||
|
|
d5e623c7ba | ||
|
|
aa82321907 | ||
|
|
62e8c87d0e | ||
|
|
62c5a8a408 | ||
|
|
5f9380dfca | ||
|
|
aa5abfa911 | ||
|
|
c2fe6a6bf4 | ||
|
|
f6048f268b | ||
|
|
2d04c448c2 | ||
|
|
c59b38775b | ||
|
|
f5a76cf88a | ||
|
|
1184d6b787 | ||
|
|
811f2b8898 | ||
|
|
4c842f5d70 | ||
|
|
b4ef42d947 | ||
|
|
ecd42ed484 | ||
|
|
c9af5481e7 | ||
|
|
a2ff2a1102 | ||
|
|
4d4cdd6ea7 | ||
|
|
70c988e702 | ||
|
|
0790f8dfd3 | ||
|
|
9c3a1c5be1 | ||
|
|
cbb0f11288 | ||
|
|
9dad61f508 | ||
|
|
971ae01caf | ||
|
|
397a400d57 | ||
|
|
2c6739ad76 | ||
|
|
cec9a98305 | ||
|
|
de7fb2c936 | ||
|
|
01988305a8 | ||
|
|
328e570309 | ||
|
|
cf5a790ea8 | ||
|
|
4d3c85fd06 | ||
|
|
28fe7c43a9 | ||
|
|
72553cb414 | ||
|
|
20a95da487 | ||
|
|
c7e585e21d | ||
|
|
5fa8b7412e | ||
|
|
b8e554dac7 | ||
|
|
765a8923a4 | ||
|
|
fa0e8d7d97 | ||
|
|
13d64e0000 | ||
|
|
a9b275abbb | ||
|
|
92c06ac8dc | ||
|
|
168a37542b | ||
|
|
edd9d63f5e | ||
|
|
43df08b52a | ||
|
|
94f71eea19 | ||
|
|
17ede3c6aa | ||
|
|
ae6e9256c6 | ||
|
|
abb5910d2f | ||
|
|
5450ec786f | ||
|
|
24a6ab3d1e | ||
|
|
ac3a557566 | ||
|
|
508a1c02da | ||
|
|
55d515d41f | ||
|
|
f6dbfd3625 | ||
|
|
f378953982 | ||
|
|
daf760220b | ||
|
|
a31eca65c3 | ||
|
|
2f90851c03 | ||
|
|
0a36b3fda9 | ||
|
|
72c4aa3895 | ||
|
|
4933c075b0 | ||
|
|
11ac4f50e6 | ||
|
|
35d93d7612 | ||
|
|
97a64c8a33 | ||
|
|
2010961d32 | ||
|
|
f21aef3cfa | ||
|
|
4298cd5de1 | ||
|
|
2bd25be712 | ||
|
|
bb9798e32c | ||
|
|
3ffa3f5318 | ||
|
|
58535890c4 | ||
|
|
ba9d98f7ce | ||
|
|
6907961ce0 | ||
|
|
d0b1f9694e | ||
|
|
1918da29be | ||
|
|
fbb6b0c180 | ||
|
|
ffe5dc14a8 | ||
|
|
d4bb8d344b | ||
|
|
ed722d55f8 | ||
|
|
311b1a7ec4 | ||
|
|
36b954d347 | ||
|
|
30857df5b2 | ||
|
|
faa508e87a | ||
|
|
82b5a606f7 | ||
|
|
56f3abdb36 | ||
|
|
089d4f3a80 | ||
|
|
f650bf892a | ||
|
|
1c89a5eeeb | ||
|
|
c04a3f083e | ||
|
|
72f0b4a258 | ||
|
|
8e7c7bbf24 | ||
|
|
1d0ec61c9d | ||
|
|
bb68fefe04 | ||
|
|
d829267f1c | ||
|
|
e0d23780d8 | ||
|
|
d1ec40f738 | ||
|
|
b6ef27cd2d | ||
|
|
2a55a0d265 | ||
|
|
d2c656533e | ||
|
|
02fd2de502 | ||
|
|
f79e5ebb5e | ||
|
|
0c8e29b05a | ||
|
|
edefc34a5b | ||
|
|
87a9f4eb25 | ||
|
|
0a2d654e68 | ||
|
|
fe310743a2 | ||
|
|
2b87a5a13b | ||
|
|
fd33fd05e1 | ||
|
|
ff7c57cf9c | ||
|
|
a2df2f242b | ||
|
|
2a8f897e61 | ||
|
|
abce381faa | ||
|
|
a415246adc |
No files matched your search
@@ -31,3 +31,14 @@ Dockerfile
|
||||
*.key
|
||||
felis
|
||||
felis.exe
|
||||
|
||||
# macOS materializes extended attributes as ._<name> sidecars (BSD tar uploads,
|
||||
# Finder copies, network volumes) and leaves .DS_Store behind. Neither is
|
||||
# source, and one is actively harmful: a ._*.sql beside the migrations is
|
||||
# //go:embed-ed into the binary and makes every `felis migrate` fail
|
||||
# ("non-numeric version") — observed live on a Mac-staged tree. Same exposure
|
||||
# for any tree the other //go:embed patterns walk (deploy/, plugins/).
|
||||
._*
|
||||
**/._*
|
||||
.DS_Store
|
||||
**/.DS_Store
|
||||
@@ -98,3 +98,42 @@ jobs:
|
||||
|
||||
- run: npm run typecheck
|
||||
working-directory: panel
|
||||
|
||||
plugins:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
# The other jobs never touch the Java layer: the plugin jars were only ever
|
||||
# compiled by bootstrap on a live host, and the three test mains under
|
||||
# plugins/*/test were run by hand. JDK 21 plus the Gradle major the plugin
|
||||
# Dockerfiles pin (8.14) is that same toolchain, in CI.
|
||||
- uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '21'
|
||||
|
||||
- uses: gradle/actions/setup-gradle@v4
|
||||
with:
|
||||
gradle-version: '8.14'
|
||||
|
||||
- run: bash plugins/test.sh
|
||||
|
||||
mods:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
# The three loader mods (Minecraft 1.20.1 / 1.20.4, Java-17 lines) compile
|
||||
# through their vendored Gradle wrappers, which fetch their own Gradle. Until
|
||||
# this job nothing ever built them: no install path touches them, and their
|
||||
# gradlew scripts were committed without the exec bit, so the README's
|
||||
# one-liners failed on a fresh clone.
|
||||
- uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '17'
|
||||
|
||||
- uses: gradle/actions/setup-gradle@v4
|
||||
|
||||
- run: bash plugins/test-mods.sh
|
||||
@@ -40,6 +40,31 @@ jobs:
|
||||
- run: go vet ./...
|
||||
- run: go test ./...
|
||||
|
||||
# The same reason, for the Java layer the binary EMBEDS: the release asset is
|
||||
# the tree's plugin sources (bootstrap_asset.go), and a tag whose plugins don't
|
||||
# compile turns every install of that release into a failed bootstrap. JDK 21
|
||||
# gates the install-time plugins + codec/invite tests; JDK 17 gates the loader
|
||||
# mods (their vendored wrappers fetch their own Gradle).
|
||||
- uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '21'
|
||||
|
||||
- uses: gradle/actions/setup-gradle@v4
|
||||
with:
|
||||
gradle-version: '8.14'
|
||||
|
||||
- run: bash plugins/test.sh
|
||||
|
||||
- uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '17'
|
||||
|
||||
- uses: gradle/actions/setup-gradle@v4
|
||||
|
||||
- run: bash plugins/test-mods.sh
|
||||
|
||||
# Both architectures, because bootstrap's default release channel DOWNLOADS these
|
||||
# rather than compiling on the target host — an arm64 host with no asset silently
|
||||
# falls back to a slow source build. Neither stage is emulated: the Dockerfile pins
|
||||
|
||||
+934
-15
@@ -1,6 +1,6 @@
|
||||
# Felis 生产就绪审计 — 2026-09-22(真机 E2E + 混沌)
|
||||
|
||||
分支:`audit-fixes-20260922`(已推送)。环境:CentOS Stream 9 / aarch64 / k3s v1.36.4,
|
||||
分支:全部修复**逐 commit 直接推 `main`**(不再用功能分支)。remote = `[email protected]:FelisMC/Felis.git`(2026-09-23 起;旧 `MliroLirrorsIngenuity/Felis` 仅靠 GitHub 301 兜住)。环境:CentOS Stream 9 / aarch64 / k3s v1.36.4,
|
||||
IPv6-only 接入(`ssh -6 -i ~/.ssh/id_ed25519 root@fdb2:2c26:f4e4:0:21c:42ff:fede:69ec`),
|
||||
面板经 `socat TCP6:443 → 127.0.0.1:30443` 中继(手动启动,重启 VM 后需重开)。
|
||||
|
||||
@@ -17,21 +17,120 @@ IPv6-only 接入(`ssh -6 -i ~/.ssh/id_ed25519 root@fdb2:2c26:f4e4:0:21c:42ff:f
|
||||
| 17 | **`ListPendingOpLogins` PG 少列**:接口注释承诺「joined to its staff username」,handler 输出 `username`/`created_at`,fake 正确填充;PG SQL 未 JOIN 也未取 `created_at` → 真机 pending 列表 username 为空、created_at 为 `0001-01-01` | 真机 internal `/op-login/pending` 响应 | `pgrepo.go` SQL 改为 JOIN users + 取 created_at。提交见分支 |
|
||||
| 18 | **绑定码并发兑换 500**:`RedeemPlayerBindCode` 无 `FOR UPDATE`(同文件 `VerifyLinkCode` 有),且裸 INSERT。6 路并发同码兑换 → 3×HTTP 500(`users_username_key` 唯一冲突)+ 1×400 + 2×200;跨码并发同 UUID 同样会撞 | 真机四组并发测试 + API 日志 `unmapped error ... duplicate key` | 同码:码行 `FOR UPDATE`(输家干净地 400 invalid_code);跨码:`INSERT users ... ON CONFLICT (username) DO NOTHING`+重读、`account_links ON CONFLICT (mc_uuid) DO NOTHING`(两路汇聚同一 user)。复测:同码×6=1×200+5×400、双码×2=2×200 同 ID、DB 干净、日志零 unmapped |
|
||||
| 19 | **reaper CronJob 渲染位置错误,永远无法调度**:`felis manifests` 把 CronJob 渲染到控制 ns,却引用 minecraft ns 的备份 PVC(Pod 不能跨 ns 挂 PVC:真机 `FailedScheduling: persistentvolumeclaim "felis-backups" not found`);修正 ns 后又发现 ServerAccount 也不能跨 ns 使用(`serviceaccount "felis-reaper" not found`)。而 reaper 的 Role/RoleBinding 本就在 minecraft ns | 真机三层取证(PVC/SA/调度)| CronJob 与 SA、RoleBinding subject 全部移到 `MinecraftNamespace`(提交 `e4f2cff`+`c839454`)。**修复后完整演练通过**(见下) |
|
||||
| 6 | **默认安装无备份能力 + 回收链路不可达/不可用**:(a) 没有任何环节渲染归档 PVC,`FELIS_BACKUP_PVC` 永远为空 → backup/restore 恒 503;(b) bootstrap 从不下传 reaper 三旗标 → 官方安装路径根本无法启用回收;(c) 即便启用,stock k3s 的 `<root>/<pvc>` 布局不存在,且 k3s storage root 是 `0700 root:root`、reaper pod 以 uid 1000 运行 → 真机 `lstat /worlds/…: permission denied`(fail-closed 跳过,但纯空转);(d) README 口径(“定时备份”)不实 | 真机全链路:默认渲染 PVC+env → 建服→marker→备份 202→jobs 端点 running→succeeded→改 marker→restore→读回原值 ✅;再建 resolvecheck 世界(20d idle,marker)→ CronJob 手动 Job:`evaluated=2 reaped=1`,归档含 marker、PVC+宿主目录回收、`world_backups` 得 `inactive_15d` 行、servers 行/CR 保留 ✅ | `platform/workloads.go` 渲染归档 PVC(minecraft ns、RWO 10Gi、默认 SC),`felis manifests --backup-pvc` 默认 `felis-backups`(`=` 空为显式关闭),bootstrap 统一下传 env/旗标并写 [archive] local_path;`resolveWorldDir` 新增精确 local-path 目录解析(读 PVC `spec.volumeName`,非 glob,杜绝陈旧 PV 目录顶替);ReaperRole 增 `pvc:get`;bootstrap 设 `FELIS_WORLDS_HOST_PATH` 时给 uid 1000 授 traverse(setfacl/o+x);README/故障手册改写。提交 `fd33fd0`+`2b87a5a` |
|
||||
| 8 | **磁盘打满灾难链**:DiskPressure → kubelet 驱逐控制面(无 PriorityClass 保护)→ 镜像被 GC(无外网、registry 空)→ 全部 ImagePullBackOff;释放后数分钟才恢复调度。恢复依赖人工 `docker save | k3s ctr images import -` | 三轮真机 drill(原缺陷复现 + 两轮修复验证):①自定义 1e6 类:游戏 pod 先走、控制面随后仍被驱逐(kubelet 日志逐条列出 ranked/evicted),随后镜像被 GC → ErrImagePull;②内建 `system-cluster-critical`(2e9):填盘至 1.7G free,kubelet 对 felis-api/operator/registry 全部报 **“cannot evict a critical pod”**,三者在整个 DiskPressure 期间保持 Running;游戏 pod(0)被驱逐;③释放磁盘后:`DiskPressure` 经 ~5 分钟(`eviction-pressure-transition-period`)转 False——即“恢复慢”的主因;被 GC 的游戏镜像按 runbook 重导入后 25s 恢复 ✅ | 控制面四类 pod 挂内建 `system-cluster-critical`(用户自定义类值上限 1e9,达不到 2e9 临界阈值;preemption 保留=管理面可调度,已记录权衡);troubleshooting §13b 固化事件链、5 分钟条件过渡与镜像恢复路径。提交见下 |
|
||||
| 9 | **升级策略 Recreate**:单副本 + Recreate:任何控制面升级=停机;坏升级(错 tag)中断约 95s 且需人工 `rollout undo`(无自动回滚) | 复核部署模板(Recreate + 单副本 + 无 leader election 的注释理由成立) | 不改策略(Recreate 是对无 leader election 的正确取舍),改为固化 runbook:troubleshooting §15 = 升级即重跑安装器;停机窗口=rollout 时长;坏镜像在安装器 180s 等待内以 `kubectl describe` 诊断呈现;回滚 `kubectl rollout undo`(镜像被 GC 时先按 §13b 重导入)。提交见下 |
|
||||
| 10 | **备份语义**:归档包含整个 /data(jar、libraries、cache),167MB;是否符合 world backup 定位待评估 | 真机 tar 清单(cache/、config/、eula.txt、server.properties、jar)+ 恢复语义(overlay+prune) | 评估结论:**整卷备份是正确语义**(回滚=整服状态回滚,含配置/插件),保留行为;文档化:troubleshooting §10「backup contains」+ OpenAPI/README 口径改为“整服数据卷”。提交见下 |
|
||||
| 20 | **验证邮箱唯一性只存在于注释里**:errors.go/repo.go 都声称迁移 0010 建了 `users_verified_email_unique` 部分唯一索引、`VerifyEmailOTP` 会返回 `ErrEmailTaken`;实际上**索引从未创建**、`ErrEmailTaken` 全仓库从未被返回 → 两个账号可同时验证一个邮箱,而登录门正是按 verified email 解析账号 → 同一邮箱的登录码归谁由数据库任意决定 | pgint 首跑即红:第二账号验证同邮箱返回 `nil`;`grep users_verified_email_unique internal/store/migrations/*.sql` 零命中。真机复验(auditfix17 邮箱 409 演练):同码二次 verify 仍 409(码未被消费) | 新增迁移 `0020_verified_email_unique.sql`(`lower(email) WHERE email_verified`);`VerifyEmailOTP` 在消费码**前**查「他人已验证」→ `ErrEmailTaken`(不消费码、不扣次数),并把并发唯一冲突映射为同一答案;handler 新增 409 `email_taken`。提交 `b6ef27c`;PG 契约测试基建 `2a55a0d`(`-tags pgint`,见 CONTRIBUTING) |
|
||||
| 21 | **admin 改邮箱不清 verified**:`UpdateUser` 写入新地址但保留 `email_verified=true` → 面板改错一个字符就能让登录码寄到别人邮箱(该 flag 正是邮箱登录解析/寄码的依据);fake 同错 | pgint 契约测试 | 改地址时同写 `email_verified = email_verified AND email IS NOT DISTINCT FROM 新值`(同值 no-op 保留证明;换值即清除);fake 同步。提交 `d1ec40f`。真机验证:PATCH→`f`、PATCH 回原值仍 `f`、玩家重验证→`t` |
|
||||
| 22 | **`role='owner'` 是死信**:迁移 0011 加了 owner 角色并把全部用户管理路由压在 `IsOwner()` 上,但**没有任何代码写过 `owner`**——breakGlass(`UpsertOwner`)、setup MC 绑定(`CompleteOwnerSetup`)、重置路径一律写 `admin` → 全新安装的整个 owner 层(用户列表/创建/编辑/禁用/删除/配额/会话)不可达;且新角色没进各处 staff 判定(op-login 只认 admin、玩家邮箱门只拒 admin、reclaim 保护与 AdminExists 只认 admin);面板一旦有 owner 行还可被降级/删除/禁用 | 真机:owner 会话加载 `/api/v1/users` 403;promote 后 200。op-login:修复前 owner start 中立不发码,修复后铸请求+finish 得 `role=owner` 会话 | `UpsertOwner`/`CompleteOwnerSetup` 改写 `role='owner'`(冲突臂重断言,即 0011 文档的 promote 路径);op-login 双端改 `staffRole`;玩家邮箱门改 `role != 'user'`;`IsProtectedAdminLink`/`AdminExists` 计入 owner;面板新增守卫:owner 行不可降级/删除/禁用(用户名/邮箱编辑仍可)。提交 `e0d2378` |
|
||||
| 23 | **配额门非原子 + storage 缓存被清零**:(a) audit #4:`QuotaCheck` 与 `ClaimServer` 两条语句,同一用户并发认领两台无主服可双双通过 `max_servers`(deferred-seams 曾把这条挂为“只能在真 PG 上关闭”);(b) 更隐蔽:PATCH 资源时 `UpdateServerResources(..., 0)` 把本不能改的 storage 缓存写 0,而缓存列是配额聚合的**唯一**输入 → 此后该服的 storage 维度在配额里凭空消失 | (a) 新增 pgint 并发测试:修复前两台全赢;修复后恰 1 赢 + 1 `ErrQuotaExceeded`,DB 只 1 行 owned;(b) hermetic 测试 `TestPatchServerPreservesStorageCache`(修复前 `resourceUpdates` 里 storage=0) | (a) 门槛进 `ClaimServer`:同一事务内 `pg_advisory_xact_lock(hashtext(user_id))` + 四维重查(与 `QuotaCheck` 共用 `quotaAllows` 防漂移)+ 行 `FOR UPDATE`,两个 claim handler 把 `ErrQuotaExceeded` 映射为与串行一致的 403;(b) resize 前读取现值并透传 storage。提交 `bb68fef`;deferred-seams 对应条目核销 |
|
||||
| 24 | **reaper 警告信从不真正投递**:`felis reaper` 从不装配任何 Warner(`r.Warner` 恒 nil),而 `maybeWarn` 对 nil warner / 投递失败一律照样 `MarkWarned` + `warned++` → 每位有主的服都在**无人收到提醒**的情况下 15 天后被静默回收,跑批日志还谎报"已警告 N 台"。红线⑤的"best-effort 不阻塞回收"被误读成了"失败也要记成已通知" | hermetic 单测 3 例(坏 notifier / nil warner → 不盖章、成功才盖章)。真机 drill(auditfix20):13d idle 实收 1 封(收件人=owner 的验证邮箱、subject 含 3d);重跑不重发;14.5d 补发 1d(elif 档位);SMTP 端口打坏 → 日志 `warn delivery failed; will retry next run` 且**不盖章**,恢复后补发;邮箱未验证 → `has no verified email` 不盖章;owner NULL → 静默跳过;边界 11d23h 不发 / 12d1m 发;全程 `world_backups` 保持 4 条不动(纯警告零备份副作用) | `mail.SendNotice` + `mailWarner`:查 owner 的 verified email → `SendNotice`;nil/失败不盖章、下次重试;reaper pod 模板加 optional `FELIS_SMTP_PASSWORD` env;`felis setup` 复制 felis-smtp 镜像并在「configure email」刷新 minecraft ns 的 felis-smtp+felis-config 镜像;docs/troubleshooting §10 更新。提交 `8e7c7bb` |
|
||||
| 25 | **idle auto-stop 从未触发(三层复合缺陷)**:条件齐备的空载服永远不停。① CRD status schema 未声明 `emptySince`,apiserver **pruning** 掉计时戳(`unknown field "status.emptySince"`),计时器每次读回都是 nil;② 即使戳幸存,Running 空载稳态**没有任何 watch 事件**(玩家进出不碰 CRD、RCON 只在 Reconcile 内探),盖章一次后 reconcile 链停摆,无人叫醒;③ auto-stop 用整对象 `Update` 写 spec 且 OperatorRole 从未有 `minecraftservers:patch/update` → 403 `cannot update resource`。三层任一都让功能永久失效,而单测(fake client 不剪 schema、手动驱动、无 RBAC)全部覆盖不到 | 真机逐层实锤:schema 修复后 `empty=2026-09-22T20:02:13Z` 首次成功持久化;静置 88s+ 无动作、operator 日志 90s 零行(②实锤);修复前日志 5 条 403(③实锤);全修后场景 1:超时戳触发即 Stopped;场景 2:起服 → 20:11:43 盖章 → **全程无干预** → 20:12:17 自动 Stopped;翻转 Running→Stopped→Running 收敛 Running;`idle=null` 后不再自停 | ① schema 补 `emptySince`(`c04a3f0`);② `reconcileRunning` 尾部返回 RequeueAfter——空载=到点精确唤醒、有人=30s 探针周期,+3 个单测(`1c89a5e`);③ auto-stop 改 merge-patch(防 status clobber,与 reaper 的 Stop 同型)+ OperatorRole 补 `patch` + rbac 测试锚点(`f650bf8`)。真机 auditfix22 全通;troubleshooting §11 增自检三连 |
|
||||
| 26 | **RCON Secret 被删 → 服务器永久锁死**:删掉 per-server RCON 密码 Secret 后,(a) operator 没有任何 watch 能看到删除(Running 稳态零事件),secret 一直不重建;(b) 一旦被任意事件带到 reconcile:新密码铸造成功,但运行中的 pod 仍持有旧密码、sts 模板无变化 → pod 永不重启 → 探针用新密码连旧密码 pod,**永久 `RconNotReachable`** 直至 300s `ReadinessTimeout`;恢复只有人工删 pod。`ensureRconSecret` 的注释却声称"heals on the next pass" | 真机:删 secret 后静置 60s 无重建、无日志;annotate 触发后 96s+ 持续 RconNotReachable(新密码 `91dd62…` vs pod 旧密码);删 pod 后 17s 恢复 Running——根因=密码漂移实锤 | ① `SetupWithManager` 增 `Owns(&corev1.Secret{})`(controller-owned,删除事件映射回 CR)——真机日志见 `source="kind source: *v1.Secret"`;② pod template 新增 `RconSecretAnnotation` = 当前密码 SHA-256 指纹(64-bit):secret 重建指纹变 → sts 自动 rollout 用上新密码,未重建则恒定不抖动;测试 2 例(跨 reconcile 稳定 / 重建必变)。提交 `56f3abd`。真机复测:删 secret 后**同秒**察觉+重建+换戳,31s 全自动恢复 Running |
|
||||
| 27 | **陈旧启动锚点 → 已恢复的服务器被误判 StartupTimeout + Provisioned 永久 False**:`status.startRequestedAt` 只在 Stop 时清,成功(markRunningReady)不清——一旦服务器曾经历一次长 Starting,其 300s 预算就悬在健康运行中;此后任意一次 pod 波动(rollout/崩溃)都会立即套用旧戳判 `Failed(StartupTimeout)`。且 `markFailed` 写入的 `ConditionProvisioned=False` 无人复位,恢复后仍永久挂着失败标记 | 真机:20:19 已恢复 Running 的 test-one,20:21 因一次 stamp 注入引发的 pod rollout 被标记 `Failed StartupTimeout`(锚点残留自 20:16);恢复后 status.conditions 里 `Provisioned=False … StartupTimeout` 持续存在 | `markRunningReady`:清 `StartRequestedAt`(每次启动/恢复尝试各有独立预算)+ 复位 `ConditionProvisioned=True`;测试 2 例(Ready 清锚点、Failed→Running 后 Provisioned 恢复)。提交 `82b5a60`。真机复测:清戳生效(`startRequestedAt: None`)、pod blip 删→33s 恢复全程无 Failed、锚点重新盖章后成功清除 |
|
||||
| 28 | **cfsetup 把 "policy already exists" 当成功吞掉 → fail-closed 保证可被旧策略顶替**:Cloudflare Access 的 `CreateAccessPolicy` 在收到 already-exists 时直接返回 nil;若该 app 上已有一条更宽松的旧策略(改 identity 后重跑、或此前手工配置),守卫 op.console 的仍是旧策略,而 Setup 报告成功——`validateFailClosed` 只校验过"我们构建的策略",从未校验证留在线上的那条 | 代码审查(集成侧无真实 CF 账号,无法真机):httptest 3 例复刻——已存在同名策略、缺席、POST 竞态。修复前第 1 例吞错返回 nil 且不发 PUT | 改为按名 upsert:lookup → PUT 覆盖守卫体 → 缺席才 POST(POST 撞 already-exists → 重查后 PUT,绝不吞)。`apiPost/apiPut` 共用一个 `apiWrite`。提交 `30857df` |
|
||||
| 29 | **"configure email" 的工作负载镜像复制从不生效**:`replicateSMTPToWorkloadNamespace` 把硬编码 `namespace: felis` 的 felis-smtp manifest 用 `-n minecraft apply` 发出——kubectl 拒绝 namespace 冲突(`the namespace from the provided object ... does not match`)→ 第一段直接 return err,**第二段 felis-config 的镜像复制根本不会执行**。于是任何"装好后再改 SMTP"的部署,reaper 的 pre-reap 警告永远拿不到新配置(这正是 8e7c7bb 添加该刷新要解决的事),且失败仅 warning 不中止 | 真机 kubectl 行为实验:`-n minecraft apply` 带 `namespace: felis` 的 manifest → `error: … does not match …`;修复后(manifest namespace=minecraft)→ `accepted-namespace=minecraft`(server dry-run);felis-config 渲染+apply 序列同为 accepted | `smtpSecretManifest(password, namespace)` 显式参数(控制面调用传 "felis",镜像传工作负载 ns);回归测试 `TestSMTPSecretManifestCarriesTargetNamespace`。提交 `ed722d5` |
|
||||
|
||||
## 待决策台账(未修)
|
||||
|
||||
| # | 主题 | 说明 |
|
||||
前两日台账的 #7、#11–#15 已全部修复(本批核销,证据见下),不再挂在"未修"里:
|
||||
|
||||
| # | 主题 | 状态 |
|
||||
|---|------|------|
|
||||
| 6 | 默认安装无备份能力 | backup/restore 端点默认 503(需 `FELIS_BACKUP_PVC`+PVC),reaper CronJob 需 `--backup-pvc/--archive-local-path/--worlds-host-path` 三旗标渲染,bootstrap 一个都不传;文档无说明;README 与"自动备份"口径不符。另:归档 3 个月过期依赖 reaper 清理,未启用则磁盘只增不减 |
|
||||
| 7 | 异步失败不可感知 | backup/restore 失败后无状态出口:restore 行不变、backup 无行;只有集群侧 Job/日志可查。建议状态字段或 `?failed` 查询 |
|
||||
| 8 | 磁盘打满灾难链 | DiskPressure → kubelet 驱逐控制面(无 PriorityClass 保护)→ 镜像被 GC(无外网、registry 空)→ 全部 ImagePullBackOff;释放后约 8 分钟才恢复调度。恢复靠 `docker save felis:* | k3s ctr images import -`(docker 守护进程存储是唯一副本,需固化回源路径)。建议:PriorityClass、镜像入内置 registry、磁盘告警 |
|
||||
| 9 | 升级策略 Recreate | 单副本 + Recreate:任何控制面升级=停机;坏升级(实测错 tag)服务中断约 95s 且需人工 `rollout undo`(无自动回滚)。建议 runbook/文档化 |
|
||||
| 10 | 备份语义 | 归档包含整个 /data(jar、libraries、cache),167MB;是否符合"world backup"定位待评估 |
|
||||
| 11 | PG 断连表现 | 会话查询失败报 401 而非 503(fail-closed 但误导;用户以为没登录) |
|
||||
| 12 | ready 门滞后 | 容器 Ready 后 6~10s 内 API 仍 409 not_running |
|
||||
| 13 | setup token 截断 | 43 字符 token + 长域名,80 列终端下 TUI 截断显示(复现:tmux 80 列) |
|
||||
| 14 | 日志噪音 | controller-runtime 未 SetLogger,首用打印整段堆栈;TLS handshake EOF 噪音(kubelet 探针) |
|
||||
| 15 | reaper 启用未演练 | **已演练完毕(2026-09-22)**:绑定挂载演练 world → 一次 pass 归档+删 PVC+保留 servers 行/CRD;过期归档驱逐(行转 deleted、文件删除、合法归档未动)。启用仍受 #6 三旗标约束;另注意 `--worlds-host-path` 的 `<path>/<pvc>` 布局在 stock local-path 下不成立(编排责任),且 VM 上演练用的 CronJob 已 **suspend** 防误删(真实世界目录不在 /srv/worlds-root)|
|
||||
| 7 | 异步失败不可感知 | ✅ 修复:`GET /api/v1/servers/{name}/jobs`(`ff7c57c`,含 RBAC `jobs:list` 与 OpenAPI);真机 drill 用它看到 running→succeeded |
|
||||
| 11 | PG 断连报 401 | ✅ 修复:会话存储故障改为 503(`2a8f897`) |
|
||||
| 12 | ready 门滞后 | ✅ 修复:operator 启动期内每 2s 重探 RCON(`a2df2f2`) |
|
||||
| 13 | setup URL 截断 | ✅ 修复:TUI 折行(`abce381`) |
|
||||
| 14 | controller-runtime 日志噪音 | ✅ 修复:SetLogger 接 slog(`a415246`) |
|
||||
| 15 | reaper 启用未演练 | ✅ 已演练(见"第二日"节;CronJob 仍 `suspend=true` 防误删) |
|
||||
|
||||
## 可达性分级(#1–#79;#62 立案后剔除)——这些缺陷真实使用中到底谁能踩到
|
||||
|
||||
回应质疑"是不是全在测边界条件 / 只有内部 hook 才能触发":
|
||||
|
||||
- **hook 的边界**:演练中使用的内部手段(读服务端日志取 OTP 码、service token 打内部面代 approve、本地 SMTP sink、tmux 驱动 TUI)只是**测试仪表**——代替"真实收邮件 / 键盘敲击 / 操作台点击";触发路径本身是公开路径,真实用户做同一动作走同一段代码。真正**本环境不可复现**的只有 #28(无真实 CF 账号,仅 httptest 复刻),另有 #4/#5/#14 属工具/卫生级(无运行时触发面)。
|
||||
- **分级口径**:① 日常=普通使用或默认安装下即会踩到;② 运维=升级/重启/故障恢复/磁盘满/删资源/闲置回收等真实运维动作会踩到;③ 窗口=需要并发或特定状态时序,真实但概率低;④ 工具=仅工具判定/审查;决策=非缺陷(评估/演练项)。
|
||||
|
||||
| # | 真实触发路径(一句话) | 分级 |
|
||||
|---|------------------------|------|
|
||||
| 1 | 恢复失败后 10 分钟内重试(必然的恢复动作)→ 202 但 Job 不动 | ② |
|
||||
| 2 | 默认安装下第一次点「备份」→ Job FailedMount(felis-config 缺在 minecraft ns) | ① |
|
||||
| 3 | #1 修复所需 RBAC(单独不产生用户可见行为) | ② |
|
||||
| 4 | gofmt/staticcheck 漂移 + CI 门禁缺失 | ④ |
|
||||
| 5 | 依赖漏洞(govulncheck 判定可达):需特定输入 | ④ |
|
||||
| 6 | 默认安装即踩:备份/恢复恒 503、回收链无旗标不可启用 | ① |
|
||||
| 7 | 任何异步操作(备份/构建)想查结果:无端点(功能缺口) | ① |
|
||||
| 8 | 磁盘打满事故链(控制面被逐 → 镜像 GC → 全体 ImagePullBackOff) | ② |
|
||||
| 9 | 非缺陷:升级策略评估 → runbook | 决策 |
|
||||
| 10 | 非缺陷:备份语义评估 → 文档化 | 决策 |
|
||||
| 11 | PG 掉线期间任何会话校验 → 401 误导 | ② |
|
||||
| 12 | 每次起服:就绪感知滞后 | ① |
|
||||
| 13 | 首次安装向导:长 URL 截断 | ① |
|
||||
| 14 | operator 日志噪音(卫生项) | ④ |
|
||||
| 15 | 流程项:reaper 启用补演练 | 决策 |
|
||||
| 16 | 邮箱登录主路径:错码/重放 → 500、5 次锁定失效 | ① |
|
||||
| 17 | 面板 op-login pending 列表字段缺失 | ① |
|
||||
| 18 | 绑定码并发兑换(双击/双端即可)→ 部分 500 | ③ |
|
||||
| 19 | 按文档启用回收 → CronJob 永远无法调度 | ② |
|
||||
| 20 | 两账号验证同一邮箱 → 登录码归属不定(安全) | ③ |
|
||||
| 21 | 面板改任一用户邮箱 → verified 未清、码寄旧地址 | ① |
|
||||
| 22 | 全新安装 owner 层不可达(用户管理全路由) | ① |
|
||||
| 23 | (a) 并发认领两服可双双过配额;(b) PATCH 任意服资源 → storage 缓存清零 | ①(a=③) |
|
||||
| 24 | 有主服闲置 12–15d:警告信从不投递、静默回收 | ② |
|
||||
| 25 | 启用空载停机后等超时:三层复合缺陷永不生效 | ② |
|
||||
| 26 | 运维删/轮换 RCON Secret → 永久锁死至人工删 pod | ② |
|
||||
| 27 | 曾长启动的服 + 任意 pod 波动 → 误判 StartupTimeout | ② |
|
||||
| 28 | 重跑 cfsetup 且线上有旧策略 → fail-closed 被顶替;⚠无真实 CF 账号,仅 httptest | ④ |
|
||||
| 29 | 装后改 SMTP(configure email)→ 复制链从未生效 | ② |
|
||||
| 30 | 对不存在/刚被删的用户 id 操作 → 500 而非 404 | ③ |
|
||||
| 31 | 真实 owner 打开 /admin/builds 被 mock 残留误判 | ① |
|
||||
| 32 | 新浏览器首访 /admin/builds 播种两个假 404 | ①(轻) |
|
||||
| 33 | 禁用/已删账号重走邮箱登录门 → 可复活(安全) | ① |
|
||||
| 34 | 同族:死账号游戏内身份面仍存活 | ① |
|
||||
| 35 | 任何真实服务器保存过一次后:所有备份/fileread 必败 | ① |
|
||||
| 36 | 任何构建失败的提交者永远看不到结果 | ① |
|
||||
| 37 | 管理端 fleet:login/lobby 行全是死操作 | ①(轻) |
|
||||
| 38 | felis manifests 输出残留过期指导 | ②(文档) |
|
||||
| 39 | 断玻璃 reset owner 用非在位名 → 铸第二 owner 永不可清 | ② |
|
||||
| 40 | 断玻璃 add operator 撞名 → 裸 SQLSTATE | ② |
|
||||
| 41 | 断玻璃 Sync 选系统服 → 裸内部错误 | ② |
|
||||
| 42 | 对无世界盘服备份/恢复 → 202 后静默卡 30 分钟 | ② |
|
||||
| 43 | #42 配套:409 一刀切文案 | ②(轻) |
|
||||
| 44 | 重跑 `felis setup` 完成 s/c reconfigure → 状态框退回“Setup complete”(信息一致性) | ①(轻) |
|
||||
| 45 | 评审看不到将被执行的 recipe(执行的是上传 blob 里的 Dockerfile)→ 盲批 | ① |
|
||||
| 46 | 大层推送打死 registry(256Mi 模板被 OOM kill;实测 475MB 层)——用户构建/安装器入仓即触发 | ① |
|
||||
| 47 | Mac 打包树构建:`._*.sql` 旁文件混入 embed → `felis migrate` 全挂 | ② |
|
||||
| 48 | 安装器重跑:每镜像 start/stop docker 触发 systemd 限流 → 批量入仓中途断 | ② |
|
||||
| 49 | 安装器重跑:`[registry]`/`[archive]` 运维配置静默回退(构建/S3/reaper 行为回默认) | ② |
|
||||
| 50 | 配过 SMTP 的安装按文档重跑安装器 → 生成的注释块被 `[smtp]` carry 吞并并每次 +1(纯膨胀) | ② |
|
||||
| 51 | 改配置/轮换 DB 凭据后:工作负载 ns 的 `felis-config` 副本永不刷新 → backup/reaper 静默读旧配置 | ② |
|
||||
| 52 | 向导内 `s`/`c` 重配置后:工作负载 ns 的 `felis-config` 镜像停在上一次运行快照,直到下次 `felis setup`/安装器才收敛 | ② |
|
||||
| 53 | 默认安装或 `felis update` 的 felis-api 检查:坐标仍指向已迁移的旧仓库,现仅靠 GitHub 301 兜住——redirect 一退休,安装器默认 URL 与更新检查全挂 | ② |
|
||||
| 54 | 完成态安装上照 `felis update` 指引升级:`sudo felis setup` 只开配置控制台,任何组件都不会动(三处 run 指引 + trailer 全假) | ②(文档) |
|
||||
| 55 | 配置的源回 200 但档案字段不可用(马虎自建/三方 Yggdrasil):坏源在前 → 经梯子的登录被静默吞掉(零日志、后续源不被询问) | ② |
|
||||
| 56 | 给"还没进过服"的玩家加白名单/封禁/踢出(面板说成功、vanilla 实际拒绝)——任何新玩家第一次被管理就撞 | ① |
|
||||
| 57 | 控制台发任何命令后无反馈(回复被丢弃、pod 日志也不回显) | ① |
|
||||
| 58 | 对从未启动过的服务器点"文件"页(90s 卡死 → 误导性 504) | ① |
|
||||
| 59 | 装了 LuckPerms 的服打开 LP 页:读不到被报成"未分配任何权限"(假事实),且历史伪造 `[RCON]` 输出 | ① |
|
||||
| 60 | 新建服务器输入非法名/子域名(前端预检比后端弱)→ 对话框直出 Go 英文错误;账户验证撞已占邮箱同理 | ① |
|
||||
| 61 | 给"还没进过服"的玩家预授权限:面板提示"仍然会实际生效"(过度承诺),实际 LuckPerms 静默丢弃(未进服的名字无法解析) | ① |
|
||||
|
||||
统计:① 26 | ② 25 | ③ 3+#23(a) | ④ 4 | 决策 3 = 61。第三批构建链另有 3 处未编号修复(Job requests 超小节点上限 → 永远 Pending、Kaniko chown、Trivy DB egress 被锁)——均属 ①/②。
|
||||
|
||||
第四十批追加:#63(demo-up 单起点化——旧镜像臂能产出"起了但无处路由"的演示机;dev/demo 路径)②;#62 经立案复核**不成立**(全部安装路径都构建 felis-velocity.jar,证据见该批节),已剔除、不计入。
|
||||
|
||||
第四十一批追加:#64(三个 vendored `gradlew` 以 100644 提交、无可执行位——README 教的构建命令在全新 clone 上直接 `Permission denied`;无任何 CI/安装路径跑过这三个模块)①;#65(三个装载器 mod 的 `license` 仍写 MIT、Forge/NeoForge 的 `issueTrackerURL` 是 `example.invalid` 占位,与仓库 AGPL-3.0-only 相悖——fabric loader 启动时会打印该字段)②(低)。
|
||||
|
||||
第四十二批追加:#66(英文 README 缺中文版"使用方式"里的私有仓库安装 workaround 与重跑升级说明——英文读者照文档第一步即 404、无任何指引)①(轻)。
|
||||
|
||||
第四十三批追加:#67(build Job 从不设 `ttlSecondsAfterFinished`——每构建一次就永久留下一个完成 Job+Pod,完成 Pod 计入节点 pod 预算(stock k3s 110),构建量上来后新构建全 Pending)②(时间维度;随构建量从②滑向①)。
|
||||
|
||||
第四十三批追加(二):#68(构建 context 拉取单次尝试——控制面滚动/重启/驱逐恰好撞上构建窗口时,fetch initContainer 一次 `connection refused` 直接打成终态 `Failed`;`BackoffLimit=0` 无第二次 Pod,代价 = 人工重审重提)②(运维:升级/重启/故障恢复撞上正在进行的构建;构建量大时概率上升)。
|
||||
|
||||
第四十五批追加:#69(README 中英"审批通过后自动构建**并部署**"过度承诺——数据模型无目标服务器、部署实为"选用该镜像";① 轻);#70(kaniko 拉内建底座默认 HTTPS——`--insecure` 只覆盖推,任何 `FROM registry.felis.svc:5000/…` 构建必败 ①);#71(kaniko 以 drop-ALL 解包底座层,chown 必败——任何非 scratch 底座必败 ①);#72(trivy 扫 jar 必拉 Java DB、被 egress 锁拒绝——含 jar 即所有真实模组包的构建必败 ①);#73(bootstrap 测试在跑 k3s 的主机上必假失败 ④ 工具);#74(console 断连测试读写竞态 flaky ④ 工具)。
|
||||
|
||||
第四十六批追加:#75(提交上传面无上限——pending 无个数上限、无存储预算、create/upload 无节流;一个账号可无限堆积上下文刷爆 uploads PVC ①);#76(提交无撤回/管理员删除路径——提交者无法自救、运维无法回收占用 ①);#77(未完成引导的会话触发受保护操作 → 面板显示"无权执行此操作"而非送往 /setup;tracker #8 ①);#78(create-if-absent 使新增 CR 字段在已装机永不落地——lobby 无 RCON 故控制台死、玩家数恒 0;tracker #1 ②);#79(troubleshooting [INERT] 图例指向已不存在字段 ④ 文档)。
|
||||
|
||||
## 已验证事实(正向清单)
|
||||
|
||||
@@ -46,6 +145,807 @@ IPv6-only 接入(`ssh -6 -i ~/.ssh/id_ed25519 root@fdb2:2c26:f4e4:0:21c:42ff:f
|
||||
- op-login 全状态机(start→status→approve→finish;早 finish 不烧码、错码扣次且请求保留、重放/重复批准/非管理员批准/未知 handle 全部按契约返回)✅
|
||||
- 并发:OTP 风暴 8 路 = 1×202 + 7×429 且仅铸 1 码;绑定码并发(同码/双码)见 #18 ✅
|
||||
- **reaper 全链路真机演练**(修复 #19 后):绑定挂载假世界 → `felis reaper` 一 pass:归档 tar 落盘(内容含标记文件)、`world_backups` 行 `inactive_15d`+90d 过期、**PVC 删除且宿主目录回收**、servers 行保留且 activity 时钟重置(红线②)、CRD 保留;第二 pass:伪造过期归档被驱逐(`expired=1`,文件删、行转 `deleted`),合法归档未动;`evaluated=2 reaped=1` 只动到期的世界 ✅
|
||||
- **S3 存储向导全链**(第三十二批):错误凭据预检零副作用、正例落地(Secret + 两 toml + API 滚 + 状态行)、上传 → MinIO 桶 → 内部面取件、UI 回滚归位、控制面/工作负载镜像对齐 ✅
|
||||
- **NetworkPolicy 栅栏真机**(第三十五批):pod 源"身份×端口"矩阵全对(允许行与拒绝行同源对照)、转发型外部源白名单闭环(拒→放→拒,ipset 与 spec 双幂等)、host/ClusterIP 路径全通、Mac→NodePort 面板 200;两条归因注记见该批 ✅
|
||||
|
||||
### 本轮新增真机证据(第二日)
|
||||
|
||||
- **默认安装的备份闭环**:apply 新 bundle → 归档 PVC 自动创建(WaitForFirstConsumer,被首个 backup Job 拉绑定)→ `POST /backup` 202 → `GET /jobs` running→succeeded → 漂移 marker → `restore-backup` → 读回备份前内容 ✅
|
||||
- **回收闭环(真实 k3s 布局 + 权限)**:resolvecheck(20d idle)→ CronJob 派 Job:`evaluated=2 reaped=1`;归档含 marker、PVC+宿主目录回收、`world_backups` 记 `inactive_15d`;期间先经历 `lstat /worlds/…: permission denied`(k3s storage root 0700)→ 修 ACL/root 权限后通过 ✅
|
||||
- **磁盘压力三连 drill**:①1e6 自定义类:游戏 pod 先走、控制面随后仍被逐(kubelet ranked/evicted 日志)→ 镜像 GC → ErrImagePull;②改内建 `system-cluster-critical` 后:`cannot evict a critical pod` × felis-api/operator/registry,全部保持 Running;③恢复阶段实测 `DiskPressure` 条件 ~5min(eviction-pressure-transition-period)转 False,被 GC 的游戏镜像按 runbook 重导入后 25s 恢复 ✅
|
||||
- **混沌**:SIGKILL k3s 主进程 → kubectl 与 felis-api ready 均在 **10s** 内恢复(api pod 本身未被重启)✅;内存压力(2.6G tmpfs 满写):无 OOM kill、无人被逐(swap 5.9G 吸收 + 内核回收),控制面不受影响 ✅
|
||||
- **构建链路定位(当时未修)**:Kaniko/Trivy 默认镜像是外部的 → 已加 `[registry] kaniko_image/trivy_image/build_cpu_limit/build_mem_limit` 覆写;上下文跨 ns 不可挂载(PVC 不能跨 ns)是设计级缺口,s3 lane 也缺凭据注入 ✅(证据:`internal/submit/blobstore.go:40`、`cmd/felis/api.go` contextBase 分支、jobspec 无 volumes/env)→ **第三批已修复,见下**
|
||||
|
||||
### 本轮新增真机证据(第三批:探针 + 构建链路全通)
|
||||
|
||||
- **控制面探针(`0c8e29b`)**:felis-api / felis-operator / registry 三个 Deployment 此前**完全没有探针**。修复后真机验证:api、operator `/readyz`+`/healthz`(internal 8081),registry `/v2/`(含命名端口解析);三个 Pod `Ready=true`、`restartCount=0`、无 `Unhealthy` 事件;operator 新增 `--health-probe-bind-address`(8081,与 metrics 8080 分离,零检查也 404 → 已注册 ping)
|
||||
- **构建链路全通(`f79e5eb` + `02fd2de`)**:
|
||||
- 传输:context_ref 改为 internal-face URL;`felis fetch-context` initContainer 走 service token(对象来自 `felis-build` 里按 `bootstrap`/`felis setup` 同款 Secret 复制机制落地的 `felis-service-token`;netpol 只放行控制 ns:8081)+ zip-slip 安全解包到限容 emptyDir;Kaniko `--context=/context` 只读挂载
|
||||
- **真机 E2E**:玩家 OTP 登录 → 建 submission → 上传 gzip context(marker `felis-e2e-build-marker`)→ Owner approve → Job:fetch ✅ → Kaniko 构建并 push `registry.felis.svc:5000/user-uploads/<sub>:latest` ✅ → Trivy 扫描(内建镜像 DB)✅ → Job `Complete`;**从 registry 拉回镜像校验 `/hello.txt` 内容 = 上传的 marker** ✅
|
||||
- 演练中真机抓到并修复 3 个单测看不到的缺陷:① Job requests=limits(2C/4Gi)在 4C/5.5G starter 上**永远 Pending**(`FailedScheduling/Insufficient memory`)→ requests 改为地板值(250m/512Mi,不超上限);② Kaniko 重拷 Dockerfile 时 chown/chmod 到源文件 owner(distroless uid 65532)在 drop-ALL 能力下失败(`copying dockerfile: chown … operation not permitted`)→ fetch 容器以 root 解包(= Kaniko 自身 uid);③ Trivy DB 默认 `mirror.gcr.io` 正被 egress 锁拒绝(fail-closed)→ 新配置 `[registry] trivy_db_repository` + 文档 §8e 固化镜像配方(`docker pull/tag/push aquasec/trivy-db:2` 进内建 registry;`--insecure` 已覆盖明文 HTTP)
|
||||
|
||||
### 本轮新增真机证据(第四批:PG 契约测试 + 验证邮箱唯一性 + owner 角色)
|
||||
|
||||
- **PG 契约测试首跑抓到 #20**:`internal/pgint` 对着真实 PG(`felis_pgint`,drop schema + 重放迁移)跑会话/OTP/op-login/绑定码/submission/build 全契约;`TestOnboardingEmailOTPRejectsTakenEmail` 当场变红 → 挖出「索引从未创建 + `ErrEmailTaken` 从未返回」。
|
||||
- **迁移 0020 真机应用**:`felis migrate up` → `applied 1 migration(s): [20]`;`schema_migrations` max=20;`pg_indexes` 出现 `users_verified_email_unique`。
|
||||
- **邮箱唯一性真机 409 演练(auditfix17)**:owner 已验地址被玩家 onboarding 流程二次验证 → 第一次 409 `email_taken`;**同码重放仍是 409**(证明码未被消费,符合契约;若被消费会是 400)。
|
||||
- **admin 改邮箱清 verified 真机演练(auditfix18)**:PATCH player.test → `[email protected]` 后 `email_verified=f`;PATCH 回原地址仍 `f`;玩家走 onboarding 重验证 → `t`。
|
||||
- **owner 角色真机闭环(auditfix18)**:owner 行按 0011 文档升级路径置 `role='owner'` → `GET /api/v1/users` 200(修复前 403);owner op-login start 铸请求+发码、finish 得 `role=owner` 会话;PATCH owner role / DELETE owner / DISABLE owner 全部 403;玩家邮箱登录回归 200。
|
||||
- **配额原子门(第五批,`bb68fef`,auditfix19 已上线)**:pgint 并发实证(真实 PG 上两路并发认领:恰 1 赢 + 1 `ErrQuotaExceeded`;修复前两路全赢);`docs/deferred-seams.md` 的 audit #4 条目核销。
|
||||
|
||||
### 本轮新增真机证据(第六批:reaper 警告链路端到端演练,auditfix20)
|
||||
|
||||
- **演练装置**:独立 hostNetwork Pod(`felis:auditfix20`,SA/卷结构复刻 live CronJob)+ 宿主本地零依赖 SMTP sink(捕获完整 RFC5322 原文)+ 专用 `felis-config-drill` Secret(SMTP 指向 `127.0.0.1:1025`);场景服 `warntest`(独立 CRD,desiredState=Stopped,不占资源),期间 operator 暂停防 reconcile。
|
||||
- **投递**:13d idle → 恰 1 封,`To: [email protected]`(owner 的**验证**邮箱)、subject 含 `3d`、正文含 server/remaining;`warned_3d_at` 盖章、`warned_1d_at` 留空;`world_backups` 保持 4 条(纯警告零备份)。
|
||||
- **去重与档位**:重跑 0 封(已盖章不重发);14.5d idle → 补发第 2 封 = `1d` 档(elif 顺序正确,不是重复 3d),`warned_1d_at` 盖章。
|
||||
- **失败语义(三连)**:SMTP 端口打坏 → 日志 `warn delivery failed; will retry next run`(err=`connection refused`)、**不盖章**,恢复端口后同一 run 补发成功;邮箱 `email_verified=false` → `resolve owner email: … has no verified email`、不盖章,恢复后补发;owner NULL → 静默跳过(无日志无警告)。
|
||||
- **边界**:idle=11d23h(< 12d 阈值)不发不盖章;idle=12d1m 发。`http_code` 冒烟:升级 auditfix20 后 panel 200 / op-login 200。
|
||||
- **环境还原**:warntest CRD+行删除、drill Secret/Pod 删除、operator 恢复;`servers` 表回到 resolvecheck+test-one,CRD 回到 lobby/login/resolvecheck/test-one。
|
||||
|
||||
### 本轮新增真机证据(第七批:idle auto-stop 三层修复,auditfix21/22)
|
||||
|
||||
- **发现路径**:operator 深挖时先怀疑"计时器无驱动",真机实验立刻抓到 pruning(操作日志 `unknown field "status.emptySince"`)+ 触发后 88s 无动作 + RBAC 403,三层各自独立、各自足以致死。
|
||||
- **场景 1(超时点唤醒 + 写权限)**:EmptySince 已超时 11 分钟 → annotate 触发一次 → 秒级内 desiredState=Stopped、sts 0/replicas、EmptySince 清、phase=Stopped。
|
||||
- **场景 2(完整自驱,决定性)**:desired=Running → pod 起 → 20:11:43 盖章 → 静置无干预 → **20:12:17 自动 Stopped**(30s 到点后 ~4s 完成 stop+scaledown+markStopped)。
|
||||
- **翻转混沌**:3 秒内 Running→Stopped→Running,最终收敛 Running ready(期间 409 竞争为控制器正常噪音,controller-runtime 重试自愈)。
|
||||
- **收尾**:`spec.idle` 删除后再无自停;test-one 回 Stopped、空 sts、无 EmptySince;VM 镜像 auditfix22 = `f650bf8`。
|
||||
|
||||
### 本轮新增真机证据(第八批:operator 自愈深挖,auditfix23/24)
|
||||
|
||||
- **RCON Secret 删除自愈(`56f3abd`)**:修复前——删 secret 静置 60s 零察觉、触发后 96s+ 永久 `RconNotReachable`、仅人工删 pod 可恢复;修复后——删 secret **同秒**(20:22:16)察觉(Secret watch 日志)→ 新密码 + stamp 换值(`3b24164f…`→`c465b627…`)→ rollout → **20:22:47 全自动恢复 Running**(31s)。
|
||||
- **陈旧锚点(`82b5a60`)**:修复前——已恢复服务器因锚点残留被 `Failed(StartupTimeout)`、`Provisioned=False` 永挂;修复后——Ready 即清锚(`startRequestedAt: None`)、`Provisioned: True`;pod blip 演练:删除→20:24:48 重锚→20:25:19 恢复→锚清空,**全程无 Failed**。
|
||||
- **StatefulSet 误删自愈**:删除 sts → **秒级**重建(PVC `world-test-one-0` 保持 Bound,Retain 策略)、29s 后 Running、stamp 保持 `c465b627…`(未触发多余 rollout)。
|
||||
- **收尾**:期间 7 条 operator ERROR 全为 CR/sts 409 竞争噪音(自愈);test-one 回 Stopped。
|
||||
|
||||
### 本轮新增真机证据(第九批:未演练端点地毯覆盖 — 迁移 / access / window / credentials)
|
||||
|
||||
- **account/migrate 四步状态机全链路(首个端到端演练,0 缺陷)**:造 user2 目标账户(绑定码 `FDTGKYAQ`)→ player.test claim test-one → 内部 start(未联动 UUID=404 not_linked;源=201 initiated)→ `passkey_required 409`(有 passkey 强制强因子)→ **passkey credentials list + DELETE 204**(顺带覆盖管理端点;删后 OTP 门自动解除)→ OTP start 202 / 错码 400 / 真码 confirmed → issue-code(self=400 / missing=400 / 真目标=201)→ redeem 负例×2(源持码=400;target 错码=403 setup_required——**onboarding 门正确拦截未验证账户**;验证邮箱后重试)→ **redeem 成功**(小写+空格码兼容,`servers_moved:1`)。
|
||||
- **retire 断言全对**:servers.owner→user2;migration=redeemed(含时间戳);源 disabled+软删;源活跃 session=0;源旧 cookie 401;double-redeem 400;源再加迁移 409 account_retired。**环境复原**:源复活 + 新 OTP 会话、test-one 归还无主。
|
||||
- **access 玩家管理全组(0 缺陷)**:players/whitelist/banlist 三个读 projector 解析正确(`There are 0 of a max of 20 players online: `→[]);kick/permission/group/luckperms-info 调用全通(demo 服无 LuckPerms → 原样透传 `Unknown or incomplete command`,API 不掩饰);输入负例 4 连 400(bad player/bad action/bad node/bad world);`whitelist add/remove` 在空档案服上得到 vanilla `That player does not exist`(fail-closed egress 无法解析 Mojang profile——**vanilla 约束非 API 缺陷**,命令已如实送达);手写 whitelist.json + `command whitelist reload` 验证成功路径解析(`players:["E2E_Tester"]`);**audit 7 条 access.\* 全记录**。
|
||||
- **updates/window**:unset→null;PUT 正例+读回;半设 400;倒置 400;clear→null 全对。
|
||||
- **/fleet**:4 服全量(含运行中 lobby/login 的 endpoint 地址)一次读全。
|
||||
- **bootstrap-assets crd**:嵌入式资产含 `emptySince`(auditfix24 二进制核对)。
|
||||
|
||||
### 本轮新增真机证据(第十批:configure email 复制链修复,auditfix25)
|
||||
|
||||
- **发现路径**:复核 `8e7c7bb` 新增的 replicate 逻辑时怀疑 manifest/-n namespace 冲突 → kubectl 真机实验裁决(hardcoded `namespace: felis` 版直接 `error: … does not match …`,证实第一段必错、第二段被跳过)。
|
||||
- **修复后行为验证**:两条命令序列(felis-smtp manifest 携带目标 ns + `felis-config` create--dry-run|apply)均被 server 接受且落点 `minecraft`。
|
||||
- **live 收敛项(随下次 `felis install/setup` 重跑)**:live CronJob 仍是旧模板(无 `FELIS_SMTP_PASSWORD` env);`felis-smtp`/镜像两 ns 均未创建(SMTP 未配置,属正常);与 operator Role 手动补 patch 同批处理。
|
||||
- **部署**:`felis:auditfix25` = `ed722d5`(api/operator/reaper 三处 set;api/operator rollout 完成,panel 200)。
|
||||
|
||||
### 本轮新增真机证据(第十一批:submission 全路由负例矩阵 + 内部面取件 + owner 会话重铸)
|
||||
|
||||
- **session 重铸路径(hook)**:owner 旧 cookie 过期(401)→ 临时插 `account_links`(owner↔假 UUID)→ op-login 三件套:start 202 → 日志取码 → 内部面 `op-login/{id}/approve`(hook,经 service token)→ status `approved:true` → finish 200 `role=owner` → op 域新 cookie 生效(`/fleet` 200);**演习后已删除假链接行**。顺带复证:approve 失败时 status=false、finish 400 `op_login_invalid`(码不烧,复用同码二次 finish 成功)。
|
||||
- **submission 车道负例矩阵(23/23 PASS,0 缺陷)**:create 未知字段/空名/非法字符/超长 → 全 400;非 gzip 上传 400、未知 id 404;跨用户上传 → **404 不可见**(非 403);玩家打 admin 三路由(queue/approve/reject)全 403;reject 空理由/1001 字/夹带字段 → 400,成功 200(`reject_reason`/`reviewed_by`=验主邮箱/`status=rejected` 全对);重复 reject 409、reject 后 approve 409;**无 blob approve → 400 且行保持 pending_review(未 stranded)**;玩家列表严格只见自己。
|
||||
- **内部面取件**:`GET /api/v1/internal/submissions/{id}/context`(service token)→ 200 + gzip magic `1f8b` + 解包内容 = 上传 marker;无 token 401;未知 id 404。
|
||||
- **演习清理**:E2E-Lane-\* 8 行已删、假链接已删;队列只留历史 "E2E *"(approved)行。
|
||||
- 备注:port-forward 再次因 API pod 重建悬死(空响应)→ 按速查重启即恢复;本轮无需新镜像(纯验证)。
|
||||
|
||||
### 本轮新增真机证据(第十二批:users 管理面全量矩阵 → 抓到并修复 #30)
|
||||
|
||||
- **/users 管理面矩阵(修复前,39 PASS / 2 NOTE)**:create/get/patch/list 权限与负例全对(未知字段/非法用户名/owner 角色/未知 id → 400/404;owner 保护 403 复证);sessions 单条吊销→旧 cookie 立即 401→重登恢复、revoke-all 同效(重登两次含 61s 冷却等待,全绿);passkeys 无凭证 200 no-op;links 幂等/跨用户 409/非法 auth_source 400/解链 404 全对;disable/enable/delete/double-delete 全对。
|
||||
- **缺陷 #30(两处 FK 500 + 一处假 200)**:`PUT /users/{unknown}/quotas` 与 `POST /users/{unknown}/links` 触碰 user_id 外键 → **500 internal**(契约要求 404);`GET /users/{unknown}/quotas` 回**零值视图(=unlimited)**,像 id 存在。修复:`PGRepo.requireLiveUser`(live 行 + `deleted_at IS NULL`)守卫三个方法,`handleGetQuotas`/`handleLinkAccount` 补 ErrNotFound→404 映射;单测 `TestAdminSubresourcesRequireLiveUser`(fake 同步 liveUserExists 契约)+ pgint 真库契约(ghost→ErrNotFound、live 对照、软删后→ErrNotFound)。commit `1918da2`。
|
||||
- **auditfix26 真机复验**:ghost 三连 → **404 not_found**;活用户对照 → PUT/GET quotas 200(读回 2)、POST link 200、delete 200;panel 200,api/operator 滚动完成(新 pod 24s Ready)。
|
||||
- 契约备注(非缺陷):软删用户 `GET /users/{id}` 仍 200(带 `disabled=true` + `deleted_at`,列表已过滤);`DELETE .../sessions/deadbeef` 幂等 200;revoke-all 对未知用户 200 no-op。
|
||||
|
||||
### 本轮新增真机证据(第十三批:CLI / 镜像面 / 直接构建 / 面板 CDP 全路由 + #31/#32)
|
||||
|
||||
- **CLI 面 19/19(VM,`felis-auditfix26`)**:`version`/`-h`/无参=exit2/未知命令=exit2;`manifests` 缺 `--velocity-cidr`、缺 `--felis-image`、坏 CIDR、坏 NodePort 全 fail-loud exit2,正例渲染 25 文档且 `--worlds-host-path` 分支出 CronJob+三条前置警示——整包 `kubectl apply --dry-run=server` **全部 `configured`**;`apply` 缺 `-f`/文件不存在/坏 JSON 各自 exit2/1/1;`migrate up` 二次幂等(`database already up to date`);`bootstrap-assets crd`(含 `emptySince`)与 `game-stack`(tar)正常;`update` 正常。
|
||||
- **镜像面负例 + builds 负例**:`/images` 列表 200、空体 400、合法 201、重复幂等 201、无 `?ref` 400、删除 **204**(契约;我的 200 预期作废)、再删 404、坏 ref 400;`/images/build/unknown` GET/cancel 均 404;空体提交 400;玩家打 `/images*` 全 403。
|
||||
- **`/images/build` 正路径(直接构建)**:借遗留 submission blob 作 context,`FROM scratch` 内联 Dockerfile → **202 → 12s succeeded**;`/logs` SSE 实收 Kaniko 流;真机 registry `tags/list` 出现 `e2e/direct-probe:latest`(镜像确实被推入)。留档:该 drill 镜像保留在 registry。
|
||||
- **`auth/options`**:未知邮箱 → `{"methods":[]}`;玩家 → `["email_otp"]`(无 passkey 时);坏邮箱 400。
|
||||
- **`PATCH /servers/{name}`**:空体/storage/空 policy/未知字段/坏名 → 400;未知服 404;displayName 设置+还原 200。**契约备注**:`displayName:""` 因 `omitempty`+merge-null 语义 = **清除该字段**(本次 drill 先误读为设空串,比对 patch 前 fleet 快照确认 test-one 原无 displayName,无数据损失)。
|
||||
- **面板 CDP 全路由巡检(owner 会话,14 路由)**:全部有标题/有内容/无崩溃;唯二发现 = #31/#32(见下);`luckperms` 页对 stopped 服 409 → **优雅降级**为「服务器已休眠 + 启动」态(browser 级 409 日志属正常资源日志,非缺陷)。玩家会话 4 路由零错误且**导航分级正确**(不见管理/平台组);未登录 `/login` 渲染正常(`/me` 401 为预期探测)、`/account` 未登录重定向 `/login`。
|
||||
- **缺陷 #31(面板 mock 残留·owner 判定)**:`ImageBuildPage` 用 `email==="[email protected]" || startsWith("owner@")` 猜 owner——真实 owner(`[email protected]`)**不被识别**(只能看 approved 提交),而任何 `owner@x` 邮箱都冒充 owner。改为消费 TierProvider 服务端 `isOwner`。commit `6907961`。
|
||||
- **缺陷 #32(面板 mock 残留·假构建种子)**:首次访问 `/admin/builds` 向 localStorage 播种 `["bld-1","bld-2"]` → 每个新浏览器打两个必然 404 的请求(页面注释自认“in production will 404”)。移除播种。同 commit。
|
||||
- **auditfix27 复验**:清掉该 localStorage 键后重访 → **只发 `/me`**、零错误、空态正确;面板 111 单测 + typecheck 全绿。
|
||||
- **passkey 全流程复验(CDP 虚拟认证器,面板 UI 实操)**:注册(对话框输入→仪式→列表出现 `E2E-Key`)→ API logout → 邮箱优先 passkey 登录(`/me` 回 `[email protected] user`、落回 `/`)→ `DELETE credentials/{id}` 204 → 列表清空(环境复原)。
|
||||
- **环境**:api/operator/reaper 镜像 = `felis:auditfix28`(= `6907961`);面板 200。**教训**:`pkill -f "port-forward …"` 会匹配到**执行该命令的 ssh 自身 cmdline**(命令里同时含明文 `port-forward svc/…`)→ 自杀式断连;改为 `ss -ltnp` 取 PID kill,另起一条 ssh 启动转发。
|
||||
|
||||
### 本轮新增真机证据(第十四批:#33 死账号复活漏洞 —— 全登录门闭环 + 资产切断)
|
||||
|
||||
- **发现路径**:users 管理面演习收尾核对时发现软删用户仍留 `account_links`/`quotas` 孤儿行 → 顺藤摸瓜确认 `UserByEmail` 与 `SessionUser` 均不过滤 `disabled`/`deleted_at`。
|
||||
- **红证据(auditfix27,真机)**:禁用用户的邮箱门重登录 `verify=200`、`/me=200`(**禁用锁死可绕过**);**已删号**用户 `start=202`(真实发码)、`verify=200`、`/me=200`(**删号可复活**)。
|
||||
- **修复(`5853589`,7 文件 / +306 行)**:① `UserByEmail` 只解析 live 账号(覆盖 email/passkey 邮箱优先/op-login/auth options 全部前置门);② `SessionUser` 同样过滤(皮带:任何门铸出的死号会话都立即失效);③ `RedeemPlayerBindCode` 两个解析臂拒绝死账号(新哨兵 `ErrPlayerAccountRetired` → 403 `account_retired`,**不消费码**,可逆重试);④ `DeleteUser` 事务内 `DELETE webauthn_credentials` + `account_links`(凭证与 `UNIQUE(mc_uuid)` 占用不再外泄);⑤ discoverable passkey resolve 补 liveness 检查。单测 `TestDeadAccountsCannotLogInOrKeepSessions` + pgint `TestDeadAccountsAreLockedOutInPG`(含"删后新铸会话不验证"与资产清点)。
|
||||
- **绿证据(auditfix28,真机)**:红期为死号铸出的会话 → **401**(皮带生效);死号 `start=202` 中立且 **90s 日志窗口 0 条码**、`verify=400 invalid_code`;禁用中 `start` 中立(日志码数 1→1 不变)→ `verify=400`;**恢复启用后 `verify=200` + `/me=200`(可逆)**;bind 门死账号 → **403 `account_retired`** 且码保留 `count=1`(retryable);删号后 `account_links`/`webauthn_credentials` 行数 = 0。
|
||||
- **演习清理**:红演练遗留(2 个测试账号的 quotas 行、1 条旧链接、2 条死号会话)已清除;`/fleet` 与 panel 冒烟保持 200。
|
||||
|
||||
### 本轮新增真机证据(第十五批:#34 死账号同族收尾 —— in-game 身份解析与链接接管,auditfix29)
|
||||
|
||||
- **发现路径**:#33 修完 Web 登录门后做同族复查——**in-game 面是 UUID 驱动、不走 Web 会话,因此 #33 的会话皮带盖不住它**。两处:① `UserByMCUUID`(claim / menu / wake 授权 / op-login vouch / QR link-status 五处共享解析)直接读 `account_links`、不过滤 users 的 liveness;② `VerifyLinkCode` 对「holder 已死」的接管语义未定义——一刀切 409 会把迁移退役这类真实场景锁死。
|
||||
- **修复(`bb9798e`,5 文件 / +182−7)**:① `UserByMCUUID` JOIN users 过滤 `disabled`+`deleted_at`——死账号的链接在游戏面**读作未链接**(claim 412、wake 落回 policy 门、vouch 403、status `linked:false`),绝不作为遗留身份存活;② `VerifyLinkCode`:**软删** holder 的链接可由新账号凭新 mint 码**接管**(账号已亡,码证明调用者仍持有该 UUID),**禁用** holder 仍 409(接管= 绕过锁死,不允许)、任何失败都不消费码。fake 单测 `TestLinkVerifyTakesOverDeletedLinkOnly`;pgint `TestVerifyLinkCodeTakesOverDeletedLink` + `TestDeadAccountsAreLockedOutInPG` 增「禁用链接无资格 / 复启用恢复」断言。
|
||||
- **绿证据(auditfix29 真机,三面六点)**:
|
||||
1. `link/status`:活 `{"linked":true}` → 禁用 `{"linked":false}` → 恢复 `{"linked":true}`;
|
||||
2. `claim`:活链 404(身份已解析、ghost 服不存在)→ 禁用 **412 not_linked** → 恢复 404;未知 UUID 对照 412;
|
||||
3. **接管正例**:临时软删 holder(`de49df52`)→ 新 mint 码 `X3G3RZZM` 由 player.test verify → **200 linked:true**;链接行迁移到 player.test(`auth_source=mojang`)、码被消费(`account_link_codes` 0 行);演练后 holder 与链接全部还原;
|
||||
4. **接管负例**:临时禁用 holder(user2)→ 新码 `X5W94S42` verify → **409 already_linked**;同码重试仍 409(**不消费**);恢复启用后**同码**由 holder verify → **200**(码保留、可逆);
|
||||
5. **wake 授权**:活 owner **202**(真实启动)→ 禁用 **403 forbidden** → 恢复 **202**;test-one 已停回 Stopped 且所有权释放;
|
||||
6. **op-login vouch**:禁用 owner UUID → **403 not_admin**(与未链接/非 staff 同一个拒绝面);活 owner UUID → 404 op_login_not_found(vouch 已解析、只是请求不存在)。
|
||||
- **环境刷新(部署态)**:热升级遗留 `FELIS_IMAGE=felis:auditfix16`(job 侧执行器 backup/restore/fileedit/forwarding-init 引用的镜像)→ 已 `set env` api/operator 双 Deployment 至 `felis:auditfix29` 并滚动完成;api/operator/reaper 三镜像一致 = auditfix29;port-forward 重启后内部面绿;panel 200、op 面 `/me` 200。
|
||||
- **运维备注**:`account_link_codes` 会积累过期行("失败不消费"的另一面),巡检顺手 `DELETE FROM account_link_codes WHERE expires_at < now()`。
|
||||
|
||||
### 本轮新增真机证据(第十六批:内部面残面地毯补测 —— ready / join-event / status / reclaim / blacklist / backup 负例)
|
||||
|
||||
- **背景**:对内部面做覆盖盘点后发现五个端点此前无真机证据:`ready`、`join-event`、`status`、`player/reclaim`、`player/blacklist`,以及 `internal/servers/{name}/backup` 的负例臂。本批全部补测(auditfix29),**0 新缺陷**。
|
||||
- **`ready`**:test-one → **204**(advisory 语义);坏名 → 400。
|
||||
- **`join-event`**(临时 UUID `3333…`):首次 **204**、重复 **204**(幂等);DB 实锤:`test-one.last_active_at` 15:02→05:53(重置)、`server_allowlist` 恰 1 行;缺 uuid → 400;未知服 → 404。演习后 allowlist 行删除、`last_active_at` 复原。
|
||||
- **`status`**:test-one → 200 全字段(phase/desiredState/endpointMode=fallback/…);未知服 → 404。
|
||||
- **`player/reclaim`**:新建(临时 UUID `4444…`)→ 200 `hold_expires_at`=+30d;**幂等重试返回同一时间戳**(首窗保留,与存储行 21:53:51.726017Z 完全一致);缺字段 → 400;**protected-admin 负例**:临时插 owner↔thirdparty 链接 → **409 protected_admin**(不 bar、不 stash)。清理后 `username_blacklist`/`player_data_holds`/`server_allowlist`/临时链接 = 0/0/0/0。
|
||||
- **`player/blacklist`**:bar 后命中 `{"blacklisted":true}`;陌生 UUID → `false`。
|
||||
- **`internal backup` 负例**:未知服 → 404;坏名 → 400;保留名(login)→ 400 `bad_name`(在入队前拒绝,RWO 闸门单测已覆盖)。
|
||||
- **探针**:`/healthz` 200、`/readyz` `{"status":"ready"}`。
|
||||
|
||||
### 本轮新增真机证据(第十七批:hasJoined 多源会话校验器 —— 假 Yggdrasil 全链路 + 三方身份重写/改名)
|
||||
|
||||
- **背景**:`hasJoined`(velocity 指向的 vanilla sessionserver 协议面,`handleHasJoined`)此前零真机覆盖。本批用「VM 主机假 Yggdrasil + 临时将 `[[auth_source]]` 换向」的方式把正/负路径全部打通。
|
||||
- **装置**:主机 python 假源(`:18099`,按 username 分流 `ftok`/`ftnotch`/`ftbadname`/`ftdown`/其余 204);`felis-config` 的 littleskin 源临时改指 `http://10.42.0.1:18099/fake`(tag=`faketest`/prefix=`FT`),api 滚动后逐项打靶。**演练后配置已还原 littleskin、装置已清理**。
|
||||
- **结果(全绿,0 缺陷)**:
|
||||
1. **三方身份重写**:`ftok` → 200 `{"id":"74409c3bbae93acabd2176e517b4f2a0",…}`,与本地按 `uuid.NewMD5(felisAuthNS, "faketest:native-123")` 的预算值**逐位一致**;
|
||||
2. **保费名冲突改名**:`ftnotch` → `47c5527a18b03fe5a49ec50c11154dcd` + `name="FT_Notch"`(Mojang 实查 Notch=200 premium);`ftok` 的 `FtPlayer` 恰也是真实 Mojang 名(`640c1672…`)→ `FT_FtPlayer`,改名按设计触发(非保费名保持原名);
|
||||
3. **敌意插件名拒绝**:假源返回 `§4admin` → **204**(不落 proxy 玩家列表);
|
||||
4. **源故障不静默**:假源 500 → **503**(velocity 报 auth servers down),对照未知玩家 → 204;
|
||||
5. **形状负例**:缺参 / 超长参 → 204;声明 body → **400 + `Connection: close`**(防 drain 挂连接)。
|
||||
- **顺带覆盖**:`isPremiumName` 的 Mojang 实查(`api.mojang.com`,超时/错误 fail-closed=改名)——404→非保费、200→保费判别实测成立。
|
||||
|
||||
### 本轮新增真机证据(第十八批:缺陷 #35 —— world 执行器 uid 1000 读不了游戏服写的世界 + 面板备份模块补全)
|
||||
|
||||
- **发现路径**:给备份页新增「立即备份 + 最近操作」后做第一次真机 drill——面板链路全对(按钮/成功消息/运行态→终态、零坏请求),**但 backup Job 真的失败了**:`Job has reached the specified backoff limit`,pod 日志 `felis backup: archive: backup: tar walk: open /world/world/level.dat: permission denied`。
|
||||
- **根因(缺陷 #35,跨模块)**:世界卷的属主是**游戏镜像自己的 UID**(我们发布的 Paper 镜像都是 root),而 Paper 保存 `level.dat` 用的是 **0600**(Files.createTempFile 默认权限)→ 固定 uid 1000 的 **backup / restore / fileedit / reaper** 四类执行器:读不了(归档 `permission denied`)、也覆盖不了(restore 写不进 600-root 的 level.dat)。此前 drill 侥幸全绿,是因为当时的世界文件由 uid-1000 工具(restore/夹具)写的;**服务器真实启动保存过一次之后**,所有备份从此必死。测试全部是 shape 断言(无集群),这条只能真机抓。
|
||||
- **修复(`2010961`)**:四类执行器统一改**以 root 运行**(`runAsNonRoot:false`,省略 fsGroup 防误 chgrp),容器保持除 `DAC_OVERRIDE` 外 drop-ALL——与 operator `init-forwarding` 容器的既定先例同源("只有 root 能可靠读写这些文件");DAC_OVERRIDE 兜住"游戏镜像是非 root UID"的任意镜像场景。四份 shape 测试同步改断言。troubleshooting §10「Permissions」段落重写(uid-1000 + setfacl 时代结束)。
|
||||
- **绿证据(auditfix31,真机四联 drill)**:
|
||||
1. **backup**(面板 CDP 实操,红→绿同场景):点击「立即备份」→ `进行中` → **`成功`**(同页面 reload 后仍在);Job pod `runAsUser:0` + DAC_OVERRIDE;新归档 `test-one-1790115428472416736.tar.gz` 实测含 `world/level.dat`(471B)与 server.properties,共 499 条。
|
||||
2. **restore**:`POST restore-backup`(bk-9df0…)→ 202 → Job **Succeeded**(root 写路径过关),日志 `restored from … into /world`。
|
||||
3. **fileedit**:`GET /servers/test-one/file?path=world/level.dat` → **200**(base64-gzip 内容 628B)——修复前该请求必然 permission denied。
|
||||
4. **reaper**(从 live CronJob 派生一次性 Job + 复刻钻取世界:1Gi PV/PVC + `world/level.dat` 512B **0600 root** + marker + 20d idle 行):pod **root+DAC_OVERRIDE**;`world reaped server=reapdrill2 …`、`evaluated=3 reaped=1`;归档 711B 含 marker 与 level.dat;PVC 删除;`world_backups` 得 `inactive_15d` 行;servers 行/CRD 保留(红线②)。**钻取现场全部清理**(job/CRD/行/PV/宿主目录/临时归档)。
|
||||
- **面板补全(`97a64c8`)**:备份页新增「立即备份」(仅 Stopped 可用,409 原文呈现)与「最近操作」卡(GET `/servers/{name}/jobs`,running 每 5s 自刷新,failed 显示 Job 失败文本——正是它把上面这次失败暴露出来的)。面板 113 单测 + typecheck 全绿。
|
||||
- **部署注意**:backup/restore/fileedit 的 Job spec 是 api **运行时渲染**,随镜像即生效;**reaper CronJob 的 pod 模板是安装期静态渲染**——本轮已按新形状热补丁 live 对象(root + DAC_OVERRIDE),`felis install/setup` 重渲染时收敛。
|
||||
|
||||
### 本轮新增真机证据(第十九批:面板补齐 owner 解绑通行密钥入口)
|
||||
|
||||
- **背景**:`DELETE /users/{id}/passkeys`(owner-tier 凭据补救:密钥丢失/被盗时切断登录脚架,且不锁死账号——邮箱码/游戏内审批仍可用)后端早已实现并审计,但面板无入口,owner 只能靠 API。属"功能缺口"而非缺陷。
|
||||
- **修复(`11ac4f5`)**:用户详情页危险操作区新增「解绑通行密钥」行 + 确认对话框(`api.unbindUserPasskeys` 客户端方法 + 双语 i18n + wire-shape 测试)。
|
||||
- **绿证据(auditfix32,CDP 真机全链)**:player.test 注册虚拟认证器密钥 `E2E-Unbind` → `/auth/options` 从 `["email_otp"]` 变为 `["passkey","email_otp"]` → owner 在 `/admin/users/<id>` 点「解绑通行密钥」→ 对话框 → 确认 → **凭据列表清空、options 回落 `["email_otp"]`**(passkey 门确实关闭);审计 `user.unbind_passkeys` 落账(同批还可见 `account.passkey.registered`/`backup.create`/`backup.restore`/`reap_world` 各审计行)。面板 114 单测全绿。
|
||||
|
||||
### 本轮新增真机证据(第二十批:缺陷 #36 —— 提交者看不到构建结果)
|
||||
|
||||
- **缺陷 #36(提交/构建模块·结果不可见)**:`/me/submissions` 只回审核状态;构建的成功/失败(含失败原因)只有 admin-tier `/images/build/{id}` 看得到 → **提交者永远不知道自己的包构建死了**。修复 `72c4aa3`:两个列表路由(玩家 `/me/submissions` + 管理 `/submissions`)对 `build_id` 非空的行附 `build_status`/`build_error`,来源是只读 `Builder.Get`(**绝不调 Sync**——状态推进归 15s reconcile 循环,列表渲染不碰集群);构建行已消失(ErrNotFound)→ 字段省略;其他存储错误照常 500,绝不静默吞。OpenAPI 的 Submission schema 同步。
|
||||
- **绿证据(auditfix33,真机双例)**:① 失败例(上下文 Dockerfile `COPY does-not-exist`)→ 提交 → owner approve → `/me/submissions` 返回 `build_status:"failed"` + `build_error:"build job failed or scan found a CRITICAL CVE"`;② 成功例(`FROM scratch`+LABEL)→ `build_status:"succeeded"`。面板 CDP(玩家会话):展开行显示「构建状态」徽章(构建失败/构建成功)+ 失败原因 + build_id,全程零 4xx。**演习残留已清**(2 行 submission+build、2 个 context blob、1 条 whitelist 条目;`sub-bf7dc1…` 那条是更早 E2E 遗留,未动)。
|
||||
|
||||
### 本轮新增真机证据(第二十一批:面板文件编辑器补齐 —— 缺口而非缺陷)
|
||||
|
||||
- **背景**:`GET /files`、`GET/PUT /file` 后端早已全绿(可读写 `level.dat`),但面板无入口——「一行 server.properties 写错导致起不来」的修复路径只有 API。属功能缺口。
|
||||
- **修复(`0a36b3f`)**:新增 `/servers/:name/files` 页:面包屑目录浏览、编辑器对话框([]byte ↔ base64 编解码)、二进制文件打开即只读(NUL/非 UTF-8 拒绝 round-trip)、>256KiB 禁用保存;**停服门前置**(世界卷 RWO,未停服时整页显示「服务器正在运行」+ 停止动作,而不是让每个调用 409);控制台右侧新增门口卡。i18n `files` 命名空间(en/zh)+ 3 条 wire-shape 测试。
|
||||
- **绿证据(auditfix34,CDP owner 全链)**:根目录 → `world/` 导航;`felis-e2e-marker.txt`(原 `v1\n`)打开 → 追加 → 保存「已保存 …」→ **API 读回一致** → 重开一致 → 还原原始字节 → 落盘复核一致(零残留);`level.dat` 打开为只读 + 二进制提示;控制台门口卡存在;`audit_logs` 两条 `file.write`(edit/restore 各一);全程零 API 4xx/5xx。面板 117 单测 + typecheck 全绿。
|
||||
|
||||
### 本轮新增真机证据(第二十二批:缺陷 #37 —— 系统服务在面板里全是死操作)
|
||||
|
||||
- **缺陷 #37(面板·fleet 死操作)**:`login`/`lobby` 是平台自建系统服务、名字在保留名单里,于是**每个 per-server 路由都用 `ValidateServerName` 拒绝**(400 `bad_name`)——但管理端 fleet 表格给这两行渲染 认领/停止/唤醒/控制台,全是死操作(控制台链接点进去也是一页 `invalid server name: reserved`)。修复 `2f90851`:`naming.IsSystemServer` 作为唯一事实源;fleet 行附 `system:true`;面板把这两行渲染为「系统服务」纯标签(owner 列 + 操作列),不再给任何动作。玩家侧不受影响(`/me/servers` 本就不含系统服务)。
|
||||
- **绿证据(auditfix35,真机)**:`GET /fleet`(owner)→ `lobby system:true`、`login system:true`、`test-one/resolvecheck` 无 flag;面板 CDP:两行「系统服务」、**0 按钮 0 链接**;`test-one` 行照常 认领/控制台、无系统标记;零 4xx。Go 侧新增 `TestIsSystemServer` + `TestFleetAdminRead` 的 system 子测试。
|
||||
|
||||
### 本轮新增真机证据(第二十三批:缺陷 #38 + 多节点回收缺口 —— reaper 提示过期 / 钉节点)
|
||||
|
||||
- **缺陷 #38(CLI·提示过期)**:`felis manifests` 渲染 reaper 时的 stderr 提示还在教“uid 1000 需要 `setfacl -m u:1000:x` 才能遍历存储根”——#35 之后 reaper 已改为 **root + DAC_OVERRIDE**,这条指导已失效且会误导运维(照做无害但白做,真问题被掩盖)。同批落 **多节点回收缺口**(结论第 5 条):新增 `--reaper-node`,reaper CronJob 的 pod 渲染 `nodeSelector kubernetes.io/hostname=<node>`;多节点集群必须钉在存世界的节点,否则可能调度到 hostPath 为空的节点。`--reaper-node` 无 `--worlds-host-path` 时 fail-loud exit 2。修复 `daf7602`(双测:`TestReaperCronJob_NodePin`、`TestManifestsReaperNodePin`)。
|
||||
- **绿证据(auditfix36,真机 render + dry-run + 收敛 diff)**:
|
||||
1. **固定渲染**:`felis manifests --felis-image felis:auditfix36 --velocity-cidr 10.211.55.6/32 --panel-node-port 30443 --worlds-host-path /var/lib/rancher/k3s/storage --archive-local-path /var/lib/felis/archives --reaper-node localhost.localdomain` → exit 0;bundle 内 `kubernetes.io/hostname: localhost.localdomain`;stderr **0 处** uid 1000 / setfacl,改为「已钉到节点 …」;
|
||||
2. **负例**:`--reaper-node` 无 `--worlds-host-path` → exit 2 + 原文;不传 node 的渲染 stderr 仍完整保留「NO nodeSelector … 多节点必须传 --reaper-node」警示,bundle 内 0 个 selector;
|
||||
3. **`kubectl apply --dry-run=server -f -`**:整包 **全部 configured**(含新 nodeSelector 的 CronJob);
|
||||
4. **再安装收敛性 diff**(`kubectl diff`,本批新增的收敛证据):与 live 对比只剩 **一个对象**(reaper CronJob)两处实质增量——`+env FELIS_SMTP_PASSWORD`(热升级期未补的模板字段)与本次显式传入的 `+nodeSelector`;其余 24 份文档(api/operator Deployment、RBAC、NetworkPolicy、PVC、registry)**零差异**——即“热补丁过的 live 对象”与“当前代码重渲染”已收敛(reaper 安全上下文 root+DAC_OVERRIDE 两侧一致,无 diff)。
|
||||
|
||||
### 本轮新增真机证据(第二十四批:S3 上传通道真机演练 —— 此前只有单测的暗路径)
|
||||
|
||||
- **背景**:`internal/submit/s3store.go`(S3ContextStore)此前只有单测,本装是 local 路径(`user_uploads_context = "/var/lib/felis/uploads"`)从未激活。本批用「VM 宿主 MinIO(quay.io 镜像,:9000)+ 临时把 felis-config 切到 `s3://felis-user-uploads` + `[registry.s3] endpoint=http://10.211.55.6:9000` + 创建 `felis-uploads-s3` Secret」把整条通道打通,**全程 0 缺陷**、演练后还原并逐项复核。
|
||||
- **绿证据(auditfix36,真机全链)**:
|
||||
1. 切 S3 后 api 滚动启动 **无** “S3 user-uploads store not configured” 告警(凭据解析成功);pod 内 busybox 探针实测可达 `http://10.211.55.6:9000/minio/health/live`(rc=0);
|
||||
2. 玩家提交 + 上传 → **200**,对象实测落桶:`sub-12c953e12b0cd144/context.tar.gz`(197B,mc ls 实见);
|
||||
3. owner approve → 构建 Job `build-bld-1790118222183394193` **status.succeeded=1**(fetch-context 从内部面流式取件 = api 自 S3 读回成功);`/me/submissions` 的 `build_status` 收敛为 `succeeded`;
|
||||
4. **回滚**:felis-config 还原(sha256 与演练前备份**逐字节一致**)、删除 `felis-uploads-s3`、api 滚动;再演练一次本地路径:新提交上传 **200** 且 blob 实测落在 uploads PVC(context.tar.gz 197B);启动日志仅剩 smtp/jwks 两条既有提示;
|
||||
5. **清理**:2 行 submission + 1 行 build 删除、回滚演练 blob 删除、S3 构建产物从 whitelist 摘除(204)、MinIO 容器 + 两个镜像移除;`/fleet` 200。
|
||||
- **遗留观察(非缺陷)**:registry 里保留本次推送的 `user-uploads/sub-12c953e12b0cd144:latest` 层数据(与早前 direct-probe 同类,filesystem registry 无删除接口);`felis setup` 的「S3 存储」向导屏本身未演练(列入 TUI 逐屏待办)。
|
||||
|
||||
### 本轮新增真机证据(第二十五批:breakGlass 控制台逐屏全量 + 备份/恢复门禁 —— 缺陷 #39–#43)
|
||||
|
||||
- **背景**:breakGlass 此前只验过 Owner 首装与 Add Operator happy path。本批把菜单四操作(Owner reset / Add Operator / Halt / Sync)+ 首装(bootstrap)分支全部逐屏真机走完,并打穿 Sync/恢复背后的 API 门禁;共抓 5 个缺陷、全部修复复验。驱动方式:VM tmux(`remain-on-exit on` 才能读回 alt-screen 撕掉后的 durable summary)。
|
||||
- **先落的正向证据(无缺陷)**:Halt——`test-one` Running → 选中 → 卡「is stopping」→ CRD `desiredState=Stopped`、pod 收敛消失、审计 `break_glass.halt`;面板 `wake`/`stop` 两个恢复杠杆均 202(复验后还原)。Sync 正路径——test-one(Stopped)→ 卡 backup started → Job 6s Complete → 归档落 `felis-backups` PVC(167MB)→ `world_backups` 行 `present` → `/api/v1/backups` 可见。Sync 负路径——对 Running 选 → 友好 409 卡(不误烧冷却)。
|
||||
- **缺陷 #39(owner 席位可被静默复制,且不可清理)**:恢复模式下用非在位席位名做「reset」→ `UpsertOwner` insert 臂**铸出第二个 owner 行**、原席位继续存活;面板对任何 owner 行都删/降/禁 403 → 永久无法收敛回单席。修复 `55d515d`:`provisionOwner` 先查在位席位(新 `PGRepo.OwnerUsername`),非席位名 → `ownerSeatTakenError`(`Is api.ErrConflict` → TUI 路由回表单并**指名**应输入的用户名);bootstrap(无席位)与同席位名复位原样。真机双验:新名被拒(表单原位显示 `an Owner already exists as "08595879-…" — enter that username to reset the Owner`);改席位名复位成功(owner 恒 1 行、id/邮箱不变);面板删除保护同步实证(对新 owner 行 DELETE → 403)。
|
||||
- **缺陷 #40(operator 撞名 = 裸 SQLSTATE,「换名重试」分支在真机从未生效)**:`InsertOperator` 冲突返回原始驱动错误(23505),TUI 却按 `api.ErrConflict` 判定「可恢复、换名重试」——fake 与 PG 漂移。真机复现:输入已存在用户名 → 控制台 exit 1 + 裸错误。修复 `55d515d`:`isUniqueViolation → ErrConflict` + pgint 契约断言。复验:撞名**回到表单**提示换名 → `drill-op-2` 成功(审计 `break_glass.operator_create` 落账,演练行已清)。
|
||||
- **缺陷 #41(Sync picker 死选项)**:picker 列出 login/lobby,而备份 API 对它们**永远失败**(保留名 + 无 servers 行);真机选 lobby → 裸内部错误 + exit 1。修复 `ac3a557`:`backupPickable` 过滤系统服(halt picker 保留它们,断玻璃完整权力)。
|
||||
- **缺陷 #42(缺失世界盘 → 202 后静默卡死 30 分钟)**:对「从未启动/已被回收」的服备份或恢复:202 → Job → Pod `persistentvolumeclaim "world-<name>-0" not found` **Pending 至 deadline**,全程零失败记录。真机用已回收的 `resolvecheck` 复现(留证后删除)。修复 `508a1c0`:`Cluster.WorldVolumeExists`(直接 Get 与 Job 挂载**同名**的 PVC)+ 两 handler 409 `no_world_volume`("start it once to create it, then retry")。**修复首跑翻出配套 RBAC 洞**:felis-api SA 无 `persistentvolumeclaims:get`(403 被吞成 500「internal error」)→ `APIMinecraftRole` 补 get-only 规则 + rbac 测试锚点。复验(auditfix38):resolvecheck 两面 409 + 友好文案;test-one 照常 202 → Job 10s Complete → 新行落库。
|
||||
- **缺陷 #43(409 一刀切文案)**:TUI 把所有 409 当停服门 → 世界盘拒绝会展示错误原因。修复 `ac3a557`:按 body 的 `error.code` 分流(无 code 的旧体仍按停服门)。复验:无盘 pick 显示 API 原文;停服门文案不变。
|
||||
- **同步完成**:① bootstrap 分支 scratch 库演练(`migrate up` 20 迁移 → 无菜单/无认证直接铸 owner;`local_auth_enabled=true`;审计 `break_glass.bootstrap`;exit 0;库/hba 规则/临时配置即测即清);② 台面收敛:reaper CronJob `suspend=true/auditfix36/无 pin` → `suspend=false / felis:auditfix38 / nodeSelector=localhost.localdomain`;③ 镜像升级 `auditfix37→38`(api/operator + `FELIS_IMAGE`)。
|
||||
- **流程修正(教训)**:pgint 一度误用**本机 Docker Desktop**(启动 daemon + 临时 PG 容器)——已完全清理(容器/镜像删除、daemon 退出),并改为**经 ssh 隧道用 VM 的 postgres** 运行(`ssh -L 15433:127.0.0.1:5432` → `postgres://felis:***@localhost:15433/felis_pgint?sslmode=disable`)。勿再在本机跑容器。
|
||||
|
||||
### 本轮新增真机证据(第二十六批:`felis setup` 重跑向导逐屏 + 构建链 pin 回验 —— 缺陷 #44)
|
||||
|
||||
- **范围**:本机已装机,故覆盖"重跑状态屏 + c/s/e 三条 reconfigure 流";首装屏(postgres/owner/connect/edge/storage/smtp/mc-bind/migration/preflight/summary)此前各批已有定点真机证据(安装闭环 / 第二~三批 / 第九批 / 第二十五批),本轮不重复。
|
||||
- **逐屏走查(auditfix38→39,VM tmux)**:
|
||||
1. 重跑 → 直落状态屏「✓ Felis is already set up.」(owner/connect 不触碰;host bootstrap 已就绪跳过)✓
|
||||
2. `c` → 三选一 chooser(Local / Cloudflare+Access / Reverse proxy + 警示语)渲染 ✓,esc 无损返回。
|
||||
3. `s` → chooser 预选当前后端;S3 分支表单(Endpoint/Bucket/Region/AK/SK + 提示)渲染 ✓。**观察:reconfigure 非只读**——选「Local disk」即 apply(写 /etc 两文件 + 重渲染 felis-config + 滚 API);从 S3 表单 esc 退回会把选择重置为 Local 预选。
|
||||
4. `e` → SMTP 表单渲染 ✓;esc 直接回状态屏、零副作用(felis-smtp 未创建)✓。
|
||||
- **缺陷 #44(CLI·重跑框脱落)**:重跑后完成 `s`/`c` reconfigure,落回首装 summary「✓ Setup complete.」——丢了 alreadySetUp 框(smtp 路径有专门分支,storage/connect 漏)。修复 `abb5910`(`showSummary` 透传 `m.result.alreadySetUp`)+ 回归测试 `TestRootReconfigureStorageKeepsStatusFraming`。真机复验(auditfix39):storage reconfigure 完成 → 「✓ Felis is already set up.」+ storage recap ✓。
|
||||
- **演练事故(自曝;环境 drift,非产品缺陷)**:首次 storage 走查意外触发 Local apply——它从 `/etc/felis/felis.pod.toml` 重渲染 Secret,而构建链 pin 值当初**只热补在 live Secret、不在 /etc 文件** → 重渲染清空 pin(放任则下次构建死在 `:latest` 拉取 + trivy DB egress)。当场修复:pin 值写回 `/etc/felis/felis.host.toml` + `/etc/felis/felis.pod.toml` → 重渲染 Secret → 滚 API;再做第二次 storage apply,重渲染后 pin 仍在(drift 修复耐久)。
|
||||
- **构建链回验(direct build,真机)**:Job 规格实证 `kaniko=gcr.io/kaniko-project/executor:v1.24.0`、`trivy=aquasec/trivy:0.74.0`、`--db-repository registry.felis.svc:5000/mirror/trivy-db:2`(registry 仍有 `mirror/trivy-db`);`POST /images/build`(context=遗留 `sub-bf7dc18…` blob,内部面取件)→ 202 → **succeeded**;registry `e2e/pins-check` 落位;whitelist 条目已摘除(204)、Job 已清。
|
||||
- **文档修正(`ae6e925`)**:§8e 原「编辑 felis.toml 后重启 felis-api」不完整(API 挂的是 Secret)→ 改为「写进 /etc 两文件 → 重渲染 Secret → roll」,并写明三种无效/易损做法(只 restart / 只改 host 文件 / 只热补 live Secret——后者会在下次 reconfigure 被冲掉)。
|
||||
- 收尾:api 1/1、panel 200、tmux 全清。
|
||||
|
||||
### 本轮新增真机证据(第二十七批:告警模块落地 —— 内部面 /metrics + 规则集 + 真实构建失败实弹演练)
|
||||
|
||||
- **范围**:把"指标 → 规则 → 告警"链路从零补到可交付:API 内部面 `/metrics`(`94f71ee`)、`deploy/alerts/` 规则与 promtool 单测(`43df08b`)、真机实弹演练(本批)。
|
||||
- **/metrics(`94f71ee`)**:`felis_image_build_failures_total` 此前只在进程内存里、无任何 scrape 出口。修复:internal face(8081)新增 `GET /metrics`(服务面 Public 路由,语义同 /healthz);单测断言外部面 404。真机:port-forward `svc/felis-api-internal 18081:8081` → 200 且含 `felis_image_build_failures_total`;operator `:8080` 提供 `felis_servers_total` / `felis_start_duration_seconds_*`。
|
||||
- **规则集(`43df08b`)**:`deploy/alerts/felis-alerts.yaml`(plain Prometheus)5 条——构建失败 increase>0 / 起服 p90>300s / 磁盘可用<15% / DiskPressure / 内存可用<10%;`felis-prometheusrule.yaml` 为 prometheus-operator twin(脚本比对两文件 groups 一致);`felis-alerts_test.yml` 为 promtool 单测。VM 上 promtool 3.14.0 实跑:`check rules` + `test rules` 双 SUCCESS。
|
||||
- **实弹演练(真实构建失败 → pending → firing)**:
|
||||
1. 打包含 `COPY does-not-exist` 的 Dockerfile 上传到 uploads PVC `sub-alertdrill` → `POST /images/build`(`e2e/alert-drill2:latest`);
|
||||
2. Kaniko `failed to get fileinfo for /context/does-not-exist` → Job Failed、build 行 `failed`;
|
||||
3. 真实 Prometheus(宿主 `:19090`)scrape `127.0.0.1:18081`(api) 与 `:18080`(operator) 双 target up;`felis_image_build_failures_total{job="felis-api"}=1`;
|
||||
4. `FelisImageBuildFailures` pending(activeAt 08:28:14Z)→ **08:33:14Z 准时 firing**(`for: 5m` 精确到期),labels/annotations 完整。
|
||||
- **清尾(残留全清)**:whitelist `e2e/alert-drill` 摘除(204);`sub-alertdrill` 目录、`/tmp/drillctx`、两个演练 Job、tmux `prom`/`fwd`/`alertpoll`、`/root/prom-drill`(promtool+prometheus 二进制)全删;DB `%alert-drill%` 行删除(whitelist/build 复核 count=0);宿主无残留监听/进程。演练期间 live api 进程内计数器=1(重启归零,属演练事实)。
|
||||
- **可达性追加**:#44 定级 ①(轻)——已装机环境重跑 `felis setup` 完成 storage/connect reconfigure 即触发。
|
||||
|
||||
### 本轮新增真机证据(第二十八批:缺陷 #45 —— 审核门"盲批":评审看不到将被执行的 recipe)
|
||||
|
||||
- **缺陷 #45(构建 lane·审核语义)**:被执行的 Dockerfile 永远来自**上传上下文压缩包根目录的 `Dockerfile`**(`build/jobspec.go` 钉死 `--dockerfile=Dockerfile`),API 的 `dockerfile` 字段**仅审计存档**(`submit.auditDockerfile`、`build.Request` 注释均已声明)——但审核者没有任何路径能看到它:`GET /api/v1/submissions` 不含 blob 内容、面板只显示 `context_ref` 文本、内部面取件路由是 service-token(评审用不了)→ "人工审核是门禁"事实上是**盲批**。同批口径缺口:`POST /api/v1/images/build` 的 `dockerfile` 字段在 OpenAPI 里无任何说明(易被当成"将被执行"),面板表单也把该框呈现为"Dockerfile 内容 *"。
|
||||
- **修复(`168a375`)**:
|
||||
- 新增 admin-tier `GET /api/v1/submissions/{id}/context`:评审下载与构建 Pod 同源同字节的 `context.tar.gz`;`Content-Disposition: attachment` + `nosniff`(攻击者提供的归档只下载、不渲染);审计 `submission.context.download`(actor=评审者、target=submission id)。
|
||||
- 内部面取件 handler 共享 `openSubmissionContext`/`streamSubmissionContext`(行为不变,原测试锁定)。
|
||||
- OpenAPI:新路由 + `/api/v1/images/build` 字段描述补全(明说"执行的是 context 根目录的 Dockerfile;`dockerfile` 仅审计")。
|
||||
- 面板:SubmissionsPage 展开区新增「下载上下文」按钮(spinner/错误呈现,i18n en/zh);ImageBuildPage 表单补审计说明行。
|
||||
- 测试:`TestAdminSubmissionContextRoute`(流式/404/503)+ admin-only 矩阵加该路由 + OpenAPI parity 强制文档;面板 117 单测 + 构建、go vet/go test 全绿。
|
||||
- **真机验证(auditfix41 已部署;owner 会话经 op-login + 内部面代 approve 重铸)**:
|
||||
- admin 下载 `sub-bf7dc18e9dd97dc2` → **200**,`attachment; filename="context.tar.gz"`、`application/gzip`、`nosniff`;sha256 `205496f2…` **与 uploads PVC blob 逐字节一致**;
|
||||
- 内部面(Bearer=felis-service-token)同 blob → 200 + 同 sha256(重构未破坏构建取件路径);无 token → 401;
|
||||
- 有效玩家会话(console host 隔离,仅 adminOnly 生效)→ **403**;匿名 → 401;不存在 id(admin)→ 404;
|
||||
- 审计落账:`audit_logs` = `[email protected] | submission.context.download | sub-bf7dc18e9dd97dc2`;
|
||||
- 面板产物:服务端 index.html 引用新构建 `index-CwFSKpgW.js`,bundle 内含新按钮逻辑(grep 命中 3 处)。
|
||||
- **可达性追加**:#45 定级 ①——每一次真实的"用户提交 → 管理员审核"都会踩到(审核者此前无法查看将被执行的内容)。
|
||||
- **hook 链补齐(可复用)**:staff 账号走邮件登录门会被设计拒绝(refuse staff)→ owner 会话铸法:`op-login/start`(email=felis-owner@example.com)→ VM 日志 grep `email-otp` 取码(no-Mailer fallback)→ 内部面 `op-login/{id}/approve`(Bearer=felis-service-token;body `approver_uuid`=owner 的 mc_uuid)→ `op-login/finish`(curl -c 存 cookie)。
|
||||
|
||||
### 本轮新增真机证据(第二十九批:镜像耐久落地 —— registry 托管 + 回环拉取路径;#46–#49)
|
||||
|
||||
- **背景(结论清单第 2 条)**:磁盘压力演练证明 kubelet 会 GC 掉"当前无人使用"的镜像 → ImagePullBackOff,恢复依赖人工重导入。本批让镜像自愈:自建镜像全部托管进内建 registry,节点侧 pull 经回环 hostPort(节点 containerd 到 Service VIP 是死路,实测 "Empty reply")。
|
||||
- **平台侧 `a9b275a`**:registry 容器端口加 `hostPort 127.0.0.1:5000`;同提交修 **#46** —— registry 独立资源模板(1 CPU / 2Gi):旧模板 256Mi 在实测推 475MB 层时被 OOM kill(dmesg `oom-kill … registry, oom_score_adj=989`,上传中断),2Gi 下同一推送 2 秒完成。
|
||||
- **安装器侧 `13d64e0` + 文档 `fa0e8d7`**:自建镜像规范 ref = `registry.felis.svc:5000/felis/{felis,limbo,lobby,paper}:demo`;import 进 containerd 就用该名(首启命中本地,免 registry round-trip),`deploy_bundle` 之后统一 `push_images_to_registry` 入仓(推 `127.0.0.1:5000`;registry 只认主机名之后的路径 —— 推/拉落点一致)。`configure_registry_mirror` 写 `registries.yaml`(`registry.felis.svc:5000 → http://127.0.0.1:5000`),内容不变不重启 k3s;`import_registry_image` 预缓存 registry:2(重跑走跳过分支)。迁移 **0021** 把 recommended 白名单重指到 registry ref(线上实查两行已落)。
|
||||
- **真机三次重跑**(`/opt/felis/src` = 765a892 快照;`FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full FELIS_IMAGE=registry.felis.svc:5000/felis/felis:auditfix42`):
|
||||
- run1 失败 = **#47**:Mac tar 的 `._*` 旁文件混入构建上下文,`._0004_*.sql` 被 //go:embed → 新二进制的 `felis migrate` 报 `non-numeric version "."`,安装器停在 run_migrations。修复 `5fa8b74`(.dockerignore 排除 `._*`/`.DS_Store`)。复验方式:故意在暂存树留 `._zz_probe_junk.sql` → 重建后 `migrations applied`(过滤生效)。
|
||||
- run2 失败 = **#48**:每个镜像一对 `systemctl start/stop docker` 触发 systemd 限流(`Start request repeated too quickly / start-limit-hit`),第 4 个镜像(paper)未入仓。修复 `c7e585e`(整批一次 start/stop,单测锚定)。run3 零 `[fail]`:4 镜像全部入仓(push digest ×4 实收)。
|
||||
- run3 收敛实查:api/operator/reaper = `registry.felis.svc:5000/felis/felis:auditfix42`(registry ref 首滚命中本地 import);registry 模板 1 CPU/2Gi 生效;reaper CronJob 同步换 ref 且 nodeSelector 保留;login/lobby Running 于 registry ref;迁移 applied。
|
||||
- **GC 演练(本批验收本体)**:
|
||||
- A 控制面:`ctr images rm …/felis/felis:auditfix42` + `ctr content prune references` → 本地 ref 消失 → `rollout restart felis-api` → 事件 `Pulling` → `Pulled … Successfully pulled image … in 25ms`,ref 恢复、pod Running。
|
||||
- B 游戏:rm `…/felis/lobby:demo` → 删 `lobby-0` → `Successfully pulled … in 10ms … Image size: 182357358 bytes`,Running。
|
||||
- #46 复验:整轮重建 + 4 推送期间 dmesg `oom-kill` 计数不变(仍 2,历史)。
|
||||
- **构建 lane 复验**:`POST /images/build`(context=遗留 `sub-bf7dc18…`;ref `registry.felis.svc:5000/e2e/durable-check2:latest`)→ 202 → succeeded;registry `e2e/durable-check2` tags 落位;Job 事件见 felis/kaniko/trivy 三镜像就绪;白名单条目摘除(204)、Job 清。
|
||||
- **#49(同一条升级路径的第二类静默回退)**:`write_felis_toml` 重写整表([registry] 仅 url+build_namespace、[archive] 仅 store+local_path)→ §8e 的 kaniko/trivy pin、构建上限、uploads 后端、[registry.s3]、reaper 的 retention/warn_before/max_local_bytes 在重跑时全部丢失(S3 安装切回 local、构建回退被 egress 拒绝的上游 executor)。修复 `765a892`+`b8e554d`(沿用 [smtp]/[[auth_source]] 的 carry 模式;url/build_namespace/store/local_path 保持安装器所有)。真机复验:预置 `retention = "30d"`,run3 后 host toml / pod toml / felis-config Secret 三层都在,且构建 lane 直接用 carry 的配置跑通(上条)。
|
||||
- **可达性**:#46 ①(用户构建大层或安装器入仓即触发;实测 475MB 层);#47 ②(Mac 打包树构建路径,真机踩中);#48 ②(安装器重跑,真机踩中);#49 ②(升级=重跑安装器,真机踩中)。
|
||||
- **口径/遗留**:`demo-up.sh` 未改(dev/demo 路径,本地 tag 导入维持原状);kaniko/trivy 编译默认值未动(§8e 改为"镜像进 registry"配方);registry 2Gi 为渲染常量(暂未开 flag);drill 残留:registry 里 `e2e/*` 小镜像留档,`durable-check2` 白名单条目已摘。
|
||||
|
||||
### 本轮新增真机证据(第三十批:缺陷 #50 —— `[smtp]` carry 吞掉下一节的注释块,每次重跑 +1)
|
||||
|
||||
- **发现路径**:为 `felis setup` 首装连续走查做前置盘点时读 live 配置——`/etc/felis/felis.host.toml` 已经堆了 **3 份**、`felis.pod.toml` **4 份**重复的 Yggdrasil 注释块(同一段文案逐次叠加);用脚本自带的提取器实测:一次重跑会把 3 份全部当作 `[smtp]` 内容带走,再叠一份模板注释 → **每次重跑 +1、无上界**(host/pod 增速不同步,现场 3 vs 4 即历史残迹)。
|
||||
- **根因**:`persisted_smtp_block` 打印"`[smtp]` 到下一个 section header 之间"的**所有行**;generated 注释块正好落在这段区间里 → 被吞并。同文件的自称"cached on first call"缓存因写方是命令替换(子 shell)从未生效,pod 写实际上重复抽取刚被重写的 host,加剧了两文件的不同步。纯注释膨胀、无功能损失,但属 #49 同族 carry 语义缺陷(把不属于自己的内容也带走了)。
|
||||
- **修复 `4d3c85f`**:carry 改为白名单(section header + 键行),与 #49 的 `[registry]`/`[archive]` 提取同型;删掉失效缓存说明。`deploy/bootstrap_test.sh` 新增用例:配置值被携带 / 不吞注释行 / 不越节 / 写回后二次抽取**字节稳定**(幂等)。
|
||||
- **真机验证(auditfix43,两次连续全量重跑)**:
|
||||
- run1(`bootstrap-auditfix43.log`):零 `[fail]`;host 注释块 **3→1**、pod **4→1**;与 run 前快照 diff 恰为 21/31 行(全部是被删掉的重复注释)——值零漂移(kaniko/trivy pin、`trivy_db_repository`、uploads local、`[registry.s3]` 空、`retention="30d"`、auth_source 原样、空 `[smtp]` 节保留;两文件 db host 仍分别为 127.0.0.1 / 10.211.55.6)。
|
||||
- run2(`bootstrap-auditfix43b.log`,紧接再跑):零 `[fail]`,仍 **1/1** —— 收敛证明(旧代码此处会 1→2 继续增长)。
|
||||
- 部署同步:api/operator = `registry.felis.svc:5000/felis/felis:auditfix43`;reaper CronJob 同 ref 且 nodeSelector 保留;registry `felis/felis` tags = auditfix41/42/43;panel 200;host 二进制已从新镜像提取。
|
||||
- **可达性**:#50 ②——配置过 email(存在 `[smtp]` 节)的安装,按文档升级=重跑安装器即触发;危害=配置注释无限膨胀(每次 +1),无功能损失。
|
||||
|
||||
### 本轮新增真机证据(第三十一批:首装连续走查 + 缺陷 #51 —— 工作负载 `felis-config` 副本永不刷新)
|
||||
|
||||
**一、`felis setup` 首装单次连续走查(队列第 1 项,完成;0 缺陷)**
|
||||
|
||||
- **装置**:scratch 库 `felis_scratch`(新建 + `felis migrate up` 21 条迁移)+ scratch 配置(真实 pod toml 副本,仅换库名;root 0600);hook 直插 `account_link_codes` 一枚绑定码(等价 `/link` 内网端点写入,代替"进服拿码");tmux 驱动 `felis setup -config`;k8s 只读复用(登录门 Ready 等待通过)。
|
||||
- **连续走查(一条会话走完)**:Preflight(自动:PG✓/迁移 21 applied/面板✓)→ **MC 绑定**(输码 → working →「✓ Owner account is ready.」+ 一次性 setup URL)→ **连接 chooser**(Local)→ **存储 chooser**(Local → working → ~20s 后「✓ Local storage configured.」,含 Secret 重渲染 + API rollout)→ **首装 Summary**(「✓ Setup complete.」+ owner/setup URL/access/storage/panel 卡片 + c/s/e 提示)→ **轨道回顾**(← 依次只读 recap Storage→Connection→Owner→Preflight,→/esc 回到前台)→ Enter 退出 → stdout 汇总(`Owner account … provisioned (passwordless)`、`Recorded as "root"`、setup URL、Admin console)→ **EXIT=0**。
|
||||
- **结果**:移动端/文案/切换全部符合设计,**0 新缺陷**;scratch 库侧复核:owner(role=owner) 1 行、绑定码已消费(0)、setup_tokens 1、`local_auth_enabled=true`。
|
||||
- **副作用与还原(如实记录)**:存储 apply 会把 `/etc/felis` 两文件重写为 Go encoder 形态(无注释、含空值键如 `[archive.s3]`,**值零漂移**:kaniko/trivy pin、uploads local、retention 30d、auth_source 原样),并重渲染控制面 Secret + 滚 API——这是该向导的既定行为;演练后按 pre 快照整文件还原(sha256 逐字节一致),两 ns Secret 重渲染复核一致,scratch 库/配置/hba 行/tmux 全部清理。
|
||||
|
||||
**二、缺陷 #51(工作负载 `felis-config` 副本永不刷新)**
|
||||
|
||||
- **发现路径**:上条还原核对时发现 minecraft ns 的 `felis-config` 是**旧形态**(encoder 式)而 felis ns 已是模板式 → 挖出 `ensureSecretReplica` 的「绝不覆盖既有副本」(凭据语义:防冲掉手工轮换值)把 **felis-config 也纳入只建不更**,而 bootstrap 只 apply 控制 ns。后果:改配置后(DB 凭据轮换、[archive] 保留策略调整、root domain 等)backup/restore/fileedit Job 与 reaper 永远读旧副本 → 静默失效(如备份认证失败)。
|
||||
- **红证据(auditfix43,真机)**:向 minecraft 副本注入 `# drill-51-stale-marker` → 运行 `felis setup` → 输出 `- config (minecraft ns): skipped (already exists)`;副本 sha `5ec2ff6e…` 保持,控制面 `e4791fe1…` 不同(陈旧坐实)。
|
||||
- **修复 `328e570`**:① setup 侧:`ensureSecretReplica` 增 `refreshExisting`(仅 felis-config 传 true)——源缺失降级 skip、内容一致 skip(`already current`)、不同则原地 Update;凭据类保持 create-if-absent;新增 `updated` 结果与「refreshed from the control namespace」文案。② bootstrap 侧:新增 `apply_felis_config_secrets()`,每次运行同时 apply 控制 ns + 工作负载 ns 两份(同渲染自最新 pod toml)。单测:Go 新增 4 例(陈旧刷新/一致跳过/空键补写/源缺失降级);`bootstrap_test.sh` 新增 4 断言(两 ns apply、同一 pod toml 渲染、恰 2 次 apply)。
|
||||
- **绿证据(auditfix44,真机)**:A) 安装器重跑(标记仍在副本中)→ 日志出现 `secret/felis-config configured`(工作负载 ns 被刷新)→ 副本 sha 与控制面一致、标记 0;B) 再注入标记 → 运行**新** `felis setup` → 输出 `- config (minecraft ns): refreshed from the control namespace`、凭据仍 `skipped (already exists)`、副本 sha 一致、`EXIT=0`。部署=auditfix44(api/operator/reaper),panel 200,host 二进制随镜像 `docker cp` 刷新。
|
||||
- **可达性**:#51 ②——升级=重跑安装器、或改完配置跑 setup 即触发;旧行为下 backup/reaper 静默使用旧配置(DB 轮换后备份全挂)。
|
||||
|
||||
**三、环境修复(非产品)**:现场 pg_hba 缺 `felis_pgint` 规则(按文档走 ssh 隧道跑 pgint 会 ident 失败)——补回 `host felis_pgint felis 127.0.0.1/32 scram-sha-256` 并复测连接成功。
|
||||
|
||||
### 本轮新增真机证据(第三十二批:S3 存储向导逐屏走查收尾 + 缺陷 #52 —— 向导 apply 不刷新工作负载 `felis-config` 镜像)
|
||||
|
||||
**一、S3 存储向导屏走查(队列第 2 项,完成;0 功能缺陷,走查自身暴露镜像缺口 → #52)**
|
||||
|
||||
- **装置**:VM 宿主 MinIO 容器(`quay.io/minio/minio`,`felis`/`felis-drill-9000`,:9000)+ bucket `felis-wizard-uploads`;tmux 驱动 `felis setup` 重跑向导;行动前先留 preS3 快照(两 toml + 两 ns Secret + sha256)。
|
||||
- **负例**:错误凭据 → `✗ Could not save storage settings.` + `submit: s3 credentials rejected: The Access Key Id you provided does not exist`(`CheckS3Access` 预检先于一切写入);**零副作用**(无 Secret、两 toml sha 不变);`esc` 返回编辑时已填值保留(密钥掩码)✓
|
||||
- **正例**:修正凭据 → ~10s working → `✓ Object storage configured.` → Enter → 状态屏 `storage s3://felis-wizard-uploads · http://10.211.55.6:9000` ✓
|
||||
- **落地核对**:`felis-uploads-s3` Secret(access_key_id/secret_access_key);两 toml `user_uploads_context` + `[registry.s3]` endpoint/refs;API 滚动;启动日志仅既有的 smtp/jwks 警告 ✓
|
||||
- **功能链(batch24 同款)**:player.test 邮箱 OTP 登录(hook 取码)→ `POST /api/v1/me/submissions` 201 → context 上传 200 → MinIO 桶实见 `sub-853a4e2ba4ba6443/context.tar.gz` → 内部面取回 200 + tar 内容正确 ✓
|
||||
- **UI 回滚(`s` → Local)**:两 toml 归位(`/var/lib/felis/uploads`、`[registry.s3]` 归空)、控制面 Secret 更新、API 滚毕;与 preS3 快照的差异仅「s3 字段回环 + encoder 形态」(注释丢失属该向导既定行为)——**值零漂移** ✓
|
||||
- **观察(不计缺陷)**:切回 Local 后 `felis-uploads-s3` Secret 残留(演练按清理流程删除;是否自动清理属产品取舍)。
|
||||
|
||||
**二、缺陷 #52(向导内 apply 只刷新控制面,工作负载镜像滞后到下一次运行)**
|
||||
|
||||
- **发现路径**:回滚后按计划复核「两 ns 重渲染」——minecraft 副本仍为 S3 内容(`4fb80ed8…`),而控制面与两 toml 已 local(`df206074…`)。
|
||||
- **红证据(auditfix44)**:① S3 方向:S3 apply(20:18:44)后副本停在 local 内容(20:22:53 快照 = `e4791fe1…`)达 4 分钟;② Local 方向:回滚 apply(20:23:41)后副本停在 S3 内容。副本 managedFields 两笔写入(12:13:58Z `kubectl-client-side-apply` = 安装器 rerun 的双 ns apply;12:23:06Z manager `felis` = 下一次 setup 启动的 refresh)都不是 apply 时刻——**apply 本身不碰副本**。
|
||||
- **根因**:`applyFelisConfigSecret`(storage/connection/edge 三条 apply 的公共出口)只 apply 控制 ns;镜像刷新只存在于 `felis setup` 启动(#51)与安装器。email 路径早有显式镜像刷新,storage/connection 是漏网的两条。
|
||||
- **修复 `de7fb2c`**:镜像刷新移入 `applyFelisConfigSecret`(best-effort + stderr 警告;控制面-only 安装无工作负载 ns 时降级不阻塞);smtp helper 去掉重复块。门禁:`gofmt`/`go vet`/`go test ./...`/`bootstrap_test.sh` 全绿。
|
||||
- **绿证据(`v0.0.0+fix52`,宿主二进制 sha `0bd49467…` 已装 `/usr/local/bin/felis`,旧版留 `/root/felis-auditfix44.bin`;真机双向)**:Run1 从 Local `s`→S3:apply 后 `control = mirror = podtoml = 4fb80ed8…`(S3 渲染;旧代码此刻镜像会停在 local);Run2 `s`→Local:`control = mirror = podtoml = df206074…`;副本 managedFields 写者 = `kubectl-client-side-apply` @ 12:33:40Z / 12:34:59Z(正是 apply 时刻)。Run1 启动块另见 `config (minecraft ns): refreshed from the control namespace`(#51 机制照常先收敛一次旧账)。
|
||||
- **可达性**:#52 ②——任何 `s`/`c` 重配置即触发;危害等级低(镜像消费者 backup/restore/fileedit/reaper 当前不读被改动字段——`UserUploadsContext` 仅 `api.go` 消费——但「配置动了、镜像没动」正是 #51 要消灭的静默滞后类,且与 email 路径的既定行为不一致)。
|
||||
|
||||
**三、清理与还原**:MinIO 容器/卷/两镜像、`felis-uploads-s3` Secret、DB 行(`image_submissions` `sub-853a4e2ba4ba6443`)、/tmp 残留(player-cookies/drill-ctx/svc-tok 等)全清;docker 停;收尾 `control = mirror = df206074`(ALIGNED)、API 滚毕 Running、panel/healthz 200 ✓。
|
||||
|
||||
### 本轮新增真机证据(第三十三批:`/updates` 维护窗口 API+UI 走查全绿;#53 跟随仓库迁移;#54 update 升级指引全假)
|
||||
|
||||
**零、仓库迁移(背景)**:remote 已改 `[email protected]:FelisMC/Felis.git`(本批两个修复随 `2c6739a`、`397a400` 直推 main)。带 token 实测:旧 `MliroLirrorsIngenuity/Felis` 路径 301(GitHub rename redirect)、新路径 200——旧坐标当前仍能工作但全靠 redirect。新仓库**尚无 stable release**(`releases/latest` 带 token 也 404):felis-api 更新检查的 404 属发布流程事实,非代码缺陷。
|
||||
|
||||
**一、`/updates` 维护窗口 API 走查(完成;全绿)**
|
||||
|
||||
- 读:GET 未设置 → `{"not_before":null,"not_after":null}`。
|
||||
- 负例全按预期拒绝:半设 / 倒序 / 相等 / 坏 JSON / 未知字段 → 400;缺 `Content-Type` → 415。
|
||||
- 正例:写入 → 回读一致 → **API pod 重启后仍在**(已落 DB,非内存态)。
|
||||
- 鉴权:无 cookie → 401;player 会话 → 403(新铸 player 会话,留 `/tmp/player-cookies.txt`)。
|
||||
- 收尾:DB 回到 `{null,null}`。
|
||||
|
||||
**二、`/updates` 面板 UI 走查(CDP,完成;全绿零 console 错误)**
|
||||
|
||||
- 状态流转逐屏:生效中 → 过期 → 计划 → 未设置(截图 `/tmp/updates-{1..5}-*.png`)。
|
||||
- 倒序提交 → 校验文案正确;清除后表单清空 + 「未设置」提示。
|
||||
- 核账:DB 收尾 `{null,null}`;审计 `updates.window_set` 恰 3 笔(20:46:18 / :20 / :21)。
|
||||
- 驱动:`/tmp/cdp-updates2.js`(bun + 原生 WebSocket;须 `Network.setCookie` 注入 owner 会话,否则新标签页 401 跳登录)。
|
||||
- 观察(不计缺陷):窗口自身存取已验证;「窗口被 runner 消费」的端到端链路(Applier/Notifier)仍属 INTEGRATION-ONLY 设计(deferred-seams),不要当缺陷重复修。
|
||||
|
||||
**三、#53(旧仓库坐标残留)—— 提交 `2c6739a`**
|
||||
|
||||
- 红(`v0.0.0+fix52` 实机):`felis update` → `github: MliroLirrorsIngenuity/Felis releases/latest returned HTTP 404 — …`。
|
||||
- 绿(`v0.0.0+fix53` 实机):同一命令 → `github: FelisMC/Felis releases/latest returned HTTP 404 — …`。
|
||||
- 范围:`internal/updater/topology.go`、`deploy/bootstrap.sh`(默认 `FELIS_REPO_URL` + 两处 UA)、README ×2、两个测试夹具(6 文件 9 处);门禁四件套全绿。
|
||||
- 可达性:②——默认安装 / 每次 `felis update` 都读该坐标;危害在 redirect 退休时兑现(安装与更新一起挂)。
|
||||
|
||||
**四、#54(`felis update` 的 apply 指引在既有安装上全是错的)—— 提交 `397a400`**
|
||||
|
||||
- 红(`v0.0.0+fix52` 实机文本):`--panel --force` → `run: sudo felis setup` + “felis setup is idempotent and re-runs the installer…” + felis-api「例外」块。
|
||||
- 决定性证据(完成态安装):`felis setup` 跑 8s 退出,`/opt/felis/velocity/velocity.jar` mtime/hash 不变、零 bootstrap 输出;同刻安装器重跑日志有 `resolving the newest Velocity 3.5.1 build`。代码侧 `shouldRunHostBootstrapBeforeConfig` 仅在 4 个 marker 不全时进 bootstrap——既有安装上 `felis setup` 只开配置 TUI。
|
||||
- 绿(`v0.0.0+fix54` 实机;宿主二进制已换 `/usr/local/bin/felis`,fix52 留档 `/root/felis-fix52.bin` sha `0bd49467…`):
|
||||
- `--panel --force` / `--velocity --force` → `run: curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash` + 单条 trailer(channel 未持久化 caveat、私有仓库 token'd form、`felis setup is not this path`);
|
||||
- `--mc` 无命令无 trailer;`--all` trailer 恰一次;`--force`/`LatestKnown` 语义不变。
|
||||
- 文案同步:`docs/troubleshooting.md §15` 删掉 “or `sudo felis setup`”、补 channel caveat;测试改为 `TestApplyGuidancePointsEveryComponentAtTheInstaller`(钉安装器路径 + 禁 `run: sudo felis setup`)。
|
||||
- 可达性:②(文档/指引;同 #38 型)。
|
||||
|
||||
### 本轮新增真机证据(第三十四批:`felis nano` 全链走查 60/60;缺陷 #55 —— 坏源静默挡住验证梯子)
|
||||
|
||||
**零、装置(全部在 VM,不进产品环境)**:`/srv/nanotest/nano-stub.py`=可控 Yggdrasil 源,路径即行为——`/ok/<name>`、`/okid/<name>/<id>`、`/props/<name>`、`/badname/<case>`(tooshort/toolong/badchar/section/noname)、`/none`=204、`/boom`=500、`/redir`=302、`/garbage`、`/emptyid`、`/slow/<secs>`;跑在 127.0.0.1:9900,临时单元 `nano-stub`,请求日志 `/srv/nanotest/stub.log`。驱动脚本:`nano-matrix.sh`(HTTP 矩阵 S1–S7)、`nano-service.sh`(systemd 层)、`nano-config.sh`(配置校验)、`nano-fw.sh`(firewalld);`nano-harness.sh` 从 HEAD 版 `bootstrap.sh` 提取产品函数在 scratch 路径跑(`felis-nano-test` 改名件,不碰产品单元)。
|
||||
|
||||
**一、HTTP 矩阵 S1–S7(60 项断言)**
|
||||
- 红(修复前 `v0.0.0+audit-nano1`):`SUMMARY pass=47 fail=13`——13 红全在 #55 语义内:5 个坏名用例「应 503+日志、实得静默 204」(状态+日志 ×2=10),failover 3(坏源在前登录不落第二源 ×2、无日志 ×1)。留档 `matrix-red.r2.log`。
|
||||
- 绿(fix55,同装置重跑):`SUMMARY pass=60 fail=0`。留档 `matrix-fix55.r2.log` 与 `matrix-fix55/`(更早原跑在 `matrix/`)。
|
||||
- 覆盖面:S1 输入校验(缺参/超长→204、POST→405、未知路径→404、带体 GET→400、HEAD→204、400 带 `Connection: close`)、S2 三方登录全链(前缀 + felisAuthNS UUID 确定性、ip/serverId 转发保真)、S3 premium 撞名改名(`LS_`)与免费名不动、S4 失败模式(不可达/500/302/garbage/204 → 跳过、记日志、503)、S5 多源优先级 + 坏源后 failover、S6 日志纪律(长 URI 截断、控制字节转义)、S7 properties 转发。
|
||||
|
||||
**二、systemd 层(`nano-service.sh`,真 unit + 真 DynamicUser;重跑落盘 `nano-service.r2.log`,0 FAIL)**:装服务→起服务;重跑语义(配置字节不动、`resolve_nano_listen` 从 unit 读回端点);drain(stop 期间 in-flight 登录照样答完,实测 stop 等待 5335ms);SIGKILL 后自动重启;端点迁移 8081→8099 后新址应答、旧址关闭;三种启动文案(loopback「Bound to loopback」/ CIDR「admits」/ 无 CIDR「WARNING」)。
|
||||
|
||||
**三、配置层(`nano-config.sh`,LoadNano 16/16,落盘 `nano-config.r2.log`)**:未知键、空/重复/保留 tag、colon、空白、坏 prefix、重复 prefix(大小写不敏感)、非 http scheme、URL 带 query、明文公网 http、缺文件、`[server] listen` 忽略警告、Mojang-only 可服务。
|
||||
|
||||
**四、firewalld(`nano-fw.sh`,3/3,落盘 `nano-fw.r2.log`)**:关旧「开全源」8081/tcp、仅对 CIDR 开 rich rule、loopback 不开洞、无 CIDR 警告、快照清理还原。
|
||||
|
||||
**五、缺陷 #55(提交 `9dad61f`)**
|
||||
- 红(fix54 实机):配置源回 200 但档案字段不可用(三方源名字进不了 MC 字符集;身份源 UUID 不解析)→ 静默 204、零日志;且筛查在 `handleHasJoined` 直接 `return` → [坏源, 好源] 顺序下 204,好源根本没被询问。
|
||||
- 根因:两个 200-后筛查在 handler 层(200 已赢下梯子之后),而同类坏答案(200 无档案/非 200/不可达)在 `resolveHasJoined` 里是「记日志 + skip + failed」——一致性缺口。
|
||||
- 修复:筛查移入 `resolveHasJoined`(身份源 `uuid.Parse`、三方源 `mcUsernameRe`)→ 命中记日志(`unusable profile name` / `unparseable profile id`)+ `failed=true` + `continue`;handler 守卫保留为最后防线(注释更新)。单测 +3(含 `log.Writer()` 捕获断言)。
|
||||
- 可达性:#55 ②——需要配置了一个「回 200 但字段不可用」的源(马虎的自建/三方 Yggdrasil);坏源在前时经梯子的登录全部被吞(静默、零日志),安全侧(坏名不落玩家列表)保留。
|
||||
|
||||
**六、观察(不计缺陷)**:premium 冷查询 fail-closed 抖动——api.mojang.com 从 VM 首查偶发接近 2s 超时 → 偶按「疑似 premium」给免费名加前缀(`FelisNanoStub1` → `LS_FelisNanoStub`);符合 `isPremiumName` 既定取舍(误加前缀=外观代价,误放行=抢名),记观察不修。
|
||||
|
||||
### 本轮新增真机证据(第三十五批:NetworkPolicy 真机强制矩阵全绿 —— 端到端白名单闭环 + felis-velocity 刷新存活验证)
|
||||
|
||||
**零、装置与现场**:三张策略 live 于 `minecraft` ns(`felis-default-deny-ingress` / `felis-allow-rcon-from-control-plane` / `felis-allow-game-from-velocity`;apply 于 09-22T05:37:24Z,与渲染收敛 diff 零漂移);k3s 参数无 `--disable-network-policy`,**强制执行实测生效**;kube-router 机制实证:per-pod `KUBE-POD-FW-*` 链、未标记流量 `REJECT --reject-with icmp-port-unreachable`、每链首条 `--src-type LOCAL -j ACCEPT`(本节点豁免)、ipBlock 落 ipset `KUBE-SRC-*`(白名单成员可直接查)。探测法:**nsenter 进真实 pod 网络命名空间(源 IP=真实 pod IP)+ netns/veth 合成「转发型外部源」10.99.0.2(模拟 velocity 在另一台机器)**;全程不改产品代码。
|
||||
|
||||
**一、pod 源矩阵(源=真实 pod netns)**
|
||||
- api(felis,命中 RCON 白名单标签)→ lobby:25575 = **OPEN**;同源 → lobby:25565 = **REFUSED**(端口级区分 ✓)
|
||||
- registry(felis,非匹配)→ lobby:25575/25565 = **REFUSED**;同 pod → api:8081 = OPEN(对照:无策略命名空间不受限)
|
||||
- coredns(kube-system)→ login:25565 = **REFUSED**
|
||||
- 收尾复跑全矩阵与首轮逐行一致(`np-matrix-run1/2.log`)。
|
||||
|
||||
**二、host(=velocity 同机侧)**:→ login/lobby 的 ClusterIP 与 podIP :25565 **全 OPEN**(velocity 注册的后端路径实际可用);→ api-internal:8081 OPEN;→ lobby:25575 亦 OPEN——归因:kube-router 每 pod 链的 `--src-type LOCAL -j ACCEPT`(**本节点流量豁免,kube-router 设计行为**,kubelet 探针等依赖它;netpol 语义无法对节点自身收口,节点 root 本在 TCB 内)。**注记①:node-local 豁免。**
|
||||
|
||||
**三、转发型外部源(最严苛模拟)**
|
||||
- 基线:netns(10.99.0.2) → 全部服务器端口(podIP 与 ClusterIP、25565/25575)= REFUSED,包级可见 netpol 的 icmp-port-unreachable(`td3.log`)。
|
||||
- **白名单闭环**:临时把 10.99.0.0/24 加入 allow-game → ipset 即时生效(members:`10.99.0.0/24` + `10.211.55.6`)→ **lobby-svc:25565 与 login-svc:25565 = OPEN**(ClusterIP=velocity 实际拨号形态);25575 仍 REFUSED(端口维度不破)→ 回滚 → spec 哈希逐字节一致(md5 `e6b07440…`)、ipset 复原、复测全部 REFUSED。
|
||||
- 层间归因:firewalld 规则含 `ct status dnat accept`(**DNAT 后的服务流量放行**;非 DNAT 转发走 forward policy 的 `admin-prohibited` 拒绝)——受支持拓扑(velocity 同机=LOCAL、ClusterIP、Mac→NodePort 面板 200)全部实测可用;「外部未经服务直连 podIP」不属于任何产品流。**注记②:firewalld 只放 DNAT 服务流。**
|
||||
- 装置保养:veth 未归区时 firewalld 会拒其转发属装置噪声(已归因);public 区临时挂载已摘、netns/veth 已删、策略零残留(spec diff 为空)。
|
||||
|
||||
**四、felis-velocity 刷新循环存活(顺带验证)**:20:18–20:42 的 `server list refresh failed` 全部落在 #52 演练的 API rollout 窗口(成功不打日志属设计);tcpdump 450s 窗口抓到 52 条 `GET /api/v1/servers` 载荷行(成对出现=双抓包点看到同一请求,折算约 26 次 ≈ 每 15–17s 一次,与 `REGISTRATION_REFRESH=15s` 常量吻合)及对应 200 响应;"keeping current registrations" 为设计降级。非缺陷。
|
||||
|
||||
**五、留档(VM `/srv/npdrill/`)**:`np-matrix.sh`+`np-matrix-run1/2.log`、`np-netns.sh`、`netns-probe.py`、`np-cidr-test.sh/.log`、`spec-before/after.json`、`td3.log`(包级归因)、`tcpdump-8081.log`(刷新存活)、`iptables-save.txt` 与 `nft-rules.txt`(现场快照)。
|
||||
|
||||
### 本轮新增真机证据(第三十六批:面板错误文案全量本地化(#60)+ Run4c 管理面写操作复核 0 缺陷)
|
||||
|
||||
**零、现场**:三个缺陷批次(#56–#59)已按「一缺陷一 commit」推 main 并部署 `auditfix59`;本批完成 #60 后部署 `auditfix60`(api/operator/reaper 三处 + 宿主 CLI,`felis version` = v0.0.0+fix60)。面板门禁:typecheck 0 错、vitest 117/117、vite build 通过(注意项目测试是 `bun run test`=vitest run;裸 `bun test` 是 Bun 内置 runner,解析不了 `@/` 别名,41 个 fail 系假象,勿误报)。
|
||||
|
||||
**一、Run4c 管理面写操作复核(4 段全真机,最终 0 缺陷;3 处为测试脚本自身误报,均已自纠)**
|
||||
1. **镜像添加+删除**:添加 `registry.felis.svc:5000/e2e/probe:1` → 列表出现、API 落库(`added_at=14:50:53Z`)✅。删除初测"未生效"系脚本用 `tr` 找行——该表是 `div` 网格,`tr` 选择器命中 0 → 从未点到按钮。换 `div[class*=grid-cols-12] → button[title]` 重测:confirm("确定删除此镜像吗?")自动接受 → 页面 1ms 内刷新、**API 侧同 ref 从 5 条降至 4 条**——删除通道无缺陷。
|
||||
2. **构建负例**:外部 registry(`docker.io/...`)+ 不存在 context → 对话框红字拒绝 `build: invalid request: image reference … must target the internal registry "registry.felis.svc:5000"`(截图 `run4c-3`)。脚本"未捕获"系其正则先命中了侧边栏"镜像"二字——误报。
|
||||
3. **创建重复用户**:对话框内正确显示 **"该用户名已被使用。"**、对话框保持打开(截图 `run6-1`)。此前 run4c 的"dialog-closed"系脚本点错页面(进到了用户详情页);且该轮实际是一次**正例**:真管理员的 username 是 UUID 串(`08595879-…`,role=`owner` 是角色名),字面用户名 `owner` 当时空闲、创建确实成功——测试件已删(DELETE→200,列表 7→6)。
|
||||
4. **创建重复子域名**:正确拒绝 **"该子域名已被占用。"**(截图 `run4c-5`)✅
|
||||
|
||||
**二、缺陷 #60(提交 `c9af548`):37 个用户可达错误码显示生英文**
|
||||
- 发现路径:run4c 复核 `email_taken` 时做了一次系统性对照——后端 `newError` 共 **71 个错误码**,面板 `humanizeError` 只映射 **23 个**;其余走 `default` 分支直接透传 `err.message`(多为 Go 包装嵌套的英文,如 `invalid server name: naming: invalid server name "BadName": must match ^[a-z0-9-]{3,32}$`)。
|
||||
- 三个代表真机验证(修复前→修复后):
|
||||
- `bad_name`:新建服务器"名称"填 `BadName`(前端只查非空)→ 前:对话框直出上述嵌套英文;后:**"服务器名称不合法:需为 3–32 位小写字母、数字或连字符,且不能使用保留名。"**(`run8-1`)
|
||||
- `bad_subdomain`:子域名填 `ab`(前端 RE 允许 1–2 字符,后端要求 ≥3)→ 后:**"子域名不合法:…"**(`run8-2`)
|
||||
- `email_taken`:DB hook 造"他人已验证邮箱" → 账户页真发码(no-mailer 日志取码)→ verify 409 → 后:**"该邮箱已在其他账户上完成验证;请直接用该邮箱登录,或换一个地址。"**(`run9-1`)
|
||||
- 修复:api.ts +37 case、zh/en errors.json 各 +37 键(纯映射,无逻辑变更)。**剩 11 个故意不补**:`bad_request`/`conflict`/`internal`/`panic`/`not_found`(通用兜底)、`forbidden`/`not_admin`/`unauthorized`(状态码分支已在 default 覆盖)、`not_ready`(内部面)、`setup_required`(bootstrap 期 SPA 流控信号,403 文案可用)、`unsupported_media_type`(CSRF 门,面板永不可达)。
|
||||
- 遗留观察(不计缺陷):构建负例的英文前缀 `build: invalid request:` 仍会透传——该错误无独立码、message 即最终文案;对管理员可读,记观察。
|
||||
|
||||
**三、观察(不计缺陷)**
|
||||
- UserDetailPage 对 owner/对自己都显示删除按钮;真的点了会得 403 + 具体 message("the owner account cannot be deleted from the panel"),但前端 403 分支统一显示"你无权执行此操作"——笼统但不算错,记观察。
|
||||
- 字面用户名 `owner` 可注册(用户名无保留名单;角色由服务端管理、无提权路径),记观察。
|
||||
|
||||
**四、留档(Mac)**:脚本 `/tmp/cdp-run5.js`(镜像删除复测)、`run6.js`(重复用户复测)、`run7/run8.js`(bad_name/bad_subdomain 前后对照)、`run9b.js`(email_taken,内置 ssh 取码);截图 `run5-1`/`run6-1`/`run7-1/2`/`run8-1/2`/`run9-1`。VM `/opt/felis/src` = `c9af548` 快照。
|
||||
|
||||
### 本轮新增真机证据(第三十七批:管理面交互收尾 0 缺陷 + 缺陷 #61 —— LuckPerms 写操作的过度承诺)
|
||||
|
||||
**零、现场**:`auditfix61`(api/operator/reaper 三处 + 宿主 CLI)。面板门禁 typecheck/vitest 117/build 全绿后部署。
|
||||
|
||||
**一、管理面交互收尾(3 项全真机,0 缺陷)**
|
||||
1. **submissions approve/reject**:player 会话(邮箱 OTP 铸造,no-mailer 日志取码)现场造两条 pending(`sub-489eda47fe4bc12d` / `sub-b1dcc4b4a2cc5f16`,各上传 tar.gz context)→ 面板「通过」→ 状态变「审核通过」且**自动构建 `bld-1790176562566802097` 到 succeeded**(点击到构建完成全链闭环);「驳回」→ 对话框必填原因 → 状态变「已拒绝」。全程 UI 零报错。
|
||||
2. **用户会话撤销**:player.test 铸 2 条新会话(共 4 条)→ UI「单独撤销」首条 → **精确生效**(cookie5 → 401、cookie4 → 200、列表 4→3);「全部撤销」→ 提示「所有会话已撤销。」、列表清空、cookie4 也 401。
|
||||
3. **ServerCard 启动交互**:面板点 test-one「启动」→ `Starting` → `Running/ready`。
|
||||
|
||||
**二、缺陷 #61(提交 `b4ef42d`):LuckPerms 页对写操作"过度承诺"**
|
||||
- 发现路径:LP 页写操作交互测试(给 `E2E_Tester` 写权限)→ 历史卡显示绿色 success +「(服务器未返回输出)」→ **落盘取证发现根本没写入**。
|
||||
- 真机实验矩阵(test-one,LP 5.5.85 / H2 存储):
|
||||
| 操作 | RCON 回包 | 落盘(`lp export` 实测) |
|
||||
|---|---|---|
|
||||
| 面板 UI:`E2E_Tester permission set e2e.ui.write.test` | 空 | ❌ |
|
||||
| 面板控制台:`E2E_Tester … set probe.test true` | 空 | ❌ |
|
||||
| 面板控制台:`<UUID> … set probe2.test true` | 空 | ✅ |
|
||||
| **独立 Python RCON 客户端**(绕开 felis):`E2E_Tester … set probe3.test` | 空 | ❌ |
|
||||
| 独立客户端:`<UUID> … set probe4.test` | "Another command…"(异步提示) | ✅ |
|
||||
- 结论:**felis 只如实转发空回包;名字写入的静默失败是 LuckPerms 自身行为**(独立客户端 1:1 复现)。
|
||||
- 根因(LP config.yml 官方注释背书):`use-server-uuid-cache: false`(LP 默认)→ "commands using a player's username will not work **unless the player has joined since LuckPerms was first installed**"——**未进过服的玩家名永远无法解析**(与有无外网无关)。
|
||||
- 面板问题:旧文案承诺"授予与撤销操作仍然会实际生效"(#59 遗留半句)→ 对"给还没来过的玩家预授权"场景是不成立的承诺。
|
||||
- 修复(纯文案,双语):`luckperms_no_reply` → "……按玩家名的操作只对「自 LuckPerms 安装以来进过本服」的玩家可靠,对没进过服的玩家名可能静默不生效";`luckperms_no_output` → "(服务器未返回输出,无法确认结果)"。真机复验(`run16b`):读提示与写占位均按新文案显示。
|
||||
- 可达性:#61 ①——给未进服的玩家预授权是日常动作,此前面板显示绿色成功构成误导。
|
||||
- 局限(记观察):读提示块只在"读取为空"时渲染;已有部分玩家数据的服上不会露出这段说明(后续增强候选:常驻说明或写前对未解析名字的提示)。
|
||||
|
||||
**三、观察(不计缺陷)**
|
||||
- LP 命令是**串行异步**执行:快速连发第二条会得到 `§7[§b§lL§3§lP§7]§r §7Another command is being executed, waiting for it to finish...`(带颜色代码原文,面板如实显示);`lp export` 回包时有时无(响应与执行解耦)。felis 层无责。
|
||||
- submissions 徽章(`submissions.json`:"审核通过/已拒绝")与过滤标签(`admin.json`:"已通过/已驳回")两套词,语义均可,记观察。
|
||||
- 控制台对空回包命令只显示 echo、无占位提示(终端风格);LP 页有占位文案。记观察。
|
||||
|
||||
**四、留档(Mac)**:`/tmp/cdp-run10.js`(approve/reject)、`run11a/11b.js`(会话撤销)、`run12a.js`(启动交互)、`run13/14/15/15b/16/16b.js`(LP 全链)。VM 独立探针:`/tmp/rcon_probe.py`、`/tmp/rcon_one.py`(+ `/tmp/rcon_pw.txt`);LP 导出样本在 `/data/plugins/LuckPerms/luckperms-2026-09-23-15-*.json.gz`。测试数据:两条 pending 提交(一 approved 一 rejected,构建 succeeded/failed 各一)。VM `/opt/felis/src` = `b4ef42d` 快照。
|
||||
|
||||
### 本轮新增真机证据(第三十八批:#56–#59 证据回填 + 服务器详情/运维面收尾 0 缺陷)
|
||||
|
||||
**零、现场**:`auditfix61`(api/operator/reaper 三处 + 宿主 CLI,`felis version` = v0.0.0+fix61)。本批两部分:把 #56–#59 四项修复的真机证据落节(此前只存在于 commit message),并把队列里剩余的服务器详情页/运维面交互复跑做完(0 缺陷)。
|
||||
|
||||
**一、缺陷 #56–#59 证据回填(部署 `auditfix59`;四项均 = 修复前真机触发 + 修复后复验)**
|
||||
|
||||
1. **#56(`0790f8d`)玩家管理操作把 RCON 回包扔掉、只报 canned 成功。** 触发路径:给"还没进过服"的玩家加白名单/封禁——vanilla 对没见过的名字回 `That player does not exist` 并拒绝,而旧面板无论如何都显示成功。修复后回包逐字上屏:`whitelist add E2E_Bad` → `Added E2E_Bad to the whitelist`(磁盘同步真写入);负例 `NoSuchPlayerXYZ` → `That player does not exist`(不再伪装成功);静默服务器回退本地化文案。
|
||||
2. **#57(`70c988e`)控制台命令的回复无处显示。** 触发路径:控制台发任何命令——`sendCommand` 拿得到 RCON 回包但被丢弃、pod 日志也不回显命令输出,等于零反馈。修复后 echo + 回包以终端样式渲染在提示符上方(实测 `list`)。
|
||||
3. **#58(`4d4cdd6`)无世界盘服务器的文件页 90s 卡死。** 触发路径:对从未启动/已回收的服点"文件"页——旧行为建 Job → Pod `FailedScheduling (pvc not found)` Pending 到 90s 超时 → 误导性 504 `files_timeout`。修复:file 路由补上与 backup/restore 同款 `WorldVolumeExists` 门 → 快速 409 `no_world_volume` + 面板双语文案。本批现场复查:`resolvecheck`(无 world PVC)→ 409 `no_world_volume`,0.03s;`test-one`(有盘)→ 200(~2s)。
|
||||
4. **#59(`a2ff2a1`)LuckPerms 页把"读不到"报成"没有"。** 触发路径:打开装了 LP 的服的管理页——LP 5.5.85 的 `lp` 命令 RCON 回包全为空(独立 RCON 客户端 1:1 复现),读投影永远为空,旧页面却断言"没有父组/没有显式节点"(假事实)、写历史伪造 `[RCON]` 行。修复:原始回包随 rosters 同款披露渲染;空回包显式提示、不再假断言;历史占位不再伪造输出。
|
||||
|
||||
**二、服务器详情页交互复跑(5 项全真机,0 缺陷)**
|
||||
|
||||
1. **ServerCard 停止**:test-one `Stopping` → `Stopped`(启动在第三十七批;收尾复查 `desiredState: Stopped`)。
|
||||
2. **ServerFiles 写流程**:编辑 motd → 保存「已保存」→ 重开读回一致(`motd=Felis E2E files drill`)→ 还原默认 `A Minecraft Server`(收尾复查确认)。
|
||||
3. **备份/恢复**:立即备份 → Job `succeeded`;恢复 → 确认对话框(破坏性警告原文)→ Job `restore-test-one` `Complete`(5s)→ UI「恢复 27秒钟前 成功」→ 唤醒 `Running/ready`(恢复后的世界可加载)。
|
||||
4. **白名单**:加/负例/移除全链复跑(证据见 #56)——终态磁盘只剩 `E2E_Tester`(88 字节)。
|
||||
5. **封禁/解封**:封禁(内联确认)→ `Banned E2E_Bad: Banned by an operator.` + 磁盘写入;解封 → `Unbanned E2E_Bad`;`banned-players.json` 终态 `[]`。
|
||||
|
||||
**三、运维面复查(3 项,0 缺陷)**
|
||||
|
||||
1. **`/admin/updates` 复跑**(队列"深挖"项):设置窗口 → 「维护窗口更新成功。」+ 状态卡更新;负例 end<start → 前端「结束时间必须在开始时间之后。」、后端 400 `end must be after start`;清除 → 「当前未设置维护窗口」。
|
||||
2. **metrics 端点**:内部面 `/metrics` 正常——`felis_*` 样本 18 条 / 3 个指标族(build 失败计数、回收计数、起服时长直方图)。
|
||||
3. **CLI 覆盖核销**:`run` 分发表 17 项(15 个用户面命令 + `bootstrap-assets`/`init-forwarding` 两个容器内部入口)在历批演练中均已有真机记录,本批逐项核销无遗漏。
|
||||
|
||||
**四、接口语义注记与收尾**
|
||||
|
||||
- 文件 API 的 `path` 是**相对路径**:空串 / `.` = 根(正常列出)、`config` 下钻正常;字面 `/` 被路径约束拒绝(400 `bad_path: path escapes from parent`)——面板从不发绝对路径(`joinPath` 只拼相对段),无用户面影响;本批复查脚本初次误用 `/` 时曾见 13s 延迟,属首次 fileedit Job 冷启动,非卡死。
|
||||
- 收尾静止态:test-one `Stopped`;三个名单 = whitelist `E2E_Tester` / banned `[]` / ops `[]`;`E2E_Bad` 仅余 latest.log 与 usercache.json(日志与 Mojang 缓存)。
|
||||
- 留档(Mac):脚本 `/tmp/cdp-run17a/17b/17c`(停止、写文件、还原)、`run18/18b/19`(备份、恢复对话框、完成等待)、`run20a/20b`(白名单加/减)、`run21/21b/22`(封禁重试、封禁、解封)、`run23`(维护窗口)+同名 `-out.json`;截图 `run17b-1`、`run18-1..4`、`run18b-1,2`、`run19-1`、`run20a-1,2`、`run20b-1`、`run21b-1`、`run23-1..3`。
|
||||
|
||||
### 本轮新增真机证据(第三十九批:reaper 多节点实机 —— VM 克隆双节点验证)
|
||||
|
||||
**零、装置**:`prlctl clone "CentOS Linux 9 Stream" --linked --name felis-node2`(linked 克隆,初始 1.4M)→ node2 改 host 名 `felis-node2`,停用/禁用 `felis-velocity`、`k3s.service`(server 形态)、docker,以 k3s-agent 加入同一集群(`https://10.211.55.6:6443`,复用 node1 node-token)。克隆副作用 = node2 自带 node1 storage 副本(6 项 / 3.1G)——已移开,模拟真实新节点。两节点均 Ready(v1.36.4+k3s1);为克隆,node1 经历一次正常重启,重启后组件/服务全回归(docker 随自启后又停回 `inactive`)。
|
||||
|
||||
**一、pin 正向**:`kubectl create job reaper-b39-ok --from=cronjob/felis-reaper`(live 模板原样,nodeSelector=`localhost.localdomain`;node2 无 taint、是合法调度候选)→ pod 落在 `localhost.localdomain`(nodeSelector 命中;backups PVC 的 affinity 亦指向同节点——两者本就该同节点),4s 跑通:`evaluated=2 reaped=0 warned=0 skipped=0 evicted=0 expired=0`、Job `Complete`。即:渲染出的 pin 在真实双节点集群把 reaper 钉在持盘节点。
|
||||
|
||||
**二、错位 pin 反向对照**:同模板把 nodeSelector 改成 `felis-node2` → pod 停在 **Pending**,scheduler 事件原文:`0/2 nodes are available: 1 node(s) didn't match PersistentVolume's node affinity, 1 node(s) didn't match Pod's node affinity/selector`——node1 被错位 selector 拒绝、node2 被 `felis-backups` PVC 的 volume node affinity 拒绝(PV `pvc-0b4fbda6-…`,nodeAffinity=`localhost.localdomain`、hostPath=`/var/lib/rancher/k3s/storage/pvc-0b4fbda6-…_minecraft_felis-backups`)。**结论:本部署形态下 pin 写错是 fail-closed(卡住 + 明确调度事件),不会在无世界节点上静默执行**;真正要防的是首跑顺序(全新多节点安装时,第一次 reaper 运行会把 backups PV 落在其所在节点)——这正是渲染默认要求 `--reaper-node` 并 stderr 警告的原因,维持现状不修。
|
||||
|
||||
**三、装置回收**:drill job ×2 删除(minecraft ns 无残留);node2 关机 → `kubectl delete node felis-node2` → `prlctl delete felis-node2`(VM+克隆件删除)。收尾 `get nodes` = 单节点 `localhost.localdomain`,felis/游戏 pod 全 Running,CronJob `suspend=false` + nodeSelector 原样,docker `inactive`。重启副作用按既有说明处理:Mac 侧 443 入面板依赖的手工 `socat` 中继(本台账开头注记「重启 VM 后需重开」)随重启消失——已按原样重开(`TCP6-LISTEN:443,ipv6only=0,reuseaddr,fork TCP:127.0.0.1:30443`)并复核 op.console 面板 / api-me / player 面板皆 200。
|
||||
|
||||
**四、留档**:node2 agent 上线日志(`k3s agent is up and running`、VXLAN subnet event 来自 10.211.55.6);两向 job 的 pod/调度事件原文(见上);`/root/node2-storage-copy/`(3.1G,随 VM 删除)。
|
||||
|
||||
### 本轮新增真机证据(第四十批:Java 插件层收口 —— #63 demo-up 单起点、CI Java 门禁、插件面真机 E2E)
|
||||
|
||||
**零、现场**:本批不改 Go/面板(控制面维持 `auditfix61`);产出 = `deploy/demo-up.sh` 重写(`f5a76cf`)+ `plugins/test.sh` 与 CI `plugins` 作业(`c59b387`)。三组真机验证(T1/T2/T3)在 VM 上针对该两文件跑完;CI 首跑即绿。
|
||||
|
||||
**一、#62 复核:不成立(立案后剔除,未计为缺陷)**
|
||||
- 主张:"fresh 非 demo 安装跳过 Velocity 插件构建 → proxy 静默不路由"。复核三条路径,**每条都构建 felis-velocity.jar**:
|
||||
1. 源码臂:`felis-install.log`(首装)L1530 `building felis-velocity.jar`、L1553 `staged`;`bootstrap-auditfix{42,42b,42c,43,43b,44}.log` 每次重跑同两行。
|
||||
2. 嵌入 tar 臂(fresh release/TUI 路径;本 VM 从未走过):`felis bootstrap-assets game-stack | tar -x` → 37 文件(含 `plugins/velocity`、`plugins/shared`,构建输入齐全)→ `docker run --rm -v /tmp/gs62:/src:z -w /src/plugins/velocity gradle:8.14-jdk21 gradle --no-daemon clean build` → **BUILD SUCCESSFUL in 35s**,产出唯一 `felis-velocity-0.1.0.jar`(71117B,与现装同尺寸)。
|
||||
3. 调用图:`build_velocity_plugin` 唯一调用点在 `build_game_stack` 末尾;`build_game_stack` 在 full 模式主链路(L2868)必经,nano 早返回,不存在可绕开的"demo 分支"。
|
||||
- 结论:代码阅读误判,剔除。真正会跳过插件构建的是 demo-up.sh 的旧镜像臂 → 即 #63(本批修复)。
|
||||
|
||||
**二、#63(`f5a76cf`):demo-up.sh 双起源 → 单起点(129 → 63 行)**
|
||||
- 红证据(读取 + 真机口径核对):① 版本分叉——demo-up 硬编码 `PAPER_MC_VERSION:=1.21.8`,bootstrap `resolve_game_jars` 从 Limbo CI 产物名推导(T3 实测当日 Limbo 2026.0.3-ALPHA / MC 26.3;同一登录两跳必须同协议);② 其镜像臂从不构建 felis-velocity.jar,导入的是本地 tag(`felis-limbo:demo` 等),与 bootstrap 写入的 registry refs 不一致 → 死件、旧基座上可致"起了但无处路由";③ 起 docker 后从不停(违反 13d64e0 起"构建方以 docker 停收尾"的约定);wiring 段在现代基座上只是 no-op。
|
||||
- 修复后 = bootstrap(未 SKIP 时)+ 三项硬校验(`felis.host.toml`、`[velocity]` 段、`felis-velocity.jar`)+ `felis setup` 交棒;镜像逻辑全删。
|
||||
- 真机验证:**T1** 健康基座(`SKIP_BOOTSTRAP=1 SKIP_SETUP=1`)→ 通过、exit 0;**T2** 移走 jar → `ERROR: /opt/felis/velocity/plugins/felis-velocity.jar missing — … silently routes nothing …`、exit 1(随即还原 71117B root:root 0644);**T3** 默认臂全量重跑(`FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full FELIS_IMAGE=…:auditfix61 FELIS_WORLDS_HOST_PATH=… SKIP_SETUP=1`,日志 `/root/demo-up-b40.log`):L733 插件构建、L752 `staged`、L941 交棒行、`[fail]`=0;收尾态 `felis-velocity active / k3s active / docker inactive`。
|
||||
- 附带实录(同一日志窗):重跑期间 felis-api 滚动,velocity 00:22:25/00:22:40 两次 `refresh failed … keeping current registrations.`(失败保旧——规格 §11 承诺的真机上演)→ 00:22:55 恢复注册 → 00:22:58 服务随 `install_velocity` 重启并重载插件(新 pid)→ 00:22:59 `routing ready` → 00:23:29 收敛到 pod IP。
|
||||
|
||||
**三、Java 层 CI 门禁(`c59b387`):三个手工测试 + 三个装机 jar 首进 CI**
|
||||
- 缺口:`plugins/*/test` 三个 main 测试从未被任何 build/CI 运行;velocity/paper/limbo 三个"装机即用"jar 只在 bootstrap/Dockerfile 编译(Go CI 从不碰 Java)。
|
||||
- 产出:`plugins/test.sh`(Maven Central 取 adventure 三 jar,pinned+sha256;三测试 javac+java;三生产编译,limbo 按 bootstrap 同源解析当日版本)+ `ci.yml` 新增 `plugins` 作业(temurin 21 + Gradle 8.14)。
|
||||
- 首跑两次真跑抓出两处"从未被跑过"的假设错误:① `InviteCardTest` 一行式缺 examination-api(adventure-api 4.26.1 的 `Component` 签名引用 `Examinable`,javac 编译期即需);② limbo 的 `com.loohp:Limbo:+` **永远不可解析**(LOOHP 仓无 maven-metadata,404 实查)→ 裸 `gradle -p plugins/limbo build`(plugins/README 原文)从来不可行;二者均已修(脚本 + 该测试 javadoc + README)。
|
||||
- 验证:VM 容器三测试 `OK (32/36/48 checks)`;velocity 21s / paper 29s / limbo 13s(2026.0.3-ALPHA)全 `BUILD SUCCESSFUL`。GitHub Actions run `35888009965`(c59b387)= go/shell/panel/plugins **4/4 success**(`plugins` 作业首跑即绿)。
|
||||
|
||||
**四、插件层真机 E2E(不依赖真实客户端的面)**
|
||||
- 自建纯 socket 探针 `/tmp/mcprobe.py`(status = 服务器列表 ping;login = 登录首包),重装前与重装后各跑一遍,结果一致:
|
||||
- 子域 MOTD(§11 只读缓存 + 相位):`test-one` → `« test-one » 休眠中,加入即唤醒 / sleeping — join to wake`;`lobby`/`login` → `« … » 在线 / online`;`resolvecheck` → 休眠中;`nosuchxyz.<root>` → 回落 `A Felis server`(非 felis 子域不被劫持)。
|
||||
- 登录边界:`LoginStart`(含现代协议 UUID 字段)→ 服务端首包 `0x01 EncryptionRequest` ⇒ 边缘 online-mode 强制成立。
|
||||
- 加载面:`Loaded plugin felis-link 0.2.0`(共 5 插件)。
|
||||
- 口径:真实账号进服(limbo 门 → 菜单 → 转服)仍按"进服跳过"决定不演练;以上为不依赖客户端的最大真机面。(首次探测因探针漏发 UUID 字段被静默关闭——探针缺陷,非服务端问题;补齐后一次通过。)
|
||||
|
||||
**五、观察(不计缺陷)**
|
||||
- ViaVersion 自报有 5.12.0(当前 5.11.0):bootstrap 的 pin 是 FL-007 实测版本,属刻意,不随提示升级。
|
||||
- limbo 插件编译期一条 deprecated API 提示(`FelisLimboPlugin` 用/覆写已弃用 API):门禁下可见、不阻塞,留观察。
|
||||
|
||||
### 本轮新增真机证据(第四十一批:装载器 mod 层收口 —— #64 exec 位、#65 元数据、编译+起服双门禁)
|
||||
|
||||
**零、现场**:控制面与面板不动;产出三个 commit:`f6048f2`(#64 修复)、`c2fe6a6`(#65 元数据)、`aa5abfa`(`plugins/test-mods.sh` + CI `mods` 作业 + release 双门禁 + README 记录)——这是最后一块"从未被任何自动化碰过"的模块面(fabric / forge / neoforge 三个装载器 mod)。
|
||||
|
||||
**一、#64(`f6048f2`):三个 vendored `gradlew` 缺可执行位——文档正路第一步即失败**
|
||||
- 红证据:`git ls-tree origin/main` 三个 `gradlew` 全为 `100644`(blob `b9bb139f…`);`plugins/README.md` 第 212–214 行原文教 `plugins/fabric/gradlew -p plugins/fabric build`(forge/neoforge 同式);VM 全新 clone 实跑 → `bash: line 1: ./gradlew: Permission denied`,exit 126。
|
||||
- 为何从未暴露:无 CI、无安装路径,README 是唯一入口——直到本批第一次真跑才现形。
|
||||
- 修复:`chmod +x` + `git update-index --chmod=+x`(mode 100644→100755 ×3)。
|
||||
- 可达性:①(全新用户照文档走的第一步)。
|
||||
|
||||
**二、#65(`c2fe6a6`):mod 元数据与仓库 LICENSE 相悖**
|
||||
- 三处 `license = "MIT"`(`fabric.mod.json`、forge/neoforge 的 `mods.toml`)vs 仓库 `README.md:73` 的 `AGPL-3.0-only` + LICENSE 全文。时间线:mods 2026-06-26 加入(`93f143f`)、LICENSE 2026-07-12 才落地(`037eb24`)——陈旧残留。
|
||||
- 另两处 `issueTrackerURL = "https://example.invalid/felis"`(占位域名)→ `https://github.com/FelisMC/Felis/issues`。
|
||||
- 用户面:fabric loader 启动打印 license 字段;`mods.toml` 被两个 loader 解析。
|
||||
- 可达性:②(低;元数据/合规面,非功能)。
|
||||
|
||||
**三、双门禁落地(`aa5abfa`)**
|
||||
- `plugins/test-mods.sh`:JDK 17 下依次跑三个模块的 vendored wrapper(`./gradlew --no-daemon build`);java 大版本非 17 直接 fail-loud(三模块目标是 Java-17 的 Minecraft 线)。
|
||||
- `ci.yml` 新增 `mods` 作业(temurin 17 + `gradle/actions/setup-gradle@v4`,版本由各模块 wrapper 自管);`release.yml` 发布前加两道 Java 门禁(JDK21 `plugins/test.sh` + JDK17 `plugins/test-mods.sh`)。
|
||||
- README:Status 更新为 compile + boot 双验证;Building 段补 limbo `-PlimboVersion=<release>` 要求与两个门禁说明。
|
||||
- CI:run `35892544563`(`aa5abfa`)——`mods` 作业首跑即绿,五作业全过(shell / mods / panel / go / plugins 5/5)。
|
||||
|
||||
**四、三个模块的真机编译 + 起服 E2E(本批核心证据)**
|
||||
- 编译(VM 容器 `eclipse-temurin:17-jdk` + 持久 gradle 缓存):三模块 `BUILD SUCCESSFUL`(首跑 4–5 分钟级/模块),产物 `felis-{fabric,forge,neoforge}-0.1.0.jar` = 29276 / 28855 / 28562 B;留档 `/root/mods-build-b41.log`。
|
||||
- 起服(三个真实专用服,控制台直驱 `/link`,随后 `linkx` 未知命令对照、`stop` 正常停服):
|
||||
- **fabric** 1.20.1 + loader 0.19.5 + fabric-api:`Felis link ready; /link is registered.` → `Done (10.670s)!` → `/link 只能由玩家执行 / /link can only be run by a player.` → `Unknown or incomplete command`(linkx)→ `Stopping server`;`EXIT_fabric_boot=0`。
|
||||
- **forge** 1.20.1-47.3.0:`Felis link ready; /link will be registered.`(modloading-worker)→ `Done (11.751s)!` → 同四段证据;`EXIT_forge_boot=0`。
|
||||
- **neoforge** 20.4.251(1.20.4):`Done (16.094s)!` → 同四段证据;`EXIT_neoforge_boot=0`。
|
||||
- 口径:三服均为"装好 mod 后真启动"的场景,同时验证 loader 真实解析修改后的元数据(license=AGPL、tracker URL)后正常装载;完整取码链(玩家在游戏内 `/link`)按"进服跳过"决定不演练,装载/注册/拒绝面已全部真机成立。
|
||||
- 留档:`/root/mods-e2e-b41.log`、`/root/mods-e2e-b41-resume{2}.log`、`/opt/felis/mods-e2e/`(496M,三服目录保留);收尾态:docker `inactive`、k3s/felis-velocity `active`、磁盘 17G free。
|
||||
|
||||
**五、演习装置自身两次修正(非产品缺陷,如实记录)**
|
||||
- 控制台驱动脚本 `/root/modserver-drive.sh` 初版把 FIFO 写端先开(`exec 3> console`)→ 自我死锁(内核 `wait_for_partner`、`State: S`),java 从未启动(无 `server.out` 实证);改 `exec 3<> console`(RDWR 打开不阻塞)后三服全通。
|
||||
- forge 安装器首跑:`libraries.minecraft.net` 连接失败 → `com.google.code.findbugs:jsr305:3.0.2` 下载失败、`There was an error during installation`、无 `run.sh`;原样重试即 `The server installed successfully`(VM 网络抖动,非 forge/产品问题)。
|
||||
|
||||
**六、观察(不计缺陷)**
|
||||
- 无新增。
|
||||
|
||||
### 本轮新增真机证据(第四十二批:覆盖面对账 + 文档/工具收尾 —— #66 README_EN、§28 图对齐、sync.sh 本地修复)
|
||||
|
||||
**零、现场**:本批不碰产品代码;对"所有模块均已演练"的结论做**独立对账**(枚举仓库全部模块/交付物 × 台账覆盖),只挖到文档与本地工具层面的残留。commits:`62c5a8a`(#66 README_EN)、`62e8c87`(§28 序列图对齐)。
|
||||
|
||||
**一、覆盖面对账(本批核心,逐项)**
|
||||
- 仓库卫生:`git ls-files` 全量扫描无 `.DS_Store`/`*.log`/`node_modules`/构建产物误入库;顶层 `felis` 二进制(83MB)被 `.gitignore` 正确忽略;旧仓库名 `MliroLirrorsIngenuity` 残留仅存在于台账(历史记录)与本地未跟踪二进制。
|
||||
- 未竟工作标记:`TODO|FIXME|XXX|HACK` 全仓(go/ts/sh/md/yml)仅 1 处命中——`internal/api/api_test.go:263` 的 `context.TODO()`(合法测试写法),无未完成实现标记。
|
||||
- CLI 派发面:`cmd/felis/run.go` 的 usage ↔ dispatch 表有双向测试强约束(`run_test.go`,含显式 `undocumentedCommands` 允许表:`bootstrap-assets`/`init-forwarding` 等 in-Pod 入口),机制已核实成立。
|
||||
- CI 覆盖:`shell` 作业按 shebang 对**全部 tracked `*.sh`** 做语法检查(`git ls-files '*.sh'` 逐文件 `bash -n`/`sh -n`)并运行 `deploy/bootstrap_test.sh`;`plugins`/`mods` 各自作业。无"存在但从无入口运行"的脚本。
|
||||
- `internal/*` 22 个包逐一对账:0 提及者(`internal/apis`)为 CRD 类型包(被 40+ 站点引用、随 operator/manifests 演练间接全覆盖);`deploy/{limbo,lobby,paper}` 为三张游戏镜像定义,随每次起服演练执行。真盲区=无。
|
||||
- `panel` 全部 17 条页面路由 × 既有批次(CDP 巡检、深挖、写流程)覆盖;`scripts/` 三脚本为**本地私有**(`.git/info/exclude`,不入库)。
|
||||
- `docs/` 四件:troubleshooting(多批引用)、openapi(含 parity 强制测试)、deferred-seams(内容为现行状态,含 CLOSED/verified-live 注记,抽查与代码一致)、sequence-diagrams(发现漂移 → 修复,见第三节)。
|
||||
|
||||
**二、#66(`62c5a8a`):README_EN 与中文 README 漂移**
|
||||
- 红证据:`README.md`(中文)"使用方式"含**必需**的私有仓库 workaround(仓库私有 → 裸 raw 命令 404;token 经 `curl --config -` 由 stdin 下发、不进 argv;`sudo -E` 传给安装器)与"重跑=升级 felis-api、通道不继承(跟 main 需 `FELIS_VERSION_BOOTSTRAP=dev`)"两段;`README_EN.md` 是 `README.md:6` 链接的英文入口,这两段**完全没有** → 英文读者照文档第一步就 404 且无指引。
|
||||
- 修复:两段译入 EN(保留原命令与语义;`grep` 校验 token/config/dev 三关键串在位)。可达性:①(轻)——英文首装路径第一步。
|
||||
|
||||
**三、§28 序列图对齐(`62e8c87`)**
|
||||
- 漂移点(两处,与修复后代码实读对照):① Claim Transaction 图注仍是 audit #4 前形态(`SELECT EXISTS` + 事务外 pre-check)→ 更新为现行实现:`pg_advisory_xact_lock(user_id)` + 行 `FOR UPDATE` + **事务内**四维配额门(图上 `QuotaAvailable` 只是 fast path);② Link Flow 更新为:取码 `FOR UPDATE`、同用户重验幂等、退役(soft-deleted)账户链接被当场接管——409 仅对"其他活跃用户"。均以 `internal/api/pgrepo.go` 实读 + handler 映射核对。
|
||||
- 惯例依据:该文件有维护史(`676407d docs(diagrams): align §28 sequence diagrams with implemented routes`)。
|
||||
|
||||
**四、本地工具修复(私有脚本,不入库):`scripts/sync.sh` 排除 `plugins/` 与 `go:embed` 冲突**
|
||||
- 红证据(本机复现):以 sync.sh 原排除清单 rsync 一个 fresh 目标 → `go build` 立即失败:`bootstrap_asset.go:30:12: pattern plugins/limbo/build.gradle: no matching files found`(`bootstrap_asset.go` embed 了 plugins/{limbo,paper,velocity,shared} 源码;排除清单成文于 embed 引入之前)。存量目标靠 rsync 对排除路径的"保护"而看似正常,fresh 目标必炸。
|
||||
- 修复:排除项从 `plugins/` 改为 `plugins/*/build|.gradle|bin/`(与 `.gitignore` 同义);`watch.sh` 同步去掉 plugins 的忽略。验证:新清单 rsync fresh 目标 → `go build` 绿(83.7MB 二进制产出);两脚本 `bash -n` 通过。因脚本被 `.git/info/exclude` 私有,修复留在工作机、无 commit。
|
||||
|
||||
**五、release.yml 静态复核(该流水线从未执行过——仓库当前无任何 tag/release)**
|
||||
- 三处输入/输出契约实核:① 发布资产名 `felis-linux-amd64/arm64` ↔ `deploy/bootstrap.sh` 的 `asset="felis-linux-${arch}"` 及收敛检查 `felis ${FELIS_REF}`(版本首行契约两侧一致);② `out/linux_amd64/usr/local/bin/felis` ↔ Dockerfile 最终层 `COPY /out/felis /usr/local/bin/felis`;③ 版本断言 `felis ${GITHUB_REF_NAME}` ↔ `cmdVersion` 的 `felis %s` 首行。
|
||||
- 唯一无法在本机完全复刻的 `file(1)` 机器类型断言:用交叉编译的 linux/arm64 二进制实测输出 `ELF 64-bit LSB executable, ARM aarch64, …`,断言串 `ARM aarch64` 匹配(BSD/Ubuntu file 同源 magic)。
|
||||
- 结论:整条流水线未跑属"尚未切 tag"的发布流程事实(第四十批已注记);切首个 tag 前无待修项。
|
||||
|
||||
**六、观察(不计缺陷)**
|
||||
- `internal/api/handlers_account.go:169` 的 reclaim 选择仍为 Java/Velocity 侧 CODE-ONLY(deferred-seams 已记 accepting)——维持。
|
||||
|
||||
### 本轮新增真机证据(第四十三批:构建 Job 收尸 #67 + #68 上下文拉取重试 + 生命周期/故障注入压测)
|
||||
|
||||
**零、现场**:控制面从 auditfix61 升到 **auditfix62**(含 #67;自建镜像推入内建 registry,api/operator/reaper 三处滚动,`felis version` = `v0.0.0+fix62`)。本批证据 = 6 轮起停循环 + 3 处故障注入 + PG 断连语义 + 并发风暴 + 持续轮询。
|
||||
|
||||
**一、#67(`2755e41`):build Job 从不回收——每构建一个、完成 Pod 永久堆积**
|
||||
- 红证据(真机):`felis-build` 里 5 个完成 Job/Pod 最长 26h 无人回收;全仓 TTL 对照——fileedit 2m / backup 10m / restore 10m / reaper CronJob 3+3 历史,**唯独 build 没有 `ttlSecondsAfterFinished`**;且 `build.go:69` 的注释写着 "(e.g. GC'd); treated as failed"——预期的 GC 从未存在。危害:完成 Pod 计入节点 pod 预算(stock k3s 110),构建量一上来先把节点塞满,新构建全部 Pending;etcd/磁盘同步膨胀。
|
||||
- 修复:`buildJobTTL = 7 * 24h`(终态后计时;日志路由是失败分诊面故取长窗;`Sync` 对终态幂等、`JobUnknown` 只影响非终态 → 删除后的唯一代价是日志 404)。单测 `TestBuildJobIsReapedAfterCompletion`。
|
||||
- 真机验证(auditfix62):① 真实路径触发构建(owner 会话 → `POST /api/v1/images/build`,借遗留 blob 作 context)→ 新 Job `ttlSecondsAfterFinished=604800`、构建 25s `succeeded`;② 对旧 Job 打 `ttl=30s` → 40s 内 Job+Pod 被 TTL 控制器收走(机制实证);③ 旧堆积 5 个里 1 个已收、4 个留存对照。
|
||||
|
||||
**二、生命周期 ×6 + 三处故障注入(operator / api / postgresql)**
|
||||
- 每轮:CR `desiredState` Running → 等 Running → Stopped → 等 pod 消失。结果:**6/6 全收敛**——Running 23–29s(注入轮与无注入轮无差)、Stopped 3s、pod 每轮如期消失;收尾 conditions 无 Failed 残留(Ready=False / RconReached=False = Stopped 正常;Provisioned=True)。
|
||||
- 注入 1(cycle 2,删 operator pod @t+2s):Running 仍 23s 达成(新 operator 立即接管 reconcile)。
|
||||
- 注入 2(cycle 4,`systemctl restart postgresql`):CR 路径无感;另做定点验证:PG 停机窗口 `/me` = **503**(非 401,#11 语义保持)、`/healthz` = 200(存活探针独立)、内部 status = 200(纯集群读);PG 恢复后 `/me` = 200。
|
||||
- 注入 3(cycle 5,删 api pod @t+2s):内部轮询出现 2 次 `http=000`(约 6s 窗口)后自愈;Running 28s 达成。
|
||||
- 观察(不计缺陷):高频翻杆期间 operator 报 9 条 `Operation cannot be fulfilled ... object has been modified`(乐观锁冲突 → 重排队自愈),目标时间无差;controller-runtime 正常重试语义,生产低频操作下更罕见。
|
||||
|
||||
**三、并发风暴 + 拒绝面 + 持续轮询**
|
||||
- 20 路并发 internal status + 10 路 `/me`:30/30 全 200。
|
||||
- 无主 + ownerOnly 服 wake:403 `forbidden`(正确拒绝面,非 5xx)。
|
||||
- 持续轮询(status + healthz + me,5s 一轮 × 300 轮 ≈ 25 分钟,900 样本):收盘 `LONG POLL DONE fails=27`——27 个失败样本**全部**落在三个自导演练窗口内(02:24:15/20 红证据杀 api〔6〕、02:29:48/53 auditfix63 滚动升级〔6〕、02:30:53–02:31:13 #68 确定性复现 scale api→0〔15〕),窗口外 873 样本全 200、零自发失败;每个窗口在动作结束后 ≤1 个探测周期(~5s)内恢复。
|
||||
|
||||
**四、留档与收尾**
|
||||
- 留档:`/root/soak43.sh` + `soak43.log`、`/root/poll43-long.sh` + `poll43-long.log`、`/root/mint-owner-43.sh`(owner 会话重铸)、`/root/felis-image-build43.log`(镜像构建)、`/tmp/owner-jar43.txt`(owner cookie);镜像 `10.43.182.43:5000/felis/felis:auditfix62` 已入 registry(`e2e/ttl-probe:latest` = 本批 drill 镜像,保留)。
|
||||
- VM 状态:docker 用毕已停(inactive)、k3s/velocity active;控制面三处 = auditfix62。
|
||||
|
||||
**五、#68(`b76d0ac`):构建 context 拉取对控制面重启零容忍——一次拒连即终态失败**
|
||||
- 红证据(真机):删 api pod 后触发构建 `bld-1790187851749257160` → Job `Failed`;pod 时间线 `18:24:12Z` 起、context-fetch `18:24:13Z` exit 1,日志原文 `dial tcp 10.43.237.249:8081: connect: connection refused`;Kaniko 从未启动(PodInitializing);新 api pod 11s 后就绪、同 blob 前后各一次构建均 25s succeeded——纯粹"单次尝试"造成的无谓失败。
|
||||
- 修复(`b76d0ac`):`fetchContextWithRetry`——传输错误/5xx 重试至 45s 窗口(3s 间隔),4xx 快速失败(是答案不是抖动);URL 校验提前为 usage error(2);保留原错误文案前缀。窗口/间隔为 vars(测试可缩窗);4 例单测:5xx 后恢复、拒连后恢复、404 不重试、窗口耗尽(全绿,含 `-race`)。
|
||||
- 真机验证(auditfix63,确定性优先,不复刻竞态):`scale api→0` → 以复刻 Job 直建(由红证据 Job 派生:换名 `build-retry68`、剥 controller 标签与 selector、镜像改 auditfix63,复用 `felis-service-token` Secret)→ fetch 连续 8 次 `connection refused; retrying`(实捕日志)→ `scale api→1` → fetch 恢复、Kaniko 构建推送、trivy 干净 → **Job Complete**(`18:30:52Z→18:31:24Z`,32s);同形场景对照旧版 = 终态 Failed。正常路径回归:api 在线直构 `bld-1790188306964162960` → `succeeded`(10s),fetch 日志**零重试行**(重试不引入正常路径开销)。
|
||||
- 附带发现(运维):本批升级沿用 `set image` 后 `FELIS_IMAGE` env 仍停 `auditfix61`(两代落后)——构建 Job 的 fetch 容器实际一直在用旧镜像;已把 api/operator 的 `FELIS_IMAGE` 与 api/operator/reaper 三处镜像全部对齐 `auditfix63`。真实安装器路径随 manifests 重渲注入该 env,手工升级须成对改(升级清单事项)。
|
||||
|
||||
### 本轮新增真机证据(第四十四批:modpack 规模上下文全链 + 并发构建)
|
||||
|
||||
**零、现场**:控制面 = auditfix63(含 #67/#68;api/operator `FELIS_IMAGE` 同版本),VM docker 停。本批目标 = 把"真实模组包大小"这条链压到生产尺度:此前所有提交/构建演练的上下文都是 KB 级。
|
||||
|
||||
**一、200 MiB 上下文全链(零缺陷)**
|
||||
- 构造:5×40 MiB `/dev/urandom` 载荷 + Dockerfile(随机数据不可压缩 = 最坏情形),tar.gz = 209,750,275 B。
|
||||
- 上传(真实 app 路由):`POST /api/v1/me/submissions/{id}/context`(owner 会话)→ 200,**0.44s / ~481 MB/s**;宿主核对 PVC 落盘 `sub-cc160b3ad19080f9/context.tar.gz`。服务端读写超时设计(ReadTimeout/WriteTimeout 有意不设、仅 ReadHeaderTimeout)与流式落盘(io.Copy → cappedReader)均按预期,无内存尖峰。
|
||||
- 批准 → 构建(`bld-1790189227767686191`):fetch init **1s**(200MB 内部面拉取 + 解压,无重试行)、kaniko **4s**(unpack/COPY/snapshot/push gzip 层 **209,738,853 B**)、trivy **4s**(0 findings);Job Complete,全链 13s(18:47:07Z→18:47:20Z)。registry manifest 实查层大小一致(推送非虚)。
|
||||
- 覆盖点随验:提交构建 Job `ttlSecondsAfterFinished=604800`(#67 对真实提交路同样生效);`/me/submissions` `build_status=succeeded`(#36 面在 200MB 规模下仍成立)。
|
||||
- 资源采样(2s 粒度,构建太快可能漏尖峰,标注局限):无 OOM——api 17.9MB / kaniko 33.4MB / registry 34.1MB / trivy 35.9MB;emptyDir+快照为临时占用,随 Job TTL 回收。
|
||||
- 1 GiB 帽的单测已有覆盖(`internal/submit/submit_test.go` oversize→ErrInvalid);帽下真机 = 本条(帽上真机不划算,不做)。
|
||||
|
||||
**二、并发构建 ×3(零缺陷)**
|
||||
- 三路同时 `POST /api/v1/images/build`(各自 ref)→ 202×3;三个 Job 均 **11s Complete**,无串扰:每个 Job destination 与 tag 一一对应(conc-b44-1/2/3 各落位)、`ttl=604800` 各自在。
|
||||
|
||||
**三、磁盘运维注记(非缺陷)**
|
||||
- 连续镜像构建(docker build)把 VM 根盘从 11G free 压到 4.3G;`docker builder prune -af` 一键回收 **6.5G**(回 11G free)。即每轮全量重建约需 5–7G 构建缓存空间;生产主机根盘建议 ≥40G 并定期 prune(如需可在部署文档补一行,本批未改码)。
|
||||
- 留档:`/root/bigctx.tar.gz`(200MiB 上下文原件)、`/root/bigctx/`、`/root/stats44.log`、`/root/stats44-build.log`;提交 `sub-cc160b3ad19080f9`(blob 与 `user-uploads` 镜像保留作尺寸样本);镜像 `e2e/conc-b44-{1,2,3}:latest`。
|
||||
|
||||
### 本轮新增真机证据(第四十五批:模组构建链三连修 #70/#71/#72 + 提交→装服全链闭环 + 文档/测试修 #69/#73/#74)
|
||||
|
||||
**零、现场**:控制面 auditfix63→66(每修一版真机重演);本批目标 = 真实用户会写的那种模组包 Dockerfile(`FROM <平台底座>` + 内容层)。此前所有构建演练都是 `FROM scratch`——用户真实形态的第一步从未跑过,本批连撞三个缺陷。
|
||||
|
||||
**一、#70(`ac403b9`):kaniko 拉底座缺 pull 侧 insecure 标志**
|
||||
- 红证据(真机):提交 `FROM registry.felis.svc:5000/felis/paper:demo`(`bld-1790189685537480076`)→ kaniko `Retrieving image manifest …` 后即败:`Get "https://registry.felis.svc:5000/v2/": http: server gave HTTP response to HTTPS client`。根因:Job 只给了 `--insecure --skip-tls-verify`——kaniko v1.24 `--help` 实测二者均只覆盖 **push**,拉取默认 HTTPS,撞上明文 registry。
|
||||
- 修复:补 `--insecure-pull` / `--skip-tls-verify-pull`(对称补齐;构建命名空间 egress 本就只许 DNS/内建 registry,不扩大可达面)。单测断言两 flag 在参。
|
||||
- 真机复验:下一次构建 `Retrieving image manifest` 不再报错,进入解包阶段(随即暴露 #71)。
|
||||
|
||||
**二、#71(`6e47730`):kaniko 解包底座层被 drop-ALL 卡死**
|
||||
- 红证据(真机):`error building image: error building stage: failed to get filesystem from image: chown /etc/gshadow: operation not permitted`——kaniko 以 root 解包 tar 层要把文件 chown 到层里记录的属主(root:shadow 等),drop-ALL 后 CAP_CHOWN/CAP_FOWNER/DAC_OVERRIDE 全无。`FROM scratch` 全用 COPY(属主= kaniko 自己)所以从未暴露;**任何真实底座**必触。
|
||||
- 修复:仅 kaniko 容器补回最小能力集 `CHOWN + DAC_OVERRIDE + FOWNER`(fetch/trivy 保持 drop-ALL 基线;单测断言"恰好三枚"防漂移)。
|
||||
- 真机复验:解包通过、COPY、推送成功(进入 trivy 阶段,撞上 #72)。
|
||||
|
||||
**三、#72(`b14bacb`):trivy 扫 jar 必拉 Java DB——egress 锁下必败**
|
||||
- 红证据(真机):kaniko 全绿后 trivy `FATAL … Unable to initialize the Java DB … failed to download artifact from mirror.gcr.io/aquasec/trivy-java-db:1: connection refused`。Java DB 按需下载:镜像一有 jar 就触发——模组包 = jar 集合,故所有真实用户构建都会倒在扫描门;旧构建全 scratch(无 jar)从未触发。
|
||||
- 修复:新增 `[registry] trivy_java_db_repository`(→ `--java-db-repository`,与 vuln-DB 旋钮对称);installer carry 白名单收录;§8e 配方补 Java DB 镜像步骤;deferred-seams 更新。
|
||||
- 真机复验(第 4 次尝试,auditfix66):Job spec 实测含 `--db-repository registry.felis.svc:5000/mirror/trivy-db:2 --java-db-repository registry.felis.svc:5000/mirror/trivy-java-db:1`;扫描 `ubuntu 26.04` + `paper/paper.jar` + `pebble` 全 0 漏洞;**Job Complete**(全链 27s,`bld-1790190788310114143`)。
|
||||
|
||||
**四、capstone:提交 → 装服 → 起服,全链闭环(零缺陷)**
|
||||
- 第 4 次尝试产物 `user-uploads/sub-c006cbd633317ccb:latest`(= `felis/paper:demo` + 用户 marker,经完整提交管道构建):
|
||||
- 白名单自动收录(`source=built`)→ 产品路由 `PATCH /api/v1/servers/test-one {"image": …}` **200**(`image_not_whitelisted` 门通过);
|
||||
- `POST /servers/test-one/wake` 202 → pod `test-one-0` Running 1/1(25s 内);
|
||||
- `kubectl -n minecraft exec test-one-0 -- cat /felis-probe-marker.txt` → **`felis-probe44-ok`**(用户构建上下文的内容确在运行中的服务器里);
|
||||
- 服务器日志 `Done (4.678s)!`(paper 完整启动);RCON 线程应答平台就绪探针;
|
||||
- 收尾:stop 202、pod 消失、镜像回滚 `felis/paper:demo`、desiredState=Stopped。
|
||||
- 口径:这是"玩家上传模组包 → 服主审批 → 自动构建 → 选用为该服镜像 → 起服"的机制全链(真实客户端进服仍按决定跳过)。
|
||||
|
||||
**五、#69(`5cbfa89`):README 过度承诺"自动部署"**
|
||||
- 复核(读全):提交数据模型无目标服务器字段;approve→build→白名单是唯一自动化;部署 = 服务器编辑框选镜像(产品路由与 UI 均在);面板文案本身只承诺"自动触发安全构建"。
|
||||
- 修复:README 中英两行改为"自动构建;产物进入镜像白名单,可直接选用为服务器镜像完成部署"。可达性 ①(轻)。
|
||||
|
||||
**六、#73(`2961beb`):bootstrap 测试在跑 k3s 的主机上必假失败(工具级)**
|
||||
- 红证据:VM(真实 k3s 主机)跑 `bootstrap_test.sh`(**原版同现**)→ `FAIL a missing worlds root is warned about`:用例拿真实路径 `/var/lib/rancher/k3s/storage` 期望 WARN,而该目录在"安装器真正要服务的机器"上必然存在 → 假红;CI 从未见到(runner 无此路径)。
|
||||
- 修复:warn 情形改用保证不存在的路径;并把 `trivy_db_repository` / `trivy_java_db_repository` carry 断言补齐。VM 复跑 **140 PASS / 0 FAIL**。
|
||||
|
||||
**七、#74(`96aa817`):CI flaky——console 断连测试读写竞态(工具级)**
|
||||
- 红证据:run `35907662213`(纯 README 提交)go 作业红:`TestServerConsoleDisconnectTeardown: expected the first event before disconnect, got ""`;`gh run rerun --failed` 即绿 → 竞态确认。
|
||||
- 根因:测试只等"源读到第一行"就 cancel,"中继把事件写入响应"尚未发生;cancel 落进缝里时 body 为空。
|
||||
- 修复:加 `firstDataWriter` 信号(写完成后再 cancel,写→读有 happens-before);本地 `-count=60` 与 `-race ×5` 全绿;CI 复跑 `5cbfa89` = success。
|
||||
|
||||
**八、运维注记(非缺陷)**
|
||||
- 升级滚动撞 kubelet **ephemeral-storage 压力**:auditfix64 滚动时新 api/operator pod `Pending 4m46s`(事件 `untolerated taint(s)`),02:58:12 kubelet eviction manager 回收后自动调度成功。诱因 = 构建把根盘压到 89%(docker 构建缓存 + kaniko/emptyDir 临时层);`docker builder prune -af` 两清共回收 ~11G,余 11G free。生产清单:根盘 ≥40G + 升级前清构建缓存(与批 44 注记合并)。
|
||||
- 中间失败构建留下的 registry 镜像(`user-uploads/sub-83f6…`、`sub-171e…` = 已推未过门;`sub-c006…` = capstone 产物)留档;测试服 `test-one` 已回 `felis/paper:demo` + Stopped;docker 停。
|
||||
|
||||
### 本轮新增真机证据(第四十六批:上传面 #75/#76 真机闭环 + tracker #8/#1 收口 #77/#78 + 文案 #79 + 发布链 rc smoke)
|
||||
|
||||
**零、现场**:代码三条 commit 先落(`43699b4` #77 面板跳转、`c57daaf` #78 converge、`b9ebc87` #79 INERT 文案),随后按 `b9ebc87` 重建镜像 = `registry.felis.svc:5000/felis/felis:auditfix77`(`v0.0.0+fix77`;宿主源 `10.43.182.43:5000`,docker build → push),控制面 api/operator + minecraft ns `felis-reaper` CronJob + api/operator 的 `FELIS_IMAGE` 四处成对齐;`/opt/felis/src` = b9ebc87 快照(上一版 `src.bak46`);宿主 drill 二进制 `/root/felis-fix77.bin`(Mac 侧 `GOOS=linux GOARCH=arm64` 交叉编译,`-X main.version=v0.0.0+fix77`)。
|
||||
|
||||
**一、#75 真机闭环(配额/限流,owner 会话,auditfix77)**——留档 `/root/probe77/create-lane.log`、`lane2.log`:
|
||||
- **create 失败不烧窗口**:坏 JSON → 400,**同一秒**合法 create → 201(`sub-811427aa7dcf6807`)——`release` 路径生效;
|
||||
- **create 冷却**:同用户 30s 内第二条 → 429 `submission_cooldown`;
|
||||
- **pending 上限**:攒到 5 条 pending → 第 6 条 → 403 `submission_quota_exceeded`;
|
||||
- **upload 失败不烧窗口**:对已审核行上传 → 409 `already_reviewed`,**同一秒**对 pending 行上传 → 200;
|
||||
- **upload 冷却**:15s 内第二次 → 429 `submission_cooldown`;等 16s → 200;
|
||||
- **存储预算**:把 S2 的 blob 稀疏 `truncate` 到让 owner 已存字节 = 2GiB−100B(`blocks=8`,不占盘)→ 上传 → 403 `submission_quota_exceeded`;管理员删除 S2(行 + 目录双清)后 → 同一上传 → 200(预算即时释放)。
|
||||
- 口径:预算按 blob 的实际占用聚合(`Blobs.Size` 遍历汇总,正是"用久了就超"的同一条读取路径);稀疏垫付只是把"已经存了 2GiB 的用户"这一状态合成出来。
|
||||
|
||||
**二、#76 真机闭环(撤回 + 管理员删除)**——留档 `/root/probe77/lane3.log`:
|
||||
- 撤回自己 pending → 200;DB 行 = 0、uploads 目录 = gone;重复撤回 → 404;对已审核 → 409 `already_reviewed`;对他人 id → 404 `not_found`(owner 判定先于状态判定,不泄露他人行状态);
|
||||
- 管理员删除(pending)→ 200;重复 → 404;行与 blob 双清(lane2 的 S2 同证:`s2 rows=0 dir=gone`);
|
||||
- 面板侧 CDP:**撤回两步确认**(展开行 → 「撤回提交」→「确认撤回」→ 行消失)与 **admin 队列每行两步删除**(trash →「确认删除」→ 行消失、计数回落)双双走过;截图 `/tmp/withdraw-{1,2}-*.png`、`/tmp/admdelete-{1,2}-*.png`(Mac),脚本 `/tmp/cdp-withdraw76.js`、`/tmp/cdp-admdelete76.js`。
|
||||
- 真机注记(非缺陷,分级 ZT 的对照实证):admin 面只在 `op.console.<root>` 主机可用——同一 owner 会话在玩家面板主机上 `is_admin=false`(admin 路由 403),在 `op.console.<root>` 上 `is_admin=true`(200);面板按要求显示「无权访问」而不是假装能点。
|
||||
|
||||
**三、#77(tracker #8,`43699b4`)真机闭环:锁定会话 → /setup**
|
||||
- 铸一个未完成引导会话(SQL:`sha256('drill-locked-77')` → `sessions`,用户 `3f2c1b0a-…`,`email_verified=f`、无 passkey)→ `GET /me` 200、`GET /me/submissions` → 403 `setup_required`(后端本就正确);
|
||||
- CDP 用该 cookie 打开 `/submissions`:页面请求 `/me/submissions` 收 403(code=`setup_required`)→ **自动跳转 `/setup`**(`location.href` 实测)→ 向导渲染「初始化你的账户 / 第一步 · 填写邮箱」(截图 `/tmp/setup8-redirect.png`;网络事件 = 200/403(/setup/status 200) 序列在脚本输出里)。
|
||||
- 顺带确认:Dashboard 首屏三请求(`/me`、`/me/servers`、`account/link/start`)都是 `SetupAllowed`——所以补丁前用户是在"点受保护操作"时才撞到那句误报「无权执行此操作」。
|
||||
|
||||
**四、#78(tracker #1,`c57daaf`)真机演练:`felis converge`**
|
||||
- 现场先剥后补:`kubectl patch` 清空 lobby `spec.rcon` 与 login `spec.startup.healthHTTPPort`(复刻老装机 CR 缺后加字段的状态)→ `/root/felis-fix77.bin converge` → 两条全部填回(`rcon{enabled:true,secretRef:lobby-rcon/password}`、`healthHTTPPort:8080`);**第二次运行 → 双双 `already converged`**(幂等);operator 侧滚动收尾,login/lobby 回 Running/Ready(日志 `converge-{1,2}.log`、`crs-before.yaml`)。
|
||||
- 语义边界(单测):非零值一律不碰(运维自换的 secretRef 生还)、缺席 CR 只提示(创建仍归 setup)、占名非系统角色 CR 拒绝收敛、未配镜像跳过。
|
||||
|
||||
**五、#79(`b9ebc87`,文档)**:troubleshooting 的 `[INERT]` 图例仍写「§12 列着唯一仍适用的字段」,而唯一候选 `spec.storage.retainOnDelete` 早已移除、§12 自述「每个字段都有 controller 读」。改为「今天没有字段处于该状态」并指向 §13 的移除记录。
|
||||
|
||||
**六、发布链 rc smoke(tag `v0.1.0-rc1` → release run `35949621233`)**:
|
||||
**六、发布链 rc smoke(tag `v0.1.0-rc1` → release run `35949621233`):首次全绿**
|
||||
- run 步骤实况:go vet/test ✅ → `plugins/test.sh` ✅ → `plugins/test-mods.sh` ✅(**release.yml 史上第一次真正跑这三关**)→ buildx 双架构构建 ✅ → **stamp 校验** ✅(`felis v0.1.0-rc1` + arm64 ELF 断言)→ publish ✅;
|
||||
- 产物释出:`felis-linux-amd64`(61,378,722 B)与 `felis-linux-arm64`(57,344,162 B)双资产,release 标记 `prerelease=true`;
|
||||
- **prerelease 语义复核**:`GET /repos/FelisMC/Felis/releases/latest` → **404**(RC 没有顶掉 latest——正是 release.yml 注释里防的那件事);bootstrap 默认通道在无 stable 时按设计给出显式指引后 die、`felis update` 的 404 报错可读——两个行为本批均实测;
|
||||
- 产物级复验(目标架构实机):arm64 资产 scp 到 VM 执行 → `felis v0.1.0-rc1`(`go1.26.8 linux/arm64`),且直接可用:`/root/felis-rc1.bin converge` 打现集群 → 双服 `already converged`;
|
||||
- 留档:`/root/felis-rc1.bin`;asset 副本 Mac `/tmp/rc1b/felis-linux-arm64`。
|
||||
- 剩余(产物决定,未代拍板):仓库仍无 **stable** release → 新装走默认 release 通道会以指引性报错 die(提示改 dev 或等 stable);切首个稳定 tag(如 `v0.1.0`)即让默认通道与 `felis update` 真正上线,建议维护者择时执行。
|
||||
|
||||
**七、运维注记(非缺陷)**
|
||||
- pg_hba 的 `host felis_pgint felis 127.0.0.1/32 scram-sha-256` 行**再次丢失**(批 31 补过一次)——补回并 reload 后 pgint 才能连。重装/动过 PG 后先查这条(已写进速查)。
|
||||
- 升级域注记:#75/#76 的 pgint 新断言(`CountPendingSubmissionsBy`、`DeletePendingSubmission`/`DeleteSubmission` CAS)首次上真 PG:**17/17 全绿**。
|
||||
- 根盘:镜像构建后 83% → `docker builder prune -af` 回收 3.5G,docker 停回 inactive。
|
||||
- CI:`43699b4` success;`c57daaf` 被后一 commit 的并发策略取消(同分支 cancel-in-progress),其树被 `b9ebc87` 的 success 完整覆盖;`b9ebc87` success。
|
||||
- tracker 收编:**关** #10/#20/#21/#22(#20→`d9246dd`、#21→`f5a76cf`、#22→`3f2b28d`、#10→`23792d6`,均附证据评论)+ **关** #1/#8(本轮实现并真机验证);**注记**(保持打开作老装机待办)#2/#3/#4;**留存** #9/#12/#13/#15(真增强,超出生产可用主干,未动)。
|
||||
|
||||
## 结论:离"生产可用"还差什么(按优先级)
|
||||
|
||||
1. ~~构建链路的上下文通道~~ ✅ **已修**(`f79e5eb`/`02fd2de`,真机全链路含拉回校验;Trivy DB 需按 §8e 镜像一次)。
|
||||
2. ~~镜像耐久~~ ✅ **已完成**(第二十九批:`a9b275a`/`13d64e0`/`fa0e8d7` + `c7e585e`;三次真机重跑 + GC 两演练——控制面与游戏镜像被 GC 后自动回拉;同批修复并复验 #46/#47/#48/#49)。
|
||||
3. ~~PG 级契约测试~~ ✅ **已落地**(`2a55a0d`):`internal/pgint`(`-tags pgint`,需 `FELIS_TEST_PG_URL` 指向名字含 `pgint` 的库,harness 会 drop schema + 重放真实迁移)已覆盖会话/OTP/op-login/绑定码/submission/build/owner 角色;**首跑即抓到 #20**(索引与 ErrEmailTaken 从未存在)。运行方式见 CONTRIBUTING.md。
|
||||
4. ~~面板把 /jobs 显示出来~~ ✅ **已完成**(`97a64c8`,备份页「最近操作」卡,正是它把 #35 暴露出来的)。
|
||||
5. ~~多节点回收~~ ✅ **已修**(`daf7602`:`--reaper-node` → CronJob pod `kubernetes.io/hostname` nodeSelector,真机 render/dry-run/收敛 diff 三连;单节点部署不传即维持原状)。附带核销缺陷 #38(渲染提示里过期的 uid-1000/setfacl 指导)。
|
||||
6. ~~告警~~ ✅ **已完成**(`94f71ee` + `43df08b` + 第二十七批实弹演练:真实构建失败 → pending → 08:33:14Z firing;规则随 `deploy/alerts/` 交付)。
|
||||
7. ~~玩家可见的构建结果~~ ✅ **已修**(`72c4aa3`,缺陷 #36:列表路由附 `build_status`/`build_error`,面板「我的提交」展开行呈现,真机双例验证)。
|
||||
8. ~~面板文件编辑器入口~~ ✅ **已补**(`0a36b3f`,缺口补齐,真机 CDP 全链)。
|
||||
9. ~~fleet 系统服务死操作~~ ✅ **已修**(`2f90851`,缺陷 #37)。
|
||||
10. ~~S3 上传通道演练 + 存储向导逐屏~~ ✅ **已演练**(第二十四批:上传通道全链、0 缺陷;第三十二批:向导屏逐屏 + 上传取件 + UI 回滚,0 功能缺陷并修 #52)。
|
||||
11. ~~breakGlass 控制台逐屏~~ ✅ **已演练**(第二十五批:菜单 4 操作 + bootstrap 分支,缺陷 #39–#43 全修全验)。~~`felis setup` 全屏向导逐屏~~ ✅ **重跑侧已逐屏**(第二十六批),**首装侧连续走查也已完成**(第三十一批:scratch 库单次连续走完全部屏幕 + 轨道回顾,0 缺陷)。
|
||||
12. ~~`felis nano` 全链~~ ✅ **已走查**(第三十四批:HTTP 矩阵红 13→绿 60/60;systemd/配置/firewalld 三层重跑全绿;同批修复 #55——坏源静默挡住验证梯子)。
|
||||
13. ~~NetworkPolicy 真机强制矩阵~~ ✅ **已演练**(第三十五批:pod 源端口级矩阵全绿、转发型外部源「拒→放→拒」白名单闭环、host/ClusterIP/NodePort 路径全通;两条已归因注记——node-local 流量豁免、firewalld 仅放 DNAT 服务流)。
|
||||
14. ~~面板错误文案全量本地化~~ ✅ **已修**(第三十六批:#60,37 个用户可达码补齐双语映射,三类真机验证;管理面写操作 Run4c 复核 0 缺陷)。
|
||||
15. ~~管理面交互级收尾(submissions / 会话撤销 / 启动)~~ ✅ **已演练**(第三十七批:点击→构建→succeeded 全链、单条/全部撤销精确验证;同批修 #61——LP 写承诺按 LP 官方语义收窄,真机复验)。
|
||||
16. ~~缺陷 #56–#59(玩家管理回包 / 控制台回显 / 文件页 fail-fast / LP 诚实化)~~ ✅ **已修并归档**(第三十八批回填真机证据:修复前触发 + 修复后复验,四项全绿)。
|
||||
17. ~~服务器详情与运维面收尾~~ ✅ **已演练**(第三十八批:ServerFiles 写流程、备份/恢复全链、白名单/封禁闭环、ServerCard 停止、`/admin/updates` 复跑、metrics、CLI 核销——0 缺陷)。
|
||||
18. ~~reaper 多节点实机~~ ✅ **已演练**(第三十九批:VM 克隆双节点 k3s——pin 命中持盘节点并跑通、错位 pin fail-closed Pending、装置全回收;`felis-backups` PVC 的 volume node affinity 构成第二层保险)。
|
||||
19. ~~Java 插件层自动化缺口~~ ✅ **已补**(第四十批:`plugins/test.sh` + CI `plugins` 作业——三个手工测试与三个装机 jar 全进门禁,CI 4/4 绿;首跑抓出两处"没跑过"的错误:InviteCardTest 编译类路径缺 examination-api、limbo `+` 版本不可解析。同批关闭 demo-up 单起点化 #63、并复核剔除 #62)。
|
||||
20. ~~装载器 mod 层(fabric/forge/neoforge)~~ ✅ **已收口**(第四十一批:#64 三个 `gradlew` 补可执行位、#65 元数据对齐 AGPL-3.0-only;`plugins/test-mods.sh` + CI `mods` 作业 + release 双门禁;三真实专用服起服 E2E——mod 装载、`/link` 注册、控制台拒绝全绿。CI run `35892544563` = 5/5,`mods` 作业首跑即绿)。
|
||||
21. ~~全仓覆盖面对账 + 文档收尾~~ ✅ **已完成**(第四十二批:22 个 internal 包 × 面板 17 条页面路由 × 全部 tracked 脚本 × 仓库卫生逐项对账,无盲区;#66 README_EN 修复;§28 两张序列图对齐现行实现;本地 `scripts/sync.sh` 排除清单与 `go:embed` 冲突——修复+复现验证〔脚本私有,不入库〕)。
|
||||
22. ~~构建层资源回收~~ ✅ **已修**(第四十三批:#67 build Job TTL——完成 Job/Pod 不再无限堆积;真机:新构建 Job `ttl=604800` + 旧 Job 打 30s TTL 实测被 TTL 控制器收走。同批:6 轮生命周期循环 + operator/api/pg 三处注入全绿、PG 503 语义复验、30 路并发全 200)。
|
||||
23. ~~构建上下文拉取的短暂断连~~ ✅ **已修**(第四十三批追加:#68——fetch-context 有界重试〔45s 窗口 / 3s 间隔;4xx 不重试〕;真机:api 停机期连续 8 次拒连全部重试、恢复后构建 32s Complete;正常路径无重试开销。附:演练升级暴露的 `FELIS_IMAGE` 两代错位已对齐 auditfix63)。
|
||||
24. ~~modpack 规模上下文全链 + 并发构建~~ ✅ **已演练**(第四十四批:200MiB 上下文上传 0.44s → 批准 → 构建全链 13s、三路并发构建全绿;TTL 与玩家可见面在规模下复验,零缺陷。见该批节)。
|
||||
25. ~~内建底座构建(`FROM registry.felis.svc:5000/…`)~~ ✅ **已修并真机闭环**(第四十五批三连修:#70 拉取缺 `--insecure-pull`、#71 drop-ALL 卡解包、#72 trivy Java DB 未镜像——"真实模组包"形态的三个必踩点;修复后 paper 底座构建 27s Complete、`paper.jar` 扫描 0 漏洞)。
|
||||
26. ~~提交 → 装服 → 起服 闭环~~ ✅ **已演练**(第四十五批 capstone:构建产物 PATCH 为服务器镜像〔白名单门通过〕→ wake → Running 1/1 → 用户 marker 可读 + `Done (4.678s)!`;test-one 已复原为 paper:demo/Stopped)。
|
||||
27. ~~文档与测试工具面~~ ✅ **已修**(第四十五批:#69 README 部署承诺对齐实现;#73 bootstrap 测试密闭化〔VM 140 PASS〕;#74 console 断连测试消抖;CI 全绿)。
|
||||
28. ~~提交上传面的上限与生命周期~~ ✅ **已修并真机闭环**(第四十六批:#75 per-user create/upload 冷却〔429〕、pending ≤5〔403〕、2GiB 存储预算〔403〕;#76 撤回〔owner+pending CAS〕与管理员删除〔行+blob 双清〕;面板两步确认 CDP 全绿)。
|
||||
29. ~~未完成引导的导航(tracker #8)~~ ✅ **已修并真机闭环**(第四十六批 #77:`403 setup_required` → 自动 `/setup`;CDP 复验)。
|
||||
30. ~~已装机系统服的新增 CR 字段(tracker #1)~~ ✅ **已修并真机闭环**(第四十六批 #78:`sudo felis converge`——零值才填、非零不覆写;剥字段→填回→幂等三连真机过)。
|
||||
31. ~~troubleshooting `[INERT]` 图例~~ ✅ **已修**(第四十六批 #79,文档级)。
|
||||
|
||||
## 剩余待演练队列(截至第四十三批)
|
||||
|
||||
- ~~reaper 多节点实机~~ ✅ 已完结(第三十九批:临时克隆第二节点组双节点 k3s 实机——pin 命中持盘节点并跑通、错位 pin fail-closed Pending;装置已回收)。
|
||||
- ~~面板交互级收尾~~ ✅ 已完结(第三十六–三十八批:管理面写操作、submissions/会话、服务器详情页全链,0 缺陷)。
|
||||
- ~~`demo-up.sh`~~ ✅ 已完结(第四十批:单起点化 #63——bootstrap 装机 + 三项硬校验 + `felis setup` 交棒;T1/T2/T3 真机全绿)。
|
||||
- ~~Java 插件层自动化~~ ✅ 已补(第四十批:`plugins/test.sh` + CI `plugins` 作业;三个手工测试与三个装机 jar 首进 CI)。
|
||||
- ~~装载器 mod 层(fabric/forge/neoforge)~~ ✅ 已收口(第四十一批:#64/#65 修复 + `plugins/test-mods.sh`/CI `mods`/release 双门禁 + 三真实专用服起服 E2E)。
|
||||
- ~~全仓覆盖面对账~~ ✅ 已复核(第四十二批:模块×台账逐项对账无盲区;批次产出为文档/本地工具层面的修复,见该批节)。
|
||||
- ~~长稳/故障注入(首轮)~~ ✅ 已完成(第四十三批:6 轮起停循环 + 3 处注入 + PG 断连 503 + 30 路并发 + 25 分钟持续轮询;见该批节)。
|
||||
- ~~modpack 规模上下文 + 并发构建~~ ✅ 已完成(第四十四批:200MiB 上下文全链 13s + 并发 ×3 全绿;见该批节)。
|
||||
- ~~提交 → 装服 全链(含平台底座 FROM)~~ ✅ 已完成(第四十五批:三连修 #70/#71/#72 + capstone 提交→装服→起服;见该批节)。
|
||||
|
||||
**队列现已清空。**(真实游戏客户端进服已按决定跳过,用内部 mint/approve hook 链替代;不依赖客户端的最大真机面见第四十批节 + 第四十一批节〔装载器三服起服〕;第四十二批为覆盖面对账复核。)
|
||||
|
||||
## 系统性观察
|
||||
|
||||
@@ -54,8 +954,27 @@ IPv6-only 接入(`ssh -6 -i ~/.ssh/id_ed25519 root@fdb2:2c26:f4e4:0:21c:42ff:f
|
||||
|
||||
## 复现入口速查
|
||||
|
||||
- 面板会话 cookie:`/tmp/felis-cookies.json`;API 助手:`/tmp/fcurl.sh`
|
||||
- 面板会话 cookie:`/tmp/owner-cookies2.txt`(2026-09-23 铸;auditfix42 部署后仍有效;staff 邮件门被设计拒绝 → 走第二十八批的 op-login + 内部面 approve hook 链);API 助手:`/tmp/fcurl.sh`
|
||||
- 测试服:`test-one`(minecraft ns,stopped);合法备份 `bk-47ee2e7e96e5a4ca9d0e51b805518bac`
|
||||
- CDP 调试口:Mac `127.0.0.1:9333`(独立 Chrome,profile `/tmp/felis-chrome2`);WebAuthn 虚拟认证器需在**同一 CDP 会话**内完成仪式,且先 `Page.bringToFront`(否则 NotAllowedError: page does not have focus)
|
||||
- 分支已部署到 VM:`felis-api` 镜像 = `felis:auditfix7`(含全部修复)
|
||||
- VM 内部面:`k3s kubectl -n felis port-forward svc/felis-api-internal 18081:8081`(Pod 重建后转发会悬死,需重启);reaper CronJob(minecraft ns)已应用但 `suspend=true`
|
||||
- 已部署到 VM:控制面(api/operator/reaper + `FELIS_IMAGE`)= `registry.felis.svc:5000/felis/felis:auditfix61`(含至 #61 的全部修复;镜像托管于内建 registry);系统服/推荐 ref = `registry.felis.svc:5000/felis/{limbo,lobby,paper}:demo`;源码快照 `/opt/felis/src`(`b4ef42d`;上一版在 `/opt/felis/src.bak`,更早 `src.old43`);host 二进制 = `/usr/local/bin/felis` = `v0.0.0+fix61`(含 #53–#61;旧件留档 `/root/felis-fix54.bin`(sha `a0b29153…`)、`/root/felis-fix52.bin`、`/root/felis-auditfix44.bin`);**docker 守护进程在第三十七批构建后已停(用前 `sudo systemctl start docker`)**;**world 执行器以 root+DAC_OVERRIDE 运行**;reaper CronJob(minecraft ns)`suspend=false / …auditfix61 / nodeSelector=localhost.localdomain`;迁移 `schema_migrations` max=21;SMTP 密码 env 待「configure email」刷新时落地
|
||||
- 安装器重跑配方(第三十批复用;先同步 `/opt/felis/src` 再跑):`cd /root && FELIS_SKIP_FETCH=1 FELIS_INSTALL_MODE=full FELIS_IMAGE=registry.felis.svc:5000/felis/felis:<tag> FELIS_WORLDS_HOST_PATH=/var/lib/rancher/k3s/storage nohup bash /opt/felis/src/deploy/bootstrap.sh > /root/bootstrap-<tag>.log 2>&1 &`;完成后 `grep -c '\[fail\]'` 应为 0
|
||||
- 第三十一批 drill 留档:`/root/preTUI43/`(首装走查前快照:两 toml + 两 ns Secret + sha256、走查后 Secret 快照)、`/root/pre51-replica.toml`/`post51-replica.toml`/`post51b-replica.toml`(#51 红/绿副本证据);走查残留(scratch 库、scratch 配置、hba 行、tmux 会话)均已清理
|
||||
- 第三十二批 drill 留档:`/root/preS3/`(S3 走查前快照:两 toml + 两 ns Secret + sha256 + deploys.txt)、`/root/s3wiz/`(posts3 快照、postroll-sha256、fix52-red.txt、fix52-green.txt)、`/root/felis-fix52`(含 #52 的宿主二进制);MinIO 容器+卷+两镜像、`felis-uploads-s3` Secret、DB 行(`image_submissions` `sub-853a4e2ba4ba6443`)、/tmp 残留均已清理,docker 已停
|
||||
- 第三十三批 drill 留档(VM):`/tmp/felis-fix53`、`/tmp/felis-fix54`(宿主二进制;sha `9afd141d…` / `a0b29153…`)、`/root/felis-fix52.bin`;`/tmp/player-cookies.txt`(player 会话);面板 CDP 驱动脚本 `/tmp/cdp-updates2.js`(Mac 侧)
|
||||
- 第三十四批 drill 留档(VM):`/srv/nanotest/`(nano 装置:`nano-stub.py` + 5 驱动脚本 + `matrix/` 红原跑 / `matrix-fix55/` 绿原跑)与四层重跑落盘 `matrix-red.r2.log`(`pass=47 fail=13`)/ `matrix-fix55.r2.log`(`pass=60 fail=0`)/ `nano-service.r2.log` / `nano-config.r2.log` / `nano-fw.r2.log`;`nano-stub` 临时单元仍在跑(收尾 `systemctl stop nano-stub`);宿主二进制 `/usr/local/bin/felis-nano-test` = fix55 改名件(sha `2c8c34fd…`,与 `/usr/local/bin/felis` 同物)
|
||||
- 第三十五批 drill 留档(VM):`/srv/npdrill/`(netpol 装置全套:`np-matrix.sh` + `np-matrix-run1/2.log`、`np-netns.sh`、`netns-probe.py`、`np-cidr-test.sh/.log`、`spec-before/after.json`、`td3.log`(包级归因)、`tcpdump-8081.log`(刷新存活)、`iptables-save.txt` 与 `nft-rules.txt`);netns/veth、firewalld 临时挂载、策略改动全部已清理/复原(spec md5 `e6b07440…` 前后一致)
|
||||
- pgint 正确跑法(走 VM 的 PG,勿在本机起容器):Mac 侧短隧道 `ssh -6 -i ~/.ssh/id_ed25519 -N -L 15433:127.0.0.1:5432 root@…` → `FELIS_TEST_PG_URL='postgres://felis:<pw>@localhost:15433/felis_pgint?sslmode=disable' go test -tags pgint ./internal/pgint/ -count=1`(pg_hba 需 `host felis_pgint felis 127.0.0.1/32 scram-sha-256`;第三十一批把现场丢失的这条补回;隧道用完即 kill)
|
||||
- breakGlass TUI 驱动法:VM tmux `new-session -d -s bg -x 160 -y 45 "/root/felis-auditfixNN breakGlass"` + `set-window-option -t bg remain-on-exit on`(否则退出摘要读不到);send-keys/capture-pane 驱动;退出码 1 = 操作失败(卡上会显示原因)
|
||||
- 构建链路 drill 现成条件:`felis-build` 里有 `felis-service-token`(Secret 复制);`felis-config` 的 kaniko/trivy pin 指向内建 registry 的 mirror 路径(`registry.felis.svc:5000/mirror/...`,见 §8e)——本批后安装器重跑会 **carry** 这些 pin(不再被写盘冲掉;#49 复验点);直构一行:`POST /api/v1/images/build`(context=`http://felis-api-internal.felis.svc.cluster.local:8081/api/v1/internal/submissions/sub-bf7dc18e9dd97dc2/context`)
|
||||
- registry 工具:`curl -s http://127.0.0.1:5000/v2/_catalog`、`/v2/<repo>/tags/list`;**取 manifest 必须带 `Accept:`(OCI index/manifest list;缺 Accept 的 GET 会 404,别误判没推上)**;GC 演练配方:`k3s ctr images rm <ref>`(要清到 blob 级再加 `k3s ctr content prune references`)→ 删 pod / `rollout restart` → `kubectl describe` 看 `Pulled … in …ms`
|
||||
- setup 重跑向导(第二十六批):状态屏 `c/s/e` 三流已走查;**reconfigure 非只读**——storage 选 Local / 提交 S3 即 apply + 滚 API(幂等),email 表单 esc 无副作用;从 S3 表单 esc 退回会重置为 Local 预选;存储屏(第三十二批补充):chooser 预选当前值(↑↓ 切换),Local 直接 enter 提交、S3 五字段 `enter next`/末字段 `enter submit`,提交后 10–40s(含 API rollout status 180s 上限)
|
||||
- VM 内部面:`k3s kubectl -n felis port-forward svc/felis-api-internal 18081:8081`(Pod 重建后转发会悬死,需重启);reaper CronJob(minecraft ns)已应用且已收敛(`suspend=false`)
|
||||
- 第三十八批收尾复查:docker `inactive`;控制面三处 = `auditfix61`;`/opt/felis/src` = `b4ef42d` tarball 展开(无 `.git`);reaper `suspend=false` + `nodeSelector=localhost.localdomain`;test-one `Stopped`;`resolvecheck` 文件接口 409 `no_world_volume`(0.03s)
|
||||
- 第三十九批 drill(已回收):临时 VM `felis-node2`(`prlctl clone --linked` 克隆 node1;node2 停用 felis-velocity/k3s-server/docker 后以 agent 加入)——流程与两向对照见该批节;node1 为克隆经历过一次正常重启(组件全回归、docker 停回 `inactive`);收尾 job/节点对象/VM 全部删除,集群回到单节点
|
||||
- 第四十批工具与留档:VM `/tmp/mcprobe.py`(纯 socket MC 探针:`python3 /tmp/mcprobe.py status <sub>.<root>`;`login <sub>.<root> <name>` 打登录首包);插件门禁跑法 = `docker run --rm -v /opt/felis/src:/src:z -w /src gradle:8.14-jdk21 bash plugins/test.sh`(或宿主 `bash plugins/test.sh`,需 JDK 21 + Gradle 8.14;日志 `/root/plugins-test-b40.log`);T3(demo-up 默认臂全量重跑)日志 `/root/demo-up-b40.log`;`/opt/felis/src` = `c59b387` 快照(上一版 `src.bak40`;此后全量重跑用 `sudo bash /opt/felis/src/deploy/demo-up.sh` 即可,SKIP 开关见脚本头);CI run `35888009965`(含新 `plugins` 作业首跑绿)
|
||||
- 第四十三批工具与留档:控制面三处 = `registry.felis.svc:5000/felis/felis:auditfix62`(含 #67;镜像源 `10.43.182.43:5000/felis/felis:auditfix62`,宿主 docker build 后 `docker push 10.43.182.43:5000/...`);`/opt/felis/src` = `2755e41` 快照(上一版 `src.bak42`);owner 会话重铸 = `bash /root/mint-owner-43.sh`(cookie 落 `/tmp/owner-jar43.txt`);直构触发(含 TTL 检查)= `curl -sk -b /tmp/owner-jar43.txt -X POST https://127.0.0.1:30443/api/v1/images/build -H 'Content-Type: application/json' -d '{"image_ref":"registry.felis.svc:5000/e2e/ttl-probe:latest","dockerfile":"FROM scratch\nLABEL felis=x\n","context_ref":"http://felis-api-internal.felis.svc.cluster.local:8081/api/v1/internal/submissions/sub-bf7dc18e9dd97dc2/context"}'` → `kubectl -n felis-build get job build-bld-<id> -o jsonpath='{.spec.ttlSecondsAfterFinished}'` 应为 `604800`;压测编排 `/root/soak43.sh`(日志 `soak43.log`,含 6 轮循环 + 3 注入)、长轮询 `/root/poll43-long.sh`(日志 `poll43-long.log`);镜像构建日志 `/root/felis-image-build43.log`。注意:`lobby`/`login` 是保留名,internal per-server 路由(status/wake 等)对它们**按设计**返回 400,勿作缺陷误报。
|
||||
- 第四十三批追加(#68)留档:控制面 api/operator/reaper 镜像 + api/operator `FELIS_IMAGE` = `registry.felis.svc:5000/felis/felis:auditfix63`(`felis version` = `v0.0.0+fix63`;升级时 `set image` 与 `FELIS_IMAGE` env 必须成对改,否则构建 Job 的 fetch 容器会悄悄用旧镜像);`/opt/felis/src` = `b76d0ac` 快照(上一版 `src.bak43` = `2755e41`);复刻 Job `/root/job68-new.json`(由红证据 Job `build-bld-1790187851749257160` 派生:换名 `build-retry68`、剥 controller uid 标签与 selector、镜像改 auditfix63、复用 `felis-service-token` Secret);重演验证法:`kubectl -n felis scale deploy/felis-api --replicas=0` → `kubectl -n felis-build create -f /root/job68-new.json` → `kubectl -n felis-build logs <pod> -c context-fetch`(应见 `connection refused; retrying`)→ `scale --replicas=1` → Job 收敛 `Complete`;对照:api 在线直构成功且 fetch 日志无重试行。
|
||||
- 第四十四批工具与留档:200MiB 上下文构造 = 5×40MiB `head -c 41943040 /dev/urandom` + `Dockerfile`(`FROM scratch` + `COPY payload /payload`),`tar -czf /root/bigctx.tar.gz Dockerfile payload`;上传 = `curl -sk -b /tmp/owner-jar43.txt -X POST -H "Content-Type: application/gzip" --data-binary @/root/bigctx.tar.gz https://127.0.0.1:30443/api/v1/me/submissions/<id>/context`;采样 = `k3s crictl stats`(此版 **无 `--no-trunc`**,列解析取 `$2=NAME $4=MEM`);docker 缓存清理 = `systemctl start docker && docker builder prune -af && systemctl stop docker`;留档 `/root/stats44.log`、`/root/stats44-build.log`、提交 `sub-cc160b3ad19080f9`、镜像 `e2e/conc-b44-{1,2,3}`。
|
||||
- 第四十五批工具与留档:从平台底座构建的配方 = 提交上下文含 `Dockerfile`(`FROM registry.felis.svc:5000/felis/paper:demo` + `COPY marker.txt /felis-probe-marker.txt`);装服验证 = `PATCH /api/v1/servers/test-one {"image":"registry.felis.svc:5000/user-uploads/<sub>:latest"}` → `POST /servers/test-one/wake` → `kubectl -n minecraft exec test-one-0 -- cat /felis-probe-marker.txt`;trivy 双 DB 镜像 = `mirror/trivy-db:2` + `mirror/trivy-java-db:1`(Java DB digest `5766dfbb…`;两键在 `/etc/felis/felis.{host,pod}.toml` 与 felis-config Secret 双副本);`bootstrap_test.sh` 需在 Linux 跑(VM 留档 `/root/btest72/`,应 140 PASS);本批提交样本 `sub-{b45d416f4af811ef(无镜像),83f677e41acd80c3,171e8f967f0de3bc,c006cbd633317ccb,cc160b3ad19080f9}`;留档 `/root/probe44/`、`/root/probe44.tar.gz`、`/root/probe44*.sid`。
|
||||
- 第四十六批工具与留档:控制面三处 = `registry.felis.svc:5000/felis/felis:auditfix77`(宿主 docker 源 `10.43.182.43:5000/felis/felis:auditfix77`,`v0.0.0+fix77`;api/operator + minecraft ns `felis-reaper` CronJob + api/operator `FELIS_IMAGE` 四处成对改);`/opt/felis/src` = `b9ebc87` 快照(上一版 `src.bak46`);宿主 drill 二进制 `/root/felis-fix77.bin`(Mac 侧 `CGO_ENABLED=0 GOOS=linux GOARCH=arm64 go build -ldflags="-s -w -X main.version=v0.0.0+fix77" ./cmd/felis`;scp 到 IPv6 主机时主机位必须写 `root@[fdb2:…]` 方括号);#75/#76/#77/#78 留档 `/root/probe77/`(`create-lane.log`、`lane2.log`、`lane3.log`〔含玩家会话矩阵 + 分级 ZT 对照 + curl 7.76 在 `-H Host:` 下**不发 jar cookie**的坑——host 路由 drill 用显式 `-H "Cookie: felis_session=…"`〕、`converge-1.log`/`converge-2.log`、`crs-before.yaml`、各步请求/响应 json);面板 CDP 截图(Mac):`/tmp/setup8-redirect.png`、`/tmp/withdraw-{1,2}-*.png`、`/tmp/admdelete-{1,2}-*.png`,脚本 `/tmp/cdp-setup8.js`、`/tmp/cdp-withdraw76.js`、`/tmp/cdp-admdelete76.js`;锁定会话铸法 = `sha256('drill-locked-77')` 直插 `sessions`(用户 `3f2c1b0a-…`);玩家会话铸法 = `[email protected]` 邮件 OTP(码在 felis-api 日志 `no Mailer configured` 行;`edge-dis2` 是软删死账号,登录门按设计排除);pgint 前置:`pg_hba` 需 `host felis_pgint felis 127.0.0.1/32 scram-sha-256`(本批第二次丢失并补回);发布链 smoke = tag `v0.1.0-rc1`(release run `35949621233`),prerelease 不移动 `/releases/latest`——无 stable 时 bootstrap 默认通道给出显式指引后 die、`felis update` 404 报错可读;首个稳定 tag 是产物决定(未代拍板)。
|
||||
@@ -67,6 +67,20 @@ go test ./internal/api
|
||||
go test ./cmd/felis
|
||||
```
|
||||
|
||||
The hermetic suites run against in-memory fakes; the business stores' SQL is
|
||||
verified separately against a real Postgres, on a throwaway database whose name
|
||||
must contain `pgint` (the harness drops and recreates its schema and replays the
|
||||
embedded migrations):
|
||||
|
||||
```bash
|
||||
FELIS_TEST_PG_URL='postgres://felis:***@127.0.0.1:5432/felis_pgint?sslmode=disable' \
|
||||
go test -tags pgint ./internal/pgint/ -v
|
||||
```
|
||||
|
||||
Run it after touching anything under `internal/api/pgrepo.go`, `internal/submit`,
|
||||
or `internal/build` that speaks SQL: the fakes encode the contract, and this
|
||||
suite exists to catch the drift between the fakes and the real queries.
|
||||
|
||||
Build the CLI:
|
||||
|
||||
```bash
|
||||
|
||||
@@ -17,10 +17,10 @@ A Kubernetes-driven Minecraft server hosting platform — one command to deploy,
|
||||
|
||||
- **即开即玩**:玩家尝试连接时自动唤醒服务器,空闲后自动休眠,像游戏主机一样省资源。
|
||||
- **Web 控制面板**:浏览器中查看服务器状态、在线玩家与资源用量,管理备份与恢复。
|
||||
- **自动备份与恢复**:定时将世界打包存档,支持从任意备份点一键回滚。
|
||||
- **智慧回收**:超过 15 天无人游玩的世界自动备份后删除,释放磁盘空间。
|
||||
- **备份与恢复**:一键把整服数据(世界、配置、插件/模组,即整个 /data 卷)打包进集群内的归档库,支持从任意备份点回滚;默认安装就已启用(归档 PVC 与路径由安装器一并生成)。
|
||||
- **智慧回收(可选开启)**:超过 15 天无人游玩的世界自动备份后删除,释放磁盘空间;安装时设置 `FELIS_WORLDS_HOST_PATH`(k3s 默认 `/var/lib/rancher/k3s/storage`)即启用每日回收,不设置则不删任何世界。
|
||||
- **多核心支持**:兼容 Paper、Fabric、Forge、NeoForge,经由 Velocity 代理统一入口。
|
||||
- **模组自助提交**:玩家自行上传模组包,服主审批通过后自动构建并部署。
|
||||
- **模组自助提交**:玩家自行上传模组包,服主审批通过后自动构建;构建产物进入镜像白名单,可直接选用为服务器镜像完成部署。
|
||||
- **Passkey 登录**:支持指纹、面容、硬件密钥等无密码认证方式。
|
||||
- **零信任安全**:面板流量由 Cloudflare Access 保护,集群内 API 不暴露到公网。
|
||||
|
||||
@@ -29,7 +29,7 @@ A Kubernetes-driven Minecraft server hosting platform — one command to deploy,
|
||||
在准备好的 Linux 主机上执行:
|
||||
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/MliroLirrorsIngenuity/Felis/main/deploy/bootstrap.sh | sudo bash
|
||||
curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash
|
||||
```
|
||||
|
||||
脚本将自动安装 K3s、部署控制平面并启动设置向导。完成后浏览器访问已配置的域名进入控制面板即可使用。
|
||||
@@ -41,7 +41,7 @@ curl -fsSL https://raw.githubusercontent.com/MliroLirrorsIngenuity/Felis/main/de
|
||||
> export FELIS_GITHUB_TOKEN=<对本仓库有读权限的 token>
|
||||
> printf 'header = "Authorization: Bearer %s"\n' "$FELIS_GITHUB_TOKEN" \
|
||||
> | curl -fsSL --config - -H "Accept: application/vnd.github.raw" \
|
||||
> https://api.github.com/repos/MliroLirrorsIngenuity/Felis/contents/deploy/bootstrap.sh \
|
||||
> https://api.github.com/repos/FelisMC/Felis/contents/deploy/bootstrap.sh \
|
||||
> | sudo -E bash
|
||||
> ```
|
||||
>
|
||||
|
||||
+25
-4
@@ -17,10 +17,10 @@ Table of Contents
|
||||
|
||||
- **Wake on Join**: Servers start automatically when a player connects, and stop when idle — like hibernate for your server.
|
||||
- **Web Dashboard**: Monitor server status, online players, and resource usage from your browser, with backup and restore management.
|
||||
- **Auto Backup & Restore**: Scheduled world backups with one-click rollback from any backup point.
|
||||
- **World Reaper**: Worlds idle for more than 15 days are automatically backed up and removed to free disk space.
|
||||
- **Backup & Restore**: One-click snapshots of a server's whole data volume (worlds, config, plugins/mods — the entire /data volume) into the cluster's archive store, with rollback from any backup point — enabled by default (the installer renders the archive PVC and its path).
|
||||
- **World Reaper** (opt in): Worlds idle for more than 15 days are automatically backed up and removed to free disk space. Enable it by setting `FELIS_WORLDS_HOST_PATH` at install time (on k3s: `/var/lib/rancher/k3s/storage`); without it, no world is ever deleted.
|
||||
- **Multi-core Support**: Compatible with Paper, Fabric, Forge, and NeoForge, federated behind a Velocity proxy.
|
||||
- **Modpack Submission**: Players submit custom modpacks; admin approval triggers automatic build and deployment.
|
||||
- **Modpack Submission**: Players submit custom modpacks; admin approval triggers an automatic build, and the result is whitelisted as a server image you can select to deploy.
|
||||
- **Passkey Login**: Passwordless authentication via fingerprint, face recognition, or hardware security keys.
|
||||
- **Zero Trust Security**: Panel traffic protected by Cloudflare Access; the internal API is never exposed to the internet.
|
||||
|
||||
@@ -29,11 +29,32 @@ Table of Contents
|
||||
On a prepared Linux host, run:
|
||||
|
||||
```bash
|
||||
curl -fsSL https://raw.githubusercontent.com/MliroLirrorsIngenuity/Felis/main/deploy/bootstrap.sh | sudo bash
|
||||
curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash
|
||||
```
|
||||
|
||||
The script installs K3s, deploys the control plane, and launches a setup wizard. Once done, open your browser at the configured domain.
|
||||
|
||||
> **This repository is currently private**, so the command above returns 404. Use the
|
||||
> credentialed form instead; the installer itself needs the same token to resolve and
|
||||
> download the release, so pass it through with `sudo -E`:
|
||||
>
|
||||
> ```bash
|
||||
> export FELIS_GITHUB_TOKEN=<a token with read access to this repository>
|
||||
> printf 'header = "Authorization: Bearer %s"\n' "$FELIS_GITHUB_TOKEN" \
|
||||
> | curl -fsSL --config - -H "Accept: application/vnd.github.raw" \
|
||||
> https://api.github.com/repos/FelisMC/Felis/contents/deploy/bootstrap.sh \
|
||||
> | sudo -E bash
|
||||
> ```
|
||||
>
|
||||
> The token reaches `curl --config -` over stdin instead of the command line: argv is
|
||||
> readable by any local user via `/proc`, and that is exactly why the installer's
|
||||
> internal `github_api` uses the same form.
|
||||
|
||||
Rerunning this command is also how you upgrade felis-api to a newer version (`felis setup`
|
||||
cannot — it uses the binary already installed on the host). The rerun keeps the installed
|
||||
root domain but **not** the channel: if this host follows main, also
|
||||
`export FELIS_VERSION_BOOTSTRAP=dev`.
|
||||
|
||||
## Build from Source
|
||||
|
||||
Felis is built with Go and Node.js:
|
||||
|
||||
+54
-14
@@ -18,6 +18,7 @@ import (
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/fileedit"
|
||||
"felis.lolicon.best/internal/mail"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/panel"
|
||||
"felis.lolicon.best/internal/passkey"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
@@ -142,10 +143,14 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
// build namespace and pushes to the internal registry. The build Pod never
|
||||
// holds DB credentials — felis-api owns the PG store and admits scanned
|
||||
// images, so the Builder is constructed here with both bindings.
|
||||
buildCfg := buildConfig(cfg)
|
||||
// The fetch initContainer runs THIS image's fetch-context entrypoint, so the
|
||||
// build config carries the api's own image (the platform sets FELIS_IMAGE).
|
||||
buildCfg.FelisImage = os.Getenv("FELIS_IMAGE")
|
||||
builder := &build.Builder{
|
||||
Store: build.NewPGStore(drv.DB()),
|
||||
Jobs: build.NewK8sJobs(cl, buildConfig(cfg)),
|
||||
Config: buildConfig(cfg),
|
||||
Jobs: build.NewK8sJobs(cl, buildCfg),
|
||||
Config: buildCfg,
|
||||
}
|
||||
|
||||
// User-modpack approval lane (user-directed extension over §16; see
|
||||
@@ -159,14 +164,19 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
// The blob upload transport is selected by the shape of user_uploads_context —
|
||||
// the two backends the setup wizard chooses between. A local path wires
|
||||
// LocalContextStore (the mounted uploads PVC); an s3:// base wires
|
||||
// S3ContextStore when its credentials resolve. Either way the store's target is
|
||||
// derived from the SAME config field the context ref uses, so the blob lands
|
||||
// exactly where Kaniko's --context points. Anything else — or an s3:// base with
|
||||
// no credentials configured — leaves Blobs nil so POST
|
||||
// S3ContextStore when its credentials resolve. Anything else — or an s3:// base
|
||||
// with no credentials configured — leaves Blobs nil so POST
|
||||
// /me/submissions/{id}/context returns 503, honest like the restore executor
|
||||
// when its PVC is not supplied. (Letting the sandboxed Kaniko build Pod READ the
|
||||
// context — PVC mount for local, creds+egress for S3 — is a separate deployment
|
||||
// integration.)
|
||||
// when its PVC is not supplied.
|
||||
//
|
||||
// Reading the blob back is the API's job, not Kaniko's: the build Pod runs in
|
||||
// another namespace and can neither mount the uploads PVC (a PVC does not cross
|
||||
// namespaces) nor hold object-store credentials, so ContextBaseURL makes the
|
||||
// derived context ref an internal-face URL that the build Job's fetch
|
||||
// initContainer streams (cmd/felis fetch-context). The platform renders this
|
||||
// address into the api Deployment (felis API base URL env); the fallback keeps
|
||||
// a hand-rolled deployment working under the platform's default control
|
||||
// namespace.
|
||||
contextBase := cfg.Registry.UserUploadsContext
|
||||
var blobs submit.Blobs
|
||||
switch {
|
||||
@@ -186,11 +196,12 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintf(stderr, "felis api: user-uploads context %q is neither a local path nor an s3:// base — modpack upload transport disabled (POST /api/v1/me/submissions/{id}/context returns 503)\n", contextBase)
|
||||
}
|
||||
submissions := &submit.Manager{
|
||||
Store: submit.NewPGStore(drv.DB()),
|
||||
Builds: builder,
|
||||
Registry: cfg.Registry.URL,
|
||||
ContextStore: contextBase,
|
||||
Blobs: blobs,
|
||||
Store: submit.NewPGStore(drv.DB()),
|
||||
Builds: builder,
|
||||
Registry: cfg.Registry.URL,
|
||||
ContextStore: contextBase,
|
||||
ContextBaseURL: internalAPIBaseURL(),
|
||||
Blobs: blobs,
|
||||
}
|
||||
|
||||
// Restore subsystem (spec §7): the weak-SA restore Job mounts the target
|
||||
@@ -259,6 +270,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
Builder: builder,
|
||||
Restorer: restorer,
|
||||
Backuper: backuper,
|
||||
JobStatus: api.NewK8sJobStatus(cl, cfg.K8s.Namespace),
|
||||
Files: files,
|
||||
Submissions: submissions,
|
||||
Mailer: mailer,
|
||||
@@ -279,6 +291,11 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int {
|
||||
AdminHostname: cfg.Auth.AdminHostname,
|
||||
PanelHostname: cfg.Auth.PanelHostname,
|
||||
WakeCooldown: 30 * time.Second,
|
||||
// The user-modpack lane's per-user throttles: a create spaces out
|
||||
// review-queue rows, an upload spaces out (up to 1 GiB) context streams.
|
||||
// Separate keys, so the normal create→upload sequence stays immediate.
|
||||
SubmitCreateCooldown: 30 * time.Second,
|
||||
SubmitUploadCooldown: 15 * time.Second,
|
||||
// Bound concurrent console/build-log SSE streams per principal. Generous enough
|
||||
// for legitimate multi-tab / multi-server watching, while capping how many
|
||||
// upstream follow connections a single caller can tie up if their streams stall.
|
||||
@@ -401,9 +418,32 @@ func buildConfig(cfg *config.Config) build.Config {
|
||||
return build.Config{
|
||||
Namespace: cfg.Registry.BuildNamespace,
|
||||
RegistryURL: cfg.Registry.URL,
|
||||
// Empty overrides fall back to the build package's defaults, so an
|
||||
// install that has not imported kaniko/trivy keeps the compiled-in refs
|
||||
// (and fails loudly on pull rather than silently building with the wrong
|
||||
// image).
|
||||
KanikoImage: cfg.Registry.KanikoImage,
|
||||
TrivyImage: cfg.Registry.TrivyImage,
|
||||
CPULimit: cfg.Registry.BuildCPULimit,
|
||||
MemLimit: cfg.Registry.BuildMemLimit,
|
||||
// Empty keeps Trivy's own default; an install with builds points this at
|
||||
// the internal DB mirror (see config.RegistryConfig.TrivyDBRepository).
|
||||
TrivyDBRepository: cfg.Registry.TrivyDBRepository,
|
||||
TrivyJavaDBRepository: cfg.Registry.TrivyJavaDBRepository,
|
||||
}
|
||||
}
|
||||
|
||||
// internalAPIBaseURL resolves the platform's internal-face base URL: the address
|
||||
// the platform rendered into this pod (felis API base URL env), or — for a
|
||||
// hand-rolled deployment that set none — the platform default control namespace,
|
||||
// the same fallback setup.go uses to hand the login gate its address.
|
||||
func internalAPIBaseURL() string {
|
||||
if base := os.Getenv(naming.EnvAPIBaseURL); base != "" {
|
||||
return base
|
||||
}
|
||||
return platform.InternalAPIBaseURL(platform.DefaultControlNamespace)
|
||||
}
|
||||
|
||||
// uploadsSchemeRE matches a leading URL scheme like "s3://" or "gs://".
|
||||
var uploadsSchemeRE = regexp.MustCompile(`^[a-zA-Z][a-zA-Z0-9+.-]*://`)
|
||||
|
||||
|
||||
@@ -44,6 +44,32 @@ func TestAuthSourcesFromConfig(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestBuildConfig_ProjectsOverrides pins the [registry] overrides reaching the
|
||||
// build subsystem: unset fields must stay EMPTY (the build package's compiled-in
|
||||
// defaults apply there, not here), and set fields must pass through verbatim —
|
||||
// an air-gapped install points these at its imported mirrors.
|
||||
func TestBuildConfig_ProjectsOverrides(t *testing.T) {
|
||||
empty := buildConfig(&config.Config{})
|
||||
if empty.KanikoImage != "" || empty.TrivyImage != "" || empty.CPULimit != "" || empty.MemLimit != "" {
|
||||
t.Errorf("empty registry config must project empty overrides (defaults live in internal/build), got %+v", empty)
|
||||
}
|
||||
full := buildConfig(&config.Config{Registry: config.RegistryConfig{
|
||||
URL: "registry.felis.svc:5000",
|
||||
BuildNamespace: "felis-build",
|
||||
KanikoImage: "reg/kaniko:v1",
|
||||
TrivyImage: "reg/trivy:v1",
|
||||
BuildCPULimit: "1",
|
||||
BuildMemLimit: "2Gi",
|
||||
}})
|
||||
if full.KanikoImage != "reg/kaniko:v1" || full.TrivyImage != "reg/trivy:v1" ||
|
||||
full.CPULimit != "1" || full.MemLimit != "2Gi" {
|
||||
t.Errorf("registry overrides did not reach build.Config: %+v", full)
|
||||
}
|
||||
if full.Namespace != "felis-build" || full.RegistryURL != "registry.felis.svc:5000" {
|
||||
t.Errorf("namespace/registry url must keep projecting: %+v", full)
|
||||
}
|
||||
}
|
||||
|
||||
// TestNewAPIServerSetsHardenedTimeouts pins the gosec-G112 hardening on every
|
||||
// felis-api listener: the shared factory must bound the header and idle phases
|
||||
// (Slowloris + idle-connection exhaustion) while leaving WriteTimeout UNSET, because
|
||||
|
||||
+33
-9
@@ -86,23 +86,31 @@ func requestBackup(ctx context.Context, hc *http.Client, baseURL, token, name, o
|
||||
|
||||
// backupErrorFromResponse turns a non-202 into a human message. The well-known codes get
|
||||
// an operator-facing explanation; anything else falls back to the API's
|
||||
// {"error":{message}} body, then the bare status code.
|
||||
// {"error":{code,message}} body, then the bare status code.
|
||||
func backupErrorFromResponse(resp *http.Response) error {
|
||||
switch resp.StatusCode {
|
||||
case http.StatusConflict: // not_stopped
|
||||
return fmt.Errorf("the server must be stopped before its world can be backed up — halt it first")
|
||||
case http.StatusServiceUnavailable: // backup_unavailable
|
||||
return fmt.Errorf("the backup subsystem is not configured on felis-api (FELIS_IMAGE / FELIS_BACKUP_PVC unset)")
|
||||
case http.StatusNotFound:
|
||||
return fmt.Errorf("no such server")
|
||||
}
|
||||
var e struct {
|
||||
Error struct {
|
||||
Code string `json:"code"`
|
||||
Message string `json:"message"`
|
||||
} `json:"error"`
|
||||
}
|
||||
raw, _ := io.ReadAll(io.LimitReader(resp.Body, 1<<16))
|
||||
_ = json.Unmarshal(raw, &e)
|
||||
|
||||
switch resp.StatusCode {
|
||||
case http.StatusConflict:
|
||||
// Two refusals share 409: the stopped gate and the missing-world-volume
|
||||
// gate. The body's code distinguishes them; a code-less body reads as the
|
||||
// stopped gate (the only 409 before the volume gate existed), and any other
|
||||
// coded 409 falls through to the API's own operator text.
|
||||
if e.Error.Code == "" || e.Error.Code == "not_stopped" {
|
||||
return fmt.Errorf("the server must be stopped before its world can be backed up — halt it first")
|
||||
}
|
||||
case http.StatusServiceUnavailable: // backup_unavailable
|
||||
return fmt.Errorf("the backup subsystem is not configured on felis-api (FELIS_IMAGE / FELIS_BACKUP_PVC unset)")
|
||||
case http.StatusNotFound:
|
||||
return fmt.Errorf("no such server")
|
||||
}
|
||||
if e.Error.Message != "" {
|
||||
return fmt.Errorf("felis-api: %s", e.Error.Message)
|
||||
}
|
||||
@@ -120,3 +128,19 @@ func performBackupNow(ctx context.Context, cl client.Client, controlNamespace, n
|
||||
hc := &http.Client{Timeout: 10 * time.Second}
|
||||
return requestBackup(ctx, hc, baseURL, token, name, osUser)
|
||||
}
|
||||
|
||||
// backupPickable narrows the backup picker to servers the backup API can accept.
|
||||
// System servers (login/lobby) are excluded: they have no row in the servers
|
||||
// table and carry reserved names, so every attempt dies in name validation —
|
||||
// offering them would be a dead pick. The halt picker keeps them on purpose
|
||||
// (break-glass retains full power over system servers); only the API-backed
|
||||
// backup op cannot reach them.
|
||||
func backupPickable(servers []haltableServer) []haltableServer {
|
||||
out := make([]haltableServer, 0, len(servers))
|
||||
for _, s := range servers {
|
||||
if !s.system {
|
||||
out = append(out, s)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -114,16 +114,23 @@ func TestRequestBackup(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
code int
|
||||
body string // optional JSON error body
|
||||
expect string
|
||||
}{
|
||||
{"409 not_stopped", http.StatusConflict, "must be stopped"},
|
||||
{"503 backup_unavailable", http.StatusServiceUnavailable, "not configured"},
|
||||
{"404 not found", http.StatusNotFound, "no such server"},
|
||||
{"409 not_stopped", http.StatusConflict, "", "must be stopped"},
|
||||
{"409 no_world_volume surfaces the API's own text", http.StatusConflict,
|
||||
`{"error":{"code":"no_world_volume","message":"this server has no world volume yet — start it once to create it, then retry"}}`,
|
||||
"no world volume yet"},
|
||||
{"503 backup_unavailable", http.StatusServiceUnavailable, "", "not configured"},
|
||||
{"404 not found", http.StatusNotFound, "", "no such server"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(tc.code)
|
||||
if tc.body != "" {
|
||||
_, _ = io.WriteString(w, tc.body)
|
||||
}
|
||||
}))
|
||||
defer srv.Close()
|
||||
_, err := requestBackup(context.Background(), hc, srv.URL, "tok", "survival", "alice")
|
||||
@@ -142,3 +149,25 @@ func TestRequestBackup(t *testing.T) {
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// The Sync picker must not offer system servers: the backup API validates names
|
||||
// and resolves a servers-table row, so a lobby/login pick can only die in
|
||||
// validation — a dead choice in an emergency console.
|
||||
func TestBackupPickable(t *testing.T) {
|
||||
got := backupPickable([]haltableServer{
|
||||
{name: "lobby", phase: "Running", system: true},
|
||||
{name: "login", phase: "Running", system: true},
|
||||
{name: "test-one", phase: "Stopped"},
|
||||
})
|
||||
if len(got) != 1 || got[0].name != "test-one" || got[0].system {
|
||||
t.Fatalf("backupPickable = %+v, want only the user server", got)
|
||||
}
|
||||
|
||||
// Survivors keep their input order (the picker's cursor math depends on it).
|
||||
got = backupPickable([]haltableServer{
|
||||
{name: "alpha"}, {name: "login", system: true}, {name: "beta"},
|
||||
})
|
||||
if len(got) != 2 || got[0].name != "alpha" || got[1].name != "beta" {
|
||||
t.Fatalf("backupPickable order = %+v, want [alpha beta]", got)
|
||||
}
|
||||
}
|
||||
+43
-12
@@ -76,12 +76,19 @@ type ownerStore interface {
|
||||
AdminExists(ctx context.Context) (bool, error)
|
||||
// UserByUsername loads a staff login projection.
|
||||
UserByUsername(ctx context.Context, username string) (*api.StaffUser, error)
|
||||
// OwnerUsername names the single active Owner seat, or "" when none exists.
|
||||
// provisionOwner refuses to re-target anything but this username: with the
|
||||
// seat occupied, a fresh name would mint a second owner row (UpsertOwner's
|
||||
// insert arm) while the existing — possibly compromised — seat stays live,
|
||||
// and no supported path can delete an owner row afterwards.
|
||||
OwnerUsername(ctx context.Context) (string, error)
|
||||
UpsertOwner(ctx context.Context, id, username, email string) error
|
||||
// InsertOperator mints a NEW Operator staff account. Unlike UpsertOwner it is
|
||||
// insert-only: a username already taken is a conflict (api.ErrConflict), never a
|
||||
// silent reset, so adding an Operator can never clobber the Owner or an existing
|
||||
// Operator. The row is role=admin, identical in shape to the Owner — Felis has no
|
||||
// separate operator DB role (migration 0003: staff = role=admin).
|
||||
// Operator. The row is role=admin — an Operator is staff BELOW the single
|
||||
// role=owner identity (migration 0011 adds that role); the two are the only
|
||||
// staff roles.
|
||||
InsertOperator(ctx context.Context, id, username, email string) error
|
||||
// CompleteOwnerSetup atomically consumes the in-game link code, creates or
|
||||
// promotes the bound Owner, enables local auth, and stores the one-time setup
|
||||
@@ -266,20 +273,31 @@ func authenticateAdmin(ctx context.Context, s ownerStore, username string) (matc
|
||||
if err != nil {
|
||||
return "", false, err
|
||||
}
|
||||
if u.Role != "admin" {
|
||||
// Staff means admin OR owner: recovery attribution must accept the Owner (the
|
||||
// primary break-glass identity), not just plain admins.
|
||||
if u.Role != "admin" && u.Role != "owner" {
|
||||
return "", false, nil
|
||||
}
|
||||
return u.Username, true, nil
|
||||
}
|
||||
|
||||
// provisionOwner mints or resets the single Owner account direct-to-Postgres,
|
||||
// passwordless. The account is role=admin with no password — the Owner completes
|
||||
// passwordless. The account is role=owner with no password — the Owner completes
|
||||
// passwordless login setup via the web setup-token flow after `felis setup`.
|
||||
// With a seat already occupied the reset must name that seat (ownerSeatTakenError
|
||||
// otherwise): the upsert's insert arm would silently mint a SECOND owner, and
|
||||
// every owner row is undeletable through the panel, so the tier could never
|
||||
// converge back to one.
|
||||
func provisionOwner(ctx context.Context, s ownerStore, username, email string) error {
|
||||
username = strings.TrimSpace(username)
|
||||
if username == "" {
|
||||
return errors.New("owner username is required")
|
||||
}
|
||||
if seat, err := s.OwnerUsername(ctx); err != nil {
|
||||
return fmt.Errorf("check the owner seat: %w", err)
|
||||
} else if seat != "" && seat != username {
|
||||
return &ownerSeatTakenError{seat: seat}
|
||||
}
|
||||
id := newOwnerID()
|
||||
if id == "" {
|
||||
return errors.New("generate owner id: entropy source failed")
|
||||
@@ -290,13 +308,26 @@ func provisionOwner(ctx context.Context, s ownerStore, username, email string) e
|
||||
return nil
|
||||
}
|
||||
|
||||
// provisionOperator mints a NEW Operator staff account direct-to-Postgres. Like the
|
||||
// Owner it is role=admin and passwordless — Felis has no separate operator DB role,
|
||||
// so an Operator is simply an additional staff admin (migration 0003). UNLIKE
|
||||
// provisionOwner, which upserts the single Owner and resets it on a username
|
||||
// conflict, this is insert-only: a username already taken returns api.ErrConflict
|
||||
// rather than overwriting a live account, so adding an Operator can never silently
|
||||
// clobber the Owner's or another Operator's account.
|
||||
// ownerSeatTakenError refuses an Owner reset that names anything but the
|
||||
// occupied seat, naming it so the operator can retype. Is reports
|
||||
// api.ErrConflict so the TUI's recoverable-error branch (shared with the
|
||||
// operator path's taken-name clash) routes back to the form instead of ending
|
||||
// the console.
|
||||
type ownerSeatTakenError struct{ seat string }
|
||||
|
||||
func (e *ownerSeatTakenError) Error() string {
|
||||
return fmt.Sprintf("an Owner already exists as %q — enter that username to reset the Owner", e.seat)
|
||||
}
|
||||
|
||||
func (e *ownerSeatTakenError) Is(target error) bool { return target == api.ErrConflict }
|
||||
|
||||
// provisionOperator mints a NEW Operator staff account direct-to-Postgres. It is
|
||||
// role=admin and passwordless — an additional staff admin below the single
|
||||
// role=owner identity (migrations 0003 + 0011). UNLIKE provisionOwner, which
|
||||
// upserts the single Owner and resets it on a username conflict, this is
|
||||
// insert-only: a username already taken returns api.ErrConflict rather than
|
||||
// overwriting a live account, so adding an Operator can never silently clobber
|
||||
// the Owner's or another Operator's account.
|
||||
func provisionOperator(ctx context.Context, s ownerStore, username, email string) error {
|
||||
username = strings.TrimSpace(username)
|
||||
if username == "" {
|
||||
@@ -387,7 +418,7 @@ func newSetupToken() (raw, hash string, err error) {
|
||||
|
||||
// performSetupMCBind is the `felis setup` Owner-establishment path: the operator
|
||||
// binds their Minecraft account via a one-time link code the login gate handed
|
||||
// them in-game, the bound user is promoted to role='admin' (passwordless Owner),
|
||||
// them in-game, the bound user is promoted to role='owner' (passwordless Owner),
|
||||
// local auth is enabled, and a one-time setup URL is minted for the first web
|
||||
// login where the Owner verifies email / enrolls a passkey. adminHostname is the
|
||||
// operator-console host the URL points at (op.console.<root>): the Owner is staff,
|
||||
|
||||
@@ -19,14 +19,15 @@ import (
|
||||
// terminal. The design is passwordless: accounts carry no credential, and the
|
||||
// Owner completes first-login through the setup-token web flow.
|
||||
type fakeOwnerStore struct {
|
||||
upserts []upsertCall
|
||||
inserts []upsertCall
|
||||
settings map[string][]byte
|
||||
audits []api.AuditEntry
|
||||
tokens []setupTokenCall
|
||||
redeems []redeemCall
|
||||
users map[string]*api.StaffUser // keyed by username
|
||||
admins bool // AdminExists answer
|
||||
upserts []upsertCall
|
||||
inserts []upsertCall
|
||||
settings map[string][]byte
|
||||
audits []api.AuditEntry
|
||||
tokens []setupTokenCall
|
||||
redeems []redeemCall
|
||||
users map[string]*api.StaffUser // keyed by username
|
||||
admins bool // AdminExists answer
|
||||
ownerSeat string // OwnerUsername answer: the occupied seat, "" when none
|
||||
|
||||
// CompleteOwnerSetup's success result. redeemUserID defaults to the fresh id
|
||||
// the caller passes (the unlinked-UUID case) when left empty.
|
||||
@@ -40,6 +41,7 @@ type fakeOwnerStore struct {
|
||||
auditErr error
|
||||
userErr error // non-not-found error from UserByUsername
|
||||
adminErr error
|
||||
seatErr error
|
||||
redeemErr error
|
||||
createTokenErr error
|
||||
}
|
||||
@@ -80,6 +82,15 @@ func (f *fakeOwnerStore) UserByUsername(_ context.Context, username string) (*ap
|
||||
return nil, api.ErrNotFound
|
||||
}
|
||||
|
||||
// OwnerUsername reports the single active Owner seat. Tests set ownerSeat; the
|
||||
// zero value models a fresh install where bootstrap is free to mint.
|
||||
func (f *fakeOwnerStore) OwnerUsername(_ context.Context) (string, error) {
|
||||
if f.seatErr != nil {
|
||||
return "", f.seatErr
|
||||
}
|
||||
return f.ownerSeat, nil
|
||||
}
|
||||
|
||||
func (f *fakeOwnerStore) UpsertOwner(_ context.Context, id, username, email string) error {
|
||||
if f.upsertErr != nil {
|
||||
return f.upsertErr
|
||||
@@ -201,6 +212,34 @@ func TestProvisionOwner(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("an occupied seat refuses any other username", func(t *testing.T) {
|
||||
// The seat is the single owner row: upserting a fresh name would take the
|
||||
// insert arm and mint a SECOND owner, while the existing seat — possibly the
|
||||
// compromised account this reset was meant to replace — stays live, and no
|
||||
// supported path can delete an owner row.
|
||||
f := &fakeOwnerStore{ownerSeat: "seat-holder"}
|
||||
err := provisionOwner(ctx, f, "someone-else", "")
|
||||
if !errors.Is(err, api.ErrConflict) {
|
||||
t.Fatalf("error = %v, want it to wrap api.ErrConflict so the TUI routes back to the form", err)
|
||||
}
|
||||
if !strings.Contains(err.Error(), `"seat-holder"`) {
|
||||
t.Errorf("error = %q, want it to name the occupied seat", err)
|
||||
}
|
||||
if len(f.upserts) != 0 {
|
||||
t.Errorf("want no write against an occupied seat, got %d", len(f.upserts))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("the occupied seat's own username still resets", func(t *testing.T) {
|
||||
f := &fakeOwnerStore{ownerSeat: "seat-holder"}
|
||||
if err := provisionOwner(ctx, f, "seat-holder", "[email protected]"); err != nil {
|
||||
t.Fatalf("provisionOwner(reset): %v", err)
|
||||
}
|
||||
if len(f.upserts) != 1 || f.upserts[0].username != "seat-holder" || f.upserts[0].email != "[email protected]" {
|
||||
t.Fatalf("want 1 reset upsert for the seat, got %+v", f.upserts)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("propagates a store error", func(t *testing.T) {
|
||||
f := &fakeOwnerStore{upsertErr: errors.New("boom")}
|
||||
if err := provisionOwner(ctx, f, "owner", ""); err == nil {
|
||||
@@ -264,6 +303,16 @@ func TestAuthenticateAdmin(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("the owner role attributes like an admin", func(t *testing.T) {
|
||||
owner := mkAdmin("root")
|
||||
owner.Role = "owner" // the platform owner is staff too (migration 0011)
|
||||
f := &fakeOwnerStore{users: map[string]*api.StaffUser{"root": owner}}
|
||||
matched, ok, err := authenticateAdmin(ctx, f, "root")
|
||||
if err != nil || !ok || matched != "root" {
|
||||
t.Fatalf("authenticateAdmin(owner) = (%q, %v, %v), want (root, true, nil)", matched, ok, err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("an unknown user is a non-match, not an error", func(t *testing.T) {
|
||||
f := &fakeOwnerStore{}
|
||||
_, ok, err := authenticateAdmin(ctx, f, "nobody")
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
)
|
||||
|
||||
// cmdConverge is the explicit convergence pass over already-installed system
|
||||
// servers (#1). Provisioning is create-if-absent, so a field the desired spec
|
||||
// gained after an install (spec.rcon, spec.startup.healthHTTPPort, a derived env
|
||||
// key) never reaches the existing CR — and nothing says so. This command fills
|
||||
// exactly those zero-value fields; see convergeSystemServers for the full contract
|
||||
// and why it is a separate, operator-timed step rather than part of setup.
|
||||
//
|
||||
// It reads the same host config as setup (the control plane's felis.toml) and
|
||||
// talks to the cluster with the local kubeconfig, so it must run as root on the
|
||||
// control-plane host.
|
||||
func cmdConverge(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("converge", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
cfgPath := fs.String("config", defaultSetupConfigPath, "path to felis.toml")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
if errors.Is(err, flag.ErrHelp) {
|
||||
return 0
|
||||
}
|
||||
return 2
|
||||
}
|
||||
if os.Geteuid() != 0 {
|
||||
fmt.Fprintln(stderr, "felis converge: refused — converging needs the cluster credentials, so it must run as root (try: sudo felis converge)")
|
||||
return 1
|
||||
}
|
||||
|
||||
cfg, err := config.Load(*cfgPath)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis converge: %v\n", err)
|
||||
fmt.Fprintln(stderr, "If this host was never installed, run `sudo felis setup` first.")
|
||||
return 1
|
||||
}
|
||||
cl, err := buildSystemServerClient()
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis converge: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
controlNS := platform.DefaultControlNamespace
|
||||
outcomes := convergeSystemServers(context.Background(), cl, cfg.K8s.Namespace,
|
||||
cfg.Velocity.LoginImage, cfg.Velocity.LobbyImage,
|
||||
platform.InternalAPIBaseURL(controlNS), cfg.Server.RootDomain,
|
||||
defaultPanelHostname(cfg.Server.RootDomain, cfg.Auth.PanelHostname))
|
||||
|
||||
fmt.Fprintln(stdout, "felis converge: filling fields an installed system server predates (operator-set values are never overwritten):")
|
||||
exit := 0
|
||||
for _, o := range outcomes {
|
||||
switch {
|
||||
case o.err != nil:
|
||||
fmt.Fprintf(stdout, " - %s: ERROR %v\n", o.name, o.err)
|
||||
exit = 1
|
||||
case len(o.changes) > 0:
|
||||
fmt.Fprintf(stdout, " - %s: updated (%s)\n", o.name, strings.Join(o.changes, ", "))
|
||||
default:
|
||||
fmt.Fprintf(stdout, " - %s: %s\n", o.name, o.skipped)
|
||||
}
|
||||
}
|
||||
return exit
|
||||
}
|
||||
@@ -0,0 +1,185 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
)
|
||||
|
||||
// converge is the explicit pass over an installed system server whose CR predates
|
||||
// a field the desired spec has since gained (#1). It must fill exactly the
|
||||
// zero-valued whitelist fields and the derived env, and must not touch anything a
|
||||
// non-zero value already occupies — that is the operator's.
|
||||
func TestConvergeSystemServersFillsPredatedFields(t *testing.T) {
|
||||
scheme := newSystemServerScheme(t)
|
||||
ctx := context.Background()
|
||||
|
||||
// An old install: the lobby CR was created before the desired spec began
|
||||
// rendering spec.rcon, and the login CR before the HTTP readiness gate existed.
|
||||
// One derived env key is absent entirely (as if it were added later), and one
|
||||
// hand-added env var plus a non-whitelisted spec field must survive.
|
||||
lobby, err := lobbySystemServer("reg/lobby:1", "minecraft")
|
||||
if err != nil {
|
||||
t.Fatalf("build lobby: %v", err)
|
||||
}
|
||||
lobby.Spec.Rcon = v1alpha1.RconSpec{}
|
||||
lobby.Spec.JavaMemory = "999Mi"
|
||||
|
||||
login, err := loginSystemServer("reg/limbo:1", "minecraft",
|
||||
"http://felis-api.felis.svc.cluster.local:8081", "mc.example.net", "console.mc.example.net")
|
||||
if err != nil {
|
||||
t.Fatalf("build login: %v", err)
|
||||
}
|
||||
login.Spec.Startup.HealthHTTPPort = 0
|
||||
kept := login.Spec.Env
|
||||
login.Spec.Env = nil
|
||||
for _, e := range kept {
|
||||
if e.Name != envPanelHostname {
|
||||
login.Spec.Env = append(login.Spec.Env, e)
|
||||
}
|
||||
}
|
||||
login.Spec.Env = append(login.Spec.Env, v1alpha1.EnvVar{Name: "OPERATOR_TUNING", Value: "keep-me"})
|
||||
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(lobby, login).Build()
|
||||
outcomes := convergeSystemServers(ctx, cl, "minecraft", "reg/limbo:1", "reg/lobby:1",
|
||||
"http://felis-api.felis.svc.cluster.local:8081", "mc.example.net", "console.mc.example.net")
|
||||
|
||||
byName := map[string]systemServerOutcome{}
|
||||
for _, o := range outcomes {
|
||||
if o.err != nil {
|
||||
t.Fatalf("%s: unexpected error: %v", o.name, o.err)
|
||||
}
|
||||
byName[o.name] = o
|
||||
}
|
||||
lobbyOut := byName[naming.SystemLobbyServer]
|
||||
if len(lobbyOut.changes) != 1 || lobbyOut.changes[0] != "spec.rcon" {
|
||||
t.Errorf("lobby changes = %v, want [spec.rcon] (only the zero-valued field)", lobbyOut.changes)
|
||||
}
|
||||
loginOut := byName[naming.SystemLoginServer]
|
||||
if !slices.Contains(loginOut.changes, "spec.startup.healthHTTPPort") || !slices.Contains(loginOut.changes, "env "+envPanelHostname) {
|
||||
t.Errorf("login changes = %v, want the health port plus the missing derived env key", loginOut.changes)
|
||||
}
|
||||
|
||||
var gotLobby v1alpha1.MinecraftServer
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLobbyServer}, &gotLobby); err != nil {
|
||||
t.Fatalf("get lobby: %v", err)
|
||||
}
|
||||
if !gotLobby.Spec.Rcon.Enabled ||
|
||||
gotLobby.Spec.Rcon.SecretRef.Name != naming.RconSecretName(naming.SystemLobbyServer) ||
|
||||
gotLobby.Spec.Rcon.SecretRef.Key != naming.RconSecretKey {
|
||||
t.Errorf("lobby rcon = %+v, want the desired block with the %s secret",
|
||||
gotLobby.Spec.Rcon, naming.RconSecretName(naming.SystemLobbyServer))
|
||||
}
|
||||
if gotLobby.Spec.JavaMemory != "999Mi" {
|
||||
t.Errorf("lobby javaMemory = %q, want 999Mi — converge fills new fields, it does not rewrite the spec", gotLobby.Spec.JavaMemory)
|
||||
}
|
||||
|
||||
var gotLogin v1alpha1.MinecraftServer
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLoginServer}, &gotLogin); err != nil {
|
||||
t.Fatalf("get login: %v", err)
|
||||
}
|
||||
if gotLogin.Spec.Startup.HealthHTTPPort != felisLimboHealthPort {
|
||||
t.Errorf("login healthHTTPPort = %d, want %d", gotLogin.Spec.Startup.HealthHTTPPort, felisLimboHealthPort)
|
||||
}
|
||||
env := map[string]string{}
|
||||
for _, e := range gotLogin.Spec.Env {
|
||||
env[e.Name] = e.Value
|
||||
}
|
||||
if env[envPanelHostname] != "console.mc.example.net" {
|
||||
t.Errorf("%s was not added back: %q", envPanelHostname, env[envPanelHostname])
|
||||
}
|
||||
if env["OPERATOR_TUNING"] != "keep-me" {
|
||||
t.Error("a hand-added env var was dropped; converge only touches config-derived names")
|
||||
}
|
||||
}
|
||||
|
||||
// A field already holding a non-zero value belongs to the operator: converge must
|
||||
// report "already converged" and write nothing.
|
||||
func TestConvergeSystemServersLeavesNonZeroFieldsAlone(t *testing.T) {
|
||||
scheme := newSystemServerScheme(t)
|
||||
ctx := context.Background()
|
||||
|
||||
lobby, err := lobbySystemServer("reg/lobby:1", "minecraft")
|
||||
if err != nil {
|
||||
t.Fatalf("build lobby: %v", err)
|
||||
}
|
||||
lobby.Spec.Rcon = v1alpha1.RconSpec{
|
||||
Enabled: true,
|
||||
SecretRef: v1alpha1.SecretKeyRef{Name: "operator-rotated", Key: "password"},
|
||||
}
|
||||
login, err := loginSystemServer("reg/limbo:1", "minecraft",
|
||||
"http://felis-api.felis.svc.cluster.local:8081", "mc.example.net", "console.mc.example.net")
|
||||
if err != nil {
|
||||
t.Fatalf("build login: %v", err)
|
||||
}
|
||||
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(lobby, login).Build()
|
||||
for _, o := range convergeSystemServers(ctx, cl, "minecraft", "reg/limbo:1", "reg/lobby:1",
|
||||
"http://felis-api.felis.svc.cluster.local:8081", "mc.example.net", "console.mc.example.net") {
|
||||
if o.err != nil {
|
||||
t.Fatalf("%s: unexpected error: %v", o.name, o.err)
|
||||
}
|
||||
if len(o.changes) != 0 || o.skipped != "already converged" {
|
||||
t.Errorf("%s outcome = %+v, want already converged with no writes", o.name, o)
|
||||
}
|
||||
}
|
||||
var got v1alpha1.MinecraftServer
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: naming.SystemLobbyServer}, &got); err != nil {
|
||||
t.Fatalf("get lobby: %v", err)
|
||||
}
|
||||
if got.Spec.Rcon.SecretRef.Name != "operator-rotated" {
|
||||
t.Errorf("lobby rcon secretRef = %q — converge overwrote a field the operator had already set",
|
||||
got.Spec.Rcon.SecretRef.Name)
|
||||
}
|
||||
}
|
||||
|
||||
// Guards: an absent CR is reported (creation is setup's job), a foreign CR is
|
||||
// refused rather than adopted, and an unset image skips like the provisioner does.
|
||||
func TestConvergeSystemServersGuards(t *testing.T) {
|
||||
scheme := newSystemServerScheme(t)
|
||||
ctx := context.Background()
|
||||
run := func(cl client.Client, loginImage, lobbyImage string) []systemServerOutcome {
|
||||
return convergeSystemServers(ctx, cl, "minecraft", loginImage, lobbyImage,
|
||||
"http://felis-api.felis.svc.cluster.local:8081", "mc.example.net", "console.mc.example.net")
|
||||
}
|
||||
|
||||
t.Run("absent CRs are reported, not created", func(t *testing.T) {
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).Build()
|
||||
for _, o := range run(cl, "reg/limbo:1", "reg/lobby:1") {
|
||||
if o.err != nil {
|
||||
t.Fatalf("%s: %v", o.name, o.err)
|
||||
}
|
||||
if o.created || !strings.Contains(o.skipped, "not present") {
|
||||
t.Errorf("%s outcome = %+v, want a not-present skip", o.name, o)
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("foreign CR is refused", func(t *testing.T) {
|
||||
foreign := &v1alpha1.MinecraftServer{}
|
||||
foreign.Name = naming.SystemLoginServer
|
||||
foreign.Namespace = "minecraft"
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(foreign).Build()
|
||||
out := run(cl, "reg/limbo:1", "")
|
||||
if len(out) != 2 {
|
||||
t.Fatalf("outcomes = %d, want 2", len(out))
|
||||
}
|
||||
if out[0].err == nil || !strings.Contains(out[0].err.Error(), "not marked") {
|
||||
t.Fatalf("login error = %v, want an unmarked-name refusal", out[0].err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("unset image skips", func(t *testing.T) {
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).Build()
|
||||
out := run(cl, "", "reg/lobby:1")
|
||||
if out[0].skipped != "image not configured" {
|
||||
t.Errorf("login skipped = %q, want %q", out[0].skipped, "image not configured")
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,202 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"archive/tar"
|
||||
"compress/gzip"
|
||||
"context"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/signal"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
// cmdFetchContext is the in-Pod entrypoint the build Job's context-fetch
|
||||
// initContainer runs. It reads the blob the platform stored for a submission
|
||||
// from the felis-api INTERNAL face (with a bounded retry — see
|
||||
// fetchContextWithRetry) and extracts it into the shared emptyDir the Kaniko
|
||||
// container then builds from.
|
||||
//
|
||||
// Why this exists: the build Pod runs in the build namespace, where it can neither
|
||||
// mount the control-plane uploads PVC (a PVC does not cross namespaces) nor hold
|
||||
// object-store credentials, so the API that WROTE the blob is the transport. The
|
||||
// route is service-token-gated; the token arrives through a namespace-local Secret
|
||||
// mounted only into this initContainer, never into Kaniko's — so the untrusted
|
||||
// Dockerfile's build steps have no credential to read (their containers share no
|
||||
// environment, no PID namespace, and Kaniko itself mounts the context read-only).
|
||||
//
|
||||
// The extraction is deliberately paranoid: the tarball is attacker-controlled
|
||||
// input, so absolute paths, ".." escapes, links, and special files are refused
|
||||
// rather than sanitized. Kaniko treats the extracted tree as hostile regardless
|
||||
// (spec §16), but the pod's own filesystem still must not be written outside the
|
||||
// context directory it was given.
|
||||
func cmdFetchContext(args []string, _, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("fetch-context", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
url := fs.String("url", "", "internal-face URL of the submission's build-context tarball")
|
||||
out := fs.String("out", "/context", "directory to extract the build context into")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
if *url == "" {
|
||||
fmt.Fprintln(stderr, "felis fetch-context: --url is required")
|
||||
return 2
|
||||
}
|
||||
token := os.Getenv("FELIS_SERVICE_TOKEN")
|
||||
if token == "" {
|
||||
fmt.Fprintln(stderr, "felis fetch-context: FELIS_SERVICE_TOKEN is empty — the internal face rejects anonymous reads")
|
||||
return 2
|
||||
}
|
||||
|
||||
ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
|
||||
defer stop()
|
||||
|
||||
// Validate the URL once up front: a bad one is a usage error (2), not
|
||||
// something to sit in the retry loop.
|
||||
if _, err := http.NewRequest(http.MethodGet, *url, nil); err != nil {
|
||||
fmt.Fprintf(stderr, "felis fetch-context: bad --url: %v\n", err)
|
||||
return 2
|
||||
}
|
||||
// No overall client timeout: a legitimate modpack context can be large and the
|
||||
// Job's activeDeadlineSeconds is the real bound. The header timeout catches a
|
||||
// wedged endpoint without capping a healthy download.
|
||||
client := &http.Client{Transport: &http.Transport{ResponseHeaderTimeout: time.Minute}}
|
||||
resp, err := fetchContextWithRetry(ctx, client, *url, token, stderr)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis fetch-context: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if err := extractTarGz(resp.Body, *out); err != nil {
|
||||
fmt.Fprintf(stderr, "felis fetch-context: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
return 0
|
||||
}
|
||||
|
||||
// fetchRetryInterval/fetchRetryWindow bound how long the fetch waits out a
|
||||
// control-plane blip before giving up. The api pod being replaced is a normal
|
||||
// event (rollout, eviction, a chaos drill), and without a retry one refused
|
||||
// dial turns it into a failed build: BackoffLimit=0 gives the Job no second
|
||||
// Pod, so the terminal verdict costs a manual re-approval — the live drill hit
|
||||
// exactly this (context-fetch exit 1 on `connect: connection refused` while
|
||||
// the api pod rolled; the new pod was serving 11 seconds later and the same
|
||||
// 198-byte blob). The window is tiny next to the Job's 30-minute
|
||||
// activeDeadline; a 4xx (missing blob, rejected token) still fails fast.
|
||||
//
|
||||
// Vars, not consts, so tests can shrink the window.
|
||||
var (
|
||||
fetchRetryInterval = 3 * time.Second
|
||||
fetchRetryWindow = 45 * time.Second
|
||||
)
|
||||
|
||||
// fetchContextWithRetry GETs the context tarball, retrying transport failures
|
||||
// and 5xx responses until fetchRetryWindow runs out. A 4xx is an answer, not a
|
||||
// blip — retrying it only delays the honest error.
|
||||
func fetchContextWithRetry(ctx context.Context, client *http.Client, url, token string, stderr io.Writer) (*http.Response, error) {
|
||||
deadline := time.Now().Add(fetchRetryWindow)
|
||||
for {
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("bad --url: %w", err)
|
||||
}
|
||||
req.Header.Set("Authorization", "Bearer "+token)
|
||||
|
||||
resp, err := client.Do(req)
|
||||
if err == nil && resp.StatusCode == http.StatusOK {
|
||||
return resp, nil
|
||||
}
|
||||
if err == nil {
|
||||
status := resp.Status
|
||||
_ = resp.Body.Close()
|
||||
err = fmt.Errorf("GET returned %s", status)
|
||||
if resp.StatusCode < 500 {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
if ctx.Err() != nil {
|
||||
return nil, fmt.Errorf("GET failed: %w", err)
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
return nil, fmt.Errorf("GET failed (retried for %s): %w", fetchRetryWindow, err)
|
||||
}
|
||||
fmt.Fprintf(stderr, "felis fetch-context: %v; retrying (the internal face may be restarting)\n", err)
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return nil, fmt.Errorf("GET failed: %w", err)
|
||||
case <-time.After(fetchRetryInterval):
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// extractTarGz streams a gzip'd tarball into root, creating directories as
|
||||
// needed. Every entry is vetted BEFORE anything is written: a path that is
|
||||
// absolute or escapes root (via ".."), a link (symlink or hardlink), or any
|
||||
// special file kind aborts the whole extraction. Refusing rather than skipping is
|
||||
// deliberate — a context that needs one of those constructs is not a context this
|
||||
// transport carries, and silently dropping entries would build from a corpus the
|
||||
// submitter did not upload.
|
||||
func extractTarGz(r io.Reader, root string) error {
|
||||
if err := os.MkdirAll(root, 0o755); err != nil {
|
||||
return fmt.Errorf("create context dir: %w", err)
|
||||
}
|
||||
zr, err := gzip.NewReader(r)
|
||||
if err != nil {
|
||||
return fmt.Errorf("context is not a valid gzip tarball: %w", err)
|
||||
}
|
||||
defer zr.Close()
|
||||
tr := tar.NewReader(zr)
|
||||
for {
|
||||
hdr, err := tr.Next()
|
||||
if errors.Is(err, io.EOF) {
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return fmt.Errorf("read context tarball: %w", err)
|
||||
}
|
||||
name := filepath.Clean(hdr.Name)
|
||||
if name == "." {
|
||||
continue
|
||||
}
|
||||
// The zip-slip guard: reject, never rewrite. filepath.Clean collapses any
|
||||
// "a/../../b", so these two checks are sufficient once Clean has run.
|
||||
if filepath.IsAbs(name) || name == ".." || strings.HasPrefix(name, ".."+string(filepath.Separator)) {
|
||||
return fmt.Errorf("context entry %q escapes the context directory", hdr.Name)
|
||||
}
|
||||
target := filepath.Join(root, name)
|
||||
switch hdr.Typeflag {
|
||||
case tar.TypeDir:
|
||||
if err := os.MkdirAll(target, 0o755); err != nil {
|
||||
return fmt.Errorf("create %q: %w", name, err)
|
||||
}
|
||||
case tar.TypeReg, tar.TypeRegA:
|
||||
if err := os.MkdirAll(filepath.Dir(target), 0o755); err != nil {
|
||||
return fmt.Errorf("create parent of %q: %w", name, err)
|
||||
}
|
||||
mode := os.FileMode(0o644)
|
||||
if hdr.FileInfo().Mode()&0o111 != 0 {
|
||||
mode = 0o755 // preserve executability (entrypoint scripts), nothing else
|
||||
}
|
||||
f, err := os.OpenFile(target, os.O_CREATE|os.O_WRONLY|os.O_TRUNC, mode)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create %q: %w", name, err)
|
||||
}
|
||||
if _, err := io.Copy(f, tr); err != nil {
|
||||
_ = f.Close()
|
||||
return fmt.Errorf("write %q: %w", name, err)
|
||||
}
|
||||
if err := f.Close(); err != nil {
|
||||
return fmt.Errorf("close %q: %w", name, err)
|
||||
}
|
||||
default:
|
||||
return fmt.Errorf("context entry %q has unsupported type %q (links and special files are refused)", hdr.Name, string(hdr.Typeflag))
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,315 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"archive/tar"
|
||||
"bytes"
|
||||
"compress/gzip"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
type tarEntry struct {
|
||||
name string
|
||||
body string
|
||||
mode int64
|
||||
typ byte
|
||||
linkname string
|
||||
}
|
||||
|
||||
// tgzBody builds an in-memory .tar.gz from entries, preserving each entry's type
|
||||
// and mode so the tests can exercise the guards with exactly the bytes an
|
||||
// attacker could upload.
|
||||
func tgzBody(t *testing.T, entries ...tarEntry) []byte {
|
||||
t.Helper()
|
||||
var buf bytes.Buffer
|
||||
zw := gzip.NewWriter(&buf)
|
||||
tw := tar.NewWriter(zw)
|
||||
for _, e := range entries {
|
||||
typ := e.typ
|
||||
if typ == 0 {
|
||||
typ = tar.TypeReg
|
||||
}
|
||||
mode := e.mode
|
||||
if mode == 0 {
|
||||
mode = 0o644
|
||||
}
|
||||
hdr := &tar.Header{Name: e.name, Typeflag: typ, Mode: mode, Size: int64(len(e.body))}
|
||||
if typ == tar.TypeSymlink {
|
||||
hdr.Linkname = e.linkname
|
||||
hdr.Size = 0
|
||||
}
|
||||
if err := tw.WriteHeader(hdr); err != nil {
|
||||
t.Fatalf("write header %q: %v", e.name, err)
|
||||
}
|
||||
if hdr.Size > 0 {
|
||||
if _, err := tw.Write([]byte(e.body)); err != nil {
|
||||
t.Fatalf("write body %q: %v", e.name, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
if err := tw.Close(); err != nil {
|
||||
t.Fatalf("close tar: %v", err)
|
||||
}
|
||||
if err := zw.Close(); err != nil {
|
||||
t.Fatalf("close gzip: %v", err)
|
||||
}
|
||||
return buf.Bytes()
|
||||
}
|
||||
|
||||
// A normal context extracts with its tree intact, and the executable bit that
|
||||
// modpack entrypoints rely on survives.
|
||||
func TestExtractTarGzRoundTrip(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
body := tgzBody(t,
|
||||
tarEntry{name: "Dockerfile", body: "FROM scratch\n"},
|
||||
tarEntry{name: "mods/example.jar", body: "jar-bytes"},
|
||||
tarEntry{name: "start.sh", body: "#!/bin/sh\n", mode: 0o755},
|
||||
tarEntry{name: "mods/", typ: tar.TypeDir, mode: 0o755},
|
||||
)
|
||||
if err := extractTarGz(bytes.NewReader(body), dir); err != nil {
|
||||
t.Fatalf("extract: %v", err)
|
||||
}
|
||||
for name, want := range map[string]string{
|
||||
"Dockerfile": "FROM scratch\n",
|
||||
"mods/example.jar": "jar-bytes",
|
||||
} {
|
||||
got, err := os.ReadFile(filepath.Join(dir, name))
|
||||
if err != nil || string(got) != want {
|
||||
t.Fatalf("%s = (%q, %v), want %q", name, got, err, want)
|
||||
}
|
||||
}
|
||||
fi, err := os.Stat(filepath.Join(dir, "start.sh"))
|
||||
if err != nil || fi.Mode()&0o111 == 0 {
|
||||
t.Fatalf("entrypoint script lost its exec bit: %v (%v)", fi, err)
|
||||
}
|
||||
}
|
||||
|
||||
// The guards: "..", absolute paths, symlinks, and special files are refused whole
|
||||
// — nothing escapes, and nothing is silently skipped.
|
||||
func TestExtractTarGzRefusesEscapes(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
entries []tarEntry
|
||||
}{
|
||||
{"dotdot", []tarEntry{{name: "../outside", body: "x"}}},
|
||||
{"nested dotdot", []tarEntry{{name: "a/../../outside", body: "x"}}},
|
||||
{"absolute", []tarEntry{{name: "/etc/outside", body: "x"}}},
|
||||
{"symlink", []tarEntry{{name: "link", typ: tar.TypeSymlink, linkname: "/etc"}}},
|
||||
{"hardlink", []tarEntry{{name: "hard", typ: tar.TypeLink, linkname: "somewhere"}}},
|
||||
{"device", []tarEntry{{name: "dev", typ: tar.TypeChar}}},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
if err := extractTarGz(bytes.NewReader(tgzBody(t, tc.entries...)), dir); err == nil {
|
||||
t.Fatal("extract accepted a hostile entry, want an error")
|
||||
}
|
||||
// Nothing may have been written outside the target (or at all).
|
||||
entries, _ := os.ReadDir(dir)
|
||||
if len(entries) != 0 {
|
||||
t.Fatalf("hostile archive left %d entries behind", len(entries))
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// The command end to end: it dials the URL with the bearer token from the
|
||||
// environment, and refuses to run without it (the internal face would 401
|
||||
// anyway; failing at parse time is the honest earlier error).
|
||||
func TestCmdFetchContextFetchAndExtract(t *testing.T) {
|
||||
body := tgzBody(t, tarEntry{name: "Dockerfile", body: "FROM scratch\n"})
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Header.Get("Authorization") != "Bearer test-token" {
|
||||
w.WriteHeader(http.StatusUnauthorized)
|
||||
return
|
||||
}
|
||||
w.Header().Set("Content-Type", "application/gzip")
|
||||
_, _ = w.Write(body)
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
dir := t.TempDir()
|
||||
t.Setenv("FELIS_SERVICE_TOKEN", "test-token")
|
||||
if code := cmdFetchContext([]string{"--url=" + srv.URL + "/sub-1/context", "--out=" + dir}, io.Discard, io.Discard); code != 0 {
|
||||
t.Fatalf("cmdFetchContext exit = %d, want 0", code)
|
||||
}
|
||||
if got, err := os.ReadFile(filepath.Join(dir, "Dockerfile")); err != nil || string(got) != "FROM scratch\n" {
|
||||
t.Fatalf("extracted Dockerfile = (%q, %v)", got, err)
|
||||
}
|
||||
|
||||
// No token: refuse before dialing.
|
||||
t.Setenv("FELIS_SERVICE_TOKEN", "")
|
||||
var stderr bytes.Buffer
|
||||
if code := cmdFetchContext([]string{"--url=" + srv.URL + "/sub-1/context", "--out=" + t.TempDir()}, io.Discard, &stderr); code != 2 {
|
||||
t.Fatalf("missing token exit = %d, want 2 (stderr %q)", code, stderr.String())
|
||||
}
|
||||
|
||||
// A non-200 answer (e.g. the route's 404 for a never-uploaded context) fails.
|
||||
srv404 := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
w.WriteHeader(http.StatusNotFound)
|
||||
}))
|
||||
defer srv404.Close()
|
||||
t.Setenv("FELIS_SERVICE_TOKEN", "test-token")
|
||||
if code := cmdFetchContext([]string{"--url=" + srv404.URL + "/sub-1/context", "--out=" + t.TempDir()}, io.Discard, io.Discard); code != 1 {
|
||||
t.Fatalf("404 exit = %d, want 1", code)
|
||||
}
|
||||
}
|
||||
|
||||
// A body that is not a gzip tarball must fail the extraction rather than produce
|
||||
// an empty (or partial) context Kaniko would then try to build.
|
||||
func TestExtractTarGzRejectsNonGzip(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
err := extractTarGz(strings.NewReader("not a tarball"), dir)
|
||||
if err == nil || !strings.Contains(err.Error(), "gzip") {
|
||||
t.Fatalf("err = %v, want a gzip complaint", err)
|
||||
}
|
||||
}
|
||||
|
||||
// shrinkFetchWindow swaps the retry knobs for a faster test and restores them
|
||||
// afterwards, so no test leaks a tiny window into another.
|
||||
func shrinkFetchWindow(t *testing.T, interval, window time.Duration) {
|
||||
t.Helper()
|
||||
oldInterval, oldWindow := fetchRetryInterval, fetchRetryWindow
|
||||
fetchRetryInterval, fetchRetryWindow = interval, window
|
||||
t.Cleanup(func() { fetchRetryInterval, fetchRetryWindow = oldInterval, oldWindow })
|
||||
}
|
||||
|
||||
// A control-plane blip mid-fetch is survived: a 5xx on the first attempt is
|
||||
// retried and the second attempt's tarball extracts. This walks back the live
|
||||
// drill's failure, where the api pod rolled mid-fetch and the single attempt
|
||||
// died, failing the build Job.
|
||||
func TestFetchContextRetriesThroughBlip(t *testing.T) {
|
||||
shrinkFetchWindow(t, 10*time.Millisecond, time.Second)
|
||||
body := tgzBody(t, tarEntry{name: "Dockerfile", body: "FROM scratch\n"})
|
||||
var calls int32
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
if atomic.AddInt32(&calls, 1) == 1 {
|
||||
w.WriteHeader(http.StatusBadGateway) // the port is up, the API is not
|
||||
return
|
||||
}
|
||||
_, _ = w.Write(body)
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
dir := t.TempDir()
|
||||
t.Setenv("FELIS_SERVICE_TOKEN", "test-token")
|
||||
var stderr bytes.Buffer
|
||||
if code := cmdFetchContext([]string{"--url=" + srv.URL + "/sub-1/context", "--out=" + dir}, io.Discard, &stderr); code != 0 {
|
||||
t.Fatalf("exit = %d, want 0 (stderr %q)", code, stderr.String())
|
||||
}
|
||||
if got, err := os.ReadFile(filepath.Join(dir, "Dockerfile")); err != nil || string(got) != "FROM scratch\n" {
|
||||
t.Fatalf("extracted Dockerfile = (%q, %v)", got, err)
|
||||
}
|
||||
if !strings.Contains(stderr.String(), "retrying") {
|
||||
t.Fatalf("stderr %q does not mention the retry", stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
// The live drill's exact shape: the dial itself is refused (the api pod is
|
||||
// gone and no endpoint answers). A refused dial is retried like any other
|
||||
// transport failure, and once the face is back the fetch completes.
|
||||
func TestFetchContextRetriesRefusedDial(t *testing.T) {
|
||||
shrinkFetchWindow(t, 10*time.Millisecond, 5*time.Second)
|
||||
body := tgzBody(t, tarEntry{name: "Dockerfile", body: "FROM scratch\n"})
|
||||
|
||||
// Borrow a listen address, then close it: the first attempts dial into a
|
||||
// refused connection, exactly like a restarting control plane.
|
||||
probe := httptest.NewServer(http.HandlerFunc(func(http.ResponseWriter, *http.Request) {}))
|
||||
addr := strings.TrimPrefix(probe.URL, "http://")
|
||||
probe.Close()
|
||||
|
||||
dir := t.TempDir()
|
||||
t.Setenv("FELIS_SERVICE_TOKEN", "test-token")
|
||||
var stderr bytes.Buffer
|
||||
// Start the fetch; while the retry loop burns refused dials, bring the same
|
||||
// address back.
|
||||
result := make(chan int, 1)
|
||||
go func() {
|
||||
result <- cmdFetchContext([]string{"--url=http://" + addr + "/sub-1/context", "--out=" + dir}, io.Discard, &stderr)
|
||||
}()
|
||||
time.Sleep(100 * time.Millisecond) // let a handful of dials be refused
|
||||
ln, err := net.Listen("tcp", addr)
|
||||
if err != nil {
|
||||
t.Fatalf("rebind %s: %v", addr, err)
|
||||
}
|
||||
back := &http.Server{Handler: http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Header.Get("Authorization") != "Bearer test-token" {
|
||||
w.WriteHeader(http.StatusUnauthorized)
|
||||
return
|
||||
}
|
||||
_, _ = w.Write(body)
|
||||
})}
|
||||
defer back.Close()
|
||||
go func() { _ = back.Serve(ln) }()
|
||||
|
||||
code := <-result
|
||||
if code != 0 {
|
||||
t.Fatalf("exit = %d, want 0 (stderr %q)", code, stderr.String())
|
||||
}
|
||||
if got, err := os.ReadFile(filepath.Join(dir, "Dockerfile")); err != nil || string(got) != "FROM scratch\n" {
|
||||
t.Fatalf("extracted Dockerfile = (%q, %v)", got, err)
|
||||
}
|
||||
if !strings.Contains(stderr.String(), "retrying") {
|
||||
t.Fatalf("stderr %q does not mention the retry", stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
// A 4xx is an answer, not a blip: a missing/never-uploaded context fails
|
||||
// immediately — no retry loop burns the build's deadline on a terminal error.
|
||||
func TestFetchContextDoesNotRetry4xx(t *testing.T) {
|
||||
shrinkFetchWindow(t, 5*time.Millisecond, 200*time.Millisecond)
|
||||
var calls int32
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
atomic.AddInt32(&calls, 1)
|
||||
w.WriteHeader(http.StatusNotFound)
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
t.Setenv("FELIS_SERVICE_TOKEN", "test-token")
|
||||
var stderr bytes.Buffer
|
||||
if code := cmdFetchContext([]string{"--url=" + srv.URL + "/sub-1/context", "--out=" + t.TempDir()}, io.Discard, &stderr); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1 (stderr %q)", code, stderr.String())
|
||||
}
|
||||
if got := atomic.LoadInt32(&calls); got != 1 {
|
||||
t.Fatalf("server saw %d attempts, want exactly 1", got)
|
||||
}
|
||||
if strings.Contains(stderr.String(), "retrying") {
|
||||
t.Fatalf("stderr %q mentions a retry for a terminal 4xx", stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
// The retry is bounded: an internal face that stays down does not hang the
|
||||
// build pod; the window runs out and the fetch reports the exhausted retries.
|
||||
func TestFetchContextGivesUpAfterWindow(t *testing.T) {
|
||||
shrinkFetchWindow(t, 5*time.Millisecond, 60*time.Millisecond)
|
||||
var calls int32
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
atomic.AddInt32(&calls, 1)
|
||||
w.WriteHeader(http.StatusServiceUnavailable)
|
||||
}))
|
||||
defer srv.Close() // the face is up but never healthy: 503 forever
|
||||
|
||||
t.Setenv("FELIS_SERVICE_TOKEN", "test-token")
|
||||
var stderr bytes.Buffer
|
||||
start := time.Now()
|
||||
if code := cmdFetchContext([]string{"--url=" + srv.URL + "/sub-1/context", "--out=" + t.TempDir()}, io.Discard, &stderr); code != 1 {
|
||||
t.Fatalf("exit = %d, want 1 (stderr %q)", code, stderr.String())
|
||||
}
|
||||
if elapsed := time.Since(start); elapsed > 5*time.Second {
|
||||
t.Fatalf("gave up after %v; the window is supposed to bound it", elapsed)
|
||||
}
|
||||
if got := atomic.LoadInt32(&calls); got < 2 {
|
||||
t.Fatalf("server saw %d attempts, want at least one retry", got)
|
||||
}
|
||||
if !strings.Contains(stderr.String(), "retried for") {
|
||||
t.Fatalf("stderr %q does not report the exhausted retry window", stderr.String())
|
||||
}
|
||||
}
|
||||
+45
-24
@@ -26,8 +26,10 @@ func (m *multiFlag) Set(v string) error {
|
||||
// felis-reaper identity only when the retention reaper is enabled, gated with
|
||||
// its CronJob), the weak build/restore Job SAs, the build/minecraft
|
||||
// NetworkPolicies, and the running control-plane workloads (felis-api/operator
|
||||
// Deployments + the in-cluster registry Deployment/Service/PVC) — as a single
|
||||
// multi-document YAML stream on stdout, ready for `kubectl apply -f -`.
|
||||
// Deployments + the in-cluster registry Deployment/Service/PVC + the
|
||||
// world-archive PVC that backs backup/restore, unless --backup-pvc is emptied)
|
||||
// — as a single multi-document YAML stream on stdout, ready for
|
||||
// `kubectl apply -f -`.
|
||||
//
|
||||
// It is a pure renderer: it never contacts a cluster and holds no credentials.
|
||||
// --velocity-cidr records the proxy host addresses allowed by the game NetworkPolicy.
|
||||
@@ -44,9 +46,10 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
panelNodePort := fs.Int("panel-node-port", int(platform.DefaultPanelNodePort), "NodePort that exposes the built-in HTTPS panel/API origin")
|
||||
felisImage := fs.String("felis-image", "", "container image the felis-api/operator Deployments run, also passed through as FELIS_IMAGE (REQUIRED)")
|
||||
registryImage := fs.String("registry-image", "", "in-cluster registry image (default: registry:2)")
|
||||
backupPVC := fs.String("backup-pvc", "", "name of the backup PVC advertised to the restore executor via FELIS_BACKUP_PVC (default none = restore endpoint returns 503)")
|
||||
worldsHostPath := fs.String("worlds-host-path", "", "node directory under which each world PVC is visible as <path>/<pvc>; enables the reaper CronJob (requires --backup-pvc and --archive-local-path)")
|
||||
backupPVC := fs.String("backup-pvc", "felis-backups", "name of the world-archive PVC this bundle renders in the Minecraft namespace and advertises to the backup/restore executors via FELIS_BACKUP_PVC (default: felis-backups; pass an empty value to render none, leaving backup/restore answering 503)")
|
||||
worldsHostPath := fs.String("worlds-host-path", "", "node directory the reaper reads worlds from: each world PVC resolves as <path>/<pvc>, or as the stock local-path directory <path>/<pv-name>_<ns>_<pvc-name> (k3s storage root: /var/lib/rancher/k3s/storage); enables the reaper CronJob (requires --archive-local-path and a non-empty --backup-pvc)")
|
||||
archiveLocalPath := fs.String("archive-local-path", "", "path the backup PVC is mounted at in the reaper CronJob; MUST equal felis.toml [archive] local_path")
|
||||
reaperNode := fs.String("reaper-node", "", "node that holds --worlds-host-path: pins the reaper CronJob's pod there via nodeSelector kubernetes.io/hostname (multi-node clusters need this, or the reaper may schedule where the hostPath is empty)")
|
||||
var velocityCIDRs multiFlag
|
||||
fs.Var(&velocityCIDRs, "velocity-cidr", "CIDR of a Velocity proxy host allowed to reach game port 25565 (repeatable, REQUIRED)")
|
||||
var packageCIDRs multiFlag
|
||||
@@ -81,36 +84,53 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
fmt.Fprintf(stderr, "felis manifests: --panel-node-port must be in Kubernetes NodePort range 30000-32767 (got %d)\n", *panelNodePort)
|
||||
return 2
|
||||
}
|
||||
// The node pin exists only for the reaper's hostPath: naming a node without the
|
||||
// worlds root would be silently dropped (no CronJob renders), so fail loud like
|
||||
// the storage-trio check below.
|
||||
if *reaperNode != "" && *worldsHostPath == "" {
|
||||
fmt.Fprintln(stderr, "felis manifests: --reaper-node requires --worlds-host-path "+
|
||||
"(it pins the reaper CronJob, which renders only with the retention storage trio)")
|
||||
return 2
|
||||
}
|
||||
|
||||
// Retention/reaper rendering is opt-in and needs all three storage coordinates
|
||||
// together: where worlds live (to read+archive them), the backup PVC (to write
|
||||
// archives into), and the path it is mounted at (which MUST equal felis.toml
|
||||
// [archive] local_path so tarLocal's absolute archive refs resolve). A partial
|
||||
// configuration is almost certainly an operator mistake, so fail loud rather than
|
||||
// silently drop retention. Asking for it without the other two is rejected; an
|
||||
// empty trio renders the bundle WITHOUT the reaper and says so.
|
||||
// Retention/reaper rendering is opt-in and needs a storage topology together:
|
||||
// where worlds live (to read+archive them), a backup PVC (to write archives
|
||||
// into — rendered from --backup-pvc), and the path it is mounted at (which MUST
|
||||
// equal felis.toml [archive] local_path so tarLocal's absolute archive refs
|
||||
// resolve). A partial configuration is almost certainly an operator mistake, so
|
||||
// fail loud rather than silently drop retention or render a reaper with nowhere
|
||||
// to write. The backup PVC itself defaults to felis-backups (it is what makes a
|
||||
// default install's backup endpoint work at all); retention additionally needs
|
||||
// --worlds-host-path.
|
||||
if *worldsHostPath != "" {
|
||||
if *backupPVC == "" || *archiveLocalPath == "" {
|
||||
fmt.Fprintln(stderr, "felis manifests: --worlds-host-path enables the reaper CronJob and requires "+
|
||||
"--backup-pvc and --archive-local-path too (--archive-local-path must equal felis.toml [archive] local_path)")
|
||||
"--archive-local-path (must equal felis.toml [archive] local_path) and a non-empty --backup-pvc "+
|
||||
"(the archive store; default felis-backups)")
|
||||
return 2
|
||||
}
|
||||
// The reaper WILL render. Two deployment preconditions this generator cannot
|
||||
// check would SILENTLY turn retention into a no-op if unmet — surface them as
|
||||
// The reaper WILL render. Two deployment facts this generator cannot check
|
||||
// would silently turn retention into a no-op if unmet — surface them as
|
||||
// loudly as the fail-closed cases above, so an operator is never left with a
|
||||
// reaper that reaps an empty directory. (Both are also in the WorldsHostPath
|
||||
// flag/field docs, but nobody deploying from stdout reads those.)
|
||||
// reaper that reaps nothing. (Both are also in the WorldsHostPath flag/field
|
||||
// docs, but nobody deploying from stdout reads those.)
|
||||
pin := "the CronJob sets NO nodeSelector: a single-node starter pins it to the worlds implicitly, but on a " +
|
||||
"multi-node cluster you MUST pass --reaper-node <name> (or add a nodeSelector) for the node holding the " +
|
||||
"worlds, or the reaper may schedule where the hostPath is empty"
|
||||
if *reaperNode != "" {
|
||||
pin = fmt.Sprintf("the CronJob is pinned to node %q via kubernetes.io/hostname — keep this pointed at the "+
|
||||
"node that actually holds the world volumes", *reaperNode)
|
||||
}
|
||||
fmt.Fprintf(stderr, "felis manifests: note: rendering the retention reaper CronJob (worlds hostPath %q). "+
|
||||
"Two preconditions are NOT verified here:\n"+
|
||||
" - each world PVC must be visible at %s/<pvc> on the node: a stock local-path-provisioner lays "+
|
||||
"volumes under PV-name paths (.../pvc-<uuid>_<ns>_<pvc>/), so unless the worlds StorageClass is "+
|
||||
"arranged to expose <path>/<pvc>, the reaper tars an empty directory;\n"+
|
||||
" - the CronJob sets NO nodeSelector: a single-node starter pins it to the worlds implicitly, but "+
|
||||
"on a multi-node cluster you MUST add a nodeSelector for the node holding the worlds, or the reaper "+
|
||||
"may schedule where the hostPath is empty.\n", *worldsHostPath, *worldsHostPath)
|
||||
"These points are NOT verified here:\n"+
|
||||
" - the node's world volumes must actually live below %s: the reaper resolves a world as "+
|
||||
"%s/<pvc>, then as the stock local-path directory <path>/<pv-name>_<ns>_<pvc-name> (what k3s "+
|
||||
"writes under /var/lib/rancher/k3s/storage). Any other provisioner needs its volumes exposed as "+
|
||||
"<path>/<pvc>, or each candidate's archive fails and the world is preserved;\n"+
|
||||
" - %s.\n", *worldsHostPath, *worldsHostPath, *worldsHostPath, pin)
|
||||
} else {
|
||||
fmt.Fprintln(stderr, "felis manifests: note: retention reaper CronJob not rendered "+
|
||||
"(pass --worlds-host-path, --backup-pvc and --archive-local-path to enable it)")
|
||||
"(pass --worlds-host-path and --archive-local-path — the archive PVC defaults to felis-backups — to enable it)")
|
||||
}
|
||||
|
||||
out, err := platform.RenderYAML(platform.Params{
|
||||
@@ -124,6 +144,7 @@ func cmdManifests(args []string, stdout, stderr io.Writer) int {
|
||||
RegistryImage: *registryImage,
|
||||
BackupPVC: *backupPVC,
|
||||
WorldsHostPath: *worldsHostPath,
|
||||
ReaperNode: *reaperNode,
|
||||
ArchiveLocalPath: *archiveLocalPath,
|
||||
VelocityCIDRs: []string(velocityCIDRs),
|
||||
PackageSourceCIDRs: []string(packageCIDRs),
|
||||
|
||||
@@ -77,6 +77,10 @@ func TestManifestsRendersBundle(t *testing.T) {
|
||||
"10.0.0.5/32",
|
||||
// The felis image flows through to the Deployments.
|
||||
"registry.felis.svc:5000/felis:v1",
|
||||
// Backup works out of the box: the archive PVC renders and the api gets
|
||||
// the env that wires the backup/restore executors to it.
|
||||
"name: felis-backups",
|
||||
"name: FELIS_BACKUP_PVC",
|
||||
} {
|
||||
if !strings.Contains(text, want) {
|
||||
t.Errorf("rendered bundle missing %q", want)
|
||||
@@ -95,15 +99,19 @@ func TestManifestsRendersBundle(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestManifestsReaperRequiresTrio proves --worlds-host-path is a fail-loud opt-in:
|
||||
// asking for the reaper without the backup PVC and its mount path (which must equal
|
||||
// [archive] local_path) is rejected rather than silently dropping retention.
|
||||
func TestManifestsReaperRequiresTrio(t *testing.T) {
|
||||
// TestManifestsReaperRequiresStorage proves --worlds-host-path is a fail-loud
|
||||
// opt-in: asking for the reaper without a writable archive store (the backup PVC,
|
||||
// which defaults to felis-backups but can be emptied) and its mount path (which
|
||||
// must equal [archive] local_path) is rejected rather than silently dropping
|
||||
// retention or deleting worlds it could not archive first.
|
||||
func TestManifestsReaperRequiresStorage(t *testing.T) {
|
||||
base := []string{"manifests", "--felis-image", "reg/felis:test", "--velocity-cidr", "10.0.0.5/32", "--worlds-host-path", "/var/lib/felis/worlds"}
|
||||
for _, extra := range [][]string{
|
||||
{}, // neither backup-pvc nor archive-local-path
|
||||
{"--backup-pvc", "felis-backups"}, // missing archive-local-path
|
||||
{"--archive-local-path", "/backups"}, // missing backup-pvc
|
||||
{}, // missing archive-local-path (backup-pvc defaults)
|
||||
{"--backup-pvc", "other"}, // still missing archive-local-path
|
||||
// A reaper with no archive store would have nowhere to write the archive
|
||||
// it must verify before deleting a world; emptying the PVC is rejected.
|
||||
{"--archive-local-path", "/backups", "--backup-pvc="},
|
||||
} {
|
||||
var out, errBuf bytes.Buffer
|
||||
code := run(append(append([]string{}, base...), extra...), &out, &errBuf)
|
||||
@@ -119,6 +127,23 @@ func TestManifestsReaperRequiresTrio(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestManifestsBackupPVCOptOut proves --backup-pvc= renders a bundle with no
|
||||
// archive store at all: no PVC and no FELIS_BACKUP_PVC env, so backup/restore
|
||||
// answer 503 instead of pointing Jobs at a claim nobody provisions.
|
||||
func TestManifestsBackupPVCOptOut(t *testing.T) {
|
||||
var out, errBuf bytes.Buffer
|
||||
code := run([]string{"manifests", "--felis-image", "reg/felis:test",
|
||||
"--velocity-cidr", "10.0.0.5/32", "--backup-pvc="}, &out, &errBuf)
|
||||
if code != 0 {
|
||||
t.Fatalf("exit code = %d, want 0; stderr=%q", code, errBuf.String())
|
||||
}
|
||||
for _, absent := range []string{"felis-backups", "FELIS_BACKUP_PVC"} {
|
||||
if strings.Contains(out.String(), absent) {
|
||||
t.Errorf("--backup-pvc= bundle must not contain %q", absent)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestManifestsRendersReaper proves the happy path with the full retention trio:
|
||||
// a batch/v1 CronJob is emitted, named felis-reaper, mounting the backup PVC at the
|
||||
// supplied archive path.
|
||||
@@ -150,9 +175,47 @@ func TestManifestsRendersReaper(t *testing.T) {
|
||||
// this generator cannot verify (else a misarranged hostPath silently no-ops
|
||||
// retention): the <path>/<pvc> arrangement-dependency and the multi-node
|
||||
// nodeSelector hazard.
|
||||
for _, want := range []string{"local-path-provisioner", "nodeSelector"} {
|
||||
for _, want := range []string{"local-path", "nodeSelector"} {
|
||||
if !strings.Contains(errBuf.String(), want) {
|
||||
t.Errorf("reaper render must warn operators about %q on stderr, got %q", want, errBuf.String())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestManifestsReaperNodePin: --reaper-node pins the rendered CronJob's pod via
|
||||
// kubernetes.io/hostname and replaces the "no nodeSelector" hazard note with the
|
||||
// pin confirmation; using it without the worlds root is a fail-loud 2.
|
||||
func TestManifestsReaperNodePin(t *testing.T) {
|
||||
var out, errBuf bytes.Buffer
|
||||
code := run([]string{
|
||||
"manifests",
|
||||
"--felis-image", "registry.felis.svc:5000/felis:v1",
|
||||
"--velocity-cidr", "10.0.0.5/32",
|
||||
"--worlds-host-path", "/var/lib/felis/worlds",
|
||||
"--archive-local-path", "/backups",
|
||||
"--reaper-node", "node-a",
|
||||
}, &out, &errBuf)
|
||||
if code != 0 {
|
||||
t.Fatalf("exit code = %d, want 0; stderr=%q", code, errBuf.String())
|
||||
}
|
||||
for _, want := range []string{
|
||||
"kubernetes.io/hostname: node-a",
|
||||
} {
|
||||
if !strings.Contains(out.String(), want) {
|
||||
t.Errorf("pinned render missing %q", want)
|
||||
}
|
||||
}
|
||||
if !strings.Contains(errBuf.String(), "node-a") {
|
||||
t.Errorf("stderr must confirm the pin, got %q", errBuf.String())
|
||||
}
|
||||
|
||||
var out2, err2 bytes.Buffer
|
||||
if code := run([]string{
|
||||
"manifests",
|
||||
"--felis-image", "registry.felis.svc:5000/felis:v1",
|
||||
"--velocity-cidr", "10.0.0.5/32",
|
||||
"--reaper-node", "node-a",
|
||||
}, &out2, &err2); code != 2 {
|
||||
t.Errorf("--reaper-node without --worlds-host-path: exit = %d, want 2", code)
|
||||
}
|
||||
}
|
||||
+31
-2
@@ -4,16 +4,19 @@ import (
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"log/slog"
|
||||
"os"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
felismetrics "felis.lolicon.best/internal/metrics"
|
||||
"felis.lolicon.best/internal/operator"
|
||||
"github.com/go-logr/logr"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
utilruntime "k8s.io/apimachinery/pkg/util/runtime"
|
||||
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
|
||||
ctrl "sigs.k8s.io/controller-runtime"
|
||||
"sigs.k8s.io/controller-runtime/pkg/cache"
|
||||
"sigs.k8s.io/controller-runtime/pkg/healthz"
|
||||
ctrlmetrics "sigs.k8s.io/controller-runtime/pkg/metrics"
|
||||
metricsserver "sigs.k8s.io/controller-runtime/pkg/metrics/server"
|
||||
)
|
||||
@@ -25,6 +28,11 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("operator", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
metricsAddr := fs.String("metrics-bind-address", ":8080", "address the metric endpoint binds to")
|
||||
// healthAddr serves the manager's health endpoints (/healthz, /readyz) that the
|
||||
// Deployment's probes dial. Without it the operator pod would carry no probe at
|
||||
// all, and a wedged manager would keep its endpoint forever. It must differ from
|
||||
// metricsAddr: the metrics server owns :8080.
|
||||
healthAddr := fs.String("health-probe-bind-address", ":8081", "address the health probe endpoint binds to")
|
||||
// namespace MUST equal the [k8s] namespace felis-api is configured with, and
|
||||
// the deployment manifests (felis manifests) render both from one value. It
|
||||
// scopes the manager's cache (informers) to a single namespace so the operator
|
||||
@@ -42,9 +50,16 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
|
||||
utilruntime.Must(clientgoscheme.AddToScheme(scheme))
|
||||
utilruntime.Must(v1alpha1.AddToScheme(scheme))
|
||||
|
||||
// controller-runtime logs through its own logr sink; without one, its first
|
||||
// reconcile prints "log.SetLogger(...) was never called" ATTACHED TO A FULL
|
||||
// GOROUTINE STACK — pure noise, not signal. Route it to slog's default handler
|
||||
// so its messages appear as ordinary stderr lines.
|
||||
ctrl.SetLogger(logr.FromSlogHandler(slog.Default().Handler()))
|
||||
|
||||
mgr, err := ctrl.NewManager(ctrl.GetConfigOrDie(), ctrl.Options{
|
||||
Scheme: scheme,
|
||||
Metrics: metricsserver.Options{BindAddress: *metricsAddr},
|
||||
Scheme: scheme,
|
||||
Metrics: metricsserver.Options{BindAddress: *metricsAddr},
|
||||
HealthProbeBindAddress: *healthAddr,
|
||||
// Scope every informer to the single watched namespace. Without this the
|
||||
// cached client (mgr.GetClient) would LIST/WATCH cluster-wide, which a
|
||||
// namespaced Role cannot grant — the operator would fail closed at runtime
|
||||
@@ -60,6 +75,20 @@ func cmdOperator(args []string, _, stderr io.Writer) int {
|
||||
}
|
||||
fmt.Fprintf(stderr, "felis operator: watching namespace %q\n", *namespace)
|
||||
|
||||
// Register the two probe endpoints. controller-runtime only mounts /healthz and
|
||||
// /readyz once at least one check is registered, so a bare listener would 404.
|
||||
// The checks are the canonical always-pass ping: the probes' contract is "the
|
||||
// manager process is up and serving", and a dependency hiccup (e.g. an API blip)
|
||||
// must not restart the operator.
|
||||
if err := mgr.AddHealthzCheck("ping", healthz.Ping); err != nil {
|
||||
fmt.Fprintf(stderr, "felis operator: register healthz check: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
if err := mgr.AddReadyzCheck("ping", healthz.Ping); err != nil {
|
||||
fmt.Fprintf(stderr, "felis operator: register readyz check: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
// Publish the named felis_* metrics (spec §23) on the endpoint the manager
|
||||
// already serves (metricsAddr). controller-runtime's metrics server exposes
|
||||
// its global Registry, so registering into it is all that is needed for
|
||||
|
||||
+135
-20
@@ -1,9 +1,13 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -12,8 +16,11 @@ import (
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/backup"
|
||||
"felis.lolicon.best/internal/config"
|
||||
"felis.lolicon.best/internal/mail"
|
||||
"felis.lolicon.best/internal/platform"
|
||||
"felis.lolicon.best/internal/reaper"
|
||||
"felis.lolicon.best/internal/store"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
"k8s.io/apimachinery/pkg/runtime"
|
||||
utilruntime "k8s.io/apimachinery/pkg/util/runtime"
|
||||
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
|
||||
@@ -30,7 +37,7 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
fs := flag.NewFlagSet("reaper", flag.ContinueOnError)
|
||||
fs.SetOutput(stderr)
|
||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||
worldsRoot := fs.String("worlds-root", "/worlds", "mount root under which world PVCs are visible (tarLocal: <root>/<pvc>)")
|
||||
worldsRoot := fs.String("worlds-root", "/worlds", "mount root under which world PVCs are visible (tarLocal: <root>/<pvc>, else the stock local-path <root>/<pv-name>_<ns>_<pvc-name>)")
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return 2
|
||||
}
|
||||
@@ -47,21 +54,8 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
return 1
|
||||
}
|
||||
|
||||
archiver, err := buildArchiver(cfg, *worldsRoot)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis reaper: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
ctx := ctrl.SetupSignalHandler()
|
||||
|
||||
drv, err := store.Open(ctx, cfg.Database.URL)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis reaper: open database: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer drv.Close()
|
||||
|
||||
scheme := runtime.NewScheme()
|
||||
utilruntime.Must(clientgoscheme.AddToScheme(scheme))
|
||||
utilruntime.Must(v1alpha1.AddToScheme(scheme))
|
||||
@@ -71,6 +65,19 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
return 1
|
||||
}
|
||||
|
||||
archiver, err := buildArchiver(ctx, cfg, *worldsRoot, cl)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis reaper: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
|
||||
drv, err := store.Open(ctx, cfg.Database.URL)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis reaper: open database: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
defer drv.Close()
|
||||
|
||||
r := &reaper.Reaper{
|
||||
Cfg: rcfg,
|
||||
Store: reaper.NewPGStore(drv.DB()),
|
||||
@@ -78,6 +85,47 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
Archiver: archiver,
|
||||
}
|
||||
|
||||
// Pre-reap warnings go out by email when [smtp] is configured (the same
|
||||
// relay and password_ref convention felis-api uses); without it the channel
|
||||
// stays nil and the reaper logs each suppressed warning instead of stamping
|
||||
// it, so a later SMTP setup still gets to warn. The owner must have a
|
||||
// VERIFIED address — that flag is what proves the mailbox.
|
||||
if cfg.SMTP.Host != "" {
|
||||
passRef := cfg.SMTP.PasswordRef
|
||||
if passRef == "" {
|
||||
passRef = platform.SMTPPasswordEnv
|
||||
}
|
||||
password := os.Getenv(passRef)
|
||||
if cfg.SMTP.Username != "" && password == "" {
|
||||
fmt.Fprintf(stderr, "felis reaper: warning: [smtp] username is set but credentials env %s is empty — warning emails will fail AUTH\n", passRef)
|
||||
}
|
||||
db := drv.DB()
|
||||
r.Warner = &mailWarner{
|
||||
lookupEmail: func(ctx context.Context, ownerID string) (string, error) {
|
||||
var email string
|
||||
switch err := db.QueryRowContext(ctx,
|
||||
`SELECT email FROM users
|
||||
WHERE id = $1 AND email_verified = true AND COALESCE(email, '') <> ''`,
|
||||
ownerID).Scan(&email); {
|
||||
case errors.Is(err, sql.ErrNoRows):
|
||||
return "", fmt.Errorf("owner %s has no verified email", ownerID)
|
||||
case err != nil:
|
||||
return "", err
|
||||
}
|
||||
return email, nil
|
||||
},
|
||||
notifier: &mail.SMTP{
|
||||
Host: cfg.SMTP.Host,
|
||||
Port: cfg.SMTP.Port,
|
||||
From: cfg.SMTP.From,
|
||||
Username: cfg.SMTP.Username,
|
||||
Password: password,
|
||||
},
|
||||
}
|
||||
} else {
|
||||
fmt.Fprintln(stderr, "felis reaper: [smtp] not configured — pre-reap warnings are logged and NOT marked sent")
|
||||
}
|
||||
|
||||
sum, err := r.RunOnce(ctx)
|
||||
if err != nil {
|
||||
fmt.Fprintf(stderr, "felis reaper: %v\n", err)
|
||||
@@ -88,6 +136,38 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
|
||||
return 0
|
||||
}
|
||||
|
||||
// mailWarner delivers a pre-reap notice to the owner's verified email — the
|
||||
// only channel this build can reach. Unowned owners and owners who never proved
|
||||
// a mailbox yield an error; the reaper retries such notices on its next run and
|
||||
// never lets them block the reap (red line ⑤).
|
||||
type mailWarner struct {
|
||||
lookupEmail func(ctx context.Context, ownerID string) (string, error)
|
||||
notifier noticeNotifier
|
||||
}
|
||||
|
||||
// noticeNotifier is the slice of mail.SMTP the warner needs (injected in tests).
|
||||
type noticeNotifier interface {
|
||||
SendNotice(ctx context.Context, email, subject, body string) error
|
||||
}
|
||||
|
||||
func (w *mailWarner) Warn(ctx context.Context, ownerID, server, remaining string) error {
|
||||
email, err := w.lookupEmail(ctx, ownerID)
|
||||
if err != nil {
|
||||
return fmt.Errorf("resolve owner email: %w", err)
|
||||
}
|
||||
subject := fmt.Sprintf("Felis: 服务器 %s 将在 %s 后回收 · server reaped in %s", server, remaining, remaining)
|
||||
body := fmt.Sprintf(
|
||||
"Felis 世界回收提醒 / world-reaper notice\r\n"+
|
||||
"\r\n"+
|
||||
"服务器 / Server: %s\r\n"+
|
||||
"距回收 / Time left: %s\r\n"+
|
||||
"\r\n"+
|
||||
"闲置的服务器会先自动备份,再释放世界;有人加入游戏即可重置倒计时。\r\n"+
|
||||
"Idle servers are backed up and then released; any join resets the countdown.\r\n",
|
||||
server, remaining)
|
||||
return w.notifier.SendNotice(ctx, email, subject, body)
|
||||
}
|
||||
|
||||
// reaperConfig derives the reaper's retention windows from felis.toml. The 15d
|
||||
// idle deadline is fixed by §18; only the warning offsets, retention, and the
|
||||
// store soft-cap are configurable (§24).
|
||||
@@ -122,22 +202,57 @@ func reaperConfig(cfg *config.Config) (reaper.Config, error) {
|
||||
}
|
||||
|
||||
// buildArchiver constructs the WorldArchiver. Only tarLocal is implemented in
|
||||
// this build; the resolver maps each world PVC to <worldsRoot>/<pvc>, the mount
|
||||
// convention the reaper Job is deployed with.
|
||||
func buildArchiver(cfg *config.Config, worldsRoot string) (backup.WorldArchiver, error) {
|
||||
// this build; the resolver maps each world PVC to its directory under worldsRoot
|
||||
// (resolveWorldDir).
|
||||
func buildArchiver(ctx context.Context, cfg *config.Config, worldsRoot string, cl client.Client) (backup.WorldArchiver, error) {
|
||||
switch cfg.Archive.Store {
|
||||
case "tarLocal":
|
||||
return &backup.TarLocal{
|
||||
BackupRoot: cfg.Archive.LocalPath,
|
||||
Resolve: func(pvc string) (string, error) {
|
||||
return filepath.Join(worldsRoot, pvc), nil
|
||||
},
|
||||
Resolve: resolveWorldDir(ctx, cl, cfg.K8s.Namespace, worldsRoot),
|
||||
}, nil
|
||||
default:
|
||||
return nil, fmt.Errorf("[archive] store %q is not implemented in this build (only tarLocal)", cfg.Archive.Store)
|
||||
}
|
||||
}
|
||||
|
||||
// resolveWorldDir maps a world PVC to its directory under worldsRoot, supporting
|
||||
// the two layouts a Felis host actually has:
|
||||
//
|
||||
// 1. <root>/<pvc> — the reaper's documented arrangement (worlds exposed by PVC
|
||||
// name, e.g. via mounting each volume or a crafted storage class).
|
||||
// 2. <root>/<pv-name>_<namespace>_<pvc-name> — what a stock k3s install gets:
|
||||
// local-path-provisioner stores every volume under its storage root as that
|
||||
// exact directory name. Without this arm, retention on a default install could
|
||||
// only ever fail to find a world (a no-op reaper, or worse an operator
|
||||
// arranging paths by hand).
|
||||
//
|
||||
// The second path is derived EXACTLY from the live PVC's spec.volumeName, never
|
||||
// from a glob: a leftover directory of an old, deleted PV must never be mistaken
|
||||
// for the world the PVC currently binds, because the reaper archives the resolved
|
||||
// directory and then deletes that PVC — archiving stale bytes and deleting the
|
||||
// real world would be data loss. When neither path exists the first is returned,
|
||||
// so the archive walk fails loudly against the documented path.
|
||||
func resolveWorldDir(ctx context.Context, cl client.Client, namespace, worldsRoot string) backup.PVCResolver {
|
||||
return func(pvc string) (string, error) {
|
||||
direct := filepath.Join(worldsRoot, pvc)
|
||||
if _, err := os.Stat(direct); err == nil {
|
||||
return direct, nil
|
||||
}
|
||||
var claim corev1.PersistentVolumeClaim
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: namespace, Name: pvc}, &claim); err != nil {
|
||||
return "", fmt.Errorf("resolve world PVC %s: %w", pvc, err)
|
||||
}
|
||||
if pv := claim.Spec.VolumeName; pv != "" {
|
||||
volDir := filepath.Join(worldsRoot, fmt.Sprintf("%s_%s_%s", pv, claim.Namespace, claim.Name))
|
||||
if _, err := os.Stat(volDir); err == nil {
|
||||
return volDir, nil
|
||||
}
|
||||
}
|
||||
return direct, nil
|
||||
}
|
||||
}
|
||||
|
||||
// parseSpanDuration parses the human spans used in felis.toml's [archive] table:
|
||||
// "3mo" (months≈30d), "15d" (days), or any time.ParseDuration unit ("12h").
|
||||
func parseSpanDuration(s string) (time.Duration, error) {
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
)
|
||||
|
||||
// TestResolveWorldDir pins the two world layouts the reaper must find, and the
|
||||
// fail-closed miss. The stock local-path arm is derived from the live PVC's
|
||||
// volumeName — a name-based guess (glob) could tar a stale deleted PV's bytes and
|
||||
// then delete the current world, which is why it is read from the API instead.
|
||||
func TestResolveWorldDir(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
root := t.TempDir()
|
||||
|
||||
// Arrange a world under the documented <root>/<pvc> layout.
|
||||
named := filepath.Join(root, "world-named-0")
|
||||
if err := os.MkdirAll(named, 0o750); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Arrange a second world the way k3s local-path stores it.
|
||||
pvDir := filepath.Join(root, "pvc-11111111-2222-3333-4444-555555555555_minecraft_world-live-0")
|
||||
if err := os.MkdirAll(pvDir, 0o750); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
claim := &corev1.PersistentVolumeClaim{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "world-live-0", Namespace: "minecraft"},
|
||||
Spec: corev1.PersistentVolumeClaimSpec{
|
||||
VolumeName: "pvc-11111111-2222-3333-4444-555555555555",
|
||||
},
|
||||
}
|
||||
cl := fake.NewClientBuilder().WithScheme(haltScheme(t)).WithObjects(claim).Build()
|
||||
resolve := resolveWorldDir(ctx, cl, "minecraft", root)
|
||||
|
||||
t.Run("documented name layout wins", func(t *testing.T) {
|
||||
got, err := resolve("world-named-0")
|
||||
if err != nil || got != named {
|
||||
t.Fatalf("resolve = (%q, %v), want (%q, nil)", got, err, named)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("stock local-path layout resolves exactly", func(t *testing.T) {
|
||||
got, err := resolve("world-live-0")
|
||||
if err != nil || got != pvDir {
|
||||
t.Fatalf("resolve = (%q, %v), want (%q, nil)", got, err, pvDir)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("neither layout present falls back to the documented path", func(t *testing.T) {
|
||||
// The claim exists but its directory does not: return the documented path so
|
||||
// the archive walk fails there, and the reaper preserves the world.
|
||||
missing := &corev1.PersistentVolumeClaim{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "world-gone-0", Namespace: "minecraft"},
|
||||
Spec: corev1.PersistentVolumeClaimSpec{VolumeName: "pvc-99999999-0000-0000-0000-000000000000"},
|
||||
}
|
||||
cl := fake.NewClientBuilder().WithScheme(haltScheme(t)).WithObjects(missing).Build()
|
||||
got, err := resolveWorldDir(ctx, cl, "minecraft", root)("world-gone-0")
|
||||
if err != nil || got != filepath.Join(root, "world-gone-0") {
|
||||
t.Fatalf("resolve = (%q, %v), want (%q, nil)", got, err, filepath.Join(root, "world-gone-0"))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("unknown pvc is an error, not a guess", func(t *testing.T) {
|
||||
_, err := resolve("world-unknown-0")
|
||||
if err == nil || !strings.Contains(err.Error(), "resolve world PVC world-unknown-0") {
|
||||
t.Fatalf("err = %v, want a resolve-world-PVC error", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// The pre-reap warner resolves the owner's VERIFIED email and hands the notice
|
||||
// to the mailer. Every failure (no verified address, relay refusal) returns an
|
||||
// error so the reaper retries on its next run instead of stamping a notice
|
||||
// nobody received.
|
||||
func TestMailWarner(t *testing.T) {
|
||||
lookup := func(email string, err error) func(context.Context, string) (string, error) {
|
||||
return func(context.Context, string) (string, error) { return email, err }
|
||||
}
|
||||
|
||||
n := &captureNotifier{}
|
||||
w := &mailWarner{lookupEmail: lookup("[email protected]", nil), notifier: n}
|
||||
if err := w.Warn(context.Background(), "u1", "survival", "3d"); err != nil {
|
||||
t.Fatalf("Warn: %v", err)
|
||||
}
|
||||
if n.email != "[email protected]" || !strings.Contains(n.subject, "survival") || !strings.Contains(n.subject, "3d") {
|
||||
t.Fatalf("notice envelope = (%q, %q)", n.email, n.subject)
|
||||
}
|
||||
if !strings.Contains(n.body, "survival") || !strings.Contains(n.body, "3d") {
|
||||
t.Fatalf("body missing server/remaining:\n%s", n.body)
|
||||
}
|
||||
|
||||
w = &mailWarner{lookupEmail: lookup("", errors.New("owner u2 has no verified email")), notifier: n}
|
||||
if err := w.Warn(context.Background(), "u2", "survival", "3d"); err == nil || !strings.Contains(err.Error(), "verified email") {
|
||||
t.Fatalf("unverified owner = %v, want the lookup error surfaced", err)
|
||||
}
|
||||
|
||||
w = &mailWarner{lookupEmail: lookup("[email protected]", nil), notifier: &captureNotifier{err: errors.New("relay down")}}
|
||||
if err := w.Warn(context.Background(), "u1", "survival", "3d"); err == nil || !strings.Contains(err.Error(), "relay down") {
|
||||
t.Fatalf("relay failure = %v, want it surfaced", err)
|
||||
}
|
||||
}
|
||||
|
||||
type captureNotifier struct {
|
||||
email, subject, body string
|
||||
err error
|
||||
}
|
||||
|
||||
func (n *captureNotifier) SendNotice(_ context.Context, email, subject, body string) error {
|
||||
if n.err != nil {
|
||||
return n.err
|
||||
}
|
||||
n.email, n.subject, n.body = email, subject, body
|
||||
return nil
|
||||
}
|
||||
@@ -19,9 +19,11 @@ Commands:
|
||||
restore Extract a world archive into a world volume (internal Job entrypoint)
|
||||
backup Archive a world into the backup store and record it (internal Job entrypoint)
|
||||
files List/read/write one file in a stopped server's world (internal Job entrypoint)
|
||||
fetch-context Fetch and extract a submission's build context (internal Job entrypoint)
|
||||
manifests Render the control-plane RBAC + NetworkPolicy install bundle as YAML
|
||||
apply Create a MinecraftServer CRD (direct K8s write; use -f server.json)
|
||||
setup Run host bootstrap + first-run setup console (TUI; requires root/sudo)
|
||||
converge Fill in fields a newer desired spec added to already-installed system servers
|
||||
version Print the build stamp of this binary
|
||||
update Report which platform components have updates available
|
||||
breakGlass Open the local break-glass emergency console (TUI; requires root/sudo)
|
||||
@@ -47,9 +49,11 @@ var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
|
||||
"restore": cmdRestore,
|
||||
"backup": cmdBackup,
|
||||
"files": cmdFiles,
|
||||
"fetch-context": cmdFetchContext,
|
||||
"manifests": cmdManifests,
|
||||
"apply": cmdApply,
|
||||
"setup": cmdSetup,
|
||||
"converge": cmdConverge,
|
||||
"breakGlass": cmdBreakGlass,
|
||||
"bootstrap-assets": cmdBootstrapAssets,
|
||||
"init-forwarding": cmdInitForwarding,
|
||||
|
||||
+28
-3
@@ -233,13 +233,36 @@ func provisionSystemServers(ctx context.Context, cfg *config.Config, out io.Writ
|
||||
// (the on-demand BACKUP Job runs in the minecraft namespace and mounts it to
|
||||
// self-record its world_backups row; without the replica the Job's volume
|
||||
// mount fails and every backup request strands in the cluster).
|
||||
// An empty build_namespace means the build system's compiled-in default; the
|
||||
// replica must target the namespace the Jobs actually run in.
|
||||
buildNS := cfg.Registry.BuildNamespace
|
||||
if buildNS == "" {
|
||||
buildNS = platform.DefaultBuildNamespace
|
||||
}
|
||||
secretOutcomes := []systemServerOutcome{
|
||||
ensureSecretReplica(ctx, cl, controlNS, cfg.K8s.Namespace,
|
||||
naming.ServiceTokenSecretName, naming.ServiceTokenSecretKey, "service-token"),
|
||||
naming.ServiceTokenSecretName, naming.ServiceTokenSecretKey, "service-token", "minecraft ns", false),
|
||||
ensureSecretReplica(ctx, cl, controlNS, cfg.K8s.Namespace,
|
||||
naming.ForwardingSecretName, naming.ForwardingSecretKey, "forwarding-secret"),
|
||||
naming.ForwardingSecretName, naming.ForwardingSecretKey, "forwarding-secret", "minecraft ns", false),
|
||||
// refresh=true: felis-config is the rendered config, not a credential. The
|
||||
// backup/restore/fileedit Jobs and the reaper mount this copy, so a re-run
|
||||
// must update it when the control plane's render has moved on (a stale copy
|
||||
// e.g. keeps an old database URL after a credential rotation).
|
||||
ensureSecretReplica(ctx, cl, controlNS, cfg.K8s.Namespace,
|
||||
"felis-config", "felis.toml", "config"),
|
||||
"felis-config", "felis.toml", "config", "minecraft ns", true),
|
||||
// The reaper's pre-reap warning emails authenticate with the same relay
|
||||
// password felis-api uses; the reaper pod runs in the minecraft namespace,
|
||||
// where a secretKeyRef resolves only against a local mirror. Skipped while
|
||||
// the relay is not configured yet — the "configure email" screen refreshes
|
||||
// both mirrors when it applies.
|
||||
ensureSecretReplica(ctx, cl, controlNS, cfg.K8s.Namespace,
|
||||
"felis-smtp", "password", "smtp", "minecraft ns", false),
|
||||
// The build namespace needs the same token: the build Job's fetch
|
||||
// initContainer reads the submission context from the internal face. Best
|
||||
// effort — a deployment that only installs the control plane simply never
|
||||
// builds a user submission.
|
||||
ensureSecretReplica(ctx, cl, controlNS, buildNS,
|
||||
naming.ServiceTokenSecretName, naming.ServiceTokenSecretKey, "service-token", "felis-build ns", false),
|
||||
}
|
||||
outcomes := ensureSystemServers(ctx, cl, cfg.K8s.Namespace, cfg.Velocity.LoginImage, cfg.Velocity.LobbyImage, apiBaseURL, cfg.Server.RootDomain, defaultPanelHostname(cfg.Server.RootDomain, cfg.Auth.PanelHostname))
|
||||
outcomes = append(secretOutcomes, outcomes...)
|
||||
@@ -250,6 +273,8 @@ func provisionSystemServers(ctx context.Context, cfg *config.Config, out io.Writ
|
||||
fmt.Fprintf(out, " - %s: ERROR %v\n", o.name, o.err)
|
||||
case o.created:
|
||||
fmt.Fprintf(out, " - %s: created (DesiredState=Running)\n", o.name)
|
||||
case o.updated:
|
||||
fmt.Fprintf(out, " - %s: refreshed from the control namespace\n", o.name)
|
||||
default:
|
||||
fmt.Fprintf(out, " - %s: skipped (%s)\n", o.name, o.skipped)
|
||||
}
|
||||
|
||||
+194
-35
@@ -1,6 +1,7 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"time"
|
||||
@@ -95,7 +96,7 @@ const felisLimboHealthPort int32 = 8080
|
||||
// fail-safes to readiness-only, so a login pod that has the URL/domain but not yet
|
||||
// the token is safe (it simply does not authenticate) rather than broken.
|
||||
const (
|
||||
envAPIBaseURL = "FELIS_API_BASE_URL"
|
||||
envAPIBaseURL = naming.EnvAPIBaseURL
|
||||
envRootDomain = "FELIS_ROOT_DOMAIN"
|
||||
envPanelHostname = "FELIS_PANEL_HOSTNAME"
|
||||
envLobbyServer = "FELIS_LOBBY_SERVER"
|
||||
@@ -255,10 +256,31 @@ func buildSystemServerClient() (client.Client, error) {
|
||||
// setup can report it without the provisioner deciding on the output format.
|
||||
type systemServerOutcome struct {
|
||||
name string
|
||||
created bool // true = we created it this run
|
||||
available bool // true = the required object now exists
|
||||
skipped string // non-empty = why it was skipped (image unset / already exists)
|
||||
err error // non-nil = create failed
|
||||
created bool // true = we created it this run
|
||||
updated bool // true = we refreshed an existing replica from the source
|
||||
available bool // true = the required object now exists
|
||||
skipped string // non-empty = why it was skipped (image unset / already exists)
|
||||
err error // non-nil = create failed
|
||||
changes []string // converge only: the fields this pass filled
|
||||
}
|
||||
|
||||
// systemServerPlan is one system service in the provisioner's table: its name,
|
||||
// the image config gives it, and the pure builder for its desired CR.
|
||||
type systemServerPlan struct {
|
||||
name string
|
||||
image string
|
||||
build func(image, namespace string) (*v1alpha1.MinecraftServer, error)
|
||||
}
|
||||
|
||||
// systemServerPlans is the single description of the login+lobby pair, shared by
|
||||
// ensureSystemServers (create-if-absent) and convergeSystemServers (field fill).
|
||||
func systemServerPlans(loginImage, lobbyImage, apiBaseURL, rootDomain, panelHostname string) []systemServerPlan {
|
||||
return []systemServerPlan{
|
||||
{name: naming.SystemLoginServer, image: loginImage, build: func(image, ns string) (*v1alpha1.MinecraftServer, error) {
|
||||
return loginSystemServer(image, ns, apiBaseURL, rootDomain, panelHostname)
|
||||
}},
|
||||
{name: naming.SystemLobbyServer, image: lobbyImage, build: lobbySystemServer},
|
||||
}
|
||||
}
|
||||
|
||||
// ensureSystemServers idempotently creates the login and lobby system services.
|
||||
@@ -269,17 +291,7 @@ type systemServerOutcome struct {
|
||||
// K8s client and namespace; this function performs no signal-handler or client
|
||||
// setup of its own.
|
||||
func ensureSystemServers(ctx context.Context, cl client.Client, namespace, loginImage, lobbyImage, apiBaseURL, rootDomain, panelHostname string) []systemServerOutcome {
|
||||
type plan struct {
|
||||
name string
|
||||
image string
|
||||
build func(image, namespace string) (*v1alpha1.MinecraftServer, error)
|
||||
}
|
||||
plans := []plan{
|
||||
{name: naming.SystemLoginServer, image: loginImage, build: func(image, ns string) (*v1alpha1.MinecraftServer, error) {
|
||||
return loginSystemServer(image, ns, apiBaseURL, rootDomain, panelHostname)
|
||||
}},
|
||||
{name: naming.SystemLobbyServer, image: lobbyImage, build: lobbySystemServer},
|
||||
}
|
||||
plans := systemServerPlans(loginImage, lobbyImage, apiBaseURL, rootDomain, panelHostname)
|
||||
|
||||
outcomes := make([]systemServerOutcome, 0, len(plans))
|
||||
for _, p := range plans {
|
||||
@@ -379,12 +391,7 @@ var derivedSystemEnv = map[string]bool{
|
||||
// deliberate removal is indistinguishable from drift and re-adding it would fight the
|
||||
// operator every run.
|
||||
func refreshDerivedEnv(ctx context.Context, cl client.Client, existing, desired *v1alpha1.MinecraftServer) (bool, error) {
|
||||
want := make(map[string]string, len(derivedSystemEnv))
|
||||
for _, e := range desired.Spec.Env {
|
||||
if derivedSystemEnv[e.Name] {
|
||||
want[e.Name] = e.Value
|
||||
}
|
||||
}
|
||||
want := derivedEnvWanted(desired)
|
||||
|
||||
changed := false
|
||||
for i, e := range existing.Spec.Env {
|
||||
@@ -402,6 +409,116 @@ func refreshDerivedEnv(ctx context.Context, cl client.Client, existing, desired
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// derivedEnvWanted maps the derived env keys of desired onto their values.
|
||||
func derivedEnvWanted(desired *v1alpha1.MinecraftServer) map[string]string {
|
||||
want := make(map[string]string, len(derivedSystemEnv))
|
||||
for _, e := range desired.Spec.Env {
|
||||
if derivedSystemEnv[e.Name] {
|
||||
want[e.Name] = e.Value
|
||||
}
|
||||
}
|
||||
return want
|
||||
}
|
||||
|
||||
// convergeSystemServers is the explicit convergence pass over already-installed
|
||||
// system servers (#1). ensureSystemServers is create-if-absent by design — an
|
||||
// existing CR is left alone so a re-run cannot clobber an operator's edits — and
|
||||
// that leaves no path for a field the DESIRED spec gained after the install:
|
||||
// spec.rcon (the lobby's write channel), spec.startup.healthHTTPPort (the login
|
||||
// gate's readiness probe), or a config-derived env key that did not exist yet.
|
||||
// Such fields sit at their zero value forever while re-running setup reports
|
||||
// success, which is exactly the reported "configuration updates never reach an
|
||||
// installed deployment" symptom.
|
||||
//
|
||||
// This pass fills exactly those zero-value fields and the config-derived env keys,
|
||||
// and nothing else: a field already holding a non-zero value is the operator's and
|
||||
// is never overwritten. It is an explicit command rather than an implicit step of
|
||||
// setup because some fills need an ordering only the operator knows — enabling
|
||||
// RCON or the HTTP readiness gate on a server whose image predates the listener
|
||||
// would hold that server in Starting until it was marked Failed. Rebuild (or
|
||||
// upgrade) the images first, then run this.
|
||||
func convergeSystemServers(ctx context.Context, cl client.Client, namespace, loginImage, lobbyImage, apiBaseURL, rootDomain, panelHostname string) []systemServerOutcome {
|
||||
outcomes := make([]systemServerOutcome, 0, 2)
|
||||
for _, p := range systemServerPlans(loginImage, lobbyImage, apiBaseURL, rootDomain, panelHostname) {
|
||||
if p.image == "" {
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name, skipped: "image not configured"})
|
||||
continue
|
||||
}
|
||||
desired, err := p.build(p.image, namespace)
|
||||
if err != nil {
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name, err: err})
|
||||
continue
|
||||
}
|
||||
|
||||
var existing v1alpha1.MinecraftServer
|
||||
switch err := cl.Get(ctx, client.ObjectKeyFromObject(desired), &existing); {
|
||||
case apierrors.IsNotFound(err):
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name,
|
||||
skipped: "not present — run `sudo felis setup` first"})
|
||||
continue
|
||||
case err != nil:
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name, err: err})
|
||||
continue
|
||||
}
|
||||
if existing.Labels[v1alpha1.LabelSystemRole] != p.name {
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name, err: fmt.Errorf(
|
||||
"existing MinecraftServer %s/%s is not marked as the Felis %q system role; refusing to converge it",
|
||||
namespace, p.name, p.name,
|
||||
)})
|
||||
continue
|
||||
}
|
||||
|
||||
var changes []string
|
||||
if existing.Spec.Rcon == (v1alpha1.RconSpec{}) && desired.Spec.Rcon != (v1alpha1.RconSpec{}) {
|
||||
existing.Spec.Rcon = desired.Spec.Rcon
|
||||
changes = append(changes, "spec.rcon")
|
||||
}
|
||||
if existing.Spec.Startup.HealthHTTPPort == 0 && desired.Spec.Startup.HealthHTTPPort != 0 {
|
||||
existing.Spec.Startup.HealthHTTPPort = desired.Spec.Startup.HealthHTTPPort
|
||||
changes = append(changes, "spec.startup.healthHTTPPort")
|
||||
}
|
||||
changes = append(changes, convergeDerivedEnv(&existing, desired)...)
|
||||
|
||||
if len(changes) == 0 {
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name, available: true, skipped: "already converged"})
|
||||
continue
|
||||
}
|
||||
if err := cl.Update(ctx, &existing); err != nil {
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name, err: fmt.Errorf("converge %s: %w", p.name, err)})
|
||||
continue
|
||||
}
|
||||
outcomes = append(outcomes, systemServerOutcome{name: p.name, available: true, updated: true, changes: changes})
|
||||
}
|
||||
return outcomes
|
||||
}
|
||||
|
||||
// convergeDerivedEnv makes the config-derived env match the desired values: a key
|
||||
// whose value drifted is overwritten, and a key missing entirely is added. This is
|
||||
// the wider half of the same explicit pass — refreshDerivedEnv's present-only loop
|
||||
// can never introduce a NEW key, which is how a derived key added after an install
|
||||
// never reached it at all.
|
||||
func convergeDerivedEnv(existing, desired *v1alpha1.MinecraftServer) []string {
|
||||
want := derivedEnvWanted(desired)
|
||||
var changes []string
|
||||
present := make(map[string]bool, len(existing.Spec.Env))
|
||||
for i := range existing.Spec.Env {
|
||||
e := &existing.Spec.Env[i]
|
||||
present[e.Name] = true
|
||||
if v, ok := want[e.Name]; ok && v != e.Value {
|
||||
e.Value = v
|
||||
changes = append(changes, "env "+e.Name)
|
||||
}
|
||||
}
|
||||
for _, e := range desired.Spec.Env {
|
||||
if !derivedSystemEnv[e.Name] || present[e.Name] {
|
||||
continue
|
||||
}
|
||||
existing.Spec.Env = append(existing.Spec.Env, e)
|
||||
changes = append(changes, "env "+e.Name)
|
||||
}
|
||||
return changes
|
||||
}
|
||||
|
||||
// The login gate is a hard prerequisite of the Owner bind, so setup waits for it
|
||||
// rather than racing it. The ceiling covers a cold image pull on a fresh node;
|
||||
// the poll is fast enough that a warm start feels immediate.
|
||||
@@ -471,26 +588,35 @@ func phaseOrPending(p v1alpha1.Phase) string {
|
||||
return string(p)
|
||||
}
|
||||
|
||||
// ensureSecretReplica copies one Secret from the control namespace into the minecraft
|
||||
// namespace so a backend pod can mount it via secretKeyRef. A secretKeyRef is
|
||||
// namespace-local, but the backends run in the minecraft namespace while the sources
|
||||
// of truth live beside the control plane — so without this replica the operator's
|
||||
// injected secretKeyRef would dangle and wedge the pod in CreateContainerConfigError.
|
||||
// ensureSecretReplica copies one Secret from the control namespace into a workload
|
||||
// namespace (minecraft — or the build namespace, whose fetch initContainer reads the
|
||||
// context from the felis-api internal face with the same token) so a pod can mount it
|
||||
// via secretKeyRef. A secretKeyRef is namespace-local, but those workloads do not run
|
||||
// beside the control plane — so without this replica the secretKeyRef would dangle and
|
||||
// wedge the pod in CreateContainerConfigError.
|
||||
//
|
||||
// Two Secrets need it, for different reasons: the service token (login only — it
|
||||
// authenticates the limbo plugin to the felis-api internal face) and the Velocity
|
||||
// modern-forwarding secret (every backend — it is how a backend knows a login really
|
||||
// came from the proxy, and so that the player's UUID is Mojang-verified rather than
|
||||
// offline-derived).
|
||||
// Three Secrets need it, for different reasons: the service token (the login limbo and
|
||||
// the build Pod's context fetch — both authenticate to the felis-api internal face),
|
||||
// the Velocity modern-forwarding secret (every backend — it is how a backend knows
|
||||
// a login really came from the proxy, and so that the player's UUID is Mojang-verified
|
||||
// rather than offline-derived), and the SMTP relay password (the reaper's pre-reap
|
||||
// warning emails; the felis-config mirror is what carries [smtp] into its pod).
|
||||
//
|
||||
// It is create-if-absent: an existing replica is left untouched so a hand-rotated
|
||||
// value in the minecraft namespace is never clobbered (to rotate, delete the replica
|
||||
// value in the workload namespace is never clobbered (to rotate, delete the replica
|
||||
// and re-run setup). Best-effort like the rest of the provisioner: a missing source or
|
||||
// a create failure degrades to a reported outcome, never a hard setup failure. It
|
||||
// copies only Type and Data — never labels/annotations/ownerRefs — so the replica
|
||||
// carries no accidental GC owner or managed-by lineage.
|
||||
func ensureSecretReplica(ctx context.Context, cl client.Client, controlNamespace, minecraftNamespace, secretName, secretKey, label string) systemServerOutcome {
|
||||
name := label + " (minecraft ns)"
|
||||
//
|
||||
// refreshExisting switches the felis-config mirror to refresh-in-place: that Secret is
|
||||
// a rendered config, never a hand-rotated credential, and the workload Jobs that mount
|
||||
// it (backup/restore/fileedit) plus the reaper silently misbehave on a stale copy —
|
||||
// e.g. after a database credential rotation the control plane moves on while every
|
||||
// backup Job keeps failing auth. Credential Secrets keep the never-overwrite rule so a
|
||||
// rotated value survives; to rotate those, delete the replica and re-run setup.
|
||||
func ensureSecretReplica(ctx context.Context, cl client.Client, controlNamespace, minecraftNamespace, secretName, secretKey, label, where string, refreshExisting bool) systemServerOutcome {
|
||||
name := label + " (" + where + ")"
|
||||
validate := func(secret *corev1.Secret, location, skipped string) systemServerOutcome {
|
||||
if len(secret.Data[secretKey]) == 0 {
|
||||
return systemServerOutcome{name: name, skipped: fmt.Sprintf(
|
||||
@@ -498,6 +624,33 @@ func ensureSecretReplica(ctx context.Context, cl client.Client, controlNamespace
|
||||
}
|
||||
return systemServerOutcome{name: name, available: true, skipped: skipped}
|
||||
}
|
||||
// refreshFromControl updates an existing replica from the control-namespace source
|
||||
// when the rendered key differs. Only the felis-config mirror opts in.
|
||||
refreshFromControl := func(existing *corev1.Secret) systemServerOutcome {
|
||||
var src corev1.Secret
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: controlNamespace, Name: secretName}, &src); err != nil {
|
||||
if apierrors.IsNotFound(err) {
|
||||
return systemServerOutcome{name: name, skipped: fmt.Sprintf(
|
||||
"source Secret %s/%s not found — provision it (deploy/bootstrap.sh), then re-run setup",
|
||||
controlNamespace, secretName)}
|
||||
}
|
||||
return systemServerOutcome{name: name, err: err}
|
||||
}
|
||||
if out := validate(&src, controlNamespace, ""); !out.available {
|
||||
return out
|
||||
}
|
||||
if bytes.Equal(existing.Data[secretKey], src.Data[secretKey]) {
|
||||
return validate(existing, minecraftNamespace, "already current")
|
||||
}
|
||||
if existing.Data == nil {
|
||||
existing.Data = map[string][]byte{}
|
||||
}
|
||||
existing.Data[secretKey] = src.Data[secretKey]
|
||||
if err := cl.Update(ctx, existing); err != nil {
|
||||
return systemServerOutcome{name: name, err: err}
|
||||
}
|
||||
return systemServerOutcome{name: name, updated: true, available: true}
|
||||
}
|
||||
if controlNamespace == minecraftNamespace {
|
||||
// Same namespace needs no replica, but the source still has to exist.
|
||||
var existing corev1.Secret
|
||||
@@ -516,6 +669,9 @@ func ensureSecretReplica(ctx context.Context, cl client.Client, controlNamespace
|
||||
var existing corev1.Secret
|
||||
getErr := cl.Get(ctx, client.ObjectKey{Namespace: minecraftNamespace, Name: secretName}, &existing)
|
||||
if getErr == nil {
|
||||
if refreshExisting {
|
||||
return refreshFromControl(&existing)
|
||||
}
|
||||
return validate(&existing, minecraftNamespace, "already exists")
|
||||
}
|
||||
if !apierrors.IsNotFound(getErr) {
|
||||
@@ -544,6 +700,9 @@ func ensureSecretReplica(ctx context.Context, cl client.Client, controlNamespace
|
||||
if getErr := cl.Get(ctx, client.ObjectKey{Namespace: minecraftNamespace, Name: secretName}, &existing); getErr != nil {
|
||||
return systemServerOutcome{name: name, err: getErr}
|
||||
}
|
||||
if refreshExisting {
|
||||
return refreshFromControl(&existing)
|
||||
}
|
||||
return validate(&existing, minecraftNamespace, "already exists")
|
||||
}
|
||||
return systemServerOutcome{name: name, err: err}
|
||||
|
||||
@@ -156,7 +156,7 @@ func TestEnsureSecretReplica(t *testing.T) {
|
||||
}
|
||||
replicate := func(cl client.Client, controlNS, mcNS string) systemServerOutcome {
|
||||
return ensureSecretReplica(ctx, cl, controlNS, mcNS,
|
||||
naming.ServiceTokenSecretName, naming.ServiceTokenSecretKey, "service-token")
|
||||
naming.ServiceTokenSecretName, naming.ServiceTokenSecretKey, "service-token", "minecraft ns", false)
|
||||
}
|
||||
|
||||
t.Run("replicates when absent", func(t *testing.T) {
|
||||
@@ -251,6 +251,87 @@ func TestEnsureSecretReplica(t *testing.T) {
|
||||
})
|
||||
}
|
||||
|
||||
// The felis-config mirror is the one replica that must refresh: it is a rendered
|
||||
// config, and a stale workload-side copy (backup/restore/fileedit Jobs, the reaper)
|
||||
// misbehaves silently — a rotated database credential keeps the control plane moving
|
||||
// while every backup Job keeps failing auth. Credential Secrets keep create-if-absent.
|
||||
func TestEnsureSecretReplicaRefresh(t *testing.T) {
|
||||
scheme := newSystemServerScheme(t)
|
||||
ctx := context.Background()
|
||||
configSecret := func(ns, body string) *corev1.Secret {
|
||||
return &corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "felis-config", Namespace: ns},
|
||||
Type: corev1.SecretTypeOpaque,
|
||||
Data: map[string][]byte{"felis.toml": []byte(body)},
|
||||
}
|
||||
}
|
||||
refresh := func(cl client.Client) systemServerOutcome {
|
||||
return ensureSecretReplica(ctx, cl, "felis", "minecraft",
|
||||
"felis-config", "felis.toml", "config", "minecraft ns", true)
|
||||
}
|
||||
replicaBody := func(t *testing.T, cl client.Client) string {
|
||||
t.Helper()
|
||||
var got corev1.Secret
|
||||
if err := cl.Get(ctx, client.ObjectKey{Namespace: "minecraft", Name: "felis-config"}, &got); err != nil {
|
||||
t.Fatalf("get replica: %v", err)
|
||||
}
|
||||
return string(got.Data["felis.toml"])
|
||||
}
|
||||
|
||||
t.Run("refreshes a stale config replica", func(t *testing.T) {
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(
|
||||
configSecret("felis", "current"),
|
||||
configSecret("minecraft", "stale"),
|
||||
).Build()
|
||||
out := refresh(cl)
|
||||
if out.err != nil || !out.updated || !out.available {
|
||||
t.Fatalf("outcome = %+v, want refreshed", out)
|
||||
}
|
||||
if got := replicaBody(t, cl); got != "current" {
|
||||
t.Errorf("replica = %q, want current", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("leaves a current config replica alone", func(t *testing.T) {
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(
|
||||
configSecret("felis", "same"),
|
||||
configSecret("minecraft", "same"),
|
||||
).Build()
|
||||
out := refresh(cl)
|
||||
if out.err != nil || out.updated || !out.available || out.skipped != "already current" {
|
||||
t.Fatalf("outcome = %+v, want already current", out)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("fills an empty-key replica", func(t *testing.T) {
|
||||
empty := &corev1.Secret{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "felis-config", Namespace: "minecraft"},
|
||||
Data: map[string][]byte{},
|
||||
}
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(
|
||||
configSecret("felis", "current"), empty).Build()
|
||||
out := refresh(cl)
|
||||
if out.err != nil || !out.updated {
|
||||
t.Fatalf("outcome = %+v, want refreshed", out)
|
||||
}
|
||||
if got := replicaBody(t, cl); got != "current" {
|
||||
t.Errorf("replica = %q, want current", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("missing source degrades to a skip", func(t *testing.T) {
|
||||
cl := fake.NewClientBuilder().WithScheme(scheme).WithObjects(
|
||||
configSecret("minecraft", "stale")).Build()
|
||||
out := refresh(cl)
|
||||
if out.err != nil || out.updated || out.available || out.skipped == "" {
|
||||
t.Fatalf("outcome = %+v, want skipped (source missing)", out)
|
||||
}
|
||||
if got := replicaBody(t, cl); got != "stale" {
|
||||
t.Errorf("replica = %q, want untouched stale", got)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestRequiredProvisioningError(t *testing.T) {
|
||||
ready := []systemServerOutcome{
|
||||
{name: "service-token (minecraft ns)", available: true},
|
||||
|
||||
@@ -86,7 +86,7 @@ func (m *backupModel) loadCmd() tea.Cmd {
|
||||
if err != nil {
|
||||
return backupListMsg{err: fmt.Errorf("list servers: %w", err)}
|
||||
}
|
||||
return backupListMsg{cl: cl, servers: servers}
|
||||
return backupListMsg{cl: cl, servers: backupPickable(servers)}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -133,6 +133,11 @@ func writeConfig(path string, cfg *config.Config) error {
|
||||
return os.Rename(tmpPath, path)
|
||||
}
|
||||
|
||||
// applyFelisConfigSecret applies the rendered config to the control namespace and
|
||||
// then converges the workload-namespace mirror best-effort. The mirror feeds the
|
||||
// backup/restore/fileedit Jobs and the reaper; without this refresh a reconfigure
|
||||
// here would leave those readers on the previous render until the next `felis
|
||||
// setup` run (startup pass) or installer re-run.
|
||||
func applyFelisConfigSecret(ctx context.Context) error {
|
||||
out, err := kubectlOutput(ctx,
|
||||
"-n", "felis", "create", "secret", "generic", "felis-config",
|
||||
@@ -142,7 +147,41 @@ func applyFelisConfigSecret(ctx context.Context) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return kubectlWithInput(ctx, out, "apply", "-f", "-")
|
||||
if err := kubectlWithInput(ctx, out, "apply", "-f", "-"); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := replicateFelisConfigToWorkloadNamespace(ctx); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "felis setup: warning: the control-plane config is applied, but the workload-namespace mirror could not be refreshed (%v); re-run felis setup once that is fixed\n", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// replicateFelisConfigToWorkloadNamespace overwrites the workload-namespace
|
||||
// felis-config mirror with the freshly rendered pod config. Deliberately a full
|
||||
// replace, not create-if-absent: a stale mirror is exactly what silently hands
|
||||
// the Jobs that mount it old settings after a reconfigure. No-op when the
|
||||
// workload namespace is unset or is the control namespace itself.
|
||||
func replicateFelisConfigToWorkloadNamespace(ctx context.Context) error {
|
||||
cfg, err := config.Load(hostSetupConfigPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ns := cfg.K8s.Namespace
|
||||
if ns == "" || ns == "felis" {
|
||||
return nil
|
||||
}
|
||||
manifest, err := kubectlOutput(ctx,
|
||||
"-n", ns, "create", "secret", "generic", "felis-config",
|
||||
"--from-file=felis.toml="+podSetupConfigPath,
|
||||
"--dry-run=client", "-o", "yaml",
|
||||
)
|
||||
if err != nil {
|
||||
return fmt.Errorf("render felis-config for %s: %w", ns, err)
|
||||
}
|
||||
if err := kubectlWithInput(ctx, manifest, "-n", ns, "apply", "-f", "-"); err != nil {
|
||||
return fmt.Errorf("replicate felis-config to %s: %w", ns, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func installCloudflaredService(ctx context.Context, cloudflaredBin, configPath string) error {
|
||||
|
||||
@@ -207,7 +207,11 @@ func (m *mcBindModel) doneView() string {
|
||||
if box.Len() > 0 {
|
||||
box.WriteString("\n")
|
||||
}
|
||||
box.WriteString(tuiLabel.Render("setup URL ") + "\n" + tuiPassword.Render(m.setupTokenURL) + "\n\n")
|
||||
box.WriteString(tuiLabel.Render("setup URL ") + "\n")
|
||||
for _, line := range wrapDisplayURL(m.setupTokenURL, 70) {
|
||||
box.WriteString(tuiPassword.Render(line) + "\n")
|
||||
}
|
||||
box.WriteString("\n")
|
||||
box.WriteString(tuiWarn.Render("Open this URL to complete passwordless login setup.\nIt is shown only once."))
|
||||
}
|
||||
if m.auditWarning != "" {
|
||||
|
||||
@@ -151,19 +151,46 @@ func TestOwnerModelProvisionErrorRouting(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a conflict on the Owner path is not a retry", func(t *testing.T) {
|
||||
// Defensive: the Owner upserts and so never conflicts, but were one ever to
|
||||
// surface it must end the session rather than loop the form — only the
|
||||
// insert-only operator path is retryable.
|
||||
t.Run("the Owner seat refusal returns to the form naming the seat", func(t *testing.T) {
|
||||
// Upserting a fresh username while a seat is occupied would mint a second
|
||||
// owner, so provisionOwner refuses with ownerSeatTakenError (Is
|
||||
// api.ErrConflict) and the console must route back for a retype — the same
|
||||
// recoverable contract as the operator clash, and the only Owner-path
|
||||
// conflict there is.
|
||||
m := newOwnerModel(ctx, &fakeOwnerStore{}, "root", true)
|
||||
seatErr := &ownerSeatTakenError{seat: "seat-holder"}
|
||||
|
||||
next, cmd := m.Update(owProvisionMsg{err: conflict})
|
||||
next, cmd := m.Update(owProvisionMsg{err: seatErr})
|
||||
om := next.(*ownerModel)
|
||||
if om.step != owProvision {
|
||||
t.Fatalf("step = %v, want owProvision — the seat refusal is recoverable", om.step)
|
||||
}
|
||||
if om.provisionErr == nil || !errors.Is(om.provisionErr, api.ErrConflict) || !strings.Contains(om.provisionErr.Error(), "seat-holder") {
|
||||
t.Errorf("provisionErr = %v, want the seat refusal naming the seat", om.provisionErr)
|
||||
}
|
||||
// Feed the rebuilt form's init message back through so its view renders;
|
||||
// then the note must carry the seat name (the operator's retype cue).
|
||||
if cmd != nil {
|
||||
if msg := cmd(); msg != nil {
|
||||
if n2, _ := om.Update(msg); n2 != nil {
|
||||
om = n2.(*ownerModel)
|
||||
}
|
||||
}
|
||||
}
|
||||
if view := om.form.View(); !strings.Contains(view, "seat-holder") {
|
||||
t.Errorf("the provision form must surface the seat refusal:\n%s", view)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a generic Owner-path fault still tears the console down", func(t *testing.T) {
|
||||
m := newOwnerModel(ctx, &fakeOwnerStore{}, "root", true)
|
||||
next, cmd := m.Update(owProvisionMsg{err: errors.New("boom")})
|
||||
om := next.(*ownerModel)
|
||||
if om.provisionErr != nil {
|
||||
t.Error("the Owner path recorded a retryable conflict; only the operator path retries")
|
||||
t.Error("a generic fault must not be treated as a retryable refusal")
|
||||
}
|
||||
if res, ok := cmd().(ownerResultMsg); !ok || res.err == nil {
|
||||
t.Error("an Owner-path conflict should tear down via an error result")
|
||||
t.Error("a generic Owner-path fault should tear down via an error result")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
+21
-10
@@ -174,12 +174,13 @@ func (m *ownerModel) Update(msg tea.Msg) (tea.Model, tea.Cmd) {
|
||||
|
||||
case owProvisionMsg:
|
||||
if msg.err != nil {
|
||||
// A taken Operator username is the expected, recoverable outcome of the
|
||||
// insert-only operator path (refusing the clash is the whole reason it is
|
||||
// insert-only, not an upsert). Route back to the form with a note so the
|
||||
// operator can pick another name, rather than tearing down the console —
|
||||
// any other error is a genuine fault and still ends the session.
|
||||
if m.operation == bgAddOperator && errors.Is(msg.err, api.ErrConflict) {
|
||||
// api.ErrConflict marks the two recoverable refusals: a taken Operator
|
||||
// username (insert-only clash) and an Owner reset naming anything but the
|
||||
// occupied seat (ownerSeatTakenError Is ErrConflict). Route back to the
|
||||
// form with a note so the operator can retype, rather than tearing down
|
||||
// the console — any other error is a genuine fault and still ends the
|
||||
// session.
|
||||
if errors.Is(msg.err, api.ErrConflict) {
|
||||
m.provisionErr = msg.err
|
||||
m.step = owProvision
|
||||
m.form = m.sized(m.buildProvisionForm())
|
||||
@@ -364,9 +365,15 @@ func (m *ownerModel) buildProvisionForm() *huh.Form {
|
||||
}
|
||||
}
|
||||
if m.provisionErr != nil {
|
||||
// The only error routed back to this form is a username clash on the insert-only
|
||||
// operator path; show a concrete prompt to choose another name.
|
||||
desc = "That username is already taken — choose a different one.\n\n" + desc
|
||||
// Recoverable refusals routed back here: the seat refusal already names the
|
||||
// username to enter, so show it verbatim; the operator-name clash gets the
|
||||
// generic retry prompt.
|
||||
note := "That username is already taken — choose a different one."
|
||||
var seatErr *ownerSeatTakenError
|
||||
if errors.As(m.provisionErr, &seatErr) {
|
||||
note = seatErr.Error()
|
||||
}
|
||||
desc = note + "\n\n" + desc
|
||||
}
|
||||
|
||||
fields := []huh.Field{
|
||||
@@ -420,7 +427,11 @@ func (m *ownerModel) doneView() string {
|
||||
var box strings.Builder
|
||||
box.WriteString(tuiLabel.Render("username ") + m.username + "\n")
|
||||
if m.setupTokenURL != "" {
|
||||
box.WriteString("\n" + tuiLabel.Render("setup URL ") + "\n" + tuiPassword.Render(m.setupTokenURL) + "\n\n")
|
||||
box.WriteString("\n" + tuiLabel.Render("setup URL ") + "\n")
|
||||
for _, line := range wrapDisplayURL(m.setupTokenURL, 70) {
|
||||
box.WriteString(tuiPassword.Render(line) + "\n")
|
||||
}
|
||||
box.WriteString("\n")
|
||||
box.WriteString(tuiWarn.Render("Open this URL to complete passwordless login setup. It is shown only once."))
|
||||
}
|
||||
if m.auditWarning != "" {
|
||||
|
||||
@@ -553,6 +553,7 @@ func (m *rootModel) showSummary() (tea.Model, tea.Cmd) {
|
||||
storageLabel: m.result.storageDetail,
|
||||
routedHosts: routed,
|
||||
localHint: m.result.connectMethod == connectLocal,
|
||||
alreadySetUp: m.result.alreadySetUp,
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -249,6 +249,34 @@ func TestRootReconfigureSMTP(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestRootReconfigureStorageKeepsStatusFraming locks the same rule for the
|
||||
// "change storage" path: on a re-run, completing it must land back on the
|
||||
// alreadySetUp status framing (with the updated recap), not "Setup complete."
|
||||
func TestRootReconfigureStorageKeepsStatusFraming(t *testing.T) {
|
||||
m := newTestRoot(true, consoleModeSetup, "")
|
||||
m = drive(t, m, preflightDoneMsg{})
|
||||
if _, ok := m.screen.(*summaryModel); !ok {
|
||||
t.Fatalf("re-run after preflight, screen = %T, want *summaryModel", m.screen)
|
||||
}
|
||||
|
||||
m = drive(t, m, reconfigureStorageMsg{})
|
||||
if _, ok := m.screen.(*storageChooserModel); !ok {
|
||||
t.Fatalf("reconfigure-storage screen = %T, want *storageChooserModel", m.screen)
|
||||
}
|
||||
|
||||
m = drive(t, m, storageResultMsg{method: storageLocal, detail: "local disk · /var/lib/felis/uploads"})
|
||||
sum, ok := m.screen.(*summaryModel)
|
||||
if !ok {
|
||||
t.Fatalf("after reconfigure-storage, screen = %T, want *summaryModel", m.screen)
|
||||
}
|
||||
if !sum.alreadySetUp {
|
||||
t.Fatalf("after reconfigure-storage, summary should keep the alreadySetUp framing")
|
||||
}
|
||||
if sum.storageLabel != "local disk · /var/lib/felis/uploads" {
|
||||
t.Fatalf("storageLabel = %q, want the updated recap", sum.storageLabel)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRootRerunLandsOnStatus(t *testing.T) {
|
||||
// adminExists at start of a setup run = re-run: preflight should skip straight
|
||||
// to the "manage in panel" status screen, never touching owner/connect.
|
||||
|
||||
+51
-5
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
@@ -338,20 +339,30 @@ func applySMTPConfig(ctx context.Context, in smtpInputs) error {
|
||||
if err := applyFelisConfigSecret(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
// Refresh the workload-namespace copies too (the reaper's warning path): the
|
||||
// OTP path is already live in the control namespace, so a replica miss is
|
||||
// reported but not fatal.
|
||||
if err := replicateSMTPToWorkloadNamespace(ctx, in.password); err != nil {
|
||||
fmt.Fprintf(os.Stderr, "felis setup: warning: email is configured, but refreshing the workload copies failed (pre-reap warning emails may stay suppressed): %v\n", err)
|
||||
}
|
||||
if err := kubectl(ctx, "-n", "felis", "rollout", "restart", "deployment/felis-api"); err != nil {
|
||||
return err
|
||||
}
|
||||
return kubectl(ctx, "-n", "felis", "rollout", "status", "deployment/felis-api", "--timeout=180s")
|
||||
}
|
||||
|
||||
// applySMTPSecret creates (or replaces) the felis-smtp Secret the felis-api
|
||||
// Deployment injects the relay password from. Rendered in-process and piped to
|
||||
// smtpSecretManifest renders the felis-smtp Secret for the given namespace, the
|
||||
// one the receiving Deployment/CronJob resolves its secretKeyRef against (felis
|
||||
// for felis-api, the workload namespace for the reaper's mirror). The namespace
|
||||
// must be IN the manifest: kubectl rejects a manifest whose namespace conflicts
|
||||
// with -n, so leaving the control namespace hardcoded made every workload-ns
|
||||
// replica fail before it started. Rendered in-process and piped to
|
||||
// `kubectl apply` — the password is never a command-line arg, so it never
|
||||
// appears in the host process table.
|
||||
func applySMTPSecret(ctx context.Context, password string) error {
|
||||
func smtpSecretManifest(password, namespace string) ([]byte, error) {
|
||||
secret := &corev1.Secret{
|
||||
TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "Secret"},
|
||||
ObjectMeta: metav1.ObjectMeta{Name: platform.SMTPSecretName, Namespace: "felis"},
|
||||
ObjectMeta: metav1.ObjectMeta{Name: platform.SMTPSecretName, Namespace: namespace},
|
||||
Type: corev1.SecretTypeOpaque,
|
||||
StringData: map[string]string{
|
||||
platform.SMTPSecretPasswordKey: password,
|
||||
@@ -359,7 +370,42 @@ func applySMTPSecret(ctx context.Context, password string) error {
|
||||
}
|
||||
manifest, err := yaml.Marshal(secret)
|
||||
if err != nil {
|
||||
return fmt.Errorf("render smtp secret: %w", err)
|
||||
return nil, fmt.Errorf("render smtp secret: %w", err)
|
||||
}
|
||||
return manifest, nil
|
||||
}
|
||||
|
||||
func applySMTPSecret(ctx context.Context, password string) error {
|
||||
manifest, err := smtpSecretManifest(password, "felis")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return kubectlWithInput(ctx, manifest, "apply", "-f", "-")
|
||||
}
|
||||
|
||||
// replicateSMTPToWorkloadNamespace refreshes the workload-namespace (minecraft)
|
||||
// copy of felis-smtp after email is reconfigured. The reaper's CronJob runs
|
||||
// there and resolves the password by local reference — a secretKeyRef is
|
||||
// namespace-local — so without this refresh a later SMTP change would never
|
||||
// reach the pre-reap warning emails. Deliberately OVERWRITES: this is a mirror
|
||||
// of the control-namespace source, and a stale mirror is exactly the failure
|
||||
// this closes. The felis-config mirror rides along in applyFelisConfigSecret,
|
||||
// which every apply path refreshes.
|
||||
func replicateSMTPToWorkloadNamespace(ctx context.Context, password string) error {
|
||||
cfg, err := config.Load(hostSetupConfigPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ns := cfg.K8s.Namespace
|
||||
if ns == "" || ns == "felis" {
|
||||
return nil
|
||||
}
|
||||
smtpManifest, err := smtpSecretManifest(password, ns)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := kubectlWithInput(ctx, smtpManifest, "-n", ns, "apply", "-f", "-"); err != nil {
|
||||
return fmt.Errorf("replicate %s to %s: %w", platform.SMTPSecretName, ns, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"sigs.k8s.io/yaml"
|
||||
)
|
||||
|
||||
// TestSMTPSecretManifestCarriesTargetNamespace pins the fix for the
|
||||
// workload-namespace replica: kubectl refuses a manifest whose namespace
|
||||
// conflicts with -n ("the namespace from the provided object ... does not
|
||||
// match"), so the mirror must render felis-smtp with the TARGET namespace —
|
||||
// otherwise the "configure email" refresh fails on the first apply and the
|
||||
// felis-config mirror never runs at all.
|
||||
func TestSMTPSecretManifestCarriesTargetNamespace(t *testing.T) {
|
||||
for _, ns := range []string{"felis", "minecraft"} {
|
||||
b, err := smtpSecretManifest("pw", ns)
|
||||
if err != nil {
|
||||
t.Fatalf("render for %s: %v", ns, err)
|
||||
}
|
||||
var got struct {
|
||||
Metadata struct {
|
||||
Namespace string `json:"namespace"`
|
||||
} `json:"metadata"`
|
||||
}
|
||||
if err := yaml.Unmarshal(b, &got); err != nil {
|
||||
t.Fatalf("unmarshal for %s: %v", ns, err)
|
||||
}
|
||||
if got.Metadata.Namespace != ns {
|
||||
t.Fatalf("manifest namespace = %q, want %q", got.Metadata.Namespace, ns)
|
||||
}
|
||||
if !strings.Contains(string(b), "name: felis-smtp") {
|
||||
t.Fatalf("manifest must still name felis-smtp: %s", b)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -29,6 +29,33 @@ func tuiSeparator() string {
|
||||
return tuiHint.Render(strings.Repeat("─", 70))
|
||||
}
|
||||
|
||||
// wrapDisplayURL breaks a long URL into lines no wider than width so the TUI
|
||||
// renderer never truncates it on a narrow terminal — the one-time setup URL
|
||||
// carries a 43-char token and overruns 80 columns. It prefers breaking right
|
||||
// after a '=' or '/' inside the window (the token then lands on its own line)
|
||||
// and hard-wraps only when no boundary is available. Lines concatenate back to
|
||||
// the original string.
|
||||
func wrapDisplayURL(u string, width int) []string {
|
||||
if width <= 0 {
|
||||
width = 70
|
||||
}
|
||||
var lines []string
|
||||
for len(u) > width {
|
||||
cut := width
|
||||
if i := strings.LastIndexByte(u[:width], '='); i >= 0 && i >= width/2 {
|
||||
cut = i + 1
|
||||
} else if i := strings.LastIndexByte(u[:width], '/'); i >= 0 && i >= width/2 {
|
||||
cut = i + 1
|
||||
}
|
||||
lines = append(lines, u[:cut])
|
||||
u = u[cut:]
|
||||
}
|
||||
if u != "" {
|
||||
lines = append(lines, u)
|
||||
}
|
||||
return lines
|
||||
}
|
||||
|
||||
// tuiStepRail renders a breadcrumb of wizard stages. Steps before `current`
|
||||
// render as done, `current` is highlighted, and later steps are dimmed.
|
||||
func tuiStepRail(steps []string, current int) string {
|
||||
|
||||
@@ -0,0 +1,26 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestWrapDisplayURL(t *testing.T) {
|
||||
u := "https://op.console.example.net/setup?token=" + strings.Repeat("A", 43)
|
||||
lines := wrapDisplayURL(u, 70)
|
||||
if got := strings.Join(lines, ""); got != u {
|
||||
t.Fatalf("concatenated lines = %q, want the original URL back", got)
|
||||
}
|
||||
for i, l := range lines {
|
||||
if len(l) > 70 {
|
||||
t.Errorf("line %d is %d cols wide: %q", i, len(l), l)
|
||||
}
|
||||
}
|
||||
if len(lines) < 2 || !strings.HasSuffix(lines[0], "token=") {
|
||||
t.Fatalf("want the first line to end at the 'token=' boundary, got %q", lines)
|
||||
}
|
||||
short := "https://a/b"
|
||||
if got := wrapDisplayURL(short, 70); len(got) != 1 || got[0] != short {
|
||||
t.Errorf("short URL should pass through unsplit, got %q", got)
|
||||
}
|
||||
}
|
||||
+29
-25
@@ -24,8 +24,9 @@ const updateTimeout = 60 * time.Second
|
||||
// The apply side is deliberately NOT implemented in this command. Every component
|
||||
// here is installed by deploy/bootstrap.sh, which is idempotent, already handles the
|
||||
// parts that are easy to get wrong (Velocity's pinned MINOR, the atomic jar install,
|
||||
// the k3s image re-import that a byte-identical StatefulSet template will not
|
||||
// trigger on its own), and is the path that gets exercised on every install. A
|
||||
// the image re-import + registry push that a byte-identical StatefulSet template
|
||||
// will not trigger on its own), and is the path that gets exercised on every
|
||||
// install. A
|
||||
// second installer living in this file would duplicate that policy, could drift from
|
||||
// it silently, and would be reachable only on a live node where a mistake takes the
|
||||
// proxy or the control plane down. So `felis update` reports, and hands the operator
|
||||
@@ -42,6 +43,18 @@ type updateTarget struct {
|
||||
command string
|
||||
}
|
||||
|
||||
// installerRerun is the tested apply path for every planner-backed selector: re-run the
|
||||
// installer. It is idempotent, and it is the only path that fetches a newer version --
|
||||
// `felis setup` skips its host-bootstrap phase on a completed install (all four install
|
||||
// markers already exist), so there it opens the config console and moves no component,
|
||||
// and even on the bootstrap path it re-images felis-api from the binary setup is already
|
||||
// running (FELIS_BOOTSTRAP_BINARY), which looks like an update and changes nothing.
|
||||
//
|
||||
// The URL is the same one-liner both READMEs hand out. While the repo is private it
|
||||
// answers 404 (raw.githubusercontent.com hides private repos), which is why the trailer
|
||||
// below points at the README's token'd form for that case.
|
||||
const installerRerun = "curl -fsSL https://raw.githubusercontent.com/FelisMC/Felis/main/deploy/bootstrap.sh | sudo bash"
|
||||
|
||||
// updateTargets is the selector table. panel and plugins both resolve to felis-api
|
||||
// because they are not separately versioned: the panel is compiled into the felis
|
||||
// binary with //go:embed, and the plugin jars are built from this same repo in the
|
||||
@@ -51,19 +64,19 @@ var updateTargets = []updateTarget{
|
||||
selector: "panel",
|
||||
component: "felis-api",
|
||||
note: "the panel is embedded in the felis binary (//go:embed), so updating it means rebuilding the felis image and rolling felis-api",
|
||||
command: "sudo felis setup",
|
||||
command: installerRerun,
|
||||
},
|
||||
{
|
||||
selector: "velocity",
|
||||
component: "velocity",
|
||||
note: "re-runs install_velocity: newest BUILD of the pinned minor (FELIS_VELOCITY_VERSION), atomic jar install, then restarts felis-velocity",
|
||||
command: "sudo felis setup",
|
||||
command: installerRerun,
|
||||
},
|
||||
{
|
||||
selector: "plugins",
|
||||
component: "felis-api",
|
||||
note: "felis-velocity.jar is a host-file swap, but felis-paper.jar and felis-limbo.jar are baked into the lobby/limbo images and need a rebuild + k3s image re-import",
|
||||
command: "sudo felis setup",
|
||||
note: "felis-velocity.jar is a host-file swap, but felis-paper.jar and felis-limbo.jar are baked into the lobby/limbo images and need a rebuild + re-mirror into the in-cluster registry (the installer re-run does both)",
|
||||
command: installerRerun,
|
||||
},
|
||||
{
|
||||
selector: "mc",
|
||||
@@ -219,7 +232,6 @@ func renderApplyGuidance(res updater.Result, selected map[string]bool, force boo
|
||||
|
||||
var b strings.Builder
|
||||
var offeredCommand bool
|
||||
var offeredFelisAPI bool
|
||||
for _, t := range updateTargets {
|
||||
if !selected[t.selector] {
|
||||
continue
|
||||
@@ -248,28 +260,20 @@ func renderApplyGuidance(res updater.Result, selected map[string]bool, force boo
|
||||
}
|
||||
fmt.Fprintf(&b, " run: %s\n", t.command)
|
||||
offeredCommand = true
|
||||
offeredFelisAPI = offeredFelisAPI || t.component == "felis-api"
|
||||
}
|
||||
// Only explain the command when one was actually offered; a --mc-only run has
|
||||
// nothing to run and the trailer would be a non-sequitur.
|
||||
if offeredCommand {
|
||||
b.WriteString("\nfelis setup is idempotent and re-runs the installer that owns these components;\nit does not reinstall what is already current. Restart game servers afterwards.\n")
|
||||
}
|
||||
// Scoped to felis-api because it is the only component setup cannot move forward.
|
||||
// velocity is fine: install_velocity re-resolves the newest build of the pinned minor
|
||||
// on every run. But setup hands deploy/bootstrap.sh the binary it is itself running
|
||||
// (FELIS_BOOTSTRAP_BINARY), and that arm skips the release lookup entirely, so it
|
||||
// rebuilds the image and rolls the deployment from the SAME binary -- a run that looks
|
||||
// like a successful update and leaves the version unchanged.
|
||||
//
|
||||
// The installer is the only thing that moves felis-api. It is safe to point at now
|
||||
// that detect_node_ip reuses the installed root domain, so what is left to warn about
|
||||
// is the channel: FELIS_VERSION_BOOTSTRAP is not persisted anywhere and defaults to
|
||||
// release, so a bare re-run on a host tracking main quietly moves it onto releases.
|
||||
// That is a channel change, not a broken install, which is why it is one clause and
|
||||
// not a paragraph.
|
||||
if offeredFelisAPI {
|
||||
b.WriteString("\nfelis-api (panel, plugins) is the exception: setup re-images it from the felis binary\nalready on this host, so it cannot install a NEWER felis-api. Re-run the bootstrap\ninstaller for that -- it keeps this install's root domain. It does default to the\nrelease channel, so pass FELIS_VERSION_BOOTSTRAP=dev if this host tracks main.\n")
|
||||
// One trailer serves every selector now: setup is not an apply path at all on a
|
||||
// completed install (shouldRunHostBootstrapBeforeConfig only enters the host
|
||||
// bootstrap while an install marker is missing), so the installer re-run is the one
|
||||
// worked path for all three components and there is no per-component exception left
|
||||
// to scope. Two caveats stay because following the advice without them bites real
|
||||
// hosts: the channel is not persisted anywhere (a bare re-run on a main host quietly
|
||||
// moves it onto releases), and the private repo's one-liner needs the read token
|
||||
// back in the environment before it can resolve anything.
|
||||
if offeredCommand {
|
||||
b.WriteString("\nRe-running the installer applies everything above: it fetches the newest version on\nthe channel in effect and re-applies the bundle (release is the default). The channel\nis not persisted, so pass FELIS_VERSION_BOOTSTRAP=dev if this host tracks main. While\nthis repo is private, the one-liner above 404s without a token; the README's install\nsection has the token'd form that works. felis setup is not this path: on a completed\ninstall it opens the config console and installs nothing newer. Restart game servers\nafterwards.\n")
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
+20
-25
@@ -107,7 +107,7 @@ func TestApplyGuidanceMinecraftOffersNoCommand(t *testing.T) {
|
||||
if !strings.Contains(out, "pinned by policy") {
|
||||
t.Fatalf("want the pin explained:\n%s", out)
|
||||
}
|
||||
if strings.Contains(out, "run:") || strings.Contains(out, "felis setup is idempotent") {
|
||||
if strings.Contains(out, "run:") || strings.Contains(out, "Re-running the installer") {
|
||||
t.Fatalf("--mc must offer no command and no command trailer:\n%s", out)
|
||||
}
|
||||
}
|
||||
@@ -133,42 +133,37 @@ func TestUpdateTargetsMatchTopology(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// `sudo felis setup` is the right answer for velocity and the wrong one for felis-api,
|
||||
// so the caveat has to be scoped rather than appended to every run. setup hands
|
||||
// bootstrap the binary it is already running, and that arm skips the release lookup:
|
||||
// the run rebuilds the image and rolls the deployment off the SAME binary, which looks
|
||||
// like a successful update and changes nothing. install_velocity, by contrast, really
|
||||
// does re-resolve the newest build on every run.
|
||||
func TestApplyGuidanceScopesTheFelisAPICaveat(t *testing.T) {
|
||||
const caveat = "cannot install a NEWER felis-api"
|
||||
|
||||
// Re-running the installer is the one apply path this table may hand out. setup is NOT an
|
||||
// updater on a completed install -- its host-bootstrap phase only runs while an install
|
||||
// marker is missing, so it opens the config console and moves no component -- and even on
|
||||
// the bootstrap path it re-images felis-api from the binary setup is already running. The
|
||||
// table used to answer with "sudo felis setup" and scope a felis-api-only exception; both
|
||||
// taught a model that does not survive contact with an installed host.
|
||||
func TestApplyGuidancePointsEveryComponentAtTheInstaller(t *testing.T) {
|
||||
api := renderApplyGuidance(
|
||||
planResult([]updates.Action{{Component: "felis-api", Kind: updates.ActionNotify, LatestKnown: true}}),
|
||||
map[string]bool{"panel": true}, false)
|
||||
if !strings.Contains(api, caveat) {
|
||||
t.Fatalf("--panel resolves to felis-api and must carry the caveat:\n%s", api)
|
||||
for _, want := range []string{"deploy/bootstrap.sh", "FELIS_VERSION_BOOTSTRAP=dev", "felis setup is not this path"} {
|
||||
if !strings.Contains(api, want) {
|
||||
t.Fatalf("--panel guidance missing %q:\n%s", want, api)
|
||||
}
|
||||
}
|
||||
// Naming the installer obliges us to name what a bare re-run still changes. The domain
|
||||
// is handled -- detect_node_ip reuses the installed one -- but the channel is not
|
||||
// persisted at all and defaults to release, so a host tracking main gets moved onto
|
||||
// releases by following this advice.
|
||||
if !strings.Contains(api, "FELIS_VERSION_BOOTSTRAP=dev") {
|
||||
t.Fatalf("pointing at the installer without the channel caveat misleads a dev host:\n%s", api)
|
||||
if strings.Contains(api, "run: sudo felis setup") {
|
||||
t.Fatalf("setup must never be offered as the apply command:\n%s", api)
|
||||
}
|
||||
|
||||
// The same path serves velocity; a scoped caveat would re-teach the old model that
|
||||
// setup fixes velocity.
|
||||
vel := renderApplyGuidance(
|
||||
planResult([]updates.Action{{Component: "velocity", Kind: updates.ActionNotify, LatestKnown: true}}),
|
||||
map[string]bool{"velocity": true}, false)
|
||||
if strings.Contains(vel, caveat) {
|
||||
t.Fatalf("velocity IS fixed by setup; the caveat would misdirect the operator:\n%s", vel)
|
||||
}
|
||||
if !strings.Contains(vel, "felis setup is idempotent") {
|
||||
t.Fatalf("velocity still wants the ordinary trailer:\n%s", vel)
|
||||
if !strings.Contains(vel, "run: curl -fsSL") || !strings.Contains(vel, "felis setup is not this path") {
|
||||
t.Fatalf("velocity gets the same installer path:\n%s", vel)
|
||||
}
|
||||
|
||||
// --mc offers no command at all, so neither trailer belongs.
|
||||
mc := renderApplyGuidance(planResult(nil), map[string]bool{"mc": true}, true)
|
||||
if strings.Contains(mc, caveat) {
|
||||
t.Fatalf("--mc offers no command; the caveat is a non-sequitur:\n%s", mc)
|
||||
if strings.Contains(mc, "deploy/bootstrap.sh") || strings.Contains(mc, "FELIS_VERSION_BOOTSTRAP") {
|
||||
t.Fatalf("--mc offers no command; the trailer is a non-sequitur:\n%s", mc)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
# Felis alert rules — plain Prometheus format (also the promtool-tested source
|
||||
# for felis-prometheusrule.yaml). See docs/troubleshooting.md §14 for scraping
|
||||
# and loading instructions.
|
||||
#
|
||||
# felis_* series come from two processes:
|
||||
# - felis-operator pod :8080/metrics → felis_servers_total, felis_start_duration_seconds
|
||||
# - felis-api internal :8081/metrics → felis_image_build_failures_total
|
||||
# node_* / kube_* series come from node-exporter / kube-state-metrics.
|
||||
groups:
|
||||
- name: felis.rules
|
||||
rules:
|
||||
- alert: FelisImageBuildFailures
|
||||
expr: increase(felis_image_build_failures_total[6h]) > 0
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "modpack/image build failed in the last 6h"
|
||||
description: >-
|
||||
felis_image_build_failures_total increased. Inspect the failed build Job
|
||||
(kubectl logs -n felis-build job/<build-job>); the same error text is on
|
||||
GET /api/v1/images/build/{id} and in the submitter's row in the panel.
|
||||
- alert: FelisSlowServerStarts
|
||||
expr: histogram_quantile(0.9, sum by (le) (rate(felis_start_duration_seconds_bucket[30m]))) > 300
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "p90 server start time exceeds 5 minutes"
|
||||
description: >-
|
||||
Starts regularly take over five minutes (felis_start_duration_seconds,
|
||||
observed when readiness is first reached). A start that never completes
|
||||
records nothing — cross-check desiredState=Running servers with no ready
|
||||
phase (troubleshooting §1).
|
||||
- name: felis.node.rules
|
||||
rules:
|
||||
- alert: FelisNodeDiskSpaceLow
|
||||
expr: >-
|
||||
node_filesystem_avail_bytes{fstype=~"ext4|xfs|btrfs"}
|
||||
/ node_filesystem_size_bytes{fstype=~"ext4|xfs|btrfs"} < 0.15
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "node filesystem {{ $labels.mountpoint }} below 15% available"
|
||||
description: >-
|
||||
Sustained disk pressure evicts game pods and garbage-collects images
|
||||
(troubleshooting §13b). Free space before kubelet raises DiskPressure.
|
||||
- alert: FelisNodeDiskPressure
|
||||
expr: kube_node_status_condition{condition="DiskPressure",status="true"} == 1
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "kubelet reports DiskPressure on {{ $labels.node }}"
|
||||
description: >-
|
||||
The eviction chain is in progress: control-plane pods hold
|
||||
system-cluster-critical and survive, game pods do not. Free disk now
|
||||
(troubleshooting §13b).
|
||||
- alert: FelisNodeMemoryLow
|
||||
expr: node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes < 0.10
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "node memory available below 10% for 15m"
|
||||
description: >-
|
||||
PostgreSQL, the control plane, the registry and game servers share one
|
||||
node; sustained memory pressure risks OOM kills.
|
||||
@@ -0,0 +1,107 @@
|
||||
# promtool unit tests: `promtool test rules felis-alerts_test.yml`
|
||||
# Proves every shipped rule actually fires on its target condition (and stays
|
||||
# silent before it).
|
||||
rule_files:
|
||||
- felis-alerts.yaml
|
||||
evaluation_interval: 1m
|
||||
tests:
|
||||
- name: build failure and slow starts
|
||||
interval: 1m
|
||||
input_series:
|
||||
# counter: quiet for 5m, then one failure per step.
|
||||
- series: 'felis_image_build_failures_total'
|
||||
values: '0x5 1x15'
|
||||
# histogram: all observations land in the (300,600] bucket.
|
||||
- series: 'felis_start_duration_seconds_bucket{le="120"}'
|
||||
values: '0x22'
|
||||
- series: 'felis_start_duration_seconds_bucket{le="300"}'
|
||||
values: '0x22'
|
||||
- series: 'felis_start_duration_seconds_bucket{le="600"}'
|
||||
values: '0+10x21'
|
||||
- series: 'felis_start_duration_seconds_bucket{le="+Inf"}'
|
||||
values: '0+10x21'
|
||||
alert_rule_test:
|
||||
- eval_time: 2m
|
||||
alertname: FelisImageBuildFailures
|
||||
exp_alerts: []
|
||||
- eval_time: 20m
|
||||
alertname: FelisImageBuildFailures
|
||||
exp_alerts:
|
||||
- exp_labels:
|
||||
severity: warning
|
||||
exp_annotations:
|
||||
summary: "modpack/image build failed in the last 6h"
|
||||
description: >-
|
||||
felis_image_build_failures_total increased. Inspect the failed build Job
|
||||
(kubectl logs -n felis-build job/<build-job>); the same error text is on
|
||||
GET /api/v1/images/build/{id} and in the submitter's row in the panel.
|
||||
- eval_time: 20m
|
||||
alertname: FelisSlowServerStarts
|
||||
exp_alerts:
|
||||
- exp_labels:
|
||||
severity: warning
|
||||
exp_annotations:
|
||||
summary: "p90 server start time exceeds 5 minutes"
|
||||
description: >-
|
||||
Starts regularly take over five minutes (felis_start_duration_seconds,
|
||||
observed when readiness is first reached). A start that never completes
|
||||
records nothing — cross-check desiredState=Running servers with no ready
|
||||
phase (troubleshooting §1).
|
||||
- name: node disk and memory thresholds
|
||||
interval: 1m
|
||||
input_series:
|
||||
- series: 'node_filesystem_avail_bytes{device="/dev/vda1",fstype="xfs",instance="node1",job="node-exporter",mountpoint="/"}'
|
||||
values: '10x26'
|
||||
- series: 'node_filesystem_size_bytes{device="/dev/vda1",fstype="xfs",instance="node1",job="node-exporter",mountpoint="/"}'
|
||||
values: '100x26'
|
||||
- series: 'kube_node_status_condition{condition="DiskPressure",node="n1",status="true"}'
|
||||
values: '0x4 1x22'
|
||||
- series: 'node_memory_MemAvailable_bytes{instance="node1",job="node-exporter"}'
|
||||
values: '5x26'
|
||||
- series: 'node_memory_MemTotal_bytes{instance="node1",job="node-exporter"}'
|
||||
values: '100x26'
|
||||
alert_rule_test:
|
||||
- eval_time: 2m
|
||||
alertname: FelisNodeDiskPressure
|
||||
exp_alerts: []
|
||||
- eval_time: 20m
|
||||
alertname: FelisNodeDiskSpaceLow
|
||||
exp_alerts:
|
||||
- exp_labels:
|
||||
device: /dev/vda1
|
||||
fstype: xfs
|
||||
instance: node1
|
||||
job: node-exporter
|
||||
mountpoint: /
|
||||
severity: warning
|
||||
exp_annotations:
|
||||
summary: "node filesystem / below 15% available"
|
||||
description: >-
|
||||
Sustained disk pressure evicts game pods and garbage-collects images
|
||||
(troubleshooting §13b). Free space before kubelet raises DiskPressure.
|
||||
- eval_time: 20m
|
||||
alertname: FelisNodeDiskPressure
|
||||
exp_alerts:
|
||||
- exp_labels:
|
||||
condition: DiskPressure
|
||||
node: n1
|
||||
status: "true"
|
||||
severity: critical
|
||||
exp_annotations:
|
||||
summary: "kubelet reports DiskPressure on n1"
|
||||
description: >-
|
||||
The eviction chain is in progress: control-plane pods hold
|
||||
system-cluster-critical and survive, game pods do not. Free disk now
|
||||
(troubleshooting §13b).
|
||||
- eval_time: 20m
|
||||
alertname: FelisNodeMemoryLow
|
||||
exp_alerts:
|
||||
- exp_labels:
|
||||
instance: node1
|
||||
job: node-exporter
|
||||
severity: warning
|
||||
exp_annotations:
|
||||
summary: "node memory available below 10% for 15m"
|
||||
description: >-
|
||||
PostgreSQL, the control plane, the registry and game servers share one
|
||||
node; sustained memory pressure risks OOM kills.
|
||||
@@ -0,0 +1,74 @@
|
||||
# prometheus-operator twin of felis-alerts.yaml (kube-prometheus-stack loads
|
||||
# rules through the PrometheusRule CRD, not rule_files). The plain file is the
|
||||
# promtool-tested source; keep the groups in sync.
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: PrometheusRule
|
||||
metadata:
|
||||
name: felis-alerts
|
||||
namespace: monitoring
|
||||
labels:
|
||||
# Change to match your stack's ruleSelector (kube-prometheus-stack's
|
||||
# default selects on the Helm release name).
|
||||
release: kube-prometheus-stack
|
||||
spec:
|
||||
groups:
|
||||
- name: felis.rules
|
||||
rules:
|
||||
- alert: FelisImageBuildFailures
|
||||
expr: increase(felis_image_build_failures_total[6h]) > 0
|
||||
for: 5m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "modpack/image build failed in the last 6h"
|
||||
description: >-
|
||||
felis_image_build_failures_total increased. Inspect the failed build Job
|
||||
(kubectl logs -n felis-build job/<build-job>); the same error text is on
|
||||
GET /api/v1/images/build/{id} and in the submitter's row in the panel.
|
||||
- alert: FelisSlowServerStarts
|
||||
expr: histogram_quantile(0.9, sum by (le) (rate(felis_start_duration_seconds_bucket[30m]))) > 300
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "p90 server start time exceeds 5 minutes"
|
||||
description: >-
|
||||
Starts regularly take over five minutes (felis_start_duration_seconds,
|
||||
observed when readiness is first reached). A start that never completes
|
||||
records nothing — cross-check desiredState=Running servers with no ready
|
||||
phase (troubleshooting §1).
|
||||
- name: felis.node.rules
|
||||
rules:
|
||||
- alert: FelisNodeDiskSpaceLow
|
||||
expr: >-
|
||||
node_filesystem_avail_bytes{fstype=~"ext4|xfs|btrfs"}
|
||||
/ node_filesystem_size_bytes{fstype=~"ext4|xfs|btrfs"} < 0.15
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "node filesystem {{ $labels.mountpoint }} below 15% available"
|
||||
description: >-
|
||||
Sustained disk pressure evicts game pods and garbage-collects images
|
||||
(troubleshooting §13b). Free space before kubelet raises DiskPressure.
|
||||
- alert: FelisNodeDiskPressure
|
||||
expr: kube_node_status_condition{condition="DiskPressure",status="true"} == 1
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: "kubelet reports DiskPressure on {{ $labels.node }}"
|
||||
description: >-
|
||||
The eviction chain is in progress: control-plane pods hold
|
||||
system-cluster-critical and survive, game pods do not. Free disk now
|
||||
(troubleshooting §13b).
|
||||
- alert: FelisNodeMemoryLow
|
||||
expr: node_memory_MemAvailable_bytes / node_memory_MemTotal_bytes < 0.10
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: "node memory available below 10% for 15m"
|
||||
description: >-
|
||||
PostgreSQL, the control plane, the registry and game servers share one
|
||||
node; sustained memory pressure risks OOM kills.
|
||||
+324
-43
@@ -59,10 +59,24 @@
|
||||
# FELIS_GITHUB_TOKEN GitHub token; REQUIRED while the repo is private
|
||||
# FELIS_REF branch/tag/sha — pins the build, overrides the channel, and forces a
|
||||
# source build (naming a ref asks for that tree, not a published asset)
|
||||
# FELIS_IMAGE local image tag (default: felis:demo — never :latest)
|
||||
# FELIS_IMAGE control-plane image ref (default:
|
||||
# registry.felis.svc:5000/felis/felis:demo — never :latest;
|
||||
# anything not under the registry is used as-is but is NOT
|
||||
# mirrored into it, so it has no pull source after an image GC)
|
||||
# FELIS_ROOT_DOMAIN deployment root domain (default: <node-ip>.nip.io)
|
||||
# FELIS_PANEL_NODEPORT local HTTPS panel/API NodePort (default: 30443)
|
||||
# FELIS_EGRESS_MODE loadbalancer|nodeport (default: nodeport — no MetalLB on a demo box)
|
||||
# FELIS_BACKUP_PVC world-archive PVC the installer renders and felis-api hands to its
|
||||
# backup/restore Jobs (default: felis-backups; empty string disables
|
||||
# backups — the endpoints answer 503)
|
||||
# FELIS_ARCHIVE_LOCAL_PATH path that PVC is mounted at inside those Jobs; written into
|
||||
# felis.toml [archive] local_path (default: /var/lib/felis/archives)
|
||||
# FELIS_WORLDS_HOST_PATH node directory holding the world volumes (on the k3s this
|
||||
# installer provisions: /var/lib/rancher/k3s/storage). Setting it
|
||||
# enables the daily retention reaper, which archives and then deletes
|
||||
# worlds idle beyond the retention window, and grants the reaper's
|
||||
# uid (1000) traverse access to that root — k3s ships it 0700
|
||||
# root:root (default: unset = no reaper)
|
||||
# PKG_LOCK_TIMEOUT seconds to wait for package-manager locks (default: 900)
|
||||
# APT_LOCK_TIMEOUT legacy alias for PKG_LOCK_TIMEOUT
|
||||
set -Eeuo pipefail
|
||||
@@ -70,7 +84,7 @@ set -Eeuo pipefail
|
||||
# ---------------------------------------------------------------------------
|
||||
# Configuration & constants
|
||||
# ---------------------------------------------------------------------------
|
||||
FELIS_REPO_URL="${FELIS_REPO_URL:-https://github.com/MliroLirrorsIngenuity/Felis.git}"
|
||||
FELIS_REPO_URL="${FELIS_REPO_URL:-https://github.com/FelisMC/Felis.git}"
|
||||
# Which version to install. "release" builds the newest published GitHub release;
|
||||
# "dev" builds the tip of main. Release is the default because an installer that
|
||||
# tracks a moving branch by default hands every new host a different, untested
|
||||
@@ -103,9 +117,29 @@ export FELIS_GITHUB_TOKEN
|
||||
# Set by resolve_install_ref/stamp_version and linked into the binary as main.version.
|
||||
FELIS_VERSION=""
|
||||
FELIS_VERSION_BASE=""
|
||||
FELIS_IMAGE="${FELIS_IMAGE:-felis:demo}"
|
||||
# Where the installer parks every image it builds, so kubelet can re-pull one the
|
||||
# image GC has collected (the disk-pressure drill's dead end: ImagePullBackOff with
|
||||
# nothing to pull from). The node pulls through the loopback hostPort the registry
|
||||
# Deployment binds (configure_registry_mirror below); pushes go through
|
||||
# REGISTRY_PUSH_HOST — docker treats 127.0.0.1 as insecure by default, so the
|
||||
# daemon needs no insecure-registries entry for it.
|
||||
REGISTRY_URL="registry.felis.svc:5000"
|
||||
REGISTRY_PUSH_HOST="127.0.0.1:${REGISTRY_URL##*:}"
|
||||
FELIS_IMAGE="${FELIS_IMAGE:-${REGISTRY_URL}/felis/felis:demo}"
|
||||
FELIS_EGRESS_MODE="${FELIS_EGRESS_MODE:-nodeport}"
|
||||
FELIS_PANEL_NODEPORT="${FELIS_PANEL_NODEPORT:-30443}"
|
||||
# World-archive storage. The installer renders this PVC (minecraft namespace) and felis-api
|
||||
# advertises it to its backup/restore Jobs; emptying it disables backups (503). The archive
|
||||
# path is written into felis.toml so the Jobs' mount and [archive] local_path agree by
|
||||
# construction — a mismatch would leave tarLocal's absolute archive refs unresolvable.
|
||||
FELIS_BACKUP_PVC="${FELIS_BACKUP_PVC:-felis-backups}"
|
||||
FELIS_ARCHIVE_LOCAL_PATH="${FELIS_ARCHIVE_LOCAL_PATH:-/var/lib/felis/archives}"
|
||||
# Retention is opt-in because it DELETES worlds (after a verified archive): point this at the
|
||||
# node directory the world volumes live under. On the k3s this installer provisions that is
|
||||
# /var/lib/rancher/k3s/storage — the reaper resolves each PVC's local-path directory exactly
|
||||
# from its volumeName. Left unset, no reaper CronJob renders and archives accumulate until
|
||||
# the backup PVC fills (then backups fail loudly; nothing is deleted).
|
||||
FELIS_WORLDS_HOST_PATH="${FELIS_WORLDS_HOST_PATH:-}"
|
||||
INSTALL_MODE="${FELIS_INSTALL_MODE:-}"
|
||||
# Loopback by default: hasJoined is an unauthenticated endpoint by protocol (Velocity
|
||||
# sends no token), so a public bind is a free auth relay — anyone can point their own
|
||||
@@ -133,12 +167,12 @@ PKG_LOCK_TIMEOUT="${PKG_LOCK_TIMEOUT:-${APT_LOCK_TIMEOUT:-900}}"
|
||||
APT_LOCK_TIMEOUT="${APT_LOCK_TIMEOUT:-$PKG_LOCK_TIMEOUT}"
|
||||
|
||||
# --- the game stack: proxy on the host, the two always-on backends in k3s ---
|
||||
FELIS_LIMBO_IMAGE="${FELIS_LIMBO_IMAGE:-felis-limbo:demo}"
|
||||
FELIS_LOBBY_IMAGE="${FELIS_LOBBY_IMAGE:-felis-lobby:demo}"
|
||||
FELIS_LIMBO_IMAGE="${FELIS_LIMBO_IMAGE:-${REGISTRY_URL}/felis/limbo:demo}"
|
||||
FELIS_LOBBY_IMAGE="${FELIS_LOBBY_IMAGE:-${REGISTRY_URL}/felis/lobby:demo}"
|
||||
# Plain Paper base recommended for a user's own server (deploy/paper). Not a system
|
||||
# server — forwarding is applied by the operator's init-forwarding initContainer, so it
|
||||
# needs no secret. Seeded recommended in 0019_recommended_paper.sql.
|
||||
FELIS_PAPER_IMAGE="${FELIS_PAPER_IMAGE:-felis-paper:demo}"
|
||||
FELIS_PAPER_IMAGE="${FELIS_PAPER_IMAGE:-${REGISTRY_URL}/felis/paper:demo}"
|
||||
# The Velocity MINOR is pinned, not discovered. PaperMC's Fill v3 groups velocity
|
||||
# builds by version group, and "newest across all groups" today means 4.0.0-SNAPSHOT —
|
||||
# an UNRELEASED proxy (the 4.0.0 group has zero published builds) that needs a Java 25
|
||||
@@ -191,7 +225,6 @@ POD_CIDR="10.42.0.0/16" # k3s default cluster CIDR
|
||||
SERVICE_CIDR="10.43.0.0/16" # k3s default service CIDR
|
||||
DB_NAME="felis"
|
||||
DB_USER="felis"
|
||||
REGISTRY_URL="registry.felis.svc:5000"
|
||||
|
||||
STATE_DIR="/etc/felis"
|
||||
SECRETS_ENV="${STATE_DIR}/secrets.env"
|
||||
@@ -210,6 +243,10 @@ VELOCITY_SERVICE="/etc/systemd/system/felis-velocity.service"
|
||||
JRE_DIR="/opt/felis/jre"
|
||||
K3S_BIN_DIR="${K3S_BIN_DIR:-/usr/local/bin}"
|
||||
K3S_BIN="${K3S_BIN_DIR}/k3s"
|
||||
# k3s's containerd mirror config, written by configure_registry_mirror. A variable
|
||||
# (not just the literal path) so bootstrap_test.sh can point the writer at a
|
||||
# scratch file.
|
||||
K3S_REGISTRIES_FILE="/etc/rancher/k3s/registries.yaml"
|
||||
APT_LOCK_FILES=(
|
||||
/var/lib/dpkg/lock-frontend
|
||||
/var/lib/dpkg/lock
|
||||
@@ -307,6 +344,21 @@ apply_literal_secret() {
|
||||
rm -f "$tmp"
|
||||
}
|
||||
|
||||
# The control plane mounts felis-config from its own namespace; the workload
|
||||
# namespace's backup/restore/fileedit Jobs and the reaper mount a local copy (a
|
||||
# secretKeyRef is namespace-local). The installer owns the rendered config, so both
|
||||
# copies are (re)applied on every run — unlike the create-if-absent credential
|
||||
# replicas `felis setup` makes, because a stale config copy keeps an old database URL
|
||||
# or archive policy after an upgrade or a credential rotation.
|
||||
apply_felis_config_secrets() {
|
||||
kube -n "$CONTROL_NS" create secret generic felis-config \
|
||||
--from-file=felis.toml="${STATE_DIR}/felis.pod.toml" \
|
||||
--dry-run=client -o yaml | kube apply -f -
|
||||
kube -n "$MINECRAFT_NS" create secret generic felis-config \
|
||||
--from-file=felis.toml="${STATE_DIR}/felis.pod.toml" \
|
||||
--dry-run=client -o yaml | kube apply -f -
|
||||
}
|
||||
|
||||
as_postgres() {
|
||||
if command -v runuser >/dev/null 2>&1; then
|
||||
runuser -u postgres -- "$@"
|
||||
@@ -790,6 +842,13 @@ install_k3s() {
|
||||
systemctl enable --now k3s
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
log "waiting for the node to become Ready"
|
||||
wait_for_node_ready
|
||||
}
|
||||
|
||||
# Waits for the (single) node to report Ready. Shared by the k3s install and the
|
||||
# registry-mirror restart below: both restart the agent, and a bootstrap that
|
||||
# proceeds early fails later with a misleading "not found"/timeout instead.
|
||||
wait_for_node_ready() {
|
||||
local i
|
||||
for i in $(seq 1 60); do
|
||||
if kube get nodes 2>/dev/null | grep -q ' Ready '; then
|
||||
@@ -802,6 +861,78 @@ install_k3s() {
|
||||
die "k3s node did not become Ready in time"
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4b. The in-cluster registry: the node-side pull path, the registry's own
|
||||
# image, and the hosting of every image this installer builds.
|
||||
#
|
||||
# Kubelet's image GC collects an unused image under disk pressure (drilled:
|
||||
# the game images were collected and ImagePullBackOff had nothing to pull
|
||||
# from). The fix is a pull source that is always there — the registry the
|
||||
# bundle already renders. Two node-level facts make that work:
|
||||
# * kubelet cannot reach the registry Service VIP (the live stack answered
|
||||
# "Empty reply"), so containerd is told to go through the loopback
|
||||
# hostPort the registry Deployment binds (the Deployment renders it) —
|
||||
# that is configure_registry_mirror below;
|
||||
# * the registry's own image (registry:2) must already be in containerd
|
||||
# before the registry Deployment can start at all —
|
||||
# import_registry_image below caches it.
|
||||
# After deploy_bundle, push_images_to_registry mirrors the built images into
|
||||
# the registry, so containerd's imported copies are a first-boot cache
|
||||
# rather than the only copy.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# The node's containerd cannot dial the registry Service VIP, so pulls arrive
|
||||
# over the loopback hostPort the registry Deployment binds. k3s reads this file
|
||||
# when the agent starts and regenerates containerd's certs.d from it — no
|
||||
# restart, no effect — so a CONTENT change restarts k3s; an identical file
|
||||
# (every re-run) restarts nothing. K3S_REGISTRIES_FILE is a variable so
|
||||
# bootstrap_test.sh can point the function at a scratch file.
|
||||
configure_registry_mirror() {
|
||||
local file="$K3S_REGISTRIES_FILE" tmp
|
||||
tmp="$(mktemp)"
|
||||
remember_temp "$tmp"
|
||||
cat > "$tmp" <<EOF
|
||||
mirrors:
|
||||
"${REGISTRY_URL}":
|
||||
endpoint:
|
||||
- "http://${REGISTRY_PUSH_HOST}"
|
||||
EOF
|
||||
if [ -f "$file" ] && cmp -s "$tmp" "$file"; then
|
||||
rm -f "$tmp"
|
||||
ok "registry mirror already configured (${REGISTRY_URL} -> http://${REGISTRY_PUSH_HOST})"
|
||||
return 0
|
||||
fi
|
||||
mkdir -p "$(dirname "$file")"
|
||||
mv "$tmp" "$file"
|
||||
log "restarting k3s to load the registry mirror (${REGISTRY_URL} -> http://${REGISTRY_PUSH_HOST})"
|
||||
systemctl restart k3s
|
||||
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
|
||||
wait_for_node_ready
|
||||
}
|
||||
|
||||
# The registry Deployment runs registry:2 (platform.defaultRegistryImage; the
|
||||
# renderer's default — this script never passes --registry-image). On a box
|
||||
# that cannot reach Docker Hub the Deployment can never start without a local
|
||||
# copy, so the installer caches one whenever it can. Best-effort by design: if
|
||||
# the pull fails the registry rollout still fails loudly at deploy_bundle, with
|
||||
# the regular diagnostics — but for every box that CAN pull, the image is
|
||||
# fetched exactly once, here, instead of at first pod start.
|
||||
import_registry_image() {
|
||||
# The name containerd normalizes "registry:2" to after any docker-save import.
|
||||
if k3s_cmd ctr images ls -q 2>/dev/null | grep -qx 'docker.io/library/registry:2'; then
|
||||
ok "registry image registry:2 already in k3s containerd"
|
||||
return 0
|
||||
fi
|
||||
log "importing the registry's own image (registry:2) into k3s containerd"
|
||||
systemctl start docker
|
||||
if docker pull registry:2 && docker save registry:2 | k3s_cmd ctr images import -; then
|
||||
ok "registry image registry:2 imported"
|
||||
else
|
||||
warn "could not import registry:2: the in-cluster registry will start only if the node can pull it from Docker Hub; on an air-gapped box import it by hand (docs/troubleshooting.md §8e)"
|
||||
fi
|
||||
systemctl stop docker docker.socket 2>/dev/null || true
|
||||
}
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. Source/binary + image build + containerd import
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -1299,7 +1430,7 @@ resolve_game_jars() {
|
||||
luckperms_latest_jar() {
|
||||
local json url
|
||||
json="$(curl -fsSL --retry 5 --retry-delay 2 \
|
||||
-A "felis-bootstrap (+https://github.com/MliroLirrorsIngenuity/Felis)" \
|
||||
-A "felis-bootstrap (+https://github.com/FelisMC/Felis)" \
|
||||
"https://metadata.luckperms.net/data/all")" || return 1
|
||||
url="$(printf '%s' "$json" \
|
||||
| grep -o 'https://download\.luckperms\.net/[0-9]\{1,\}/bukkit/loader/[^"]*\.jar' || true)"
|
||||
@@ -1321,7 +1452,7 @@ luckperms_latest_jar() {
|
||||
papermc_latest_jar() {
|
||||
local project="$1" version="$2" json urls url sha
|
||||
json="$(curl -fsSL --retry 5 --retry-delay 2 \
|
||||
-A "felis-bootstrap (+https://github.com/MliroLirrorsIngenuity/Felis)" \
|
||||
-A "felis-bootstrap (+https://github.com/FelisMC/Felis)" \
|
||||
"https://fill.papermc.io/v3/projects/${project}/versions/${version}/builds/latest")" || return 1
|
||||
urls="$(printf '%s' "$json" | grep -o 'https://fill-data\.papermc\.io/[^"]*\.jar' || true)"
|
||||
url="${urls%%$'\n'*}"
|
||||
@@ -1971,24 +2102,30 @@ EOF
|
||||
# family as the root_domain loss fixed in ecbeb20 -- generated file, hand-set
|
||||
# value, no carry-forward.
|
||||
#
|
||||
# Cached on first call because write_felis_toml clobbers felis.host.toml before
|
||||
# it is called again for felis.pod.toml: by then the file this would read from
|
||||
# no longer has the block. The pod toml is the fallback for exactly that window.
|
||||
# The carry is the section's header and key lines only. Printing every line up to
|
||||
# the next section header hoarded the generated [[auth_source]] comment block
|
||||
# that sits below [smtp] into this carry: each re-run then re-emitted the hoard
|
||||
# plus a fresh template copy, growing both config files by one comment block per
|
||||
# run (audit #50). The extraction is idempotent, which is also why re-reading the
|
||||
# freshly rewritten host file on the pod pass is safe. The pod toml remains the
|
||||
# fallback for a host file with no [smtp] section at all.
|
||||
persisted_smtp_block() {
|
||||
if [ -z "${SMTP_BLOCK_CACHED:-}" ]; then
|
||||
SMTP_BLOCK_CACHED=1
|
||||
SMTP_BLOCK=""
|
||||
local f
|
||||
for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do
|
||||
[ -r "$f" ] || continue
|
||||
# Print from [smtp] up to (not including) the next section header.
|
||||
SMTP_BLOCK="$(awk '/^[[:space:]]*\[smtp\]/ { f=1 }
|
||||
f && /^[[:space:]]*\[/ && !/^[[:space:]]*\[smtp\]/ { exit }
|
||||
f { print }' "$f")"
|
||||
[ -n "$SMTP_BLOCK" ] && break
|
||||
done
|
||||
fi
|
||||
printf '%s' "$SMTP_BLOCK"
|
||||
local f out
|
||||
for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do
|
||||
[ -r "$f" ] || continue
|
||||
out="$(awk '
|
||||
/^[[:space:]]*\[/ {
|
||||
if (insmtp) exit
|
||||
insmtp = ($0 ~ /^[[:space:]]*\[smtp\][[:space:]]*$/)
|
||||
if (insmtp) print
|
||||
next
|
||||
}
|
||||
insmtp && /^[[:space:]]*("[A-Za-z_][A-Za-z0-9_]*"|[A-Za-z_][A-Za-z0-9_]*)[[:space:]]*=/ { print }
|
||||
' "$f")"
|
||||
[ -n "$out" ] || continue
|
||||
printf '%s' "$out"
|
||||
return 0
|
||||
done
|
||||
}
|
||||
|
||||
# persisted_auth_source_blocks echoes the [[auth_source]] tables an earlier run left
|
||||
@@ -2013,17 +2150,78 @@ persisted_auth_source_blocks() {
|
||||
'url = "https://littleskin.cn/api/yggdrasil/sessionserver/session/minecraft/hasJoined"'
|
||||
}
|
||||
|
||||
# persisted_archive_block echoes the operator-owned [archive] keys an earlier run
|
||||
# left behind — the retention window, the pre-reap warn offsets, the local cap —
|
||||
# so a re-run does not silently revert them to the built-ins the reaper carries
|
||||
# (felis reaper reads these from the config Secret at run time; defaults: 90d
|
||||
# retention, 3d/1d warnings, no cap). store and local_path are NOT carried: this
|
||||
# script owns them (FELIS_ARCHIVE_LOCAL_PATH must equal the mount). Same
|
||||
# first-readable-file rule as persisted_smtp_block; warn_before must be a
|
||||
# single-line TOML array (the shape every writer here emits).
|
||||
persisted_archive_block() {
|
||||
local f out
|
||||
for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do
|
||||
[ -r "$f" ] || continue
|
||||
out="$(awk '
|
||||
/^[[:space:]]*\[/ { sect = $0; next }
|
||||
sect ~ /^[[:space:]]*\[archive\][[:space:]]*$/ &&
|
||||
/^[[:space:]]*(retention|warn_before|max_local_bytes)[[:space:]]*=/ { print }
|
||||
' "$f")"
|
||||
[ -n "$out" ] || continue
|
||||
printf '%s\n' "$out"
|
||||
return 0
|
||||
done
|
||||
}
|
||||
|
||||
# persisted_registry_block echoes the operator-owned [registry] keys an earlier run
|
||||
# left behind — the build-lane executor mirrors, the resource caps, the uploads
|
||||
# backend and its [registry.s3] subtable — so §15's upgrade path (re-run the
|
||||
# installer) does not silently revert them. Nothing in this script's inputs
|
||||
# derives these: they are hand-written per docs/troubleshooting.md §8e or stamped
|
||||
# by the storage wizard. url and build_namespace are NOT carried: this script
|
||||
# owns them (they must match REGISTRY_URL / BUILD_NS). Same first-readable-file
|
||||
# rule as persisted_smtp_block.
|
||||
persisted_registry_block() {
|
||||
local f out
|
||||
for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do
|
||||
[ -r "$f" ] || continue
|
||||
out="$(awk '
|
||||
/^[[:space:]]*\[/ { sect = $0; next }
|
||||
sect ~ /^[[:space:]]*\[registry\][[:space:]]*$/ &&
|
||||
/^[[:space:]]*(kaniko_image|trivy_image|trivy_db_repository|trivy_java_db_repository|build_cpu_limit|build_mem_limit|user_uploads_context)[[:space:]]*=/ { print }
|
||||
sect ~ /^[[:space:]]*\[registry\.s3\][[:space:]]*$/ && /^[[:space:]]*[A-Za-z_]+[[:space:]]*=/ {
|
||||
if (!s3hdr) { printf "[registry.s3]\n"; s3hdr = 1 }
|
||||
print
|
||||
}
|
||||
' "$f")"
|
||||
[ -n "$out" ] || continue
|
||||
printf '%s\n' "$out"
|
||||
return 0
|
||||
done
|
||||
}
|
||||
|
||||
write_felis_toml() {
|
||||
local target="$1" db_host="$2" smtp_block auth_source_blocks
|
||||
local target="$1" db_host="$2" smtp_block auth_source_blocks registry_block archive_block
|
||||
smtp_block="$(persisted_smtp_block)"
|
||||
if [ -n "$smtp_block" ]; then
|
||||
log "carrying forward the configured [smtp] relay"
|
||||
smtp_block="${smtp_block}"$'\n' # keep a blank line before the next section
|
||||
fi
|
||||
auth_source_blocks="$(persisted_auth_source_blocks)"
|
||||
registry_block="$(persisted_registry_block)"
|
||||
if [ -n "$registry_block" ]; then
|
||||
log "carrying forward the configured [registry] overrides"
|
||||
registry_block="${registry_block}"$'\n' # keep a blank line before the next section
|
||||
fi
|
||||
archive_block="$(persisted_archive_block)"
|
||||
if [ -n "$archive_block" ]; then
|
||||
log "carrying forward the configured [archive] overrides"
|
||||
archive_block="${archive_block}"$'\n' # keep a blank line before the next section
|
||||
fi
|
||||
cat > "$target" <<EOF
|
||||
# Generated by deploy/bootstrap.sh; rerun the installer to regenerate. Hand edits are
|
||||
# overwritten, except [smtp] and [[auth_source]], which carry forward.
|
||||
# overwritten, except [smtp], [[auth_source]], and the operator-owned [registry] /
|
||||
# [archive] overrides, which carry forward.
|
||||
[server]
|
||||
listen = "0.0.0.0:8080"
|
||||
root_domain = "${FELIS_ROOT_DOMAIN}"
|
||||
@@ -2044,11 +2242,11 @@ lobby_image = "${FELIS_LOBBY_IMAGE}"
|
||||
[registry]
|
||||
url = "${REGISTRY_URL}"
|
||||
build_namespace = "${BUILD_NS}"
|
||||
|
||||
${registry_block}
|
||||
[archive]
|
||||
store = "tarLocal"
|
||||
local_path = "/var/lib/felis/archives"
|
||||
|
||||
local_path = "${FELIS_ARCHIVE_LOCAL_PATH}"
|
||||
${archive_block}
|
||||
[auth]
|
||||
admin_hostname = "op.console.${FELIS_ROOT_DOMAIN}"
|
||||
panel_hostname = "console.${FELIS_ROOT_DOMAIN}"
|
||||
@@ -2121,10 +2319,12 @@ deploy_bundle() {
|
||||
done
|
||||
|
||||
log "provisioning felis-config + felis-service-token + felis-forwarding-secret + panel TLS secrets (out-of-band, never in the bundle)"
|
||||
kube -n "$CONTROL_NS" create secret generic felis-config \
|
||||
--from-file=felis.toml="${STATE_DIR}/felis.pod.toml" \
|
||||
--dry-run=client -o yaml | kube apply -f -
|
||||
apply_felis_config_secrets
|
||||
apply_literal_secret "$CONTROL_NS" felis-service-token token "$SERVICE_TOKEN"
|
||||
# The build namespace needs the same token: the build Job's fetch initContainer
|
||||
# streams a submission's build context from the felis-api internal face, and a
|
||||
# secretKeyRef is namespace-local (a PVC cannot carry it across either).
|
||||
apply_literal_secret "$BUILD_NS" felis-service-token token "$SERVICE_TOKEN"
|
||||
# The forwarding key every backend verifies the proxy's handshake with. `felis setup`
|
||||
# replicates it into the minecraft namespace (ensureSecretReplica) before it creates
|
||||
# the pods that mount it; the operator injects it into EVERY backend, because Velocity's
|
||||
@@ -2137,11 +2337,40 @@ deploy_bundle() {
|
||||
--dry-run=client -o yaml | kube apply -f -
|
||||
|
||||
log "rendering + applying the control-plane bundle"
|
||||
"$HOST_BIN" manifests \
|
||||
--felis-image "$FELIS_IMAGE" \
|
||||
--panel-node-port "$FELIS_PANEL_NODEPORT" \
|
||||
--velocity-cidr "${NODE_IP}/32" \
|
||||
| kube apply -f -
|
||||
local -a manifest_args=(
|
||||
--felis-image "$FELIS_IMAGE"
|
||||
--panel-node-port "$FELIS_PANEL_NODEPORT"
|
||||
--velocity-cidr "${NODE_IP}/32"
|
||||
)
|
||||
# Backups are on by default (the renderer's own default names felis-backups); an emptied
|
||||
# FELIS_BACKUP_PVC asks for the no-backup shape explicitly, and a custom name must be
|
||||
# passed through or the api would advertise a PVC the bundle never created.
|
||||
if [ -n "$FELIS_BACKUP_PVC" ]; then
|
||||
manifest_args+=(--backup-pvc "$FELIS_BACKUP_PVC")
|
||||
else
|
||||
manifest_args+=(--backup-pvc=)
|
||||
fi
|
||||
# Retention renders only when the operator names where the worlds live; the archive path
|
||||
# always travels with it because it must equal the [archive] local_path written above.
|
||||
if [ -n "$FELIS_WORLDS_HOST_PATH" ]; then
|
||||
log "retention enabled: the daily reaper will read worlds from ${FELIS_WORLDS_HOST_PATH}"
|
||||
# The reaper pod runs as the tree's non-root uid (1000, platform.workloads.nonRootUID)
|
||||
# and must traverse into the per-volume directories under this root. k3s's own storage
|
||||
# root ships 0700 root:root, so grant traverse — an ACL entry when the host has setfacl,
|
||||
# otherwise the equivalent o+x. Traverse only: no listing either way, and the per-volume
|
||||
# directories themselves are world-accessible (local-path creates them 0777).
|
||||
if [ -d "$FELIS_WORLDS_HOST_PATH" ]; then
|
||||
if command -v setfacl >/dev/null 2>&1; then
|
||||
setfacl -m u:1000:x "$FELIS_WORLDS_HOST_PATH" || chmod o+x "$FELIS_WORLDS_HOST_PATH"
|
||||
else
|
||||
chmod o+x "$FELIS_WORLDS_HOST_PATH"
|
||||
fi
|
||||
else
|
||||
warn "worlds root ${FELIS_WORLDS_HOST_PATH} does not exist yet; the reaper CronJob cannot start until it does (hostPath type Directory)"
|
||||
fi
|
||||
manifest_args+=(--worlds-host-path "$FELIS_WORLDS_HOST_PATH" --archive-local-path "$FELIS_ARCHIVE_LOCAL_PATH")
|
||||
fi
|
||||
"$HOST_BIN" manifests "${manifest_args[@]}" | kube apply -f -
|
||||
restart_existing_control_plane "$had_api" "$had_operator"
|
||||
|
||||
log "waiting for control-plane rollouts"
|
||||
@@ -2167,10 +2396,54 @@ restart_existing_control_plane() {
|
||||
if [ "$had_operator" = "1" ]; then kube -n "$CONTROL_NS" rollout restart deployment/felis-operator; fi
|
||||
}
|
||||
|
||||
# The login/lobby images use local mutable tags. Importing a replacement updates
|
||||
# containerd, but an existing StatefulSet template is byte-for-byte unchanged and
|
||||
# Kubernetes will not roll it. Recreate only the two always-on system pods so a
|
||||
# convergent bootstrap actually starts the images it just imported.
|
||||
# push_image_to_registry <ref> re-tags a locally built image for the node's
|
||||
# loopback push endpoint and uploads it. The registry keys a repository by the
|
||||
# path AFTER the host, so pushing 127.0.0.1:5000/felis/felis:demo lands exactly
|
||||
# where a later kubelet pull of registry.felis.svc:5000/felis/felis:demo (the
|
||||
# mirror rewrites the host) will look. A ref not under REGISTRY_URL is not
|
||||
# mirrored — warn, don't fail: the install is still self-consistent, that image
|
||||
# just has no pull source once GC collects its containerd copy.
|
||||
push_image_to_registry() {
|
||||
local ref="$1" push_ref
|
||||
case "$ref" in
|
||||
"${REGISTRY_URL}/"*)
|
||||
push_ref="${REGISTRY_PUSH_HOST}/${ref#"${REGISTRY_URL}/"}"
|
||||
;;
|
||||
*)
|
||||
warn "not mirroring ${ref} into the internal registry: it is not under ${REGISTRY_URL}; once the image GC collects that tag, nothing can re-pull it"
|
||||
return 0
|
||||
;;
|
||||
esac
|
||||
log "mirroring ${ref} into the internal registry"
|
||||
docker tag "$ref" "$push_ref" || die "could not tag ${ref} as ${push_ref} — is docker healthy?"
|
||||
docker push "$push_ref" || die "could not mirror ${ref} into the internal registry — check the registry Deployment/pod and its PVC"
|
||||
docker rmi "$push_ref" >/dev/null 2>&1 || true
|
||||
}
|
||||
|
||||
# Every image this installer builds is hosted in the registry, so the copies it
|
||||
# imported into containerd are a first-boot cache, not the only copy: kubelet
|
||||
# re-pulls from the registry after any image GC. Runs AFTER deploy_bundle — the
|
||||
# registry it pushes into does not exist before that.
|
||||
#
|
||||
# Docker is started once for the whole batch and stopped once at the end. A
|
||||
# start/stop pair per image trips systemd's start rate limit — observed live on
|
||||
# a re-run: three fast pushes, then "Start request repeated too quickly /
|
||||
# start-limit-hit" and the fourth image never got mirrored. docker.service is
|
||||
# socket-triggered, so each cycle counts twice against the burst limit.
|
||||
push_images_to_registry() {
|
||||
local img
|
||||
systemctl start docker
|
||||
for img in "$FELIS_IMAGE" "$FELIS_LIMBO_IMAGE" "$FELIS_LOBBY_IMAGE" "$FELIS_PAPER_IMAGE"; do
|
||||
[ -n "$img" ] || continue
|
||||
push_image_to_registry "$img"
|
||||
done
|
||||
systemctl stop docker docker.socket 2>/dev/null || true
|
||||
}
|
||||
|
||||
# The login/lobby images use mutable :demo tags. Importing/pushing a replacement
|
||||
# updates containerd, but an existing StatefulSet template is byte-for-byte
|
||||
# unchanged and Kubernetes will not roll it. Recreate only the two always-on
|
||||
# system pods so a convergent bootstrap actually starts the images it just built.
|
||||
restart_existing_system_servers() {
|
||||
local name pods
|
||||
for name in "$LOGIN_SERVER" "$LOBBY_SERVER"; do
|
||||
@@ -2575,6 +2848,11 @@ main() {
|
||||
ensure_panel_tls_cert
|
||||
install_docker
|
||||
install_k3s
|
||||
# The registry mirror must exist before the bundle's pods start pulling (and
|
||||
# before any re-run's rollouts); the registry's own image must be in containerd
|
||||
# before its Deployment can start at all.
|
||||
configure_registry_mirror
|
||||
import_registry_image
|
||||
# Three ways to end up with a felis binary, in preference order. The release download is
|
||||
# the only one that skips compiling: it is the CI artifact for this exact tag, panel
|
||||
# included. Both other arms leave HAVE_PREBUILT_BINARY unset where a source build is what
|
||||
@@ -2592,6 +2870,9 @@ main() {
|
||||
configure_postgres
|
||||
run_migrations
|
||||
deploy_bundle
|
||||
# AFTER deploy_bundle: the registry the built images are mirrored into is part
|
||||
# of that bundle.
|
||||
push_images_to_registry
|
||||
restart_existing_system_servers
|
||||
# After deploy_bundle: the proxy dials felis-api's internal ClusterIP, which does not
|
||||
# exist until the bundle is applied.
|
||||
|
||||
@@ -192,6 +192,83 @@ for hdr in '[[ auth_source ]]' '[["auth_source"]]' "[['auth_source']]"; do
|
||||
esac
|
||||
done
|
||||
|
||||
# --- [smtp] carry-forward does not hoard the auth_source comment block -------------------
|
||||
# persisted_smtp_block used to print every line between [smtp] and the next section
|
||||
# header -- which includes the generated Yggdrasil comment block that sits above
|
||||
# [[auth_source]]. Each re-run re-emitted that hoard plus a fresh template copy, so both
|
||||
# config files grew by one comment block per run (audit #50). The carry must be the
|
||||
# section's header and keys only, and must be byte-stable when written back.
|
||||
|
||||
sblock="$(awk '/^persisted_smtp_block\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$sblock" ] || { echo "FAIL: no persisted_smtp_block found in $BS"; exit 1; }
|
||||
[ "$(printf '%s\n' "$sblock" | wc -l)" -lt 40 ] \
|
||||
|| { echo "FAIL: the extracted block is not the function -- did its closing brace move?"; exit 1; }
|
||||
|
||||
sfn="$(mktemp)"
|
||||
printf '%s\n' "$sblock" > "$sfn"
|
||||
smtp_dir="$(mktemp -d)"
|
||||
trap 'rm -f "$jar" "$sfn"; rm -rf "$vdir" "$sdir" "$smtp_dir"' EXIT
|
||||
|
||||
run_smtp() { # state-dir
|
||||
STATE_DIR="$1" SBLOCK_FILE="$sfn" bash -c '. "$SBLOCK_FILE"; persisted_smtp_block'
|
||||
}
|
||||
|
||||
cat > "$smtp_dir/felis.host.toml" <<'TOML'
|
||||
[server]
|
||||
listen = "0.0.0.0:8080"
|
||||
|
||||
[smtp]
|
||||
host = "mail.example"
|
||||
port = 587
|
||||
from = "[email protected]"
|
||||
username = "relay-user"
|
||||
password_ref = "smtp-password"
|
||||
|
||||
# Third-party Yggdrasil sources federated by the hasJoined multiplexer. Mojang is
|
||||
# always the code-owned identity anchor (premium-first), prepended in Go; sources here
|
||||
# append as namespace-rewritten guests. A fresh install federates LittleSkin. Edit the
|
||||
# list in /etc/felis/felis.host.toml and rerun the installer; re-runs keep it as it
|
||||
# is, and with no [[auth_source]] at all the server is Mojang-only.
|
||||
|
||||
[[auth_source]]
|
||||
tag = "littleskin"
|
||||
prefix = "LS"
|
||||
url = "https://littleskin.cn/api/yggdrasil/sessionserver/session/minecraft/hasJoined"
|
||||
TOML
|
||||
|
||||
out="$(run_smtp "$smtp_dir")"
|
||||
expect "a configured [smtp] relay is carried" 'host = "mail.example"' "$out"
|
||||
expect "its port survives the carry" 'port = 587' "$out"
|
||||
expect "its credentials reference survives" 'password_ref = "smtp-password"' "$out"
|
||||
case "$out" in
|
||||
*"#"*)
|
||||
echo "FAIL: the carry hoards comment lines:"; printf '%s\n' "$out"; fails=$((fails + 1)) ;;
|
||||
*) echo "PASS the carry is header and keys only -- no comment hoard" ;;
|
||||
esac
|
||||
case "$out" in
|
||||
*"[["*)
|
||||
echo "FAIL: the carry ran into the next section:"; printf '%s\n' "$out"; fails=$((fails + 1)) ;;
|
||||
*) echo "PASS the carry stops at the next section header" ;;
|
||||
esac
|
||||
|
||||
# Write the carry back the way write_felis_toml does (carry + one fresh template block +
|
||||
# the tables) and extract again: a second re-run must add nothing.
|
||||
{
|
||||
printf '%s\n' "$out"
|
||||
printf '\n%s\n' '# Third-party Yggdrasil sources federated by the hasJoined multiplexer. Mojang is'
|
||||
printf '%s\n' '[[auth_source]]' ' tag = "littleskin"' ' prefix = "LS"' \
|
||||
' url = "https://littleskin.cn/api/yggdrasil/sessionserver/session/minecraft/hasJoined"'
|
||||
} > "$smtp_dir/felis.host.toml"
|
||||
out2="$(run_smtp "$smtp_dir")"
|
||||
printf '%s\n' "$out" > "$smtp_dir/first"
|
||||
printf '%s\n' "$out2" > "$smtp_dir/second"
|
||||
if cmp -s "$smtp_dir/first" "$smtp_dir/second"; then
|
||||
echo "PASS a carried-forward [smtp] converges (a second re-run adds nothing)"
|
||||
else
|
||||
echo "FAIL: carrying [smtp] is not idempotent:"; diff "$smtp_dir/first" "$smtp_dir/second" | head
|
||||
fails=$((fails + 1))
|
||||
fi
|
||||
|
||||
# --- write_nano_config leaves the unit able to read its config ---------------------------
|
||||
# felis-nano runs as a DynamicUser, so the directory must be searchable by others under a
|
||||
# hardened umask too, including one an older installer left at 0750 -- but the full
|
||||
@@ -574,6 +651,295 @@ mkdir -p "$sdir/src/.git"
|
||||
expect "a failed fetch into an existing checkout names the token" "set FELIS_GITHUB_TOKEN" \
|
||||
"$(run_fetch "$sdir/src")"
|
||||
|
||||
# --- default install keeps backups, and retention envs reach the renderer ----------------
|
||||
# A default install must render the world-archive PVC (without one, backup/restore answer an
|
||||
# honest 503), and FELIS_WORLDS_HOST_PATH must turn into the reaper's two flags or an
|
||||
# operator's retention enablement silently renders no CronJob. Extracted, not retyped.
|
||||
|
||||
mblock="$(awk '/^ log "rendering \+ applying the control-plane bundle"/,/kube apply -f -/' "$BS")"
|
||||
[ -n "$mblock" ] || { echo "FAIL: no manifest_args block found in $BS"; exit 1; }
|
||||
[ "$(printf '%s\n' "$mblock" | wc -l)" -lt 40 ] \
|
||||
|| { echo "FAIL: the extracted block is not the manifest_args block -- did it move?"; exit 1; }
|
||||
|
||||
run_bundle_flags() { # backup-pvc worlds-host-path
|
||||
FELIS_IMAGE=reg/felis:test FELIS_PANEL_NODEPORT=30443 NODE_IP=10.0.0.5 \
|
||||
FELIS_BACKUP_PVC="$1" FELIS_WORLDS_HOST_PATH="$2" FELIS_ARCHIVE_LOCAL_PATH=/var/lib/felis/archives \
|
||||
HOST_BIN=myManifests bash -c '
|
||||
log() { :; }
|
||||
warn() { printf "WARN: %s\n" "$*"; }
|
||||
kube() { cat; }
|
||||
myManifests() { printf "%s\n" "$@"; }
|
||||
setfacl() { printf "SETFACL %s\n" "$*"; }
|
||||
run_bundle() {
|
||||
'"$mblock"'
|
||||
}
|
||||
run_bundle'
|
||||
}
|
||||
|
||||
out="$(run_bundle_flags felis-backups '')"
|
||||
expect "a default install asks the renderer for the archive PVC" "--backup-pvc
|
||||
felis-backups" "$out"
|
||||
case "$out" in
|
||||
*--worlds-host-path*) echo "FAIL: no reaper flags may render without FELIS_WORLDS_HOST_PATH"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
|
||||
out="$(run_bundle_flags '' '')"
|
||||
expect "an emptied FELIS_BACKUP_PVC is the explicit no-backup shape" "--backup-pvc=" "$out"
|
||||
|
||||
out="$(run_bundle_flags felis-backups /var/lib/rancher/k3s/storage)"
|
||||
expect "enabling retention passes the worlds root" "--worlds-host-path
|
||||
/var/lib/rancher/k3s/storage" "$out"
|
||||
expect "enabling retention passes the archive mount that must match felis.toml" "--archive-local-path
|
||||
/var/lib/felis/archives" "$out"
|
||||
|
||||
# The warn fires only when the root is ABSENT (hostPath type Directory would fail);
|
||||
# the case above passes a path that exists on any host already running k3s, so it
|
||||
# must not also demand the warning — probing the real /var/lib/rancher path made
|
||||
# this suite red on exactly the hosts the installer is for. Point the warn case at
|
||||
# a path guaranteed missing.
|
||||
missing="/tmp/felis-worlds-root-must-not-exist-$$"
|
||||
out="$(run_bundle_flags felis-backups "$missing")"
|
||||
expect "a missing worlds root is warned about, not silently skipped" "WARN: worlds root $missing does not exist yet" "$out"
|
||||
|
||||
# The reaper pod is non-root (uid 1000) and k3s ships the storage root 0700 root:root, so
|
||||
# the installer must grant traverse or every archive dies with permission denied.
|
||||
wdir="$(mktemp -d)"
|
||||
out="$(run_bundle_flags felis-backups "$wdir")"
|
||||
expect "enabling retention grants the reaper uid traverse on the worlds root" "SETFACL -m u:1000:x $wdir" "$out"
|
||||
|
||||
# --- the registry mirror writer -----------------------------------------------------------
|
||||
# k3s only consults registries.yaml at agent start, so a CONTENT change must restart k3s and
|
||||
# an identical file (every re-run) must restart nothing. The k3s restart is the expensive,
|
||||
# disruptive half of the pair -- getting the idempotence wrong bounces the whole cluster on
|
||||
# every installer re-run, so both halves are pinned here against the extracted function.
|
||||
|
||||
cmblock="$(awk '/^configure_registry_mirror\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$cmblock" ] || { echo "FAIL: no configure_registry_mirror found in $BS"; exit 1; }
|
||||
|
||||
run_mirror() { # scratch-file
|
||||
K3S_REGISTRIES_FILE="$1" REGISTRY_URL=registry.felis.svc:5000 REGISTRY_PUSH_HOST=127.0.0.1:5000 \
|
||||
bash -c '
|
||||
log() { printf "LOG: %s\n" "$*"; }
|
||||
ok() { printf "OK: %s\n" "$*"; }
|
||||
die() { printf "DIE: %s\n" "$*"; exit 1; }
|
||||
warn() { printf "WARN: %s\n" "$*"; }
|
||||
remember_temp() { :; }
|
||||
systemctl() { printf "SYSTEMCTL %s\n" "$*"; }
|
||||
kube() { printf "n Ready \n"; }
|
||||
wait_for_node_ready() { kube get nodes | grep -q " Ready " && ok "k3s node Ready"; }
|
||||
'"$cmblock"'
|
||||
configure_registry_mirror'
|
||||
}
|
||||
|
||||
mfile="$(mktemp -u)"
|
||||
out="$(run_mirror "$mfile")"
|
||||
expect "a missing registries.yaml is written" "\"registry.felis.svc:5000\":" "$(cat "$mfile" 2>/dev/null)"
|
||||
expect "the mirror endpoint is the node loopback push/pull host" "\"http://127.0.0.1:5000\"" "$(cat "$mfile" 2>/dev/null)"
|
||||
expect "a content change restarts k3s" "SYSTEMCTL restart k3s" "$out"
|
||||
|
||||
out="$(run_mirror "$mfile")"
|
||||
expect "an identical registries.yaml is recognised" "already configured" "$out"
|
||||
case "$out" in
|
||||
*"SYSTEMCTL restart"*) echo "FAIL: a re-run with identical content must not restart k3s"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
|
||||
printf 'mirrors: {}\n' >"$mfile"
|
||||
out="$(run_mirror "$mfile")"
|
||||
expect "changed content restarts k3s again" "SYSTEMCTL restart k3s" "$out"
|
||||
rm -f "$mfile"
|
||||
|
||||
# --- image mirroring ----------------------------------------------------------------------
|
||||
# The registry keys a repository by the path AFTER the host, so the push must swap the
|
||||
# registry host for the node's loopback endpoint and nothing else. A ref outside the
|
||||
# registry must be warned about, not silently pushed somewhere unintended.
|
||||
|
||||
pblock="$(awk '/^push_image_to_registry\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$pblock" ] || { echo "FAIL: no push_image_to_registry found in $BS"; exit 1; }
|
||||
|
||||
run_push() { # ref [docker-push-exit]
|
||||
REF="$1" PUSH_EXIT="${2:-0}" \
|
||||
REGISTRY_URL=registry.felis.svc:5000 REGISTRY_PUSH_HOST=127.0.0.1:5000 \
|
||||
bash -c '
|
||||
log() { printf "LOG: %s\n" "$*"; }
|
||||
warn() { printf "WARN: %s\n" "$*"; }
|
||||
die() { printf "DIE: %s\n" "$*"; exit 1; }
|
||||
ok() { :; }
|
||||
systemctl() { :; }
|
||||
docker() {
|
||||
case "$1" in
|
||||
push) printf "DOCKER %s\n" "$*"; return "$PUSH_EXIT" ;;
|
||||
*) printf "DOCKER %s\n" "$*" ;;
|
||||
esac
|
||||
}
|
||||
'"$pblock"'
|
||||
push_image_to_registry "$REF"'
|
||||
}
|
||||
|
||||
out="$(run_push registry.felis.svc:5000/felis/felis:demo)"
|
||||
expect "a registry ref is re-tagged onto the node loopback endpoint" \
|
||||
"DOCKER tag registry.felis.svc:5000/felis/felis:demo 127.0.0.1:5000/felis/felis:demo" "$out"
|
||||
expect "and pushed to exactly that endpoint" "DOCKER push 127.0.0.1:5000/felis/felis:demo" "$out"
|
||||
|
||||
out="$(run_push registry.felis.svc:50000/felis/felis:demo)"
|
||||
expect "a ref outside the registry is refused with a warning" "WARN: not mirroring" "$out"
|
||||
case "$out" in
|
||||
*"DOCKER push"*) echo "FAIL: a non-registry ref must not be pushed"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
|
||||
out="$(run_push registry.felis.svc:5000/felis/felis:demo 1)"
|
||||
expect "a failed push fails the install loudly" "DIE: could not mirror" "$out"
|
||||
|
||||
# docker must be started ONCE for the whole batch: a start/stop pair per image trips
|
||||
# systemd's start rate limit ("start-limit-hit" — observed live; the 4th image was never
|
||||
# mirrored because docker.service is socket-triggered and each cycle counts twice).
|
||||
wiblock="$(awk '/^push_images_to_registry\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$wiblock" ] || { echo "FAIL: no push_images_to_registry found in $BS"; exit 1; }
|
||||
out="$(
|
||||
FELIS_IMAGE=a FELIS_LIMBO_IMAGE=b FELIS_LOBBY_IMAGE=c FELIS_PAPER_IMAGE=d bash -c '
|
||||
systemctl() { printf "SYSTEMCTL %s\n" "$*"; }
|
||||
push_image_to_registry() { printf "PUSH %s\n" "$1"; }
|
||||
'"$wiblock"'
|
||||
push_images_to_registry'
|
||||
)"
|
||||
starts="$(printf '%s\n' "$out" | grep -c 'SYSTEMCTL start docker')"
|
||||
stops="$(printf '%s\n' "$out" | grep -c 'SYSTEMCTL stop docker')"
|
||||
[ "$starts" = 1 ] && [ "$stops" = 1 ] && [ "$(printf '%s\n' "$out" | grep -c '^PUSH')" = 4 ] \
|
||||
&& echo "PASS the batch wraps all four pushes in ONE docker start/stop" \
|
||||
|| { echo "FAIL: expected 1 start / 1 stop / 4 pushes, got:"; printf '%s\n' "$out"; fails=$((fails + 1)); }
|
||||
|
||||
# --- the registry's own image must not be re-pulled on every run --------------------------
|
||||
iblock="$(awk '/^import_registry_image\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$iblock" ] || { echo "FAIL: no import_registry_image found in $BS"; exit 1; }
|
||||
|
||||
out="$(
|
||||
bash -c '
|
||||
log() { printf "LOG: %s\n" "$*"; }
|
||||
ok() { printf "OK: %s\n" "$*"; }
|
||||
warn() { printf "WARN: %s\n" "$*"; }
|
||||
die() { printf "DIE: %s\n" "$*"; exit 1; }
|
||||
systemctl() { :; }
|
||||
k3s_cmd() { case "$*" in "ctr images ls -q") printf "docker.io/library/registry:2\n" ;; esac; }
|
||||
docker() { printf "DOCKER %s\n" "$*"; return 1; }
|
||||
'"$iblock"'
|
||||
import_registry_image'
|
||||
)"
|
||||
expect "an already-imported registry:2 is left alone" "already in k3s containerd" "$out"
|
||||
case "$out" in
|
||||
*DOCKER*) echo "FAIL: a present registry:2 must not trigger a docker pull"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
|
||||
# --- installer re-runs refresh the workload namespace's felis-config copy ---------------
|
||||
# The backup/restore/fileedit Jobs and the reaper mount the workload namespace's own
|
||||
# felis-config (a secretKeyRef is namespace-local). `felis setup` makes that replica
|
||||
# create-if-absent -- right for credentials, wrong for a rendered config -- so the
|
||||
# installer must refresh it every run; a stale copy keeps old DB/archive settings.
|
||||
|
||||
fcblock="$(awk '/^apply_felis_config_secrets\(\) \{/,/^}/' "$BS")"
|
||||
[ -n "$fcblock" ] || { echo "FAIL: no apply_felis_config_secrets found in $BS"; exit 1; }
|
||||
[ "$(printf '%s\n' "$fcblock" | wc -l)" -lt 20 ] \
|
||||
|| { echo "FAIL: the extracted block is not the function -- did its closing brace move?"; exit 1; }
|
||||
|
||||
kubcalls="$(mktemp)"
|
||||
run_fc() {
|
||||
: > "$kubcalls"
|
||||
CONTROL_NS=felis MINECRAFT_NS=minecraft STATE_DIR=/tmp/fc KUBCALLS="$kubcalls" bash -c '
|
||||
kube() { printf "%s\n" "$*" >> "$KUBCALLS"; }
|
||||
'"$fcblock"'
|
||||
apply_felis_config_secrets'
|
||||
cat "$kubcalls"
|
||||
}
|
||||
out="$(run_fc)"
|
||||
expect "the control plane's felis-config is applied" \
|
||||
"-n felis create secret generic felis-config" "$out"
|
||||
expect "the workload namespace's copy is applied too" \
|
||||
"-n minecraft create secret generic felis-config" "$out"
|
||||
expect "both copies render from the pod config" \
|
||||
"felis.toml=/tmp/fc/felis.pod.toml" "$out"
|
||||
applies="$(printf '%s\n' "$out" | grep -c '^apply -f -$')"
|
||||
if [ "$applies" -eq 2 ]; then
|
||||
echo "PASS both rendered copies are piped to kubectl apply"
|
||||
else
|
||||
echo "FAIL: expected 2 applies, got $applies:"; printf '%s\n' "$out"; fails=$((fails + 1))
|
||||
fi
|
||||
rm -f "$kubcalls"
|
||||
|
||||
# --- installer re-runs keep the operator's [registry] overrides --------------------------
|
||||
# §15's upgrade path is re-running the installer, but the build-lane mirrors and the
|
||||
# uploads backend live in [registry] as hand-written keys (docs/troubleshooting.md §8e or
|
||||
# the storage wizard) that nothing in this script's inputs derives. A re-run must carry
|
||||
# them forward — without letting a stale url/build_namespace survive (installer-owned).
|
||||
|
||||
wrblock="$(awk '/^write_felis_toml\(\) \{/,/^}/' "$BS")"
|
||||
prblock="$(awk '/^persisted_registry_block\(\) \{/,/^}/' "$BS")"
|
||||
pablock="$(awk '/^persisted_archive_block\(\) \{/,/^}/' "$BS")"
|
||||
{ [ -n "$wrblock" ] && [ -n "$prblock" ] && [ -n "$pablock" ]; } \
|
||||
|| { echo "FAIL: write_felis_toml / persisted_{registry,archive}_block not found in $BS"; exit 1; }
|
||||
# The blocks quote themselves (the awk program uses single quotes), so they are
|
||||
# sourced from a file instead of being spliced into a single-quoted bash -c.
|
||||
fnfile="$(mktemp)"
|
||||
printf '%s\n%s\n%s\n' "$prblock" "$pablock" "$wrblock" > "$fnfile"
|
||||
|
||||
rdir="$(mktemp -d)"
|
||||
cat > "$rdir/felis.host.toml" <<'TOML'
|
||||
[registry]
|
||||
url = "stale.invalid:5000"
|
||||
build_namespace = "stale-ns"
|
||||
kaniko_image = "registry.felis.svc:5000/mirror/kaniko-executor:v1.24.0"
|
||||
trivy_db_repository = "registry.felis.svc:5000/mirror/trivy-db:2"
|
||||
trivy_java_db_repository = "registry.felis.svc:5000/mirror/trivy-java-db:1"
|
||||
|
||||
[registry.s3]
|
||||
endpoint = "https://s3.example"
|
||||
region = "us-east-1"
|
||||
|
||||
[archive]
|
||||
store = "tarLocal"
|
||||
local_path = "/stale/path"
|
||||
retention = "30d"
|
||||
TOML
|
||||
|
||||
run_write() { # out-file
|
||||
STATE_DIR="$rdir" OUT_TOML="$1" FNFILE="$fnfile" bash -c '
|
||||
log() { :; }
|
||||
persisted_smtp_block() { :; }
|
||||
persisted_auth_source_blocks() { :; }
|
||||
. "$FNFILE"
|
||||
FELIS_ROOT_DOMAIN=r.example.com DB_USER=u DB_PASSWORD=p DB_NAME=d MINECRAFT_NS=minecraft \
|
||||
FELIS_EGRESS_MODE=nodeport FELIS_LIMBO_IMAGE=li FELIS_LOBBY_IMAGE=lo \
|
||||
REGISTRY_URL=registry.felis.svc:5000 BUILD_NS=felis-build FELIS_ARCHIVE_LOCAL_PATH=/a \
|
||||
write_felis_toml "$OUT_TOML" 127.0.0.1'
|
||||
}
|
||||
|
||||
run_write "$rdir/out.toml"
|
||||
out="$(cat "$rdir/out.toml")"
|
||||
expect "a re-run carries the build-lane executor mirrors" \
|
||||
'kaniko_image = "registry.felis.svc:5000/mirror/kaniko-executor:v1.24.0"' "$out"
|
||||
expect "a re-run carries the trivy vulnerability-DB mirror" \
|
||||
'trivy_db_repository = "registry.felis.svc:5000/mirror/trivy-db:2"' "$out"
|
||||
expect "a re-run carries the trivy java-DB mirror" \
|
||||
'trivy_java_db_repository = "registry.felis.svc:5000/mirror/trivy-java-db:1"' "$out"
|
||||
expect "a re-run carries the [registry.s3] uploads subtable" "[registry.s3]" "$out"
|
||||
expect "the carried subtable keeps its keys" 'endpoint = "https://s3.example"' "$out"
|
||||
expect "url stays installer-owned" 'url = "registry.felis.svc:5000"' "$out"
|
||||
expect "a re-run carries the archive retention window" 'retention = "30d"' "$out"
|
||||
expect "the archive mount stays installer-owned" 'local_path = "/a"' "$out"
|
||||
case "$out" in
|
||||
*stale.invalid* | *stale-ns* | *stale/path*)
|
||||
echo "FAIL: stale installer-owned values survived the re-run"; fails=$((fails + 1)) ;;
|
||||
esac
|
||||
|
||||
cp "$rdir/out.toml" "$rdir/felis.host.toml"
|
||||
run_write "$rdir/out2.toml"
|
||||
if cmp -s "$rdir/out.toml" "$rdir/out2.toml"; then
|
||||
echo "PASS a carried-forward config converges (the second re-run is a no-op)"
|
||||
else
|
||||
echo "FAIL: carrying [registry] overrides is not idempotent"
|
||||
diff "$rdir/out.toml" "$rdir/out2.toml" | head
|
||||
fails=$((fails + 1))
|
||||
fi
|
||||
|
||||
rm -f "$fnfile"
|
||||
|
||||
# ---------------------------------------------------------------------------------------
|
||||
if [ "$fails" -eq 0 ]; then
|
||||
echo "ALL PASS"
|
||||
|
||||
@@ -347,6 +347,14 @@ spec:
|
||||
- type
|
||||
type: object
|
||||
type: array
|
||||
emptySince:
|
||||
description: |-
|
||||
EmptySince is when the operator first observed 0 online players during
|
||||
a Running phase (spec §8 idle auto-stop). It is reset when a player joins
|
||||
or the server stops, so the empty-duration counter starts fresh each time
|
||||
the server becomes unoccupied.
|
||||
format: date-time
|
||||
type: string
|
||||
endpoint:
|
||||
description: Endpoint is where the proxy should route traffic.
|
||||
properties:
|
||||
|
||||
+25
-91
@@ -1,30 +1,28 @@
|
||||
#!/bin/bash
|
||||
# demo-up.sh — one-shot Felis demo bring-up.
|
||||
#
|
||||
# Collapses the four manual steps (bootstrap -> build/import limbo+lobby images ->
|
||||
# edit felis.toml -> felis setup) into a single command:
|
||||
#
|
||||
# sudo bash deploy/demo-up.sh
|
||||
#
|
||||
# It ends by exec'ing the interactive `felis setup` TUI (create the Owner account) —
|
||||
# that human step is the only thing this script cannot do for you.
|
||||
# Every piece a demo box needs — base platform (k3s + felis + docker + cloudflared +
|
||||
# control plane), the limbo/lobby/paper images, the felis-velocity proxy plugin, and
|
||||
# the [velocity] wiring in felis.host.toml — is built by deploy/bootstrap.sh. This
|
||||
# wrapper adds only the one step the installer cannot do: the interactive
|
||||
# `felis setup` TUI that creates the Owner account.
|
||||
#
|
||||
# Image source, in order of preference:
|
||||
# 1. Prebuilt tars at deploy/images/felis-limbo.tar + felis-lobby.tar (imported as-is).
|
||||
# 2. Otherwise built on this host with docker, resolving the LOOHP/Limbo CI jar and
|
||||
# the latest stable Paper jar automatically. Override any of:
|
||||
# LIMBO_JAR_URL LIMBO_SCHEM_URL LIMBO_VERSION PAPER_JAR_URL PAPER_JAR_SHA256
|
||||
# PAPER_MC_VERSION
|
||||
# It used to rebuild the game images here with its own copy of that logic, written
|
||||
# before bootstrap grew the job. The copy drifted: it pinned Paper 1.21.8 while the
|
||||
# installer derives one version from the Limbo login gate (both hops of a login must
|
||||
# speak one protocol), never built felis-velocity.jar (so the proxy it wired had
|
||||
# nowhere to route), and left docker running. The installer is the single origin.
|
||||
#
|
||||
# Toggles: SKIP_BOOTSTRAP=1 (base already up), SKIP_SETUP=1 (stop before the TUI).
|
||||
# Toggles: SKIP_BOOTSTRAP=1 (base + game stack already installed by a full bootstrap),
|
||||
# SKIP_SETUP=1 (stop before the TUI).
|
||||
set -Eeuo pipefail
|
||||
|
||||
SRC_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
STATE_DIR=/etc/felis
|
||||
HOST_TOML="$STATE_DIR/felis.host.toml"
|
||||
IMG_DIR="$SRC_DIR/deploy/images"
|
||||
LIMBO_IMAGE="felis-limbo:demo"
|
||||
LOBBY_IMAGE="felis-lobby:demo"
|
||||
PLUGIN_JAR=/opt/felis/velocity/plugins/felis-velocity.jar
|
||||
K3S=/usr/local/bin/k3s
|
||||
FELIS=/usr/local/bin/felis
|
||||
|
||||
@@ -33,11 +31,11 @@ die() { printf '\033[1;31mERROR: %s\033[0m\n' "$*" >&2; exit 1; }
|
||||
|
||||
[ "$(id -u)" -eq 0 ] || die "run as root (sudo bash deploy/demo-up.sh)"
|
||||
|
||||
# 1. base platform (k3s + felis + docker + cloudflared + control plane) ----------
|
||||
# 1. base platform + full game stack (deploy/bootstrap.sh) -----------------------
|
||||
if [ "${SKIP_BOOTSTRAP:-0}" = 1 ]; then
|
||||
log "SKIP_BOOTSTRAP=1 — assuming the base platform is already up"
|
||||
else
|
||||
log "bringing up the base platform (deploy/bootstrap.sh)"
|
||||
log "bringing up the base platform and the game stack (deploy/bootstrap.sh)"
|
||||
bash "$SRC_DIR/deploy/bootstrap.sh"
|
||||
fi
|
||||
|
||||
@@ -46,83 +44,19 @@ command -v "$K3S" >/dev/null 2>&1 || K3S=k3s
|
||||
command -v "$K3S" >/dev/null 2>&1 || die "k3s not found — did bootstrap complete?"
|
||||
command -v "$FELIS" >/dev/null 2>&1 || die "felis not found — did bootstrap complete?"
|
||||
|
||||
# 2. get the two game images into k3s containerd --------------------------------
|
||||
if [ -f "$IMG_DIR/felis-limbo.tar" ] && [ -f "$IMG_DIR/felis-lobby.tar" ]; then
|
||||
log "importing prebuilt image tars from $IMG_DIR"
|
||||
"$K3S" ctr images import "$IMG_DIR/felis-limbo.tar"
|
||||
"$K3S" ctr images import "$IMG_DIR/felis-lobby.tar"
|
||||
# Optional: the plain-Paper recommended base, if a tar was staged for it.
|
||||
[ -f "$IMG_DIR/felis-paper.tar" ] && "$K3S" ctr images import "$IMG_DIR/felis-paper.tar"
|
||||
else
|
||||
log "no prebuilt tars in $IMG_DIR — building on this host with docker"
|
||||
command -v docker >/dev/null 2>&1 || die "docker not found; cannot build images"
|
||||
|
||||
rel=$(curl -fsSL --max-time 30 "https://ci.loohpjames.com/job/Limbo/lastSuccessfulBuild/api/json" \
|
||||
| grep -oE 'target/Limbo-[0-9][^"]+\.jar' | head -1) || true
|
||||
: "${LIMBO_JAR_URL:=https://ci.loohpjames.com/job/Limbo/lastSuccessfulBuild/artifact/$rel}"
|
||||
: "${LIMBO_SCHEM_URL:=https://ci.loohpjames.com/job/Limbo/lastSuccessfulBuild/artifact/spawn.schem}"
|
||||
: "${LIMBO_VERSION:=$(basename "$rel" | sed -E 's/^Limbo-//; s/\.jar$//; s/-[0-9]+\.[0-9]+$//')}"
|
||||
[ -n "$rel" ] || [ -n "${LIMBO_JAR_URL##*artifact/}" ] || die "could not resolve the Limbo jar; set LIMBO_JAR_URL"
|
||||
log "building $LIMBO_IMAGE (Limbo $LIMBO_VERSION)"
|
||||
docker build -f "$SRC_DIR/deploy/limbo/Dockerfile" \
|
||||
--build-arg LIMBO_JAR_URL="$LIMBO_JAR_URL" \
|
||||
--build-arg LIMBO_SCHEM_URL="$LIMBO_SCHEM_URL" \
|
||||
--build-arg LIMBO_VERSION="$LIMBO_VERSION" \
|
||||
-t "$LIMBO_IMAGE" "$SRC_DIR"
|
||||
docker save "$LIMBO_IMAGE" | "$K3S" ctr images import -
|
||||
|
||||
: "${PAPER_MC_VERSION:=1.21.8}"
|
||||
: "${PAPER_JAR_URL:=$(curl -fsSL --max-time 30 "https://fill.papermc.io/v3/projects/paper/versions/${PAPER_MC_VERSION}/builds/latest" | grep -oE 'https://fill-data\.papermc\.io/[^"]+\.jar' | head -1)}"
|
||||
[ -n "$PAPER_JAR_URL" ] || die "could not resolve the Paper jar; set PAPER_JAR_URL"
|
||||
# Both Dockerfiles require the jar's digest. The fill-data URL is content-addressed
|
||||
# (the objects/ path segment IS the sha256), so it is derived rather than asked for;
|
||||
# a mirror override carries no such segment and must bring its own digest.
|
||||
if [ -z "${PAPER_JAR_SHA256:-}" ]; then
|
||||
sha="${PAPER_JAR_URL#*/objects/}"
|
||||
sha="${sha%%/*}"
|
||||
case "$sha" in
|
||||
*[!0-9a-f]*|"") sha="" ;;
|
||||
esac
|
||||
if [ "${#sha}" -ne 64 ]; then
|
||||
die "cannot derive the Paper jar sha256 from PAPER_JAR_URL (not a content-addressed fill-data URL); set PAPER_JAR_SHA256"
|
||||
fi
|
||||
PAPER_JAR_SHA256="$sha"
|
||||
fi
|
||||
log "building $LOBBY_IMAGE (Paper $PAPER_MC_VERSION)"
|
||||
docker build -f "$SRC_DIR/deploy/lobby/Dockerfile" \
|
||||
--build-arg PAPER_JAR_URL="$PAPER_JAR_URL" \
|
||||
--build-arg PAPER_JAR_SHA256="$PAPER_JAR_SHA256" \
|
||||
-t "$LOBBY_IMAGE" "$SRC_DIR"
|
||||
docker save "$LOBBY_IMAGE" | "$K3S" ctr images import -
|
||||
|
||||
# Plain Paper recommended base — same PAPER_JAR_URL, no plugins, no secret gate.
|
||||
: "${PAPER_IMAGE:=felis-paper:demo}"
|
||||
log "building $PAPER_IMAGE (plain Paper $PAPER_MC_VERSION, forwarding via the operator initContainer)"
|
||||
docker build -f "$SRC_DIR/deploy/paper/Dockerfile" \
|
||||
--build-arg PAPER_JAR_URL="$PAPER_JAR_URL" \
|
||||
--build-arg PAPER_JAR_SHA256="$PAPER_JAR_SHA256" \
|
||||
-t "$PAPER_IMAGE" "$SRC_DIR"
|
||||
docker save "$PAPER_IMAGE" | "$K3S" ctr images import -
|
||||
fi
|
||||
|
||||
# 3. wire the images into the config `felis setup` reads ------------------------
|
||||
log "wiring [velocity] images into $HOST_TOML"
|
||||
# SKIP_BOOTSTRAP=1 trusts an earlier run to be complete. Check that it actually left
|
||||
# the full stack behind: a base from before the game-stack installer, or one whose
|
||||
# pieces were pruned by hand, must fail here with a pointer — not present as a proxy
|
||||
# that accepts logins and routes nowhere, with nothing in any log to say why.
|
||||
[ -f "$HOST_TOML" ] || die "missing $HOST_TOML — did bootstrap run?"
|
||||
if grep -q '^\[velocity\]' "$HOST_TOML"; then
|
||||
echo " [velocity] table already present — leaving it untouched"
|
||||
else
|
||||
cat >> "$HOST_TOML" <<EOF
|
||||
grep -q '^\[velocity\]' "$HOST_TOML" \
|
||||
|| die "$HOST_TOML has no [velocity] section — re-run the installer without SKIP_BOOTSTRAP so the system servers get wired"
|
||||
[ -f "$PLUGIN_JAR" ] \
|
||||
|| die "$PLUGIN_JAR missing — this base did not finish the full installer, and a proxy without it silently routes nothing; re-run the installer without SKIP_BOOTSTRAP"
|
||||
|
||||
[velocity]
|
||||
login_image = "$LIMBO_IMAGE"
|
||||
lobby_image = "$LOBBY_IMAGE"
|
||||
EOF
|
||||
echo " appended login_image=$LIMBO_IMAGE / lobby_image=$LOBBY_IMAGE"
|
||||
fi
|
||||
|
||||
# 4. interactive Owner creation + system-server provisioning -------------------
|
||||
# 2. interactive Owner creation + system-server provisioning --------------------
|
||||
if [ "${SKIP_SETUP:-0}" = 1 ]; then
|
||||
log "SKIP_SETUP=1 — base + images + config ready. Finish with: sudo felis setup"
|
||||
log "SKIP_SETUP=1 — base + game stack ready. Finish with: sudo felis setup"
|
||||
else
|
||||
log "launching 'felis setup' — create the Owner account (this is the only interactive step)"
|
||||
exec "$FELIS" setup
|
||||
|
||||
+10
-3
@@ -90,14 +90,21 @@ docker build -f deploy/limbo/Dockerfile \
|
||||
version `2026.0.2-ALPHA` (the `-26.2` CI qualifier is not published to the
|
||||
maven repo).
|
||||
|
||||
Import into k3s and point config at it:
|
||||
Publish it into the cluster's registry and point config at it. On the node
|
||||
itself (docker treats `127.0.0.1` as insecure by default):
|
||||
|
||||
```
|
||||
docker save felis-limbo:demo | sudo k3s ctr images import -
|
||||
# felis.toml → [velocity] login_image = "felis-limbo:demo"
|
||||
docker tag felis-limbo:demo 127.0.0.1:5000/felis/limbo:demo
|
||||
docker push 127.0.0.1:5000/felis/limbo:demo
|
||||
# felis.toml → [velocity] login_image = "registry.felis.svc:5000/felis/limbo:demo"
|
||||
sudo felis setup
|
||||
```
|
||||
|
||||
The registry keys a repository by the path after the host, so pushing through a
|
||||
`kubectl -n felis port-forward svc/registry 5000:5000` from another machine is
|
||||
equivalent. Hosting the image in the registry (rather than only importing it
|
||||
into containerd) is what lets kubelet re-pull it after an image GC.
|
||||
|
||||
## Ports (handled for you)
|
||||
|
||||
The entrypoint (`deploy/limbo/entrypoint.sh`) pins Limbo's `server-port` to
|
||||
|
||||
@@ -31,8 +31,12 @@ docker build -f deploy/lobby/Dockerfile \
|
||||
--build-arg PAPER_JAR_URL=https://<mirror>/paper-1.21.x-<build>.jar \
|
||||
--build-arg PAPER_JAR_SHA256=<sha256 of that jar> \
|
||||
-t felis-lobby:demo .
|
||||
docker save felis-lobby:demo | sudo k3s ctr images import -
|
||||
# felis.toml → [velocity] lobby_image = "felis-lobby:demo"
|
||||
# Publish into the cluster's registry (on the node; docker treats 127.0.0.1 as
|
||||
# insecure by default — or through a `kubectl -n felis port-forward svc/registry
|
||||
# 5000:5000`, which is equivalent: only the path after the host matters).
|
||||
docker tag felis-lobby:demo 127.0.0.1:5000/felis/lobby:demo
|
||||
docker push 127.0.0.1:5000/felis/lobby:demo
|
||||
# felis.toml → [velocity] lobby_image = "registry.felis.svc:5000/felis/lobby:demo"
|
||||
sudo felis setup
|
||||
```
|
||||
|
||||
|
||||
+33
-7
@@ -42,9 +42,27 @@ A grep across `*.md` and `*.go` returns both sets; only the Go ones are seams.
|
||||
does not exist. `felis update` runs with a zero window, under which every
|
||||
`Scheduled` component degrades to a notify, so no path can currently claim an
|
||||
apply is under way.
|
||||
- `internal/submit/blobstore.go:40` — the uploads PVC is mounted into felis-api but
|
||||
not into the Kaniko build Pod, so a submitted context is durable at the derived
|
||||
location without yet being readable by the build that consumes it.
|
||||
- `internal/submit/blobstore.go` — CLOSED 2026-09-22. The uploads PVC still cannot
|
||||
cross namespaces, so the transport went through the API instead of a mount: the
|
||||
derived context ref is now the internal-face URL
|
||||
(`/api/v1/internal/submissions/{id}/context`, service-token gated), the build
|
||||
Job's `context-fetch` initContainer streams it with `felis fetch-context` and
|
||||
extracts under a zip-slip guard into a size-limited emptyDir, and Kaniko builds
|
||||
`--context=/context`. The token reaches the build namespace through the same
|
||||
Secret-replica mechanism the login gate uses (bootstrap + `felis setup`), and the
|
||||
build egress lock allows exactly the control namespace on the internal port.
|
||||
Uniform for local and s3:// stores — neither hands the sandboxed build Pod a
|
||||
filesystem view or object-store credentials. Kaniko/Trivy images are
|
||||
external-only by default; `[registry] kaniko_image / trivy_image /
|
||||
build_cpu_limit / build_mem_limit` override them for mirrored or air-gapped
|
||||
installs. Trivy's vulnerability DB is the same story, and now has its own knob:
|
||||
`[registry] trivy_db_repository` points `--db-repository` at an internal mirror
|
||||
(recipe in docs/troubleshooting.md §8e); `trivy_java_db_repository` does the
|
||||
same for the Java DB, which Trivy fetches so soon as the scanned image contains
|
||||
a jar — i.e. for every real modpack build. Left unset on an egress-locked box
|
||||
the scan step fails closed — Kaniko pushes, Trivy exits on the DB download —
|
||||
which is the correct fail direction but leaves the build unfinished, so the
|
||||
mirrors are part of a production build install.
|
||||
|
||||
## Built; only its I/O is unverifiable from this repo
|
||||
|
||||
@@ -58,17 +76,25 @@ or a real upstream account to run it against — not an implementation.
|
||||
drives the whole flow through a fake.
|
||||
- `internal/api/console.go:39`, `internal/api/logstream.go:236`,
|
||||
`internal/api/logstream.go:306`, `internal/fileedit/k8sjobs.go:45` — each needs a
|
||||
live cluster (RCON, `pods/log` follow, a Job).
|
||||
live cluster (RCON, `pods/log` follow, a Job). **Verified live 2026-09-22/23
|
||||
(auditfix7–25):** the RCON command spine (wake → probe → `command`/access
|
||||
mutations/stop), the log SSE stream, and the fileedit Job have each run
|
||||
end-to-end on the drill cluster.
|
||||
- `internal/api/handlers_access.go:170,490` — parsing real vanilla and LuckPerms
|
||||
command output.
|
||||
command output. **Verified live 2026-09-23:** players / whitelist / banlist
|
||||
parses matched a live Paper server's replies (LuckPerms not installed → the raw
|
||||
reply falls through as documented; the input guards held on four negative cases).
|
||||
|
||||
## Deliberately accepted, not scheduled to close
|
||||
|
||||
These are decisions, not backlog. Each names the condition under which it would be
|
||||
worth revisiting.
|
||||
|
||||
- `internal/api/pgrepo.go:281` — the quota check and `ClaimServer` are two statements
|
||||
(audit #4 TOCTOU). Closeable only against a real Postgres.
|
||||
- ~~`internal/api/pgrepo.go:281` — the quota check and `ClaimServer` are two statements
|
||||
(audit #4 TOCTOU). Closeable only against a real Postgres.~~ **Closed** — the gate
|
||||
moved inside `ClaimServer` (advisory lock + re-check + UPDATE in one transaction),
|
||||
red-then-green in the pgint suite, which is exactly the real-Postgres harness this
|
||||
line was waiting for.
|
||||
- `internal/api/api.go:671` — `cooldownLimiter` is process-local, so across N api
|
||||
replicas a caller could draw up to N OTP codes per window. The intra-replica burst
|
||||
is closed; cross-replica bounding needs a shared store, out of scope for a
|
||||
|
||||
+267
-12
@@ -337,6 +337,19 @@ components:
|
||||
build_id:
|
||||
type: string
|
||||
description: image_builds.id, set only after the build hand-off succeeds.
|
||||
build_status:
|
||||
type: string
|
||||
enum: [pending, building, succeeded, failed, cancelled]
|
||||
description: >-
|
||||
The linked build's outcome, attached by the LIST routes
|
||||
(/me/submissions, /submissions) — for a submitter this is the only
|
||||
visible outlet for a failed build. Omitted until a build is linked
|
||||
and its row is readable.
|
||||
build_error:
|
||||
type: string
|
||||
description: >-
|
||||
The build's recorded failure text (e.g. a CRITICAL CVE scan
|
||||
failure), attached alongside build_status.
|
||||
reviewed_by: { type: string }
|
||||
reject_reason: { type: string }
|
||||
created_at: { type: string, format: date-time }
|
||||
@@ -454,6 +467,26 @@ paths:
|
||||
'503':
|
||||
$ref: '#/components/responses/ServiceUnavailable'
|
||||
|
||||
/metrics:
|
||||
get:
|
||||
tags: [metrics]
|
||||
operationId: metrics
|
||||
summary: Prometheus metrics (felis_* collectors) on the internal face.
|
||||
description: >-
|
||||
Scrape-only infrastructure route, not a product API: the internal listener is
|
||||
ClusterIP-only and a Prometheus scrape carries no token, the same stance as the
|
||||
probes. Serves the felis_* exposition documented in troubleshooting §14; the
|
||||
external face never serves it.
|
||||
x-felis-face: [internal]
|
||||
x-felis-tier: public
|
||||
security: []
|
||||
responses:
|
||||
'200':
|
||||
description: Prometheus text exposition format.
|
||||
content:
|
||||
text/plain:
|
||||
schema: { type: string }
|
||||
|
||||
/session/minecraft/hasJoined:
|
||||
get:
|
||||
tags: [nano]
|
||||
@@ -607,6 +640,34 @@ paths:
|
||||
'404':
|
||||
$ref: '#/components/responses/NotFound'
|
||||
|
||||
/api/v1/internal/submissions/{id}/context:
|
||||
get:
|
||||
tags: [submissions-internal]
|
||||
operationId: internalSubmissionContext
|
||||
summary: Stream a submission's stored build-context tarball to the build Pod.
|
||||
description: >-
|
||||
The build Job's fetch initContainer cannot mount the control-plane uploads
|
||||
PVC (a PVC does not cross namespaces) and holds no object-store
|
||||
credentials, so the API that stored the blob streams it here. Served on
|
||||
the internal face (service token, no Zero Trust).
|
||||
x-felis-face: [internal]
|
||||
x-felis-tier: service
|
||||
security: [{ serviceToken: [] }]
|
||||
parameters:
|
||||
- { name: id, in: path, required: true, schema: { type: string } }
|
||||
responses:
|
||||
'200':
|
||||
description: The stored gzip tarball, verbatim.
|
||||
content:
|
||||
application/gzip:
|
||||
schema: { type: string, format: binary }
|
||||
'401':
|
||||
$ref: '#/components/responses/Unauthorized'
|
||||
'404':
|
||||
$ref: '#/components/responses/NotFound'
|
||||
'503':
|
||||
$ref: '#/components/responses/ServiceUnavailable'
|
||||
|
||||
/api/v1/internal/servers/{name}/join-event:
|
||||
post:
|
||||
tags: [servers-internal]
|
||||
@@ -2690,6 +2751,13 @@ paths:
|
||||
The owner's display identity (email, or username when
|
||||
the address is absent). Absent for an unclaimed server
|
||||
or when the best-effort owner lookup failed.
|
||||
system:
|
||||
type: boolean
|
||||
description: >-
|
||||
True for a platform-provisioned system service (the login
|
||||
gate, the lobby). Their reserved names are rejected by
|
||||
every per-server route, so the cockpit renders them
|
||||
read-only instead of offering actions that would 400.
|
||||
'401':
|
||||
$ref: '#/components/responses/Unauthorized'
|
||||
'403':
|
||||
@@ -2771,11 +2839,12 @@ paths:
|
||||
post:
|
||||
tags: [backups]
|
||||
operationId: backupNow
|
||||
summary: Back up a server's world on demand (owner-or-admin; server must be stopped).
|
||||
summary: Back up a server's data volume on demand (owner-or-admin; server must be stopped).
|
||||
description: >-
|
||||
Snapshots the server's world into the archive store as a first-class
|
||||
world_backups row (reason "manual"), restorable later like an inactivity
|
||||
backup. The world PVC is RWO and held by a running server, so the server must
|
||||
Snapshots the server's whole data volume (worlds, config, plugins/mods,
|
||||
jars, libraries — not just world folders) into the archive store as a
|
||||
first-class world_backups row (reason "manual"), restorable later like an
|
||||
inactivity backup. A restore replaces the volume with the archive. The world PVC is RWO and held by a running server, so the server must
|
||||
be fully stopped first (409 not_stopped otherwise). The backup runs
|
||||
asynchronously as a Job, so success is 202 (backing_up).
|
||||
x-felis-face: [external]
|
||||
@@ -2811,6 +2880,56 @@ paths:
|
||||
'503':
|
||||
$ref: '#/components/responses/ServiceUnavailable'
|
||||
|
||||
# -------------------------------------------------- async job status (app) ---
|
||||
/api/v1/servers/{name}/jobs:
|
||||
get:
|
||||
tags: [backups]
|
||||
operationId: listServerJobs
|
||||
summary: Latest async world operations (backup/restore) for a server (owner-or-admin).
|
||||
description: >-
|
||||
Backup and restore run as cluster Jobs, so a 202 that later failed left
|
||||
its only trace in the Job object. This route projects the newest such
|
||||
Jobs, newest first, so failures are observable without kubectl. State is
|
||||
"running" | "succeeded" | "failed".
|
||||
x-felis-face: [external]
|
||||
x-felis-tier: app
|
||||
security: [{ accessJWT: [] }]
|
||||
parameters:
|
||||
- { name: name, in: path, required: true, schema: { type: string } }
|
||||
responses:
|
||||
'200':
|
||||
description: The server's newest backup/restore jobs.
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
type: object
|
||||
required: [server, jobs]
|
||||
properties:
|
||||
server: { type: string }
|
||||
jobs:
|
||||
type: array
|
||||
items:
|
||||
type: object
|
||||
required: [name, kind, state]
|
||||
properties:
|
||||
name: { type: string }
|
||||
kind: { type: string, enum: [backup, restore] }
|
||||
state: { type: string, enum: [running, succeeded, failed] }
|
||||
message: { type: string }
|
||||
started_at: { type: string, format: date-time }
|
||||
finished_at: { type: string, format: date-time }
|
||||
'401':
|
||||
$ref: '#/components/responses/Unauthorized'
|
||||
'403':
|
||||
$ref: '#/components/responses/Forbidden'
|
||||
'404':
|
||||
description: Unknown server.
|
||||
content:
|
||||
application/json:
|
||||
schema: { $ref: '#/components/schemas/Error' }
|
||||
'503':
|
||||
$ref: '#/components/responses/ServiceUnavailable'
|
||||
|
||||
# ------------------------------------------------- server file editor (app) ---
|
||||
/api/v1/servers/{name}/files:
|
||||
get:
|
||||
@@ -4147,18 +4266,30 @@ paths:
|
||||
$ref: '#/components/responses/BadRequest'
|
||||
'401':
|
||||
$ref: '#/components/responses/Unauthorized'
|
||||
'403':
|
||||
description: >-
|
||||
The per-user submission allowance is spent — too many of the
|
||||
caller's submissions are awaiting review, or their stored-upload
|
||||
budget is full (submission_quota_exceeded).
|
||||
'429':
|
||||
description: >-
|
||||
A submission was created within the per-user cooldown window
|
||||
(submission_cooldown).
|
||||
'503':
|
||||
$ref: '#/components/responses/ServiceUnavailable'
|
||||
get:
|
||||
tags: [submissions]
|
||||
operationId: mySubmissions
|
||||
summary: List the caller's own modpack submissions (user-directed lane over §16).
|
||||
summary: List the caller's own modpack submissions with each linked build's outcome (user-directed lane over §16).
|
||||
x-felis-face: [external]
|
||||
x-felis-tier: app
|
||||
security: [{ accessJWT: [] }]
|
||||
responses:
|
||||
'200':
|
||||
description: The caller's submissions, newest first.
|
||||
description: >-
|
||||
The caller's submissions, newest first; rows with a linked build
|
||||
additionally carry build_status/build_error so the submitter can see
|
||||
whether their build succeeded or failed (and why).
|
||||
content:
|
||||
application/json:
|
||||
schema:
|
||||
@@ -4185,8 +4316,10 @@ paths:
|
||||
principal; a submission the caller does not own is reported as 404, so
|
||||
this endpoint cannot upload to or probe another user's submission. Only a
|
||||
pending_review submission accepts a context (409 otherwise); a wrong-format
|
||||
or oversize body is rejected with 400. Returns 503 when the deployment's
|
||||
context store has no implemented upload transport.
|
||||
or oversize body is rejected with 400, and an upload that would push the
|
||||
caller past their per-user stored-context budget is refused with 403
|
||||
before the excess is persisted. Returns 503 when the deployment's context
|
||||
store has no implemented upload transport.
|
||||
x-felis-face: [external]
|
||||
x-felis-tier: app
|
||||
security: [{ accessJWT: [] }]
|
||||
@@ -4207,10 +4340,53 @@ paths:
|
||||
$ref: '#/components/responses/BadRequest'
|
||||
'401':
|
||||
$ref: '#/components/responses/Unauthorized'
|
||||
'403':
|
||||
description: >-
|
||||
The upload would exceed the caller's per-user stored-context budget
|
||||
(submission_quota_exceeded).
|
||||
'404':
|
||||
$ref: '#/components/responses/NotFound'
|
||||
'409':
|
||||
$ref: '#/components/responses/Conflict'
|
||||
'429':
|
||||
description: >-
|
||||
An upload was accepted within the per-user cooldown window
|
||||
(submission_cooldown).
|
||||
'503':
|
||||
$ref: '#/components/responses/ServiceUnavailable'
|
||||
|
||||
/api/v1/me/submissions/{id}:
|
||||
delete:
|
||||
tags: [submissions]
|
||||
operationId: withdrawSubmission
|
||||
summary: Withdraw your own pending submission (user side; user-directed lane over §16).
|
||||
description: >-
|
||||
Retracts the caller's own submission while it is still pending review:
|
||||
the row and its uploaded build context are deleted, freeing the pending
|
||||
slot and the per-user storage budget for a fresh submission. A reviewed
|
||||
submission is frozen (409 — its build may already be consuming the
|
||||
context), and a submission the caller does not own reads back as 404, so
|
||||
this endpoint cannot probe or clear another user's uploads.
|
||||
x-felis-face: [external]
|
||||
x-felis-tier: app
|
||||
security: [{ accessJWT: [] }]
|
||||
parameters:
|
||||
- { name: id, in: path, required: true, schema: { type: string } }
|
||||
responses:
|
||||
'200':
|
||||
description: The withdrawn submission, as it was before the deletion.
|
||||
content:
|
||||
application/json:
|
||||
schema: { $ref: '#/components/schemas/Submission' }
|
||||
'401':
|
||||
$ref: '#/components/responses/Unauthorized'
|
||||
'404':
|
||||
$ref: '#/components/responses/NotFound'
|
||||
'409':
|
||||
description: Submission has already been reviewed and cannot be withdrawn.
|
||||
content:
|
||||
application/json:
|
||||
schema: { $ref: '#/components/schemas/Error' }
|
||||
'503':
|
||||
$ref: '#/components/responses/ServiceUnavailable'
|
||||
|
||||
@@ -4285,10 +4461,24 @@ paths:
|
||||
type: object
|
||||
required: [image_ref, dockerfile, context_ref]
|
||||
properties:
|
||||
image_ref: { type: string }
|
||||
dockerfile: { type: string }
|
||||
context_ref: { type: string }
|
||||
base_image: { type: string }
|
||||
image_ref:
|
||||
type: string
|
||||
description: Push target under the internal registry (e.g. registry.felis.svc:5000/foo:1.0).
|
||||
dockerfile:
|
||||
type: string
|
||||
description: >-
|
||||
Audit archive of the recipe, recorded on the build row and shown in the
|
||||
panel — the executed Dockerfile is the file named `Dockerfile` at the
|
||||
root of the context tarball (Kaniko runs --dockerfile=Dockerfile), so
|
||||
this field is never executed.
|
||||
context_ref:
|
||||
type: string
|
||||
description: >-
|
||||
Location of the uploaded gzip build context; its root must contain the
|
||||
Dockerfile that gets executed.
|
||||
base_image:
|
||||
type: string
|
||||
description: Resolved FROM, recorded for audit only — not a build gate.
|
||||
responses:
|
||||
'202':
|
||||
description: Build accepted.
|
||||
@@ -4522,6 +4712,39 @@ paths:
|
||||
'503':
|
||||
$ref: '#/components/responses/ServiceUnavailable'
|
||||
|
||||
/api/v1/submissions/{id}/context:
|
||||
get:
|
||||
tags: [submissions]
|
||||
operationId: downloadSubmissionContext
|
||||
summary: Download a submission's uploaded build context (admin; user-directed lane over §16).
|
||||
description: >-
|
||||
The reviewer's read path to the artifact they are about to approve: the
|
||||
executed Dockerfile lives inside this tarball (Kaniko runs the context's
|
||||
root `Dockerfile`), so without it the human gate would be blind. Streams
|
||||
the stored context.tar.gz verbatim with an attachment disposition — the
|
||||
same bytes the build Pod fetches over the internal face. 404 when the
|
||||
submission is unknown or has no uploaded context; 503 when the
|
||||
deployment's context store has no implemented transport.
|
||||
x-felis-face: [external]
|
||||
x-felis-tier: admin
|
||||
security: [{ accessJWT: [] }]
|
||||
parameters:
|
||||
- { name: id, in: path, required: true, schema: { type: string } }
|
||||
responses:
|
||||
'200':
|
||||
description: The stored build context (gzip tarball), served as an attachment.
|
||||
content:
|
||||
application/gzip:
|
||||
schema: { type: string, format: binary }
|
||||
'401':
|
||||
$ref: '#/components/responses/Unauthorized'
|
||||
'403':
|
||||
$ref: '#/components/responses/Forbidden'
|
||||
'404':
|
||||
$ref: '#/components/responses/NotFound'
|
||||
'503':
|
||||
$ref: '#/components/responses/ServiceUnavailable'
|
||||
|
||||
/api/v1/submissions/{id}/reject:
|
||||
post:
|
||||
tags: [submissions]
|
||||
@@ -4562,3 +4785,35 @@ paths:
|
||||
schema: { $ref: '#/components/schemas/Error' }
|
||||
'503':
|
||||
$ref: '#/components/responses/ServiceUnavailable'
|
||||
|
||||
/api/v1/submissions/{id}:
|
||||
delete:
|
||||
tags: [submissions]
|
||||
operationId: deleteSubmission
|
||||
summary: Retire a submission outright — row and uploaded context (admin; user-directed lane over §16).
|
||||
description: >-
|
||||
Removes the submission and its uploaded build context, any status — the
|
||||
lane's only lifecycle valve, and the path that reclaims a rejected or
|
||||
consumed upload from the uploads PVC. The reviewer identity is recorded
|
||||
in the audit event, not on the (now deleted) row. Deleting an approved
|
||||
submission whose build is still running fails that build's context
|
||||
fetch; the admin has explicitly chosen to retire the artifact.
|
||||
x-felis-face: [external]
|
||||
x-felis-tier: admin
|
||||
security: [{ accessJWT: [] }]
|
||||
parameters:
|
||||
- { name: id, in: path, required: true, schema: { type: string } }
|
||||
responses:
|
||||
'200':
|
||||
description: The deleted submission, as it was before the deletion.
|
||||
content:
|
||||
application/json:
|
||||
schema: { $ref: '#/components/schemas/Submission' }
|
||||
'401':
|
||||
$ref: '#/components/responses/Unauthorized'
|
||||
'403':
|
||||
$ref: '#/components/responses/Forbidden'
|
||||
'404':
|
||||
$ref: '#/components/responses/NotFound'
|
||||
'503':
|
||||
$ref: '#/components/responses/ServiceUnavailable'
|
||||
@@ -87,7 +87,7 @@ sequenceDiagram
|
||||
else quota available
|
||||
Repo-->>API: true
|
||||
API->>Repo: ClaimServer(name, user_id)
|
||||
Note over Repo: SELECT EXISTS(server); then atomic UPDATE servers SET owner_id=$2, claimed_at=now() WHERE name=$1 AND owner_id IS NULL AND deleted_at IS NULL
|
||||
Note over Repo: ONE transaction: pg_advisory_xact_lock(user_id) serializes this user's claim lane; SELECT FROM servers WHERE name=$1 AND deleted_at IS NULL FOR UPDATE; re-run the four-dimension quota gate (authoritative — the pre-check above is a fast path); then UPDATE servers SET owner_id=$2, claimed_at=now() WHERE name=$1 AND owner_id IS NULL AND deleted_at IS NULL
|
||||
alt server missing
|
||||
Repo-->>API: ErrNotFound
|
||||
API-->>Panel: 404 not_found
|
||||
@@ -137,14 +137,14 @@ sequenceDiagram
|
||||
Panel->>APIExternal: POST /api/v1/account/link/verify {code}
|
||||
APIExternal->>APIExternal: trim and uppercase code
|
||||
APIExternal->>Repo: VerifyLinkCode(user_id, code, now)
|
||||
Repo->>Repo: SELECT non-expired code
|
||||
Repo->>Repo: SELECT mc_uuid, auth_source FROM account_link_codes WHERE code=$1 AND expires_at>$2 FOR UPDATE
|
||||
alt missing or expired code
|
||||
Repo-->>APIExternal: ErrLinkCodeInvalid
|
||||
APIExternal-->>Panel: 400 invalid_code
|
||||
else UUID linked to another user
|
||||
else UUID linked to a different, live user
|
||||
Repo-->>APIExternal: ErrConflict
|
||||
APIExternal-->>Panel: 409 already_linked
|
||||
else valid code
|
||||
else valid code (re-verify by the same user is idempotent; a retired/soft-deleted owner's link is taken over)
|
||||
Repo->>Repo: INSERT account_links(user_id, mc_uuid, auth_source) ON CONFLICT (user_id, mc_uuid) DO UPDATE auth_source
|
||||
Repo->>Repo: DELETE account_link_codes WHERE code=$1
|
||||
Repo-->>APIExternal: mc_uuid, auth_source
|
||||
|
||||
+336
-3
@@ -20,8 +20,11 @@ graded for how far the in-repo Go test suite proves the behaviour:
|
||||
containerd, Postgres, or the network, not by Felis Go code; you will see it in
|
||||
`kubectl describe` / pod logs, never in `MinecraftServer.status`.
|
||||
- **[INERT]** — the configuration field exists in the CRD but no controller
|
||||
reads it. Tuning it does nothing. §12 lists the one field this still applies
|
||||
to, alongside the fields that *are* read and the condition each depends on.
|
||||
reads it. Tuning it does nothing. **No CRD field carries this status today**;
|
||||
the last one, `spec.storage.retainOnDelete`, was removed rather than
|
||||
implemented (§13 records why). A field whose change seems ignored is almost
|
||||
always a condition instead — §12 lists the fields that *are* read and the
|
||||
condition each depends on.
|
||||
|
||||
The operator never invents the parent domain; routing identity is
|
||||
`spec.subdomain` under the deployment zone. Examples below use
|
||||
@@ -391,6 +394,95 @@ Inspect:
|
||||
kubectl logs -n felis-build job/<build-job>
|
||||
```
|
||||
|
||||
### 8e. Build Pods never start: executor images and air-gapped installs
|
||||
|
||||
The build Job runs Kaniko and Trivy from external registries by default
|
||||
(`gcr.io/kaniko-project/executor:latest`, `aquasec/trivy:latest`). On a box whose
|
||||
build namespace cannot reach those registries (the egress policy allows only
|
||||
DNS, the internal registry and `--package-cidr` mirrors — and an air-gapped box
|
||||
has no route at all), the Pods sit in `ImagePullBackOff`/`ErrImagePull` and the
|
||||
build stays `building` until its deadline. Point the overrides at images **in
|
||||
the internal registry** — the one pull source that survives an image GC (a bare
|
||||
node-containerd import does not: kubelet's image GC collects unused images under
|
||||
disk pressure, and an air-gapped box then has nothing to restore them from) —
|
||||
in `felis.toml`:
|
||||
|
||||
```toml
|
||||
[registry]
|
||||
url = "registry.felis.svc:5000"
|
||||
build_namespace = "felis-build"
|
||||
kaniko_image = "registry.felis.svc:5000/mirror/kaniko-executor:v1.24.0"
|
||||
trivy_image = "registry.felis.svc:5000/mirror/trivy:0.74.0"
|
||||
trivy_db_repository = "registry.felis.svc:5000/mirror/trivy-db:2"
|
||||
trivy_java_db_repository = "registry.felis.svc:5000/mirror/trivy-java-db:1"
|
||||
build_cpu_limit = "2"
|
||||
build_mem_limit = "4Gi"
|
||||
```
|
||||
|
||||
Mirror the executor images into the registry once. On the node itself, push
|
||||
through the loopback hostPort the registry Deployment binds (docker treats
|
||||
`127.0.0.1` as insecure by default; the installer leaves the daemon stopped, so
|
||||
`sudo systemctl start docker` first):
|
||||
|
||||
```sh
|
||||
docker pull gcr.io/kaniko-project/executor:v1.24.0 # any versions you pin
|
||||
docker pull aquasec/trivy:0.74.0
|
||||
docker pull mirror.gcr.io/aquasec/trivy-java-db:1
|
||||
docker tag gcr.io/kaniko-project/executor:v1.24.0 127.0.0.1:5000/mirror/kaniko-executor:v1.24.0
|
||||
docker tag aquasec/trivy:0.74.0 127.0.0.1:5000/mirror/trivy:0.74.0
|
||||
docker tag mirror.gcr.io/aquasec/trivy-java-db:1 127.0.0.1:5000/mirror/trivy-java-db:1
|
||||
docker push 127.0.0.1:5000/mirror/kaniko-executor:v1.24.0
|
||||
docker push 127.0.0.1:5000/mirror/trivy:0.74.0
|
||||
docker push 127.0.0.1:5000/mirror/trivy-java-db:1
|
||||
```
|
||||
|
||||
From another machine, port-forward the registry instead (`kubectl -n felis
|
||||
port-forward svc/registry 5000:5000`) and push to `localhost:5000/...` — the
|
||||
registry keys a repository by the path after the host, so pushes through either
|
||||
door land in the same place the build Pods will pull from.
|
||||
|
||||
Put them in **both** `/etc/felis/felis.host.toml` (host-side CLI) and
|
||||
`/etc/felis/felis.pod.toml` (the file rendered into the API's `felis-config`
|
||||
Secret — the two differ only in the database URL; the setup screens re-render
|
||||
the Secret from the pod file, so edits made only through `kubectl` on the live
|
||||
Secret are lost at the next reconfigure). A Deployment restart alone is NOT
|
||||
enough — the API Pod mounts the Secret, never the host file. Re-render the
|
||||
Secret from the pod file, then roll `felis-api`:
|
||||
|
||||
```sh
|
||||
kubectl -n felis create secret generic felis-config \
|
||||
--from-file=felis.toml=/etc/felis/felis.pod.toml --dry-run=client -o yaml | kubectl apply -f -
|
||||
kubectl -n felis rollout restart deployment/felis-api
|
||||
```
|
||||
|
||||
Unset fields keep the defaults.
|
||||
|
||||
`trivy_db_repository` is not optional on an egress-locked box. Trivy fetches its
|
||||
vulnerability DB from `mirror.gcr.io`/`ghcr.io` unless told otherwise, and the
|
||||
build egress policy denies those hosts — so the scan step fails closed
|
||||
(`failed to download vulnerability DB`) and NO build ever completes, even though
|
||||
Kaniko pushed the image. Mirror the DB into the internal registry once:
|
||||
|
||||
```
|
||||
# On the node (docker treats 127.0.0.1 as insecure by default), or through the
|
||||
# port-forward above:
|
||||
# docker pull mirror.gcr.io/aquasec/trivy-db:2
|
||||
# docker tag mirror.gcr.io/aquasec/trivy-db:2 127.0.0.1:5000/mirror/trivy-db:2
|
||||
# docker push 127.0.0.1:5000/mirror/trivy-db:2
|
||||
```
|
||||
|
||||
The Job's Trivy container already runs with `--insecure`, so the internal
|
||||
registry's plain HTTP works for the DB pull exactly as it does for the scanned
|
||||
image. Re-mirror the tag periodically (Trivy refreshes the DB several times a
|
||||
day upstream; a stale mirror only means stale CVE data, never a failed gate).
|
||||
|
||||
`trivy_java_db_repository` is the same story one step lazier: Trivy downloads
|
||||
the Java DB on demand the first time it scans an image containing Java
|
||||
artifacts — every real modpack — and that download fails closed too. Mirror
|
||||
`mirror.gcr.io/aquasec/trivy-java-db:1` alongside the vulnerability DB (commands
|
||||
above); the Java DB refreshes far less often than the vulnerability DB, so a
|
||||
one-off mirror is usually fine.
|
||||
|
||||
---
|
||||
|
||||
## 9. Registry push/pull failures (spec §15)
|
||||
@@ -408,6 +500,16 @@ control namespace (or `--registry-namespace`):
|
||||
- **Storage:** PVC is RWO, `10Gi`, mounted at `/var/lib/registry`, **no
|
||||
`storageClassName`** → binds the cluster default class. If the cluster has no
|
||||
default StorageClass the PVC stays `Pending` and the registry never starts.
|
||||
- **Node-side pulls:** containerd cannot dial the Service VIP (the live stack
|
||||
answered "Empty reply"), so the registry Deployment binds a loopback hostPort
|
||||
(`127.0.0.1:<port>`) and the installer writes a `/etc/rancher/k3s/registries.yaml`
|
||||
mirror relaying `registry.<ns>.svc:<port>` onto it. That pair is what lets
|
||||
kubelet re-pull a garbage-collected image; both halves must survive together
|
||||
(remove either and every pull after an image GC fails).
|
||||
- **Memory:** the registry's limit is 2Gi, deliberately larger than the other
|
||||
control-plane pods' 256Mi — a live 475MB-layer push OOM-killed the 256Mi
|
||||
template mid-upload (audit #46). Very large layers need headroom here, not
|
||||
more CPU.
|
||||
- **Selector quirk worth knowing:** the registry Service selector is only
|
||||
`name + component=registry` — it deliberately lacks the
|
||||
`part-of=felis-control-plane` label, so the registry is *invisible* to the
|
||||
@@ -426,6 +528,16 @@ reaps a world only when `now - last_active_at > 15d` (`inactive_15d`); the 15-da
|
||||
deadline is **hard-fixed in code** (only `warn_before` / `retention` /
|
||||
`max_local_bytes` are configurable from `felis.toml [archive]`).
|
||||
|
||||
### What a "backup" contains
|
||||
|
||||
A backup tars the server's ENTIRE data volume — the same volume the server mounts
|
||||
at `/data`: world folders, `server.properties`, plugins/mods, configs, jars,
|
||||
libraries, logs and cache, not just the `world/` directory. A restore replaces the
|
||||
volume's contents with the archive (files added since the backup are pruned), so a
|
||||
restore also rolls config/plugin changes back. Sizes are dominated by
|
||||
libraries/cache on stock Paper servers (~170MB for a fresh instance before any
|
||||
world growth) — do not size the archive PVC as if only world data were stored.
|
||||
|
||||
### The backup-before-delete invariant
|
||||
|
||||
The reap sequence (all [GO-TESTED] hermetically) preserves the world unless a
|
||||
@@ -449,6 +561,29 @@ So a missing backup never results in a deleted world. [GO-TESTED:
|
||||
- CRD missing → logs `reaper: CRD missing, skipping`, skipped.
|
||||
- Idle `≤ 15d` → not yet eligible.
|
||||
|
||||
### Pre-reap warnings (the `warn_before` offsets)
|
||||
|
||||
An OWNED server inside a warning window gets an email notice (`3d`/`1d` before
|
||||
the deadline, `warn_before` from `[archive]`) to the owner's **verified** email —
|
||||
the same `[smtp]` relay felis-api uses. The `warned_3d_at` / `warned_1d_at`
|
||||
stamps record a **delivered** notice:
|
||||
|
||||
- No `[smtp]` configured (or owner has no verified address): the run logs
|
||||
`reaper: warning suppressed — no warner wired` / a delivery error and does
|
||||
NOT stamp. Nothing is falsely recorded as sent, and the day SMTP is
|
||||
configured the pending warning can still go out.
|
||||
- Delivery failure (relay down): logged and retried on the next daily run —
|
||||
bounded by the warning window, since the reap removes the candidate anyway.
|
||||
- `warned=` in the run output counts DELIVERED notices, not attempts.
|
||||
|
||||
The reaper runs in the minecraft namespace and reads the **mirrors** of
|
||||
`felis-smtp` and `felis-config` there (a `secretKeyRef` is namespace-local). The
|
||||
installed `felis setup`'s "configure email" screen refreshes both mirrors when it
|
||||
applies, so configuring SMTP after install is enough; a manual edit of the
|
||||
control-namespace Secret alone is not. [GO-TESTED: the delivered/retried/
|
||||
suppressed matrix in `internal/reaper`; live-drilled end to end against a local
|
||||
SMTP sink.]
|
||||
|
||||
### Genuine false-delete risk vectors
|
||||
|
||||
- **Stale `last_active_at`.** The keep-alive is `RecordJoin`, called from the
|
||||
@@ -468,6 +603,32 @@ Only `TarLocal` (tar+gzip) archiving is implemented; VolumeSnapshot/Longhorn
|
||||
backends return `not implemented in this build`. The live PVC delete / Postgres
|
||||
store paths are [INTEGRATION-ONLY].
|
||||
|
||||
### Where worlds are read from (hostPath resolution)
|
||||
|
||||
The CronJob mounts `--worlds-host-path` read-only at `/worlds`; the resolver
|
||||
runs `cmd/felis/reaper.resolveWorldDir`: it looks for `<root>/<pvc>`, then for
|
||||
the stock local-path directory `<root>/<pv-name>_<ns>_<pvc-name>` derived from
|
||||
the live PVC's `spec.volumeName` (never a glob — a leftover directory of a
|
||||
deleted PV must not stand in for the world the PVC currently binds). Pointing
|
||||
the flag at k3s's storage root (`/var/lib/rancher/k3s/storage`) is therefore the
|
||||
supported way to enable retention on a stock install. Two deployment facts the
|
||||
resolver cannot fix:
|
||||
|
||||
- **Permissions.** The reaper Pod runs as **root** and carries `DAC_OVERRIDE`:
|
||||
worlds are written by the game image's own UID (root for every Paper image we
|
||||
ship), and Paper saves `level.dat` mode-0600, so any fixed non-root identity
|
||||
(the previous uid-1000 convention, and the ACL setup that went with it) could
|
||||
neither walk the tree nor read the files — every archive failed
|
||||
`open …/level.dat: permission denied` and the same defect failed on-demand
|
||||
backups/restores. Root is the same identity the game container itself runs as
|
||||
(see the operator's forwarding-init note); `DAC_OVERRIDE` extends the archive
|
||||
to game images with a different UID. If a world is still **preserved** while a
|
||||
reap was expected, it is now a different cause: check the run's ERROR logs for
|
||||
the resolver's `lstat` messages before suspecting permissions.
|
||||
- **Node placement.** Multi-node clusters: the world's directory exists only on
|
||||
the node holding its volume, and the CronJob sets no `nodeSelector`, so add
|
||||
one (single-node starters are pinned implicitly).
|
||||
|
||||
---
|
||||
|
||||
## 11. Idle auto-stop never fires; player count always shows 0
|
||||
@@ -502,6 +663,32 @@ Both fields set and still nothing happens? Then the probe is failing rather than
|
||||
disabled: the server would be stuck in `Starting` with `RconNotReachable`
|
||||
(`reconciler.go:156`), which is §1's symptom, not this one.
|
||||
|
||||
This path used to fail even with everything configured correctly, through three
|
||||
stacked defects proven and fixed on a live cluster (auditfix21/22): the
|
||||
`emptySince` stamp was pruned by a missing CRD status field, a quiescent empty
|
||||
server produced no watch events to re-check the timer, and the Role lacked the
|
||||
`minecraftservers:patch` grant the stop write needs. If auto-stop ever looks
|
||||
dead again, check these three in order (each is now pinned by a test):
|
||||
|
||||
```sh
|
||||
# ① The stamp must persist — should print a timestamp, not an empty string,
|
||||
# a few seconds after a server goes Ready with zero players.
|
||||
kubectl get minecraftserver <name> -o jsonpath='{.status.emptySince}'
|
||||
|
||||
# ② The operator must be able to write spec.desiredState (403 in the operator
|
||||
# log = missing patch grant on Role felis-operator).
|
||||
kubectl auth can-i patch minecraftservers -n <ns> --as=system:serviceaccount:<ctl-ns>:felis-operator
|
||||
|
||||
# ③ A wake-up must be scheduled: while empty, expect whatever you set
|
||||
# as emptySecondsBeforeStop to elapse and the box to flip to Stopped without
|
||||
# any external action.
|
||||
```
|
||||
|
||||
While players are online the operator re-probes on a 30s cadence so it notices
|
||||
the moment the last one leaves; while empty it schedules a wake-up exactly at
|
||||
the deadline. Quiet operator logs on an idle server are normal — the action is
|
||||
the scheduled wake-up, not a stream of reconciles.
|
||||
|
||||
Note the reaper's `last_active_at` (§10) is a *different* subsystem (Postgres
|
||||
business layer, bumped by join events) — it keeps worlds alive against the
|
||||
reaper, but it does **not** auto-stop empty running servers.
|
||||
@@ -526,6 +713,30 @@ deadline for the whole start, applied on the RCON-probe branch.
|
||||
|
||||
---
|
||||
|
||||
## 12b. A newer field never reaches an already-installed system server (`felis converge`)
|
||||
|
||||
Provisioning is create-if-absent: `felis setup` never rewrites an existing
|
||||
`login`/`lobby` `MinecraftServer` beyond the config-derived env it owns, so a
|
||||
field the desired spec gained after your install sits absent forever — this is
|
||||
how a deployment ends up with a lobby that has no `spec.rcon` (a dead console
|
||||
and an online-player count that is always 0) and a login gate without
|
||||
`spec.startup.healthHTTPPort`. `felis converge` is the explicit pass that fills
|
||||
exactly those zero-valued fields (and re-adds a derived env key that is
|
||||
missing). It never overwrites a value that already holds one — an operator's
|
||||
RCON secretRef or tuning survives.
|
||||
|
||||
```
|
||||
sudo felis converge
|
||||
```
|
||||
|
||||
Run it **after the images are in place**. Enabling RCON, or the HTTP readiness
|
||||
gate, on a server whose image predates the listener would hold that server in
|
||||
`Starting` until the operator marks it `Failed` — that ordering is the reason
|
||||
this is a command you run rather than something setup does on every re-run.
|
||||
System servers that are already current report `already converged`.
|
||||
|
||||
---
|
||||
|
||||
## 13. World PVC survives after I deleted the MinecraftServer
|
||||
|
||||
This is expected. The world PVC is a StatefulSet `VolumeClaimTemplate`. There is
|
||||
@@ -557,6 +768,69 @@ changes, because retention was never conditional in the first place.
|
||||
|
||||
---
|
||||
|
||||
## 13b. Node runs out of disk: what survives, and how to recover
|
||||
|
||||
A full disk is the most destructive failure this stack sees: kubelet evicts game
|
||||
pods (the control plane is protected below), and its image GC then collects
|
||||
images nothing is running. The images have a pull source now — the in-cluster
|
||||
registry — so they come back without an operator re-import; freeing space is
|
||||
what completes the recovery.
|
||||
|
||||
**Eviction.** Every control-plane pod (api, operator, reaper, registry) runs
|
||||
under the BUILT-IN `system-cluster-critical` PriorityClass (value 2e9). Kubelet's
|
||||
node-pressure eviction refuses to touch those pods — the log shows
|
||||
*"Eviction manager: cannot evict a critical pod"* for each of them — while
|
||||
game-server pods at the default priority 0 are evicted first. A drill that filled
|
||||
the disk to 1.7G free saw exactly this: login/lobby evicted, the whole control
|
||||
plane still Running (before the fix the same drill evicted the api, operator and
|
||||
registry too, and the image-GC stage below followed). User-defined
|
||||
PriorityClasses cannot substitute: the API caps them at 1e9, below kubelet's
|
||||
critical threshold. The built-in class allows preemption (its policy is fixed),
|
||||
so a control-plane pod that cannot fit may preempt a game pod — deliberate: the
|
||||
management plane must be placeable.
|
||||
|
||||
**The pressure condition clears slowly.** After you free space, the node can stay
|
||||
`DiskPressure:True` for up to ~5 minutes
|
||||
(`--eviction-pressure-transition-period` defaults to 5m, to stop the condition
|
||||
flapping); pods that need scheduling wait for it. This is the bulk of the
|
||||
"recovery takes minutes" observation, not a stuck node.
|
||||
|
||||
**The images may be gone — they come back on their own.** If pods were evicted,
|
||||
the kubelet can garbage-collect their images (unused > 2 minutes under imagefs
|
||||
pressure). Every image this platform runs is ALSO hosted in the in-cluster
|
||||
registry: the installer builds each one as `registry.<ns>.svc:5000/felis/…`
|
||||
and mirrors it there, and the node's containerd is configured (a registries.yaml
|
||||
mirror onto the registry's loopback hostPort) to relay those refs back through
|
||||
it. So a GC'd image is re-pulled on the next attempt with no operator action —
|
||||
delete the stuck pod to force an immediate retry (or wait out the backoff), and
|
||||
the workload converges.
|
||||
|
||||
If a pull does NOT come back:
|
||||
|
||||
1. Free disk on the node (`df -h /var/lib/rancher`; the biggest consumers are
|
||||
`k3s ctr images ls -q` and the world/backup PVCs under
|
||||
`/var/lib/rancher/k3s/storage`).
|
||||
2. Check the registry: `kubectl -n felis get pods -l
|
||||
app.kubernetes.io/component=registry` and, on the node,
|
||||
`curl -s http://127.0.0.1:5000/v2/` (expect `{}`).
|
||||
3. Check the mirror file: `/etc/rancher/k3s/registries.yaml` must map
|
||||
`registry.felis.svc:5000` to `http://127.0.0.1:5000`. Missing or changed:
|
||||
re-run the installer (it rewrites the file and restarts k3s only when the
|
||||
content changed).
|
||||
4. Re-mirror a tag the registry does not have (hand-built images were never
|
||||
pushed): `sudo systemctl start docker` (the installer leaves the daemon
|
||||
stopped), then `docker tag <ref> 127.0.0.1:5000/<repo>:<tag> && docker push
|
||||
127.0.0.1:5000/<repo>:<tag>`.
|
||||
|
||||
For an image that is in neither place, the old fallback still stands: re-run the
|
||||
installer (it rebuilds/re-imports from the local Docker store AND mirrors into
|
||||
the registry), or for a single image
|
||||
`docker save felis:<tag> | k3s ctr images import -`. The Docker store remains a
|
||||
deliberate second copy on the node; treat it as the recovery path, not as free
|
||||
space.
|
||||
|
||||
---
|
||||
|
||||
## 14. Metrics for diagnosis (spec §23)
|
||||
|
||||
All four mandated metrics have real producers; scrape them when triaging:
|
||||
@@ -571,8 +845,65 @@ All four mandated metrics have real producers; scrape them when triaging:
|
||||
actually deleted post-backup (§10); a spike here means worlds crossed the 15d
|
||||
idle line — cross-check that join events are flowing (§10 risk vectors).
|
||||
|
||||
### Scraping
|
||||
|
||||
The series come from two processes:
|
||||
|
||||
- `felis-operator` pod `:8080/metrics` — `felis_servers_total`,
|
||||
`felis_start_duration_seconds` (no Service; scrape pod-scoped, e.g. a
|
||||
PodMonitor targeting port `metrics`).
|
||||
- `felis-api` internal face `:8081/metrics` (Service `felis-api-internal`) —
|
||||
`felis_image_build_failures_total`. Unauthenticated like the probes;
|
||||
ClusterIP-only, and the external face never serves it.
|
||||
- `felis_reaper_worlds_deleted_total` is produced inside the one-shot reaper
|
||||
CronJob, which exits long before any scrape interval — without a pushgateway
|
||||
it has no scrape path. Read the reaper Pod log or the `world_backups` table
|
||||
for deletions instead.
|
||||
|
||||
### Alert rules
|
||||
|
||||
`deploy/alerts/` ships ready-made rules: build failures, slow starts, node
|
||||
disk/memory thresholds, and the kubelet `DiskPressure` condition.
|
||||
|
||||
- Plain Prometheus: add `felis-alerts.yaml` to `rule_files`. Check and unit-test
|
||||
it standalone with `promtool check rules felis-alerts.yaml` and
|
||||
`promtool test rules felis-alerts_test.yml` (the tests pin exactly when each
|
||||
alert fires).
|
||||
- kube-prometheus-stack / prometheus-operator: `kubectl apply -f
|
||||
felis-prometheusrule.yaml` (adjust its `release:` label to your stack's
|
||||
ruleSelector).
|
||||
|
||||
---
|
||||
|
||||
## 15. Control-plane upgrades, and rolling back a bad one
|
||||
|
||||
There is no in-place updater: an upgrade is re-running the installer
|
||||
(`curl -fsSL <installer URL> | sudo bash`), which rebuilds/re-imports the image
|
||||
and re-applies the bundle. (`sudo felis setup` is not this path; on a completed
|
||||
install it only opens the config console.) The channel is not persisted across
|
||||
the re-run, so pass `FELIS_VERSION_BOOTSTRAP=dev` on a host that tracks main.
|
||||
Two properties of the control plane matter when you do:
|
||||
|
||||
- Both Deployments use strategy **Recreate** (single replica, no leader election:
|
||||
two overlapping instances would fight over the same cluster). An upgrade takes
|
||||
the panel/API down for the rollout window — seconds normally, longer if the new
|
||||
image still has to be imported.
|
||||
- If the new pod cannot start (bad tag, missing image), the installer's rollout
|
||||
wait fails after 180s and prints `kubectl describe` diagnostics: you see
|
||||
`ErrImagePull`/`ImagePullBackOff` there instead of a silent hang.
|
||||
|
||||
Roll back with:
|
||||
|
||||
```
|
||||
kubectl -n felis rollout undo deploy/felis-api
|
||||
kubectl -n felis rollout status deploy/felis-api
|
||||
```
|
||||
|
||||
(the same for `felis-operator` and `registry`). `rollout undo` returns to the
|
||||
previous ReplicaSet, whose image is normally still on the node; if the image GC
|
||||
collected it, the registry re-serves it automatically (§13b) for every tag the
|
||||
installer built — only hand-built tags need a manual re-mirror.
|
||||
|
||||
## Quick reference: symptom → section
|
||||
|
||||
| Symptom | Section |
|
||||
@@ -588,10 +919,12 @@ All four mandated metrics have real producers; scrape them when triaging:
|
||||
| Local password login rejected | §5c |
|
||||
| Internal callers 401 (service token) | §6 |
|
||||
| Link/claim 400/409/412/403/404 | §7 |
|
||||
| Build push 400 / SA denied / egress hang / Failed | §8 |
|
||||
| Build push 400 / SA denied / egress hang / Failed / executor ImagePullBackOff | §8, §8e |
|
||||
| Registry push/pull unreachable | §9 |
|
||||
| World deleted unexpectedly / backup skipped | §10 |
|
||||
| Idle auto-stop not firing; player count 0 | §11 |
|
||||
| A config field seems ignored | §12 |
|
||||
| PVC left behind after delete | §13 |
|
||||
| Node out of disk; pods evicted / ImagePullBackOff | §13b |
|
||||
| Which metric to scrape | §14 |
|
||||
| Upgrade / roll back a bad control-plane image | §15 |
|
||||
@@ -9,6 +9,7 @@ require (
|
||||
github.com/charmbracelet/huh v1.0.0
|
||||
github.com/charmbracelet/lipgloss v1.1.0
|
||||
github.com/descope/virtualwebauthn v1.0.5
|
||||
github.com/go-logr/logr v1.4.2
|
||||
github.com/go-webauthn/webauthn v0.17.4
|
||||
github.com/golang-jwt/jwt/v5 v5.3.1
|
||||
github.com/google/uuid v1.6.0
|
||||
@@ -42,7 +43,6 @@ require (
|
||||
github.com/erikgeiser/coninput v0.0.0-20211004153227-1c3628e74d0f // indirect
|
||||
github.com/evanphx/json-patch/v5 v5.9.0 // indirect
|
||||
github.com/fxamacker/cbor/v2 v2.9.2 // indirect
|
||||
github.com/go-logr/logr v1.4.2 // indirect
|
||||
github.com/go-openapi/jsonpointer v0.19.6 // indirect
|
||||
github.com/go-openapi/jsonreference v0.20.2 // indirect
|
||||
github.com/go-openapi/swag v0.22.4 // indirect
|
||||
|
||||
+52
-1
@@ -63,6 +63,11 @@ type API struct {
|
||||
// authorization boundary is exercised before the backup-Job executor is wired.
|
||||
Backuper Backuper
|
||||
|
||||
// JobStatus reads the latest backup/restore Job outcomes for GET
|
||||
// /servers/{name}/jobs — the status outlet for async failures (the enqueue
|
||||
// endpoints only answer 202). Optional: nil → that route reports 503.
|
||||
JobStatus JobStatusReader
|
||||
|
||||
// Files is the server file editor (list / read / write a file in a stopped
|
||||
// server's world volume — the "one wrong line in server.properties" repair).
|
||||
// Like Restorer and Backuper it is optional: when nil the file routes report
|
||||
@@ -118,6 +123,15 @@ type API struct {
|
||||
// on the wake lever). Zero disables throttling.
|
||||
WakeCooldown time.Duration
|
||||
|
||||
// SubmitCreateCooldown / SubmitUploadCooldown throttle the user-modpack
|
||||
// submission lane per user: create bounds how quickly review-queue rows can
|
||||
// appear, upload bounds how often a user may stream a (up to 1 GiB) build
|
||||
// context. The keys are separate, so the lane's normal shape — create, then
|
||||
// upload — is never blocked by its own throttle. Zero disables each lever
|
||||
// (the same idiom as WakeCooldown); cmd/felis wires positive values.
|
||||
SubmitCreateCooldown time.Duration
|
||||
SubmitUploadCooldown time.Duration
|
||||
|
||||
// MaxRunningServers caps how many servers may be desired-Running cluster-wide
|
||||
// (spec §9.1: the concurrency-上限 lever hanging on the same wake chokepoint as
|
||||
// cooldown and autostartPolicy). Zero — the default — disables it: §9.2 wires
|
||||
@@ -153,6 +167,9 @@ type API struct {
|
||||
otpCooldownOnce sync.Once
|
||||
otpCooldown *cooldownLimiter
|
||||
|
||||
submitCooldownOnce sync.Once
|
||||
submitCooldown *cooldownLimiter
|
||||
|
||||
streamCapOnce sync.Once
|
||||
streamCap *streamLimiter
|
||||
}
|
||||
@@ -199,6 +216,20 @@ func (a *API) otpLimiter() *cooldownLimiter {
|
||||
return a.otpCooldown
|
||||
}
|
||||
|
||||
// submitLimiter lazily builds a SEPARATE cooldown limiter for the user-modpack
|
||||
// submission lane, so its throttles never share state with the wake or OTP
|
||||
// keyspaces. One limiter backs both levers with prefixed keys (see the
|
||||
// submissionCreateKey/UploadKey constants), so create and upload never contend
|
||||
// with each other. Like the other cooldowns it is process-local; with multiple
|
||||
// api replicas the effective spacing is per-replica, the same accepted
|
||||
// KNOWN-LIMITATION the OTP resend throttle carries.
|
||||
func (a *API) submitLimiter() *cooldownLimiter {
|
||||
a.submitCooldownOnce.Do(func() {
|
||||
a.submitCooldown = &cooldownLimiter{now: a.now, last: map[string]time.Time{}}
|
||||
})
|
||||
return a.submitCooldown
|
||||
}
|
||||
|
||||
// streamGate lazily builds the per-principal SSE stream cap bound to
|
||||
// MaxStreamsPerPrincipal. A zero cap yields a disabled limiter that admits every
|
||||
// stream, so a deployment (or test) that leaves it unset pays nothing.
|
||||
@@ -259,13 +290,20 @@ type apiRoute struct {
|
||||
}
|
||||
|
||||
// internalAPIRoutes is the internal face's served route table (spec §7, §14):
|
||||
// service-token auth, never Zero Trust. It carries both health probes.
|
||||
// service-token auth, never Zero Trust. It carries both health probes and the
|
||||
// metrics scrape.
|
||||
func (a *API) internalAPIRoutes() []apiRoute {
|
||||
return []apiRoute{
|
||||
{Method: "GET", Pattern: "/healthz", Public: true, h: a.handleHealthz},
|
||||
{Method: "GET", Pattern: "/readyz", Public: true, h: a.handleReadyz},
|
||||
// Prometheus scrape (felis_* collectors); public because a scrape carries
|
||||
// no token, internal-only so it is never exposed off-cluster.
|
||||
{Method: "GET", Pattern: "/metrics", Public: true, h: a.handleMetrics},
|
||||
|
||||
{Method: "GET", Pattern: "/api/v1/servers", h: a.handleListServers},
|
||||
// The build Pod's context-fetch initContainer streams a submission's stored
|
||||
// modpack through this route (build namespace cannot mount the uploads PVC).
|
||||
{Method: "GET", Pattern: "/api/v1/internal/submissions/{id}/context", h: a.handleInternalSubmissionContext},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/servers/{name}/ready", h: a.handleReady},
|
||||
{Method: "POST", Pattern: "/api/v1/internal/servers/{name}/join-event", h: a.handleJoinEvent},
|
||||
// Domain-autostart (spec §9.1, §14): velocity drives the wake lever and polls
|
||||
@@ -407,6 +445,7 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
// owned), and restore is gated by owner-or-admin PLUS a former-owner match, so
|
||||
// neither sits behind adminOnly.
|
||||
{Method: "GET", Pattern: "/api/v1/backups", h: a.handleListBackups},
|
||||
{Method: "GET", Pattern: "/api/v1/servers/{name}/jobs", h: a.handleServerJobs},
|
||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/restore-backup", h: a.handleRestoreBackup},
|
||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/backup", h: a.handleBackupNow},
|
||||
// Server file editor: list / read / write a file in a STOPPED server's world
|
||||
@@ -479,6 +518,11 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
// App-tier and owner-scoped (the id must belong to the principal), exactly
|
||||
// like the create/list routes above.
|
||||
{Method: "POST", Pattern: "/api/v1/me/submissions/{id}/context", h: a.handleUploadSubmissionContext},
|
||||
// Withdraw the caller's OWN pending submission: the row and its uploaded
|
||||
// context are deleted, freeing the pending slot and storage budget. Same
|
||||
// owner-scoping as the upload route — a reviewed submission is frozen (409)
|
||||
// and another user's id is invisible (404).
|
||||
{Method: "DELETE", Pattern: "/api/v1/me/submissions/{id}", h: a.handleWithdrawSubmission},
|
||||
// Admin (Zero-Trust) tier: create / mutate spec / image admission. These gate
|
||||
// on Principal.IsAdmin() inside the handler via the adminOnly wrapper, so the
|
||||
// boundary is exercised even where the body is a later-phase stub.
|
||||
@@ -508,6 +552,13 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
{Method: "GET", Pattern: "/api/v1/submissions", Admin: true, h: a.handleListSubmissions},
|
||||
{Method: "POST", Pattern: "/api/v1/submissions/{id}/approve", Admin: true, h: a.handleApproveSubmission},
|
||||
{Method: "POST", Pattern: "/api/v1/submissions/{id}/reject", Admin: true, h: a.handleRejectSubmission},
|
||||
// Retire a submission outright (row + uploaded context), any status. The
|
||||
// lane's lifecycle valve: without it, rejected/consumed uploads accumulated
|
||||
// on the uploads PVC forever — there is no other delete path.
|
||||
{Method: "DELETE", Pattern: "/api/v1/submissions/{id}", Admin: true, h: a.handleDeleteSubmission},
|
||||
// The reviewer's read path to the uploaded blob: the executed Dockerfile
|
||||
// lives inside it, so approval would otherwise be blind.
|
||||
{Method: "GET", Pattern: "/api/v1/submissions/{id}/context", Admin: true, h: a.handleAdminSubmissionContext},
|
||||
// Auto-update maintenance window (spec §B; decision core internal/updates).
|
||||
// Admin-tier: it governs whether Felis may apply an update to itself, so setting
|
||||
// it requires the admin Zero-Trust path, not a mere session. API+persistence
|
||||
|
||||
+248
-14
@@ -38,8 +38,16 @@ type fakeRepo struct {
|
||||
owners map[string]string
|
||||
ownersErr error
|
||||
claimOK map[string]bool // name -> claim succeeds; absent name -> ErrNotFound
|
||||
audits []AuditEntry
|
||||
joins []string
|
||||
// claimQuotaRefuse simulates ClaimServer's atomic quota gate (audit #4)
|
||||
// refusing a name whose advisory pre-check already passed.
|
||||
claimQuotaRefuse map[string]bool
|
||||
// serverResources / resourceUpdates mirror the cached resource columns:
|
||||
// ServerResources is what the resize path reads (to preserve storage), and
|
||||
// UpdateServerResources records the write for assertions.
|
||||
serverResources map[string]ResourceSpec
|
||||
resourceUpdates map[string]ResourceSpec
|
||||
audits []AuditEntry
|
||||
joins []string
|
||||
// create-server seeding (spec §15)
|
||||
seeded map[string]bool // name -> servers row exists
|
||||
aliases map[string]string // subdomain -> bound server name
|
||||
@@ -59,6 +67,10 @@ type fakeRepo struct {
|
||||
staff map[string]*StaffUser // username -> staff login row
|
||||
sessions map[string]*fakeSession // token_hash -> session
|
||||
settings map[string][]byte // key -> jsonb value
|
||||
// failSessionUser / failGetSetting force those reads to fail with a generic
|
||||
// (non-ErrNotFound) error, simulating a store outage for the 503 auth path.
|
||||
failSessionUser error
|
||||
failGetSetting error
|
||||
// player email OTPs (spec §B2). Keyed by row id; the verify path scans for the
|
||||
// newest live (user, purpose) just as the PG query does.
|
||||
otps map[string]*fakeEmailOTP
|
||||
@@ -91,6 +103,10 @@ type fakeRepo struct {
|
||||
// user admin fakes
|
||||
seededUsers []seededUser
|
||||
fakeQuotas map[string]*QuotaView
|
||||
// deletedIDs remembers soft-deleted user ids: DeleteUser drops the row from
|
||||
// seededUsers (so listings hide it, mirroring the WHERE deleted_at IS NULL
|
||||
// query), and this set keeps the account dead for the liveness guards.
|
||||
deletedIDs map[string]bool
|
||||
// pingErr, when non-nil, is returned by Ping to simulate DB liveness check
|
||||
// failures in /readyz tests.
|
||||
pingErr error
|
||||
@@ -202,8 +218,9 @@ func newFakeRepo() *fakeRepo {
|
||||
allowlist: map[string]map[string]bool{}, allowUUID: map[string]map[string]bool{},
|
||||
mine: map[string][]MyServerView{},
|
||||
owners: map[string]string{},
|
||||
claimOK: map[string]bool{},
|
||||
seeded: map[string]bool{}, aliases: map[string]string{},
|
||||
claimOK: map[string]bool{}, claimQuotaRefuse: map[string]bool{},
|
||||
serverResources: map[string]ResourceSpec{}, resourceUpdates: map[string]ResourceSpec{},
|
||||
seeded: map[string]bool{}, aliases: map[string]string{},
|
||||
linkCodes: map[string]fakeLinkCode{}, links: map[string]string{},
|
||||
linkAuthSource: map[string]string{},
|
||||
staff: map[string]*StaffUser{},
|
||||
@@ -220,6 +237,7 @@ func newFakeRepo() *fakeRepo {
|
||||
discoverableChallenges: map[string]*fakeDiscoverableChallenge{},
|
||||
fakeQuotas: map[string]*QuotaView{},
|
||||
migrations: map[string]*fakeMigration{},
|
||||
deletedIDs: map[string]bool{},
|
||||
}
|
||||
}
|
||||
|
||||
@@ -245,10 +263,13 @@ func (f *fakeRepo) QuotaCheck(_ context.Context, userID string, _ string, _ Reso
|
||||
return f.QuotaAvailable(context.TODO(), userID)
|
||||
}
|
||||
|
||||
func (f *fakeRepo) UpdateServerResources(_ context.Context, _ string, _, _, _ int) error { return nil }
|
||||
func (f *fakeRepo) UpdateServerResources(_ context.Context, name string, cpu, mem, stor int) error {
|
||||
f.resourceUpdates[name] = ResourceSpec{CPUMilli: cpu, MemoryMB: mem, StorageMB: stor}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (f *fakeRepo) ServerResources(_ context.Context, _ string) (ResourceSpec, error) {
|
||||
return ResourceSpec{}, nil
|
||||
func (f *fakeRepo) ServerResources(_ context.Context, name string) (ResourceSpec, error) {
|
||||
return f.serverResources[name], nil
|
||||
}
|
||||
func (f *fakeRepo) CreateLinkCode(_ context.Context, code, mcUUID, authSource string, expiresAt time.Time) error {
|
||||
f.linkCodes[code] = fakeLinkCode{mcUUID: mcUUID, authSource: authSource, expiresAt: expiresAt}
|
||||
@@ -266,7 +287,13 @@ func (f *fakeRepo) VerifyLinkCode(_ context.Context, userID, code string, now ti
|
||||
return "", "", ErrLinkCodeInvalid
|
||||
}
|
||||
if existing, ok := f.links[rec.mcUUID]; ok && existing != userID {
|
||||
return "", "", ErrConflict // do not consume another user's pending code
|
||||
// A soft-deleted link's identity is unclaimed: the fresh in-game code lets a
|
||||
// live caller take it over (mirrors PGRepo). Disabled-but-not-deleted stays a
|
||||
// conflict — takeover there would bypass the lockout. Neither arm consumes
|
||||
// the code.
|
||||
if !f.seededDeleted(existing) {
|
||||
return "", "", ErrConflict
|
||||
}
|
||||
}
|
||||
f.links[rec.mcUUID] = userID
|
||||
f.linkAuthSource[rec.mcUUID] = rec.authSource // copy/refresh, mirrors DO UPDATE
|
||||
@@ -299,6 +326,9 @@ func (f *fakeRepo) RedeemPlayerBindCode(_ context.Context, newUserID, code strin
|
||||
return "", "", "", ErrPlayerBindForbidden // staff must use op.console; do not consume
|
||||
}
|
||||
}
|
||||
if f.seededDead(existing) {
|
||||
return "", "", "", ErrPlayerAccountRetired // dead account; do not consume
|
||||
}
|
||||
delete(f.linkCodes, code)
|
||||
return existing, rec.mcUUID, rec.authSource, nil
|
||||
}
|
||||
@@ -350,6 +380,13 @@ func (f *fakeRepo) VerifyEmailOTP(_ context.Context, userID, purpose, codeHash s
|
||||
live.attempts++ // a typo costs an attempt but does not consume the code
|
||||
return "", ErrOTPInvalid
|
||||
}
|
||||
// A DIFFERENT verified holder of the same address → ErrEmailTaken, code left
|
||||
// live — mirrors PGRepo's guard + the users_verified_email_unique index.
|
||||
for _, u := range f.staff {
|
||||
if u.ID != userID && u.EmailVerified && strings.EqualFold(u.Email, live.email) {
|
||||
return "", ErrEmailTaken
|
||||
}
|
||||
}
|
||||
live.consumed = true
|
||||
for _, u := range f.staff { // flip the user row verified (UPDATE users ...)
|
||||
if u.ID == userID {
|
||||
@@ -595,7 +632,7 @@ func (f *fakeRepo) UUIDInAllowlist(_ context.Context, n, uuid string) (bool, err
|
||||
return f.allowUUID[n][uuid], nil
|
||||
}
|
||||
func (f *fakeRepo) UserByMCUUID(_ context.Context, uuid string) (string, error) {
|
||||
if u, ok := f.links[uuid]; ok {
|
||||
if u, ok := f.links[uuid]; ok && !f.seededDead(u) {
|
||||
return u, nil
|
||||
}
|
||||
return "", ErrNotFound
|
||||
@@ -619,15 +656,15 @@ func (f *fakeRepo) IsUsernameBlacklisted(_ context.Context, mcUUID string) (bool
|
||||
}
|
||||
|
||||
// IsProtectedAdminLink mirrors PGRepo's JOIN of account_links to users: linked,
|
||||
// auth_source 'thirdparty', and the linked user an admin — no password-hash test, so
|
||||
// an SSO Operator (role='admin', with no password) is protected like any other.
|
||||
// auth_source 'thirdparty', and the linked user staff (admin OR owner) — no
|
||||
// password-hash test, so an SSO Operator or the Owner is protected like any other.
|
||||
func (f *fakeRepo) IsProtectedAdminLink(_ context.Context, mcUUID string) (bool, error) {
|
||||
userID, ok := f.links[mcUUID]
|
||||
if !ok || f.linkAuthSource[mcUUID] != authSourceThirdParty {
|
||||
return false, nil
|
||||
}
|
||||
for _, u := range f.staff {
|
||||
if u.ID == userID && u.Role == "admin" {
|
||||
if u.ID == userID && staffRole(u.Role) {
|
||||
return true, nil
|
||||
}
|
||||
}
|
||||
@@ -638,6 +675,9 @@ func (f *fakeRepo) ClaimServer(_ context.Context, n, u string) (bool, error) {
|
||||
if !present {
|
||||
return false, ErrNotFound
|
||||
}
|
||||
if ok && f.claimQuotaRefuse[n] {
|
||||
return false, ErrQuotaExceeded // mirrors the atomic gate losing the race
|
||||
}
|
||||
return ok, nil
|
||||
}
|
||||
func (f *fakeRepo) RecordJoin(_ context.Context, n, uuid string) error {
|
||||
@@ -768,12 +808,18 @@ func (f *fakeRepo) CreateSession(_ context.Context, tokenHash, userID string, ex
|
||||
return nil
|
||||
}
|
||||
func (f *fakeRepo) SessionUser(_ context.Context, tokenHash string, now time.Time) (*SessionedUser, error) {
|
||||
if f.failSessionUser != nil {
|
||||
return nil, f.failSessionUser
|
||||
}
|
||||
s, ok := f.sessions[tokenHash]
|
||||
if !ok || s.revoked || !s.expiresAt.After(now) {
|
||||
return nil, ErrNotFound
|
||||
}
|
||||
for _, u := range f.staff {
|
||||
if u.ID == s.userID {
|
||||
if f.seededDead(u.ID) {
|
||||
return nil, ErrNotFound
|
||||
}
|
||||
return &SessionedUser{
|
||||
ID: u.ID, Email: u.Email, Role: u.Role,
|
||||
}, nil
|
||||
@@ -788,6 +834,9 @@ func (f *fakeRepo) RevokeSession(_ context.Context, tokenHash string) error {
|
||||
return nil
|
||||
}
|
||||
func (f *fakeRepo) GetSetting(_ context.Context, key string) ([]byte, error) {
|
||||
if f.failGetSetting != nil {
|
||||
return nil, f.failGetSetting
|
||||
}
|
||||
if v, ok := f.settings[key]; ok {
|
||||
return v, nil
|
||||
}
|
||||
@@ -863,11 +912,29 @@ func (f *fakeRepo) ListUsers(_ context.Context, opts ListUsersOpts) ([]UserView,
|
||||
}
|
||||
|
||||
func (f *fakeRepo) UserDetail(_ context.Context, userID string) (*UserDetail, error) {
|
||||
deletedAt := time.Unix(1_700_000_000, 0)
|
||||
for _, su := range f.seededUsers {
|
||||
if su.view.ID == userID {
|
||||
return &su.detail, nil
|
||||
}
|
||||
}
|
||||
// Legacy fixtures seeded only into f.staff are live accounts (nothing marked
|
||||
// them disabled or deleted), so detail reads must resolve them too — the
|
||||
// liveness guards (discoverable login, owner protection) treat "unknown" as a
|
||||
// fault, and these fixtures are known.
|
||||
for _, u := range f.staff {
|
||||
if u.ID == userID {
|
||||
if f.deletedIDs[userID] {
|
||||
// A soft-deleted account still HAS a detail row; it is flagged, not gone.
|
||||
return &UserDetail{UserView: UserView{
|
||||
ID: u.ID, Username: u.Username, Role: u.Role, Disabled: true,
|
||||
}, DeletedAt: &deletedAt}, nil
|
||||
}
|
||||
return &UserDetail{UserView: UserView{
|
||||
ID: u.ID, Username: u.Username, Email: u.Email, Role: u.Role,
|
||||
}}, nil
|
||||
}
|
||||
}
|
||||
return nil, ErrNotFound
|
||||
}
|
||||
|
||||
@@ -904,6 +971,12 @@ func (f *fakeRepo) UpdateUser(_ context.Context, userID string, patch UpdateUser
|
||||
f.seededUsers[i].detail.Username = *patch.Username
|
||||
}
|
||||
if patch.Email != nil {
|
||||
// Changing the address voids the proof of it, exactly like PGRepo:
|
||||
// only VerifyEmailOTP may assert a verified address.
|
||||
if *patch.Email != f.seededUsers[i].view.Email {
|
||||
f.seededUsers[i].view.EmailVerified = false
|
||||
f.seededUsers[i].detail.EmailVerified = false
|
||||
}
|
||||
f.seededUsers[i].view.Email = *patch.Email
|
||||
f.seededUsers[i].detail.Email = *patch.Email
|
||||
}
|
||||
@@ -921,6 +994,20 @@ func (f *fakeRepo) DeleteUser(_ context.Context, userID, _ string) error {
|
||||
for i, su := range f.seededUsers {
|
||||
if su.view.ID == userID {
|
||||
f.seededUsers = append(f.seededUsers[:i], f.seededUsers[i+1:]...)
|
||||
f.deletedIDs[userID] = true
|
||||
// Mirror PGRepo: deletion severs the account's identity assets so the
|
||||
// closed account keeps neither a login credential nor a MC-UUID claim.
|
||||
for uuid, uid := range f.links {
|
||||
if uid == userID {
|
||||
delete(f.links, uuid)
|
||||
delete(f.linkAuthSource, uuid)
|
||||
}
|
||||
}
|
||||
for cid, cred := range f.passkeyCreds {
|
||||
if cred.UserID == userID {
|
||||
delete(f.passkeyCreds, cid)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
}
|
||||
@@ -1068,7 +1155,52 @@ func (f *fakeRepo) RedeemMigration(_ context.Context, targetUserID, codeHash str
|
||||
|
||||
// ---- quota admin fakes ----
|
||||
|
||||
// liveUserExists mirrors PGRepo.requireLiveUser: the admin quota/link fakes
|
||||
// only act on a live seeded row, so a unit test can drive the unknown-user 404
|
||||
// the real FK would otherwise turn into a 500.
|
||||
func (f *fakeRepo) liveUserExists(id string) bool {
|
||||
for _, su := range f.seededUsers {
|
||||
if su.view.ID == id && su.detail.DeletedAt == nil {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// seededDead mirrors PGRepo's liveness filters (audit #33): a seeded user that was
|
||||
// disabled or soft-deleted is dead for the login doors and session validation. A
|
||||
// fixture that was never seeded (legacy tests put it straight into f.staff) is
|
||||
// treated as live, matching the fakes' pre-existing behavior.
|
||||
func (f *fakeRepo) seededDead(id string) bool {
|
||||
if f.deletedIDs[id] {
|
||||
return true
|
||||
}
|
||||
for _, su := range f.seededUsers {
|
||||
if su.view.ID == id {
|
||||
return su.view.Disabled || su.detail.DeletedAt != nil
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// seededDeleted is the narrower liveness query: soft-deleted only (a disabled
|
||||
// account still holds its identity, mirroring VerifyLinkCode's takeover rule).
|
||||
func (f *fakeRepo) seededDeleted(id string) bool {
|
||||
if f.deletedIDs[id] {
|
||||
return true
|
||||
}
|
||||
for _, su := range f.seededUsers {
|
||||
if su.view.ID == id {
|
||||
return su.detail.DeletedAt != nil
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func (f *fakeRepo) GetQuotas(_ context.Context, userID string) (*QuotaView, error) {
|
||||
if !f.liveUserExists(userID) {
|
||||
return nil, ErrNotFound
|
||||
}
|
||||
v := &QuotaView{UserID: userID}
|
||||
if f.fakeQuotas == nil {
|
||||
return v, nil
|
||||
@@ -1083,6 +1215,9 @@ func (f *fakeRepo) GetQuotas(_ context.Context, userID string) (*QuotaView, erro
|
||||
}
|
||||
|
||||
func (f *fakeRepo) SetQuotas(_ context.Context, userID string, qi QuotaInput, _ string) (*QuotaView, error) {
|
||||
if !f.liveUserExists(userID) {
|
||||
return nil, ErrNotFound
|
||||
}
|
||||
if f.fakeQuotas == nil {
|
||||
f.fakeQuotas = map[string]*QuotaView{}
|
||||
}
|
||||
@@ -1135,6 +1270,9 @@ func (f *fakeRepo) UnlinkAccount(_ context.Context, userID, mcUUID string) error
|
||||
}
|
||||
|
||||
func (f *fakeRepo) LinkAccount(_ context.Context, userID, mcUUID, authSource string) error {
|
||||
if !f.liveUserExists(userID) {
|
||||
return ErrNotFound
|
||||
}
|
||||
if existing, ok := f.links[mcUUID]; ok && existing != userID {
|
||||
return ErrConflict
|
||||
}
|
||||
@@ -1153,7 +1291,7 @@ func (f *fakeRepo) LinkAccount(_ context.Context, userID, mcUUID, authSource str
|
||||
// is indistinguishable from no account: both yield ErrNotFound.
|
||||
func (f *fakeRepo) UserByEmail(_ context.Context, email string) (*StaffUser, error) {
|
||||
for _, u := range f.staff {
|
||||
if u.EmailVerified && strings.EqualFold(u.Email, email) {
|
||||
if u.EmailVerified && strings.EqualFold(u.Email, email) && !f.seededDead(u.ID) {
|
||||
su := *u
|
||||
return &su, nil
|
||||
}
|
||||
@@ -1313,6 +1451,7 @@ type fakeCluster struct {
|
||||
desired map[string]v1alpha1.DesiredState
|
||||
created map[string]CreateServerInput // name -> the validated input it was created from
|
||||
patched map[string]ServerSpecPatch // name -> the validated spec patch it received
|
||||
noWorld map[string]bool // server names modeled WITHOUT a world volume (never started / reaped)
|
||||
createErr error
|
||||
pingErr error
|
||||
}
|
||||
@@ -1320,7 +1459,7 @@ type fakeCluster struct {
|
||||
func newFakeCluster() *fakeCluster {
|
||||
return &fakeCluster{byName: map[string]*ServerInfo{}, bySub: map[string]*ServerInfo{},
|
||||
desired: map[string]v1alpha1.DesiredState{}, created: map[string]CreateServerInput{},
|
||||
patched: map[string]ServerSpecPatch{}}
|
||||
patched: map[string]ServerSpecPatch{}, noWorld: map[string]bool{}}
|
||||
}
|
||||
func (c *fakeCluster) GetServer(_ context.Context, n string) (*ServerInfo, error) {
|
||||
if s, ok := c.byName[n]; ok {
|
||||
@@ -1336,6 +1475,13 @@ func (c *fakeCluster) GetBySubdomain(_ context.Context, s string) (*ServerInfo,
|
||||
}
|
||||
func (c *fakeCluster) ListServers(_ context.Context) ([]ServerInfo, error) { return c.list, nil }
|
||||
func (c *fakeCluster) Ping(_ context.Context) error { return c.pingErr }
|
||||
|
||||
// WorldVolumeExists models the world PVC: present unless the test named the
|
||||
// server in noWorld (never started / already reaped).
|
||||
func (c *fakeCluster) WorldVolumeExists(_ context.Context, n string) (bool, error) {
|
||||
return !c.noWorld[n], nil
|
||||
}
|
||||
|
||||
func (c *fakeCluster) SetDesiredState(_ context.Context, n string, s v1alpha1.DesiredState) error {
|
||||
c.desired[n] = s
|
||||
return nil
|
||||
@@ -1671,6 +1817,44 @@ func TestFleetAdminRead(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("system services are marked read-only", func(t *testing.T) {
|
||||
// The login gate and the lobby carry reserved names, so every per-server
|
||||
// route rejects them; the fleet row must say "system" so the cockpit
|
||||
// renders them without actions that would 400.
|
||||
sysCl := newFakeCluster()
|
||||
sysCl.list = []ServerInfo{
|
||||
{Name: "login", Phase: "Running", Ready: true},
|
||||
{Name: "lobby", Phase: "Running", Ready: true},
|
||||
{Name: "survival", Phase: "Stopped"},
|
||||
}
|
||||
api := newTestAPI(newFakeRepo(), sysCl)
|
||||
api.External = staticExternal{p: &Principal{UserID: "a1", Email: "[email protected]",
|
||||
Role: "admin", ViaAdminAccess: true}}
|
||||
w := do(api.ExternalHandler(), "GET", "/api/v1/fleet", "", nil)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
var got struct {
|
||||
Servers []struct {
|
||||
Name string `json:"name"`
|
||||
System bool `json:"system"`
|
||||
} `json:"servers"`
|
||||
}
|
||||
if err := json.Unmarshal(w.Body.Bytes(), &got); err != nil {
|
||||
t.Fatalf("body not JSON: %v", err)
|
||||
}
|
||||
byName := map[string]bool{}
|
||||
for _, r := range got.Servers {
|
||||
byName[r.Name] = r.System
|
||||
}
|
||||
if !byName["login"] || !byName["lobby"] {
|
||||
t.Errorf("system flags = %+v, want login+lobby marked", byName)
|
||||
}
|
||||
if byName["survival"] {
|
||||
t.Errorf("survival marked system; only platform services are")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("owner merges for claimed, absent for unclaimed", func(t *testing.T) {
|
||||
repo := newFakeRepo()
|
||||
// Only "survival" is claimed; "creative"/"skyblock" stay unowned.
|
||||
@@ -1760,6 +1944,21 @@ func TestClaimStateMachine(t *testing.T) {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
t.Run("atomic gate refusal -> 403 quota_exceeded", func(t *testing.T) {
|
||||
// The advisory pre-check passed, but ClaimServer's serialized re-check
|
||||
// (audit #4) refuses: the caller must see the same 403, not a 500.
|
||||
repo := newFakeRepo()
|
||||
repo.linked["u1"] = true
|
||||
repo.quota["u1"] = true
|
||||
repo.claimOK["survival"] = true
|
||||
repo.claimQuotaRefuse["survival"] = true
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.External = staticExternal{p: user}
|
||||
w := do(api.ExternalHandler(), "POST", "/api/v1/servers/survival/claim", "", nil)
|
||||
if w.Code != http.StatusForbidden || decodeErr(t, w) != "quota_exceeded" {
|
||||
t.Fatalf("code = %d body %s, want 403 quota_exceeded", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
t.Run("already claimed -> 409", func(t *testing.T) {
|
||||
repo := newFakeRepo()
|
||||
repo.linked["u1"] = true
|
||||
@@ -2195,3 +2394,38 @@ func TestAccessVerifier(t *testing.T) {
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestSessionAuthOutageIs503Not401: a session-store outage must surface as 503
|
||||
// auth_unavailable, not a 401 that reads as "please log in again". Both failure
|
||||
// points are covered — the local_auth_enabled read and the session row read —
|
||||
// plus the regression that a genuinely missing session still answers 401.
|
||||
func TestSessionAuthOutageIs503Not401(t *testing.T) {
|
||||
apiWith := func(repo *fakeRepo) *API {
|
||||
a := newTestAPI(repo, newFakeCluster())
|
||||
a.External = SessionAuth{Repo: repo, RootDomain: testRoot, AdminHostname: "op.console." + testRoot}
|
||||
return a
|
||||
}
|
||||
cookie := map[string]string{"Cookie": sessionCookieName + "=any"}
|
||||
outage := errors.New("dial tcp 10.0.0.5:5432: connect: connection refused")
|
||||
|
||||
repo := newFakeRepo()
|
||||
repo.settings[LocalAuthEnabledKey] = []byte("true")
|
||||
repo.failGetSetting = outage
|
||||
if w := do(apiWith(repo).ExternalHandler(), "GET", "/api/v1/me", "", cookie); w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "auth_unavailable" {
|
||||
t.Fatalf("settings read outage = %d body %s, want 503 auth_unavailable", w.Code, w.Body.String())
|
||||
}
|
||||
|
||||
repo = newFakeRepo()
|
||||
repo.settings[LocalAuthEnabledKey] = []byte("true")
|
||||
repo.failSessionUser = outage
|
||||
if w := do(apiWith(repo).ExternalHandler(), "GET", "/api/v1/me", "", cookie); w.Code != http.StatusServiceUnavailable || decodeErr(t, w) != "auth_unavailable" {
|
||||
t.Fatalf("session read outage = %d body %s, want 503 auth_unavailable", w.Code, w.Body.String())
|
||||
}
|
||||
|
||||
// Regression: fail-closed auth (missing/invalid session) stays a 401.
|
||||
repo = newFakeRepo()
|
||||
repo.settings[LocalAuthEnabledKey] = []byte("true")
|
||||
if w := do(apiWith(repo).ExternalHandler(), "GET", "/api/v1/me", "", cookie); w.Code != http.StatusUnauthorized || decodeErr(t, w) != "unauthorized" {
|
||||
t.Fatalf("missing session = %d body %s, want 401 unauthorized", w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
@@ -17,13 +17,13 @@ type Principal struct {
|
||||
UserID string
|
||||
// Email is the audited actor identity (spec §14: audit actor = Access email).
|
||||
Email string
|
||||
// Role is "admin" or "user" (mirrors users.role).
|
||||
// Role is "owner", "admin", or "user" (mirrors users.role).
|
||||
Role string
|
||||
// ViaAdminAccess is true only when the request arrived through an admin-graded
|
||||
// path: the admin.* Zero-Trust hostname (Cloudflare Access, the remote face) OR
|
||||
// a local session presented on the op.console host (SessionAuth, the
|
||||
// passwordless face). Admin-tier operations require it in addition to
|
||||
// Role=="admin" (spec §14: ZT is graded by operation). A role=admin session
|
||||
// a staff role (spec §14: ZT is graded by operation). A staff session
|
||||
// arriving on the player console (console.*) never sets it.
|
||||
ViaAdminAccess bool
|
||||
// EmailVerified mirrors users.email_verified. The lockdown middleware gates
|
||||
@@ -49,7 +49,7 @@ func staffRole(role string) bool {
|
||||
}
|
||||
|
||||
// IsAdmin reports whether the principal may perform admin-tier operations.
|
||||
// Both the role claim and the admin Access path are required: a role=admin
|
||||
// Both the role claim and the admin Access path are required: a staff
|
||||
// session arriving on panel.* must not bypass the Zero-Trust boundary.
|
||||
// An owner implicitly passes this check (the owner role is a superset of admin).
|
||||
func (p *Principal) IsAdmin() bool {
|
||||
|
||||
@@ -81,6 +81,12 @@ type Cluster interface {
|
||||
|
||||
// GetServer reads one MinecraftServer's lifecycle view, or ErrNotFound.
|
||||
GetServer(ctx context.Context, name string) (*ServerInfo, error)
|
||||
// WorldVolumeExists reports whether the server's world PVC exists in the
|
||||
// server namespace. A server that never started — or whose world the
|
||||
// retention reaper already archived and deleted — has no claim, and a
|
||||
// backup/restore Job would hang Pending on the missing volume with nothing
|
||||
// ever recorded, so both handlers refuse those up front.
|
||||
WorldVolumeExists(ctx context.Context, name string) (bool, error)
|
||||
// GetBySubdomain finds the MinecraftServer whose spec.subdomain matches, or
|
||||
// ErrNotFound.
|
||||
GetBySubdomain(ctx context.Context, subdomain string) (*ServerInfo, error)
|
||||
|
||||
+28
-8
@@ -15,6 +15,13 @@ var (
|
||||
ErrNotFound = errors.New("not found")
|
||||
// ErrConflict means an atomic precondition failed (e.g. claim lost the race).
|
||||
ErrConflict = errors.New("conflict")
|
||||
// ErrQuotaExceeded means an ownership write would push the user over a quota
|
||||
// cap (spec §9.3). ClaimServer — the atomic gate — returns it when a claim
|
||||
// passes the handler's advisory pre-check but loses the serialized re-check
|
||||
// (two concurrent claims by one user); handlers map it to a 403
|
||||
// quota_exceeded, the same answer the pre-check gives, so the CONCURRENT case
|
||||
// and the SEQUENTIAL case are indistinguishable to the caller.
|
||||
ErrQuotaExceeded = errors.New("server quota exhausted")
|
||||
// ErrLinkCodeInvalid means an account-link code is unknown or expired (spec
|
||||
// §10). It is a client error (the verify endpoint exists; the code is bad), so
|
||||
// handlers map it to 400, not 404.
|
||||
@@ -44,14 +51,22 @@ var (
|
||||
// consumed, or expired) — so handlers map it to 400, not 404.
|
||||
ErrPasskeyChallengeInvalid = errors.New("passkey challenge invalid or expired")
|
||||
// ErrPlayerBindForbidden means a public Bind-Code redemption resolved to a STAFF
|
||||
// account (role=admin), which the player-console bootstrap refuses (console-tier
|
||||
// access model). Operators authenticate at op.console behind Zero Trust, never via
|
||||
// the account-less console.<root_domain> door, so the public bootstrap provably
|
||||
// never mints a session for an admin identity. It is distinct from ErrConflict so
|
||||
// the handler answers 403 (wrong door) rather than 409 (already linked).
|
||||
// account (admin or owner), which the player-console bootstrap refuses
|
||||
// (console-tier access model). Staff authenticate at op.console behind Zero Trust,
|
||||
// never via the account-less console.<root_domain> door, so the public bootstrap
|
||||
// provably never mints a session for a staff identity. It is distinct from
|
||||
// ErrConflict so the handler answers 403 (wrong door) rather than 409.
|
||||
ErrPlayerBindForbidden = errors.New("bind code belongs to a staff account")
|
||||
// ErrPlayerAccountRetired means a Bind-Code redemption resolved to an account the
|
||||
// platform has closed: an owner soft-deleted it, or it is disabled (locked out).
|
||||
// Reusing the row would mint a fresh session for a dead account — the same
|
||||
// resurrection the login doors refuse by resolving only live accounts — so the
|
||||
// redeemer gets an explicit 403 instead. The code is NOT consumed, so re-enabling
|
||||
// the account and retrying still works within the code's TTL.
|
||||
ErrPlayerAccountRetired = errors.New("player account is retired or disabled")
|
||||
// ErrEmailTaken means a verified email would collide with another account's
|
||||
// already-verified address (spec §B email-first login foundation, migration 0010).
|
||||
// already-verified address (spec §B email-first login foundation; the
|
||||
// users_verified_email_unique index ships in migration 0020).
|
||||
// VerifyEmailOTP returns it — WITHOUT consuming the code, since the address, not
|
||||
// the code, is the problem — when a DIFFERENT user has already proven the same
|
||||
// address case-insensitively. It is the clean, application-level counterpart of
|
||||
@@ -89,8 +104,13 @@ func newError(status int, code, format string, a ...any) *apiError {
|
||||
// Common errors reused across handlers.
|
||||
var (
|
||||
errUnauthorized = newError(http.StatusUnauthorized, "unauthorized", "authentication required")
|
||||
errForbidden = newError(http.StatusForbidden, "forbidden", "not permitted")
|
||||
errBadRequest = newError(http.StatusBadRequest, "bad_request", "invalid request")
|
||||
// errAuthUnavailable answers when the session store itself is unreachable
|
||||
// (Postgres down): an outage is not a credential verdict, so the caller gets
|
||||
// 503 "retry" instead of a 401 that reads as "log in again".
|
||||
errAuthUnavailable = newError(http.StatusServiceUnavailable, "auth_unavailable",
|
||||
"authentication is temporarily unavailable; retry shortly")
|
||||
errForbidden = newError(http.StatusForbidden, "forbidden", "not permitted")
|
||||
errBadRequest = newError(http.StatusBadRequest, "bad_request", "invalid request")
|
||||
)
|
||||
|
||||
// writeJSON writes v as an indented JSON body with the given status.
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
@@ -252,6 +254,58 @@ func TestLinkVerifyIdempotent(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A link whose account was SOFT-DELETED is unclaimed: a fresh in-game code lets a
|
||||
// live account take it over (the migrated-source path — retire keeps the link but
|
||||
// kills the account), while a merely disabled holder keeps its identity so the
|
||||
// lockout cannot be re-linked away, and neither dead link has in-game standing
|
||||
// (UserByMCUUID reads it exactly like an unlinked UUID). Audit #33's in-game half.
|
||||
func TestLinkVerifyTakesOverDeletedLinkOnly(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
const mcGone = "55555555-5555-5555-5555-555555555555"
|
||||
const mcLocked = "66666666-6666-6666-6666-666666666666"
|
||||
user := &Principal{UserID: "u-take", Email: "[email protected]", Role: "user"}
|
||||
repo := newFakeRepo()
|
||||
repo.seedUser(UserView{ID: "u-gone", Username: "gone", Email: "[email protected]", Role: "user"})
|
||||
repo.seedUser(UserView{ID: "u-locked", Username: "locked", Email: "[email protected]", Role: "user"})
|
||||
if err := repo.DeleteUser(ctx, "u-gone", "test"); err != nil {
|
||||
t.Fatalf("DeleteUser: %v", err)
|
||||
}
|
||||
repo.links[mcGone] = "u-gone" // a retired source's link outlives the account
|
||||
|
||||
if _, err := repo.UserByMCUUID(ctx, mcGone); !errors.Is(err, ErrNotFound) {
|
||||
t.Fatalf("UserByMCUUID(deleted link) = %v, want ErrNotFound (no in-game standing)", err)
|
||||
}
|
||||
|
||||
repo.linkCodes["TAKEOVER"] = fakeLinkCode{mcUUID: mcGone, expiresAt: time.Unix(1_700_000_600, 0)}
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.External = staticExternal{p: user}
|
||||
if w := do(api.ExternalHandler(), "POST", "/api/v1/account/link/verify", `{"code":"TAKEOVER"}`, nil); w.Code != http.StatusOK {
|
||||
t.Fatalf("takeover verify: code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if repo.links[mcGone] != "u-take" {
|
||||
t.Errorf("links[%s] = %q after takeover, want u-take", mcGone, repo.links[mcGone])
|
||||
}
|
||||
|
||||
// A disabled (not deleted) holder keeps the identity: 409, code survives, link unmoved.
|
||||
if err := repo.SetUserDisabled(ctx, "u-locked", true); err != nil {
|
||||
t.Fatalf("disable: %v", err)
|
||||
}
|
||||
repo.links[mcLocked] = "u-locked"
|
||||
repo.linkCodes["LOCKED12"] = fakeLinkCode{mcUUID: mcLocked, expiresAt: time.Unix(1_700_000_600, 0)}
|
||||
if _, err := repo.UserByMCUUID(ctx, mcLocked); !errors.Is(err, ErrNotFound) {
|
||||
t.Fatalf("UserByMCUUID(disabled link) = %v, want ErrNotFound", err)
|
||||
}
|
||||
if w := do(api.ExternalHandler(), "POST", "/api/v1/account/link/verify", `{"code":"LOCKED12"}`, nil); w.Code != http.StatusConflict {
|
||||
t.Fatalf("takeover of a disabled holder: code = %d, want 409 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if repo.links[mcLocked] != "u-locked" {
|
||||
t.Errorf("disabled holder's link moved to %q", repo.links[mcLocked])
|
||||
}
|
||||
if _, ok := repo.linkCodes["LOCKED12"]; !ok {
|
||||
t.Error("refused verify consumed the code")
|
||||
}
|
||||
}
|
||||
|
||||
// TestLinkAuthSourcePropagates proves auth_source survives the whole §10 flow: a
|
||||
// thirdparty source captured in-game at mint reaches the durable link and the
|
||||
// verify response — the value the web side can never originate itself.
|
||||
|
||||
@@ -241,7 +241,9 @@ func (a *API) handleLoginEmailVerify(w http.ResponseWriter, r *http.Request) {
|
||||
// Code redeemed. Refuse staff here — never before the verify — so op.console keeps
|
||||
// its Zero-Trust + in-game-approval gates and this public door provably yields only
|
||||
// a role=user player session (mirrors handleBindRedeem's refuse-staff contract).
|
||||
if u.Role == "admin" {
|
||||
// Staff means anything above role=user: an admin OR the role=owner identity. The
|
||||
// player door must yield only player sessions.
|
||||
if u.Role != "user" {
|
||||
writeError(w, r, newError(http.StatusForbidden, "staff_account",
|
||||
"that account is staff; sign in at the operator console"))
|
||||
return
|
||||
|
||||
@@ -1,7 +1,9 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
@@ -510,12 +512,20 @@ func TestLoginEmailPurposeSeparation(t *testing.T) {
|
||||
// a wrong code for a staff address answers the same invalid_code as for anyone, and
|
||||
// the 403 costs the valid code (verify-then-refuse), so it cannot be farmed as an
|
||||
// is-this-address-staff oracle.
|
||||
// Staff means admin AND owner (migration 0011): both must be refused at the
|
||||
// player door, after the code proves mailbox control.
|
||||
func TestLoginEmailVerifyRefusesStaff(t *testing.T) {
|
||||
for _, role := range []string{"admin", "owner"} {
|
||||
t.Run(role, func(t *testing.T) { verifyStaffRefusedAtPlayerDoor(t, role) })
|
||||
}
|
||||
}
|
||||
|
||||
func verifyStaffRefusedAtPlayerDoor(t *testing.T, role string) {
|
||||
repo := newFakeRepo()
|
||||
repo.settings[LocalAuthEnabledKey] = []byte("true")
|
||||
repo.staff["owner"] = &StaffUser{
|
||||
ID: "a1", Username: "owner", Email: "[email protected]",
|
||||
Role: "admin", EmailVerified: true,
|
||||
Role: role, EmailVerified: true,
|
||||
}
|
||||
mailer := &captureMailer{}
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
@@ -610,3 +620,94 @@ func TestLoginEmailStartFailedDeliveryReleasesCooldown(t *testing.T) {
|
||||
t.Errorf("mailer calls = %d, want 2 (one failed, one delivered)", mailer.calls)
|
||||
}
|
||||
}
|
||||
|
||||
// A disabled or soft-deleted account is DEAD at every door: the pre-session
|
||||
// resolvers refuse it (uniformly, so the door stays no-oracle), a session that
|
||||
// was live a moment ago stops authenticating, DeleteUser severs the account's
|
||||
// passkeys and Minecraft links, and the bind door refuses to reuse the retired
|
||||
// identity instead of minting a session for it. Audit #33 found the opposite
|
||||
// live: a deleted user re-logged-in through the email door and GET /me answered
|
||||
// 200 — deletion and the disable lockout were both bypassable by logging in again.
|
||||
func TestDeadAccountsCannotLogInOrKeepSessions(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
now := time.Unix(1_700_000_000, 0)
|
||||
|
||||
repo := newFakeRepo()
|
||||
repo.settings[LocalAuthEnabledKey] = []byte("true")
|
||||
repo.seedUser(UserView{ID: "u-dead", Username: "dead", Email: "[email protected]", Role: "user"})
|
||||
repo.staff["dead"].EmailVerified = true
|
||||
mailer := &captureMailer{}
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.Mailer = mailer
|
||||
eh := api.ExternalHandler()
|
||||
|
||||
// Control: alive — the door resolves the account and mails a real code, and a
|
||||
// session minted for it authenticates.
|
||||
if w := do(eh, "POST", "/api/v1/auth/email/start", `{"email":"[email protected]"}`, jsonHeader); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("alive start: code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if mailer.calls != 1 {
|
||||
t.Fatalf("alive start mailed %d codes, want 1", mailer.calls)
|
||||
}
|
||||
repo.sessions["h-live"] = &fakeSession{userID: "u-dead", expiresAt: now.Add(time.Hour)}
|
||||
if _, err := repo.SessionUser(ctx, "h-live", now); err != nil {
|
||||
t.Fatalf("live SessionUser: %v", err)
|
||||
}
|
||||
|
||||
// Disabled: the start is neutral (no mail), verify refuses, session dies.
|
||||
if err := repo.SetUserDisabled(ctx, "u-dead", true); err != nil {
|
||||
t.Fatalf("disable: %v", err)
|
||||
}
|
||||
// The alive start's reservation must not mask the neutral branch: clear the
|
||||
// throttle's window (test-only; the limiter itself is rebuilt lazily once).
|
||||
lim := api.otpLimiter()
|
||||
lim.mu.Lock()
|
||||
lim.last = map[string]time.Time{}
|
||||
lim.mu.Unlock()
|
||||
if w := do(eh, "POST", "/api/v1/auth/email/start", `{"email":"[email protected]"}`, jsonHeader); w.Code != http.StatusAccepted {
|
||||
t.Fatalf("disabled start: code = %d, want 202 neutral (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if mailer.calls != 1 {
|
||||
t.Fatalf("disabled start mailed a code (%d calls) — a dead account must resolve to nothing", mailer.calls)
|
||||
}
|
||||
if w := do(eh, "POST", "/api/v1/auth/email/verify", `{"email":"[email protected]","code":"000000"}`, jsonHeader); w.Code != http.StatusBadRequest || decodeErr(t, w) != "invalid_code" {
|
||||
t.Fatalf("disabled verify: code = %d body %s, want 400 invalid_code", w.Code, w.Body.String())
|
||||
}
|
||||
if _, err := repo.SessionUser(ctx, "h-live", now); !errors.Is(err, ErrNotFound) {
|
||||
t.Fatalf("disabled SessionUser = %v, want ErrNotFound", err)
|
||||
}
|
||||
|
||||
// Deleted: same refusals; assets severed (links released, passkeys dropped).
|
||||
if err := repo.SetUserDisabled(ctx, "u-dead", false); err != nil {
|
||||
t.Fatalf("re-enable: %v", err)
|
||||
}
|
||||
uuid := "11111111-2222-3333-4444-555555555555"
|
||||
repo.links[uuid] = "u-dead"
|
||||
repo.passkeyCreds["pk-1"] = PasskeyCredential{ID: "pk-1", UserID: "u-dead", CredentialID: "cred-1"}
|
||||
if err := repo.DeleteUser(ctx, "u-dead", "test"); err != nil {
|
||||
t.Fatalf("delete: %v", err)
|
||||
}
|
||||
if w := do(eh, "POST", "/api/v1/auth/email/verify", `{"email":"[email protected]","code":"000000"}`, jsonHeader); w.Code != http.StatusBadRequest || decodeErr(t, w) != "invalid_code" {
|
||||
t.Fatalf("deleted verify: code = %d body %s, want 400 invalid_code", w.Code, w.Body.String())
|
||||
}
|
||||
if _, err := repo.SessionUser(ctx, "h-live", now); !errors.Is(err, ErrNotFound) {
|
||||
t.Fatalf("deleted SessionUser = %v, want ErrNotFound", err)
|
||||
}
|
||||
if _, ok := repo.links[uuid]; ok {
|
||||
t.Error("DeleteUser left the Minecraft link: the UUID stays claimed forever")
|
||||
}
|
||||
if _, ok := repo.passkeyCreds["pk-1"]; ok {
|
||||
t.Error("DeleteUser left the passkey: a login credential outlives the account")
|
||||
}
|
||||
|
||||
// The bind door refuses to reuse the retired identity (a stale link that
|
||||
// predates the fix, or a username squatted by the deleted row).
|
||||
repo.links[uuid] = "u-dead"
|
||||
repo.linkCodes["CODE1234"] = fakeLinkCode{mcUUID: uuid, authSource: "mojang", expiresAt: now.Add(time.Hour)}
|
||||
if _, _, _, err := repo.RedeemPlayerBindCode(ctx, "u-new", "CODE1234", now); !errors.Is(err, ErrPlayerAccountRetired) {
|
||||
t.Fatalf("bind redeem onto a deleted account = %v, want ErrPlayerAccountRetired", err)
|
||||
}
|
||||
if _, ok := repo.linkCodes["CODE1234"]; !ok {
|
||||
t.Error("refused redeem consumed the code; re-enabling the account must stay retryable within TTL")
|
||||
}
|
||||
}
|
||||
@@ -134,6 +134,22 @@ func TestBackupNow(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("no world volume -> 409 no_world_volume, no backup", func(t *testing.T) {
|
||||
// A never-started (or reaped) server has no world PVC: the Job would hang
|
||||
// Pending on the missing claim with nothing recorded, so the gate must
|
||||
// refuse before the backuper is reached.
|
||||
api, _, cl, backuper := mk()
|
||||
cl.noWorld["survival"] = true
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "no_world_volume" {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
}
|
||||
if backuper.calls != 0 {
|
||||
t.Fatal("a world-less server must not reach the backuper")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("nil Backuper -> 503 backup_unavailable", func(t *testing.T) {
|
||||
api, _, _, _ := mk()
|
||||
api.Backuper = nil
|
||||
@@ -240,6 +256,20 @@ func TestInternalBackup(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("no world volume -> 409 no_world_volume, no backup", func(t *testing.T) {
|
||||
// The break-glass face shares enqueueBackup, so the world-volume gate must
|
||||
// hold here too — this is the face the TUI's Sync picker drives.
|
||||
api, _, cl, backuper := mk()
|
||||
cl.noWorld["survival"] = true
|
||||
w := do(api.InternalHandler(), "POST", path, "", jsonHeader)
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "no_world_volume" {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
}
|
||||
if backuper.calls != 0 {
|
||||
t.Fatal("a world-less server must not reach the backuper")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("nil Backuper -> 503 backup_unavailable", func(t *testing.T) {
|
||||
api, _, _, _ := mk()
|
||||
api.Backuper = nil
|
||||
|
||||
@@ -9,6 +9,16 @@ import (
|
||||
"felis.lolicon.best/internal/naming"
|
||||
)
|
||||
|
||||
// errNoWorldVolume is the shared 409 for backup and restore when the server's
|
||||
// world PVC does not exist: the Job would only hang Pending on the missing
|
||||
// claim — invisible to the caller and to the backups list — so the handlers
|
||||
// refuse up front. Starting the server once (which creates the claim via the
|
||||
// StatefulSet volumeClaimTemplate) unlocks both ops.
|
||||
func errNoWorldVolume() error {
|
||||
return newError(http.StatusConflict, "no_world_volume",
|
||||
"this server has no world volume yet — start it once to create it, then retry")
|
||||
}
|
||||
|
||||
// handleListBackups lists the world backups visible to the caller (spec §7 GET
|
||||
// /api/v1/backups; world_backups in §22). It is app-tier: an admin sees every
|
||||
// present backup; a regular user sees only the backups of worlds they formerly
|
||||
@@ -154,6 +164,19 @@ func (a *API) handleRestoreBackup(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
// World-volume gate: the restore Job mounts the world PVC read-write to unpack
|
||||
// the archive into it, so a missing claim means a Pod stuck Pending — a 202
|
||||
// "restoring" with nothing ever written. Same refusal as the backup face
|
||||
// (shared errNoWorldVolume): the operator starts the server once to create the
|
||||
// claim, then restores into it.
|
||||
if exists, err := a.Cluster.WorldVolumeExists(r.Context(), name); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
} else if !exists {
|
||||
writeError(w, r, errNoWorldVolume())
|
||||
return
|
||||
}
|
||||
|
||||
// Restorer is optional: when unwired the endpoint reports 503 rather than
|
||||
// panicking, so the authorization boundary above is exercised even before the
|
||||
// restore-Job executor is wired (see Restorer).
|
||||
@@ -285,6 +308,18 @@ func (a *API) enqueueBackup(w http.ResponseWriter, r *http.Request, name string,
|
||||
return
|
||||
}
|
||||
|
||||
// World-volume gate: the Job mounts the world PVC by claim name, and a missing
|
||||
// claim would leave its Pod Pending — a 202 "backing_up" with nothing ever
|
||||
// recorded anywhere. A never-started or already-reaped server is refused with
|
||||
// the same specificity as the stopped gate.
|
||||
if exists, err := a.Cluster.WorldVolumeExists(r.Context(), name); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
} else if !exists {
|
||||
writeError(w, r, errNoWorldVolume())
|
||||
return
|
||||
}
|
||||
|
||||
// Backuper is optional: when unwired the endpoint reports 503 rather than
|
||||
// panicking, so the authorization boundary above is exercised even before the
|
||||
// backup-Job executor is wired (see Backuper).
|
||||
|
||||
@@ -269,6 +269,21 @@ func TestRestoreBackup(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("no world volume -> 409 no_world_volume, no restore", func(t *testing.T) {
|
||||
// Restoring into a missing world PVC would leave the Job Pending on the
|
||||
// missing claim — a 202 "restoring" that never writes anything.
|
||||
api, _, cl, restorer := mk()
|
||||
cl.noWorld["survival"] = true
|
||||
api.External = staticExternal{p: owner}
|
||||
w := do(api.ExternalHandler(), "POST", path, "", nil)
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "no_world_volume" {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
}
|
||||
if restorer.calls != 0 {
|
||||
t.Fatal("a world-less server must not reach the restorer")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("nil Restorer -> 503 restore_unavailable", func(t *testing.T) {
|
||||
api, _, _, _ := mk()
|
||||
api.Restorer = nil
|
||||
|
||||
@@ -187,9 +187,11 @@ type emailOTPVerifyRequest struct {
|
||||
|
||||
// handleEmailOTPVerify redeems a code for the caller (spec §B2, external app face).
|
||||
// Outcomes mirror the link-verify shape: an invalid/expired/mismatched code → 400
|
||||
// invalid_code, a locked code (too many wrong guesses) → 429 otp_locked, and on
|
||||
// success the user's email is written and email_verified flips true. The verified
|
||||
// address is echoed so the panel can render it.
|
||||
// invalid_code, a locked code (too many wrong guesses) → 429 otp_locked, an address
|
||||
// another account already proved → 409 email_taken (the code stays live — the
|
||||
// address, not the code, is the problem), and on success the user's email is
|
||||
// written and email_verified flips true. The verified address is echoed so the
|
||||
// panel can render it.
|
||||
func (a *API) handleEmailOTPVerify(w http.ResponseWriter, r *http.Request) {
|
||||
p := principalFromContext(r.Context())
|
||||
var req emailOTPVerifyRequest
|
||||
@@ -211,6 +213,10 @@ func (a *API) handleEmailOTPVerify(w http.ResponseWriter, r *http.Request) {
|
||||
case errors.Is(err, ErrOTPInvalid):
|
||||
writeError(w, r, newError(http.StatusBadRequest, "invalid_code", "email code is invalid or expired"))
|
||||
return
|
||||
case errors.Is(err, ErrEmailTaken):
|
||||
writeError(w, r, newError(http.StatusConflict, "email_taken",
|
||||
"that email is already verified on another account; sign in with it or use another address"))
|
||||
return
|
||||
case err != nil:
|
||||
writeError(w, r, err)
|
||||
return
|
||||
|
||||
@@ -339,6 +339,20 @@ func TestEmailOTPVerifyRejections(t *testing.T) {
|
||||
t.Fatalf("code = %d body %s, want 429 otp_locked", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
t.Run("address proven elsewhere -> 409 email_taken, not consumed", func(t *testing.T) {
|
||||
repo := newFakeRepo()
|
||||
live(repo, "tk", otpCodeHash("123456"), time.Unix(1_700_000_600, 0), 0)
|
||||
// Another account already proved the same address: login resolves accounts
|
||||
// BY verified email, so the second proof must be refused.
|
||||
repo.staff["other"] = &StaffUser{ID: "u2", Email: "[email protected]", EmailVerified: true}
|
||||
w := do(mk(repo), "POST", "/api/v1/account/email/verify", `{"code":"123456"}`, nil)
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "email_taken" {
|
||||
t.Fatalf("code = %d body %s, want 409 email_taken", w.Code, w.Body.String())
|
||||
}
|
||||
if repo.otps["tk"].consumed {
|
||||
t.Error("a taken address must not consume the code")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestEmailOTPBruteForceLockout drives the lockout end-to-end through the handler:
|
||||
|
||||
@@ -210,6 +210,19 @@ func (a *API) authorizeFileOp(w http.ResponseWriter, r *http.Request) (string, b
|
||||
return "", false
|
||||
}
|
||||
|
||||
// World-volume gate, matching the backup/restore faces: the Job mounts the
|
||||
// world PVC by claim name, so a server that has never started (or was already
|
||||
// reaped) has no claim to mount and its Pod sits Pending until the executor's
|
||||
// wait expires — a knowably impossible request answered by a 90s hang and a
|
||||
// misleading 504. Refuse up front with the same specific 409.
|
||||
if exists, err := a.Cluster.WorldVolumeExists(r.Context(), name); err != nil {
|
||||
writeError(w, r, err)
|
||||
return "", false
|
||||
} else if !exists {
|
||||
writeError(w, r, errNoWorldVolume())
|
||||
return "", false
|
||||
}
|
||||
|
||||
// Files is optional: when unwired the endpoints report 503 rather than
|
||||
// panicking, so the authorization boundary above is exercised even before the
|
||||
// file-Job executor is wired (see FileEditor).
|
||||
|
||||
@@ -119,6 +119,47 @@ func TestFileEditorStoppedGate(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestFileEditorWorldVolumeGate pins the second physical gate: a server with no
|
||||
// world PVC (never started, or already reaped) has no claim for the Job to mount,
|
||||
// so its Pod would sit Pending until the executor's wait timed out — a 90s hang
|
||||
// and a misleading 504 files_timeout for a request that is knowably impossible.
|
||||
// All three routes must refuse BEFORE creating a Job, with the same specific 409
|
||||
// the backup/restore faces use.
|
||||
func TestFileEditorWorldVolumeGate(t *testing.T) {
|
||||
owner := &Principal{UserID: "owner1", Email: "[email protected]", Role: "user"}
|
||||
|
||||
routes := []struct {
|
||||
name string
|
||||
method string
|
||||
path string
|
||||
body string
|
||||
}{
|
||||
{"list", "GET", "/api/v1/servers/survival/files?path=config", ""},
|
||||
{"read", "GET", "/api/v1/servers/survival/file?path=server.properties", ""},
|
||||
{"write", "PUT", "/api/v1/servers/survival/file?path=server.properties", `{"content":"aGk="}`},
|
||||
}
|
||||
|
||||
for _, rt := range routes {
|
||||
t.Run(rt.name+" without a world volume -> 409 no_world_volume", func(t *testing.T) {
|
||||
api, _, cl, files := mkFiles()
|
||||
cl.noWorld["survival"] = true
|
||||
api.External = staticExternal{p: owner}
|
||||
|
||||
var hdr map[string]string
|
||||
if rt.body != "" {
|
||||
hdr = jsonHeader
|
||||
}
|
||||
w := do(api.ExternalHandler(), rt.method, rt.path, rt.body, hdr)
|
||||
if w.Code != http.StatusConflict || decodeErr(t, w) != "no_world_volume" {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
}
|
||||
if files.calls != 0 {
|
||||
t.Fatal("no claim to mount — the file Job must never be created")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestFileEditorAuthorization pins who may touch a world's files. It is the same
|
||||
// owner-or-admin rule the backup routes enforce, and it must hold on all three
|
||||
// routes — a read-only route leaking another owner's config (an RCON password
|
||||
|
||||
@@ -142,7 +142,9 @@ func (a *API) handleHasJoined(w http.ResponseWriter, r *http.Request) {
|
||||
|
||||
// Canonicalize identity. A trusted (Mojang) source keeps its UUID; a self-asserted
|
||||
// source is rewritten into felisAuthNS so it can never land in Mojang's UUID space
|
||||
// nor onto another source's. An unparseable identity UUID is not trustworthy → reject.
|
||||
// nor onto another source's. resolveHasJoined has already screened both shapes and
|
||||
// skipped unusable ones as failed; the two guards below are the last line before
|
||||
// anything leaves, kept even though nothing reaches them.
|
||||
var canonical uuid.UUID
|
||||
if src.Identity {
|
||||
id, err := uuid.Parse(prof.ID)
|
||||
@@ -154,8 +156,8 @@ func (a *API) handleHasJoined(w http.ResponseWriter, r *http.Request) {
|
||||
} else {
|
||||
// A third-party source is untrusted input, its name included: nothing stops a
|
||||
// hostile or sloppy root from answering with "§4admin", an empty string, or 200
|
||||
// characters, all of which this handler would otherwise relay straight into the
|
||||
// proxy's player list.
|
||||
// characters, all of which must not be relayed straight into the proxy's player
|
||||
// list. Screened in resolveHasJoined; this is the last-line guard.
|
||||
if !mcUsernameRe.MatchString(prof.Name) {
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
return
|
||||
@@ -320,9 +322,10 @@ func lookupPremiumName(ctx context.Context, username string) (bool, error) {
|
||||
|
||||
// resolveHasJoined queries each configured source in priority order and returns the
|
||||
// first that validates the session (200 with a profile). 204 is "not my player". Any other
|
||||
// outcome (unreachable, another status, a body that is not a profile) skips the source too,
|
||||
// but is logged with its tag and reported as failed: otherwise a dead or mistyped source
|
||||
// looks exactly like a player it does not know, and nobody finds out.
|
||||
// outcome (unreachable, another status, a body that is not a profile, or a profile this
|
||||
// multiplexer will not emit) skips the source too, but is logged with its tag and reported
|
||||
// as failed: otherwise a dead or mistyped source looks exactly like a player it does not
|
||||
// know, and nobody finds out.
|
||||
func (a *API) resolveHasJoined(ctx context.Context, username, serverID, ip string) (prof *sessionProfile, src AuthSource, failed bool) {
|
||||
for _, src := range a.AuthSources {
|
||||
u := src.URL + "?username=" + url.QueryEscape(username) + "&serverId=" + url.QueryEscape(serverID)
|
||||
@@ -362,6 +365,23 @@ func (a *API) resolveHasJoined(ctx context.Context, username, serverID, ip strin
|
||||
failed = true
|
||||
continue
|
||||
}
|
||||
// Screen what a 200 is allowed to mean BEFORE it can win. A profile this
|
||||
// multiplexer will not emit — an identity UUID that does not parse, a third-party
|
||||
// name outside the Minecraft charset — is the same class as a body that is not a
|
||||
// profile: skip, log, count as failed. Letting it win would stop the ladder on one
|
||||
// sloppy root (every source behind it silently unreachable) and read as "nobody
|
||||
// knows this player" to Velocity while a source had actually answered.
|
||||
if src.Identity {
|
||||
if _, perr := uuid.Parse(p.ID); perr != nil {
|
||||
log.Printf("hasJoined: identity source %q answered 200 with an unparseable profile id %q", src.Tag, p.ID)
|
||||
failed = true
|
||||
continue
|
||||
}
|
||||
} else if !mcUsernameRe.MatchString(p.Name) {
|
||||
log.Printf("hasJoined: source %q answered 200 with an unusable profile name %q", src.Tag, p.Name)
|
||||
failed = true
|
||||
continue
|
||||
}
|
||||
return &p, src, failed
|
||||
}
|
||||
return nil, AuthSource{}, failed
|
||||
|
||||
@@ -2,12 +2,14 @@ package api
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"log"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
@@ -337,14 +339,26 @@ func TestHasJoined(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
// Mojang is trusted for its UUIDs, which is exactly why one that does not parse must not
|
||||
// be emitted as some default: every such login would share the nil UUID.
|
||||
t.Run("identity source with an unparseable id -> 204", func(t *testing.T) {
|
||||
// Mojang is trusted for its UUIDs, which is exactly why one that does not parse must
|
||||
// not be emitted as some default: every such login would share the nil UUID. The
|
||||
// unusable 200 is surfaced like every other bad answer — skipped, logged, and 503 when
|
||||
// nothing else validates — instead of reading as "no such session".
|
||||
t.Run("identity source with an unparseable id -> skipped and surfaced", func(t *testing.T) {
|
||||
mojang := fakeYgg(t, "not-a-uuid", "Notch")
|
||||
api := newTestAPI(newFakeRepo(), newFakeCluster())
|
||||
api.AuthSources = []AuthSource{{Tag: "mojang", URL: mojang.URL, Identity: true}}
|
||||
if w := getHasJoined(api.InternalHandler(), "Notch", "abc"); w.Code != http.StatusNoContent {
|
||||
t.Fatalf("code = %d, want 204 (%q)", w.Code, w.Body.String())
|
||||
|
||||
var buf bytes.Buffer
|
||||
prev := log.Writer()
|
||||
log.SetOutput(&buf)
|
||||
w := getHasJoined(api.InternalHandler(), "Notch", "abc")
|
||||
log.SetOutput(prev)
|
||||
|
||||
if w.Code != http.StatusServiceUnavailable {
|
||||
t.Fatalf("code = %d, want 503 (%q)", w.Code, w.Body.String())
|
||||
}
|
||||
if !strings.Contains(buf.String(), "unparseable profile id") {
|
||||
t.Fatalf("the skip was not logged: %q", buf.String())
|
||||
}
|
||||
})
|
||||
|
||||
@@ -559,16 +573,51 @@ func TestHasJoinedPremiumNameRename(t *testing.T) {
|
||||
}
|
||||
})
|
||||
|
||||
// A third-party root is untrusted input, its name field included.
|
||||
t.Run("hostile upstream name -> 204", func(t *testing.T) {
|
||||
// A third-party root is untrusted input, its name field included. An unusable name is
|
||||
// refused AND surfaced: skipped like every other bad answer (logged, counted as
|
||||
// failed), so it can never be relayed and can never be mistaken for "no such player".
|
||||
t.Run("hostile upstream name is skipped, logged and surfaced", func(t *testing.T) {
|
||||
stubMojangNames(t)
|
||||
for _, bad := range []string{"§4admin", "not a name", "ab", strings.Repeat("x", 17), ""} {
|
||||
third := fakeYgg(t, notchMojangID, bad)
|
||||
api := newTestAPI(newFakeRepo(), newFakeCluster())
|
||||
api.AuthSources = []AuthSource{{Tag: "littleskin", Prefix: "LS", URL: third.URL, Identity: false}}
|
||||
if w := getHasJoined(api.InternalHandler(), "Notch", "abc"); w.Code != http.StatusNoContent {
|
||||
t.Fatalf("upstream name %q: code = %d, want 204 (%q)", bad, w.Code, w.Body.String())
|
||||
|
||||
var buf bytes.Buffer
|
||||
prev := log.Writer()
|
||||
log.SetOutput(&buf)
|
||||
w := getHasJoined(api.InternalHandler(), "Notch", "abc")
|
||||
log.SetOutput(prev)
|
||||
|
||||
if w.Code != http.StatusServiceUnavailable {
|
||||
t.Fatalf("upstream name %q: code = %d, want 503 (%q)", bad, w.Code, w.Body.String())
|
||||
}
|
||||
if !strings.Contains(buf.String(), "unusable profile name") {
|
||||
t.Fatalf("upstream name %q: the skip was not logged: %q", bad, buf.String())
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
// The reason the skip must CONTINUE the ladder rather than stop it: a broken root early
|
||||
// in the list would otherwise silently block a login its later source would have
|
||||
// validated (live on the audit box, the bad name produced a silent 204 and the valid
|
||||
// source was never asked).
|
||||
t.Run("hostile name cannot block a later source", func(t *testing.T) {
|
||||
stubMojangNames(t)
|
||||
bad := fakeYgg(t, notchMojangID, "§4admin")
|
||||
good := fakeYgg(t, notchMojangID, "Notch")
|
||||
api := newTestAPI(newFakeRepo(), newFakeCluster())
|
||||
api.AuthSources = []AuthSource{
|
||||
{Tag: "broken", Prefix: "BR", URL: bad.URL, Identity: false},
|
||||
{Tag: "littleskin", Prefix: "LS", URL: good.URL, Identity: false},
|
||||
}
|
||||
w := getHasJoined(api.InternalHandler(), "Notch", "abc")
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("code = %d, want 200 from the second source (%q)", w.Code, w.Body.String())
|
||||
}
|
||||
p := profileOf(t, w)
|
||||
if want := undashed(uuid.NewMD5(felisAuthNS, []byte("littleskin:"+notchMojangID))); p.ID != want {
|
||||
t.Fatalf("id = %q, want the second source's canonical %q", p.ID, want)
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -240,6 +240,12 @@ func (a *API) handleInternalClaim(w http.ResponseWriter, r *http.Request) {
|
||||
// 412 above; a lost race (0 rows) is 409.
|
||||
claimed, err := a.Repo.ClaimServer(r.Context(), name, userID)
|
||||
if err != nil {
|
||||
// Same atomic quota gate as the external face (audit #4): the concurrent
|
||||
// loser gets the sequential 403, never an over-provisioned tenant.
|
||||
if errors.Is(err, ErrQuotaExceeded) {
|
||||
writeError(w, r, newError(http.StatusForbidden, "quota_exceeded", "server quota exhausted"))
|
||||
return
|
||||
}
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
}
|
||||
|
||||
@@ -486,8 +486,9 @@ func TestServerConsoleDisconnectTeardown(t *testing.T) {
|
||||
}}
|
||||
api.Logs = streamer
|
||||
|
||||
rec := httptest.NewRecorder()
|
||||
w := &firstDataWriter{ResponseWriter: rec, data: make(chan struct{})}
|
||||
r := httptest.NewRequest("GET", "/api/v1/servers/survival/console", nil).WithContext(ctx)
|
||||
w := httptest.NewRecorder()
|
||||
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
@@ -502,6 +503,15 @@ func TestServerConsoleDisconnectTeardown(t *testing.T) {
|
||||
case <-time.After(2 * time.Second):
|
||||
t.Fatal("relay never read the first log line")
|
||||
}
|
||||
// The first read is not enough: the relay still has to WRITE the event, and
|
||||
// cancelling in that gap raced the write against the teardown (a live CI flake
|
||||
// left the body empty). Wait for the write itself — the signal is closed after
|
||||
// the recorder's Write returns, so the body read below happens-after it.
|
||||
select {
|
||||
case <-w.data:
|
||||
case <-time.After(2 * time.Second):
|
||||
t.Fatal("relay never wrote the first data event to the response")
|
||||
}
|
||||
cancel()
|
||||
|
||||
// The handler must return promptly...
|
||||
@@ -525,8 +535,32 @@ func TestServerConsoleDisconnectTeardown(t *testing.T) {
|
||||
default:
|
||||
t.Fatal("StreamLogs did not receive the cancellable request context (gotCtx not Done)")
|
||||
}
|
||||
if !strings.Contains(w.Body.String(), "data: boot progress 50%\n\n") {
|
||||
t.Fatalf("expected the first event before disconnect, got %q", w.Body.String())
|
||||
if !strings.Contains(rec.Body.String(), "data: boot progress 50%\n\n") {
|
||||
t.Fatalf("expected the first event before disconnect, got %q", rec.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
// firstDataWriter signals once the relay has written a `data:` event into the
|
||||
// wrapped recorder, so a test can disconnect only after the event is actually
|
||||
// observable in the body — inspecting p per Write is race-free because the relay
|
||||
// is the lone writer and each relay event is a single Write.
|
||||
type firstDataWriter struct {
|
||||
http.ResponseWriter
|
||||
data chan struct{}
|
||||
once sync.Once
|
||||
}
|
||||
|
||||
func (b *firstDataWriter) Write(p []byte) (int, error) {
|
||||
n, err := b.ResponseWriter.Write(p)
|
||||
if strings.HasPrefix(string(p), "data:") {
|
||||
b.once.Do(func() { close(b.data) })
|
||||
}
|
||||
return n, err
|
||||
}
|
||||
|
||||
func (b *firstDataWriter) Flush() {
|
||||
if f, ok := b.ResponseWriter.(http.Flusher); ok {
|
||||
f.Flush()
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -38,11 +38,11 @@ import (
|
||||
// in-game identity is the root of trust, so re-minting a code always re-grants a
|
||||
// session even after email/passkey are bound. "登录并非强制,但没登录什么都干不了".
|
||||
//
|
||||
// op.console stays behind Zero Trust. A code whose UUID belongs to STAFF (role=admin)
|
||||
// is refused here (ErrPlayerBindForbidden → 403), so the public bootstrap provably
|
||||
// never mints a session for an admin identity — the sole tier it yields is a role=user
|
||||
// player session, host-only to console.<root_domain> (never sent to op.console) and
|
||||
// carrying ViaAdminAccess=false. "op.console 必须得 Auth".
|
||||
// op.console stays behind Zero Trust. A code whose UUID belongs to STAFF (admin or
|
||||
// owner) is refused here (ErrPlayerBindForbidden → 403), so the public bootstrap
|
||||
// provably never mints a session for a staff identity — the sole tier it yields is a
|
||||
// role=user player session, host-only to console.<root_domain> (never sent to
|
||||
// op.console) and carrying ViaAdminAccess=false. "op.console 必须得 Auth".
|
||||
|
||||
// newUserID returns an opaque random user id (128 bits, hex), matching the shape of
|
||||
// the ids break-glass mints for staff rows.
|
||||
@@ -109,6 +109,14 @@ func (a *API) handleBindRedeem(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, r, newError(http.StatusForbidden, "staff_account",
|
||||
"that Minecraft account belongs to staff; sign in at the operator console"))
|
||||
return
|
||||
case errors.Is(err, ErrPlayerAccountRetired):
|
||||
// The linked Felis account is disabled or soft-deleted: the door refuses to
|
||||
// reuse it, because minting a session here would resurrect the account the
|
||||
// owner just retired (audit #33). The code survives, so re-enabling the
|
||||
// account and retrying within its TTL still works.
|
||||
writeError(w, r, newError(http.StatusForbidden, "account_retired",
|
||||
"this Minecraft account's Felis account is disabled or deleted; contact the operator"))
|
||||
return
|
||||
case err != nil:
|
||||
writeError(w, r, err)
|
||||
return
|
||||
|
||||
@@ -28,7 +28,7 @@ import (
|
||||
// column keeps it from ever colliding with a console login_email or onboard code).
|
||||
// - An in-game vouch — an already-trusted admin who is ONLINE approves the pending
|
||||
// request via velocity's /felis command (internal approve). The API's own user
|
||||
// table is the sole authority: only a UUID linked to a role=admin account may
|
||||
// table is the sole authority: only a UUID linked to a staff account may
|
||||
// approve (velocity's command runs for any player and relies on this check).
|
||||
//
|
||||
// finish mints the session only when BOTH have landed. Neither factor alone — a mailed
|
||||
@@ -137,9 +137,10 @@ func (a *API) handleOpLoginStart(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
// op.console is the STAFF door: a non-admin who typed their address here (they belong
|
||||
// op.console is the STAFF door: a player who typed their address here (they belong
|
||||
// on console.<root_domain>) gets the neutral response, never a request or a code.
|
||||
if u.Role != "admin" {
|
||||
// Staff means admin OR owner — the Owner is the primary op.console user.
|
||||
if !staffRole(u.Role) {
|
||||
neutral()
|
||||
return
|
||||
}
|
||||
@@ -290,15 +291,15 @@ func (a *API) handleOpLoginFinish(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
// Load the staff account for the session + response. Re-assert admin as defence in
|
||||
// depth: only admins ever get a request minted, but the session must never be issued
|
||||
// to a non-admin identity even if the row were somehow otherwise.
|
||||
// Load the staff account for the session + response. Re-assert staff as defence in
|
||||
// depth: only staff ever get a request minted, but the session must never be issued
|
||||
// to a non-staff identity even if the row were somehow otherwise.
|
||||
u, err := a.Repo.UserByID(r.Context(), loginReq.UserID)
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
if u.Role != "admin" {
|
||||
if !staffRole(u.Role) {
|
||||
writeError(w, r, newError(http.StatusForbidden, "staff_account", "that account is not an operator"))
|
||||
return
|
||||
}
|
||||
@@ -343,7 +344,7 @@ func (a *API) handleOpLoginPending(w http.ResponseWriter, r *http.Request) {
|
||||
|
||||
// opLoginApproveRequest is the internal approve body: the online-mode UUID of the
|
||||
// in-game admin running /felis web op approve. The API resolves it to a linked account
|
||||
// and refuses unless that account is role=admin — this check against the API's
|
||||
// and refuses unless that account is staff (admin or owner) — this check against the API's
|
||||
// authoritative user table is the only gate; velocity's command itself is unprivileged.
|
||||
type opLoginApproveRequest struct {
|
||||
ApproverUUID string `json:"approver_uuid"`
|
||||
@@ -351,10 +352,10 @@ type opLoginApproveRequest struct {
|
||||
|
||||
// handleOpLoginApprove records an in-game admin's vouch for a pending staff login
|
||||
// (internal face), supplying the second factor. It resolves the approver UUID to a
|
||||
// linked role=admin account (else 403), then flips the request approved. A missing or
|
||||
// no-longer-pending request is 404. Self-approval is allowed: a staff member online as
|
||||
// their own admin identity supplies a genuine second factor (in-game session control)
|
||||
// distinct from the mailbox factor.
|
||||
// linked staff account (admin or owner; else 403), then flips the request approved.
|
||||
// A missing or no-longer-pending request is 404. Self-approval is allowed: a staff
|
||||
// member online as their own admin identity supplies a genuine second factor
|
||||
// (in-game session control) distinct from the mailbox factor.
|
||||
func (a *API) handleOpLoginApprove(w http.ResponseWriter, r *http.Request) {
|
||||
id := r.PathValue("id")
|
||||
var req opLoginApproveRequest
|
||||
|
||||
@@ -170,6 +170,44 @@ func TestOpLoginVertical(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestOpLoginOwnerAdmitted pins that the staff door admits the Owner (role=owner),
|
||||
// not just plain admins: the Owner is the primary op.console identity, so a
|
||||
// role check of "admin only" would strand it outside its own console.
|
||||
func TestOpLoginOwnerAdmitted(t *testing.T) {
|
||||
repo := newFakeRepo()
|
||||
repo.settings[LocalAuthEnabledKey] = []byte("true")
|
||||
repo.staff["owner"] = &StaffUser{
|
||||
ID: "o1", Username: "owner", Email: "[email protected]",
|
||||
Role: "owner", EmailVerified: true,
|
||||
}
|
||||
repo.links[opUUID] = "o1"
|
||||
mailer := &captureMailer{}
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.Mailer = mailer
|
||||
eh := api.ExternalHandler()
|
||||
ih := api.InternalHandler()
|
||||
|
||||
w := startOp(eh, "[email protected]")
|
||||
if w.Code != http.StatusAccepted {
|
||||
t.Fatalf("owner start: code = %d, want 202 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
reqID, _ := acctBody(t, w)["request_id"].(string)
|
||||
if reqID == "" || mailer.calls != 1 || len(repo.opLogins) != 1 {
|
||||
t.Fatalf("owner start must mint a request + mail a code: req=%q mails=%d rows=%d",
|
||||
reqID, mailer.calls, len(repo.opLogins))
|
||||
}
|
||||
if w := approveOp(ih, reqID, opUUID); w.Code != http.StatusOK {
|
||||
t.Fatalf("approve: code = %d (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
w = finishOp(eh, reqID, mailer.code)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("owner finish: code = %d body %s, want 200", w.Code, w.Body.String())
|
||||
}
|
||||
if vb := acctBody(t, w); vb["role"] != "owner" {
|
||||
t.Fatalf("finish role = %v, want owner", vb["role"])
|
||||
}
|
||||
}
|
||||
|
||||
// TestOpLoginStartNeutral pins the start-side anti-enumeration contract: op.console is
|
||||
// the STAFF door, so a non-admin account AND an unknown address both get a 202 carrying
|
||||
// a request_id + expires_at, mint/mail nothing, and still burn the per-recipient
|
||||
|
||||
@@ -148,6 +148,13 @@ func (a *API) handlePasskeyLoginDiscoverableFinish(w http.ResponseWriter, r *htt
|
||||
if err != nil {
|
||||
return PasskeyUser{}, err
|
||||
}
|
||||
// A disabled or soft-deleted account must not complete a login even when it
|
||||
// still holds a credential (UserByID is an unfiltered lookup shared with admin
|
||||
// reads, so the liveness check lives here, at the door). Fail closed with the
|
||||
// same opaque outcome as an unknown handle (audit #33).
|
||||
if d, err := a.Repo.UserDetail(r.Context(), u.ID); err != nil || d.Disabled || d.DeletedAt != nil {
|
||||
return PasskeyUser{}, ErrNotFound
|
||||
}
|
||||
creds, err := a.Repo.PasskeyCredentialsForUser(r.Context(), u.ID)
|
||||
if err != nil {
|
||||
return PasskeyUser{}, err
|
||||
|
||||
@@ -127,6 +127,26 @@ func TestPatchServerMemoryOverride(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestPatchServerPreservesStorageCache pins the storage dimension of the quota
|
||||
// aggregate: a resources patch cannot change storage, so the cached storage
|
||||
// must survive it — passing 0 would silently zero the owner's aggregate (the
|
||||
// cached columns are QuotaCheck's only input) from that patch onward.
|
||||
func TestPatchServerPreservesStorageCache(t *testing.T) {
|
||||
api, repo, _, _ := newPatchAPI()
|
||||
repo.byName["survival"].OwnerID = "u1"
|
||||
repo.quota["u1"] = true
|
||||
repo.serverResources["survival"] = ResourceSpec{StorageMB: 10240}
|
||||
|
||||
w := patchSurvival(api, `{"memory":"2Gi","resources":{"memory":"4Gi","cpu":"2"}}`)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
got := repo.resourceUpdates["survival"]
|
||||
if got.CPUMilli != 2000 || got.MemoryMB != 4096 || got.StorageMB != 10240 {
|
||||
t.Fatalf("resource cache = %+v, want cpu 2000 / mem 4096 / storage preserved 10240", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPatchServerRejections is the validation matrix: each malformed request is
|
||||
// rejected with the right status and stable error code, and (critically) NOTHING
|
||||
// reaches the cluster on a rejection — the analog of create's "no CRD written".
|
||||
|
||||
@@ -142,6 +142,13 @@ func (a *API) handleClaim(w http.ResponseWriter, r *http.Request) {
|
||||
// ③ atomic claim
|
||||
claimed, err := a.Repo.ClaimServer(r.Context(), name, p.UserID)
|
||||
if err != nil {
|
||||
// The atomic gate re-checks quota under the per-user lock (audit #4): a
|
||||
// concurrent claim that spent the last slot surfaces here, with the same
|
||||
// 403 the pre-check gives sequentially.
|
||||
if errors.Is(err, ErrQuotaExceeded) {
|
||||
writeError(w, r, newError(http.StatusForbidden, "quota_exceeded", "server quota exhausted"))
|
||||
return
|
||||
}
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
}
|
||||
@@ -252,7 +259,8 @@ func (a *API) handleFleet(w http.ResponseWriter, r *http.Request) {
|
||||
owners, _ := a.Repo.ServerOwners(r.Context())
|
||||
views := make([]fleetServerView, len(servers))
|
||||
for i, s := range servers {
|
||||
views[i] = fleetServerView{ServerInfo: s, Owner: owners[s.Name]}
|
||||
views[i] = fleetServerView{ServerInfo: s, Owner: owners[s.Name],
|
||||
System: naming.IsSystemServer(s.Name)}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"servers": views})
|
||||
}
|
||||
@@ -260,13 +268,18 @@ func (a *API) handleFleet(w http.ResponseWriter, r *http.Request) {
|
||||
// fleetServerView is one row of the SysAdmin cockpit's fleet read: the CRD
|
||||
// lifecycle view (ServerInfo, §1 authority) with the owner's display identity
|
||||
// joined alongside. The embed keeps every lifecycle field flat in the JSON so the
|
||||
// shape is a strict superset of ServerInfo; Owner is the only addition.
|
||||
// shape is a strict superset of ServerInfo.
|
||||
type fleetServerView struct {
|
||||
ServerInfo
|
||||
// Owner is the claiming user's display identity (email, or username when the
|
||||
// address is absent), or "" when the server is unclaimed or the best-effort
|
||||
// owner lookup failed — the cockpit renders "" as "unclaimed".
|
||||
Owner string `json:"owner,omitempty"`
|
||||
// System marks a platform-provisioned system service (the login gate and the
|
||||
// lobby, naming.IsSystemServer). Their names are reserved, so every per-server
|
||||
// API route rejects them — the cockpit must render them read-only rather than
|
||||
// offer claim/wake/stop/console actions that would answer 400.
|
||||
System bool `json:"system,omitempty"`
|
||||
}
|
||||
|
||||
// createServerRequest is the structured §15 create-server form. This is the
|
||||
@@ -759,7 +772,19 @@ func (a *API) handlePatchServer(w http.ResponseWriter, r *http.Request) {
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
}
|
||||
_ = a.Repo.UpdateServerResources(r.Context(), name, newCPU, newMemMB, 0)
|
||||
// A resource patch cannot change storage, so its cached contribution must
|
||||
// be preserved: passing 0 would silently zero the storage dimension of the
|
||||
// owner's four-cap aggregate (the cached columns are its only input).
|
||||
storMB := 0
|
||||
if rec != nil {
|
||||
cur, err := a.Repo.ServerResources(r.Context(), name)
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
storMB = cur.StorageMB
|
||||
}
|
||||
_ = a.Repo.UpdateServerResources(r.Context(), name, newCPU, newMemMB, storMB)
|
||||
} else {
|
||||
if err := a.Cluster.PatchServerSpec(r.Context(), name, patch); err != nil {
|
||||
a.writeLookupError(w, r, err)
|
||||
|
||||
@@ -140,6 +140,18 @@ func (a *API) handlePatchUser(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
// Owner protection (migration 0011): the owner row is the one identity the
|
||||
// panel may never demote — only the local break-glass console resets it.
|
||||
// Username/email edits on it stay allowed. A failed detail read falls through;
|
||||
// UpdateUser then answers the real 404.
|
||||
if body.Role != nil && *body.Role != "owner" {
|
||||
if d, err := a.Repo.UserDetail(r.Context(), id); err == nil && d.Role == "owner" {
|
||||
writeError(w, r, newError(http.StatusForbidden, "forbidden",
|
||||
"the owner account's role cannot be changed from the panel"))
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
if body.Username != nil {
|
||||
if err := validateUsername(*body.Username); err != nil {
|
||||
writeError(w, r, err)
|
||||
@@ -186,6 +198,14 @@ func (a *API) handleDeleteUser(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
// Same owner protection as the role guard above: only break-glass retires the
|
||||
// owner identity. A failed detail read falls through to the real 404.
|
||||
if d, err := a.Repo.UserDetail(r.Context(), id); err == nil && d.Role == "owner" {
|
||||
writeError(w, r, newError(http.StatusForbidden, "forbidden",
|
||||
"the owner account cannot be deleted from the panel"))
|
||||
return
|
||||
}
|
||||
|
||||
if err := a.Repo.DeleteUser(r.Context(), id, p.Email); err != nil {
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
writeError(w, r, newError(http.StatusNotFound, "not_found", "user not found"))
|
||||
@@ -222,6 +242,17 @@ func (a *API) handleDisableUser(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
// Owner protection (migration 0011): disabling locks the owner out and revokes
|
||||
// its sessions — effectively a demotion, so the panel refuses it; only
|
||||
// break-glass touches the owner identity. Re-enabling stays allowed.
|
||||
if body.Disabled {
|
||||
if d, err := a.Repo.UserDetail(r.Context(), id); err == nil && d.Role == "owner" {
|
||||
writeError(w, r, newError(http.StatusForbidden, "forbidden",
|
||||
"the owner account cannot be disabled from the panel"))
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
if err := a.Repo.SetUserDisabled(r.Context(), id, body.Disabled); err != nil {
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
writeError(w, r, newError(http.StatusNotFound, "not_found", "user not found"))
|
||||
@@ -250,6 +281,10 @@ func (a *API) handleGetQuotas(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
v, err := a.Repo.GetQuotas(r.Context(), id)
|
||||
if err != nil {
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
writeError(w, r, newError(http.StatusNotFound, "not_found", "user not found"))
|
||||
return
|
||||
}
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
@@ -446,6 +481,10 @@ func (a *API) handleLinkAccount(w http.ResponseWriter, r *http.Request) {
|
||||
"this UUID is already linked to a different user"))
|
||||
return
|
||||
}
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
writeError(w, r, newError(http.StatusNotFound, "not_found", "user not found"))
|
||||
return
|
||||
}
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
|
||||
@@ -0,0 +1,99 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// The owner row is the one identity the panel may never demote, delete, or
|
||||
// disable (migration 0011) — only the local break-glass console resets it.
|
||||
// These guards became load-bearing the moment provisioning started actually
|
||||
// writing role='owner'; before that the owner tier was simply unreachable, so
|
||||
// nothing could reach them.
|
||||
func TestOwnerAccountProtectedFromPanelMutations(t *testing.T) {
|
||||
owner := &Principal{UserID: "usr-root", Role: "owner", ViaAdminAccess: true}
|
||||
repo := newFakeRepo()
|
||||
repo.seedUser(UserView{ID: "usr-root", Username: "root", Role: "owner"})
|
||||
repo.seedUser(UserView{ID: "usr-owner2", Username: "spare-owner", Role: "owner"})
|
||||
repo.seedUser(UserView{ID: "u2", Username: "alice", Role: "user"})
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.External = staticExternal{p: owner}
|
||||
eh := api.ExternalHandler()
|
||||
|
||||
t.Run("role change refused", func(t *testing.T) {
|
||||
w := do(eh, "PATCH", "/api/v1/users/usr-owner2", `{"role":"admin"}`, jsonHeader)
|
||||
if w.Code != http.StatusForbidden {
|
||||
t.Fatalf("demote owner: code = %d body %s, want 403", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
t.Run("delete refused", func(t *testing.T) {
|
||||
w := do(eh, "DELETE", "/api/v1/users/usr-owner2", "", nil)
|
||||
if w.Code != http.StatusForbidden {
|
||||
t.Fatalf("delete owner: code = %d body %s, want 403", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
t.Run("disable refused", func(t *testing.T) {
|
||||
w := do(eh, "POST", "/api/v1/users/usr-owner2/disable", `{"disabled":true}`, jsonHeader)
|
||||
if w.Code != http.StatusForbidden {
|
||||
t.Fatalf("disable owner: code = %d body %s, want 403", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
t.Run("email edits on an owner stay allowed", func(t *testing.T) {
|
||||
w := do(eh, "PATCH", "/api/v1/users/usr-owner2", `{"email":"[email protected]"}`, jsonHeader)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("edit owner email: code = %d body %s, want 200", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
t.Run("control: a normal user can still be promoted", func(t *testing.T) {
|
||||
w := do(eh, "PATCH", "/api/v1/users/u2", `{"role":"admin"}`, jsonHeader)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("promote user: code = %d body %s, want 200", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// The user-scoped admin sub-resources (quotas, account links) must answer 404
|
||||
// for an unknown user id. Before the requireLiveUser guard the quota upsert and
|
||||
// the link insert reached the users(id) foreign key and surfaced as an opaque
|
||||
// 500 (found live against the drill cluster, audit #30), and the quotas read
|
||||
// answered a zero-value "unlimited" view as if the id existed.
|
||||
func TestAdminSubresourcesRequireLiveUser(t *testing.T) {
|
||||
owner := &Principal{UserID: "usr-root", Role: "owner", ViaAdminAccess: true}
|
||||
repo := newFakeRepo()
|
||||
repo.seedUser(UserView{ID: "usr-root", Username: "root", Role: "owner"})
|
||||
repo.seedUser(UserView{ID: "u2", Username: "alice", Role: "user"})
|
||||
api := newTestAPI(repo, newFakeCluster())
|
||||
api.External = staticExternal{p: owner}
|
||||
eh := api.ExternalHandler()
|
||||
|
||||
t.Run("quotas read of an unknown user", func(t *testing.T) {
|
||||
w := do(eh, "GET", "/api/v1/users/usr-nope/quotas", "", nil)
|
||||
if w.Code != http.StatusNotFound {
|
||||
t.Fatalf("code = %d body %s, want 404", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
t.Run("quotas write of an unknown user", func(t *testing.T) {
|
||||
w := do(eh, "PUT", "/api/v1/users/usr-nope/quotas", `{"max_servers":1}`, jsonHeader)
|
||||
if w.Code != http.StatusNotFound {
|
||||
t.Fatalf("code = %d body %s, want 404", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
t.Run("account link of an unknown user", func(t *testing.T) {
|
||||
w := do(eh, "POST", "/api/v1/users/usr-nope/links",
|
||||
`{"mc_uuid":"22222222-3333-4444-5555-666666666666"}`, jsonHeader)
|
||||
if w.Code != http.StatusNotFound {
|
||||
t.Fatalf("code = %d body %s, want 404", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
t.Run("control: a live user still accepts both writes", func(t *testing.T) {
|
||||
w := do(eh, "PUT", "/api/v1/users/u2/quotas", `{"max_servers":2}`, jsonHeader)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("set quotas: code = %d body %s, want 200", w.Code, w.Body.String())
|
||||
}
|
||||
w = do(eh, "POST", "/api/v1/users/u2/links",
|
||||
`{"mc_uuid":"22222222-3333-4444-5555-666666666677"}`, jsonHeader)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("link: code = %d body %s, want 200", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -15,6 +15,11 @@ import (
|
||||
// admin-tier (Zero Trust), enforced by adminOnly before these handlers run.
|
||||
type ImageBuilder interface {
|
||||
Submit(ctx context.Context, req build.Request) (*build.Build, error)
|
||||
// Get reads one build row without reconciling it — the read-only lookup the
|
||||
// submission views use to surface a build's outcome to its submitter (the
|
||||
// /images/build routes are admin-tier). State advance belongs to the
|
||||
// reconcile loop (Sync/SyncAll), so a list render never touches the cluster.
|
||||
Get(ctx context.Context, id string) (*build.Build, error)
|
||||
// Sync reconciles a build against its Job and returns the current view, so a
|
||||
// GET doubles as the reconcile tick (idempotent on terminal builds).
|
||||
Sync(ctx context.Context, id string) (*build.Build, error)
|
||||
|
||||
@@ -16,6 +16,8 @@ import (
|
||||
type fakeBuilder struct {
|
||||
submitted *build.Request
|
||||
submitErr error
|
||||
getBuilds map[string]*build.Build
|
||||
getErr error
|
||||
syncErr error
|
||||
cancelErr error
|
||||
addedRef string
|
||||
@@ -40,6 +42,17 @@ func (f *fakeBuilder) Submit(_ context.Context, req build.Request) (*build.Build
|
||||
RequestedBy: req.RequestedBy}, nil
|
||||
}
|
||||
|
||||
func (f *fakeBuilder) Get(_ context.Context, id string) (*build.Build, error) {
|
||||
f.lastBuildID = id
|
||||
if f.getErr != nil {
|
||||
return nil, f.getErr
|
||||
}
|
||||
if b, ok := f.getBuilds[id]; ok {
|
||||
return b, nil
|
||||
}
|
||||
return nil, build.ErrNotFound
|
||||
}
|
||||
|
||||
func (f *fakeBuilder) Sync(_ context.Context, id string) (*build.Build, error) {
|
||||
f.lastBuildID = id
|
||||
if f.syncErr != nil {
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/naming"
|
||||
)
|
||||
|
||||
// AsyncJob is the observable outcome of one asynchronous world operation. The API
|
||||
// only ENQUEUES backup/restore Jobs — the work, and any failure, happens in the
|
||||
// cluster — so without this projection a failed job left its only trace in a Job
|
||||
// object an operator with kubectl could read. The route is the API-side outlet.
|
||||
type AsyncJob struct {
|
||||
Name string `json:"name"`
|
||||
Kind string `json:"kind"` // "backup" | "restore"
|
||||
State string `json:"state"` // "running" | "succeeded" | "failed"
|
||||
Message string `json:"message,omitempty"`
|
||||
StartedAt time.Time `json:"started_at,omitempty"`
|
||||
FinishedAt time.Time `json:"finished_at,omitempty"`
|
||||
}
|
||||
|
||||
// JobStatusReader reads the newest backup/restore Jobs for a server, newest
|
||||
// first. Optional like Restorer/Backuper: when nil the route answers 503.
|
||||
type JobStatusReader interface {
|
||||
LatestJobs(ctx context.Context, serverName string) ([]AsyncJob, error)
|
||||
}
|
||||
|
||||
// handleServerJobs serves GET /api/v1/servers/{name}/jobs — the latest async world
|
||||
// operations for one server, so a 202 that later failed is visible without kubectl.
|
||||
// Authorization mirrors the backup/restore gates' front half (owner-or-admin); the
|
||||
// message is free-form Job text and can name paths the owner already sees through
|
||||
// the file editor.
|
||||
func (a *API) handleServerJobs(w http.ResponseWriter, r *http.Request) {
|
||||
p := principalFromContext(r.Context())
|
||||
name := r.PathValue("name")
|
||||
if err := naming.ValidateServerName(name); err != nil {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
|
||||
return
|
||||
}
|
||||
rec, err := a.Repo.ServerByName(r.Context(), name)
|
||||
if err != nil {
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
}
|
||||
if !a.isOwnerOrAdmin(p, rec) {
|
||||
writeError(w, r, errForbidden)
|
||||
return
|
||||
}
|
||||
if a.JobStatus == nil {
|
||||
writeError(w, r, newError(http.StatusServiceUnavailable, "jobs_unavailable",
|
||||
"job status is not configured"))
|
||||
return
|
||||
}
|
||||
jobs, err := a.JobStatus.LatestJobs(r.Context(), name)
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
if jobs == nil {
|
||||
jobs = []AsyncJob{}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"server": name, "jobs": jobs})
|
||||
}
|
||||
@@ -0,0 +1,147 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
)
|
||||
|
||||
type fakeJobStatus struct {
|
||||
jobs []AsyncJob
|
||||
err error
|
||||
got string
|
||||
}
|
||||
|
||||
func (f *fakeJobStatus) LatestJobs(_ context.Context, server string) ([]AsyncJob, error) {
|
||||
f.got = server
|
||||
return f.jobs, f.err
|
||||
}
|
||||
|
||||
// TestServerJobsHandler covers GET /api/v1/servers/{name}/jobs: the owner sees the
|
||||
// recorded outcomes, strangers are refused, a nil reader is an honest 503, and a
|
||||
// reader error surfaces as 500.
|
||||
func TestServerJobsHandler(t *testing.T) {
|
||||
owner := &Principal{UserID: "owner1", Email: "[email protected]", Role: "user"}
|
||||
stranger := &Principal{UserID: "other", Email: "[email protected]", Role: "user"}
|
||||
|
||||
mk := func(reader JobStatusReader) (*API, *fakeJobStatus) {
|
||||
repo := newFakeRepo()
|
||||
repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"}
|
||||
a := newTestAPI(repo, newFakeCluster())
|
||||
a.External = staticExternal{p: owner}
|
||||
var fjs *fakeJobStatus
|
||||
if reader != nil {
|
||||
a.JobStatus = reader
|
||||
}
|
||||
if fj, ok := reader.(*fakeJobStatus); ok {
|
||||
fjs = fj
|
||||
}
|
||||
return a, fjs
|
||||
}
|
||||
|
||||
t.Run("owner sees failed job", func(t *testing.T) {
|
||||
fjs := &fakeJobStatus{jobs: []AsyncJob{{Name: "restore-survival", Kind: "restore", State: "failed", Message: "backoff limit exceeded"}}}
|
||||
a, _ := mk(fjs)
|
||||
w := do(a.ExternalHandler(), "GET", "/api/v1/servers/survival/jobs", "", nil)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
var resp struct {
|
||||
Server string `json:"server"`
|
||||
Jobs []AsyncJob `json:"jobs"`
|
||||
}
|
||||
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
|
||||
t.Fatalf("bad JSON: %v", err)
|
||||
}
|
||||
if resp.Server != "survival" || len(resp.Jobs) != 1 || resp.Jobs[0].State != "failed" {
|
||||
t.Fatalf("unexpected payload %+v", resp)
|
||||
}
|
||||
if fjs.got != "survival" {
|
||||
t.Fatalf("reader asked for %q", fjs.got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("stranger -> 403", func(t *testing.T) {
|
||||
fjs := &fakeJobStatus{}
|
||||
a, _ := mk(fjs)
|
||||
a.External = staticExternal{p: stranger}
|
||||
if w := do(a.ExternalHandler(), "GET", "/api/v1/servers/survival/jobs", "", nil); w.Code != http.StatusForbidden {
|
||||
t.Fatalf("code = %d, want 403", w.Code)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("nil reader -> 503", func(t *testing.T) {
|
||||
a, _ := mk(nil)
|
||||
if w := do(a.ExternalHandler(), "GET", "/api/v1/servers/survival/jobs", "", nil); w.Code != http.StatusServiceUnavailable {
|
||||
t.Fatalf("code = %d, want 503", w.Code)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("reader error -> 500", func(t *testing.T) {
|
||||
fjs := &fakeJobStatus{err: errors.New("apiserver down")}
|
||||
a, _ := mk(fjs)
|
||||
if w := do(a.ExternalHandler(), "GET", "/api/v1/servers/survival/jobs", "", nil); w.Code != http.StatusInternalServerError {
|
||||
t.Fatalf("code = %d, want 500", w.Code)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("unknown server -> 404", func(t *testing.T) {
|
||||
a, _ := mk(&fakeJobStatus{})
|
||||
if w := do(a.ExternalHandler(), "GET", "/api/v1/servers/ghost/jobs", "", nil); w.Code != http.StatusNotFound {
|
||||
t.Fatalf("code = %d, want 404", w.Code)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestJobToAsyncJob pins the Job→AsyncJob projection: kinds come from managed-by,
|
||||
// terminal conditions decide state, and unknown owners are dropped.
|
||||
func TestJobToAsyncJob(t *testing.T) {
|
||||
start := metav1.NewTime(time.Date(2026, 9, 22, 10, 0, 0, 0, time.UTC))
|
||||
done := metav1.NewTime(time.Date(2026, 9, 22, 10, 5, 0, 0, time.UTC))
|
||||
|
||||
failed := &batchv1.Job{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "restore-survival", Labels: map[string]string{jobManagedByLabel: jobManagedByRestore}},
|
||||
Status: batchv1.JobStatus{
|
||||
StartTime: &start,
|
||||
Conditions: []batchv1.JobCondition{{
|
||||
Type: batchv1.JobFailed, Status: corev1.ConditionTrue,
|
||||
Reason: "BackoffLimitExceeded", Message: "Job has reached the specified backoff limit",
|
||||
LastTransitionTime: done,
|
||||
}},
|
||||
},
|
||||
}
|
||||
aj, ok := jobToAsyncJob(failed)
|
||||
if !ok || aj.Kind != "restore" || aj.State != "failed" || aj.Name != "restore-survival" {
|
||||
t.Fatalf("failed job projection = %+v ok=%v", aj, ok)
|
||||
}
|
||||
if !aj.FinishedAt.Equal(done.Time) || !aj.StartedAt.Equal(start.Time) {
|
||||
t.Fatalf("timestamps = %+v", aj)
|
||||
}
|
||||
|
||||
complete := &batchv1.Job{
|
||||
ObjectMeta: metav1.ObjectMeta{Name: "backup-survival-1", Labels: map[string]string{jobManagedByLabel: jobManagedByBackup}},
|
||||
Status: batchv1.JobStatus{Conditions: []batchv1.JobCondition{{
|
||||
Type: batchv1.JobComplete, Status: corev1.ConditionTrue, LastTransitionTime: done,
|
||||
}}},
|
||||
}
|
||||
if aj, ok := jobToAsyncJob(complete); !ok || aj.Kind != "backup" || aj.State != "succeeded" {
|
||||
t.Fatalf("complete job projection = %+v ok=%v", aj, ok)
|
||||
}
|
||||
|
||||
running := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Labels: map[string]string{jobManagedByLabel: jobManagedByBackup}}}
|
||||
if aj, ok := jobToAsyncJob(running); !ok || aj.State != "running" {
|
||||
t.Fatalf("running job projection = %+v ok=%v", aj, ok)
|
||||
}
|
||||
|
||||
foreign := &batchv1.Job{ObjectMeta: metav1.ObjectMeta{Labels: map[string]string{jobManagedByLabel: "someone-else"}}}
|
||||
if _, ok := jobToAsyncJob(foreign); ok {
|
||||
t.Fatal("foreign job must be dropped")
|
||||
}
|
||||
}
|
||||
@@ -43,6 +43,22 @@ func (k *K8sCluster) GetServer(ctx context.Context, name string) (*ServerInfo, e
|
||||
return serverInfo(&ms), nil
|
||||
}
|
||||
|
||||
// WorldVolumeExists reads the world PVC the operator's StatefulSet
|
||||
// volumeClaimTemplate creates (naming.WorldPVCName — the same name the backup
|
||||
// and restore Jobs mount), so existence here is exactly existence at Job mount
|
||||
// time. NotFound is (false, nil): the caller refuses with a specific 409.
|
||||
func (k *K8sCluster) WorldVolumeExists(ctx context.Context, name string) (bool, error) {
|
||||
var pvc corev1.PersistentVolumeClaim
|
||||
err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: naming.WorldPVCName(name)}, &pvc)
|
||||
if apierrors.IsNotFound(err) {
|
||||
return false, nil
|
||||
}
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
|
||||
func (k *K8sCluster) GetBySubdomain(ctx context.Context, subdomain string) (*ServerInfo, error) {
|
||||
var list v1alpha1.MinecraftServerList
|
||||
if err := k.c.List(ctx, &list, client.InNamespace(k.namespace)); err != nil {
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sort"
|
||||
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
"sigs.k8s.io/controller-runtime/pkg/client"
|
||||
)
|
||||
|
||||
// Labels the backup/restore executors apply (internal/backupjob and
|
||||
// internal/restore keep their own copies). Literals on purpose: api defines the
|
||||
// read seam and must not import the executors — their dependency direction is
|
||||
// "executors implement api's interfaces", never the reverse.
|
||||
const (
|
||||
jobServerLabel = "felis.lolicon.best/server"
|
||||
jobManagedByLabel = "app.kubernetes.io/managed-by"
|
||||
|
||||
jobManagedByBackup = "felis-backup"
|
||||
jobManagedByRestore = "felis-restore"
|
||||
)
|
||||
|
||||
// K8sJobStatus reads the async Jobs the executors created, by the server label
|
||||
// both apply.
|
||||
type K8sJobStatus struct {
|
||||
c client.Client
|
||||
namespace string
|
||||
}
|
||||
|
||||
// NewK8sJobStatus builds the reader over the cluster client and the namespace the
|
||||
// server workloads (and their Jobs) live in.
|
||||
func NewK8sJobStatus(c client.Client, namespace string) *K8sJobStatus {
|
||||
return &K8sJobStatus{c: c, namespace: namespace}
|
||||
}
|
||||
|
||||
// LatestJobs lists this server's backup/restore Jobs newest-first, capped so a
|
||||
// long history cannot balloon the response. Jobs the label selector catches but
|
||||
// another component created (unknown managed-by) are dropped.
|
||||
func (k *K8sJobStatus) LatestJobs(ctx context.Context, serverName string) ([]AsyncJob, error) {
|
||||
var list batchv1.JobList
|
||||
if err := k.c.List(ctx, &list, client.InNamespace(k.namespace),
|
||||
client.MatchingLabels{jobServerLabel: serverName}); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := make([]AsyncJob, 0, len(list.Items))
|
||||
for i := range list.Items {
|
||||
if aj, ok := jobToAsyncJob(&list.Items[i]); ok {
|
||||
out = append(out, aj)
|
||||
}
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool { return out[i].StartedAt.After(out[j].StartedAt) })
|
||||
if len(out) > 20 {
|
||||
out = out[:20]
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// jobToAsyncJob projects one Job onto its kind/state/message. Complete condition →
|
||||
// succeeded, Failed → failed with its reason (Job conditions carry the generic
|
||||
// "backoff limit exceeded" text; the pod log holds the underlying error), anything
|
||||
// else is still running.
|
||||
func jobToAsyncJob(j *batchv1.Job) (AsyncJob, bool) {
|
||||
kind := ""
|
||||
switch j.Labels[jobManagedByLabel] {
|
||||
case jobManagedByBackup:
|
||||
kind = "backup"
|
||||
case jobManagedByRestore:
|
||||
kind = "restore"
|
||||
default:
|
||||
return AsyncJob{}, false
|
||||
}
|
||||
aj := AsyncJob{Name: j.Name, Kind: kind, State: "running", StartedAt: j.CreationTimestamp.Time}
|
||||
if j.Status.StartTime != nil {
|
||||
aj.StartedAt = j.Status.StartTime.Time
|
||||
}
|
||||
for _, c := range j.Status.Conditions {
|
||||
switch {
|
||||
case c.Type == batchv1.JobComplete && c.Status == corev1.ConditionTrue:
|
||||
aj.State = "succeeded"
|
||||
aj.FinishedAt = c.LastTransitionTime.Time
|
||||
case c.Type == batchv1.JobFailed && c.Status == corev1.ConditionTrue:
|
||||
aj.State = "failed"
|
||||
aj.FinishedAt = c.LastTransitionTime.Time
|
||||
aj.Message = c.Message
|
||||
if aj.Message == "" {
|
||||
aj.Message = c.Reason
|
||||
}
|
||||
}
|
||||
}
|
||||
return aj, true
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
|
||||
"felis.lolicon.best/internal/metrics"
|
||||
|
||||
"github.com/prometheus/client_golang/prometheus"
|
||||
"github.com/prometheus/client_golang/prometheus/promhttp"
|
||||
)
|
||||
|
||||
// apiMetricsHandler serves the felis_* collectors from a process-local registry.
|
||||
// The API process is the only producer of felis_image_build_failures_total (the
|
||||
// build reconcile loop lives in cmd/felis.reconcileBuilds), so this endpoint is
|
||||
// that counter's sole scrape path; the operator's :8080 carries the fleet
|
||||
// gauges instead.
|
||||
var apiMetricsHandler = newAPIMetricsHandler()
|
||||
|
||||
func newAPIMetricsHandler() http.Handler {
|
||||
reg := prometheus.NewRegistry()
|
||||
// A fresh registry cannot already hold a collector, so Register's
|
||||
// AlreadyRegistered tolerance arm never triggers here.
|
||||
_ = metrics.Register(reg)
|
||||
return promhttp.HandlerFor(reg, promhttp.HandlerOpts{})
|
||||
}
|
||||
|
||||
// handleMetrics is mounted as a Public route on the internal face (the same
|
||||
// stance as the health probes): the internal listener is ClusterIP-only and a
|
||||
// Prometheus scrape carries no token. The external face never serves metrics.
|
||||
// See docs/troubleshooting.md §14 for what to scrape and the alert rules that
|
||||
// consume these series.
|
||||
func (a *API) handleMetrics(w http.ResponseWriter, r *http.Request) {
|
||||
apiMetricsHandler.ServeHTTP(w, r)
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/metrics"
|
||||
)
|
||||
|
||||
// TestMetricsEndpoint locks the scrape surface: the internal face serves the
|
||||
// felis_* collectors (the API process is felis_image_build_failures_total's
|
||||
// only producer), and the external face never serves metrics.
|
||||
func TestMetricsEndpoint(t *testing.T) {
|
||||
// Give every collector a sample so the exposition carries all four families:
|
||||
// an empty vector family is omitted from the text output entirely. The bumps
|
||||
// stay inside this test binary — the collectors are per-process globals.
|
||||
metrics.SyncServerGauge([]string{"Running"})
|
||||
metrics.StartDurationSeconds.Observe(3)
|
||||
metrics.ImageBuildFailuresTotal.Inc()
|
||||
metrics.ReaperWorldsDeletedTotal.Inc()
|
||||
|
||||
a := &API{}
|
||||
|
||||
w := do(a.InternalHandler(), "GET", "/metrics", "", nil)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("internal GET /metrics = %d, want 200", w.Code)
|
||||
}
|
||||
body := w.Body.String()
|
||||
for _, want := range []string{
|
||||
`felis_servers_total{state="Running"} 1`,
|
||||
"felis_start_duration_seconds_count 1",
|
||||
"felis_image_build_failures_total",
|
||||
"felis_reaper_worlds_deleted_total",
|
||||
} {
|
||||
if !strings.Contains(body, want) {
|
||||
t.Errorf("metrics exposition missing %q", want)
|
||||
}
|
||||
}
|
||||
|
||||
if w := do(a.ExternalHandler(), "GET", "/metrics", "", nil); w.Code != http.StatusNotFound {
|
||||
t.Fatalf("external GET /metrics = %d, want 404 (metrics stay internal)", w.Code)
|
||||
}
|
||||
}
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/hex"
|
||||
"errors"
|
||||
"log"
|
||||
"net/http"
|
||||
"runtime/debug"
|
||||
@@ -95,6 +96,11 @@ func (a *API) requireInternal(next http.Handler) http.Handler {
|
||||
func (a *API) requireExternal(next http.Handler) http.Handler {
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
p, err := a.External.Authenticate(r)
|
||||
if errors.Is(err, errAuthBackend) {
|
||||
// Session store unreachable — an outage, not a missing credential.
|
||||
writeError(w, r, errAuthUnavailable)
|
||||
return
|
||||
}
|
||||
if err != nil || p == nil {
|
||||
writeError(w, r, errUnauthorized)
|
||||
return
|
||||
@@ -118,7 +124,8 @@ func (a *API) requireExternal(next http.Handler) http.Handler {
|
||||
|
||||
// adminOnly gates an external-face handler on the admin Zero-Trust path. The
|
||||
// Access middleware has already authenticated; this enforces that admin-tier
|
||||
// operations both carry role=admin AND arrived via admin.* (spec §14).
|
||||
// operations both carry a staff role (admin or owner) AND arrived via admin.*
|
||||
// (spec §14).
|
||||
func (a *API) adminOnly(next http.HandlerFunc) http.HandlerFunc {
|
||||
return func(w http.ResponseWriter, r *http.Request) {
|
||||
if p := principalFromContext(r.Context()); !p.IsAdmin() {
|
||||
|
||||
+291
-85
@@ -94,8 +94,13 @@ func (p *PGRepo) VerifyLinkCode(ctx context.Context, userID, code string, now ti
|
||||
return "", "", err
|
||||
}
|
||||
|
||||
// If this UUID is already linked, only the same user may re-verify (idempotent);
|
||||
// a different user is a conflict and must not consume the code.
|
||||
// If this UUID is already linked, only the same user may re-verify (idempotent).
|
||||
// A different LIVE user is a conflict and must not consume the code. A link whose
|
||||
// account was soft-deleted is the exception: the identity is unclaimed (the
|
||||
// account is gone; e.g. a migrated source, whose retire keeps the link but is
|
||||
// otherwise dead), and the fresh in-game code proves the caller still holds this
|
||||
// UUID, so the live caller takes the link over. Disabled-but-not-deleted stays a
|
||||
// conflict — taking over a locked account's identity would bypass the lockout.
|
||||
var existingUser string
|
||||
switch err := tx.QueryRowContext(ctx,
|
||||
`SELECT user_id FROM account_links WHERE mc_uuid = $1`, mcUUID).Scan(&existingUser); {
|
||||
@@ -105,7 +110,21 @@ func (p *PGRepo) VerifyLinkCode(ctx context.Context, userID, code string, now ti
|
||||
return "", "", err
|
||||
default:
|
||||
if existingUser != userID {
|
||||
return "", "", ErrConflict
|
||||
var linkedDeleted bool
|
||||
if err := tx.QueryRowContext(ctx,
|
||||
`SELECT deleted_at IS NOT NULL FROM users WHERE id = $1`,
|
||||
existingUser).Scan(&linkedDeleted); err != nil {
|
||||
return "", "", err
|
||||
}
|
||||
if !linkedDeleted {
|
||||
return "", "", ErrConflict
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
`UPDATE account_links SET user_id = $1, auth_source = $2, verified_at = now()
|
||||
WHERE mc_uuid = $3`,
|
||||
userID, authSource, mcUUID); err != nil {
|
||||
return "", "", fmt.Errorf("take over retired link: %w", err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -159,14 +178,19 @@ func (p *PGRepo) RedeemPlayerBindCode(ctx context.Context, newUserID, code strin
|
||||
// together), where both transactions reach the inserts before either commits.
|
||||
|
||||
// Create-or-fetch keyed on the verified UUID. An already-linked role='user' player
|
||||
// is fetched (idempotent "log in via the game"); a role='admin' STAFF account is
|
||||
// refused (op.console only) BEFORE any consume, so the code survives; an unlinked
|
||||
// UUID births a fresh role='user' player with a uuid-derived unique username.
|
||||
// is fetched (idempotent "log in via the game"); any STAFF account (admin or
|
||||
// owner, i.e. role != 'user') is refused (op.console only) BEFORE any consume, so
|
||||
// the code survives; a DISABLED or soft-deleted account is refused the same way
|
||||
// (audit #33 — a dead account must not resurrect through the bind door); an
|
||||
// unlinked UUID births a fresh role='user' player with a uuid-derived unique
|
||||
// username.
|
||||
userID := newUserID
|
||||
var existingRole string
|
||||
var disabled, deleted bool
|
||||
switch err := tx.QueryRowContext(ctx,
|
||||
`SELECT u.id, u.role::text FROM account_links al JOIN users u ON u.id = al.user_id WHERE al.mc_uuid = $1`,
|
||||
mcUUID).Scan(&userID, &existingRole); {
|
||||
`SELECT u.id, u.role::text, u.disabled, u.deleted_at IS NOT NULL
|
||||
FROM account_links al JOIN users u ON u.id = al.user_id WHERE al.mc_uuid = $1`,
|
||||
mcUUID).Scan(&userID, &existingRole, &disabled, &deleted); {
|
||||
case errors.Is(err, sql.ErrNoRows):
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
`INSERT INTO users (id, username, role) VALUES ($1, $2, 'user')
|
||||
@@ -175,15 +199,21 @@ func (p *PGRepo) RedeemPlayerBindCode(ctx context.Context, newUserID, code strin
|
||||
return "", "", "", fmt.Errorf("create player: %w", err)
|
||||
}
|
||||
// Re-read by username so a cross-code race converges on the winner's row
|
||||
// (our id was discarded by DO NOTHING) instead of a bare 500.
|
||||
// (our id was discarded by DO NOTHING) instead of a bare 500 — and so a
|
||||
// deleted row squatting on the username is refused rather than reused.
|
||||
var role string
|
||||
var dis, del bool
|
||||
if err := tx.QueryRowContext(ctx,
|
||||
`SELECT id, role::text FROM users WHERE username = $1`, mcUUID).Scan(&userID, &role); err != nil {
|
||||
`SELECT id, role::text, disabled, deleted_at IS NOT NULL FROM users WHERE username = $1`,
|
||||
mcUUID).Scan(&userID, &role, &dis, &del); err != nil {
|
||||
return "", "", "", fmt.Errorf("create player: %w", err)
|
||||
}
|
||||
if role != "user" {
|
||||
return "", "", "", ErrPlayerBindForbidden
|
||||
}
|
||||
if dis || del {
|
||||
return "", "", "", ErrPlayerAccountRetired
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
`INSERT INTO account_links (user_id, mc_uuid, auth_source) VALUES ($1, $2, $3)
|
||||
ON CONFLICT (mc_uuid) DO NOTHING`,
|
||||
@@ -196,6 +226,9 @@ func (p *PGRepo) RedeemPlayerBindCode(ctx context.Context, newUserID, code strin
|
||||
if existingRole != "user" {
|
||||
return "", "", "", ErrPlayerBindForbidden // staff must use op.console
|
||||
}
|
||||
if disabled || deleted {
|
||||
return "", "", "", ErrPlayerAccountRetired
|
||||
}
|
||||
}
|
||||
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
@@ -209,7 +242,7 @@ func (p *PGRepo) RedeemPlayerBindCode(ctx context.Context, newUserID, code strin
|
||||
}
|
||||
|
||||
// CompleteOwnerSetup consumes an in-game link code, creates-or-promotes the bound
|
||||
// account to the passwordless Owner (role='admin'), enables local auth, and stores
|
||||
// account to the passwordless Owner (role='owner'), enables local auth, and stores
|
||||
// the one-time first-login token in one transaction. It is the `felis setup`
|
||||
// MC-bind path: the operator enters limbo, runs /link, and types the code here.
|
||||
// Unlike RedeemPlayerBindCode — which refuses an already-staff account so a game
|
||||
@@ -240,16 +273,16 @@ func (p *PGRepo) CompleteOwnerSetup(ctx context.Context, newUserID, code string,
|
||||
}
|
||||
|
||||
// Create-or-promote keyed on the verified UUID. An unlinked UUID births a fresh
|
||||
// staff row (role='admin') with a uuid-derived username; an already-linked
|
||||
// account is promoted to role='admin' in place (idempotent when it is already
|
||||
// staff), keeping its id and username. Setup elevates on purpose, so there is no
|
||||
// staff refusal here — that guard belongs to the player path only.
|
||||
// staff row (role='owner') with a uuid-derived username; an already-linked
|
||||
// account is promoted to role='owner' in place (idempotent when it already is),
|
||||
// keeping its id and username. Setup elevates on purpose, so there is no staff
|
||||
// refusal here — that guard belongs to the player path only.
|
||||
userID := newUserID
|
||||
switch err := tx.QueryRowContext(ctx,
|
||||
`SELECT user_id FROM account_links WHERE mc_uuid = $1`, mcUUID).Scan(&userID); {
|
||||
case errors.Is(err, sql.ErrNoRows):
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
`INSERT INTO users (id, username, role) VALUES ($1, $2, 'admin')`,
|
||||
`INSERT INTO users (id, username, role) VALUES ($1, $2, 'owner')`,
|
||||
newUserID, mcUUID); err != nil {
|
||||
return "", "", "", fmt.Errorf("create owner: %w", err)
|
||||
}
|
||||
@@ -263,7 +296,7 @@ func (p *PGRepo) CompleteOwnerSetup(ctx context.Context, newUserID, code string,
|
||||
return "", "", "", err
|
||||
default:
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
`UPDATE users SET role = 'admin' WHERE id = $1`, userID); err != nil {
|
||||
`UPDATE users SET role = 'owner' WHERE id = $1`, userID); err != nil {
|
||||
return "", "", "", fmt.Errorf("promote owner: %w", err)
|
||||
}
|
||||
}
|
||||
@@ -292,22 +325,10 @@ func (p *PGRepo) CompleteOwnerSetup(ctx context.Context, newUserID, code string,
|
||||
|
||||
// QuotaAvailable treats a missing quota row or a NULL max_servers as unlimited;
|
||||
// otherwise it compares the live owned-server count against the cap (spec §9.3).
|
||||
//
|
||||
// KNOWN-LIMITATION (audit #4, quota TOCTOU): this check and ClaimServer are two
|
||||
// separate statements, not one transaction, so the count read here is not serialized
|
||||
// against a concurrent claim's UPDATE. Two claims by the same user for two DIFFERENT
|
||||
// ownerless servers can both read count < max_servers (under READ COMMITTED neither
|
||||
// sees the other's uncommitted UPDATE) and both succeed, leaving the user one server
|
||||
// over quota. Severity is low: it over-provisions the quota by a small margin under a
|
||||
// deliberate concurrent burst — it is NOT an authorization, ownership, or isolation
|
||||
// break (each server is still claimed atomically via UPDATE ... WHERE owner_id IS
|
||||
// NULL, so two users never share one server). Closing it needs Postgres transaction
|
||||
// semantics: wrap the count and a conditional UPDATE (gated on count < max_servers) in
|
||||
// one tx under pg_advisory_xact_lock(hashtext(user_id)) — or SERIALIZABLE with a retry
|
||||
// loop — folding the gate out of the two handlers (handleClaim and the internal UUID
|
||||
// claim) into a single repo method. That is INTEGRATION-dependent: it is verifiable
|
||||
// only against a real Postgres, not the hermetic fakeRepo suite, so it is documented
|
||||
// here rather than patched blind.
|
||||
// It is the single-dimension convenience read; handlers use the four-dimension
|
||||
// QuotaCheck. The former audit-#4 TOCTOU (check and claim in separate statements)
|
||||
// is closed inside ClaimServer, which re-runs the four-dimension gate under a
|
||||
// per-user advisory lock in the SAME transaction as the ownership write.
|
||||
func (p *PGRepo) QuotaAvailable(ctx context.Context, userID string) (bool, error) {
|
||||
var maxServers sql.NullInt64
|
||||
switch err := p.db.QueryRowContext(ctx,
|
||||
@@ -333,8 +354,11 @@ func (p *PGRepo) QuotaAvailable(ctx context.Context, userID string) (bool, error
|
||||
// cached resources should be excluded ("" for a fresh claim where the row
|
||||
// doesn't exist yet). Four dimensions are checked: server count, CPU millicores,
|
||||
// memory MB, and storage MB. A NULL or missing quota row/column means unlimited
|
||||
// for that dimension. Like QuotaAvailable, the count check and the write are not
|
||||
// serialized — see the QuotaAvailable TOCTOU docstring.
|
||||
// for that dimension. As a standalone read it is advisory — it backs the
|
||||
// handler's fast-path 403 — while the AUTHORITATIVE gate for claims is the one
|
||||
// ClaimServer re-runs atomically; the resize path (server PATCH) keeps this
|
||||
// advisory shape because its write goes through the Kubernetes API, not this
|
||||
// transaction.
|
||||
func (p *PGRepo) QuotaCheck(ctx context.Context, userID string, excludeName string, incoming ResourceSpec) (bool, error) {
|
||||
var maxServers, maxCPU, maxMem, maxStor sql.NullInt64
|
||||
switch err := p.db.QueryRowContext(ctx,
|
||||
@@ -347,8 +371,7 @@ func (p *PGRepo) QuotaCheck(ctx context.Context, userID string, excludeName stri
|
||||
return false, err
|
||||
}
|
||||
|
||||
var count int64
|
||||
var cpuSum, memSum, storSum sql.NullInt64
|
||||
var count, cpuSum, memSum, storSum int64
|
||||
switch err := p.db.QueryRowContext(ctx,
|
||||
`SELECT COUNT(*), COALESCE(SUM(cached_cpu_milli), 0), COALESCE(SUM(cached_memory_mb), 0), COALESCE(SUM(cached_storage_mb), 0)
|
||||
FROM servers WHERE owner_id = $1 AND deleted_at IS NULL AND name != $2`,
|
||||
@@ -357,19 +380,7 @@ func (p *PGRepo) QuotaCheck(ctx context.Context, userID string, excludeName stri
|
||||
return false, err
|
||||
}
|
||||
|
||||
if maxServers.Valid && count >= maxServers.Int64 {
|
||||
return false, nil
|
||||
}
|
||||
if maxCPU.Valid && cpuSum.Int64+int64(incoming.CPUMilli) > maxCPU.Int64 {
|
||||
return false, nil
|
||||
}
|
||||
if maxMem.Valid && memSum.Int64+int64(incoming.MemoryMB) > maxMem.Int64 {
|
||||
return false, nil
|
||||
}
|
||||
if maxStor.Valid && storSum.Int64+int64(incoming.StorageMB) > maxStor.Int64*1024 {
|
||||
return false, nil
|
||||
}
|
||||
return true, nil
|
||||
return quotaAllows(maxServers, maxCPU, maxMem, maxStor, count, cpuSum, memSum, storSum, incoming), nil
|
||||
}
|
||||
|
||||
// UpdateServerResources updates the resource cache for a server after a spec
|
||||
@@ -398,17 +409,67 @@ func (p *PGRepo) ServerResources(ctx context.Context, name string) (ResourceSpec
|
||||
|
||||
// ClaimServer performs the atomic ownership transfer (spec §9.3). A missing
|
||||
// server is ErrNotFound; an existing-but-owned server yields claimed=false so the
|
||||
// handler can answer 409.
|
||||
// handler can answer 409; a claim that would push the user over any of the four
|
||||
// quota caps yields ErrQuotaExceeded (the handler's pre-check is a fast path,
|
||||
// this gate is the authoritative one). The whole decision — quota read,
|
||||
// per-owner aggregate, and the ownership UPDATE — runs in ONE transaction under
|
||||
// pg_advisory_xact_lock(hashtext(user_id)), so two concurrent claims by the same
|
||||
// user for two DIFFERENT ownerless servers serialize instead of both passing the
|
||||
// gate (audit #4); the row is additionally taken FOR UPDATE so concurrent claims
|
||||
// of the SAME server still resolve to exactly one winner.
|
||||
func (p *PGRepo) ClaimServer(ctx context.Context, name, userID string) (bool, error) {
|
||||
var exists bool
|
||||
if err := p.db.QueryRowContext(ctx,
|
||||
`SELECT EXISTS(SELECT 1 FROM servers WHERE name = $1 AND deleted_at IS NULL)`, name).Scan(&exists); err != nil {
|
||||
tx, err := p.db.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
if !exists {
|
||||
return false, ErrNotFound
|
||||
defer tx.Rollback() //nolint:errcheck // no-op after commit
|
||||
|
||||
// Serialize this user's claim lane: the aggregate read below and the
|
||||
// ownership write must observe one consistent quota state. A hashtext
|
||||
// collision across users merely serializes unrelated claims — never waives a
|
||||
// cap.
|
||||
if _, err := tx.ExecContext(ctx, `SELECT pg_advisory_xact_lock(hashtext($1))`, userID); err != nil {
|
||||
return false, err
|
||||
}
|
||||
res, err := p.db.ExecContext(ctx,
|
||||
|
||||
var owned sql.NullString
|
||||
var cpu, mem, stor int
|
||||
switch err := tx.QueryRowContext(ctx,
|
||||
`SELECT owner_id, cached_cpu_milli, cached_memory_mb, cached_storage_mb
|
||||
FROM servers WHERE name = $1 AND deleted_at IS NULL FOR UPDATE`,
|
||||
name).Scan(&owned, &cpu, &mem, &stor); {
|
||||
case errors.Is(err, sql.ErrNoRows):
|
||||
return false, ErrNotFound
|
||||
case err != nil:
|
||||
return false, err
|
||||
}
|
||||
if owned.Valid {
|
||||
return false, nil // already claimed → 409 at the handler
|
||||
}
|
||||
|
||||
// The four-dimension gate, re-run inside the transaction. A missing quota
|
||||
// row leaves every NullInt64 invalid → quotaAllows treats each dimension as
|
||||
// unlimited, matching QuotaCheck.
|
||||
var maxServers, maxCPU, maxMem, maxStor sql.NullInt64
|
||||
if err := tx.QueryRowContext(ctx,
|
||||
`SELECT max_servers, max_cpu_milli, max_memory_mb, max_storage_gb
|
||||
FROM quotas WHERE user_id = $1`, userID).Scan(
|
||||
&maxServers, &maxCPU, &maxMem, &maxStor); err != nil && !errors.Is(err, sql.ErrNoRows) {
|
||||
return false, err
|
||||
}
|
||||
var count, cpuSum, memSum, storSum int64
|
||||
if err := tx.QueryRowContext(ctx,
|
||||
`SELECT COUNT(*), COALESCE(SUM(cached_cpu_milli), 0), COALESCE(SUM(cached_memory_mb), 0), COALESCE(SUM(cached_storage_mb), 0)
|
||||
FROM servers WHERE owner_id = $1 AND deleted_at IS NULL AND name != $2`,
|
||||
userID, name).Scan(&count, &cpuSum, &memSum, &storSum); err != nil {
|
||||
return false, err
|
||||
}
|
||||
if !quotaAllows(maxServers, maxCPU, maxMem, maxStor, count, cpuSum, memSum, storSum,
|
||||
ResourceSpec{CPUMilli: cpu, MemoryMB: mem, StorageMB: stor}) {
|
||||
return false, ErrQuotaExceeded
|
||||
}
|
||||
|
||||
res, err := tx.ExecContext(ctx,
|
||||
`UPDATE servers SET owner_id = $2, claimed_at = now() WHERE name = $1 AND owner_id IS NULL AND deleted_at IS NULL`,
|
||||
name, userID)
|
||||
if err != nil {
|
||||
@@ -418,7 +479,35 @@ func (p *PGRepo) ClaimServer(ctx context.Context, name, userID string) (bool, er
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
return n == 1, nil
|
||||
if n != 1 {
|
||||
return false, nil
|
||||
}
|
||||
if err := tx.Commit(); err != nil {
|
||||
return false, err
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
|
||||
// quotaAllows applies the four spec §9.3 caps to one per-owner aggregate plus
|
||||
// the incoming spec. Shared by QuotaCheck (the advisory pre-check) and
|
||||
// ClaimServer (the atomic gate) so the two can never drift. An invalid (NULL or
|
||||
// missing) cap means unlimited for that dimension; storage is compared in MB
|
||||
// against max_storage_gb × 1024.
|
||||
func quotaAllows(maxServers, maxCPU, maxMem, maxStor sql.NullInt64,
|
||||
count, cpuSum, memSum, storSum int64, incoming ResourceSpec) bool {
|
||||
if maxServers.Valid && count >= maxServers.Int64 {
|
||||
return false
|
||||
}
|
||||
if maxCPU.Valid && cpuSum+int64(incoming.CPUMilli) > maxCPU.Int64 {
|
||||
return false
|
||||
}
|
||||
if maxMem.Valid && memSum+int64(incoming.MemoryMB) > maxMem.Int64 {
|
||||
return false
|
||||
}
|
||||
if maxStor.Valid && storSum+int64(incoming.StorageMB) > maxStor.Int64*1024 {
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func (p *PGRepo) UserInAllowlist(ctx context.Context, name, userID string) (bool, error) {
|
||||
@@ -443,11 +532,16 @@ func (p *PGRepo) UUIDInAllowlist(ctx context.Context, name, mcUUID string) (bool
|
||||
}
|
||||
|
||||
// UserByMCUUID resolves a verified in-game UUID to its linked user_id (spec §10
|
||||
// account_links), or ErrNotFound when the UUID is not linked to any account.
|
||||
// account_links), or ErrNotFound when the UUID is not linked to any account. A
|
||||
// link whose account is dead reads the same as no link at all (audit #33), so the
|
||||
// in-game doors never act as a retired identity.
|
||||
func (p *PGRepo) UserByMCUUID(ctx context.Context, mcUUID string) (string, error) {
|
||||
var userID string
|
||||
switch err := p.db.QueryRowContext(ctx,
|
||||
`SELECT user_id FROM account_links WHERE mc_uuid = $1`, mcUUID).Scan(&userID); {
|
||||
`SELECT al.user_id FROM account_links al
|
||||
JOIN users u ON u.id = al.user_id
|
||||
WHERE al.mc_uuid = $1 AND u.disabled = false AND u.deleted_at IS NULL`,
|
||||
mcUUID).Scan(&userID); {
|
||||
case errors.Is(err, sql.ErrNoRows):
|
||||
return "", ErrNotFound
|
||||
case err != nil:
|
||||
@@ -695,7 +789,9 @@ func (p *PGRepo) CreateEmailOTP(ctx context.Context, id, userID, email, codeHash
|
||||
// never silently accepted, and a hash mismatch costs an attempt (UPDATE attempts+1)
|
||||
// without consuming the code — a typo must not burn a still-valid code. On a match
|
||||
// the code is consumed and the user row is flipped verified, returning the proven
|
||||
// address. ErrOTPInvalid / ErrOTPLocked are the only domain errors.
|
||||
// address — unless a DIFFERENT account already proved the same address, which is
|
||||
// ErrEmailTaken with the code left unconsumed (the address, not the guess, is the
|
||||
// problem). ErrOTPInvalid / ErrOTPLocked / ErrEmailTaken are the only domain errors.
|
||||
func (p *PGRepo) VerifyEmailOTP(ctx context.Context, userID, purpose, codeHash string, now time.Time) (string, error) {
|
||||
tx, err := p.db.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
@@ -738,12 +834,35 @@ func (p *PGRepo) VerifyEmailOTP(ctx context.Context, userID, purpose, codeHash s
|
||||
return "", ErrOTPInvalid
|
||||
}
|
||||
|
||||
// A DIFFERENT account may not also prove this address: the pre-session login
|
||||
// door resolves accounts BY verified email (UserByEmail), so a second verified
|
||||
// holder would make the identity ambiguous. The code is NOT consumed and no
|
||||
// attempt is charged — the address, not the guess, is the problem. The check is
|
||||
// the application-level counterpart of the users_verified_email_unique index
|
||||
// (migration 0020), which catches a cross-user race that passes this SELECT.
|
||||
var taken bool
|
||||
if err := tx.QueryRowContext(ctx,
|
||||
`SELECT EXISTS (SELECT 1 FROM users
|
||||
WHERE lower(email) = lower($1) AND email_verified = true AND id <> $2)`,
|
||||
email, userID).Scan(&taken); err != nil {
|
||||
return "", fmt.Errorf("check verified-email uniqueness: %w", err)
|
||||
}
|
||||
if taken {
|
||||
return "", ErrEmailTaken
|
||||
}
|
||||
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
`UPDATE email_otps SET consumed_at = $2 WHERE id = $1`, id, now); err != nil {
|
||||
return "", fmt.Errorf("consume otp: %w", err)
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
`UPDATE users SET email = $2, email_verified = true WHERE id = $1`, userID, email); err != nil {
|
||||
// Lost the race the guarded SELECT above cannot serialise: the index rejects
|
||||
// the second write, and it reads as the same answer the sequential path gives.
|
||||
// The rollback undoes the OTP consumption with it, so the code stays live.
|
||||
if isUniqueViolation(err) {
|
||||
return "", ErrEmailTaken
|
||||
}
|
||||
return "", fmt.Errorf("mark email verified: %w", err)
|
||||
}
|
||||
if err := tx.Commit(); err != nil {
|
||||
@@ -833,17 +952,17 @@ func (p *PGRepo) IsUsernameBlacklisted(ctx context.Context, mcUUID string) (bool
|
||||
// who authenticates through the third-party Yggdrasil — the admin-on-Yggdrasil reclaim
|
||||
// exception (spec §B3). The EXISTS joins account_links to users on exactly three
|
||||
// conjuncts: the UUID is linked, that link authenticated via 'thirdparty', and the
|
||||
// linked user is an admin. It intentionally does not test HOW the account signs in:
|
||||
// an Operator may authenticate via SSO (Cloudflare Access, §14) or any local
|
||||
// passwordless door and must be protected just the same — the sign-in method is
|
||||
// orthogonal to "is staff" and "logs in via the Login Server". Keyed by UUID, the
|
||||
// only identity velocity holds.
|
||||
// linked user is staff (admin OR owner — the Owner is the one identity that must never
|
||||
// be displaced). It intentionally does not test HOW the account signs in: staff may
|
||||
// authenticate via SSO (Cloudflare Access, §14) or any local passwordless door and
|
||||
// must be protected just the same — the sign-in method is orthogonal to "is staff"
|
||||
// and "logs in via the Login Server". Keyed by UUID, the only identity velocity holds.
|
||||
func (p *PGRepo) IsProtectedAdminLink(ctx context.Context, mcUUID string) (bool, error) {
|
||||
var ok bool
|
||||
err := p.db.QueryRowContext(ctx,
|
||||
`SELECT EXISTS(
|
||||
SELECT 1 FROM account_links al JOIN users u ON u.id = al.user_id
|
||||
WHERE al.mc_uuid = $1 AND al.auth_source = 'thirdparty' AND u.role = 'admin')`,
|
||||
WHERE al.mc_uuid = $1 AND al.auth_source = 'thirdparty' AND u.role IN ('admin', 'owner'))`,
|
||||
mcUUID).Scan(&ok)
|
||||
return ok, err
|
||||
}
|
||||
@@ -867,13 +986,13 @@ func (p *PGRepo) UserByUsername(ctx context.Context, username string) (*StaffUse
|
||||
return &u, nil
|
||||
}
|
||||
|
||||
// AdminExists reports whether any admin account already exists. It is the
|
||||
// AdminExists reports whether any staff account (admin or owner) already exists. It is the
|
||||
// break-glass console's bootstrap-vs-recovery switch: false means the typed
|
||||
// credential mints the first Owner (no prior identity to verify against), true
|
||||
// means the operator must identify against an existing admin for accountability.
|
||||
// means the operator must identify against an existing staff account for accountability.
|
||||
// It is not on the Repo interface because only the break-glass CLI consults it.
|
||||
func (p *PGRepo) AdminExists(ctx context.Context) (bool, error) {
|
||||
const q = `SELECT 1 FROM users WHERE role = 'admin' LIMIT 1`
|
||||
const q = `SELECT 1 FROM users WHERE role IN ('admin', 'owner') LIMIT 1`
|
||||
var one int
|
||||
switch err := p.db.QueryRowContext(ctx, q).Scan(&one); {
|
||||
case errors.Is(err, sql.ErrNoRows):
|
||||
@@ -901,29 +1020,55 @@ func (p *PGRepo) UserByID(ctx context.Context, id string) (*StaffUser, error) {
|
||||
}
|
||||
|
||||
// UpsertOwner creates or resets the Owner account direct-to-Postgres (the
|
||||
// break-glass first-run / recovery path). role is forced to 'admin' — the
|
||||
// platform-level identity. On a username conflict the email is overwritten
|
||||
// while the existing id is preserved, so live sessions referencing it survive
|
||||
// a reset. The account is passwordless by design. The empty email is stored
|
||||
// as NULL (users.email is nullable).
|
||||
// break-glass first-run / recovery path). role is forced to 'owner' — the
|
||||
// platform-level identity above admin (migration 0011); every owner-tier route
|
||||
// and the panel's owner surfaces gate on exactly this role, so writing a plain
|
||||
// 'admin' here would silently strand them. On a username conflict the email is
|
||||
// overwritten while the existing id is preserved, so live sessions referencing
|
||||
// it survive a reset — and the role is re-asserted, which is also the documented
|
||||
// promotion path for a pre-0011 install whose Owner row is still 'admin'. The
|
||||
// account is passwordless by design. The empty email is stored as NULL
|
||||
// (users.email is nullable).
|
||||
func (p *PGRepo) UpsertOwner(ctx context.Context, id, username, email string) error {
|
||||
_, err := p.db.ExecContext(ctx,
|
||||
`INSERT INTO users (id, username, email, role) VALUES ($1, $2, NULLIF($3, ''), 'admin')
|
||||
ON CONFLICT (username) DO UPDATE SET email = EXCLUDED.email`,
|
||||
`INSERT INTO users (id, username, email, role) VALUES ($1, $2, NULLIF($3, ''), 'owner')
|
||||
ON CONFLICT (username) DO UPDATE SET email = EXCLUDED.email, role = 'owner'`,
|
||||
id, username, email)
|
||||
return err
|
||||
}
|
||||
|
||||
// OwnerUsername names the single active Owner seat, or "" when no owner exists.
|
||||
// It backs the console's single-seat guard: once a seat is occupied only that
|
||||
// username may be re-targeted (see cmd/felis provisionOwner), because a fresh
|
||||
// name would take the upsert's insert arm and mint a SECOND owner row that no
|
||||
// supported path can remove (the panel protects every owner row).
|
||||
func (p *PGRepo) OwnerUsername(ctx context.Context) (string, error) {
|
||||
var name string
|
||||
switch err := p.db.QueryRowContext(ctx,
|
||||
`SELECT username FROM users WHERE role = 'owner' AND deleted_at IS NULL ORDER BY created_at LIMIT 1`).Scan(&name); {
|
||||
case errors.Is(err, sql.ErrNoRows):
|
||||
return "", nil
|
||||
case err != nil:
|
||||
return "", err
|
||||
default:
|
||||
return name, nil
|
||||
}
|
||||
}
|
||||
|
||||
// InsertOperator mints a NEW Operator (additional staff admin) account
|
||||
// direct-to-Postgres. role is forced to 'admin'. UNLIKE UpsertOwner this is
|
||||
// insert-only: a username conflict is left untouched and surfaces as a driver
|
||||
// error, so adding an Operator can never silently reset the Owner's or another
|
||||
// insert-only: a username conflict leaves the existing row untouched and
|
||||
// surfaces as ErrConflict — the console routes a rename off that sentinel — so
|
||||
// adding an Operator can never silently reset the Owner's or another
|
||||
// Operator's row. The account is passwordless by design. The empty email is
|
||||
// stored as NULL.
|
||||
func (p *PGRepo) InsertOperator(ctx context.Context, id, username, email string) error {
|
||||
_, err := p.db.ExecContext(ctx,
|
||||
`INSERT INTO users (id, username, email, role) VALUES ($1, $2, NULLIF($3, ''), 'admin')`,
|
||||
id, username, email)
|
||||
if err != nil && isUniqueViolation(err) {
|
||||
return ErrConflict
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -939,9 +1084,14 @@ func (p *PGRepo) CreateSession(ctx context.Context, tokenHash, userID string, ex
|
||||
// SessionUser resolves a live (unrevoked, unexpired at now) session hash to its
|
||||
// user, or ErrNotFound.
|
||||
func (p *PGRepo) SessionUser(ctx context.Context, tokenHash string, now time.Time) (*SessionedUser, error) {
|
||||
// The disabled/deleted filter is the belt to the doors' braces: even a session
|
||||
// minted for an account that was alive a moment ago stops authenticating the
|
||||
// instant the account is disabled or soft-deleted, so every authenticated route
|
||||
// is fail-closed regardless of which door minted the cookie (audit #33).
|
||||
const q = `SELECT u.id, COALESCE(u.email, ''), u.role::text, COALESCE(u.email_verified, false)
|
||||
FROM sessions s JOIN users u ON u.id = s.user_id
|
||||
WHERE s.token_hash = $1 AND s.revoked_at IS NULL AND s.expires_at > $2`
|
||||
WHERE s.token_hash = $1 AND s.revoked_at IS NULL AND s.expires_at > $2
|
||||
AND u.disabled = false AND u.deleted_at IS NULL`
|
||||
var u SessionedUser
|
||||
switch err := p.db.QueryRowContext(ctx, q, tokenHash, now).Scan(
|
||||
&u.ID, &u.Email, &u.Role, &u.EmailVerified); {
|
||||
@@ -1397,7 +1547,15 @@ func (p *PGRepo) UpdateUser(ctx context.Context, userID string, patch UpdateUser
|
||||
}
|
||||
if patch.Email != nil {
|
||||
argn++
|
||||
sets = append(sets, fmt.Sprintf("email = NULLIF($%d, '')", argn))
|
||||
// Changing the address voids any proof of it: only VerifyEmailOTP may assert
|
||||
// a verified address (mirrors SetUserEmail's rationale — a fresh, unproven
|
||||
// value must not keep a stale verified flag that would let the pre-session
|
||||
// email login resolve the account). A no-op edit that passes the same value
|
||||
// keeps the flag; the second expression reads the OLD row, so comparing
|
||||
// there is exact.
|
||||
sets = append(sets,
|
||||
fmt.Sprintf("email = NULLIF($%d, '')", argn),
|
||||
fmt.Sprintf("email_verified = (email_verified AND email IS NOT DISTINCT FROM NULLIF($%d, ''))", argn))
|
||||
args = append(args, *patch.Email)
|
||||
}
|
||||
if patch.Role != nil {
|
||||
@@ -1463,8 +1621,11 @@ func (p *PGRepo) userView(ctx context.Context, userID string) (*UserView, error)
|
||||
}
|
||||
|
||||
// DeleteUser soft-deletes a user in one transaction: sets deleted_at, revokes
|
||||
// every live session, and releases every owned server. The row is preserved so
|
||||
// audit_logs.actor references survive.
|
||||
// every live session, releases every owned server, and severs the account's
|
||||
// identity assets (passkey credentials, Minecraft links) so a closed account
|
||||
// cannot keep a login credential or pin an in-game identity via
|
||||
// UNIQUE(mc_uuid)(audit #33). The user row is preserved so audit_logs.actor
|
||||
// references survive.
|
||||
func (p *PGRepo) DeleteUser(ctx context.Context, userID, _ string) error {
|
||||
tx, err := p.db.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
@@ -1497,6 +1658,18 @@ func (p *PGRepo) DeleteUser(ctx context.Context, userID, _ string) error {
|
||||
return err
|
||||
}
|
||||
|
||||
// Sever the login credentials and in-game bindings: a passkey is a standing
|
||||
// login foothold and occupies credential_id UNIQUE, and an account_links row
|
||||
// would keep the Minecraft UUID claimed forever, blocking any future binding.
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
`DELETE FROM webauthn_credentials WHERE user_id = $1`, userID); err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
`DELETE FROM account_links WHERE user_id = $1`, userID); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// Soft-delete the user row.
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
`UPDATE users SET disabled = true, deleted_at = now() WHERE id = $1`,
|
||||
@@ -1538,8 +1711,13 @@ func (p *PGRepo) SetUserDisabled(ctx context.Context, userID string, disabled bo
|
||||
// ---- quota admin ----
|
||||
|
||||
// GetQuotas returns the quotas row for a user, or a zero-value view when no
|
||||
// row exists (meaning unlimited).
|
||||
// row exists (meaning unlimited). An unknown or soft-deleted user id is
|
||||
// ErrNotFound, never a zero-value "unlimited" answer — the admin sub-resource
|
||||
// routes all 404 on a user that has no live row.
|
||||
func (p *PGRepo) GetQuotas(ctx context.Context, userID string) (*QuotaView, error) {
|
||||
if err := p.requireLiveUser(ctx, userID); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
const q = `SELECT user_id, max_servers, max_cpu_milli, max_memory_mb, max_storage_gb
|
||||
FROM quotas WHERE user_id = $1`
|
||||
v := QuotaView{UserID: userID}
|
||||
@@ -1553,9 +1731,29 @@ func (p *PGRepo) GetQuotas(ctx context.Context, userID string) (*QuotaView, erro
|
||||
return &v, nil
|
||||
}
|
||||
|
||||
// requireLiveUser gates the user-scoped admin sub-resources (quotas, account
|
||||
// links) on a live users row. Without it a write would hit the user_id foreign
|
||||
// key and surface as an opaque 500, and the read would answer as if a
|
||||
// never-existed id did; every admin route answers ErrNotFound → 404 instead.
|
||||
func (p *PGRepo) requireLiveUser(ctx context.Context, userID string) error {
|
||||
var ok bool
|
||||
if err := p.db.QueryRowContext(ctx,
|
||||
`SELECT EXISTS(SELECT 1 FROM users WHERE id = $1 AND deleted_at IS NULL)`,
|
||||
userID).Scan(&ok); err != nil {
|
||||
return err
|
||||
}
|
||||
if !ok {
|
||||
return ErrNotFound
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// SetQuotas upserts a quotas row. Nil fields are left unchanged; a non-nil
|
||||
// zero-value field clears the cap.
|
||||
func (p *PGRepo) SetQuotas(ctx context.Context, userID string, qi QuotaInput, setBy string) (*QuotaView, error) {
|
||||
if err := p.requireLiveUser(ctx, userID); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
type col struct {
|
||||
name string
|
||||
value *int
|
||||
@@ -1660,6 +1858,9 @@ func (p *PGRepo) UnlinkAccount(ctx context.Context, userID, mcUUID string) error
|
||||
// NOTHING on the UNIQUE(mc_uuid) constraint, plus an idempotency check via
|
||||
// EXISTS).
|
||||
func (p *PGRepo) LinkAccount(ctx context.Context, userID, mcUUID, authSource string) error {
|
||||
if err := p.requireLiveUser(ctx, userID); err != nil {
|
||||
return err
|
||||
}
|
||||
// Check idempotency first: already linked to this user → success.
|
||||
var exists bool
|
||||
if err := p.db.QueryRowContext(ctx,
|
||||
@@ -1880,8 +2081,13 @@ func (p *PGRepo) RedeemMigration(ctx context.Context, targetUserID, codeHash str
|
||||
func (p *PGRepo) UserByEmail(ctx context.Context, email string) (*StaffUser, error) {
|
||||
// lower() on both sides honors the interface's case-insensitivity contract
|
||||
// and matches the users_verified_email_unique index (lower(email)).
|
||||
// Disabled and soft-deleted accounts are invisible here on purpose: every caller
|
||||
// is a pre-session LOGIN door (console email/passkey, op-login, /auth/options),
|
||||
// and a dead account must not be able to mint a session again — deletion and the
|
||||
// disable lockout would otherwise be bypassable by simply logging in (audit #33).
|
||||
const q = `SELECT id, username, COALESCE(email, ''), role::text, email_verified
|
||||
FROM users WHERE lower(email) = lower($1) AND email_verified = true`
|
||||
FROM users WHERE lower(email) = lower($1) AND email_verified = true
|
||||
AND disabled = false AND deleted_at IS NULL`
|
||||
var u StaffUser
|
||||
switch err := p.db.QueryRowContext(ctx, q, email).Scan(
|
||||
&u.ID, &u.Username, &u.Email, &u.Role, &u.EmailVerified); {
|
||||
|
||||
+29
-14
@@ -77,10 +77,10 @@ type BackupRecord struct {
|
||||
}
|
||||
|
||||
// StaffUser is the login-side projection of a users row (spec §B passwordless
|
||||
// auth). Owner/Operator are role=admin rows, minted by `felis setup` (MC link)
|
||||
// and recovered by `felis breakGlass` (email OTP); players are role=user rows.
|
||||
// There is no password column — staff authenticate via email-OTP / passkey +
|
||||
// in-game approve, never a password.
|
||||
// auth). The Owner is the role=owner row, minted by `felis setup` (MC link) and
|
||||
// recovered by `felis breakGlass` (email OTP); Operators are role=admin;
|
||||
// players are role=user. There is no password column — staff authenticate via
|
||||
// email-OTP / passkey + in-game approve, never a password.
|
||||
type StaffUser struct {
|
||||
ID string
|
||||
Username string
|
||||
@@ -207,9 +207,10 @@ type Repo interface {
|
||||
// return newUserID;
|
||||
// - the uuid is already linked to a role='user' player → return THAT user
|
||||
// (idempotent "log in via the game"), consuming the code;
|
||||
// - the uuid is linked to a role='admin' STAFF account → ErrPlayerBindForbidden
|
||||
// WITHOUT consuming the code (operators use op.console behind Zero Trust; the
|
||||
// public bootstrap never mints a session for an admin identity).
|
||||
// - the uuid is linked to a STAFF account (role != 'user', i.e. admin or owner)
|
||||
// → ErrPlayerBindForbidden WITHOUT consuming the code (staff use op.console
|
||||
// behind Zero Trust; the public bootstrap never mints a session for a staff
|
||||
// identity).
|
||||
//
|
||||
// Safe as an unauthenticated entrypoint because a Bind Code is minted internal-face
|
||||
// only (CreateLinkCode), against an online-mode-verified UUID, short-TTL and
|
||||
@@ -230,7 +231,11 @@ type Repo interface {
|
||||
QuotaCheck(ctx context.Context, userID string, excludeName string, incoming ResourceSpec) (bool, error)
|
||||
// ClaimServer atomically sets owner_id where it is currently NULL and returns
|
||||
// whether a row changed. false means the server was already claimed (spec §9.3:
|
||||
// 0 rows → 409).
|
||||
// 0 rows → 409). The claim runs in one transaction that re-checks the four quota
|
||||
// dimensions under pg_advisory_xact_lock(hashtext(user_id)), so it is the
|
||||
// authoritative gate: a claim that would exceed a cap → ErrQuotaExceeded (403),
|
||||
// and two concurrent claims by one user cannot both pass (audit #4).
|
||||
// QuotaCheck remains the advisory pre-check for the handler's fast-path 403.
|
||||
ClaimServer(ctx context.Context, name, userID string) (bool, error)
|
||||
// UserInAllowlist reports whether the user's linked UUID is on the server
|
||||
// allowlist (spec §9.4).
|
||||
@@ -246,6 +251,10 @@ type Repo interface {
|
||||
// (spec §10 account_links), or ErrNotFound when the UUID is not linked. The
|
||||
// internal-face wake uses it to apply the owner bypass for a player known only
|
||||
// by UUID; an unlinked UUID simply falls through to the autostartPolicy gate.
|
||||
// Only a LIVE account resolves (audit #33): a link whose account is disabled or
|
||||
// soft-deleted carries no standing on the in-game doors — claim, menu, wake
|
||||
// authorization, op-login vouch and the QR link-status poll all read a dead
|
||||
// account exactly like an unlinked UUID, never as a retired identity.
|
||||
UserByMCUUID(ctx context.Context, mcUUID string) (userID string, err error)
|
||||
// RecordJoin updates last_active_at, clears reaper warnings, and auto-appends
|
||||
// the UUID to the allowlist (spec §7 join-event, §9.4).
|
||||
@@ -309,7 +318,10 @@ type Repo interface {
|
||||
// mismatch increments attempts and returns ErrOTPInvalid WITHOUT consuming the
|
||||
// code (so a typo does not burn it). On a match the code is consumed and the
|
||||
// user row is flipped to email=<the proven address>, email_verified=true; the
|
||||
// proven email is returned. now is the API clock so expiry is testable.
|
||||
// proven email is returned — unless a DIFFERENT account has already proven the
|
||||
// same address, which is ErrEmailTaken with the code left unconsumed (the
|
||||
// address, not the code, is the problem). now is the API clock so expiry is
|
||||
// testable.
|
||||
//
|
||||
// This is the ONBOARDING primitive: verifying the code is the moment the address
|
||||
// becomes proven, so the write is load-bearing. The pre-session LOGIN door must
|
||||
@@ -460,12 +472,13 @@ type Repo interface {
|
||||
// is staff logging in via the Login Server, not a Mojang squatter, so a
|
||||
// Mojang-priority reclaim must never bar them. The predicate is exactly three
|
||||
// conjuncts: the UUID is linked (account_links), that link authenticated via
|
||||
// 'thirdparty' (auth_source), and the linked user is an admin (role='admin').
|
||||
// 'thirdparty' (auth_source), and the linked user is staff — admin or owner
|
||||
// (migration 0011), the Owner being the identity most in need of the exception.
|
||||
// It deliberately does NOT ask HOW the staff account signs in: an Operator may
|
||||
// authenticate via SSO (Cloudflare Access, IdP-agnostic per §14) or any local
|
||||
// passwordless door, and must be protected all the same — the sign-in method is
|
||||
// orthogonal to both "is staff" and "logs in via the Login Server". An unlinked
|
||||
// UUID, a Mojang-sourced link, or a non-admin link all yield false, so the
|
||||
// UUID, a Mojang-sourced link, or a player link all yield false, so the
|
||||
// exception never broadens to ordinary thirdparty players (Mojang priority still
|
||||
// displaces them) nor to Mojang-authenticated identities (who have no Login-Server
|
||||
// name to protect). Keyed by UUID — the only identity velocity knows.
|
||||
@@ -490,7 +503,7 @@ type Repo interface {
|
||||
// (email_verified true), so a merely-asserted or unverified address never
|
||||
// resolves to a session-mintable identity — an attacker cannot claim someone
|
||||
// else's login by typing their email. Matching is on lower(email) to align with
|
||||
// the users_verified_email_unique partial index (migration 0010), which
|
||||
// the users_verified_email_unique partial index (migration 0020), which
|
||||
// guarantees at most one verified row per normalized address, so the result is
|
||||
// unambiguous. A player (role='user') row resolves too — email-first login is
|
||||
// passwordless and role-agnostic here; the door that consumes this result decides
|
||||
@@ -498,8 +511,10 @@ type Repo interface {
|
||||
UserByEmail(ctx context.Context, email string) (*StaffUser, error)
|
||||
// UpsertOwner creates or resets the single Owner account direct-to-Postgres
|
||||
// (the `felis setup` / `felis breakGlass` recovery path). role is forced to
|
||||
// 'admin'; on a username conflict the existing row's email is overwritten so
|
||||
// a reset is idempotent. The account is passwordless by design.
|
||||
// 'owner' (migration 0011 — the tier every user-admin route gates on); on a
|
||||
// username conflict the existing row's email is overwritten and the role
|
||||
// re-asserted, so a reset is idempotent and a pre-0011 'admin' Owner row is
|
||||
// promoted. The account is passwordless by design.
|
||||
UpsertOwner(ctx context.Context, id, username, email string) error
|
||||
// CreateSession records a minted session: the sha-256 of the opaque cookie
|
||||
// value, its owner, and its expiry (spec §B sessions). Only the hash is stored,
|
||||
|
||||
+35
-6
@@ -7,6 +7,7 @@ import (
|
||||
"encoding/base64"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net"
|
||||
"net/http"
|
||||
@@ -145,14 +146,24 @@ func (s SessionAuth) Authenticate(r *http.Request) (*Principal, error) {
|
||||
}
|
||||
|
||||
ctx := r.Context()
|
||||
if !localAuthEnabled(ctx, s.Repo) {
|
||||
enabled, err := localAuthEnabledStatus(ctx, s.Repo)
|
||||
if err != nil {
|
||||
// The session store is unreachable: this is an outage, not a verdict on
|
||||
// the caller's credentials, so the middleware answers 503 rather than a
|
||||
// misleading "please log in".
|
||||
return nil, fmt.Errorf("%w: %v", errAuthBackend, err)
|
||||
}
|
||||
if !enabled {
|
||||
// A cookie was presented but local auth is off: reject, never fall through.
|
||||
return nil, fmt.Errorf("local auth disabled")
|
||||
}
|
||||
|
||||
u, err := s.Repo.SessionUser(ctx, hashCookie(cookie.Value), s.now())
|
||||
if err != nil {
|
||||
switch {
|
||||
case errors.Is(err, ErrNotFound):
|
||||
return nil, fmt.Errorf("invalid session: %w", err)
|
||||
case err != nil:
|
||||
return nil, fmt.Errorf("%w: %v", errAuthBackend, err)
|
||||
}
|
||||
return &Principal{
|
||||
UserID: u.ID,
|
||||
@@ -164,6 +175,11 @@ func (s SessionAuth) Authenticate(r *http.Request) (*Principal, error) {
|
||||
}, nil
|
||||
}
|
||||
|
||||
// errAuthBackend marks an authentication failure caused by the session store
|
||||
// being unreachable (e.g. Postgres down) rather than by a missing or invalid
|
||||
// credential. Middleware maps it to 503 so an outage is not misreported as 401.
|
||||
var errAuthBackend = errors.New("auth backend unavailable")
|
||||
|
||||
// localAuthEnabled reports whether the runtime local_auth_enabled toggle is true.
|
||||
// A missing setting, a read error, or a non-true value all read as disabled — the
|
||||
// gate fails closed so local sessions are honored, and new ones minted, only on an
|
||||
@@ -171,15 +187,28 @@ func (s SessionAuth) Authenticate(r *http.Request) (*Principal, error) {
|
||||
// (minting one) consult it, so the two never disagree about whether local auth is
|
||||
// live.
|
||||
func localAuthEnabled(ctx context.Context, repo Repo) bool {
|
||||
enabled, _ := localAuthEnabledStatus(ctx, repo)
|
||||
return enabled
|
||||
}
|
||||
|
||||
// localAuthEnabledStatus is localAuthEnabled with the outage case kept apart: a
|
||||
// MISSING setting (ErrNotFound — never enabled) reads as (false, nil), while a
|
||||
// store read failure reads as (false, err) so SessionAuth can tell "local auth
|
||||
// is off" (401) from "the database is down" (503). An unreadable value still
|
||||
// fails closed as disabled — it is a config fault, not an outage.
|
||||
func localAuthEnabledStatus(ctx context.Context, repo Repo) (bool, error) {
|
||||
raw, err := repo.GetSetting(ctx, LocalAuthEnabledKey)
|
||||
if err != nil {
|
||||
return false // ErrNotFound (never enabled) or a transient read error → closed
|
||||
switch {
|
||||
case errors.Is(err, ErrNotFound):
|
||||
return false, nil
|
||||
case err != nil:
|
||||
return false, err
|
||||
}
|
||||
var enabled bool
|
||||
if err := json.Unmarshal(raw, &enabled); err != nil {
|
||||
return false
|
||||
return false, nil
|
||||
}
|
||||
return enabled
|
||||
return enabled, nil
|
||||
}
|
||||
|
||||
// ensure SessionAuth satisfies ExternalAuth at compile time.
|
||||
|
||||
+234
-7
@@ -6,6 +6,7 @@ import (
|
||||
"io"
|
||||
"net/http"
|
||||
|
||||
"felis.lolicon.best/internal/build"
|
||||
"felis.lolicon.best/internal/submit"
|
||||
)
|
||||
|
||||
@@ -42,6 +43,18 @@ type SubmissionService interface {
|
||||
// Reject is the admin's other verdict: pending_review -> rejected with a
|
||||
// required reason; it starts no build.
|
||||
Reject(ctx context.Context, id, reviewedBy, reason string) (*submit.Submission, error)
|
||||
// Withdraw retracts the caller's OWN pending submission: the row and its
|
||||
// uploaded context are deleted. A reviewed submission is frozen (409) and a
|
||||
// submission the caller does not own reads back as 404, like the upload route.
|
||||
Withdraw(ctx context.Context, id, submittedBy string) (*submit.Submission, error)
|
||||
// Delete retires any submission outright (the admin lifecycle valve): the row
|
||||
// and its uploaded context are removed, any status.
|
||||
Delete(ctx context.Context, id string) (*submit.Submission, error)
|
||||
// OpenContext returns the stored build-context blob for the internal
|
||||
// context-fetch route: the build Pod's initContainer cannot mount the uploads
|
||||
// PVC across namespaces and holds no object-store credentials, so it streams
|
||||
// the blob from the API over the service-token-gated internal face instead.
|
||||
OpenContext(ctx context.Context, id string) (io.ReadCloser, error)
|
||||
}
|
||||
|
||||
// createSubmissionRequest is the POST /me/submissions body. The user
|
||||
@@ -53,6 +66,15 @@ type createSubmissionRequest struct {
|
||||
DisplayName string `json:"display_name"`
|
||||
}
|
||||
|
||||
// The submission lane's two cooldown keys, prefixed into the shared submit
|
||||
// limiter's per-user keys. Create and upload are separate levers on purpose:
|
||||
// creating a submission and then immediately uploading its context is the lane's
|
||||
// normal shape, so one must never consume the other's window.
|
||||
const (
|
||||
submissionCreateKey = "create:"
|
||||
submissionUploadKey = "upload:"
|
||||
)
|
||||
|
||||
// rejectSubmissionRequest is the POST /submissions/{id}/reject body. A reason is
|
||||
// required (the submit layer rejects an empty one with 400).
|
||||
type rejectSubmissionRequest struct {
|
||||
@@ -68,6 +90,26 @@ func (a *API) handleCreateSubmission(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
p := principalFromContext(r.Context())
|
||||
// Reserve the per-user create cooldown BEFORE the store write. Unlike the
|
||||
// wake lever's allowed→record (whose real gate is the running cap and whose
|
||||
// effect is idempotent), a create is a non-idempotent row insertion with no
|
||||
// other bound on its rate, so a burst of truly concurrent creates must yield
|
||||
// exactly one winner per window. The deferred rollback frees the window
|
||||
// whenever the create fails — a 400 typo, a spent quota, a store error — so
|
||||
// only a row that was actually recorded consumes it.
|
||||
lim := a.submitLimiter()
|
||||
reservedAt, ok := lim.reserve(submissionCreateKey+p.UserID, a.SubmitCreateCooldown)
|
||||
if !ok {
|
||||
writeError(w, r, newError(http.StatusTooManyRequests, "submission_cooldown",
|
||||
"a submission was created recently; wait a moment before creating another"))
|
||||
return
|
||||
}
|
||||
committed := false
|
||||
defer func() {
|
||||
if !committed {
|
||||
lim.release(submissionCreateKey+p.UserID, reservedAt)
|
||||
}
|
||||
}()
|
||||
var body createSubmissionRequest
|
||||
if err := decodeJSON(w, r, &body); err != nil {
|
||||
writeError(w, r, err)
|
||||
@@ -81,6 +123,7 @@ func (a *API) handleCreateSubmission(w http.ResponseWriter, r *http.Request) {
|
||||
writeSubmitError(w, r, err)
|
||||
return
|
||||
}
|
||||
committed = true
|
||||
a.audit(r, p.Email, "submission.create", sub.ID)
|
||||
writeJSON(w, http.StatusCreated, sub)
|
||||
}
|
||||
@@ -101,19 +144,43 @@ func (a *API) handleUploadSubmissionContext(w http.ResponseWriter, r *http.Reque
|
||||
return
|
||||
}
|
||||
p := principalFromContext(r.Context())
|
||||
// Reserve the per-user upload cooldown BEFORE streaming. The body is the
|
||||
// expensive part (up to the 1 GiB blob cap), so without a reservation the
|
||||
// throttle would never bound the resource it exists for: a caller could
|
||||
// repeatedly start long uploads and abort them. Reserving also collapses the
|
||||
// lane's parallel overshoot — a burst of concurrent uploads from one user
|
||||
// yields exactly one admitted stream per replica. The rollback keeps a failed
|
||||
// upload (aborted transfer, wrong format, spent quota) from burning the
|
||||
// window, so a legit retry after a genuine failure is not punished.
|
||||
lim := a.submitLimiter()
|
||||
reservedAt, ok := lim.reserve(submissionUploadKey+p.UserID, a.SubmitUploadCooldown)
|
||||
if !ok {
|
||||
writeError(w, r, newError(http.StatusTooManyRequests, "submission_cooldown",
|
||||
"an upload was accepted recently; wait a moment before uploading again"))
|
||||
return
|
||||
}
|
||||
committed := false
|
||||
defer func() {
|
||||
if !committed {
|
||||
lim.release(submissionUploadKey+p.UserID, reservedAt)
|
||||
}
|
||||
}()
|
||||
id := r.PathValue("id")
|
||||
sub, err := a.Submissions.UploadContext(r.Context(), id, p.UserID, r.Body)
|
||||
if err != nil {
|
||||
writeSubmitError(w, r, err)
|
||||
return
|
||||
}
|
||||
committed = true
|
||||
a.audit(r, p.Email, "submission.upload", sub.ID)
|
||||
writeJSON(w, http.StatusOK, sub)
|
||||
}
|
||||
|
||||
// handleMySubmissions lists the caller's own submissions (app-tier). It scopes
|
||||
// strictly to the principal's id; there is no parameter that could widen the
|
||||
// query to another user's uploads.
|
||||
// query to another user's uploads. Each row is enriched with its linked build's
|
||||
// outcome — this list is the only player-visible outlet for a build result, so
|
||||
// a failed build is not invisible to the person who submitted it.
|
||||
func (a *API) handleMySubmissions(w http.ResponseWriter, r *http.Request) {
|
||||
if a.Submissions == nil {
|
||||
writeError(w, r, errSubmissionsUnavailable)
|
||||
@@ -125,12 +192,18 @@ func (a *API) handleMySubmissions(w http.ResponseWriter, r *http.Request) {
|
||||
writeSubmitError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"submissions": subs})
|
||||
views, err := a.submissionViews(r.Context(), subs)
|
||||
if err != nil {
|
||||
writeSubmitError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"submissions": views})
|
||||
}
|
||||
|
||||
// handleListSubmissions is the admin review queue: every submission across all
|
||||
// users, newest first (admin-tier — it reads other users' uploads, so it gates
|
||||
// on the admin Zero-Trust path via adminOnly).
|
||||
// on the admin Zero-Trust path via adminOnly). Rows carry the same build
|
||||
// outcome enrichment as /me/submissions.
|
||||
func (a *API) handleListSubmissions(w http.ResponseWriter, r *http.Request) {
|
||||
if a.Submissions == nil {
|
||||
writeError(w, r, errSubmissionsUnavailable)
|
||||
@@ -141,7 +214,49 @@ func (a *API) handleListSubmissions(w http.ResponseWriter, r *http.Request) {
|
||||
writeSubmitError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"submissions": subs})
|
||||
views, err := a.submissionViews(r.Context(), subs)
|
||||
if err != nil {
|
||||
writeSubmitError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"submissions": views})
|
||||
}
|
||||
|
||||
// submissionView is one submission row enriched with its linked build's
|
||||
// outcome. The row itself is embedded unchanged, so the wire shape only gains
|
||||
// the two optional fields; they appear solely once a build has been linked
|
||||
// (BuildID set) and its record is still readable.
|
||||
type submissionView struct {
|
||||
submit.Submission
|
||||
BuildStatus string `json:"build_status,omitempty"`
|
||||
BuildError string `json:"build_error,omitempty"`
|
||||
}
|
||||
|
||||
// submissionViews enriches each submission with its linked build's status via a
|
||||
// read-only Builder.Get — deliberately never Sync, because the 15s reconcile
|
||||
// loop owns state advance and rendering a list must not touch the cluster. A
|
||||
// submission with no linked build (never approved, or approved before the
|
||||
// hand-off could record the id), no Builder wired, or a build row that is gone
|
||||
// (ErrNotFound) renders without the extra fields; any other store failure is
|
||||
// returned so the handler reports it rather than silently dropping the outcome.
|
||||
func (a *API) submissionViews(ctx context.Context, subs []submit.Submission) ([]submissionView, error) {
|
||||
views := make([]submissionView, len(subs))
|
||||
for i, s := range subs {
|
||||
views[i] = submissionView{Submission: s}
|
||||
if a.Builder == nil || s.BuildID == "" {
|
||||
continue
|
||||
}
|
||||
bld, err := a.Builder.Get(ctx, s.BuildID)
|
||||
if errors.Is(err, build.ErrNotFound) {
|
||||
continue
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
views[i].BuildStatus = string(bld.Status)
|
||||
views[i].BuildError = bld.Error
|
||||
}
|
||||
return views, nil
|
||||
}
|
||||
|
||||
// handleApproveSubmission is the admin approve gate (admin-tier). The reviewer is
|
||||
@@ -186,6 +301,48 @@ func (a *API) handleRejectSubmission(w http.ResponseWriter, r *http.Request) {
|
||||
writeJSON(w, http.StatusOK, sub)
|
||||
}
|
||||
|
||||
// handleWithdrawSubmission retracts the caller's own pending submission
|
||||
// (app-tier): the row and its uploaded context are deleted, freeing the pending
|
||||
// slot and the storage budget for a fresh submission. The submitter is the
|
||||
// principal, never the body; a submission the caller does not own is reported as
|
||||
// 404, so this endpoint cannot probe or clear another user's uploads, and a
|
||||
// reviewed submission is 409 (its build may already be consuming the context).
|
||||
func (a *API) handleWithdrawSubmission(w http.ResponseWriter, r *http.Request) {
|
||||
if a.Submissions == nil {
|
||||
writeError(w, r, errSubmissionsUnavailable)
|
||||
return
|
||||
}
|
||||
p := principalFromContext(r.Context())
|
||||
sub, err := a.Submissions.Withdraw(r.Context(), r.PathValue("id"), p.UserID)
|
||||
if err != nil {
|
||||
writeSubmitError(w, r, err)
|
||||
return
|
||||
}
|
||||
a.audit(r, p.Email, "submission.withdraw", sub.ID)
|
||||
writeJSON(w, http.StatusOK, sub)
|
||||
}
|
||||
|
||||
// handleDeleteSubmission retires any submission outright (admin-tier): the row
|
||||
// and its uploaded context are removed, any status. This is the lane's lifecycle
|
||||
// valve — the only path that reclaims a rejected or consumed upload from the
|
||||
// uploads PVC. The reviewer identity goes to the audit event, not the (now
|
||||
// nonexistent) row. Deleting an approved submission whose build is still running
|
||||
// fails that build's context fetch; the admin has explicitly chosen to retire it.
|
||||
func (a *API) handleDeleteSubmission(w http.ResponseWriter, r *http.Request) {
|
||||
if a.Submissions == nil {
|
||||
writeError(w, r, errSubmissionsUnavailable)
|
||||
return
|
||||
}
|
||||
p := principalFromContext(r.Context())
|
||||
sub, err := a.Submissions.Delete(r.Context(), r.PathValue("id"))
|
||||
if err != nil {
|
||||
writeSubmitError(w, r, err)
|
||||
return
|
||||
}
|
||||
a.audit(r, p.Email, "submission.delete", sub.ID)
|
||||
writeJSON(w, http.StatusOK, sub)
|
||||
}
|
||||
|
||||
// errSubmissionsUnavailable is returned when the approval lane is not configured
|
||||
// on this api instance (a nil Submissions service), so the admin/app boundary is
|
||||
// still exercised before the subsystem is wired in.
|
||||
@@ -194,9 +351,10 @@ var errSubmissionsUnavailable = newError(http.StatusServiceUnavailable, "submiss
|
||||
|
||||
// writeSubmitError maps submit-package errors onto HTTP status codes. Only the
|
||||
// business sentinels are client-facing: a validation failure is 400, a missing
|
||||
// submission is 404, an already-reviewed submission is 409, and an unconfigured
|
||||
// upload transport is 503 (the store this deployment set has no implemented
|
||||
// transport — an honest "not available here", not a client error). Everything
|
||||
// submission is 404, an already-reviewed submission is 409, a spent per-user
|
||||
// allowance is 403 (the same status the server-resource quota answers with), and
|
||||
// an unconfigured upload transport is 503 (the store this deployment set has no
|
||||
// implemented transport — an honest "not available here", not a client error). Everything
|
||||
// else — including a build.ErrInvalid raised by the pre-CAS build.Validate (a
|
||||
// platform registry/context MISCONFIGURATION, never client input, since every
|
||||
// build input is platform-derived) and a post-CAS Submit hand-off failure — is a
|
||||
@@ -211,6 +369,11 @@ func writeSubmitError(w http.ResponseWriter, r *http.Request, err error) {
|
||||
case errors.Is(err, submit.ErrAlreadyReviewed):
|
||||
writeError(w, r, newError(http.StatusConflict, "already_reviewed",
|
||||
"submission has already been reviewed"))
|
||||
case errors.Is(err, submit.ErrQuotaExceeded):
|
||||
writeError(w, r, newError(http.StatusForbidden, "submission_quota_exceeded",
|
||||
"submission quota reached"))
|
||||
case errors.Is(err, submit.ErrBlobNotFound):
|
||||
writeError(w, r, newError(http.StatusNotFound, "not_found", "no context uploaded for this submission"))
|
||||
case errors.Is(err, submit.ErrUploadsUnavailable):
|
||||
writeError(w, r, newError(http.StatusServiceUnavailable, "uploads_unavailable",
|
||||
"modpack upload transport is not configured"))
|
||||
@@ -219,5 +382,69 @@ func writeSubmitError(w http.ResponseWriter, r *http.Request, err error) {
|
||||
}
|
||||
}
|
||||
|
||||
// handleInternalSubmissionContext streams a submission's stored build-context
|
||||
// tarball to the build Pod's `felis fetch-context` initContainer. It lives on the
|
||||
// internal face (service-token, no Zero Trust) because its only caller is
|
||||
// in-cluster infrastructure: the build Job runs in the build namespace, where it
|
||||
// can neither mount the uploads PVC nor hold object-store credentials, so the API
|
||||
// — which wrote the blob — is the transport. The blob is served verbatim; the
|
||||
// fetcher extracts it under a zip-slip guard, and Kaniko treats the result as
|
||||
// hostile regardless (spec §16).
|
||||
func (a *API) handleInternalSubmissionContext(w http.ResponseWriter, r *http.Request) {
|
||||
rc, ok := a.openSubmissionContext(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
streamSubmissionContext(w, rc)
|
||||
}
|
||||
|
||||
// handleAdminSubmissionContext streams a submission's stored build-context
|
||||
// tarball to a reviewing admin (admin-tier). Review is only a real gate if the
|
||||
// reviewer can inspect what they approve: the executed Dockerfile lives INSIDE
|
||||
// this tarball (build/jobspec.go pins --dockerfile=Dockerfile), so without this
|
||||
// route the human gate could not see the recipe at all. The bytes are the same
|
||||
// ones the build Pod fetches over the internal face; the attachment disposition
|
||||
// makes the browser download the attacker-supplied archive, never render it.
|
||||
func (a *API) handleAdminSubmissionContext(w http.ResponseWriter, r *http.Request) {
|
||||
rc, ok := a.openSubmissionContext(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
w.Header().Set("Content-Disposition", `attachment; filename="context.tar.gz"`)
|
||||
w.Header().Set("X-Content-Type-Options", "nosniff")
|
||||
a.audit(r, principalFromContext(r.Context()).Email, "submission.context.download", r.PathValue("id"))
|
||||
streamSubmissionContext(w, rc)
|
||||
}
|
||||
|
||||
// openSubmissionContext resolves the build-context blob named in the request
|
||||
// path, mapping the submit-layer errors onto the shared submission statuses (a
|
||||
// missing blob is 404, an unwired transport 503). On failure the error response
|
||||
// is already written and the caller must return.
|
||||
func (a *API) openSubmissionContext(w http.ResponseWriter, r *http.Request) (io.ReadCloser, bool) {
|
||||
if a.Submissions == nil {
|
||||
writeError(w, r, errSubmissionsUnavailable)
|
||||
return nil, false
|
||||
}
|
||||
rc, err := a.Submissions.OpenContext(r.Context(), r.PathValue("id"))
|
||||
if err != nil {
|
||||
writeSubmitError(w, r, err)
|
||||
return nil, false
|
||||
}
|
||||
return rc, true
|
||||
}
|
||||
|
||||
// streamSubmissionContext copies the blob to w verbatim and closes it. The
|
||||
// caller must have set every header already: the copy commits the response, so
|
||||
// a failure mid-stream can only truncate it.
|
||||
func streamSubmissionContext(w http.ResponseWriter, rc io.ReadCloser) {
|
||||
defer rc.Close()
|
||||
w.Header().Set("Content-Type", "application/gzip")
|
||||
if _, err := io.Copy(w, rc); err != nil {
|
||||
// The status is already committed; the client sees a truncated stream and
|
||||
// the fetch fails on size/extract, so there is nothing left to write here.
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// Compile-time proof that the production Manager satisfies the API interface.
|
||||
var _ SubmissionService = (*submit.Manager)(nil)
|
||||
@@ -7,8 +7,11 @@ import (
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/build"
|
||||
"felis.lolicon.best/internal/submit"
|
||||
)
|
||||
|
||||
@@ -17,24 +20,32 @@ import (
|
||||
// what the handler forwarded (the point of the owner-scoping checks: the
|
||||
// submitter and reviewer must come from the principal, never the body).
|
||||
type fakeSubmissions struct {
|
||||
created *submit.CreateRequest
|
||||
createErr error
|
||||
uploadedID string
|
||||
uploadedBy string
|
||||
uploadedN int64
|
||||
uploadErr error
|
||||
listedBy string
|
||||
byResult []submit.Submission
|
||||
byErr error
|
||||
listed []submit.Submission
|
||||
listErr error
|
||||
approvedID string
|
||||
approvedBy string
|
||||
approveErr error
|
||||
rejectedID string
|
||||
rejectedBy string
|
||||
rejectReas string
|
||||
rejectErr error
|
||||
created *submit.CreateRequest
|
||||
createErr error
|
||||
uploadedID string
|
||||
uploadedBy string
|
||||
uploadedN int64
|
||||
uploadErr error
|
||||
listedBy string
|
||||
byResult []submit.Submission
|
||||
byErr error
|
||||
listed []submit.Submission
|
||||
listErr error
|
||||
approvedID string
|
||||
approvedBy string
|
||||
approveErr error
|
||||
rejectedID string
|
||||
rejectedBy string
|
||||
rejectReas string
|
||||
rejectErr error
|
||||
withdrawnID string
|
||||
withdrawBy string
|
||||
withdrawErr error
|
||||
deletedID string
|
||||
deleteErr error
|
||||
openedID string
|
||||
openBody string
|
||||
openErr error
|
||||
}
|
||||
|
||||
func (f *fakeSubmissions) Create(_ context.Context, req submit.CreateRequest) (*submit.Submission, error) {
|
||||
@@ -82,6 +93,32 @@ func (f *fakeSubmissions) Reject(_ context.Context, id, reviewedBy, reason strin
|
||||
return &submit.Submission{ID: id, Status: submit.StatusRejected, ReviewedBy: reviewedBy, RejectReason: reason}, nil
|
||||
}
|
||||
|
||||
func (f *fakeSubmissions) Withdraw(_ context.Context, id, submittedBy string) (*submit.Submission, error) {
|
||||
f.withdrawnID, f.withdrawBy = id, submittedBy
|
||||
if f.withdrawErr != nil {
|
||||
return nil, f.withdrawErr
|
||||
}
|
||||
return &submit.Submission{ID: id, SubmittedBy: submittedBy, Status: submit.StatusPendingReview}, nil
|
||||
}
|
||||
|
||||
func (f *fakeSubmissions) Delete(_ context.Context, id string) (*submit.Submission, error) {
|
||||
f.deletedID = id
|
||||
if f.deleteErr != nil {
|
||||
return nil, f.deleteErr
|
||||
}
|
||||
return &submit.Submission{ID: id, Status: submit.StatusRejected}, nil
|
||||
}
|
||||
|
||||
// openErr injects the OpenContext outcome; the body recorder lets the internal
|
||||
// route test assert byte-exact streaming and the 404 mapping.
|
||||
func (f *fakeSubmissions) OpenContext(_ context.Context, id string) (io.ReadCloser, error) {
|
||||
f.openedID = id
|
||||
if f.openErr != nil {
|
||||
return nil, f.openErr
|
||||
}
|
||||
return io.NopCloser(strings.NewReader(f.openBody)), nil
|
||||
}
|
||||
|
||||
// appSubAPI wires a submissions service behind an ordinary user principal (the
|
||||
// app tier — /me/submissions). [email protected] / .test are deliberately not the
|
||||
// deployment domain.
|
||||
@@ -233,6 +270,69 @@ func TestUploadSubmissionContextWithoutServiceIs503(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// Withdraw retracts the caller's OWN pending submission: the submitter is the
|
||||
// principal (never the body), a reviewed submission is 409, and a foreign id is
|
||||
// 404 — the same posture as the upload route.
|
||||
func TestWithdrawSubmission(t *testing.T) {
|
||||
t.Run("withdraws as the principal", func(t *testing.T) {
|
||||
fs := &fakeSubmissions{}
|
||||
w := do(appSubAPI(fs).ExternalHandler(), "DELETE", "/api/v1/me/submissions/sub-3", "", nil)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if fs.withdrawnID != "sub-3" || fs.withdrawBy != "user-7" {
|
||||
t.Fatalf("withdraw forwarded (%q, %q), want (sub-3, user-7)", fs.withdrawnID, fs.withdrawBy)
|
||||
}
|
||||
})
|
||||
t.Run("reviewed submission is 409", func(t *testing.T) {
|
||||
fs := &fakeSubmissions{withdrawErr: submit.ErrAlreadyReviewed}
|
||||
w := do(appSubAPI(fs).ExternalHandler(), "DELETE", "/api/v1/me/submissions/sub-3", "", nil)
|
||||
if w.Code != http.StatusConflict {
|
||||
t.Fatalf("code = %d, want 409 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if got := decodeErr(t, w); got != "already_reviewed" {
|
||||
t.Errorf("error code = %q, want already_reviewed", got)
|
||||
}
|
||||
})
|
||||
t.Run("foreign or unknown id is 404", func(t *testing.T) {
|
||||
fs := &fakeSubmissions{withdrawErr: submit.ErrNotFound}
|
||||
w := do(appSubAPI(fs).ExternalHandler(), "DELETE", "/api/v1/me/submissions/sub-x", "", nil)
|
||||
if w.Code != http.StatusNotFound {
|
||||
t.Fatalf("code = %d, want 404 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
t.Run("no service is 503", func(t *testing.T) {
|
||||
app := appSubAPI(nil)
|
||||
app.Submissions = nil
|
||||
w := do(app.ExternalHandler(), "DELETE", "/api/v1/me/submissions/sub-3", "", nil)
|
||||
if w.Code != http.StatusServiceUnavailable {
|
||||
t.Fatalf("code = %d, want 503 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// The admin delete retires any submission and maps the lane's 404; the route's
|
||||
// admin gate itself is pinned by TestSubmissionAdminRoutesAreAdminOnly.
|
||||
func TestDeleteSubmissionAdmin(t *testing.T) {
|
||||
t.Run("deletes the named row", func(t *testing.T) {
|
||||
fs := &fakeSubmissions{}
|
||||
w := do(adminSubAPI(fs).ExternalHandler(), "DELETE", "/api/v1/submissions/sub-8", "", nil)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if fs.deletedID != "sub-8" {
|
||||
t.Fatalf("delete forwarded id %q, want sub-8", fs.deletedID)
|
||||
}
|
||||
})
|
||||
t.Run("unknown is 404", func(t *testing.T) {
|
||||
fs := &fakeSubmissions{deleteErr: submit.ErrNotFound}
|
||||
w := do(adminSubAPI(fs).ExternalHandler(), "DELETE", "/api/v1/submissions/sub-x", "", nil)
|
||||
if w.Code != http.StatusNotFound {
|
||||
t.Fatalf("code = %d, want 404 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// The "my uploads" list scopes strictly to the principal's id — there is no
|
||||
// parameter that could widen it to another user's submissions.
|
||||
func TestMySubmissionsScopesToPrincipal(t *testing.T) {
|
||||
@@ -256,6 +356,68 @@ func TestMySubmissionsScopesToPrincipal(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The "my uploads" list carries the linked build's outcome — for a submitter it
|
||||
// is the only visible outlet for a failed build (the /images/build routes are
|
||||
// admin-tier). A row with no linked build gains no build fields.
|
||||
func TestMySubmissionsCarriesBuildOutcome(t *testing.T) {
|
||||
fs := &fakeSubmissions{byResult: []submit.Submission{
|
||||
{ID: "sub-1", SubmittedBy: "user-7", Status: submit.StatusApproved, BuildID: "bld-9"},
|
||||
{ID: "sub-2", SubmittedBy: "user-7", Status: submit.StatusPendingReview},
|
||||
}}
|
||||
api := appSubAPI(fs)
|
||||
api.Builder = &fakeBuilder{getBuilds: map[string]*build.Build{
|
||||
"bld-9": {ID: "bld-9", Status: build.StatusFailed, Error: "build job failed or scan found a CRITICAL CVE"},
|
||||
}}
|
||||
w := do(api.ExternalHandler(), "GET", "/api/v1/me/submissions", "", nil)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
var got struct {
|
||||
Submissions []struct {
|
||||
ID string `json:"id"`
|
||||
BuildStatus string `json:"build_status"`
|
||||
BuildError string `json:"build_error"`
|
||||
} `json:"submissions"`
|
||||
}
|
||||
if err := json.Unmarshal(w.Body.Bytes(), &got); err != nil {
|
||||
t.Fatalf("body not JSON: %v", err)
|
||||
}
|
||||
if len(got.Submissions) != 2 {
|
||||
t.Fatalf("submissions = %d, want 2", len(got.Submissions))
|
||||
}
|
||||
if got.Submissions[0].BuildStatus != "failed" || got.Submissions[0].BuildError == "" {
|
||||
t.Errorf("sub-1 outcome = %+v, want failed with the error text", got.Submissions[0])
|
||||
}
|
||||
if got.Submissions[1].BuildStatus != "" || got.Submissions[1].BuildError != "" {
|
||||
t.Errorf("sub-2 outcome = %+v, want no build fields without a linked build", got.Submissions[1])
|
||||
}
|
||||
}
|
||||
|
||||
// A linked build whose row is gone renders as "no outcome" rather than failing
|
||||
// the whole list; any other lookup failure must surface, never be swallowed.
|
||||
func TestMySubmissionsBuildLookupSemantics(t *testing.T) {
|
||||
// Missing row (ErrNotFound): 200 with no build fields.
|
||||
fs := &fakeSubmissions{byResult: []submit.Submission{{ID: "sub-1", BuildID: "bld-gone"}}}
|
||||
api := appSubAPI(fs)
|
||||
api.Builder = &fakeBuilder{}
|
||||
w := do(api.ExternalHandler(), "GET", "/api/v1/me/submissions", "", nil)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("missing build row: code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if strings.Contains(w.Body.String(), "build_status") {
|
||||
t.Errorf("missing build row: body carries build fields: %s", w.Body.String())
|
||||
}
|
||||
|
||||
// Store fault: the failure is reported, not hidden behind a 200.
|
||||
fs = &fakeSubmissions{byResult: []submit.Submission{{ID: "sub-1", BuildID: "bld-1"}}}
|
||||
api = appSubAPI(fs)
|
||||
api.Builder = &fakeBuilder{getErr: errors.New("db down")}
|
||||
w = do(api.ExternalHandler(), "GET", "/api/v1/me/submissions", "", nil)
|
||||
if w.Code != http.StatusInternalServerError {
|
||||
t.Fatalf("store fault: code = %d, want 500 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
// Every /submissions route is admin-tier: a plain user is rejected before the
|
||||
// handler runs.
|
||||
func TestSubmissionAdminRoutesAreAdminOnly(t *testing.T) {
|
||||
@@ -265,6 +427,8 @@ func TestSubmissionAdminRoutesAreAdminOnly(t *testing.T) {
|
||||
{"GET", "/api/v1/submissions", ""},
|
||||
{"POST", "/api/v1/submissions/sub-1/approve", ""},
|
||||
{"POST", "/api/v1/submissions/sub-1/reject", `{"reason":"no"}`},
|
||||
{"DELETE", "/api/v1/submissions/sub-1", ""},
|
||||
{"GET", "/api/v1/submissions/sub-1/context", ""},
|
||||
}
|
||||
for _, c := range cases {
|
||||
api := adminSubAPI(&fakeSubmissions{})
|
||||
@@ -280,9 +444,12 @@ func TestSubmissionAdminRoutesAreAdminOnly(t *testing.T) {
|
||||
func TestListSubmissionsAdmin(t *testing.T) {
|
||||
fs := &fakeSubmissions{listed: []submit.Submission{
|
||||
{ID: "sub-1", SubmittedBy: "user-7", Status: submit.StatusPendingReview},
|
||||
{ID: "sub-2", SubmittedBy: "user-9", Status: submit.StatusApproved},
|
||||
{ID: "sub-2", SubmittedBy: "user-9", Status: submit.StatusApproved, BuildID: "bld-2"},
|
||||
}}
|
||||
api := adminSubAPI(fs)
|
||||
api.Builder = &fakeBuilder{getBuilds: map[string]*build.Build{
|
||||
"bld-2": {ID: "bld-2", Status: build.StatusSucceeded},
|
||||
}}
|
||||
w := do(api.ExternalHandler(), "GET", "/api/v1/submissions", "", nil)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
@@ -294,6 +461,10 @@ func TestListSubmissionsAdmin(t *testing.T) {
|
||||
if len(got["submissions"]) != 2 {
|
||||
t.Fatalf("submissions = %d, want 2", len(got["submissions"]))
|
||||
}
|
||||
// The admin queue carries the same build outcome enrichment.
|
||||
if !strings.Contains(w.Body.String(), `"build_status":"succeeded"`) {
|
||||
t.Errorf("admin queue lacks the linked build outcome: %s", w.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
// The reviewer is the admin principal's email, never client input.
|
||||
@@ -396,3 +567,198 @@ func TestSubmissionRoutesWithoutServiceAre503(t *testing.T) {
|
||||
t.Fatalf("admin route: code = %d, want 503", w.Code)
|
||||
}
|
||||
}
|
||||
|
||||
// The internal context route is the build Pod's only read path to a submission's
|
||||
// blob: it streams the bytes verbatim, and its error mapping distinguishes a
|
||||
// missing blob (404) from an unwired transport (503).
|
||||
func TestInternalSubmissionContextRoute(t *testing.T) {
|
||||
newAPI := func(s SubmissionService) *API {
|
||||
api := newTestAPI(newFakeRepo(), newFakeCluster())
|
||||
api.Submissions = s
|
||||
return api
|
||||
}
|
||||
|
||||
t.Run("streams the blob", func(t *testing.T) {
|
||||
fs := &fakeSubmissions{openBody: "\x1f\x8b\x08\x00blob"}
|
||||
w := do(newAPI(fs).InternalHandler(), "GET", "/api/v1/internal/submissions/sub-7/context", "", nil)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
}
|
||||
if w.Body.String() != fs.openBody {
|
||||
t.Fatalf("body = %q, want the stored blob %q", w.Body.String(), fs.openBody)
|
||||
}
|
||||
if fs.openedID != "sub-7" {
|
||||
t.Fatalf("opened id = %q, want the path id", fs.openedID)
|
||||
}
|
||||
if ct := w.Header().Get("Content-Type"); ct != "application/gzip" {
|
||||
t.Fatalf("content-type = %q, want application/gzip", ct)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("missing blob is 404", func(t *testing.T) {
|
||||
fs := &fakeSubmissions{openErr: fmt.Errorf("%w: gone", submit.ErrBlobNotFound)}
|
||||
w := do(newAPI(fs).InternalHandler(), "GET", "/api/v1/internal/submissions/sub-7/context", "", nil)
|
||||
if w.Code != http.StatusNotFound {
|
||||
t.Fatalf("code = %d, want 404", w.Code)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("unwired transport is 503", func(t *testing.T) {
|
||||
w := do(newAPI(nil).InternalHandler(), "GET", "/api/v1/internal/submissions/sub-7/context", "", nil)
|
||||
if w.Code != http.StatusServiceUnavailable {
|
||||
t.Fatalf("code = %d, want 503", w.Code)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// The admin context route is the reviewer's read path to the blob they are
|
||||
// approving: an admin streams the stored bytes with a download disposition,
|
||||
// and the error mapping matches the internal route (404 missing, 503 unwired).
|
||||
// The admin-only gate itself is pinned by TestSubmissionAdminRoutesAreAdminOnly.
|
||||
func TestAdminSubmissionContextRoute(t *testing.T) {
|
||||
t.Run("streams the blob with a download disposition", func(t *testing.T) {
|
||||
fs := &fakeSubmissions{openBody: "\x1f\x8b\x08\x00blob"}
|
||||
w := do(adminSubAPI(fs).ExternalHandler(), "GET", "/api/v1/submissions/sub-7/context", "", nil)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("code = %d body %s", w.Code, w.Body.String())
|
||||
}
|
||||
if w.Body.String() != fs.openBody {
|
||||
t.Fatalf("body = %q, want the stored blob %q", w.Body.String(), fs.openBody)
|
||||
}
|
||||
if fs.openedID != "sub-7" {
|
||||
t.Fatalf("opened id = %q, want the path id", fs.openedID)
|
||||
}
|
||||
if ct := w.Header().Get("Content-Type"); ct != "application/gzip" {
|
||||
t.Fatalf("content-type = %q, want application/gzip", ct)
|
||||
}
|
||||
if cd := w.Header().Get("Content-Disposition"); cd != `attachment; filename="context.tar.gz"` {
|
||||
t.Fatalf("content-disposition = %q", cd)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("missing blob is 404", func(t *testing.T) {
|
||||
fs := &fakeSubmissions{openErr: fmt.Errorf("%w: gone", submit.ErrBlobNotFound)}
|
||||
w := do(adminSubAPI(fs).ExternalHandler(), "GET", "/api/v1/submissions/sub-7/context", "", nil)
|
||||
if w.Code != http.StatusNotFound {
|
||||
t.Fatalf("code = %d, want 404", w.Code)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("unwired transport is 503", func(t *testing.T) {
|
||||
api := adminSubAPI(nil)
|
||||
api.Submissions = nil
|
||||
w := do(api.ExternalHandler(), "GET", "/api/v1/submissions/sub-7/context", "", nil)
|
||||
if w.Code != http.StatusServiceUnavailable {
|
||||
t.Fatalf("code = %d, want 503", w.Code)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// A spent per-user allowance is 403 submission_quota_exceeded on both the create
|
||||
// and the upload path — distinctly NOT the 400 a malformed request gets, and not
|
||||
// the 429 the cooldown answers with.
|
||||
func TestSubmissionQuotaIs403(t *testing.T) {
|
||||
t.Run("create", func(t *testing.T) {
|
||||
fs := &fakeSubmissions{createErr: fmt.Errorf("%w: 5 submissions are already awaiting review", submit.ErrQuotaExceeded)}
|
||||
w := do(appSubAPI(fs).ExternalHandler(), "POST", "/api/v1/me/submissions", `{"display_name":"Pack"}`, nil)
|
||||
if w.Code != http.StatusForbidden {
|
||||
t.Fatalf("code = %d, want 403 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if got := decodeErr(t, w); got != "submission_quota_exceeded" {
|
||||
t.Errorf("error code = %q, want submission_quota_exceeded", got)
|
||||
}
|
||||
})
|
||||
t.Run("upload", func(t *testing.T) {
|
||||
fs := &fakeSubmissions{uploadErr: fmt.Errorf("%w: exceeds your remaining storage allowance", submit.ErrQuotaExceeded)}
|
||||
w := do(appSubAPI(fs).ExternalHandler(), "POST", "/api/v1/me/submissions/sub-9/context", "\x1f\x8bdata", nil)
|
||||
if w.Code != http.StatusForbidden {
|
||||
t.Fatalf("code = %d, want 403 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
if got := decodeErr(t, w); got != "submission_quota_exceeded" {
|
||||
t.Errorf("error code = %q, want submission_quota_exceeded", got)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// The per-user create cooldown bounds review-queue growth: a second create in
|
||||
// the same window is 429 submission_cooldown and never reaches the service; the
|
||||
// window recovers afterwards.
|
||||
func TestCreateSubmissionRateLimited(t *testing.T) {
|
||||
fs := &fakeSubmissions{}
|
||||
api := appSubAPI(fs)
|
||||
clock := time.Unix(1_700_000_000, 0)
|
||||
api.Now = func() time.Time { return clock }
|
||||
api.SubmitCreateCooldown = time.Minute
|
||||
eh := api.ExternalHandler()
|
||||
|
||||
if w := do(eh, "POST", "/api/v1/me/submissions", `{"display_name":"First"}`, nil); w.Code != http.StatusCreated {
|
||||
t.Fatalf("first create: code = %d, want 201 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
w := do(eh, "POST", "/api/v1/me/submissions", `{"display_name":"Second"}`, nil)
|
||||
if w.Code != http.StatusTooManyRequests || decodeErr(t, w) != "submission_cooldown" {
|
||||
t.Fatalf("immediate second create: code = %d body %s, want 429 submission_cooldown", w.Code, w.Body.String())
|
||||
}
|
||||
// The gate sits before the body handling: even a malformed request is
|
||||
// refused while the window is closed, so it cannot be used to probe.
|
||||
if w := do(eh, "POST", "/api/v1/me/submissions", `{`, nil); w.Code != http.StatusTooManyRequests {
|
||||
t.Fatalf("malformed create during cooldown: code = %d, want 429", w.Code)
|
||||
}
|
||||
clock = clock.Add(time.Minute + time.Second)
|
||||
if w := do(eh, "POST", "/api/v1/me/submissions", `{"display_name":"Third"}`, nil); w.Code != http.StatusCreated {
|
||||
t.Fatalf("post-cooldown create: code = %d, want 201 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
// A failed create frees the window: only a row that was actually recorded burns
|
||||
// the cooldown, so a validation typo is not punished with a wait.
|
||||
func TestCreateSubmissionFailureDoesNotBurnCooldown(t *testing.T) {
|
||||
fs := &fakeSubmissions{createErr: fmt.Errorf("%w: display name is required", submit.ErrInvalid)}
|
||||
api := appSubAPI(fs)
|
||||
api.Now = func() time.Time { return time.Unix(1_700_000_000, 0) }
|
||||
api.SubmitCreateCooldown = time.Minute
|
||||
eh := api.ExternalHandler()
|
||||
|
||||
if w := do(eh, "POST", "/api/v1/me/submissions", `{"display_name":""}`, nil); w.Code != http.StatusBadRequest {
|
||||
t.Fatalf("failed create: code = %d, want 400", w.Code)
|
||||
}
|
||||
fs.createErr = nil
|
||||
if w := do(eh, "POST", "/api/v1/me/submissions", `{"display_name":"Fixed"}`, nil); w.Code != http.StatusCreated {
|
||||
t.Fatalf("retry at the same instant: code = %d, want 201 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
// The per-user upload cooldown bounds context streaming: a second upload in the
|
||||
// same window is 429 submission_cooldown, and a FAILED upload frees the window
|
||||
// for an immediate retry.
|
||||
func TestUploadSubmissionContextRateLimited(t *testing.T) {
|
||||
fs := &fakeSubmissions{}
|
||||
api := appSubAPI(fs)
|
||||
clock := time.Unix(1_700_000_000, 0)
|
||||
api.Now = func() time.Time { return clock }
|
||||
api.SubmitUploadCooldown = time.Minute
|
||||
eh := api.ExternalHandler()
|
||||
body := "\x1f\x8b\x08\x00 the modpack bytes"
|
||||
|
||||
if w := do(eh, "POST", "/api/v1/me/submissions/sub-9/context", body, nil); w.Code != http.StatusOK {
|
||||
t.Fatalf("first upload: code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
w := do(eh, "POST", "/api/v1/me/submissions/sub-9/context", body, nil)
|
||||
if w.Code != http.StatusTooManyRequests || decodeErr(t, w) != "submission_cooldown" {
|
||||
t.Fatalf("immediate second upload: code = %d body %s, want 429 submission_cooldown", w.Code, w.Body.String())
|
||||
}
|
||||
|
||||
// A failed upload releases its reservation, so the user is not punished for
|
||||
// a genuine failure (aborted transfer, spent quota) with a cooldown wait.
|
||||
fs2 := &fakeSubmissions{uploadErr: submit.ErrUploadsUnavailable}
|
||||
api2 := appSubAPI(fs2)
|
||||
api2.Now = func() time.Time { return clock }
|
||||
api2.SubmitUploadCooldown = time.Minute
|
||||
eh2 := api2.ExternalHandler()
|
||||
if w := do(eh2, "POST", "/api/v1/me/submissions/sub-9/context", body, nil); w.Code != http.StatusServiceUnavailable {
|
||||
t.Fatalf("failed upload: code = %d, want 503", w.Code)
|
||||
}
|
||||
fs2.uploadErr = nil
|
||||
if w := do(eh2, "POST", "/api/v1/me/submissions/sub-9/context", body, nil); w.Code != http.StatusOK {
|
||||
t.Fatalf("retry at the same instant after failure: code = %d, want 200 (%s)", w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
@@ -87,9 +87,14 @@ type Config struct {
|
||||
// CPULimit / MemLimit cap the backup container.
|
||||
CPULimit string
|
||||
MemLimit string
|
||||
// RunAsUser / RunAsGroup / FSGroup are the Pod's runtime identity. FSGroup MUST
|
||||
// match the operator StatefulSet's runtime group so the read-only world mount is
|
||||
// readable by this Pod's uid.
|
||||
// RunAsUser / RunAsGroup / FSGroup are the Pod's runtime identity. They default
|
||||
// to ROOT (0:0) for the same reason the operator's forwarding-init container
|
||||
// runs as root: the world volume is written by the game image's own UID (root
|
||||
// for every Paper image we ship), and Paper saves files a non-root uid can
|
||||
// never read — level.dat is written mode 0600 (tar walk: permission denied,
|
||||
// verified live). DAC_OVERRIDE on the container covers images whose UID is
|
||||
// neither root nor ours. Set 0/0/0 explicitly for root; FSGroup is omitted
|
||||
// when zero.
|
||||
RunAsUser int64
|
||||
RunAsGroup int64
|
||||
FSGroup int64
|
||||
@@ -111,7 +116,6 @@ const (
|
||||
defaultDeadline = 30 * time.Minute
|
||||
defaultCPULimit = "1"
|
||||
defaultMemLimit = "1Gi"
|
||||
defaultRunAsID = int64(1000)
|
||||
defaultTTL = 10 * time.Minute
|
||||
)
|
||||
|
||||
@@ -145,15 +149,6 @@ func (c Config) withDefaults() Config {
|
||||
if c.MemLimit == "" {
|
||||
c.MemLimit = defaultMemLimit
|
||||
}
|
||||
if c.RunAsUser == 0 {
|
||||
c.RunAsUser = defaultRunAsID
|
||||
}
|
||||
if c.RunAsGroup == 0 {
|
||||
c.RunAsGroup = defaultRunAsID
|
||||
}
|
||||
if c.FSGroup == 0 {
|
||||
c.FSGroup = defaultRunAsID
|
||||
}
|
||||
if c.TTLAfterFinished <= 0 {
|
||||
c.TTLAfterFinished = defaultTTL
|
||||
}
|
||||
|
||||
@@ -148,7 +148,14 @@ func BackupJob(p JobParams) (*batchv1.Job, error) {
|
||||
Privileged: boolPtr(false),
|
||||
AllowPrivilegeEscalation: boolPtr(false),
|
||||
ReadOnlyRootFilesystem: boolPtr(true),
|
||||
Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}},
|
||||
// DAC_OVERRIDE is granted on top of dropping ALL: the pod runs as root,
|
||||
// but the world may have been written by a game image whose UID is
|
||||
// neither root nor ours, and Paper's own files are mode 0600. It is the
|
||||
// minimal extra power that makes the archive read every world shape.
|
||||
Capabilities: &corev1.Capabilities{
|
||||
Drop: []corev1.Capability{"ALL"},
|
||||
Add: []corev1.Capability{"DAC_OVERRIDE"},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
@@ -175,13 +182,12 @@ func BackupJob(p JobParams) (*batchv1.Job, error) {
|
||||
RestartPolicy: corev1.RestartPolicyNever,
|
||||
ServiceAccountName: p.ServiceAccount,
|
||||
AutomountServiceAccountToken: boolPtr(false),
|
||||
SecurityContext: &corev1.PodSecurityContext{
|
||||
RunAsNonRoot: boolPtr(true),
|
||||
RunAsUser: int64Ptr(p.RunAsUser),
|
||||
RunAsGroup: int64Ptr(p.RunAsGroup),
|
||||
FSGroup: int64Ptr(p.FSGroup),
|
||||
},
|
||||
Containers: []corev1.Container{container},
|
||||
// Root by default (see Config.RunAsUser): the world volume's
|
||||
// owner is the game image's UID, so only an owner-matching or
|
||||
// DAC-overriding uid can read it. FSGroup is omitted when unset
|
||||
// so a root pod never triggers a volume chgrp.
|
||||
SecurityContext: backupPodSecurityContext(p),
|
||||
Containers: []corev1.Container{container},
|
||||
Volumes: []corev1.Volume{
|
||||
{
|
||||
Name: worldVolume,
|
||||
@@ -240,3 +246,20 @@ func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
|
||||
func boolPtr(b bool) *bool { return &b }
|
||||
func int32Ptr(i int32) *int32 { return &i }
|
||||
func int64Ptr(i int64) *int64 { return &i }
|
||||
|
||||
// backupPodSecurityContext pins the Pod identity. RunAsNonRoot is false because
|
||||
// the default identity is root: worlds are owned by the game image's UID (root
|
||||
// for the images we ship), and Paper writes mode-0600 files a non-root reader
|
||||
// cannot open. FSGroup stays unset unless configured — a root executor must not
|
||||
// needlessly chgrp the world volume.
|
||||
func backupPodSecurityContext(p JobParams) *corev1.PodSecurityContext {
|
||||
sc := &corev1.PodSecurityContext{
|
||||
RunAsNonRoot: boolPtr(false),
|
||||
RunAsUser: int64Ptr(p.RunAsUser),
|
||||
RunAsGroup: int64Ptr(p.RunAsGroup),
|
||||
}
|
||||
if p.FSGroup > 0 {
|
||||
sc.FSGroup = int64Ptr(p.FSGroup)
|
||||
}
|
||||
return sc
|
||||
}
|
||||
@@ -23,9 +23,9 @@ func sampleJobParams() JobParams {
|
||||
Deadline: 30 * time.Minute,
|
||||
CPULimit: "1",
|
||||
MemLimit: "1Gi",
|
||||
RunAsUser: 1000,
|
||||
RunAsGroup: 1000,
|
||||
FSGroup: 1000,
|
||||
RunAsUser: 0,
|
||||
RunAsGroup: 0,
|
||||
FSGroup: 0,
|
||||
TTLAfterFinished: 10 * time.Minute,
|
||||
}
|
||||
}
|
||||
@@ -109,13 +109,30 @@ func TestBackupJobMountsTwoPVCsPlusConfigSecretOnly(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The backup container is hardened exactly like the restore/build Job containers:
|
||||
// no privilege, no escalation, read-only root fs, drop ALL capabilities.
|
||||
// The backup container is hardened like the restore/build Job containers: no
|
||||
// privilege, no escalation, read-only root fs, ALL capabilities dropped — plus
|
||||
// DAC_OVERRIDE, because the Pod runs as root and the world may have been written
|
||||
// by a game image with a different UID (verified live: a uid-1000 executor cannot
|
||||
// read Paper's mode-0600 level.dat).
|
||||
func TestBackupJobContainerIsHardened(t *testing.T) {
|
||||
job, err := BackupJob(sampleJobParams())
|
||||
if err != nil {
|
||||
t.Fatalf("BackupJob: %v", err)
|
||||
}
|
||||
pod := job.Spec.Template.Spec
|
||||
if pod.SecurityContext == nil {
|
||||
t.Fatal("pod SecurityContext is nil")
|
||||
}
|
||||
if pod.SecurityContext.RunAsNonRoot == nil || *pod.SecurityContext.RunAsNonRoot {
|
||||
t.Error("pod must NOT require non-root: root is the owner-matching default for game-image worlds")
|
||||
}
|
||||
if pod.SecurityContext.RunAsUser == nil || *pod.SecurityContext.RunAsUser != 0 ||
|
||||
pod.SecurityContext.RunAsGroup == nil || *pod.SecurityContext.RunAsGroup != 0 {
|
||||
t.Errorf("pod must run as 0:0 by default, got %+v", pod.SecurityContext)
|
||||
}
|
||||
if pod.SecurityContext.FSGroup != nil {
|
||||
t.Error("fsGroup must stay unset when zero (a root executor must not chgrp the world volume)")
|
||||
}
|
||||
sc := job.Spec.Template.Spec.Containers[0].SecurityContext
|
||||
if sc == nil {
|
||||
t.Fatal("container SecurityContext is nil")
|
||||
@@ -132,6 +149,9 @@ func TestBackupJobContainerIsHardened(t *testing.T) {
|
||||
if sc.Capabilities == nil || len(sc.Capabilities.Drop) != 1 || sc.Capabilities.Drop[0] != "ALL" {
|
||||
t.Error("capabilities must drop ALL")
|
||||
}
|
||||
if len(sc.Capabilities.Add) != 1 || sc.Capabilities.Add[0] != "DAC_OVERRIDE" {
|
||||
t.Errorf("capabilities must add exactly DAC_OVERRIDE, got %v", sc.Capabilities.Add)
|
||||
}
|
||||
}
|
||||
|
||||
// One-shot: a wedged archive must not loop, and a deadline caps it.
|
||||
|
||||
@@ -24,10 +24,10 @@ func NewK8sJobs(c client.Client) *K8sJobs {
|
||||
return &K8sJobs{c: c}
|
||||
}
|
||||
|
||||
// CreateBackupJob renders and applies the backup Job. Its name is a deterministic
|
||||
// function of the server (BackupJobName), so a concurrent backup of the same server
|
||||
// collides on Create; that collision is mapped to ErrAlreadyExists, which the
|
||||
// Backuper treats as success (idempotent enqueue).
|
||||
// CreateBackupJob renders and applies the backup Job. Its name carries a fresh
|
||||
// random suffix (see Backuper.Backup), so a repeated backup never collides with a
|
||||
// just-finished Job inside its TTL window; ErrAlreadyExists survives only as the
|
||||
// defensive no-op for the astronomically unlikely suffix collision.
|
||||
func (k *K8sJobs) CreateBackupJob(ctx context.Context, p JobParams) error {
|
||||
job, err := BackupJob(p)
|
||||
if err != nil {
|
||||
|
||||
+29
-11
@@ -215,6 +215,21 @@ type Config struct {
|
||||
// RegistryURL is the internal registry the build pushes to and Trivy scans
|
||||
// (spec §17). Image refs are validated to be under it.
|
||||
RegistryURL string
|
||||
// FelisImage is the platform image whose `fetch-context` entrypoint streams a
|
||||
// submission's context from the internal face into the build Pod. Required
|
||||
// only when a build's ContextRef is an http(s) URL (the submit lane's derived
|
||||
// shape); an install that never builds user submissions can leave it empty.
|
||||
FelisImage string
|
||||
// TrivyDBRepository overrides where Trivy fetches its vulnerability DB
|
||||
// (--db-repository). Empty keeps Trivy's upstream default, which the build
|
||||
// egress lock denies — an install with builds must point this at an internal
|
||||
// mirror (see config.RegistryConfig.TrivyDBRepository).
|
||||
TrivyDBRepository string
|
||||
// TrivyJavaDBRepository overrides where Trivy fetches its Java DB
|
||||
// (--java-db-repository), downloaded lazily for images that contain Java
|
||||
// artifacts — i.e. every real modpack. Same egress story as the
|
||||
// vulnerability DB (see config.RegistryConfig.TrivyJavaDBRepository).
|
||||
TrivyJavaDBRepository string
|
||||
// KanikoImage / TrivyImage are the executor images.
|
||||
KanikoImage string
|
||||
TrivyImage string
|
||||
@@ -349,17 +364,20 @@ func (b *Builder) Submit(ctx context.Context, req Request) (*Build, error) {
|
||||
// jobParams projects a build + config onto the inputs jobspec.go renders.
|
||||
func (b *Builder) jobParams(bld *Build, cfg Config) JobParams {
|
||||
return JobParams{
|
||||
BuildID: bld.ID,
|
||||
ImageRef: bld.ImageRef,
|
||||
ContextRef: bld.ContextRef,
|
||||
Namespace: cfg.Namespace,
|
||||
ServiceAccount: cfg.ServiceAccount,
|
||||
RegistryURL: cfg.RegistryURL,
|
||||
KanikoImage: cfg.KanikoImage,
|
||||
TrivyImage: cfg.TrivyImage,
|
||||
Deadline: cfg.Deadline,
|
||||
CPULimit: cfg.CPULimit,
|
||||
MemLimit: cfg.MemLimit,
|
||||
BuildID: bld.ID,
|
||||
ImageRef: bld.ImageRef,
|
||||
ContextRef: bld.ContextRef,
|
||||
Namespace: cfg.Namespace,
|
||||
ServiceAccount: cfg.ServiceAccount,
|
||||
RegistryURL: cfg.RegistryURL,
|
||||
FelisImage: cfg.FelisImage,
|
||||
TrivyDBRepository: cfg.TrivyDBRepository,
|
||||
TrivyJavaDBRepository: cfg.TrivyJavaDBRepository,
|
||||
KanikoImage: cfg.KanikoImage,
|
||||
TrivyImage: cfg.TrivyImage,
|
||||
Deadline: cfg.Deadline,
|
||||
CPULimit: cfg.CPULimit,
|
||||
MemLimit: cfg.MemLimit,
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+238
-25
@@ -2,8 +2,10 @@ package build
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/naming"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
networkingv1 "k8s.io/api/networking/v1"
|
||||
@@ -32,8 +34,37 @@ const (
|
||||
const (
|
||||
ContainerKaniko = "kaniko"
|
||||
ContainerTrivy = "trivy"
|
||||
// ContainerFetch is the initContainer that pulls a submission's build context
|
||||
// from the felis-api internal face and extracts it into the shared emptyDir.
|
||||
// It exists only for an http(s) ContextRef (see BuildJob); a ref Kaniko can
|
||||
// read natively (s3://) or a pre-mounted path renders no such container.
|
||||
ContainerFetch = "context-fetch"
|
||||
|
||||
// contextVolume/contextMountPath carry a fetched build context: the fetch
|
||||
// initContainer writes the extracted tree there, Kaniko reads it read-only.
|
||||
contextVolume = "context"
|
||||
contextMountPath = "/context"
|
||||
)
|
||||
|
||||
// contextSizeLimit bounds the extracted (attacker-controlled) context tree so a
|
||||
// tarball bomb wedges the build pod instead of the node's disk. The compressed
|
||||
// upload is capped at 1 GiB by the submit lane; 4 GiB leaves expansion room.
|
||||
var contextSizeLimit = resource.MustParse("4Gi")
|
||||
|
||||
// buildJobTTL is how long a finished build Job survives before the Job
|
||||
// controller deletes it — and with it the Pod whose kaniko log is the admin
|
||||
// failure-triage surface (GET /api/v1/images/build/{id}/logs).
|
||||
//
|
||||
// Every other Job family the platform renders carries a TTL (fileedit 2m,
|
||||
// backup/restore 10m); the build lane deliberately keeps a much longer one
|
||||
// because the logs are the point. Without ANY TTL the Job and its completed
|
||||
// Pod accumulate one pair per build forever: they count against the node's
|
||||
// pod budget (110 on stock k3s), grow etcd, and eventually block new builds.
|
||||
// Sync already tolerates a vanished Job (JobUnknown → failed; terminal builds
|
||||
// are returned unchanged), so a week-old log falling off costs a 404, not a
|
||||
// status flip.
|
||||
const buildJobTTL = 7 * 24 * time.Hour
|
||||
|
||||
// JobParams are the rendered inputs to a build Job. They are derived from a
|
||||
// Build + Config by the Builder; jobspec is a pure function of them so the
|
||||
// security-critical Job shape is unit-tested without a cluster.
|
||||
@@ -44,11 +75,23 @@ type JobParams struct {
|
||||
Namespace string
|
||||
ServiceAccount string
|
||||
RegistryURL string
|
||||
KanikoImage string
|
||||
TrivyImage string
|
||||
Deadline time.Duration
|
||||
CPULimit string
|
||||
MemLimit string
|
||||
// FelisImage runs the context-fetch initContainer (the felis binary's
|
||||
// fetch-context entrypoint). Required when ContextRef is an http(s) URL.
|
||||
FelisImage string
|
||||
// TrivyDBRepository overrides Trivy's vulnerability-DB source (the
|
||||
// --db-repository flag). Empty keeps Trivy's own default; see
|
||||
// build.Config.TrivyDBRepository for why an in-cluster install sets it.
|
||||
TrivyDBRepository string
|
||||
// TrivyJavaDBRepository overrides Trivy's Java-DB source (the
|
||||
// --java-db-repository flag), fetched lazily when the image contains Java
|
||||
// artifacts; empty keeps Trivy's own default, which the build egress lock
|
||||
// denies — a jar-bearing image then fails the scan.
|
||||
TrivyJavaDBRepository string
|
||||
KanikoImage string
|
||||
TrivyImage string
|
||||
Deadline time.Duration
|
||||
CPULimit string
|
||||
MemLimit string
|
||||
}
|
||||
|
||||
// BuildJobName is the deterministic Job name for a build id.
|
||||
@@ -90,44 +133,144 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
|
||||
deadline = int64(defaultDeadline / time.Second)
|
||||
}
|
||||
|
||||
// Hardened container security context shared by both build containers: no
|
||||
// privilege, no privilege escalation, drop all capabilities. Kaniko needs a
|
||||
// writable root filesystem to unpack layers, so we do not force read-only
|
||||
// root here, but it gains no privilege.
|
||||
// Hardened container security context baseline: no privilege, no privilege
|
||||
// escalation, drop all capabilities. Kaniko needs a writable root filesystem
|
||||
// to unpack layers, so we do not force read-only root here; it also needs a
|
||||
// minimal capability subset added back (kanikoSec below), while fetch and
|
||||
// trivy run with exactly this baseline.
|
||||
sec := &corev1.SecurityContext{
|
||||
Privileged: boolPtr(false),
|
||||
AllowPrivilegeEscalation: boolPtr(false),
|
||||
Capabilities: &corev1.Capabilities{Drop: []corev1.Capability{"ALL"}},
|
||||
}
|
||||
// Kaniko unpacks base-image layers as root, and the tar apply must chown/chmod
|
||||
// files to the owners the layer recorded — impossible under drop-ALL (live:
|
||||
// "failed to get filesystem from image: chown /etc/gshadow: operation not
|
||||
// permitted" for any FROM <base image>; scratch builds masked this because
|
||||
// COPY only ever creates files kaniko itself owns). Add back exactly the caps
|
||||
// the unpack needs and nothing else: CHOWN/FOWNER for the ownership and mode
|
||||
// restore, DAC_OVERRIDE to write entries whose bits would otherwise exclude
|
||||
// even root once the capability-based exemption is gone.
|
||||
kanikoSec := sec.DeepCopy()
|
||||
kanikoSec.Capabilities = &corev1.Capabilities{
|
||||
Drop: []corev1.Capability{"ALL"},
|
||||
Add: []corev1.Capability{"CHOWN", "DAC_OVERRIDE", "FOWNER"},
|
||||
}
|
||||
|
||||
// The context Kaniko reads. An http(s) ref (the submit lane's derived ref: the
|
||||
// API streams the blob on its internal face, because the build Pod can neither
|
||||
// mount the control-plane uploads PVC across namespaces nor hold object-store
|
||||
// credentials) is first fetched into a shared emptyDir; a ref Kaniko can read
|
||||
// in place (s3://, or a path an installer pre-mounted) passes through untouched.
|
||||
contextPath := p.ContextRef
|
||||
initContainers := []corev1.Container{}
|
||||
var kanikoMounts []corev1.VolumeMount
|
||||
var podVolumes []corev1.Volume
|
||||
if isHTTPContextRef(p.ContextRef) {
|
||||
if p.FelisImage == "" {
|
||||
return nil, fmt.Errorf("build: context ref %q needs FelisImage for the fetch initContainer", p.ContextRef)
|
||||
}
|
||||
contextPath = contextMountPath
|
||||
// The fetch container runs as root while Kaniko keeps the image default
|
||||
// (also root): Kaniko re-copies the Dockerfile out of the context and
|
||||
// chowns/chmods it to the SOURCE file's owner, which fails for any other
|
||||
// owner without CAP_CHOWN/CAP_FOWNER — capabilities this pod deliberately
|
||||
// drops (the live drill hit exactly this: "copying dockerfile: chown
|
||||
// /kaniko/Dockerfile: operation not permitted" with the distroless uid
|
||||
// 65532). Extracting as root, the uid Kaniko itself runs as, keeps the
|
||||
// context owned by the only user that can satisfy that copy. The pod is
|
||||
// root by necessity regardless: Kaniko unpacks base-image layers into its
|
||||
// own filesystem.
|
||||
fetchSec := sec.DeepCopy()
|
||||
fetchSec.RunAsUser = int64Ptr(0)
|
||||
fetchSec.RunAsGroup = int64Ptr(0)
|
||||
fetch := corev1.Container{
|
||||
Name: ContainerFetch,
|
||||
Image: p.FelisImage,
|
||||
Args: []string{
|
||||
"fetch-context",
|
||||
"--url=" + p.ContextRef,
|
||||
"--out=" + contextMountPath,
|
||||
},
|
||||
// The internal face is service-token gated, and the token is read from a
|
||||
// Secret the installer materializes in THIS namespace (secretKeyRef is
|
||||
// namespace-local). It is mounted into this initContainer only: the Kaniko
|
||||
// container executes the untrusted Dockerfile and must never hold it, and
|
||||
// pod containers share neither environment nor PID namespace.
|
||||
Env: []corev1.EnvVar{{
|
||||
Name: "FELIS_SERVICE_TOKEN",
|
||||
ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{
|
||||
LocalObjectReference: corev1.LocalObjectReference{Name: naming.ServiceTokenSecretName},
|
||||
Key: naming.ServiceTokenSecretKey,
|
||||
}},
|
||||
}},
|
||||
VolumeMounts: []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath}},
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
|
||||
SecurityContext: fetchSec,
|
||||
}
|
||||
initContainers = append(initContainers, fetch)
|
||||
kanikoMounts = []corev1.VolumeMount{{Name: contextVolume, MountPath: contextMountPath, ReadOnly: true}}
|
||||
podVolumes = []corev1.Volume{{
|
||||
Name: contextVolume,
|
||||
VolumeSource: corev1.VolumeSource{EmptyDir: &corev1.EmptyDirVolumeSource{
|
||||
// The extracted tree is attacker-controlled; bound it so a tarball
|
||||
// bomb wedges THIS pod (admitted failure) instead of filling the
|
||||
// node's disk. The compressed upload is capped at 1 GiB by the
|
||||
// submit lane, and 4 GiB leaves room for a typical expansion.
|
||||
SizeLimit: sizeLimitPtr(),
|
||||
}},
|
||||
}}
|
||||
}
|
||||
|
||||
kaniko := corev1.Container{
|
||||
Name: ContainerKaniko,
|
||||
Image: p.KanikoImage,
|
||||
Args: []string{
|
||||
"--dockerfile=Dockerfile",
|
||||
"--context=" + p.ContextRef,
|
||||
"--context=" + contextPath,
|
||||
"--destination=" + p.ImageRef,
|
||||
// The internal registry is in-cluster only and may serve plain HTTP;
|
||||
// it is never a public ingress (spec §17).
|
||||
// it is never a public ingress (spec §17). Both directions need the
|
||||
// insecure flags: --insecure/--skip-tls-verify cover the PUSH, while
|
||||
// the pull side needs its own pair — a Dockerfile's `FROM
|
||||
// registry.felis.svc:5000/...` otherwise fails with "server gave
|
||||
// HTTP response to HTTPS client", breaking every build based on a
|
||||
// platform image (the canonical modpack shape).
|
||||
"--insecure",
|
||||
"--skip-tls-verify",
|
||||
"--insecure-pull",
|
||||
"--skip-tls-verify-pull",
|
||||
},
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
|
||||
SecurityContext: sec,
|
||||
VolumeMounts: kanikoMounts,
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
|
||||
SecurityContext: kanikoSec,
|
||||
}
|
||||
initContainers = append(initContainers, kaniko)
|
||||
|
||||
trivyArgs := []string{
|
||||
"image",
|
||||
"--exit-code", "1",
|
||||
"--severity", "CRITICAL",
|
||||
"--no-progress",
|
||||
"--insecure",
|
||||
}
|
||||
// The DB source is configurable because the default (mirror.gcr.io/ghcr.io)
|
||||
// is exactly what the build egress lock denies: an install that never mirrors
|
||||
// the DB cannot complete a scan, and the gate fails closed on purpose. The
|
||||
// supported shape is the internal registry (`--insecure` above already covers
|
||||
// its plain HTTP).
|
||||
if p.TrivyDBRepository != "" {
|
||||
trivyArgs = append(trivyArgs, "--db-repository", p.TrivyDBRepository)
|
||||
}
|
||||
if p.TrivyJavaDBRepository != "" {
|
||||
trivyArgs = append(trivyArgs, "--java-db-repository", p.TrivyJavaDBRepository)
|
||||
}
|
||||
trivyArgs = append(trivyArgs, p.ImageRef)
|
||||
trivy := corev1.Container{
|
||||
Name: ContainerTrivy,
|
||||
Image: p.TrivyImage,
|
||||
Args: []string{
|
||||
"image",
|
||||
"--exit-code", "1",
|
||||
"--severity", "CRITICAL",
|
||||
"--no-progress",
|
||||
"--insecure",
|
||||
p.ImageRef,
|
||||
},
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: limits},
|
||||
Name: ContainerTrivy,
|
||||
Image: p.TrivyImage,
|
||||
Args: trivyArgs,
|
||||
Resources: corev1.ResourceRequirements{Limits: limits, Requests: buildRequests(limits)},
|
||||
SecurityContext: sec,
|
||||
}
|
||||
|
||||
@@ -141,14 +284,17 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
|
||||
// A poisoned build must not loop — one shot, then a terminal verdict.
|
||||
BackoffLimit: int32Ptr(0),
|
||||
ActiveDeadlineSeconds: int64Ptr(deadline),
|
||||
// ...and a finished one must not linger forever (see buildJobTTL).
|
||||
TTLSecondsAfterFinished: int32Ptr(int32(buildJobTTL / time.Second)),
|
||||
Template: corev1.PodTemplateSpec{
|
||||
ObjectMeta: metav1.ObjectMeta{Labels: buildLabels(p)},
|
||||
Spec: corev1.PodSpec{
|
||||
RestartPolicy: corev1.RestartPolicyNever,
|
||||
ServiceAccountName: p.ServiceAccount,
|
||||
AutomountServiceAccountToken: boolPtr(false),
|
||||
InitContainers: []corev1.Container{kaniko},
|
||||
InitContainers: initContainers,
|
||||
Containers: []corev1.Container{trivy},
|
||||
Volumes: podVolumes,
|
||||
},
|
||||
},
|
||||
},
|
||||
@@ -156,11 +302,24 @@ func BuildJob(p JobParams) (*batchv1.Job, error) {
|
||||
return job, nil
|
||||
}
|
||||
|
||||
// isHTTPContextRef reports whether ref is an http(s) URL — the shape the submit
|
||||
// lane derives when the API is the blob transport — i.e. a context only the
|
||||
// fetch initContainer can turn into a local path for Kaniko.
|
||||
func isHTTPContextRef(ref string) bool {
|
||||
return strings.HasPrefix(ref, "http://") || strings.HasPrefix(ref, "https://")
|
||||
}
|
||||
|
||||
// NetPolParams parameterises the build-namespace egress lock.
|
||||
type NetPolParams struct {
|
||||
Namespace string
|
||||
RegistryNamespace string
|
||||
RegistryPort int32
|
||||
// ControlNamespace and APIPort are where the felis-api internal face lives:
|
||||
// the fetch initContainer's only egress besides DNS and the registry. Both
|
||||
// defaults (felis, 8081) match platform.DefaultControlNamespace and the
|
||||
// internal listener, so an unset Params is still the safe shape.
|
||||
ControlNamespace string
|
||||
APIPort int32
|
||||
// PackageSourceCIDRs is an optional, explicit allowlist of external package
|
||||
// mirrors (spec §16: egress 仅 registry + 包源). Empty means the most
|
||||
// locked-down default — no internet egress at all (默认拒外网).
|
||||
@@ -177,10 +336,19 @@ func BuildNetworkPolicy(p NetPolParams) *networkingv1.NetworkPolicy {
|
||||
if port == 0 {
|
||||
port = 5000
|
||||
}
|
||||
controlNS := p.ControlNamespace
|
||||
if controlNS == "" {
|
||||
controlNS = "felis"
|
||||
}
|
||||
apiPort := p.APIPort
|
||||
if apiPort == 0 {
|
||||
apiPort = 8081
|
||||
}
|
||||
dnsUDP := corev1.ProtocolUDP
|
||||
dnsTCP := corev1.ProtocolTCP
|
||||
dns53 := intstr.FromInt32(53)
|
||||
regPort := intstr.FromInt32(port)
|
||||
ctxPort := intstr.FromInt32(apiPort)
|
||||
|
||||
egress := []networkingv1.NetworkPolicyEgressRule{
|
||||
// DNS resolution: port-restricted to 53, so this is not an open-internet
|
||||
@@ -203,6 +371,21 @@ func BuildNetworkPolicy(p NetPolParams) *networkingv1.NetworkPolicy {
|
||||
{Protocol: &dnsTCP, Port: ®Port},
|
||||
},
|
||||
},
|
||||
// felis-api's internal face, where the fetch initContainer streams the
|
||||
// submission's build context from. Without this rule the build Pod could
|
||||
// not read the context and every user build would fail in its first init
|
||||
// step — the default-deny here is exactly why the transport had to be
|
||||
// planned, not assumed.
|
||||
{
|
||||
To: []networkingv1.NetworkPolicyPeer{{
|
||||
NamespaceSelector: &metav1.LabelSelector{
|
||||
MatchLabels: map[string]string{"kubernetes.io/metadata.name": controlNS},
|
||||
},
|
||||
}},
|
||||
Ports: []networkingv1.NetworkPolicyPort{
|
||||
{Protocol: &dnsTCP, Port: &ctxPort},
|
||||
},
|
||||
},
|
||||
}
|
||||
// Explicit package-mirror CIDRs, when configured. No CIDR ⇒ no internet.
|
||||
for _, cidr := range p.PackageSourceCIDRs {
|
||||
@@ -279,6 +462,36 @@ func resourceLimits(cpu, mem string) (corev1.ResourceList, error) {
|
||||
}, nil
|
||||
}
|
||||
|
||||
// buildRequests is the scheduler floor a build container asks for while its
|
||||
// configured limit stays the safety cap. Reserving the full cap as a request is
|
||||
// what once made a default install on the platform's starter node (4 vCPU /
|
||||
// 5.5 GiB) unable to schedule ANY build — caught by the live end-to-end drill, not
|
||||
// by any unit test. A build is best-effort batch work: it may be throttled or
|
||||
// evicted under contention, which fails the Job loudly, and the caps still stop a
|
||||
// runaway build from exhausting the node.
|
||||
func buildRequests(limits corev1.ResourceList) corev1.ResourceList {
|
||||
req := corev1.ResourceList{}
|
||||
for res, floor := range map[corev1.ResourceName]resource.Quantity{
|
||||
corev1.ResourceCPU: resource.MustParse("250m"),
|
||||
corev1.ResourceMemory: resource.MustParse("512Mi"),
|
||||
} {
|
||||
limit, ok := limits[res]
|
||||
if ok && limit.Cmp(floor) < 0 {
|
||||
floor = limit // never ask for more than the cap
|
||||
}
|
||||
req[res] = floor
|
||||
}
|
||||
return req
|
||||
}
|
||||
|
||||
func boolPtr(b bool) *bool { return &b }
|
||||
func int32Ptr(i int32) *int32 { return &i }
|
||||
|
||||
// sizeLimitPtr returns a copy of contextSizeLimit for a VolumeSource (the API
|
||||
// object only ever gets serialized, but a shared pointer across rendered Jobs
|
||||
// invites accidental aliasing).
|
||||
func sizeLimitPtr() *resource.Quantity {
|
||||
q := contextSizeLimit
|
||||
return &q
|
||||
}
|
||||
func int64Ptr(i int64) *int64 { return &i }
|
||||
Loaded 100 of 179 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user