feat(offsite): 世界归档与数据库备份加密同步到异地 S3,reaper 确认异地副本后才删除世界

This commit is contained in:
Lemon-miaow committed 2026-09-24 19:25:16 +08:00
1 parent fa8db4c7e8
commit 47890ca913
27 files changed
+2928 -51

No files matched your search

+1
View File
@@ -963,6 +963,7 @@ IPv6-only 接入(`ssh -6 -i ~/.ssh/id_ed25519 root@fdb2:2c26:f4e4:0:21c:42ff:f
30. ~~已装机系统服的新增 CR 字段(tracker #1)~~ ✅ **已修并真机闭环**(第四十六批 #78:`sudo felis converge`——零值才填、非零不覆写;剥字段→填回→幂等三连真机过)。
31. ~~troubleshooting `[INERT]` 图例~~ ✅ **已修**(第四十六批 #79,文档级)。
32. ~~稳定版发布与 release 安装/升级通道~~ ✅ **已实证**(第四十七批:tag `v0.1.0`(`b9c97ff`)+ release run `35950722509` 双资产、`/releases/latest` 解析到 v0.1.0;无 checkout 全量安装 2m17s / 零 fail / 零 warn、registry mirror 四镜像、reaper CronJob 对齐 `felis:v0.1.0`;`felis update` 双向报告〔rc1→v0.1.0 notify / v0.1.0 up to date〕)。
33. ~~世界归档与数据库备份只在本机~~ ✅ **已修并真机闭环**(2026-09-24:`felis offsite` + `felis-offsite.timer` 每小时把世界归档与 DB bundle 以 AES-256-GCM 分段加密推到 S3 兼容桶,`world_backups.offsite_at` 记账〔迁移 0024〕;配了 `[offsite]` 后 reaper 只在归档的异地副本确认后才删 PVC;watchdog 12 小时无成功同步即告警;安装器 `FELIS_OFFSITE_*` 生成密钥并在收尾横幅要求离机保存。真机(MinIO):首次同步 8 世界 1.2 GiB + 12 bundle、两条"记录在册但盘上已无"的历史归档如实列出;`fetch-db latest` sha256 与本机一致、错误密钥拒绝且不留半成品;`fetch-worlds` 取回被挪走的归档 sha256 一致;reaper 演练〔20 天空闲世界〕第一轮 `awaiting_offsite=1` 且 PVC 保留 → 同步后第二轮 `reaped=1`、PVC 删除、所有权释放;watchdog 把 status 改成 35 小时前 → `[warning] offsite`。演练装置已清理)。
## 剩余待演练队列(截至第四十三批)
+556
View File
@@ -0,0 +1,556 @@
package main
import (
"bufio"
"context"
"errors"
"flag"
"fmt"
"io"
"os"
"path/filepath"
"strings"
"time"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/dbbackup"
"felis.lolicon.best/internal/offsite"
"felis.lolicon.best/internal/platform"
"felis.lolicon.best/internal/store"
appsv1 "k8s.io/api/apps/v1"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/types"
"sigs.k8s.io/controller-runtime/pkg/client"
)
const offsiteUsage = `usage:
felis offsite sync [-config path] [-archive-dir dir] [-db-dir dir] [-status-file path]
felis offsite status [-config path] [-status-file path]
felis offsite list [-config path]
felis offsite fetch-db [-config path | -endpoint url -bucket name [-region r] [-prefix p]]
[-dir dir] latest|<bundle>
felis offsite fetch-worlds [-config path] [-archive-dir dir]
felis offsite keygen
Every verb but keygen reads the bucket credentials and the encryption key from
the variables [offsite] names (default FELIS_OFFSITE_ACCESS_KEY,
FELIS_OFFSITE_SECRET_KEY, FELIS_OFFSITE_KEY), taking any that are unset from
-env-file (default /etc/felis/offsite.env).
`
// defaultOffsiteEnvFile is where bootstrap keeps the [offsite] secrets; the
// felis-offsite.service unit loads it as its EnvironmentFile.
const defaultOffsiteEnvFile = "/etc/felis/offsite.env"
// cmdOffsite implements `felis offsite`: the off-site copy of the world
// archives and the database bundles (internal/offsite). felis-offsite.timer
// runs `sync` hourly on the host; the fetch verbs are the way back after the
// node is lost (docs/troubleshooting.md §16).
func cmdOffsite(args []string, stdout, stderr io.Writer) int {
if len(args) == 0 {
fmt.Fprint(stderr, offsiteUsage)
return 2
}
verb, rest := args[0], args[1:]
fs := flag.NewFlagSet("offsite "+verb, flag.ContinueOnError)
fs.SetOutput(stderr)
fs.Usage = func() { fmt.Fprint(stderr, offsiteUsage) }
switch verb {
case "sync":
return offsiteSync(fs, rest, stdout, stderr)
case "status":
return offsiteStatus(fs, rest, stdout, stderr)
case "list":
return offsiteList(fs, rest, stdout, stderr)
case "fetch-db":
return offsiteFetchDB(fs, rest, stdout, stderr)
case "fetch-worlds":
return offsiteFetchWorlds(fs, rest, stdout, stderr)
case "keygen":
k, err := offsite.NewKey()
if err != nil {
fmt.Fprintf(stderr, "felis offsite keygen: %v\n", err)
return 1
}
fmt.Fprintln(stdout, k)
return 0
case "-h", "--help", "help":
fmt.Fprint(stdout, offsiteUsage)
return 0
}
fmt.Fprintf(stderr, "felis offsite: unknown verb %q\n%s", verb, offsiteUsage)
return 2
}
// offsiteEnv is the resolved [offsite] binding: the bucket and the key.
type offsiteEnv struct {
cfg config.OffsiteConfig
bucket *offsite.S3
key []byte
}
// loadOffsiteEnvFile sets each KEY=VALUE of path that is not already in the
// environment, so a root shell reaches the bucket the same way the unit does.
// A missing file is not an error.
func loadOffsiteEnvFile(path string) error {
if path == "" {
return nil
}
f, err := os.Open(path)
if errors.Is(err, os.ErrNotExist) {
return nil
}
if err != nil {
return err
}
defer f.Close()
sc := bufio.NewScanner(f)
for sc.Scan() {
line := strings.TrimSpace(sc.Text())
if line == "" || strings.HasPrefix(line, "#") {
continue
}
k, v, ok := strings.Cut(line, "=")
if !ok {
continue
}
k = strings.TrimSpace(strings.TrimPrefix(k, "export "))
v = strings.TrimSpace(v)
if len(v) >= 2 && (v[0] == '"' || v[0] == '\'') && v[len(v)-1] == v[0] {
v = v[1 : len(v)-1]
}
if os.Getenv(k) == "" {
os.Setenv(k, v)
}
}
return sc.Err()
}
// resolveOffsite builds the bucket client and parses the key for c.
func resolveOffsite(c config.OffsiteConfig) (*offsiteEnv, error) {
if !c.Enabled() {
return nil, errors.New("no [offsite] bucket is configured (docs/troubleshooting.md §16, \"Keep a copy somewhere else\")")
}
need := func(ref, what string) (string, error) {
v := os.Getenv(ref)
if v == "" {
return "", fmt.Errorf("%s: environment variable %s is empty (set it, or put it in %s)", what, ref, defaultOffsiteEnvFile)
}
return v, nil
}
ak, err := need(c.AccessKeyRef, "access key")
if err != nil {
return nil, err
}
sk, err := need(c.SecretKeyRef, "secret key")
if err != nil {
return nil, err
}
rawKey, err := need(c.KeyRef, "encryption key")
if err != nil {
return nil, err
}
key, err := offsite.ParseKey(rawKey)
if err != nil {
return nil, err
}
b, err := offsite.NewS3(offsite.S3Config{
Endpoint: c.Endpoint, Region: c.Region, Bucket: c.Bucket, Prefix: c.Prefix,
AccessKey: ak, SecretKey: sk,
})
if err != nil {
return nil, err
}
return &offsiteEnv{cfg: c, bucket: b, key: key}, nil
}
// loadOffsite loads felis.toml and the env file and resolves [offsite].
func loadOffsite(cfgPath, envFile string) (*config.Config, *offsiteEnv, error) {
if err := loadOffsiteEnvFile(envFile); err != nil {
return nil, nil, fmt.Errorf("read %s: %w", envFile, err)
}
cfg, err := config.Load(cfgPath)
if err != nil {
return nil, nil, err
}
env, err := resolveOffsite(cfg.Offsite)
if err != nil {
return cfg, nil, err
}
return cfg, env, nil
}
func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml (the host copy, which reaches PostgreSQL on 127.0.0.1)")
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
archiveDir := fs.String("archive-dir", "", "host directory of the world archive volume (default: resolved from the backup PVC through the cluster)")
backupPVC := fs.String("backup-pvc", "felis-backups", `the world archive PVC, in the [k8s] namespace ("" when backups are off)`)
dbDir := fs.String("db-dir", dbbackup.DefaultDir, `database bundle directory ("" copies no bundles)`)
statusFile := fs.String("status-file", offsite.DefaultStatusFile, "where the result of this run is recorded for the watchdog and `status`")
if err := fs.Parse(args); err != nil {
return 2
}
cfg, env, err := loadOffsite(*cfgPath, *envFile)
if err != nil {
fmt.Fprintf(stderr, "felis offsite sync: %v\n", err)
return 1
}
st := offsite.Status{
LastAttempt: time.Now().UTC(), Endpoint: env.cfg.Endpoint, Bucket: env.cfg.Bucket,
Prefix: env.cfg.Prefix, KeyID: offsite.KeyID(env.key),
}
if prev, _ := offsite.ReadStatus(*statusFile); prev != nil {
st.LastSuccess = prev.LastSuccess
}
res, err := runOffsiteSync(cfg, env, *archiveDir, *backupPVC, *dbDir, stderr)
st.Result = res
if err != nil {
st.LastError = err.Error()
} else {
st.LastSuccess = st.LastAttempt
}
if werr := offsite.WriteStatus(*statusFile, st); werr != nil {
fmt.Fprintf(stderr, "felis offsite sync: record status: %v\n", werr)
}
fmt.Fprintf(stdout, "felis offsite sync: worlds copied=%d pending=%d missing=%d expired=%d; bundles copied=%d pruned=%d; bucket holds %d worlds (%s) and %d bundles\n",
res.WorldsUploaded, res.WorldsPending, len(res.WorldsMissing), res.WorldsExpired,
res.DBUploaded, res.DBPruned, res.RemoteWorlds, offsite.HumanBytes(res.RemoteBytes), res.RemoteDB)
for _, m := range res.WorldsMissing {
fmt.Fprintf(stderr, "felis offsite sync: recorded archive not on the volume, nothing to copy: %s\n", m)
}
if err != nil {
fmt.Fprintf(stderr, "felis offsite sync: %v\n", err)
return 1
}
return 0
}
func runOffsiteSync(cfg *config.Config, env *offsiteEnv, archiveDir, backupPVC, dbDir string, log io.Writer) (offsite.Result, error) {
ctx, cancel := context.WithTimeout(context.Background(), 50*time.Minute)
defer cancel()
checkCtx, checkCancel := context.WithTimeout(ctx, 30*time.Second)
err := env.bucket.Check(checkCtx)
checkCancel()
if err != nil {
return offsite.Result{}, err
}
if archiveDir == "" && backupPVC != "" {
dir, err := resolveArchiveDir(ctx, cfg.K8s.Namespace, backupPVC, false, log)
if err != nil {
return offsite.Result{}, err
}
archiveDir = dir
}
drv, err := store.Open(ctx, cfg.Database.URL)
if err != nil {
return offsite.Result{}, fmt.Errorf("open database: %w", err)
}
defer drv.Close()
s := &offsite.Syncer{
Bucket: env.bucket, Catalog: offsite.PGCatalog{DB: drv.DB()}, Key: env.key,
ArchiveDir: archiveDir, DBDir: dbDir, DBKeep: env.cfg.DBKeep, Log: log,
}
return s.Run(ctx)
}
// resolveArchiveDir finds the host directory behind the world archive PVC: a
// local-path volume is a directory on this node. A PVC still waiting for its
// first consumer holds nothing yet: without bind that is "" (no archives),
// with bind it is bound first, for fetch-worlds to write into.
func resolveArchiveDir(ctx context.Context, ns, pvcName string, bind bool, log io.Writer) (string, error) {
if ns == "" {
ns = platform.DefaultMinecraftNamespace
}
cl, err := buildSystemServerClient()
if err != nil {
return "", fmt.Errorf("reach the cluster to find the archive volume (or pass -archive-dir): %w", err)
}
var pvc corev1.PersistentVolumeClaim
if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: pvcName}, &pvc); err != nil {
return "", fmt.Errorf("archive volume %s/%s: %w", ns, pvcName, err)
}
if pvc.Spec.VolumeName == "" {
if !bind {
fmt.Fprintf(log, "felis offsite: archive volume %s/%s is not bound yet; no world has been archived\n", ns, pvcName)
return "", nil
}
if err := bindVolume(ctx, cl, ns, pvcName, log); err != nil {
return "", err
}
if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: pvcName}, &pvc); err != nil {
return "", err
}
}
var pv corev1.PersistentVolume
if err := cl.Get(ctx, types.NamespacedName{Name: pvc.Spec.VolumeName}, &pv); err != nil {
return "", fmt.Errorf("archive volume %s: %w", pvc.Spec.VolumeName, err)
}
var dir string
switch {
case pv.Spec.Local != nil:
dir = pv.Spec.Local.Path
case pv.Spec.HostPath != nil:
dir = pv.Spec.HostPath.Path
default:
return "", fmt.Errorf("archive volume %s is not a directory on a node (local or hostPath); pass -archive-dir with where it is mounted on this host", pv.Name)
}
if fi, err := os.Stat(dir); err != nil || !fi.IsDir() {
return "", fmt.Errorf("archive volume %s is %s on its node, which is not a directory here; run this on the node that holds it, or pass -archive-dir", pv.Name, dir)
}
return dir, nil
}
// bindVolume runs a pod that mounts the PVC and exits, which is what makes a
// WaitForFirstConsumer volume (k3s local-path) get provisioned. The pod uses
// the control plane's own image, which every install already has.
func bindVolume(ctx context.Context, cl client.Client, ns, pvcName string, log io.Writer) error {
var api appsv1.Deployment
if err := cl.Get(ctx, types.NamespacedName{Namespace: platform.DefaultControlNamespace, Name: "felis-api"}, &api); err != nil {
return fmt.Errorf("find the felis image to bind the archive volume with: %w", err)
}
if len(api.Spec.Template.Spec.Containers) == 0 {
return errors.New("felis-api has no container to take the image from")
}
image := api.Spec.Template.Spec.Containers[0].Image
pod := platform.VolumeBinderPod(ns, pvcName, image)
if err := cl.Create(ctx, pod); err != nil {
return fmt.Errorf("start a pod to bind the archive volume: %w", err)
}
fmt.Fprintf(log, "felis offsite: binding the archive volume %s/%s (pod %s)\n", ns, pvcName, pod.Name)
defer func() {
_ = cl.Delete(context.Background(), pod, client.PropagationPolicy(metav1.DeletePropagationBackground))
}()
deadline := time.Now().Add(3 * time.Minute)
for time.Now().Before(deadline) {
var pvc corev1.PersistentVolumeClaim
if err := cl.Get(ctx, types.NamespacedName{Namespace: ns, Name: pvcName}, &pvc); err == nil && pvc.Spec.VolumeName != "" && pvc.Status.Phase == corev1.ClaimBound {
return nil
}
select {
case <-ctx.Done():
return ctx.Err()
case <-time.After(2 * time.Second):
}
}
return fmt.Errorf("the archive volume %s/%s did not bind within 3 minutes; see kubectl -n %s describe pod %s", ns, pvcName, ns, pod.Name)
}
func offsiteStatus(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
statusFile := fs.String("status-file", offsite.DefaultStatusFile, "the record `sync` writes")
if err := fs.Parse(args); err != nil {
return 2
}
cfg, err := config.Load(*cfgPath)
if err != nil {
fmt.Fprintf(stderr, "felis offsite status: %v\n", err)
return 1
}
if !cfg.Offsite.Enabled() {
fmt.Fprintln(stdout, "off-site copy: not configured. World archives and database bundles exist on this machine only.")
fmt.Fprintln(stdout, "See docs/troubleshooting.md §16, \"Keep a copy somewhere else\".")
return 1
}
o := cfg.Offsite
fmt.Fprintf(stdout, "bucket: %s at %s", o.Bucket, o.Endpoint)
if o.Prefix != "" {
fmt.Fprintf(stdout, ", prefix %s", o.Prefix)
}
fmt.Fprintln(stdout)
st, err := offsite.ReadStatus(*statusFile)
if err != nil {
fmt.Fprintf(stderr, "felis offsite status: %v\n", err)
return 1
}
if st == nil {
fmt.Fprintln(stdout, "last sync: never (sudo systemctl start felis-offsite.service)")
return 1
}
now := time.Now()
fmt.Fprintf(stdout, "key id: %s\n", st.KeyID)
fmt.Fprintf(stdout, "last attempt: %s (%s ago)\n", st.LastAttempt.Local().Format(time.DateTime), dbbackup.Age(now.Sub(st.LastAttempt)))
if st.LastSuccess.IsZero() {
fmt.Fprintln(stdout, "last success: never")
} else {
fmt.Fprintf(stdout, "last success: %s (%s ago)\n", st.LastSuccess.Local().Format(time.DateTime), dbbackup.Age(now.Sub(st.LastSuccess)))
}
if st.LastError != "" {
fmt.Fprintf(stdout, "last error: %s\n", st.LastError)
}
r := st.Result
fmt.Fprintf(stdout, "bucket holds: %d world archives (%s), %d database bundles, newest %s\n",
r.RemoteWorlds, offsite.HumanBytes(r.RemoteBytes), r.RemoteDB, orNone(r.NewestDB))
fmt.Fprintf(stdout, "waiting: %d world archives not yet copied\n", r.WorldsPending)
for _, m := range r.WorldsMissing {
fmt.Fprintf(stdout, "missing: %s is recorded but not on the volume\n", m)
}
if st.LastSuccess.IsZero() || now.Sub(st.LastSuccess) > offsite.StaleAfter {
fmt.Fprintf(stdout, "\nThe last successful sync is older than %s: journalctl -u felis-offsite -n 50\n", dbbackup.Age(offsite.StaleAfter))
return 1
}
return 0
}
func orNone(s string) string {
if s == "" {
return "none"
}
return s
}
func offsiteList(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
if err := fs.Parse(args); err != nil {
return 2
}
_, env, err := loadOffsite(*cfgPath, *envFile)
if err != nil {
fmt.Fprintf(stderr, "felis offsite list: %v\n", err)
return 1
}
return printOffsiteList(env, stdout, stderr)
}
func printOffsiteList(env *offsiteEnv, stdout, stderr io.Writer) int {
ctx, cancel := context.WithTimeout(context.Background(), 2*time.Minute)
defer cancel()
bundles, err := offsite.ListDB(ctx, env.bucket)
if err != nil {
fmt.Fprintf(stderr, "felis offsite list: %v\n", err)
return 1
}
worlds, err := env.bucket.List(ctx, "worlds/")
if err != nil {
fmt.Fprintf(stderr, "felis offsite list: %v\n", err)
return 1
}
fmt.Fprintf(stdout, "database bundles (%d, newest first):\n", len(bundles))
for _, b := range bundles {
fmt.Fprintf(stdout, " %s %s\n", b.Key, offsite.HumanBytes(b.Size))
}
var total int64
for _, w := range worlds {
total += w.Size
}
fmt.Fprintf(stdout, "world archives: %d (%s)\n", len(worlds), offsite.HumanBytes(total))
return 0
}
func offsiteFetchDB(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml; on a host with no install yet, give -endpoint and -bucket instead")
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
endpoint := fs.String("endpoint", "", "bucket endpoint, when there is no felis.toml")
bucket := fs.String("bucket", "", "bucket name, when there is no felis.toml")
region := fs.String("region", "", "bucket region, when there is no felis.toml")
prefix := fs.String("prefix", "", "key prefix, when there is no felis.toml")
dir := fs.String("dir", dbbackup.DefaultDir, "directory to write the bundle to")
arg, ok := parseWithArg(fs, args)
if !ok {
return 2
}
if arg == "" {
fmt.Fprint(stderr, offsiteUsage)
return 2
}
if err := loadOffsiteEnvFile(*envFile); err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: read %s: %v\n", *envFile, err)
return 1
}
var oc config.OffsiteConfig
if *bucket != "" {
oc = config.OffsiteConfig{
Endpoint: *endpoint, Bucket: *bucket, Region: *region, Prefix: *prefix,
AccessKeyRef: config.DefaultOffsiteAccessKeyEnv, SecretKeyRef: config.DefaultOffsiteSecretKeyEnv,
KeyRef: config.DefaultOffsiteKeyEnv,
}
} else {
cfg, err := config.Load(*cfgPath)
if err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: %v (on a host with no install yet, pass -endpoint and -bucket)\n", err)
return 1
}
oc = cfg.Offsite
}
env, err := resolveOffsite(oc)
if err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
return 1
}
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Minute)
defer cancel()
name := arg
if name == "latest" {
bundles, err := offsite.ListDB(ctx, env.bucket)
if err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
return 1
}
if len(bundles) == 0 {
fmt.Fprintln(stderr, "felis offsite fetch-db: the bucket holds no database bundle")
return 1
}
name = bundles[0].Key
}
if _, _, ok := dbbackup.ParseBundleName(name); !ok {
fmt.Fprintf(stderr, "felis offsite fetch-db: %q is not a bundle name (felis-db-<stamp>-<label>.tar); see `felis offsite list`\n", name)
return 2
}
if err := os.MkdirAll(*dir, 0o700); err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
return 1
}
dst := filepath.Join(*dir, name)
if err := offsite.FetchObject(ctx, env.bucket, env.key, offsite.DBKey(name), dst, 0o600); err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: %v\n", err)
return 1
}
if _, err := dbbackup.Verify(dst); err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-db: fetched %s but it does not verify: %v\n", dst, err)
return 1
}
fmt.Fprintf(stdout, "felis offsite fetch-db: wrote %s (verified)\n", dst)
return 0
}
func offsiteFetchWorlds(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int {
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml (the host copy)")
envFile := fs.String("env-file", defaultOffsiteEnvFile, "file with the [offsite] secrets, for variables not already set")
archiveDir := fs.String("archive-dir", "", "host directory of the world archive volume (default: resolved from the backup PVC, binding it if needed)")
backupPVC := fs.String("backup-pvc", "felis-backups", "the world archive PVC, in the [k8s] namespace")
if err := fs.Parse(args); err != nil {
return 2
}
cfg, env, err := loadOffsite(*cfgPath, *envFile)
if err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-worlds: %v\n", err)
return 1
}
ctx, cancel := context.WithTimeout(context.Background(), 6*time.Hour)
defer cancel()
dir := *archiveDir
if dir == "" {
if dir, err = resolveArchiveDir(ctx, cfg.K8s.Namespace, *backupPVC, true, stderr); err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-worlds: %v\n", err)
return 1
}
}
drv, err := store.Open(ctx, cfg.Database.URL)
if err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-worlds: open database: %v\n", err)
return 1
}
defer drv.Close()
res, err := offsite.FetchWorlds(ctx, env.bucket, offsite.PGCatalog{DB: drv.DB()}, env.key, dir, stderr)
fmt.Fprintf(stdout, "felis offsite fetch-worlds: %d recorded archives, %d fetched into %s, %d with no copy in the bucket\n",
res.Present, len(res.Fetched), dir, len(res.Missing))
for _, m := range res.Missing {
fmt.Fprintf(stdout, " no off-site copy: %s\n", m)
}
if err != nil {
fmt.Fprintf(stderr, "felis offsite fetch-worlds: %v\n", err)
return 1
}
return 0
}
+96
View File
@@ -0,0 +1,96 @@
package main
import (
"bytes"
"os"
"path/filepath"
"strings"
"testing"
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/offsite"
)
func TestLoadOffsiteEnvFile(t *testing.T) {
path := filepath.Join(t.TempDir(), "offsite.env")
body := `# written by bootstrap
FELIS_OFFSITE_ACCESS_KEY=AKIA123
export FELIS_OFFSITE_SECRET_KEY="se=cret"
FELIS_OFFSITE_KEY='k'
not a line
`
if err := os.WriteFile(path, []byte(body), 0o600); err != nil {
t.Fatal(err)
}
t.Setenv("FELIS_OFFSITE_ACCESS_KEY", "from-the-shell")
t.Setenv("FELIS_OFFSITE_SECRET_KEY", "")
t.Setenv("FELIS_OFFSITE_KEY", "")
if err := loadOffsiteEnvFile(path); err != nil {
t.Fatal(err)
}
for k, want := range map[string]string{
"FELIS_OFFSITE_ACCESS_KEY": "from-the-shell", // the environment wins
"FELIS_OFFSITE_SECRET_KEY": "se=cret",
"FELIS_OFFSITE_KEY": "k",
} {
if got := os.Getenv(k); got != want {
t.Errorf("%s = %q, want %q", k, got, want)
}
}
if err := loadOffsiteEnvFile(filepath.Join(t.TempDir(), "absent")); err != nil {
t.Errorf("a missing env file is not an error: %v", err)
}
}
func TestResolveOffsiteNamesTheMissingVariable(t *testing.T) {
c := config.OffsiteConfig{
Endpoint: "https://s3.example", Bucket: "b",
AccessKeyRef: "T_AK", SecretKeyRef: "T_SK", KeyRef: "T_KEY",
}
t.Setenv("T_AK", "ak")
t.Setenv("T_SK", "sk")
t.Setenv("T_KEY", "")
if _, err := resolveOffsite(c); err == nil || !strings.Contains(err.Error(), "T_KEY") {
t.Fatalf("err = %v, want it to name T_KEY", err)
}
t.Setenv("T_KEY", "not base64 at all")
if _, err := resolveOffsite(c); err == nil {
t.Fatal("a malformed key was accepted")
}
key, _ := offsite.NewKey()
t.Setenv("T_KEY", key)
env, err := resolveOffsite(c)
if err != nil {
t.Fatal(err)
}
if len(env.key) != offsite.KeySize {
t.Fatalf("key is %d bytes", len(env.key))
}
if _, err := resolveOffsite(config.OffsiteConfig{}); err == nil {
t.Fatal("an unconfigured [offsite] resolved")
}
}
func TestOffsiteKeygen(t *testing.T) {
var out, errb bytes.Buffer
if code := cmdOffsite([]string{"keygen"}, &out, &errb); code != 0 {
t.Fatalf("exit %d: %s", code, errb.String())
}
if _, err := offsite.ParseKey(strings.TrimSpace(out.String())); err != nil {
t.Fatalf("keygen printed %q: %v", out.String(), err)
}
}
func TestOffsiteFetchDBRejectsOddNames(t *testing.T) {
key, _ := offsite.NewKey()
t.Setenv("FELIS_OFFSITE_ACCESS_KEY", "ak")
t.Setenv("FELIS_OFFSITE_SECRET_KEY", "sk")
t.Setenv("FELIS_OFFSITE_KEY", key)
var out, errb bytes.Buffer
code := cmdOffsite([]string{"fetch-db", "-env-file", "", "-endpoint", "http://127.0.0.1:1", "-bucket", "b",
"-dir", t.TempDir(), "../../etc/shadow"}, &out, &errb)
if code != 2 || !strings.Contains(errb.String(), "not a bundle name") {
t.Fatalf("exit %d: %s", code, errb.String())
}
}
+3 -2
View File
@@ -131,8 +131,8 @@ func cmdReaper(args []string, stdout, stderr io.Writer) int {
fmt.Fprintf(stderr, "felis reaper: %v\n", err)
return 1
}
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d warned=%d skipped=%d evicted=%d expired=%d\n",
sum.Evaluated, sum.WorldsReaped, sum.Warned, sum.Skipped, sum.EvictedEarly, sum.BackupsExpired)
fmt.Fprintf(stdout, "felis reaper: evaluated=%d reaped=%d awaiting_offsite=%d warned=%d skipped=%d evicted=%d expired=%d\n",
sum.Evaluated, sum.WorldsReaped, sum.AwaitingOffsite, sum.Warned, sum.Skipped, sum.EvictedEarly, sum.BackupsExpired)
return 0
}
@@ -198,6 +198,7 @@ func reaperConfig(cfg *config.Config) (reaper.Config, error) {
}
rc.MaxLocalBytes = b
}
rc.RequireOffsite = cfg.Offsite.Enabled()
return rc, nil
}
+2
View File
@@ -13,6 +13,7 @@ Usage:
Commands:
migrate up Apply embedded database migrations under an advisory lock (snapshots the database first)
db Back up, verify, list and restore the control-plane database (backup|restore|verify|list|check)
offsite Copy world archives and database bundles to an off-site bucket, and fetch them back (sync|status|list|fetch-db|fetch-worlds|keygen)
operator Run the MinecraftServer controller-manager
api Run the felis-api HTTP server
nano Run the Felis-nano hasJoined multiplexer (multi-Yggdrasil, no control plane)
@@ -47,6 +48,7 @@ Run "felis <command> -h" for command-specific flags.
var commands = map[string]func(args []string, stdout, stderr io.Writer) int{
"migrate": cmdMigrate,
"db": cmdDB,
"offsite": cmdOffsite,
"operator": cmdOperator,
"api": cmdAPI,
"nano": cmdNano,
+5
View File
@@ -14,6 +14,7 @@ import (
"felis.lolicon.best/internal/config"
"felis.lolicon.best/internal/mail"
"felis.lolicon.best/internal/offsite"
"felis.lolicon.best/internal/platform"
"felis.lolicon.best/internal/store"
"felis.lolicon.best/internal/watchdog"
@@ -40,6 +41,7 @@ func cmdWatchdog(args []string, stdout, stderr io.Writer) int {
diskPaths := fs.String("disk-paths", "/,/var/lib/rancher/k3s,/var/lib/postgresql,/var/lib/felis", "comma-separated paths whose filesystems must keep free space")
proxyAddr := fs.String("proxy-addr", "", `game proxy address to dial, e.g. 127.0.0.1:25565 ("" skips the check)`)
controlNS := fs.String("control-namespace", platform.DefaultControlNamespace, "namespace of the control plane")
offsiteStatus := fs.String("offsite-status", offsite.DefaultStatusFile, "the record `felis offsite sync` leaves, checked when [offsite] is configured")
dryRun := fs.Bool("dry-run", false, "print every finding and the mail that is due; send nothing and keep the state as it was")
if err := fs.Parse(args); err != nil {
if errors.Is(err, flag.ErrHelp) {
@@ -102,6 +104,9 @@ func cmdWatchdog(args []string, stdout, stderr io.Writer) int {
if *backupDir != "" {
add(watchdog.BackupFinding(*backupDir, now))
}
if cfg.Offsite.Enabled() {
add(watchdog.OffsiteFinding(*offsiteStatus, now))
}
report.Findings = append(report.Findings, watchdog.DiskFindings(splitList(*diskPaths))...)
add(watchdog.MemoryFinding("/proc/meminfo"))
+226 -4
View File
@@ -153,6 +153,19 @@ FELIS_DB_BACKUP_METRICS="${FELIS_DB_BACKUP_METRICS:-/var/lib/node_exporter/textf
# 0 migrates without the pre-migration snapshot, e.g. against an external database newer
# than this host's pg_dump. The upgrade stops if the snapshot fails and this is not set.
FELIS_PRE_MIGRATE_BACKUP="${FELIS_PRE_MIGRATE_BACKUP:-1}"
# The off-site copy (felis offsite, troubleshooting §16). Every hour the host encrypts each
# world archive and the newest database bundles and copies them to an S3-compatible bucket,
# and the reaper deletes an idle world only once its archive is there. Without a bucket
# every backup lives on this one machine, and losing its disk loses them all. Set the
# bucket and its endpoint to turn it on; FELIS_OFFSITE_ACCESS_KEY / FELIS_OFFSITE_SECRET_KEY
# (and optionally a FELIS_OFFSITE_KEY from `felis offsite keygen`) go into
# /etc/felis/offsite.env, mode 0600, with the encryption key generated when there is none.
# A re-run without these keeps the [offsite] felis.host.toml already has.
FELIS_OFFSITE_ENDPOINT="${FELIS_OFFSITE_ENDPOINT:-}"
FELIS_OFFSITE_BUCKET="${FELIS_OFFSITE_BUCKET:-}"
FELIS_OFFSITE_REGION="${FELIS_OFFSITE_REGION:-}"
FELIS_OFFSITE_PREFIX="${FELIS_OFFSITE_PREFIX:-}"
FELIS_OFFSITE_DB_KEEP="${FELIS_OFFSITE_DB_KEEP:-}"
INSTALL_MODE="${FELIS_INSTALL_MODE:-}"
# Loopback by default: hasJoined is an unauthenticated endpoint by protocol (Velocity
# sends no token), so a public bind is a free auth relay — anyone can point their own
@@ -255,6 +268,9 @@ DB_BACKUP_TIMER="/etc/systemd/system/felis-db-backup.timer"
WATCHDOG_SERVICE="/etc/systemd/system/felis-watchdog.service"
WATCHDOG_TIMER="/etc/systemd/system/felis-watchdog.timer"
WATCHDOG_STATE="/var/lib/felis/watchdog/state.json"
OFFSITE_ENV="${STATE_DIR}/offsite.env"
OFFSITE_SERVICE="/etc/systemd/system/felis-offsite.service"
OFFSITE_TIMER="/etc/systemd/system/felis-offsite.timer"
# While this marker holds a future Unix time, felis watchdog mails nothing: an install
# restarts the control plane and the system servers on purpose. cleanup removes it; the
# time in it is the backstop for an installer killed before its EXIT trap runs.
@@ -600,6 +616,34 @@ validate_settings() {
validate_nodeport FELIS_PANEL_NODEPORT "$FELIS_PANEL_NODEPORT"
validate_listen FELIS_NANO_LISTEN "$FELIS_NANO_LISTEN"
validate_cidr FELIS_NANO_PROXY_CIDR "$FELIS_NANO_PROXY_CIDR"
validate_offsite_settings
}
# validate_offsite_settings checks the FELIS_OFFSITE_* inputs before anything is
# installed. They are written into felis.toml as TOML strings and the secrets into a
# single-quoted env file, so a quote, a backslash or a line break is refused outright.
validate_offsite_settings() {
local name value
for name in FELIS_OFFSITE_ENDPOINT FELIS_OFFSITE_BUCKET FELIS_OFFSITE_REGION FELIS_OFFSITE_PREFIX \
FELIS_OFFSITE_ACCESS_KEY FELIS_OFFSITE_SECRET_KEY FELIS_OFFSITE_KEY; do
value="${!name:-}"
case "$value" in
*\"* | *\'* | *\\* | *$'\n'* | *$'\r'*) die "${name} must not contain quotes, backslashes or line breaks" ;;
esac
done
if [ -z "$FELIS_OFFSITE_BUCKET" ]; then
if [ -n "$FELIS_OFFSITE_ENDPOINT$FELIS_OFFSITE_REGION$FELIS_OFFSITE_PREFIX$FELIS_OFFSITE_DB_KEEP" ]; then
die "FELIS_OFFSITE_* is set without FELIS_OFFSITE_BUCKET; name the bucket too"
fi
return 0
fi
[ -n "$FELIS_OFFSITE_ENDPOINT" ] || die "FELIS_OFFSITE_BUCKET needs FELIS_OFFSITE_ENDPOINT (https://host[:port] of the S3-compatible store)"
case "$FELIS_OFFSITE_BUCKET" in
*/* | *' '*) die "FELIS_OFFSITE_BUCKET is a bucket name; put a path inside it in FELIS_OFFSITE_PREFIX" ;;
esac
if [ -n "$FELIS_OFFSITE_DB_KEEP" ] && ! [[ "$FELIS_OFFSITE_DB_KEEP" =~ ^[1-9][0-9]*$ ]]; then
die "FELIS_OFFSITE_DB_KEEP must be a positive number of bundles, got: ${FELIS_OFFSITE_DB_KEEP}"
fi
}
# ---------------------------------------------------------------------------
@@ -2319,8 +2363,52 @@ persisted_registry_block() {
done
}
# persisted_offsite_block echoes the [offsite] section an earlier run (or the operator)
# left behind: after the first install the bucket is configured by editing felis.host.toml
# or by re-running with FELIS_OFFSITE_*, and a plain re-run must keep it. Deleting the
# section and re-running is how the off-site copy is turned off. Same first-readable-file
# rule as persisted_smtp_block.
persisted_offsite_block() {
local f out
for f in "${STATE_DIR}/felis.host.toml" "${STATE_DIR}/felis.pod.toml"; do
[ -r "$f" ] || continue
out="$(awk '
/^[[:space:]]*\[/ {
if (inoff) exit
inoff = ($0 ~ /^[[:space:]]*\[offsite\][[:space:]]*$/)
if (inoff) print
next
}
inoff && /^[[:space:]]*(endpoint|region|bucket|prefix|access_key_ref|secret_key_ref|key_ref|db_keep)[[:space:]]*=/ { print }
' "$f")"
[ -n "$out" ] || continue
printf '%s\n' "$out"
return 0
done
}
# offsite_block is the [offsite] section this run writes: from FELIS_OFFSITE_* when the
# bucket is given, else the one an earlier run left.
offsite_block() {
if [ -z "$FELIS_OFFSITE_BUCKET" ]; then
persisted_offsite_block
return 0
fi
printf '[offsite]\n'
printf 'endpoint = "%s"\n' "$FELIS_OFFSITE_ENDPOINT"
printf 'bucket = "%s"\n' "$FELIS_OFFSITE_BUCKET"
if [ -n "$FELIS_OFFSITE_REGION" ]; then printf 'region = "%s"\n' "$FELIS_OFFSITE_REGION"; fi
if [ -n "$FELIS_OFFSITE_PREFIX" ]; then printf 'prefix = "%s"\n' "$FELIS_OFFSITE_PREFIX"; fi
if [ -n "$FELIS_OFFSITE_DB_KEEP" ]; then printf 'db_keep = %s\n' "$FELIS_OFFSITE_DB_KEEP"; fi
}
# offsite_enabled: the [offsite] section this run writes names a bucket.
offsite_enabled() {
offsite_block | grep -Eq '^[[:space:]]*bucket[[:space:]]*=[[:space:]]*"[^"]+"'
}
write_felis_toml() {
local target="$1" db_host="$2" smtp_block auth_source_blocks registry_block archive_block
local target="$1" db_host="$2" smtp_block auth_source_blocks registry_block archive_block offsite_section
smtp_block="$(persisted_smtp_block)"
if [ -n "$smtp_block" ]; then
log "carrying forward the configured [smtp] relay"
@@ -2337,10 +2425,16 @@ write_felis_toml() {
log "carrying forward the configured [archive] overrides"
archive_block="${archive_block}"$'\n' # keep a blank line before the next section
fi
# The pod copy needs it as much as the host one: the reaper reads [offsite] from the
# config Secret to know it must wait for each archive's off-site copy.
offsite_section="$(offsite_block)"
if [ -n "$offsite_section" ]; then
offsite_section="${offsite_section}"$'\n\n' # keep a blank line before the next section
fi
cat > "$target" <<EOF
# Generated by deploy/bootstrap.sh; rerun the installer to regenerate. Hand edits are
# overwritten, except [smtp], [[auth_source]], and the operator-owned [registry] /
# [archive] overrides, which carry forward.
# overwritten, except [smtp], [[auth_source]], [offsite], and the operator-owned
# [registry] / [archive] overrides, which carry forward.
[server]
listen = "0.0.0.0:8080"
root_domain = "${FELIS_ROOT_DOMAIN}"
@@ -2366,7 +2460,7 @@ ${registry_block}
store = "tarLocal"
local_path = "${FELIS_ARCHIVE_LOCAL_PATH}"
${archive_block}
[auth]
${offsite_section}[auth]
admin_hostname = "op.console.${FELIS_ROOT_DOMAIN}"
panel_hostname = "console.${FELIS_ROOT_DOMAIN}"
@@ -2432,6 +2526,82 @@ quiet_watchdog() {
printf '%s\n' "$(( $(date +%s) + 7200 ))" > "$WATCHDOG_QUIET_FILE"
}
# The hourly off-site copy. The bucket is checked now, in the install, so wrong
# credentials or an unreachable endpoint show up here; the first copy itself runs in the
# background, since a host with many archives can take a long while to upload them.
# With no bucket configured the timer is removed (the operator deleted [offsite]) and the
# install says loudly that every backup is on this machine only.
install_offsite_timer() {
if [ "${OFFSITE_ENABLED:-0}" != 1 ]; then
if [ -e "$OFFSITE_TIMER" ] || [ -e "$OFFSITE_SERVICE" ]; then
systemctl disable --now felis-offsite.timer >/dev/null 2>&1 || true
rm -f "$OFFSITE_TIMER" "$OFFSITE_SERVICE"
systemctl daemon-reload
warn "no [offsite] bucket is configured any more; the off-site copy timer was removed"
fi
return 0
fi
cat > "$OFFSITE_SERVICE" <<EOF
[Unit]
Description=Felis off-site copy (world archives and database bundles, encrypted, to the [offsite] bucket)
After=network-online.target k3s.service postgresql.service felis-db-backup.service
Wants=network-online.target
[Service]
Type=oneshot
EnvironmentFile=${OFFSITE_ENV}
ExecStart=${HOST_BIN} offsite sync -config ${STATE_DIR}/felis.host.toml -env-file ${OFFSITE_ENV} -db-dir ${FELIS_DB_BACKUP_DIR} -backup-pvc "${FELIS_BACKUP_PVC}"
TimeoutStartSec=55min
Nice=10
IOSchedulingClass=idle
PrivateTmp=yes
NoNewPrivileges=yes
EOF
cat > "$OFFSITE_TIMER" <<EOF
[Unit]
Description=Hourly Felis off-site copy
[Timer]
OnCalendar=hourly
RandomizedDelaySec=10min
Persistent=true
[Install]
WantedBy=timers.target
EOF
systemctl daemon-reload
systemctl enable --now felis-offsite.timer
if "$HOST_BIN" offsite list -config "${STATE_DIR}/felis.host.toml" -env-file "$OFFSITE_ENV" >/dev/null; then
systemctl start --no-block felis-offsite.service
ok "off-site copy: hourly to the [offsite] bucket, first copy started (sudo felis offsite status; journalctl -u felis-offsite)"
else
warn "the [offsite] bucket did not answer (error above); nothing is copied off this machine until it does: fix ${OFFSITE_ENV} or [offsite] in ${STATE_DIR}/felis.host.toml, then sudo systemctl start felis-offsite.service"
fi
}
# summary_offsite is the installer's last word on where the backups live.
summary_offsite() {
if [ "${OFFSITE_ENABLED:-0}" != 1 ]; then
warn "NO OFF-SITE COPY: every world archive and database backup is on this machine only."
warn "Losing its disk loses them all. Set FELIS_OFFSITE_BUCKET, FELIS_OFFSITE_ENDPOINT,"
warn "FELIS_OFFSITE_ACCESS_KEY and FELIS_OFFSITE_SECRET_KEY and re-run (docs/troubleshooting.md §16)."
return 0
fi
if [ "${OFFSITE_KEY_NEW:-0}" = 1 ]; then
echo
warn "================================================================================"
warn "The off-site copies are encrypted with this key. Store it NOW somewhere other than"
warn "this machine (a password manager): without it nothing in the bucket can be read."
warn ""
warn " FELIS_OFFSITE_KEY=${FELIS_OFFSITE_KEY}"
warn ""
warn "It is also in ${OFFSITE_ENV}, which is lost with this machine."
warn "================================================================================"
else
log "Off-site copy: the encryption key is in ${OFFSITE_ENV}; keep a copy of it off this machine."
fi
}
# The platform watchdog: every two minutes it checks the control plane, the login gate,
# the fleet, PostgreSQL, the game proxy, the database backups and the host's disks and
# memory, and mails the owners (their verified addresses, over the [smtp] relay) what
@@ -2520,6 +2690,53 @@ EOF
fi
}
# configure_offsite writes OFFSITE_ENV, the secrets behind [offsite]: the bucket's access
# keys and the key every off-site object is sealed with. Values in this run's environment
# replace the file's (rotating the bucket credentials is a re-run); the encryption key is
# generated when neither has one. A different key than the file's is refused: every object
# already in the bucket is sealed with the old one, and swapping it would make them
# unreadable without a word.
configure_offsite() {
OFFSITE_ENABLED=0
OFFSITE_KEY_NEW=0
offsite_enabled || return 0
OFFSITE_ENABLED=1
local env_ak="${FELIS_OFFSITE_ACCESS_KEY:-}" env_sk="${FELIS_OFFSITE_SECRET_KEY:-}" env_key="${FELIS_OFFSITE_KEY:-}"
local file_key=""
FELIS_OFFSITE_ACCESS_KEY="" FELIS_OFFSITE_SECRET_KEY="" FELIS_OFFSITE_KEY=""
if [ -f "$OFFSITE_ENV" ]; then
# shellcheck disable=SC1090
. "$OFFSITE_ENV"
file_key="$FELIS_OFFSITE_KEY"
fi
FELIS_OFFSITE_ACCESS_KEY="${env_ak:-$FELIS_OFFSITE_ACCESS_KEY}"
FELIS_OFFSITE_SECRET_KEY="${env_sk:-$FELIS_OFFSITE_SECRET_KEY}"
if [ -n "$env_key" ] && [ -n "$file_key" ] && [ "$env_key" != "$file_key" ]; then
die "FELIS_OFFSITE_KEY differs from the key in ${OFFSITE_ENV}; the objects already in the bucket are sealed with that one. Unset FELIS_OFFSITE_KEY to keep it (docs/troubleshooting.md §16)"
fi
FELIS_OFFSITE_KEY="${env_key:-$file_key}"
if [ -z "$FELIS_OFFSITE_ACCESS_KEY" ] || [ -z "$FELIS_OFFSITE_SECRET_KEY" ]; then
die "[offsite] names a bucket but there are no credentials for it: set FELIS_OFFSITE_ACCESS_KEY and FELIS_OFFSITE_SECRET_KEY (they are kept in ${OFFSITE_ENV})"
fi
if [ -z "$FELIS_OFFSITE_KEY" ]; then
FELIS_OFFSITE_KEY="$(openssl rand -base64 32)"
OFFSITE_KEY_NEW=1
fi
(
umask 077
cat > "$OFFSITE_ENV" <<EOF
# The off-site copy's secrets (felis offsite, docs/troubleshooting.md §16). Keep a copy of
# FELIS_OFFSITE_KEY somewhere other than this machine: without it the copies in the
# bucket cannot be read, and this file goes with the machine.
FELIS_OFFSITE_ACCESS_KEY='${FELIS_OFFSITE_ACCESS_KEY}'
FELIS_OFFSITE_SECRET_KEY='${FELIS_OFFSITE_SECRET_KEY}'
FELIS_OFFSITE_KEY='${FELIS_OFFSITE_KEY}'
EOF
)
chmod 0600 "$OFFSITE_ENV"
ok "off-site copy: secrets in ${OFFSITE_ENV}"
}
deploy_bundle() {
local had_api=0 had_operator=0
export KUBECONFIG=/etc/rancher/k3s/k3s.yaml
@@ -2767,6 +2984,8 @@ summary() {
log "Mojang account rather than a password."
log "Use 'sudo felis breakGlass' only for emergency local Owner recovery/reset."
echo
summary_offsite
echo
}
# ---------------------------------------------------------------------------
@@ -3099,6 +3318,7 @@ main() {
bootstrap_from_tui || [ -n "${FELIS_SKIP_FETCH:-}" ] || resolve_install_ref
install_cloudflared
load_or_make_secrets
configure_offsite
ensure_panel_tls_cert
install_docker
install_k3s
@@ -3138,6 +3358,8 @@ main() {
install_velocity
# After deploy_bundle: the bundle's MinecraftServer export reads the cluster.
install_db_backup_timer
# After the backup timer: its first bundle is part of the first copy.
install_offsite_timer
# Last: its first run should see the platform as this install leaves it.
install_watchdog_timer
mark_bootstrap_done
+167 -4
View File
@@ -1020,12 +1020,14 @@ rm -f "$kubcalls"
wrblock="$(awk '/^write_felis_toml\(\) \{/,/^}/' "$BS")"
prblock="$(awk '/^persisted_registry_block\(\) \{/,/^}/' "$BS")"
pablock="$(awk '/^persisted_archive_block\(\) \{/,/^}/' "$BS")"
{ [ -n "$wrblock" ] && [ -n "$prblock" ] && [ -n "$pablock" ]; } \
|| { echo "FAIL: write_felis_toml / persisted_{registry,archive}_block not found in $BS"; exit 1; }
poblock="$(awk '/^persisted_offsite_block\(\) \{/,/^}/' "$BS")"
oblock="$(awk '/^offsite_block\(\) \{/,/^}/' "$BS")"
{ [ -n "$wrblock" ] && [ -n "$prblock" ] && [ -n "$pablock" ] && [ -n "$poblock" ] && [ -n "$oblock" ]; } \
|| { echo "FAIL: write_felis_toml / persisted_{registry,archive,offsite}_block / offsite_block not found in $BS"; exit 1; }
# The blocks quote themselves (the awk program uses single quotes), so they are
# sourced from a file instead of being spliced into a single-quoted bash -c.
fnfile="$(mktemp)"
printf '%s\n%s\n%s\n' "$prblock" "$pablock" "$wrblock" > "$fnfile"
printf '%s\n%s\n%s\n%s\n%s\n' "$prblock" "$pablock" "$poblock" "$oblock" "$wrblock" > "$fnfile"
rdir="$(mktemp -d)"
cat > "$rdir/felis.host.toml" <<'TOML'
@@ -1044,6 +1046,10 @@ region = "us-east-1"
store = "tarLocal"
local_path = "/stale/path"
retention = "30d"
[offsite]
endpoint = "https://objects.example"
bucket = "felis-offsite"
TOML
run_write() { # out-file
@@ -1055,7 +1061,7 @@ run_write() { # out-file
FELIS_ROOT_DOMAIN=r.example.com DB_USER=u DB_PASSWORD=p DB_NAME=d MINECRAFT_NS=minecraft \
FELIS_EGRESS_MODE=nodeport FELIS_LIMBO_IMAGE=li FELIS_LOBBY_IMAGE=lo \
REGISTRY_URL=registry.felis.svc:5000 BUILD_NS=felis-build FELIS_ARCHIVE_LOCAL_PATH=/a \
write_felis_toml "$OUT_TOML" 127.0.0.1'
FELIS_OFFSITE_BUCKET= write_felis_toml "$OUT_TOML" 127.0.0.1'
}
run_write "$rdir/out.toml"
@@ -1071,6 +1077,11 @@ expect "the carried subtable keeps its keys" 'endpoint = "https://s3.example"' "
expect "url stays installer-owned" 'url = "registry.felis.svc:5000"' "$out"
expect "a re-run carries the archive retention window" 'retention = "30d"' "$out"
expect "the archive mount stays installer-owned" 'local_path = "/a"' "$out"
expect "a re-run keeps the off-site bucket, set apart from the next section" '[offsite]
endpoint = "https://objects.example"
bucket = "felis-offsite"
[auth]' "$out"
case "$out" in
*stale.invalid* | *stale-ns* | *stale/path*)
echo "FAIL: stale installer-owned values survived the re-run"; fails=$((fails + 1)) ;;
@@ -1220,6 +1231,158 @@ case "$main_block" in
echo "PASS main quiets the watchdog first and installs it last" ;;
*) echo "FAIL main must call quiet_watchdog before any restart and install_watchdog_timer after the backup timer"; fails=$((fails + 1)) ;;
esac
case "$main_block" in
*load_or_make_secrets*configure_offsite*run_migrations*install_db_backup_timer*install_offsite_timer*install_watchdog_timer*)
echo "PASS main configures the off-site copy before the toml is written and starts it after the first backup" ;;
*) echo "FAIL main must call configure_offsite before run_migrations and install_offsite_timer between the backup and watchdog timers"; fails=$((fails + 1)) ;;
esac
# --- the off-site copy: [offsite], its secrets file, the hourly timer ---------------------
# Without it every backup is on one disk. A re-run must keep the bucket, the encryption key
# must never change under objects sealed with the old one, and an install without a bucket
# must say so loudly.
ofile="$(mktemp)"
for fn in validate_offsite_settings persisted_offsite_block offsite_block offsite_enabled configure_offsite install_offsite_timer summary_offsite; do
blk="$(awk "/^${fn}\\(\\) \\{/,/^}/" "$BS")"
[ -n "$blk" ] || { echo "FAIL: no ${fn} found in $BS"; exit 1; }
[ "$(printf '%s\n' "$blk" | wc -l)" -lt 90 ] \
|| { echo "FAIL: the extracted block is not ${fn} -- did its closing brace move?"; exit 1; }
printf '%s\n' "$blk" >> "$ofile"
done
odir="$(mktemp -d)"
run_offsite() { # script; runs with the off-site functions sourced
STATE_DIR="$odir" OFFSITE_ENV="$odir/offsite.env" OFFSITE_SERVICE="$odir/felis-offsite.service" \
OFFSITE_TIMER="$odir/felis-offsite.timer" FNFILE="$ofile" HOST_BIN=fakefelis \
FELIS_DB_BACKUP_DIR=/var/lib/felis/db-backups FELIS_BACKUP_PVC=felis-backups bash -c '
set -Eeuo pipefail
die() { printf "DIE: %s\n" "$*"; exit 1; }
log() { printf "LOG: %s\n" "$*"; }; ok() { printf "OK: %s\n" "$*"; }; warn() { printf "WARN: %s\n" "$*"; }
systemctl() { printf "SYSTEMCTL: %s\n" "$*" >&2; }
fakefelis() { printf "RUN: %s\n" "$*" >&2; [ -z "${CHECK_FAILS:-}" ] || { echo "bucket: access denied" >&2; return 1; }; }
FELIS_OFFSITE_ENDPOINT="${FELIS_OFFSITE_ENDPOINT:-}" FELIS_OFFSITE_BUCKET="${FELIS_OFFSITE_BUCKET:-}"
FELIS_OFFSITE_REGION="${FELIS_OFFSITE_REGION:-}" FELIS_OFFSITE_PREFIX="${FELIS_OFFSITE_PREFIX:-}"
FELIS_OFFSITE_DB_KEEP="${FELIS_OFFSITE_DB_KEEP:-}"
. "$FNFILE"
'"$1" 2>&1
}
out="$(FELIS_OFFSITE_BUCKET=b run_offsite validate_offsite_settings)"
expect "a bucket without an endpoint is refused" "DIE: FELIS_OFFSITE_BUCKET needs FELIS_OFFSITE_ENDPOINT" "$out"
out="$(FELIS_OFFSITE_ENDPOINT=https://s3.example FELIS_OFFSITE_BUCKET='b"x' run_offsite validate_offsite_settings)"
expect "a quote cannot reach the generated toml" "DIE: FELIS_OFFSITE_BUCKET must not contain quotes" "$out"
out="$(FELIS_OFFSITE_ENDPOINT=https://s3.example FELIS_OFFSITE_BUCKET=b FELIS_OFFSITE_SECRET_KEY="se'cret" run_offsite validate_offsite_settings)"
expect "a quote cannot reach the secrets file" "DIE: FELIS_OFFSITE_SECRET_KEY must not contain quotes" "$out"
out="$(FELIS_OFFSITE_ENDPOINT=https://s3.example FELIS_OFFSITE_BUCKET=b/sub run_offsite validate_offsite_settings)"
expect "a path in the bucket name is refused" "DIE: FELIS_OFFSITE_BUCKET is a bucket name" "$out"
out="$(FELIS_OFFSITE_ENDPOINT=https://s3.example run_offsite validate_offsite_settings)"
expect "an endpoint without a bucket is refused" "DIE: FELIS_OFFSITE_* is set without FELIS_OFFSITE_BUCKET" "$out"
out="$(FELIS_OFFSITE_ENDPOINT=https://s3.example FELIS_OFFSITE_BUCKET=b FELIS_OFFSITE_DB_KEEP=0 run_offsite validate_offsite_settings)"
expect "db_keep must be positive" "DIE: FELIS_OFFSITE_DB_KEEP must be a positive number" "$out"
out="$(run_offsite 'validate_offsite_settings; echo fine')"
expect "no off-site settings is a valid install" "fine" "$out"
out="$(FELIS_OFFSITE_ENDPOINT=https://s3.example FELIS_OFFSITE_BUCKET=felis FELIS_OFFSITE_PREFIX=host1 run_offsite offsite_block)"
expect "the environment writes [offsite]" '[offsite]
endpoint = "https://s3.example"
bucket = "felis"
prefix = "host1"' "$out"
cat > "$odir/felis.host.toml" <<'TOML'
[archive]
store = "tarLocal"
[offsite]
endpoint = "https://s3.example"
bucket = "kept"
key_ref = "MY_KEY"
[auth]
admin_hostname = "x"
TOML
out="$(run_offsite offsite_block)"
expect "a re-run keeps the configured bucket" 'bucket = "kept"' "$out"
expect "a re-run keeps the key reference" 'key_ref = "MY_KEY"' "$out"
case "$out" in
*admin_hostname*) echo "FAIL the carried [offsite] swallowed the next section"; fails=$((fails + 1)) ;;
*) echo "PASS the carried [offsite] stops at the next section" ;;
esac
# First configured install: the key is generated, the file is private, the key is shown once.
rm -f "$odir/felis.host.toml"
out="$(FELIS_OFFSITE_ENDPOINT=https://s3.example FELIS_OFFSITE_BUCKET=felis \
FELIS_OFFSITE_ACCESS_KEY=AK FELIS_OFFSITE_SECRET_KEY=SK run_offsite 'configure_offsite; summary_offsite')"
envf="$(cat "$odir/offsite.env" 2>/dev/null)"
expect "the access key is kept for the unit" "FELIS_OFFSITE_ACCESS_KEY='AK'" "$envf"
expect "an encryption key is generated" "FELIS_OFFSITE_KEY='" "$envf"
key="$(sed -n "s/^FELIS_OFFSITE_KEY='\(.*\)'$/\1/p" "$odir/offsite.env")"
if [ "$(printf '%s' "$key" | base64 -d 2>/dev/null | wc -c | tr -d ' ')" = 32 ]; then
echo "PASS the generated key is 32 random bytes in base64"
else
echo "FAIL the generated key '$key' is not 32 bytes of base64"; fails=$((fails + 1))
fi
if [ "$(stat -c %a "$odir/offsite.env" 2>/dev/null || stat -f %Lp "$odir/offsite.env")" = 600 ]; then
echo "PASS the off-site secrets file is private"
else
echo "FAIL offsite.env must be 0600"; fails=$((fails + 1))
fi
expect "a new key is shown once, with the warning to keep it elsewhere" "WARN: FELIS_OFFSITE_KEY=${key}" "$out"
# A re-run with new credentials keeps the key; a different key is refused.
out="$(FELIS_OFFSITE_ENDPOINT=https://s3.example FELIS_OFFSITE_BUCKET=felis \
FELIS_OFFSITE_ACCESS_KEY=AK2 run_offsite 'configure_offsite; summary_offsite')"
expect "rotated credentials replace the old ones" "FELIS_OFFSITE_ACCESS_KEY='AK2'" "$(cat "$odir/offsite.env")"
expect "the secret key the re-run did not give is kept" "FELIS_OFFSITE_SECRET_KEY='SK'" "$(cat "$odir/offsite.env")"
expect "the key survives a re-run" "FELIS_OFFSITE_KEY='${key}'" "$(cat "$odir/offsite.env")"
case "$out" in
*"FELIS_OFFSITE_KEY="*) echo "FAIL a re-run printed the key again"; fails=$((fails + 1)) ;;
*) echo "PASS a re-run does not print the key again" ;;
esac
out="$(FELIS_OFFSITE_ENDPOINT=https://s3.example FELIS_OFFSITE_BUCKET=felis \
FELIS_OFFSITE_KEY=c29tZXRoaW5nIGVsc2UgZW50aXJlbHkgZGlmZmVyZW50IQ== run_offsite configure_offsite)"
expect "a different key is refused" "DIE: FELIS_OFFSITE_KEY differs from the key in" "$out"
expect "the refusal leaves the key alone" "FELIS_OFFSITE_KEY='${key}'" "$(cat "$odir/offsite.env")"
rm -f "$odir/offsite.env"
out="$(FELIS_OFFSITE_ENDPOINT=https://s3.example FELIS_OFFSITE_BUCKET=felis run_offsite configure_offsite)"
expect "a bucket without credentials is refused" "DIE: [offsite] names a bucket but there are no credentials" "$out"
out="$(run_offsite 'configure_offsite; install_offsite_timer; summary_offsite; echo "enabled=$OFFSITE_ENABLED"')"
expect "no bucket leaves the off-site copy off" "enabled=0" "$out"
expect "no bucket is a loud warning" "WARN: NO OFF-SITE COPY" "$out"
out="$(OFFSITE_ENABLED=1 run_offsite install_offsite_timer)"
unit="$(cat "$odir/felis-offsite.service")"
timer="$(cat "$odir/felis-offsite.timer")"
expect "the unit loads the secrets" "EnvironmentFile=$odir/offsite.env" "$unit"
expect "the unit syncs from the host config" \
"ExecStart=fakefelis offsite sync -config $odir/felis.host.toml -env-file $odir/offsite.env -db-dir /var/lib/felis/db-backups -backup-pvc \"felis-backups\"" "$unit"
expect "a run ends before the next hour's" "TimeoutStartSec=55min" "$unit"
expect "the copy runs hourly" "OnCalendar=hourly" "$timer"
expect "a missed run catches up at boot" "Persistent=true" "$timer"
expect "the timer is enabled" "SYSTEMCTL: enable --now felis-offsite.timer" "$out"
expect "the bucket is checked during the install" "RUN: offsite list -config $odir/felis.host.toml -env-file $odir/offsite.env" "$out"
expect "the first copy runs in the background" "SYSTEMCTL: start --no-block felis-offsite.service" "$out"
out="$(OFFSITE_ENABLED=1 CHECK_FAILS=1 run_offsite install_offsite_timer)"
expect "an unreachable bucket shows why" "bucket: access denied" "$out"
expect "an unreachable bucket is a loud warning" "WARN: the [offsite] bucket did not answer" "$out"
case "$out" in
*"start --no-block"*) echo "FAIL a failed check still started the copy"; fails=$((fails + 1)) ;;
*) echo "PASS a failed check does not start the copy" ;;
esac
out="$(OFFSITE_ENABLED=0 run_offsite 'systemctl() { printf "SYSTEMCTL: %s\n" "$*" >> "$STATE_DIR/systemctl.log"; }; install_offsite_timer')"
expect "removing [offsite] disables the timer" "SYSTEMCTL: disable --now felis-offsite.timer" "$(cat "$odir/systemctl.log")"
expect "removing [offsite] says so" "WARN: no [offsite] bucket is configured any more" "$out"
if [ -e "$odir/felis-offsite.service" ] || [ -e "$odir/felis-offsite.timer" ]; then
echo "FAIL removing [offsite] left the units behind"; fails=$((fails + 1))
else
echo "PASS removing [offsite] removes the units"
fi
rm -rf "$odir" "$ofile"
# ---------------------------------------------------------------------------------------
if [ "$fails" -eq 0 ]; then
+105 -17
View File
@@ -643,13 +643,23 @@ The reap sequence (all [GO-TESTED] hermetically) preserves the world unless a
**preserved** (not deleted).
2. `Archiver.Archive` fails ⇒ world **preserved**, PVC untouched.
3. `InsertBackup` (DB) fails ⇒ the orphan archive is deleted, PVC **untouched**.
4. **Only then** `DeletePVC` → `ReleaseWorld` → `Stop` (cosmetic) → audit →
4. With an `[offsite]` bucket configured (§16), the archive must also be in the
bucket: until `felis offsite sync` has copied it and set `offsite_at` on its
`world_backups` row, the run logs `world archived, kept until the archive's
off-site copy is confirmed`, counts it in `awaiting_offsite=` and leaves the
PVC alone. The next daily run after the copy reuses the same archive and
deletes. [GO-TESTED: `TestReapWaitsForOffsiteCopy`]
5. **Only then** `DeletePVC` → `ReleaseWorld` → `Stop` (cosmetic) → audit →
`felis_reaper_worlds_deleted_total++`.
So a missing backup never results in a deleted world. [GO-TESTED:
So a missing backup never results in a deleted world, and with a bucket
configured neither does a backup that exists on this disk only. [GO-TESTED:
`TestReapArchiveFailurePreservesWorld`,
`TestReapInsertBackupFailurePreservesWorld`, `TestReapIdleWorldFullSequence`.]
`awaiting_offsite` that stays above zero for more than a day means the copy is
failing: `sudo felis offsite status` (§16).
### Exemptions (world never reaped)
- `spec.reaperExempt=true` → skipped entirely (system servers). [GO-TESTED
@@ -1297,12 +1307,26 @@ older installer version to bring the host binary back in line.
### Rebuild on a new host (the old one is gone)
This needs a bundle that was copied off the old host (next section).
This needs the off-site copy (next section) or a bundle you copied off the old
host yourself, plus the off-site encryption key if the copy is in the bucket.
1. Check the copy: `sha256sum -c felis-db-....tar.sha256`.
1. Get the newest bundle. From the bucket, with any `felis` binary of the same
or a newer release (the `felis-linux-<arch>` release asset runs on its own;
this works from any machine that can reach the bucket):
```
export FELIS_OFFSITE_ACCESS_KEY=... FELIS_OFFSITE_SECRET_KEY=... FELIS_OFFSITE_KEY=...
sudo -E felis offsite fetch-db -endpoint https://s3.example.com -bucket felis-backups \
[-region ...] [-prefix ...] latest
```
It writes the bundle into `/var/lib/felis/db-backups` (`-dir` to change),
checks it (`felis db verify`) and names it. A wrong key fails with
`object does not decrypt with this key` and writes nothing. For a copy you
made yourself, check it with `sha256sum -c felis-db-....tar.sha256`.
2. Put the old host's state in place **before** installing, so the installer
reuses the same DB password, session secret and forwarding secret (the
Velocity proxy and existing sessions keep working):
reuses the same DB password, session secret, forwarding secret and the
`[offsite]` bucket with its credentials and key (`offsite.env`):
```
sudo install -d -m 0700 /etc/felis
@@ -1311,34 +1335,97 @@ This needs a bundle that was copied off the old host (next section).
3. Run the installer as for a first install. `bootstrap.done` is not in the
bundle, so it takes the fresh-install path, creates the empty database with
the restored password and migrates it.
the restored password and migrates it. It finds `[offsite]` in the restored
`felis.host.toml` and turns the hourly copy back on.
4. Restore the database and bring the servers back:
```
kubectl -n felis scale deployment felis-api felis-operator --replicas=0
sudo felis db restore -yes -no-safety-backup /path/to/felis-db-....tar
sudo felis db restore -yes -no-safety-backup /var/lib/felis/db-backups/felis-db-....tar
sudo felis migrate up -config /etc/felis/felis.host.toml
kubectl -n felis scale deployment felis-api felis-operator --replicas=1
tar -xOf felis-db-....tar k8s/minecraftservers.json | kubectl apply -f -
```
5. Worlds come back from their own archives (§10), which are a separate volume
and need their own off-host copy. Custom images built on the old host are
rebuilt from their submissions (§8), or re-pushed.
5. Bring the world archives back into the archive volume:
```
sudo felis offsite fetch-worlds
```
It fetches every archive the restored `world_backups` index lists as present
and the volume lacks, provisioning the `felis-backups` volume first if
nothing has used it yet (a short-lived `felis-bind-felis-backups-*` pod). It
lists any it could not find in the bucket. Restore a world from its archive
as usual (§10, §13). Custom images built on the old host are rebuilt from
their submissions (§8), or re-pushed.
### Keep a copy somewhere else
A bundle on the same disk as the database protects against mistakes and bad
upgrades, not against losing the disk. Copy the directory off the host on a
schedule of your own, for example from another machine:
upgrades, and a world archive on the same disk as the worlds protects against
a deleted server. Neither survives losing the disk. The installer's off-site
copy sends both to an S3-compatible bucket (AWS S3, Cloudflare R2, Backblaze
B2, MinIO, ...), encrypted on this host:
```
rsync -a --delete root@felis-host:/var/lib/felis/db-backups/ /backups/felis-db/
FELIS_OFFSITE_ENDPOINT=https://<account>.r2.cloudflarestorage.com \
FELIS_OFFSITE_BUCKET=felis-backups \
FELIS_OFFSITE_ACCESS_KEY=... FELIS_OFFSITE_SECRET_KEY=... \
bash deploy/bootstrap.sh # or the curl | sudo bash one-liner
```
or with `rclone copy /var/lib/felis/db-backups remote:felis-db` from a systemd
timer on the host. Copy the `.sha256` sidecars too; `sha256sum -c` on the far
side proves the copy.
Optional: `FELIS_OFFSITE_REGION`, `FELIS_OFFSITE_PREFIX` (a key prefix, so one
bucket can hold several installs) and `FELIS_OFFSITE_DB_KEEP` (default 30).
The installer writes `[offsite]` into `felis.toml`, keeps the credentials and
a generated encryption key in `/etc/felis/offsite.env` (mode 0600), and
**prints the key once**. Store it in a password manager: the bucket holds only
sealed objects, and without the key they cannot be read. A later re-run keeps
the key; it refuses a `FELIS_OFFSITE_KEY` that differs from the one in
`offsite.env`, since every object already in the bucket is sealed with it.
Without a bucket the installer ends with `NO OFF-SITE COPY`.
What runs:
- **`felis-offsite.timer`** runs `felis offsite sync` hourly (plus up to
10 min random delay, `Persistent=true`). Each run copies every world archive
whose row has no `offsite_at` yet and records it, copies the newest
`db_keep` database bundles the bucket lacks and prunes older ones there, and
deletes a world archive from the bucket once its row has been deleted and
its retention (`expires_at`) has passed. An object already in the bucket at
the right size is recorded without being sent again, so a run cut short
resumes. [GO-TESTED: `internal/offsite`]
- Objects are `worlds/<archive>.fenc` and `db/<bundle>.fenc`: AES-256-GCM in
64 KiB segments, so truncation, reordering and a wrong key are all refused
on the way back.
- The reaper deletes an idle world only after its archive is in the bucket
(§10).
- The watchdog mails the owners when no sync has completed for 12 hours
(`the off-site copy last completed ... ago`).
Checking it:
```
sudo felis offsite status # last run, errors, what the bucket holds, what waits
sudo felis offsite list # the bundles in the bucket, newest first
sudo journalctl -u felis-offsite -n 50 --no-pager
sudo systemctl start felis-offsite.service # run one now
```
`status` exits 1 when no sync has completed in 12 hours. `missing:` lines are
archives the database records but the volume no longer has (an archive
removed by hand); there is nothing left to copy for those.
To change the bucket, edit `[offsite]` in `/etc/felis/felis.host.toml` (and
`offsite.env` for new credentials) and re-run the installer; to turn the copy
off, delete the section and re-run. Keeping the same key across buckets keeps
old copies readable.
Without a bucket, copy the backup directory off the host on a schedule of your
own (`rsync -a root@felis-host:/var/lib/felis/db-backups/ /backups/felis-db/`,
with the `.sha256` sidecars; `sha256sum -c` on the far side proves the copy).
That covers the database only; the world archives are under the
`felis-backups` volume's directory in `/var/lib/rancher/k3s/storage/`.
### `FELIS_PRE_MIGRATE_BACKUP=0`
@@ -1472,6 +1559,7 @@ for 10 seconds (the Free plan's limits).
| `pre-migration backup failed, nothing applied` during an upgrade | §16 |
| Undo a mistaken change / restore the control-plane database | §16 |
| Host lost: rebuild from a database bundle | §16 |
| `the off-site copy last completed ... ago` / `NO OFF-SITE COPY` / reaper `awaiting_offsite` stays above 0 | §16, §10 |
| Sign-in 429 `rate_limited` for everyone at once | §17 |
| 429 `mail_rate_limited` / `FelisMailBudgetExhausted` | §17 |
| Right code refused; `otp_account_locked` / `FelisOTPAccountLocked` | §17 |
+88
View File
@@ -23,6 +23,7 @@ type Config struct {
K8s K8sConfig `toml:"k8s"`
Registry RegistryConfig `toml:"registry"`
Archive ArchiveConfig `toml:"archive"`
Offsite OffsiteConfig `toml:"offsite"`
SMTP SMTPConfig `toml:"smtp"`
// AuthSources is the [[auth_source]] array-of-tables: the third-party Yggdrasil
// roots the Felis-nano hasJoined multiplexer federates over, in priority order
@@ -236,6 +237,48 @@ type ArchiveS3Config struct {
SecretKeyRef string `toml:"secret_key_ref"`
}
// OffsiteConfig is the [offsite] table: the S3-compatible bucket, away from
// this machine, that `felis offsite sync` (felis-offsite.timer on the host)
// copies every world archive and the newest database bundles into, encrypted
// (internal/offsite). An empty bucket means no off-site copy: the archives and
// the database then share the node's disk with the worlds.
//
// When it is set the reaper deletes an idle world only once the archive it made
// has its off-site copy, so the reaper pod reads this table too. The secrets
// follow the credential rule of [archive.s3]: the *_ref fields NAME the
// environment variables holding them (bootstrap writes /etc/felis/offsite.env),
// and they are never written into felis.toml.
type OffsiteConfig struct {
// Endpoint is https://host[:port]; http:// only for a store on a trusted
// network. A bare host means TLS.
Endpoint string `toml:"endpoint"`
Region string `toml:"region"`
Bucket string `toml:"bucket"`
// Prefix places every object under this key prefix, so one bucket can hold
// several installs.
Prefix string `toml:"prefix"`
AccessKeyRef string `toml:"access_key_ref"`
SecretKeyRef string `toml:"secret_key_ref"`
// KeyRef names the variable holding the encryption key (`felis offsite
// keygen`). The copies are unreadable without it, so it must also be kept
// somewhere other than this machine.
KeyRef string `toml:"key_ref"`
// DBKeep is how many of the newest database bundles the bucket keeps.
DBKeep int `toml:"db_keep"`
}
// Enabled reports whether an off-site bucket is configured.
func (o OffsiteConfig) Enabled() bool { return o.Bucket != "" }
// Default environment variable names for the [offsite] secrets, and the bundle
// count kept off-site.
const (
DefaultOffsiteAccessKeyEnv = "FELIS_OFFSITE_ACCESS_KEY"
DefaultOffsiteSecretKeyEnv = "FELIS_OFFSITE_SECRET_KEY"
DefaultOffsiteKeyEnv = "FELIS_OFFSITE_KEY"
DefaultOffsiteDBKeep = 30
)
// archive store backends recognized by §19.
var archiveStores = map[string]struct{}{
"tarLocal": {},
@@ -347,6 +390,20 @@ func (c *Config) applyDefaults() {
if c.SMTP.Host != "" && c.SMTP.Port == 0 {
c.SMTP.Port = defaultSMTPPort
}
if c.Offsite.Enabled() {
if c.Offsite.AccessKeyRef == "" {
c.Offsite.AccessKeyRef = DefaultOffsiteAccessKeyEnv
}
if c.Offsite.SecretKeyRef == "" {
c.Offsite.SecretKeyRef = DefaultOffsiteSecretKeyEnv
}
if c.Offsite.KeyRef == "" {
c.Offsite.KeyRef = DefaultOffsiteKeyEnv
}
if c.Offsite.DBKeep == 0 {
c.Offsite.DBKeep = DefaultOffsiteDBKeep
}
}
}
// Validate enforces the mandatory fields (spec §24: database.url is 强制) and
@@ -398,12 +455,43 @@ func (c *Config) Validate() error {
if c.SMTP.MaxPerHour < 0 {
return fmt.Errorf("config: [smtp] max_per_hour %d must be positive (0 means the default %d)", c.SMTP.MaxPerHour, DefaultMailPerHour)
}
if err := c.Offsite.validate(); err != nil {
return err
}
if h := c.Auth.ClientIPHeader; strings.ContainsAny(h, " :\t\r\n") {
return fmt.Errorf("config: [auth] client_ip_header %q must be a bare header name such as CF-Connecting-IP or X-Forwarded-For", h)
}
return c.validateAuthSources()
}
// validate checks a configured [offsite] table. Nothing is required of an
// unconfigured one; a half-filled one (an endpoint and no bucket) is refused,
// since it reads as configured while nothing is copied anywhere.
func (o OffsiteConfig) validate() error {
if !o.Enabled() {
if o.Endpoint != "" || o.Prefix != "" {
return fmt.Errorf("config: [offsite] names an endpoint or prefix but no bucket; set bucket, or remove the table")
}
return nil
}
if strings.TrimSpace(o.Endpoint) == "" {
return fmt.Errorf("config: [offsite] endpoint is required with bucket %q (e.g. https://s3.eu-central-1.amazonaws.com)", o.Bucket)
}
if u, err := url.Parse(o.Endpoint); strings.Contains(o.Endpoint, "://") && (err != nil || (u.Scheme != "http" && u.Scheme != "https") || u.Host == "" || strings.Trim(u.Path, "/") != "") {
return fmt.Errorf("config: [offsite] endpoint %q must be http(s)://host[:port] with no path; the bucket goes in bucket", o.Endpoint)
}
if strings.ContainsAny(o.Bucket, "/ ") {
return fmt.Errorf("config: [offsite] bucket %q must be a bare bucket name; put a key prefix in prefix", o.Bucket)
}
if strings.Contains(o.Prefix, "..") {
return fmt.Errorf("config: [offsite] prefix %q must not contain ..", o.Prefix)
}
if o.DBKeep < 1 {
return fmt.Errorf("config: [offsite] db_keep %d must be at least 1", o.DBKeep)
}
return nil
}
// authSourcePrefixRe is the shape of a prefix. It is prepended to a real Minecraft
// username (LS_steve), so it is confined to the username charset and kept short enough to
// leave a legible name behind after truncation.
+47
View File
@@ -609,3 +609,50 @@ url = "postgres://felis@db/felis"
t.Fatal("negative max_per_hour accepted")
}
}
// TestLoadOffsite: a bucket turns the table on and fills in the secret refs and
// bundle count; a half-filled or malformed table fails at load.
func TestLoadOffsite(t *testing.T) {
const head = `
[server]
root_domain = "mc.example.net"
[database]
url = "postgres://felis@db/felis"
`
cfg, err := config.Load(writeTOML(t, head))
if err != nil {
t.Fatalf("Load without [offsite]: %v", err)
}
if cfg.Offsite.Enabled() || cfg.Offsite.KeyRef != "" {
t.Fatalf("absent [offsite] must stay off and zero, got %+v", cfg.Offsite)
}
cfg, err = config.Load(writeTOML(t, head+`
[offsite]
endpoint = "https://s3.example.net"
bucket = "felis-copies"
prefix = "site-a"
`))
if err != nil {
t.Fatalf("Load: %v", err)
}
o := cfg.Offsite
if !o.Enabled() || o.AccessKeyRef != "FELIS_OFFSITE_ACCESS_KEY" || o.SecretKeyRef != "FELIS_OFFSITE_SECRET_KEY" ||
o.KeyRef != "FELIS_OFFSITE_KEY" || o.DBKeep != 30 {
t.Fatalf("defaults not applied: %+v", o)
}
for name, body := range map[string]string{
"no bucket": "endpoint = \"https://s3.example.net\"",
"no endpoint": "bucket = \"b\"",
"path in endpoint": "endpoint = \"https://s3.example.net/b\"\nbucket = \"b\"",
"ftp endpoint": "endpoint = \"ftp://s3.example.net\"\nbucket = \"b\"",
"bucket with key": "endpoint = \"s3.example.net\"\nbucket = \"b/x\"",
"negative keep": "endpoint = \"s3.example.net\"\nbucket = \"b\"\ndb_keep = -1",
"secret in toml": "endpoint = \"s3.example.net\"\nbucket = \"b\"\nsecret_key = \"x\"",
} {
if _, err := config.Load(writeTOML(t, head+"[offsite]\n"+body+"\n")); err == nil {
t.Errorf("%s: loaded", name)
}
}
}
+3 -3
View File
@@ -266,8 +266,8 @@ func BundleName(t time.Time, label string) string {
return bundlePrefix + t.UTC().Format(stampLayout) + "-" + label + bundleExt
}
// parseBundleName reverses BundleName.
func parseBundleName(name string) (time.Time, string, bool) {
// ParseBundleName reverses BundleName: the stamp and label of a bundle file name.
func ParseBundleName(name string) (time.Time, string, bool) {
if !strings.HasPrefix(name, bundlePrefix) || !strings.HasSuffix(name, bundleExt) {
return time.Time{}, "", false
}
@@ -297,7 +297,7 @@ func List(dir string) ([]Bundle, error) {
if !e.Type().IsRegular() {
continue
}
t, label, ok := parseBundleName(e.Name())
t, label, ok := ParseBundleName(e.Name())
if !ok {
continue
}
+2 -2
View File
@@ -474,8 +474,8 @@ func TestParseBundleName(t *testing.T) {
"felis-db-20260924T033000Z-daily.tar.sha256": false,
"other.tar": false,
} {
if _, _, got := parseBundleName(name); got != ok {
t.Errorf("parseBundleName(%q) ok = %v, want %v", name, got, ok)
if _, _, got := ParseBundleName(name); got != ok {
t.Errorf("ParseBundleName(%q) ok = %v, want %v", name, got, ok)
}
}
}
+160
View File
@@ -0,0 +1,160 @@
package offsite
import (
"context"
"errors"
"fmt"
"io"
"net/http"
"strings"
"time"
"github.com/minio/minio-go/v7"
"github.com/minio/minio-go/v7/pkg/credentials"
)
// Object is one object in the bucket, its key relative to the configured
// prefix.
type Object struct {
Key string
Size int64
Modified time.Time
}
// Bucket is the object store the sync writes to. S3 is the production one; the
// tests use an in-memory map.
type Bucket interface {
// Put stores exactly size bytes read from r under key.
Put(ctx context.Context, key string, r io.Reader, size int64) error
// Get opens key. A missing key is ErrNotFound.
Get(ctx context.Context, key string) (io.ReadCloser, error)
// List returns every object whose key starts with prefix.
List(ctx context.Context, prefix string) ([]Object, error)
Remove(ctx context.Context, key string) error
}
// ErrNotFound is a key the bucket does not hold.
var ErrNotFound = errors.New("offsite: no such object")
// S3Config locates an S3-compatible bucket. Endpoint takes an http:// or
// https:// scheme; a bare host means TLS.
type S3Config struct {
Endpoint string
Region string
Bucket string
Prefix string
AccessKey string
SecretKey string
}
// S3 is a Bucket on an S3-compatible store (AWS S3, Backblaze B2, Cloudflare
// R2, Wasabi, MinIO...). Every key is placed under Prefix.
type S3 struct {
client *minio.Client
bucket string
prefix string
}
// partSize bounds what one multipart upload holds in memory.
const partSize = 16 << 20
// NewS3 builds the client. It does not touch the network; Check does.
func NewS3(c S3Config) (*S3, error) {
host, secure, err := splitEndpoint(c.Endpoint)
if err != nil {
return nil, err
}
if c.Bucket == "" {
return nil, errors.New("offsite: no bucket configured")
}
if c.AccessKey == "" || c.SecretKey == "" {
return nil, errors.New("offsite: the access key or the secret key is empty")
}
cl, err := minio.New(host, &minio.Options{
Creds: credentials.NewStaticV4(c.AccessKey, c.SecretKey, ""),
Secure: secure,
Region: c.Region,
})
if err != nil {
return nil, fmt.Errorf("offsite: %w", err)
}
return &S3{client: cl, bucket: c.Bucket, prefix: cleanPrefix(c.Prefix)}, nil
}
func splitEndpoint(ep string) (host string, secure bool, err error) {
ep = strings.TrimSpace(ep)
switch {
case ep == "":
return "", false, errors.New("offsite: no endpoint configured")
case strings.HasPrefix(ep, "https://"):
return strings.Trim(strings.TrimPrefix(ep, "https://"), "/"), true, nil
case strings.HasPrefix(ep, "http://"):
return strings.Trim(strings.TrimPrefix(ep, "http://"), "/"), false, nil
default:
return strings.Trim(ep, "/"), true, nil
}
}
func cleanPrefix(p string) string {
p = strings.Trim(p, "/")
if p == "" {
return ""
}
return p + "/"
}
// Check proves the bucket is reachable and the credentials may use it.
func (s *S3) Check(ctx context.Context) error {
ok, err := s.client.BucketExists(ctx, s.bucket)
if err != nil {
return fmt.Errorf("offsite: cannot reach bucket %q (endpoint unreachable or credentials rejected): %w", s.bucket, err)
}
if !ok {
return fmt.Errorf("offsite: bucket %q does not exist; create it first", s.bucket)
}
return nil
}
func (s *S3) Put(ctx context.Context, key string, r io.Reader, size int64) error {
_, err := s.client.PutObject(ctx, s.bucket, s.prefix+key, r, size, minio.PutObjectOptions{
ContentType: "application/octet-stream",
PartSize: partSize,
})
return err
}
func (s *S3) Get(ctx context.Context, key string) (io.ReadCloser, error) {
obj, err := s.client.GetObject(ctx, s.bucket, s.prefix+key, minio.GetObjectOptions{})
if err != nil {
return nil, err
}
// GetObject is lazy; Stat surfaces a missing key before the first read.
if _, err := obj.Stat(); err != nil {
obj.Close()
if isNotFound(err) {
return nil, fmt.Errorf("%w: %s", ErrNotFound, key)
}
return nil, err
}
return obj, nil
}
func (s *S3) List(ctx context.Context, prefix string) ([]Object, error) {
var out []Object
for o := range s.client.ListObjects(ctx, s.bucket, minio.ListObjectsOptions{Prefix: s.prefix + prefix, Recursive: true}) {
if o.Err != nil {
return nil, o.Err
}
out = append(out, Object{Key: strings.TrimPrefix(o.Key, s.prefix), Size: o.Size, Modified: o.LastModified})
}
return out, nil
}
func (s *S3) Remove(ctx context.Context, key string) error {
return s.client.RemoveObject(ctx, s.bucket, s.prefix+key, minio.RemoveObjectOptions{})
}
func isNotFound(err error) bool {
resp := minio.ToErrorResponse(err)
return resp.Code == "NoSuchKey" || resp.StatusCode == http.StatusNotFound
}
+195
View File
@@ -0,0 +1,195 @@
package offsite
import (
"bufio"
"crypto/aes"
"crypto/cipher"
"crypto/rand"
"crypto/sha256"
"encoding/base64"
"encoding/binary"
"encoding/hex"
"errors"
"fmt"
"io"
"strings"
)
// An object in the bucket is its file encrypted with AES-256-GCM in fixed
// segments, so a multi-gigabyte world streams through in constant memory:
//
// magic "FELISOS1" | 7-byte random nonce prefix | segment 0 | segment 1 | ...
//
// Segment i seals up to segmentSize bytes under the nonce
// prefix || uint32(i) || last, where last is 1 on the final segment only. The
// counter stops segments being reordered or dropped, and the last flag stops
// the object being cut short at a segment boundary: either change fails to
// authenticate.
const (
magic = "FELISOS1"
prefixSize = 7
segmentSize = 64 << 10
headerSize = len(magic) + prefixSize
// KeySize is the key length: AES-256.
KeySize = 32
)
// ErrAuth is a segment that does not authenticate: the wrong key, or an object
// damaged in the bucket or on the way.
var ErrAuth = errors.New("offsite: object does not decrypt with this key (wrong key, or the object is damaged)")
// NewKey returns a fresh random key in the text form ParseKey reads.
func NewKey() (string, error) {
k := make([]byte, KeySize)
if _, err := rand.Read(k); err != nil {
return "", err
}
return base64.StdEncoding.EncodeToString(k), nil
}
// ParseKey decodes a key written by NewKey (standard base64 of 32 bytes).
func ParseKey(s string) ([]byte, error) {
k, err := base64.StdEncoding.DecodeString(strings.TrimSpace(s))
if err != nil || len(k) != KeySize {
return nil, fmt.Errorf("offsite: the key must be %d random bytes in base64 (generate one with `felis offsite keygen`)", KeySize)
}
return k, nil
}
// KeyID names a key without revealing it, so status output and logs can say
// which key a bucket's objects were written with.
func KeyID(key []byte) string {
h := sha256.Sum256(append([]byte("felis-offsite-key-id\x00"), key...))
return hex.EncodeToString(h[:8])
}
// SealedSize is the size of the object Encrypt makes from n plaintext bytes.
// Uploads need it up front: an S3 upload of unknown length buffers far more.
func SealedSize(n int64) int64 {
segs := (n + segmentSize - 1) / segmentSize
if segs == 0 {
segs = 1
}
return int64(headerSize) + n + segs*16
}
func newAEAD(key []byte) (cipher.AEAD, error) {
if len(key) != KeySize {
return nil, fmt.Errorf("offsite: key is %d bytes, want %d", len(key), KeySize)
}
block, err := aes.NewCipher(key)
if err != nil {
return nil, err
}
return cipher.NewGCM(block)
}
func segmentNonce(prefix []byte, i uint32, last bool) []byte {
n := make([]byte, 12)
copy(n, prefix)
binary.BigEndian.PutUint32(n[prefixSize:], i)
if last {
n[11] = 1
}
return n
}
// Encrypt writes src to dst in the segmented format.
func Encrypt(dst io.Writer, src io.Reader, key []byte) error {
aead, err := newAEAD(key)
if err != nil {
return err
}
prefix := make([]byte, prefixSize)
if _, err := rand.Read(prefix); err != nil {
return err
}
if _, err := io.WriteString(dst, magic); err != nil {
return err
}
if _, err := dst.Write(prefix); err != nil {
return err
}
in := bufio.NewReaderSize(src, segmentSize+1)
buf := make([]byte, segmentSize)
out := make([]byte, 0, segmentSize+aead.Overhead())
for i := uint32(0); ; i++ {
n, err := io.ReadFull(in, buf)
switch {
case err == io.EOF || err == io.ErrUnexpectedEOF:
err = nil
case err != nil:
return err
}
last := n < segmentSize
if !last {
if _, perr := in.Peek(1); perr == io.EOF {
last = true
} else if perr != nil {
return perr
}
}
out = aead.Seal(out[:0], segmentNonce(prefix, i, last), buf[:n], []byte(magic))
if _, err := dst.Write(out); err != nil {
return err
}
if last {
return nil
}
if i == ^uint32(0) {
return errors.New("offsite: file too large to encrypt")
}
}
}
// Decrypt reverses Encrypt. Nothing unauthenticated is written: each segment
// is checked before its plaintext reaches dst, and a missing tail is an error.
// A caller writing to a file still has to discard it on error, since earlier
// segments were already written.
func Decrypt(dst io.Writer, src io.Reader, key []byte) error {
aead, err := newAEAD(key)
if err != nil {
return err
}
hdr := make([]byte, headerSize)
if _, err := io.ReadFull(src, hdr); err != nil {
return fmt.Errorf("offsite: object too short for its header: %w", err)
}
if string(hdr[:len(magic)]) != magic {
return errors.New("offsite: not a Felis off-site object (bad magic)")
}
prefix := hdr[len(magic):]
sealed := segmentSize + aead.Overhead()
in := bufio.NewReaderSize(src, sealed+1)
buf := make([]byte, sealed)
var plain []byte
for i := uint32(0); ; i++ {
n, err := io.ReadFull(in, buf)
switch {
case err == io.EOF:
return fmt.Errorf("offsite: object is cut short after %d segments", i)
case err == io.ErrUnexpectedEOF:
err = nil
case err != nil:
return err
}
last := n < sealed
if !last {
if _, perr := in.Peek(1); perr == io.EOF {
last = true
} else if perr != nil {
return perr
}
}
plain, err = aead.Open(plain[:0], segmentNonce(prefix, i, last), buf[:n], []byte(magic))
if err != nil {
return ErrAuth
}
if _, err := dst.Write(plain); err != nil {
return err
}
if last {
return nil
}
}
}
+446
View File
@@ -0,0 +1,446 @@
package offsite
import (
"bytes"
"context"
"crypto/rand"
"errors"
"fmt"
"io"
"os"
"path/filepath"
"sort"
"strings"
"sync"
"testing"
"time"
)
func testKey(t *testing.T) []byte {
t.Helper()
s, err := NewKey()
if err != nil {
t.Fatal(err)
}
k, err := ParseKey(s)
if err != nil {
t.Fatal(err)
}
return k
}
func seal(t *testing.T, plain, key []byte) []byte {
t.Helper()
var buf bytes.Buffer
if err := Encrypt(&buf, bytes.NewReader(plain), key); err != nil {
t.Fatal(err)
}
return buf.Bytes()
}
func TestEncryptRoundTrip(t *testing.T) {
key := testKey(t)
for _, n := range []int{0, 1, segmentSize - 1, segmentSize, segmentSize + 1, 3*segmentSize + 5} {
plain := make([]byte, n)
rand.Read(plain)
sealed := seal(t, plain, key)
if int64(len(sealed)) != SealedSize(int64(n)) {
t.Errorf("n=%d: sealed %d bytes, SealedSize says %d", n, len(sealed), SealedSize(int64(n)))
}
var out bytes.Buffer
if err := Decrypt(&out, bytes.NewReader(sealed), key); err != nil {
t.Fatalf("n=%d: decrypt: %v", n, err)
}
if !bytes.Equal(out.Bytes(), plain) {
t.Fatalf("n=%d: round trip changed the data", n)
}
}
}
// TestDecryptRejectsTampering: a wrong key, a flipped bit, a missing tail at a
// segment boundary and swapped segments all fail instead of yielding data.
func TestDecryptRejectsTampering(t *testing.T) {
key := testKey(t)
plain := make([]byte, 3*segmentSize)
rand.Read(plain)
sealed := seal(t, plain, key)
seg := segmentSize + 16
flipped := bytes.Clone(sealed)
flipped[headerSize+10] ^= 1
cut := sealed[:headerSize+2*seg]
swapped := bytes.Clone(sealed)
copy(swapped[headerSize:], sealed[headerSize+seg:headerSize+2*seg])
copy(swapped[headerSize+seg:], sealed[headerSize:headerSize+seg])
cases := map[string]struct {
data []byte
key []byte
}{
"wrong key": {sealed, testKey(t)},
"flipped bit": {flipped, key},
"cut short": {cut, key},
"swapped": {swapped, key},
"header only": {sealed[:headerSize], key},
"not ours": {[]byte("PK\x03\x04 a zip file, not a sealed object"), key},
}
for name, c := range cases {
if err := Decrypt(io.Discard, bytes.NewReader(c.data), c.key); err == nil {
t.Errorf("%s: decrypted without error", name)
}
}
if err := Decrypt(io.Discard, bytes.NewReader(sealed), testKey(t)); !errors.Is(err, ErrAuth) {
t.Errorf("wrong key error = %v, want ErrAuth", err)
}
}
func TestParseKey(t *testing.T) {
s, _ := NewKey()
k, err := ParseKey(" " + s + "\n")
if err != nil || len(k) != KeySize {
t.Fatalf("ParseKey(NewKey()) = %d bytes, %v", len(k), err)
}
for _, bad := range []string{"", "short", strings.Repeat("A", 40)} {
if _, err := ParseKey(bad); err == nil {
t.Errorf("ParseKey(%q) accepted", bad)
}
}
if KeyID(k) == KeyID(testKey(t)) || len(KeyID(k)) != 16 {
t.Errorf("KeyID = %q, want 16 hex chars that differ per key", KeyID(k))
}
}
// ---- fakes ----------------------------------------------------------------
type memBucket struct {
mu sync.Mutex
objs map[string][]byte
putErr map[string]error
puts int
removed []string
}
func newMemBucket() *memBucket {
return &memBucket{objs: map[string][]byte{}, putErr: map[string]error{}}
}
func (b *memBucket) Put(_ context.Context, key string, r io.Reader, size int64) error {
if err := b.putErr[key]; err != nil {
io.Copy(io.Discard, r)
return err
}
data, err := io.ReadAll(r)
if err != nil {
return err
}
if int64(len(data)) != size {
return fmt.Errorf("put %s: got %d bytes, declared %d", key, len(data), size)
}
b.mu.Lock()
defer b.mu.Unlock()
b.objs[key] = data
b.puts++
return nil
}
func (b *memBucket) Get(_ context.Context, key string) (io.ReadCloser, error) {
b.mu.Lock()
defer b.mu.Unlock()
data, ok := b.objs[key]
if !ok {
return nil, fmt.Errorf("%w: %s", ErrNotFound, key)
}
return io.NopCloser(bytes.NewReader(data)), nil
}
func (b *memBucket) List(_ context.Context, prefix string) ([]Object, error) {
b.mu.Lock()
defer b.mu.Unlock()
var out []Object
for k, v := range b.objs {
if strings.HasPrefix(k, prefix) {
out = append(out, Object{Key: k, Size: int64(len(v))})
}
}
sort.Slice(out, func(i, j int) bool { return out[i].Key < out[j].Key })
return out, nil
}
func (b *memBucket) Remove(_ context.Context, key string) error {
b.mu.Lock()
defer b.mu.Unlock()
delete(b.objs, key)
b.removed = append(b.removed, key)
return nil
}
type row struct {
WorldBackup
status string
expires time.Time
offsite time.Time
}
type fakeCatalog struct{ rows []*row }
func (c *fakeCatalog) PendingWorlds(context.Context) ([]WorldBackup, error) {
var out []WorldBackup
for _, r := range c.rows {
if r.status == "present" && r.offsite.IsZero() {
out = append(out, r.WorldBackup)
}
}
return out, nil
}
func (c *fakeCatalog) PresentWorlds(context.Context) ([]WorldBackup, error) {
var out []WorldBackup
for _, r := range c.rows {
if r.status == "present" {
out = append(out, r.WorldBackup)
}
}
return out, nil
}
func (c *fakeCatalog) MarkOffsite(_ context.Context, id string, at time.Time) error {
for _, r := range c.rows {
if r.ID == id {
r.offsite = at
}
}
return nil
}
func (c *fakeCatalog) ExpiredRefs(_ context.Context, now time.Time) ([]string, error) {
var out []string
for _, r := range c.rows {
if r.status == "deleted" && r.expires.Before(now) && !r.offsite.IsZero() {
out = append(out, r.Ref)
}
}
return out, nil
}
func writeFile(t *testing.T, dir, name string, size int) []byte {
t.Helper()
data := make([]byte, size)
rand.Read(data)
if err := os.WriteFile(filepath.Join(dir, name), data, 0o600); err != nil {
t.Fatal(err)
}
return data
}
var now = time.Date(2026, 9, 24, 12, 0, 0, 0, time.UTC)
func newSyncer(t *testing.T, cat *fakeCatalog) (*Syncer, *memBucket) {
t.Helper()
b := newMemBucket()
return &Syncer{
Bucket: b, Catalog: cat, Key: testKey(t),
ArchiveDir: t.TempDir(), DBDir: t.TempDir(), DBKeep: 2,
Now: func() time.Time { return now },
}, b
}
// TestSyncCopiesWorldsAndRecordsThem: each pending archive is encrypted into
// the bucket and marked; a second run sends nothing again.
func TestSyncCopiesWorldsAndRecordsThem(t *testing.T) {
cat := &fakeCatalog{rows: []*row{
{WorldBackup: WorldBackup{ID: "b1", Server: "alpha", Ref: "/var/lib/felis/archives/alpha-1.tar.gz"}, status: "present"},
{WorldBackup: WorldBackup{ID: "b2", Server: "beta", Ref: "/var/lib/felis/archives/beta-2.tar.gz"}, status: "present"},
}}
s, b := newSyncer(t, cat)
alpha := writeFile(t, s.ArchiveDir, "alpha-1.tar.gz", 3*segmentSize+7)
writeFile(t, s.ArchiveDir, "beta-2.tar.gz", 10)
res, err := s.Run(context.Background())
if err != nil {
t.Fatalf("run: %v (%+v)", err, res)
}
if res.WorldsUploaded != 2 || res.WorldsPending != 0 || res.RemoteWorlds != 2 {
t.Fatalf("result = %+v, want 2 uploaded, 0 pending, 2 remote", res)
}
for _, r := range cat.rows {
if !r.offsite.Equal(now) {
t.Errorf("row %s offsite_at = %v, want %v", r.ID, r.offsite, now)
}
}
var out bytes.Buffer
if err := Decrypt(&out, bytes.NewReader(b.objs["worlds/alpha-1.tar.gz.fenc"]), s.Key); err != nil || !bytes.Equal(out.Bytes(), alpha) {
t.Fatalf("stored alpha does not decrypt to the archive (err %v)", err)
}
puts := b.puts
if res, err := s.Run(context.Background()); err != nil || res.WorldsUploaded != 0 || b.puts != puts {
t.Fatalf("second run uploaded again: %+v, puts %d→%d, err %v", res, puts, b.puts, err)
}
}
// TestSyncResumesWithoutResending: an object stored by a run that died before
// recording it is recorded, not uploaded twice; a partial one is replaced.
func TestSyncResumesWithoutResending(t *testing.T) {
cat := &fakeCatalog{rows: []*row{
{WorldBackup: WorldBackup{ID: "b1", Server: "alpha", Ref: "/a/alpha-1.tar.gz"}, status: "present"},
{WorldBackup: WorldBackup{ID: "b2", Server: "beta", Ref: "/a/beta-2.tar.gz"}, status: "present"},
}}
s, b := newSyncer(t, cat)
alpha := writeFile(t, s.ArchiveDir, "alpha-1.tar.gz", 1000)
writeFile(t, s.ArchiveDir, "beta-2.tar.gz", 1000)
b.objs["worlds/alpha-1.tar.gz.fenc"] = seal(t, alpha, s.Key)
b.objs["worlds/beta-2.tar.gz.fenc"] = []byte("partial")
res, err := s.Run(context.Background())
if err != nil {
t.Fatal(err)
}
if res.WorldsUploaded != 1 || b.puts != 1 {
t.Fatalf("uploaded %d (puts %d), want only the partial beta resent", res.WorldsUploaded, b.puts)
}
if cat.rows[0].offsite.IsZero() || cat.rows[1].offsite.IsZero() {
t.Fatal("both rows should be recorded as copied")
}
}
// TestSyncFailureLeavesRowPending: a failed upload is reported, the row stays
// unmarked (so the reaper keeps the world), and the rest still go.
func TestSyncFailureLeavesRowPending(t *testing.T) {
cat := &fakeCatalog{rows: []*row{
{WorldBackup: WorldBackup{ID: "b1", Server: "alpha", Ref: "/a/alpha-1.tar.gz"}, status: "present"},
{WorldBackup: WorldBackup{ID: "b2", Server: "beta", Ref: "/a/beta-2.tar.gz"}, status: "present"},
{WorldBackup: WorldBackup{ID: "b3", Server: "gamma", Ref: "/a/gamma-3.tar.gz"}, status: "present"},
}}
s, b := newSyncer(t, cat)
writeFile(t, s.ArchiveDir, "alpha-1.tar.gz", 100)
writeFile(t, s.ArchiveDir, "beta-2.tar.gz", 100)
b.putErr["worlds/alpha-1.tar.gz.fenc"] = errors.New("503 slow down")
res, err := s.Run(context.Background())
if err == nil {
t.Fatal("run reported success despite a failed upload")
}
if !cat.rows[0].offsite.IsZero() {
t.Error("the failed archive was recorded as copied")
}
if cat.rows[1].offsite.IsZero() {
t.Error("the archive after the failure was not copied")
}
if res.WorldsPending != 1 || len(res.WorldsMissing) != 1 || !strings.Contains(res.WorldsMissing[0], "gamma") {
t.Fatalf("result = %+v, want alpha pending, gamma missing", res)
}
}
// TestSyncExpiresOnlyPastRetention: a remote archive goes once its row has
// expired; one evicted early from the local disk stays until then, and an
// object with no row at all is left alone.
func TestSyncExpiresOnlyPastRetention(t *testing.T) {
cat := &fakeCatalog{rows: []*row{
{WorldBackup: WorldBackup{ID: "old", Ref: "/a/old.tar.gz"}, status: "deleted", expires: now.Add(-time.Hour), offsite: now.Add(-100 * 24 * time.Hour)},
{WorldBackup: WorldBackup{ID: "evicted", Ref: "/a/evicted.tar.gz"}, status: "deleted", expires: now.Add(30 * 24 * time.Hour), offsite: now.Add(-24 * time.Hour)},
}}
s, b := newSyncer(t, cat)
for _, k := range []string{"worlds/old.tar.gz.fenc", "worlds/evicted.tar.gz.fenc", "worlds/unknown.tar.gz.fenc"} {
b.objs[k] = []byte("x")
}
res, err := s.Run(context.Background())
if err != nil {
t.Fatal(err)
}
if res.WorldsExpired != 1 || len(b.removed) != 1 || b.removed[0] != "worlds/old.tar.gz.fenc" {
t.Fatalf("removed %v (expired %d), want only the expired archive", b.removed, res.WorldsExpired)
}
if res.RemoteWorlds != 2 {
t.Fatalf("remote worlds = %d, want 2 left", res.RemoteWorlds)
}
}
// TestSyncDBKeepsNewest: only the newest DBKeep bundles are sent, and older
// remote bundles are pruned down to DBKeep.
func TestSyncDBKeepsNewest(t *testing.T) {
s, b := newSyncer(t, &fakeCatalog{})
for _, name := range []string{
"felis-db-20260920T030000Z-daily.tar",
"felis-db-20260922T030000Z-daily.tar",
"felis-db-20260923T030000Z-manual.tar",
"felis-db-20260924T030000Z-daily.tar",
} {
writeFile(t, s.DBDir, name, 50)
}
b.objs["db/felis-db-20260901T030000Z-daily.tar.fenc"] = []byte("old")
b.objs["db/notes.txt"] = []byte("not ours")
res, err := s.Run(context.Background())
if err != nil {
t.Fatal(err)
}
if res.DBUploaded != 2 || res.RemoteDB != 2 || res.NewestDB != "felis-db-20260924T030000Z-daily.tar" {
t.Fatalf("result = %+v, want the newest 2 uploaded", res)
}
var keys []string
for k := range b.objs {
keys = append(keys, k)
}
sort.Strings(keys)
want := "db/felis-db-20260923T030000Z-manual.tar.fenc db/felis-db-20260924T030000Z-daily.tar.fenc db/notes.txt"
if strings.Join(keys, " ") != want {
t.Fatalf("bucket = %v, want %s", keys, want)
}
listed, err := ListDB(context.Background(), b)
if err != nil || len(listed) != 2 || listed[0].Key != "felis-db-20260924T030000Z-daily.tar" {
t.Fatalf("ListDB = %+v, %v", listed, err)
}
}
// TestFetchWorldsRestoresVolume: after a rebuild, every present archive the
// volume lacks comes back byte for byte; ones already there are left alone.
func TestFetchWorldsRestoresVolume(t *testing.T) {
cat := &fakeCatalog{rows: []*row{
{WorldBackup: WorldBackup{ID: "b1", Server: "alpha", Ref: "/a/alpha-1.tar.gz"}, status: "present"},
{WorldBackup: WorldBackup{ID: "b2", Server: "beta", Ref: "/a/beta-2.tar.gz"}, status: "present"},
{WorldBackup: WorldBackup{ID: "b3", Server: "gamma", Ref: "/a/gamma-3.tar.gz"}, status: "present"},
}}
s, b := newSyncer(t, cat)
alpha := writeFile(t, s.ArchiveDir, "alpha-1.tar.gz", segmentSize+3)
writeFile(t, s.ArchiveDir, "beta-2.tar.gz", 20)
if res, err := s.Run(context.Background()); err != nil || len(res.WorldsMissing) != 1 {
t.Fatalf("run = %+v, %v; want gamma reported missing", res, err)
}
fresh := t.TempDir()
os.WriteFile(filepath.Join(fresh, "beta-2.tar.gz"), []byte("kept"), 0o600)
res, err := FetchWorlds(context.Background(), b, cat, s.Key, fresh, nil)
if err != nil {
t.Fatal(err)
}
if strings.Join(res.Fetched, ",") != "alpha-1.tar.gz" || len(res.Missing) != 1 {
t.Fatalf("fetch = %+v, want alpha fetched, gamma missing", res)
}
got, _ := os.ReadFile(filepath.Join(fresh, "alpha-1.tar.gz"))
if !bytes.Equal(got, alpha) {
t.Fatal("fetched alpha differs from the original")
}
if kept, _ := os.ReadFile(filepath.Join(fresh, "beta-2.tar.gz")); string(kept) != "kept" {
t.Fatal("fetch overwrote an archive already on the volume")
}
// A wrong key leaves no file behind, partial or whole.
other := t.TempDir()
if err := FetchObject(context.Background(), b, testKey(t), "worlds/alpha-1.tar.gz.fenc", filepath.Join(other, "alpha-1.tar.gz"), 0o600); !errors.Is(err, ErrAuth) {
t.Fatalf("wrong key fetch = %v, want ErrAuth", err)
}
if entries, _ := os.ReadDir(other); len(entries) != 0 {
t.Fatalf("wrong key left files: %v", entries)
}
}
func TestWorldKeyRejectsOddRefs(t *testing.T) {
if k, ok := WorldKey("/var/lib/felis/archives/alpha-1.tar.gz"); !ok || k != "worlds/alpha-1.tar.gz.fenc" {
t.Fatalf("WorldKey = %q, %v", k, ok)
}
for _, ref := range []string{"", "/", "..", "/a/.."} {
if _, ok := WorldKey(ref); ok {
t.Errorf("WorldKey(%q) accepted", ref)
}
}
}
+73
View File
@@ -0,0 +1,73 @@
package offsite
import (
"context"
"database/sql"
"time"
)
// PGCatalog is the Catalog on the felis database.
type PGCatalog struct{ DB *sql.DB }
func (c PGCatalog) PendingWorlds(ctx context.Context) ([]WorldBackup, error) {
return c.query(ctx, `SELECT id, server_name, backup_ref, created_at FROM world_backups
WHERE status = 'present' AND offsite_at IS NULL ORDER BY created_at`)
}
func (c PGCatalog) PresentWorlds(ctx context.Context) ([]WorldBackup, error) {
return c.query(ctx, `SELECT id, server_name, backup_ref, created_at FROM world_backups
WHERE status = 'present' ORDER BY created_at`)
}
func (c PGCatalog) query(ctx context.Context, q string, args ...any) ([]WorldBackup, error) {
rows, err := c.DB.QueryContext(ctx, q, args...)
if err != nil {
return nil, err
}
defer rows.Close()
var out []WorldBackup
for rows.Next() {
var w WorldBackup
if err := rows.Scan(&w.ID, &w.Server, &w.Ref, &w.Created); err != nil {
return nil, err
}
out = append(out, w)
}
return out, rows.Err()
}
func (c PGCatalog) MarkOffsite(ctx context.Context, id string, at time.Time) error {
_, err := c.DB.ExecContext(ctx,
`UPDATE world_backups SET offsite_at = $2 WHERE id = $1 AND offsite_at IS NULL`, id, at)
return err
}
func (c PGCatalog) ExpiredRefs(ctx context.Context, now time.Time) ([]string, error) {
rows, err := c.DB.QueryContext(ctx, `SELECT backup_ref FROM world_backups
WHERE status = 'deleted' AND expires_at < $1 AND offsite_at IS NOT NULL`, now)
if err != nil {
return nil, err
}
defer rows.Close()
var out []string
for rows.Next() {
var ref string
if err := rows.Scan(&ref); err != nil {
return nil, err
}
out = append(out, ref)
}
return out, rows.Err()
}
// PendingCount is how many present archives wait for their copy, and since
// when the oldest has waited.
func (c PGCatalog) PendingCount(ctx context.Context) (int, time.Time, error) {
var (
n int
oldest sql.NullTime
)
err := c.DB.QueryRowContext(ctx, `SELECT count(*), min(created_at) FROM world_backups
WHERE status = 'present' AND offsite_at IS NULL`).Scan(&n, &oldest)
return n, oldest.Time, err
}
+64
View File
@@ -0,0 +1,64 @@
package offsite
import (
"encoding/json"
"os"
"path/filepath"
"time"
)
// DefaultStatusFile is where `felis offsite sync` records its last run; the
// watchdog and `felis offsite status` read it.
const DefaultStatusFile = "/var/lib/felis/offsite/status.json"
// StaleAfter is how long the sync may go without a clean run before the
// watchdog mails the owners. The timer runs hourly, so this rides out a
// provider's bad morning, and still leaves a day's database bundle uncopied
// for at most half a day.
const StaleAfter = 12 * time.Hour
// Status is the record of the last run.
type Status struct {
LastAttempt time.Time `json:"last_attempt"`
// LastSuccess is the last run in which every step succeeded.
LastSuccess time.Time `json:"last_success,omitempty"`
LastError string `json:"last_error,omitempty"`
Endpoint string `json:"endpoint"`
Bucket string `json:"bucket"`
Prefix string `json:"prefix,omitempty"`
KeyID string `json:"key_id"`
Result Result `json:"result"`
}
// ReadStatus reads the status file. A missing file is (nil, nil): no sync has
// run yet.
func ReadStatus(p string) (*Status, error) {
raw, err := os.ReadFile(p)
if os.IsNotExist(err) {
return nil, nil
}
if err != nil {
return nil, err
}
var st Status
if err := json.Unmarshal(raw, &st); err != nil {
return nil, err
}
return &st, nil
}
// WriteStatus replaces the status file atomically.
func WriteStatus(p string, st Status) error {
if err := os.MkdirAll(filepath.Dir(p), 0o700); err != nil {
return err
}
raw, err := json.MarshalIndent(st, "", " ")
if err != nil {
return err
}
tmp := p + ".tmp"
if err := os.WriteFile(tmp, append(raw, '\n'), 0o600); err != nil {
return err
}
return os.Rename(tmp, p)
}
+464
View File
@@ -0,0 +1,464 @@
// Package offsite keeps a second copy of what a lost node would take with it:
// every world archive (world_backups) and the newest control-plane database
// bundles (internal/dbbackup), encrypted, in an S3-compatible bucket off the
// machine. `felis offsite sync` runs it from felis-offsite.timer on the host,
// which is where both the archive volume and the bundle directory live.
//
// The database records the copy: world_backups.offsite_at is set once an
// archive's object is in the bucket, and with [offsite] configured the reaper
// deletes an idle world only after that (internal/reaper). Remote world
// objects go when their row has expired, so the bucket keeps each archive for
// the same retention the panel promises, including archives evicted early from
// the local disk to make room. Remote bundles are pruned to the newest DBKeep.
package offsite
import (
"context"
"errors"
"fmt"
"io"
"os"
"path"
"path/filepath"
"sort"
"strings"
"syscall"
"time"
"felis.lolicon.best/internal/dbbackup"
)
// Object key layout under the configured prefix.
const (
worldsDir = "worlds/"
dbDir = "db/"
objExt = ".fenc"
)
// WorldKey is the object key of the world archive stored at ref (the
// world_backups.backup_ref, an in-pod path whose last element is the file name
// on the archive volume).
func WorldKey(ref string) (string, bool) {
name := path.Base(ref)
if !safeName(name) {
return "", false
}
return worldsDir + name + objExt, true
}
// DBKey is the object key of a database bundle.
func DBKey(bundle string) string { return dbDir + bundle + objExt }
func safeName(name string) bool {
return name != "" && name != "." && name != ".." && name != "/" && !strings.ContainsAny(name, `/\`)
}
// WorldBackup is one world_backups row the sync works on.
type WorldBackup struct {
ID string
Server string
Ref string
Created time.Time
}
// Catalog is the world_backups view the sync needs; PGCatalog in production.
type Catalog interface {
// PendingWorlds lists present archives without an off-site copy yet,
// oldest first.
PendingWorlds(ctx context.Context) ([]WorldBackup, error)
// MarkOffsite records that the archive of row id is in the bucket.
MarkOffsite(ctx context.Context, id string, at time.Time) error
// ExpiredRefs lists the backup_ref of every row past its retention
// (deleted and expires_at < now) whose archive was copied off-site.
ExpiredRefs(ctx context.Context, now time.Time) ([]string, error)
// PresentWorlds lists every present archive, for a restore of the volume.
PresentWorlds(ctx context.Context) ([]WorldBackup, error)
}
// Syncer copies what is missing from the bucket and prunes what has expired.
type Syncer struct {
Bucket Bucket
Catalog Catalog
Key []byte
// ArchiveDir is the host directory of the world archive volume; "" when
// there is none yet (nothing has been archived on this install).
ArchiveDir string
// DBDir holds the database bundles; DBKeep is how many of the newest the
// bucket keeps.
DBDir string
DBKeep int
Now func() time.Time
Log io.Writer
}
// Result is what one Run did and found.
type Result struct {
WorldsUploaded int `json:"worlds_uploaded"`
BytesUploaded int64 `json:"bytes_uploaded"`
// WorldsPending are present archives still without an off-site copy
// after this run; the missing ones below are counted there alone.
WorldsPending int `json:"worlds_pending"`
// WorldsMissing are present rows whose archive file is not on the volume:
// nothing to copy, and nothing a restore could use.
WorldsMissing []string `json:"worlds_missing,omitempty"`
WorldsExpired int `json:"worlds_expired"`
RemoteWorlds int `json:"remote_worlds"`
RemoteBytes int64 `json:"remote_bytes"`
DBUploaded int `json:"db_uploaded"`
DBPruned int `json:"db_pruned"`
RemoteDB int `json:"remote_db"`
NewestDB string `json:"newest_db,omitempty"`
Errors []string `json:"errors,omitempty"`
}
func (s *Syncer) now() time.Time {
if s.Now != nil {
return s.Now()
}
return time.Now()
}
func (s *Syncer) logf(format string, args ...any) {
if s.Log != nil {
fmt.Fprintf(s.Log, "felis offsite: "+format+"\n", args...)
}
}
// Run does one pass: world archives, then database bundles, then expiry. A
// failure on one item is recorded and the pass carries on; the returned error
// is non-nil when anything failed.
func (s *Syncer) Run(ctx context.Context) (Result, error) {
var res Result
fail := func(format string, args ...any) {
msg := fmt.Sprintf(format, args...)
res.Errors = append(res.Errors, msg)
s.logf("%s", msg)
}
remoteWorlds, err := s.listSizes(ctx, worldsDir)
if err != nil {
return res, fmt.Errorf("list %s in the bucket: %w", worldsDir, err)
}
s.syncWorlds(ctx, remoteWorlds, &res, fail)
s.syncDB(ctx, &res, fail)
s.expireWorlds(ctx, remoteWorlds, &res, fail)
for _, size := range remoteWorlds {
res.RemoteWorlds++
res.RemoteBytes += size
}
if len(res.Errors) > 0 {
return res, fmt.Errorf("%d of this run's steps failed; first: %s", len(res.Errors), res.Errors[0])
}
return res, nil
}
func (s *Syncer) listSizes(ctx context.Context, prefix string) (map[string]int64, error) {
objs, err := s.Bucket.List(ctx, prefix)
if err != nil {
return nil, err
}
out := make(map[string]int64, len(objs))
for _, o := range objs {
out[o.Key] = o.Size
}
return out, nil
}
func (s *Syncer) syncWorlds(ctx context.Context, remote map[string]int64, res *Result, fail func(string, ...any)) {
pending, err := s.Catalog.PendingWorlds(ctx)
if err != nil {
fail("list world archives waiting for a copy: %v", err)
return
}
if s.ArchiveDir == "" {
res.WorldsPending = len(pending)
if len(pending) > 0 {
fail("%d world archives wait for a copy, but there is no archive volume to read them from", len(pending))
}
return
}
for _, w := range pending {
if ctx.Err() != nil {
fail("stopped: %v", ctx.Err())
res.WorldsPending++
continue
}
key, ok := WorldKey(w.Ref)
if !ok {
fail("world archive %s of %s has an unusable path %q", w.ID, w.Server, w.Ref)
res.WorldsPending++
continue
}
local := filepath.Join(s.ArchiveDir, path.Base(w.Ref))
fi, err := os.Stat(local)
if errors.Is(err, os.ErrNotExist) {
res.WorldsMissing = append(res.WorldsMissing, fmt.Sprintf("%s (%s, backup %s)", path.Base(w.Ref), w.Server, w.ID))
continue
}
if err != nil {
fail("world archive %s: %v", local, err)
res.WorldsPending++
continue
}
want := SealedSize(fi.Size())
// An earlier run may have stored the object and died before recording
// it; a complete object is only recorded, not sent again.
if size, ok := remote[key]; !ok || size != want {
if err := s.putFile(ctx, key, local, fi.Size()); err != nil {
fail("upload %s (%s): %v", path.Base(w.Ref), w.Server, err)
res.WorldsPending++
continue
}
remote[key] = want
res.WorldsUploaded++
res.BytesUploaded += fi.Size()
s.logf("copied world archive %s (%s, %s)", path.Base(w.Ref), w.Server, HumanBytes(fi.Size()))
}
if err := s.Catalog.MarkOffsite(ctx, w.ID, s.now()); err != nil {
fail("record the copy of %s: %v", path.Base(w.Ref), err)
res.WorldsPending++
}
}
}
func (s *Syncer) syncDB(ctx context.Context, res *Result, fail func(string, ...any)) {
if s.DBDir == "" {
return
}
keep := s.DBKeep
if keep < 1 {
keep = 1
}
local, err := dbbackup.List(s.DBDir)
if err != nil {
fail("list database bundles in %s: %v", s.DBDir, err)
return
}
remote, err := s.listSizes(ctx, dbDir)
if err != nil {
fail("list %s in the bucket: %v", dbDir, err)
return
}
// Only the newest keep bundles are worth sending: older ones would be
// pruned again at the end of this very pass.
if len(local) > keep {
local = local[:keep]
}
for _, b := range local {
key := DBKey(b.Name)
want := SealedSize(b.Size)
if size, ok := remote[key]; ok && size == want {
continue
}
if err := s.putFile(ctx, key, b.Path, b.Size); err != nil {
fail("upload database bundle %s: %v", b.Name, err)
continue
}
remote[key] = want
res.DBUploaded++
res.BytesUploaded += b.Size
s.logf("copied database bundle %s (%s)", b.Name, HumanBytes(b.Size))
}
var names []string
for key := range remote {
name := strings.TrimSuffix(strings.TrimPrefix(key, dbDir), objExt)
if _, _, ok := dbbackup.ParseBundleName(name); ok && strings.HasSuffix(key, objExt) {
names = append(names, name)
}
}
// Bundle names start with their UTC stamp, so newest sorts last.
sort.Sort(sort.Reverse(sort.StringSlice(names)))
for i, name := range names {
if i < keep {
continue
}
if err := s.Bucket.Remove(ctx, DBKey(name)); err != nil {
fail("prune database bundle %s: %v", name, err)
continue
}
res.DBPruned++
}
res.RemoteDB = min(len(names), keep)
if len(names) > 0 {
res.NewestDB = names[0]
}
}
func (s *Syncer) expireWorlds(ctx context.Context, remote map[string]int64, res *Result, fail func(string, ...any)) {
refs, err := s.Catalog.ExpiredRefs(ctx, s.now())
if err != nil {
fail("list expired world archives: %v", err)
return
}
for _, ref := range refs {
key, ok := WorldKey(ref)
if !ok {
continue
}
if _, ok := remote[key]; !ok {
continue
}
if err := s.Bucket.Remove(ctx, key); err != nil {
fail("remove expired %s: %v", key, err)
continue
}
delete(remote, key)
res.WorldsExpired++
}
}
// putFile encrypts the file at p into key. The sealed size is known in
// advance, so the upload streams: nothing larger than one part is buffered.
func (s *Syncer) putFile(ctx context.Context, key, p string, size int64) error {
f, err := os.Open(p)
if err != nil {
return err
}
defer f.Close()
pr, pw := io.Pipe()
go func() {
// A file that changed size under us would not match the declared
// length; LimitReader keeps the stream to the size we announced and
// the length check below catches a short one.
err := Encrypt(pw, io.LimitReader(f, size), s.Key)
pw.CloseWithError(err)
}()
err = s.Bucket.Put(ctx, key, pr, SealedSize(size))
pr.CloseWithError(errors.New("upload finished"))
return err
}
// FetchResult is what Fetch did.
type FetchResult struct {
Fetched []string
Present int
Missing []string // present rows with no object in the bucket
Failures []string
}
// FetchWorlds downloads every present archive the volume lacks: the volume
// half of a rebuild, after the database came back from a bundle.
func FetchWorlds(ctx context.Context, b Bucket, cat Catalog, key []byte, archiveDir string, log io.Writer) (FetchResult, error) {
var res FetchResult
worlds, err := cat.PresentWorlds(ctx)
if err != nil {
return res, err
}
res.Present = len(worlds)
for _, w := range worlds {
objKey, ok := WorldKey(w.Ref)
if !ok {
res.Failures = append(res.Failures, fmt.Sprintf("%s: unusable path %q", w.ID, w.Ref))
continue
}
dst := filepath.Join(archiveDir, path.Base(w.Ref))
if _, err := os.Stat(dst); err == nil {
continue
}
// World-readable like the archives the backup Jobs write: the restore
// Job reads them as its own non-root user.
switch err := FetchObject(ctx, b, key, objKey, dst, 0o644); {
case errors.Is(err, ErrNotFound):
res.Missing = append(res.Missing, fmt.Sprintf("%s (%s)", path.Base(w.Ref), w.Server))
case err != nil:
res.Failures = append(res.Failures, fmt.Sprintf("%s: %v", path.Base(w.Ref), err))
default:
alignOwner(dst, archiveDir)
res.Fetched = append(res.Fetched, path.Base(w.Ref))
if log != nil {
fmt.Fprintf(log, "felis offsite: restored %s (%s)\n", path.Base(w.Ref), w.Server)
}
}
}
if len(res.Failures) > 0 {
return res, fmt.Errorf("%d archives failed; first: %s", len(res.Failures), res.Failures[0])
}
return res, nil
}
// alignOwner gives a restored archive the archive volume's owner, which is the
// user the backup Jobs write as, so the volume looks as they left it. Only root
// can; for anyone else the file stays theirs, readable all the same.
func alignOwner(file, dir string) {
if os.Geteuid() != 0 {
return
}
fi, err := os.Stat(dir)
if err != nil {
return
}
if st, ok := fi.Sys().(*syscall.Stat_t); ok {
_ = os.Lchown(file, int(st.Uid), int(st.Gid))
}
}
// FetchObject downloads and decrypts key into dst, created with mode. The file
// appears under its name only once it has decrypted completely; until then it
// is a hidden .partial next to it.
func FetchObject(ctx context.Context, b Bucket, key []byte, objKey, dst string, mode os.FileMode) error {
rc, err := b.Get(ctx, objKey)
if err != nil {
return err
}
defer rc.Close()
tmp := filepath.Join(filepath.Dir(dst), "."+filepath.Base(dst)+".partial")
f, err := os.OpenFile(tmp, os.O_CREATE|os.O_TRUNC|os.O_WRONLY, 0o600)
if err != nil {
return err
}
if err := f.Chmod(mode); err != nil {
f.Close()
os.Remove(tmp)
return err
}
if err := Decrypt(f, rc, key); err != nil {
f.Close()
os.Remove(tmp)
return err
}
if err := f.Sync(); err != nil {
f.Close()
os.Remove(tmp)
return err
}
if err := f.Close(); err != nil {
os.Remove(tmp)
return err
}
return os.Rename(tmp, dst)
}
// ListDB returns the database bundles in the bucket, newest first.
func ListDB(ctx context.Context, b Bucket) ([]Object, error) {
objs, err := b.List(ctx, dbDir)
if err != nil {
return nil, err
}
var out []Object
for _, o := range objs {
name := strings.TrimSuffix(strings.TrimPrefix(o.Key, dbDir), objExt)
if _, _, ok := dbbackup.ParseBundleName(name); !ok || !strings.HasSuffix(o.Key, objExt) {
continue
}
o.Key = name
out = append(out, o)
}
sort.Slice(out, func(i, j int) bool { return out[i].Key > out[j].Key })
return out, nil
}
// HumanBytes formats n in binary units.
func HumanBytes(n int64) string {
const unit = 1024
if n < unit {
return fmt.Sprintf("%d B", n)
}
div, exp := int64(unit), 0
for m := n / unit; m >= unit; m /= unit {
div *= unit
exp++
}
return fmt.Sprintf("%.1f %ciB", float64(n)/float64(div), "KMGTPE"[exp])
}
+37
View File
@@ -1029,6 +1029,43 @@ func backupPVC(p Params) *corev1.PersistentVolumeClaim {
}
}
// VolumeBinderPod mounts pvc and exits at once. A WaitForFirstConsumer volume
// (k3s local-path) has no directory until a pod uses it, and on a rebuilt node
// nothing has yet, so `felis offsite fetch-worlds` runs this to have the
// archive volume provisioned before it writes the archives back into it. It
// runs as the control-plane identity, which the archive volume's files already
// belong to (fsGroup).
func VolumeBinderPod(ns, pvc, image string) *corev1.Pod {
return &corev1.Pod{
TypeMeta: metav1.TypeMeta{APIVersion: "v1", Kind: "Pod"},
ObjectMeta: metav1.ObjectMeta{
GenerateName: "felis-bind-" + pvc + "-",
Namespace: ns,
Labels: map[string]string{LabelName: appName, LabelComponent: "volume-binder"},
},
Spec: corev1.PodSpec{
RestartPolicy: corev1.RestartPolicyNever,
AutomountServiceAccountToken: boolPtr(false),
SecurityContext: hardenedPodSecurityContext(),
Containers: []corev1.Container{{
Name: "bind",
Image: image,
Command: []string{felisBinaryPath, "version"},
SecurityContext: hardenedContainerSecurityContext(),
VolumeMounts: []corev1.VolumeMount{{Name: "archives", MountPath: "/archives"}},
Resources: corev1.ResourceRequirements{
Requests: corev1.ResourceList{corev1.ResourceCPU: resource.MustParse("10m"), corev1.ResourceMemory: resource.MustParse("16Mi")},
Limits: corev1.ResourceList{corev1.ResourceCPU: resource.MustParse("100m"), corev1.ResourceMemory: resource.MustParse("64Mi")},
},
}},
Volumes: []corev1.Volume{{
Name: "archives",
VolumeSource: corev1.VolumeSource{PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ClaimName: pvc}},
}},
},
}
}
// registryLabels are the registry's recommended labels. Note the absence of
// part-of=felis-control-plane: that is what keeps the registry out of the RCON
// NetworkPolicy peer's reach (asserted in workloads_test.go).
+29
View File
@@ -164,6 +164,35 @@ func TestControlPlanePods_Hardened(t *testing.T) {
}
}
// TestVolumeBinderPod pins the pod `felis offsite fetch-worlds` runs to get the
// archive volume provisioned: hardened like the control plane (it runs in the
// Minecraft namespace under PSA), mounting the named claim, gone once it exits.
func TestVolumeBinderPod(t *testing.T) {
pod := VolumeBinderPod("minecraft", "felis-backups", "registry.example/felis:1")
ps := pod.Spec
if pod.Namespace != "minecraft" || pod.GenerateName == "" || ps.RestartPolicy != corev1.RestartPolicyNever {
t.Fatalf("pod meta = %+v, restart %s", pod.ObjectMeta, ps.RestartPolicy)
}
if ps.SecurityContext == nil || ps.SecurityContext.RunAsNonRoot == nil || !*ps.SecurityContext.RunAsNonRoot ||
ps.SecurityContext.SeccompProfile == nil || ps.SecurityContext.SeccompProfile.Type != corev1.SeccompProfileTypeRuntimeDefault {
t.Errorf("pod security context = %+v", ps.SecurityContext)
}
if ps.AutomountServiceAccountToken == nil || *ps.AutomountServiceAccountToken {
t.Error("the binder needs no API token")
}
c := ps.Containers[0]
if got := append(append([]string{}, c.Command...), c.Args...); !containsSeq(got, []string{felisBinaryPath, "version"}) {
t.Errorf("command = %v", got)
}
if sc := c.SecurityContext; sc == nil || sc.AllowPrivilegeEscalation == nil || *sc.AllowPrivilegeEscalation ||
sc.Capabilities == nil || len(sc.Capabilities.Drop) == 0 || sc.Capabilities.Drop[0] != "ALL" {
t.Errorf("container security context = %+v", c.SecurityContext)
}
if len(ps.Volumes) != 1 || ps.Volumes[0].PersistentVolumeClaim == nil || ps.Volumes[0].PersistentVolumeClaim.ClaimName != "felis-backups" {
t.Errorf("volumes = %+v", ps.Volumes)
}
}
// TestAPIDeployment_Wiring pins the api entrypoint, the credential plumbing, and
// the FELIS_IMAGE passthrough.
func TestAPIDeployment_Wiring(t *testing.T) {
+8 -8
View File
@@ -49,18 +49,18 @@ func (s *PGStore) ListActiveServers(ctx context.Context) ([]Candidate, error) {
return out, rows.Err()
}
func (s *PGStore) FreshBackup(ctx context.Context, server string, since time.Time) (string, bool, error) {
const q = `SELECT backup_ref FROM world_backups
func (s *PGStore) FreshBackup(ctx context.Context, server string, since time.Time) (Fresh, bool, error) {
const q = `SELECT backup_ref, offsite_at IS NOT NULL FROM world_backups
WHERE server_name = $1 AND status = 'present' AND created_at >= $2
ORDER BY created_at DESC LIMIT 1`
var ref string
switch err := s.db.QueryRowContext(ctx, q, server, since).Scan(&ref); {
ORDER BY offsite_at IS NOT NULL DESC, created_at DESC LIMIT 1`
var f Fresh
switch err := s.db.QueryRowContext(ctx, q, server, since).Scan(&f.Ref, &f.Offsite); {
case err == sql.ErrNoRows:
return "", false, nil
return Fresh{}, false, nil
case err != nil:
return "", false, err
return Fresh{}, false, err
}
return ref, true, nil
return f, true, nil
}
func (s *PGStore) InsertBackup(ctx context.Context, rec BackupRecord) error {
+32 -6
View File
@@ -81,6 +81,12 @@ type Config struct {
WarnBefore []time.Duration // §24: warn these long before the deadline (default 3d, 1d)
Retention time.Duration // §18: keep a backup this long after deletion (default 3mo≈90d)
MaxLocalBytes int64 // §26: backup store soft cap; 0 = unlimited
// RequireOffsite holds each deletion until the world's archive has its
// off-site copy ([offsite] configured; internal/offsite records the copy).
// The archive is written on the run that finds the world idle, and the
// world is deleted on the first run after the copy lands, normally the
// next day.
RequireOffsite bool
}
// DefaultConfig is the spec's §24 default window set.
@@ -131,6 +137,13 @@ type BackupRecord struct {
ExpiresAt time.Time
}
// Fresh is the backup FreshBackup found.
type Fresh struct {
Ref string
// Offsite reports that the archive has its off-site copy (offsite_at).
Offsite bool
}
// StoredBackup is an existing world_backups row, used by both the expiry pass
// and the capacity-eviction path.
type StoredBackup struct {
@@ -156,10 +169,11 @@ type Store interface {
ListActiveServers(ctx context.Context) ([]Candidate, error)
// FreshBackup reports an existing present backup for server whose world is
// still current — created at or after since (the world's last_active_at).
// It makes a reap idempotent across a DeletePVC failure: the retry reuses
// the archive instead of writing a duplicate.
FreshBackup(ctx context.Context, server string, since time.Time) (ref string, ok bool, err error)
// still current — created at or after since (the world's last_active_at),
// preferring one already copied off-site. It makes a reap idempotent across
// a DeletePVC failure, and across the wait for the off-site copy: the retry
// reuses the archive instead of writing a duplicate.
FreshBackup(ctx context.Context, server string, since time.Time) (b Fresh, ok bool, err error)
// InsertBackup records a world_backups row (status=present).
InsertBackup(ctx context.Context, rec BackupRecord) error
@@ -232,6 +246,9 @@ type Summary struct {
Skipped int // exempt, CRD gone, or could not back up
EvictedEarly int
BackupsExpired int
// AwaitingOffsite are idle worlds that are archived and kept until the
// archive's off-site copy lands.
AwaitingOffsite int
}
func (r *Reaper) now() time.Time {
@@ -336,10 +353,11 @@ func (r *Reaper) reap(ctx context.Context, now time.Time, c Candidate, crd Serve
// world but failed before deleting the PVC, reuse that backup rather than
// writing a duplicate. The world has not changed since last_active_at, so
// any present backup created after it still describes the current world.
ref, ok, err := r.Store.FreshBackup(ctx, c.Name, c.LastActiveAt)
fresh, ok, err := r.Store.FreshBackup(ctx, c.Name, c.LastActiveAt)
if err != nil {
return fmt.Errorf("lookup fresh backup: %w", err)
}
ref, offsite := fresh.Ref, fresh.Offsite
if !ok {
aref, size, err := r.Archiver.Archive(ctx, c.Name, crd.PVC)
if err != nil {
@@ -364,7 +382,15 @@ func (r *Reaper) reap(ctx context.Context, now time.Time, c Candidate, crd Serve
}
return fmt.Errorf("insert backup: %w", err)
}
ref = string(aref)
ref, offsite = string(aref), false
}
// With an off-site bucket configured, the archive on this node's disk is
// not enough on its own: a lost disk would take it along with the world.
if r.Cfg.RequireOffsite && !offsite {
sum.AwaitingOffsite++
r.log().Info("reaper: world archived, kept until the archive's off-site copy is confirmed", "server", c.Name, "backup_ref", ref)
return nil
}
// World is safely archived and recorded — now (and only now) delete it.
+42 -3
View File
@@ -107,6 +107,7 @@ type fakeBackup struct {
size int64
status string // present | deleted
createdAt, expires time.Time
offsite bool
}
type fakeStore struct {
@@ -133,13 +134,19 @@ func (s *fakeStore) ListActiveServers(context.Context) ([]Candidate, error) {
return out, nil
}
func (s *fakeStore) FreshBackup(_ context.Context, server string, since time.Time) (string, bool, error) {
func (s *fakeStore) FreshBackup(_ context.Context, server string, since time.Time) (Fresh, bool, error) {
var found *fakeBackup
for _, b := range s.backups {
if b.server == server && b.status == "present" && !b.createdAt.Before(since) {
return b.ref, true, nil
if found == nil || (b.offsite && !found.offsite) {
found = b
}
}
return "", false, nil
}
if found == nil {
return Fresh{}, false, nil
}
return Fresh{Ref: found.ref, Offsite: found.offsite}, true, nil
}
func (s *fakeStore) InsertBackup(_ context.Context, rec BackupRecord) error {
@@ -426,6 +433,38 @@ func TestReapDeletePVCFailureIsIdempotent(t *testing.T) {
}
}
// With an off-site bucket configured the reaper archives an idle world but
// keeps it until the archive's copy is confirmed; the next run reuses that
// archive and deletes the world, without archiving it again.
func TestReapWaitsForOffsiteCopy(t *testing.T) {
cfg := DefaultConfig()
cfg.RequireOffsite = true
r, st, cl, ar := newReaper(cfg,
Candidate{Name: "echo", OwnerID: "user-5", LastActiveAt: idleBy(20 * Day)})
sum := mustRun(t, r)
if sum.WorldsReaped != 0 || sum.AwaitingOffsite != 1 || sum.Skipped != 0 {
t.Fatalf("run1 summary = %+v, want archived and awaiting the copy", sum)
}
if ar.archives != 1 || len(st.backups) != 1 || cl.deletePVCCalls != 0 {
t.Fatalf("run1: archives=%d backups=%d deletes=%d, want 1/1/0", ar.archives, len(st.backups), cl.deletePVCCalls)
}
// The copy has not landed yet: still kept, still one archive.
if sum := mustRun(t, r); sum.AwaitingOffsite != 1 || ar.archives != 1 || cl.deletePVCCalls != 0 {
t.Fatalf("run2 = %+v archives=%d deletes=%d, want the world still kept", sum, ar.archives, cl.deletePVCCalls)
}
st.backups[0].offsite = true
sum = mustRun(t, r)
if sum.WorldsReaped != 1 || sum.AwaitingOffsite != 0 {
t.Fatalf("run3 summary = %+v, want reaped", sum)
}
if ar.archives != 1 || len(cl.deletedPVCs) != 1 {
t.Fatalf("run3: archives=%d deleted=%v, want the copied archive reused and the PVC gone", ar.archives, cl.deletedPVCs)
}
}
// Red line ⑤ (reap arm): an unowned server is still reaped on time; the audit
// records an empty former_owner.
func TestReapUnownedServerStillReaped(t *testing.T) {
@@ -0,0 +1,11 @@
-- Off-site copies of world archives. offsite_at is when `felis offsite sync`
-- (felis-offsite.timer on the host) confirmed the archive's encrypted copy in
-- the [offsite] bucket; NULL means the archive exists on this node only. With
-- [offsite] configured the reaper deletes an idle world only once the archive
-- it made has an off-site copy, so losing the node's disk cannot take a reaped
-- world with it.
ALTER TABLE world_backups ADD COLUMN offsite_at timestamptz;
-- The sync's work list: present archives still waiting for their copy.
CREATE INDEX world_backups_offsite_pending_idx ON world_backups (created_at)
WHERE status = 'present' AND offsite_at IS NULL;
+40 -1
View File
@@ -13,6 +13,7 @@ import (
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/dbbackup"
"felis.lolicon.best/internal/naming"
"felis.lolicon.best/internal/offsite"
"felis.lolicon.best/internal/platform"
appsv1 "k8s.io/api/apps/v1"
batchv1 "k8s.io/api/batch/v1"
@@ -333,6 +334,43 @@ func BackupFinding(dir string, now time.Time) *Finding {
return f
}
// OffsiteFinding reports an off-site copy that has not completed a clean run
// within offsite.StaleAfter, going by the record `felis offsite sync` leaves
// in statusFile. It is a warning: the local copies are intact, but a lost
// node would now lose what the bucket lacks, and the reaper keeps idle worlds
// on disk until their archives reach the bucket.
func OffsiteFinding(statusFile string, now time.Time) *Finding {
const hint = "journalctl -u felis-offsite -n 50; `sudo felis offsite status` (docs/troubleshooting.md §16)"
st, err := offsite.ReadStatus(statusFile)
if err != nil {
return &Finding{
Key: "offsite", Severity: Warning, For: backupFor,
Summary: fmt.Sprintf("无法读取异地备份状态 %s", statusFile),
SummaryEN: fmt.Sprintf("cannot read the off-site copy status %s: %v", statusFile, err),
Hint: hint,
}
}
if st != nil && !st.LastSuccess.IsZero() && now.Sub(st.LastSuccess) <= offsite.StaleAfter {
return nil
}
f := &Finding{
Key: "offsite", Severity: Warning, For: backupFor,
Summary: "异地备份从未成功同步过,世界归档与数据库备份只在本机",
SummaryEN: "the off-site copy has never completed; world archives and database bundles exist on this machine only",
Hint: hint,
}
if st != nil && !st.LastSuccess.IsZero() {
age := roundHours(now.Sub(st.LastSuccess))
f.Summary = fmt.Sprintf("异地备份已有 %s 没有成功同步", age)
f.SummaryEN = fmt.Sprintf("the off-site copy last completed %s ago", age)
}
if st != nil && st.LastError != "" {
f.Summary += "(最近一次错误:" + st.LastError + ")"
f.SummaryEN += " (last error: " + st.LastError + ")"
}
return f
}
// DiskFindings reports each filesystem under paths that is running out of
// space. Paths on one filesystem are reported once, under the first of them; a
// path that does not exist is skipped (a feature that is not in use).
@@ -407,8 +445,9 @@ func MemoryFinding(meminfo string) *Finding {
}
}
// roundHours prints a duration to the hour, "35h" rather than "35h0m0s".
func roundHours(d time.Duration) string {
return d.Round(time.Hour).String()
return strings.TrimSuffix(d.Round(time.Hour).String(), "0m0s")
}
func humanBytes(b uint64) string {
+26 -1
View File
@@ -12,6 +12,7 @@ import (
"felis.lolicon.best/internal/apis/felis/v1alpha1"
"felis.lolicon.best/internal/dbbackup"
"felis.lolicon.best/internal/offsite"
appsv1 "k8s.io/api/apps/v1"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
@@ -146,7 +147,7 @@ func TestBackupFinding(t *testing.T) {
}
}
touch(t0.Add(-30 * time.Hour))
if f := BackupFinding(dir, t0); f == nil || !strings.Contains(f.SummaryEN, "30h0m0s old") {
if f := BackupFinding(dir, t0); f == nil || !strings.Contains(f.SummaryEN, "30h old") {
t.Fatalf("stale: %+v", f)
}
touch(t0.Add(-2 * time.Hour))
@@ -155,6 +156,30 @@ func TestBackupFinding(t *testing.T) {
}
}
func TestOffsiteFinding(t *testing.T) {
path := filepath.Join(t.TempDir(), "offsite", "status.json")
if f := OffsiteFinding(path, t0); f == nil || !strings.Contains(f.SummaryEN, "never completed") {
t.Fatalf("no status: %+v", f)
}
write := func(st offsite.Status) {
if err := offsite.WriteStatus(path, st); err != nil {
t.Fatal(err)
}
}
write(offsite.Status{LastAttempt: t0.Add(-time.Hour), LastError: "bucket unreachable"})
if f := OffsiteFinding(path, t0); f == nil || !strings.Contains(f.SummaryEN, "machine only (last error: bucket unreachable)") {
t.Fatalf("failing from the start: %+v", f)
}
write(offsite.Status{LastAttempt: t0.Add(-time.Hour), LastSuccess: t0.Add(-13 * time.Hour), LastError: "access denied"})
if f := OffsiteFinding(path, t0); f == nil || f.Severity != Warning || !strings.Contains(f.SummaryEN, "13h ago (last error: access denied)") {
t.Fatalf("stale: %+v", f)
}
write(offsite.Status{LastAttempt: t0.Add(-time.Hour), LastSuccess: t0.Add(-2 * time.Hour), LastError: "one bundle failed"})
if f := OffsiteFinding(path, t0); f != nil {
t.Fatalf("a success within %s reported: %+v", offsite.StaleAfter, f)
}
}
func TestMemoryFinding(t *testing.T) {
path := filepath.Join(t.TempDir(), "meminfo")
write := func(avail int) {