fix(offsite): 复制新归档后补打并上传 DB 包,桶内无记录的归档过最长保留期后清理
This commit is contained in:
10 files changed
+669
-28
No files matched your search
+2
-1
@@ -36,6 +36,7 @@ var defaultKeep = map[string]int{
|
|||||||
dbbackup.LabelDaily: 14,
|
dbbackup.LabelDaily: 14,
|
||||||
dbbackup.LabelPreMigrate: 10,
|
dbbackup.LabelPreMigrate: 10,
|
||||||
dbbackup.LabelPreRestore: 5,
|
dbbackup.LabelPreRestore: 5,
|
||||||
|
dbbackup.LabelOffsite: 1,
|
||||||
}
|
}
|
||||||
|
|
||||||
// cmdDB implements `felis db`: logical backups of the control-plane database
|
// cmdDB implements `felis db`: logical backups of the control-plane database
|
||||||
@@ -136,7 +137,7 @@ func libpqQuote(v string) string {
|
|||||||
func dbBackup(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Writer) int {
|
func dbBackup(fs *flag.FlagSet, dir *string, args []string, stdout, stderr io.Writer) int {
|
||||||
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
cfgPath := fs.String("config", "/etc/felis/felis.toml", "path to felis.toml")
|
||||||
label := fs.String("label", dbbackup.LabelManual, "bundle label; daily/pre-migrate/pre-restore bundles are pruned, manual ones never")
|
label := fs.String("label", dbbackup.LabelManual, "bundle label; daily/pre-migrate/pre-restore bundles are pruned, manual ones never")
|
||||||
keep := fs.Int("keep", -1, "bundles of this label to keep (default: daily 14, pre-migrate 10, pre-restore 5, manual all)")
|
keep := fs.Int("keep", -1, "bundles of this label to keep (default: daily 14, pre-migrate 10, pre-restore 5, offsite 1, manual all)")
|
||||||
stateDir := fs.String("state-dir", dbbackup.DefaultStateDir, `host state directory to bundle ("" for none)`)
|
stateDir := fs.String("state-dir", dbbackup.DefaultStateDir, `host state directory to bundle ("" for none)`)
|
||||||
noServers := fs.Bool("no-servers", false, "leave the MinecraftServer objects out of the bundle")
|
noServers := fs.Bool("no-servers", false, "leave the MinecraftServer objects out of the bundle")
|
||||||
metrics := fs.String("metrics-file", "", "node-exporter textfile to rewrite on success (e.g. /var/lib/node_exporter/textfile_collector/felis_db_backup.prom)")
|
metrics := fs.String("metrics-file", "", "node-exporter textfile to rewrite on success (e.g. /var/lib/node_exporter/textfile_collector/felis_db_backup.prom)")
|
||||||
|
|||||||
+55
-9
@@ -211,6 +211,7 @@ func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int
|
|||||||
archiveDir := fs.String("archive-dir", "", "host directory of the world archive volume (default: resolved from the backup PVC through the cluster)")
|
archiveDir := fs.String("archive-dir", "", "host directory of the world archive volume (default: resolved from the backup PVC through the cluster)")
|
||||||
backupPVC := fs.String("backup-pvc", "felis-backups", `the world archive PVC, in the [k8s] namespace ("" when backups are off)`)
|
backupPVC := fs.String("backup-pvc", "felis-backups", `the world archive PVC, in the [k8s] namespace ("" when backups are off)`)
|
||||||
dbDir := fs.String("db-dir", dbbackup.DefaultDir, `database bundle directory ("" copies no bundles)`)
|
dbDir := fs.String("db-dir", dbbackup.DefaultDir, `database bundle directory ("" copies no bundles)`)
|
||||||
|
stateDir := fs.String("state-dir", dbbackup.DefaultStateDir, `host state directory bundled into the database bundle taken after archives are copied ("" for none)`)
|
||||||
registry := fs.String("registry", "", `host[:port] of the registry whose user images are copied (default: the in-cluster registry's loopback hostPort; "off" copies none)`)
|
registry := fs.String("registry", "", `host[:port] of the registry whose user images are copied (default: the in-cluster registry's loopback hostPort; "off" copies none)`)
|
||||||
uploadsDir := fs.String("uploads-dir", "", "host directory of the submission uploads volume (default: resolved from the uploads PVC through the cluster)")
|
uploadsDir := fs.String("uploads-dir", "", "host directory of the submission uploads volume (default: resolved from the uploads PVC through the cluster)")
|
||||||
uploadsPVC := fs.String("uploads-pvc", platform.UploadsPVCName, `the submission uploads PVC, in the control-plane namespace ("" copies no uploads)`)
|
uploadsPVC := fs.String("uploads-pvc", platform.UploadsPVCName, `the submission uploads PVC, in the control-plane namespace ("" copies no uploads)`)
|
||||||
@@ -225,7 +226,7 @@ func offsiteSync(fs *flag.FlagSet, args []string, stdout, stderr io.Writer) int
|
|||||||
}
|
}
|
||||||
st, lease := startRun(env.cfg, env.key, *statusFile, time.Now())
|
st, lease := startRun(env.cfg, env.key, *statusFile, time.Now())
|
||||||
res, err := runOffsiteSync(cfg, env, offsiteSources{
|
res, err := runOffsiteSync(cfg, env, offsiteSources{
|
||||||
archiveDir: *archiveDir, backupPVC: *backupPVC, dbDir: *dbDir,
|
archiveDir: *archiveDir, backupPVC: *backupPVC, dbDir: *dbDir, stateDir: *stateDir,
|
||||||
registry: offsiteRegistryEndpoint(*registry, cfg.Registry),
|
registry: offsiteRegistryEndpoint(*registry, cfg.Registry),
|
||||||
uploadsDir: *uploadsDir, uploadsPVC: *uploadsPVC,
|
uploadsDir: *uploadsDir, uploadsPVC: *uploadsPVC,
|
||||||
}, &lease, stderr)
|
}, &lease, stderr)
|
||||||
@@ -299,12 +300,13 @@ func recordRun(st *offsite.Status, res offsite.Result, err error, lease offsite.
|
|||||||
}
|
}
|
||||||
|
|
||||||
// offsiteSources is where one sync pass reads from: the world archive volume
|
// offsiteSources is where one sync pass reads from: the world archive volume
|
||||||
// (archiveDir, or the backupPVC's directory), the bundle directory, the
|
// (archiveDir, or the backupPVC's directory), the bundle directory (with the
|
||||||
// registry's loopback endpoint and the uploads volume (uploadsDir, or the
|
// host state the pass bundles, stateDir), the registry's loopback endpoint and
|
||||||
// uploadsPVC's directory). An empty source is skipped.
|
// the uploads volume (uploadsDir, or the uploadsPVC's directory). An empty
|
||||||
|
// source is skipped.
|
||||||
type offsiteSources struct {
|
type offsiteSources struct {
|
||||||
archiveDir, backupPVC string
|
archiveDir, backupPVC string
|
||||||
dbDir string
|
dbDir, stateDir string
|
||||||
registry string
|
registry string
|
||||||
uploadsDir, uploadsPVC string
|
uploadsDir, uploadsPVC string
|
||||||
}
|
}
|
||||||
@@ -346,10 +348,8 @@ func runOffsiteSync(cfg *config.Config, env *offsiteEnv, src offsiteSources, lea
|
|||||||
return offsite.Result{}, fmt.Errorf("open database: %w", err)
|
return offsite.Result{}, fmt.Errorf("open database: %w", err)
|
||||||
}
|
}
|
||||||
defer drv.Close()
|
defer drv.Close()
|
||||||
s := &offsite.Syncer{
|
s := offsiteSyncer(cfg, env, src, archiveDir, uploadsDir, lease, log)
|
||||||
Bucket: env.bucket, Catalog: offsite.PGCatalog{DB: drv.DB()}, Key: env.key,
|
s.Catalog = offsite.PGCatalog{DB: drv.DB()}
|
||||||
ArchiveDir: archiveDir, DBDir: src.dbDir, DBKeep: env.cfg.DBKeep, UploadsDir: uploadsDir, Lease: lease, Log: log,
|
|
||||||
}
|
|
||||||
if src.registry != "" {
|
if src.registry != "" {
|
||||||
s.Images = newRegistryImages(src.registry)
|
s.Images = newRegistryImages(src.registry)
|
||||||
s.ImagePins = imagePins(drv.DB(), cfg.Registry.URL)
|
s.ImagePins = imagePins(drv.DB(), cfg.Registry.URL)
|
||||||
@@ -357,6 +357,52 @@ func runOffsiteSync(cfg *config.Config, env *offsiteEnv, src offsiteSources, lea
|
|||||||
return s.Run(ctx)
|
return s.Run(ctx)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// offsiteSyncer is the pass runOffsiteSync runs over the resolved archive and
|
||||||
|
// uploads directories, before its catalog and registry are attached. It
|
||||||
|
// snapshots the database into the bundle directory after copying archives, and
|
||||||
|
// sweeps world objects no backup records once they outlive every retention in
|
||||||
|
// [archive]; a retention that does not parse sweeps none.
|
||||||
|
func offsiteSyncer(cfg *config.Config, env *offsiteEnv, src offsiteSources, archiveDir, uploadsDir string, lease *offsite.Lease, log io.Writer) *offsite.Syncer {
|
||||||
|
s := &offsite.Syncer{
|
||||||
|
Bucket: env.bucket, Key: env.key,
|
||||||
|
ArchiveDir: archiveDir, DBDir: src.dbDir, DBKeep: env.cfg.DBKeep, UploadsDir: uploadsDir, Lease: lease, Log: log,
|
||||||
|
}
|
||||||
|
if src.dbDir != "" {
|
||||||
|
s.Snapshot = offsiteSnapshot(cfg.Database, src.dbDir, src.stateDir, log)
|
||||||
|
}
|
||||||
|
if rc, err := reaperConfig(cfg); err != nil {
|
||||||
|
fmt.Fprintf(log, "felis offsite: world objects no backup records are kept: %v\n", err)
|
||||||
|
} else {
|
||||||
|
s.OrphanAfter = max(rc.Retention, rc.ManualRetention, rc.ScheduledRetention)
|
||||||
|
}
|
||||||
|
return s
|
||||||
|
}
|
||||||
|
|
||||||
|
// offsiteSnapshot takes the bundle a pass sends after copying world archives
|
||||||
|
// (offsite.Syncer.Snapshot): what `felis db backup` takes, labelled offsite,
|
||||||
|
// with the newest one kept in dir. It is not recorded for the panel, whose
|
||||||
|
// backup card watches felis-db-backup.timer: snapshots come only when archives
|
||||||
|
// are copied, and would hide a daily timer that stopped.
|
||||||
|
func offsiteSnapshot(db config.DatabaseConfig, dir, stateDir string, log io.Writer) func(context.Context) error {
|
||||||
|
return func(ctx context.Context) error {
|
||||||
|
tools, err := dbTools(db)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
ctx, cancel := context.WithTimeout(ctx, 30*time.Minute)
|
||||||
|
defer cancel()
|
||||||
|
path, err := dbbackup.Backup(ctx, dbbackup.BackupOptions{
|
||||||
|
DatabaseURL: db.URL, Tools: tools, Dir: dir, Label: dbbackup.LabelOffsite,
|
||||||
|
Keep: defaultKeep[dbbackup.LabelOffsite], StateDir: stateDir, Version: resolvedVersion(),
|
||||||
|
ExportServers: exportMinecraftServers, Log: log,
|
||||||
|
})
|
||||||
|
if err == nil {
|
||||||
|
fmt.Fprintf(log, "felis offsite: took database bundle %s, which lists the archives just copied\n", filepath.Base(path))
|
||||||
|
}
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// volumeKind names a PVC the off-site copy reads or restores, for messages,
|
// volumeKind names a PVC the off-site copy reads or restores, for messages,
|
||||||
// with the flag that bypasses finding it through the cluster.
|
// with the flag that bypasses finding it through the cluster.
|
||||||
type volumeKind struct{ what, dirFlag, empty string }
|
type volumeKind struct{ what, dirFlag, empty string }
|
||||||
|
|||||||
@@ -673,3 +673,67 @@ func TestRestoredHostKeepsStandingBy(t *testing.T) {
|
|||||||
t.Errorf("upgraded host's status = %+v", st)
|
t.Errorf("upgraded host's status = %+v", st)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestOffsiteSyncerSnapshotsAndSweeps: the pass `offsite sync` runs takes its
|
||||||
|
// snapshot the way `felis db backup` does, in the database pod, into the
|
||||||
|
// bundle directory, keeping the newest one there; and it sweeps unrecorded
|
||||||
|
// world objects only past the longest retention [archive] gives any backup.
|
||||||
|
func TestOffsiteSyncerSnapshotsAndSweeps(t *testing.T) {
|
||||||
|
dir := newPodRig(t)
|
||||||
|
bundles := filepath.Join(dir, "bundles")
|
||||||
|
env := &offsiteEnv{cfg: config.OffsiteConfig{DBKeep: 5}}
|
||||||
|
var log bytes.Buffer
|
||||||
|
cfg := &config.Config{Database: podDB, Archive: config.ArchiveConfig{Retention: "120d"}}
|
||||||
|
s := offsiteSyncer(cfg, env, offsiteSources{dbDir: bundles}, "/archives", "/uploads", nil, &log)
|
||||||
|
if s.DBDir != bundles || s.DBKeep != 5 || s.ArchiveDir != "/archives" || s.UploadsDir != "/uploads" {
|
||||||
|
t.Fatalf("syncer = %+v", s)
|
||||||
|
}
|
||||||
|
if s.OrphanAfter != 120*24*time.Hour {
|
||||||
|
t.Errorf("OrphanAfter = %s, want the 120d retention", s.OrphanAfter)
|
||||||
|
}
|
||||||
|
if s.Snapshot == nil {
|
||||||
|
t.Fatal("the pass takes no snapshot after copying archives")
|
||||||
|
}
|
||||||
|
for i := 0; i < 2; i++ {
|
||||||
|
if err := s.Snapshot(context.Background()); err != nil {
|
||||||
|
t.Fatalf("snapshot %d: %v", i, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
got, err := dbbackup.List(bundles)
|
||||||
|
if err != nil || len(got) != 1 || got[0].Label != dbbackup.LabelOffsite {
|
||||||
|
t.Fatalf("bundle directory = %+v, %v; want the newest offsite bundle alone", got, err)
|
||||||
|
}
|
||||||
|
if _, err := dbbackupVerify(got[0].Path); err != nil {
|
||||||
|
t.Fatalf("the snapshot does not verify: %v", err)
|
||||||
|
}
|
||||||
|
// The MinecraftServer objects are exported alongside, as in the daily bundle.
|
||||||
|
argv, _ := os.ReadFile(filepath.Join(dir, "k3s.args"))
|
||||||
|
if ran := string(argv); !strings.Contains(ran, podExecPrefix+"pg_dump --format=custom") || !strings.Contains(ran, "kubectl get minecraftservers") {
|
||||||
|
t.Errorf("k3s ran %q, want pg_dump in the pod and the server export", ran)
|
||||||
|
}
|
||||||
|
if !strings.Contains(log.String(), "took database bundle "+got[0].Name) {
|
||||||
|
t.Errorf("the snapshot is not logged:\n%s", log.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, c := range []struct {
|
||||||
|
archive config.ArchiveConfig
|
||||||
|
want time.Duration
|
||||||
|
}{
|
||||||
|
{config.ArchiveConfig{}, 90 * 24 * time.Hour},
|
||||||
|
{config.ArchiveConfig{ScheduledRetention: "200d"}, 200 * 24 * time.Hour},
|
||||||
|
{config.ArchiveConfig{ManualRetention: "150d", Retention: "30d", ScheduledRetention: "60d"}, 150 * 24 * time.Hour},
|
||||||
|
} {
|
||||||
|
s := offsiteSyncer(&config.Config{Database: podDB, Archive: c.archive}, env, offsiteSources{dbDir: bundles}, "", "", nil, io.Discard)
|
||||||
|
if s.OrphanAfter != c.want {
|
||||||
|
t.Errorf("%+v: OrphanAfter = %s, want %s", c.archive, s.OrphanAfter, c.want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
log.Reset()
|
||||||
|
s = offsiteSyncer(&config.Config{Database: podDB, Archive: config.ArchiveConfig{Retention: "soon"}}, env, offsiteSources{}, "", "", nil, &log)
|
||||||
|
if s.OrphanAfter != 0 || !strings.Contains(log.String(), "world objects no backup records are kept") {
|
||||||
|
t.Errorf("a retention that does not parse: OrphanAfter %s, log %q; want no sweep, said", s.OrphanAfter, log.String())
|
||||||
|
}
|
||||||
|
if s.Snapshot != nil {
|
||||||
|
t.Error("a pass that copies no bundles takes a snapshot")
|
||||||
|
}
|
||||||
|
}
|
||||||
+27
-7
@@ -2024,8 +2024,8 @@ along). One bundle is `felis-db-<UTC stamp>-<label>.tar`:
|
|||||||
|
|
||||||
next to a `.sha256` sidecar in `sha256sum` format. **A bundle contains the
|
next to a `.sha256` sidecar in `sha256sum` format. **A bundle contains the
|
||||||
secrets; treat it like `/etc/felis` itself.** Retention per label: `daily` 14
|
secrets; treat it like `/etc/felis` itself.** Retention per label: `daily` 14
|
||||||
(`FELIS_DB_BACKUP_KEEP`), `pre-migrate` 10, `pre-restore` 5, `manual` never
|
(`FELIS_DB_BACKUP_KEEP`), `pre-migrate` 10, `pre-restore` 5, `offsite` 1
|
||||||
pruned.
|
(taken by the off-site copy, §16), `manual` never pruned.
|
||||||
|
|
||||||
Installer knobs: `FELIS_DB_BACKUP_DIR`, `FELIS_DB_BACKUP_KEEP`,
|
Installer knobs: `FELIS_DB_BACKUP_DIR`, `FELIS_DB_BACKUP_KEEP`,
|
||||||
`FELIS_DB_BACKUP_TIME`, `FELIS_DB_BACKUP_METRICS` and
|
`FELIS_DB_BACKUP_TIME`, `FELIS_DB_BACKUP_METRICS` and
|
||||||
@@ -2175,10 +2175,10 @@ off-site bucket (next sections) plus a fresh install. What the host holds:
|
|||||||
|
|
||||||
| Data | On the host | In the bucket | Brought back by | Lost at most |
|
| Data | On the host | In the bucket | Brought back by | Lost at most |
|
||||||
|---|---|---|---|---|
|
|---|---|---|---|---|
|
||||||
| Control-plane database (accounts, passkeys, ownership, quotas, audit, submissions, the `world_backups` index) | felis-postgres, `/var/lib/felis/postgres` | every bundle, copied within the hour of being written | `fetch-db`, `db restore` | changes since the newest bundle: up to a day plus an hour with the daily timer |
|
| Control-plane database (accounts, passkeys, ownership, quotas, audit, submissions, the `world_backups` index) | felis-postgres, `/var/lib/felis/postgres` | every bundle, copied within the hour of being written, plus a fresh one after every pass that copied a world archive | `fetch-db`, `db restore` | changes since the newest bundle: up to a day plus an hour with the daily timer |
|
||||||
| Host state (`/etc/felis`: secrets, both `felis.toml` copies, `offsite.env`, the mail relay password and uploads bucket keys, panel TLS pair) | `/etc/felis` | inside every bundle | `tar -x` of the bundle's `state/` | as the database |
|
| Host state (`/etc/felis`: secrets, both `felis.toml` copies, `offsite.env`, the mail relay password and uploads bucket keys, panel TLS pair) | `/etc/felis` | inside every bundle | `tar -x` of the bundle's `state/` | as the database |
|
||||||
| MinecraftServer objects | k3s | inside every bundle (`k8s/minecraftservers.json`) | `kubectl apply` | as the database |
|
| MinecraftServer objects | k3s | inside every bundle (`k8s/minecraftservers.json`) | `kubectl apply` | as the database |
|
||||||
| World archives (reaper, "Back up now", pre-restore snapshots) | `felis-backups` volume | each one within the hour | `fetch-worlds` | archives written in the last hour |
|
| World archives (reaper, "Back up now", pre-restore snapshots) | `felis-backups` volume | each one within the hour, followed by a bundle listing it | `fetch-worlds` | archives written in the last hour |
|
||||||
| Live worlds | `world-*` volumes under `/var/lib/rancher/k3s/storage` | **only as their archives** | a restore from the newest archive (§10) | everything since that world's newest archive |
|
| Live worlds | `world-*` volumes under `/var/lib/rancher/k3s/storage` | **only as their archives** | a restore from the newest archive (§10) | everything since that world's newest archive |
|
||||||
| User images | `registry` volume | hourly; image lists kept 14 days | `fetch-images` | images pushed in the last hour |
|
| User images | `registry` volume | hourly; image lists kept 14 days | `fetch-images` | images pushed in the last hour |
|
||||||
| Submission uploads (modpacks awaiting or past review) | `felis-uploads` volume | hourly; upload lists kept 14 days | `fetch-uploads` | uploads of the last hour |
|
| Submission uploads (modpacks awaiting or past review) | `felis-uploads` volume | hourly; upload lists kept 14 days | `fetch-uploads` | uploads of the last hour |
|
||||||
@@ -2313,7 +2313,9 @@ host yourself, plus the off-site encryption key if the copy is in the bucket.
|
|||||||
and the volume lacks, provisioning the `felis-backups` volume first if
|
and the volume lacks, provisioning the `felis-backups` volume first if
|
||||||
nothing has used it yet (a short-lived `felis-bind-felis-backups-*` pod). It
|
nothing has used it yet (a short-lived `felis-bind-felis-backups-*` pod). It
|
||||||
lists any it could not find in the bucket. Restore a world from its archive
|
lists any it could not find in the bucket. Restore a world from its archive
|
||||||
as usual (§10, §13).
|
as usual (§10, §13). With the newest bundle restored, every archive in the
|
||||||
|
bucket is listed; an older bundle leaves the archives copied after it
|
||||||
|
unlisted, and the sync removes those once they pass the longest retention.
|
||||||
8. Make this host the one that writes the bucket, and send its first copy:
|
8. Make this host the one that writes the bucket, and send its first copy:
|
||||||
|
|
||||||
```
|
```
|
||||||
@@ -2418,6 +2420,24 @@ What runs:
|
|||||||
its retention (`expires_at`) has passed. An object already in the bucket at
|
its retention (`expires_at`) has passed. An object already in the bucket at
|
||||||
the right size is recorded without being sent again, so a run cut short
|
the right size is recorded without being sent again, so a run cut short
|
||||||
resumes. [GO-TESTED: `internal/offsite`]
|
resumes. [GO-TESTED: `internal/offsite`]
|
||||||
|
- A run that copied a world archive then takes a fresh `offsite` database
|
||||||
|
bundle (`felis-db-<stamp>-offsite.tar`, the same layout as a daily one, host
|
||||||
|
state from `-state-dir`, default `/etc/felis`) and sends it, so the newest
|
||||||
|
bundle in the bucket lists every archive there and a restore from it fetches
|
||||||
|
them all. The host keeps one such bundle locally, and the bucket keeps it
|
||||||
|
only while it is the newest; the `db_keep` count covers the other labels, so
|
||||||
|
a busy day of snapshots never pushes the dailies out. A run that copied
|
||||||
|
nothing, or whose newest bundle already postdates the last copy, takes none.
|
||||||
|
The panel's backup card keeps watching `felis-db-backup.timer` alone.
|
||||||
|
[GO-TESTED: `internal/offsite`, `TestOffsiteSyncerSnapshotsAndSweeps`]
|
||||||
|
[PG-TESTED]
|
||||||
|
- A world archive in the bucket that no row lists (the database came back
|
||||||
|
from a bundle older than the archive) is removed once it has been in the
|
||||||
|
bucket longer than the longest `[archive]` retention (`retention`,
|
||||||
|
`manual_retention`, `scheduled_retention`); younger ones stay, and the run
|
||||||
|
logs how many and when each goes. The sweep skips a database that lists no
|
||||||
|
archive at all, so a sync against a database not restored yet removes
|
||||||
|
nothing. [GO-TESTED: `internal/offsite`]
|
||||||
- The same run copies the user images in the platform registry: every
|
- The same run copies the user images in the platform registry: every
|
||||||
repository outside `felis/` and `mirror/`, each manifest the registry's index
|
repository outside `felis/` and `mirror/`, each manifest the registry's index
|
||||||
lists and every layer it names, read through the loopback hostPort. A layer
|
lists and every layer it names, read through the loopback hostPort. A layer
|
||||||
@@ -2450,8 +2470,8 @@ What runs:
|
|||||||
truncation, reordering and a wrong key are all refused on the way back.
|
truncation, reordering and a wrong key are all refused on the way back.
|
||||||
`felis-key-id` and `felis-writer` next to them hold the key's id and the
|
`felis-key-id` and `felis-writer` next to them hold the key's id and the
|
||||||
host writing the bucket in the clear.
|
host writing the bucket in the clear.
|
||||||
- A pass sends the database bundles first, then world archives, images and
|
- A pass sends the database bundles first, then world archives, the bundle
|
||||||
uploads. Each object has its own time limit: 10 minutes plus its size at
|
listing them, images and uploads. Each object has its own time limit: 10 minutes plus its size at
|
||||||
512 KiB/s (about 6 hours for 10 GiB). An archive the uplink cannot send in
|
512 KiB/s (about 6 hours for 10 GiB). An archive the uplink cannot send in
|
||||||
that time fails alone, stays pending and is tried again next pass; the rest
|
that time fails alone, stays pending and is tried again next pass; the rest
|
||||||
of the pass still goes. A pass over a big archive can run for hours; the
|
of the pass still goes. A pass over a big archive can run for hours; the
|
||||||
|
|||||||
@@ -62,6 +62,10 @@ const (
|
|||||||
LabelPreMigrate = "pre-migrate"
|
LabelPreMigrate = "pre-migrate"
|
||||||
LabelPreRestore = "pre-restore"
|
LabelPreRestore = "pre-restore"
|
||||||
LabelManual = "manual"
|
LabelManual = "manual"
|
||||||
|
// LabelOffsite is the bundle an off-site copy takes after copying world
|
||||||
|
// archives (internal/offsite), so the newest bundle off the machine lists
|
||||||
|
// them.
|
||||||
|
LabelOffsite = "offsite"
|
||||||
|
|
||||||
manifestEntry = "MANIFEST.json"
|
manifestEntry = "MANIFEST.json"
|
||||||
dumpEntry = "db.dump"
|
dumpEntry = "db.dump"
|
||||||
|
|||||||
@@ -240,6 +240,26 @@ func (c *fakeCatalog) ExpiredRefs(_ context.Context, now time.Time) ([]string, e
|
|||||||
return out, nil
|
return out, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (c *fakeCatalog) NewestOffsite(context.Context) (time.Time, error) {
|
||||||
|
var at time.Time
|
||||||
|
for _, r := range c.rows {
|
||||||
|
if r.offsite.After(at) {
|
||||||
|
at = r.offsite
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return at, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *fakeCatalog) KeptRefs(_ context.Context, now time.Time) ([]string, error) {
|
||||||
|
var out []string
|
||||||
|
for _, r := range c.rows {
|
||||||
|
if r.status == "present" || !r.expires.Before(now) {
|
||||||
|
out = append(out, r.Ref)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
func writeFile(t *testing.T, dir, name string, size int) []byte {
|
func writeFile(t *testing.T, dir, name string, size int) []byte {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
data := make([]byte, size)
|
data := make([]byte, size)
|
||||||
|
|||||||
@@ -43,8 +43,12 @@ func (c PGCatalog) MarkOffsite(ctx context.Context, id string, at time.Time) err
|
|||||||
}
|
}
|
||||||
|
|
||||||
func (c PGCatalog) ExpiredRefs(ctx context.Context, now time.Time) ([]string, error) {
|
func (c PGCatalog) ExpiredRefs(ctx context.Context, now time.Time) ([]string, error) {
|
||||||
rows, err := c.DB.QueryContext(ctx, `SELECT backup_ref FROM world_backups
|
return c.refs(ctx, `SELECT backup_ref FROM world_backups
|
||||||
WHERE status = 'deleted' AND expires_at < $1 AND offsite_at IS NOT NULL`, now)
|
WHERE status = 'deleted' AND expires_at < $1 AND offsite_at IS NOT NULL`, now)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c PGCatalog) refs(ctx context.Context, q string, args ...any) ([]string, error) {
|
||||||
|
rows, err := c.DB.QueryContext(ctx, q, args...)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
@@ -60,6 +64,16 @@ func (c PGCatalog) ExpiredRefs(ctx context.Context, now time.Time) ([]string, er
|
|||||||
return out, rows.Err()
|
return out, rows.Err()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func (c PGCatalog) NewestOffsite(ctx context.Context) (time.Time, error) {
|
||||||
|
var at sql.NullTime
|
||||||
|
err := c.DB.QueryRowContext(ctx, `SELECT max(offsite_at) FROM world_backups`).Scan(&at)
|
||||||
|
return at.Time, err
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c PGCatalog) KeptRefs(ctx context.Context, now time.Time) ([]string, error) {
|
||||||
|
return c.refs(ctx, `SELECT backup_ref FROM world_backups WHERE status = 'present' OR expires_at >= $1`, now)
|
||||||
|
}
|
||||||
|
|
||||||
// PendingCount is how many present archives wait for their copy, and since
|
// PendingCount is how many present archives wait for their copy, and since
|
||||||
// when the oldest has waited.
|
// when the oldest has waited.
|
||||||
func (c PGCatalog) PendingCount(ctx context.Context) (int, time.Time, error) {
|
func (c PGCatalog) PendingCount(ctx context.Context) (int, time.Time, error) {
|
||||||
|
|||||||
@@ -0,0 +1,286 @@
|
|||||||
|
package offsite
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"context"
|
||||||
|
"errors"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"slices"
|
||||||
|
"sort"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/dbbackup"
|
||||||
|
)
|
||||||
|
|
||||||
|
// snapshotter stands in for `felis db backup -label offsite`: each call writes
|
||||||
|
// a bundle stamped with the syncer's clock into its DBDir.
|
||||||
|
type snapshotter struct {
|
||||||
|
t *testing.T
|
||||||
|
s *Syncer
|
||||||
|
calls int
|
||||||
|
err error
|
||||||
|
}
|
||||||
|
|
||||||
|
func (sn *snapshotter) take(context.Context) error {
|
||||||
|
sn.calls++
|
||||||
|
if sn.err != nil {
|
||||||
|
return sn.err
|
||||||
|
}
|
||||||
|
name := dbbackup.BundleName(sn.s.now(), dbbackup.LabelOffsite)
|
||||||
|
data := dbBundle(sn.t, sn.s.now(), &dbbackup.Counts{Users: 3, Servers: 2}, 64)
|
||||||
|
return os.WriteFile(filepath.Join(sn.s.DBDir, name), data, 0o600)
|
||||||
|
}
|
||||||
|
|
||||||
|
func withSnapshot(t *testing.T, s *Syncer) *snapshotter {
|
||||||
|
sn := &snapshotter{t: t, s: s}
|
||||||
|
s.Snapshot = sn.take
|
||||||
|
return sn
|
||||||
|
}
|
||||||
|
|
||||||
|
func dbKeys(b *memBucket) []string {
|
||||||
|
var keys []string
|
||||||
|
for k := range b.objs {
|
||||||
|
if strings.HasPrefix(k, dbDir) {
|
||||||
|
keys = append(keys, strings.TrimSuffix(strings.TrimPrefix(k, dbDir), objExt))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
sort.Strings(keys)
|
||||||
|
return keys
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSyncSnapshotsAfterCopyingWorlds: a pass that copies an archive leaves a
|
||||||
|
// bundle in the bucket taken after the copy, which a restore takes as the
|
||||||
|
// latest and whose rows list the archive; the pass counts the bucket's bundles
|
||||||
|
// once, with that one newest. A pass that copies nothing takes none, and the
|
||||||
|
// next daily bundle replaces it.
|
||||||
|
func TestSyncSnapshotsAfterCopyingWorlds(t *testing.T) {
|
||||||
|
cat := &fakeCatalog{rows: []*row{
|
||||||
|
{WorldBackup: WorldBackup{ID: "b1", Server: "alpha", Ref: "/a/alpha-1.tar.gz"}, status: "present"},
|
||||||
|
}}
|
||||||
|
s, b := newSyncer(t, cat)
|
||||||
|
sn := withSnapshot(t, s)
|
||||||
|
writeFile(t, s.ArchiveDir, "alpha-1.tar.gz", 100)
|
||||||
|
daily := "felis-db-20260924T030000Z-daily.tar"
|
||||||
|
writeFile(t, s.DBDir, daily, 50)
|
||||||
|
|
||||||
|
res, err := s.Run(context.Background())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("run: %v (%+v)", err, res)
|
||||||
|
}
|
||||||
|
snap := dbbackup.BundleName(now, dbbackup.LabelOffsite)
|
||||||
|
if sn.calls != 1 {
|
||||||
|
t.Fatalf("snapshots taken = %d, want 1 after copying an archive", sn.calls)
|
||||||
|
}
|
||||||
|
if got := dbKeys(b); !slices.Equal(got, []string{daily, snap}) {
|
||||||
|
t.Fatalf("bucket bundles = %v, want the daily and the snapshot", got)
|
||||||
|
}
|
||||||
|
if res.NewestDB != snap || res.RemoteDB != 2 || res.DBUploaded != 2 || res.WorldsUploaded != 1 {
|
||||||
|
t.Fatalf("result = %+v, want the snapshot newest of 2 bundles, 2 bundles and 1 archive copied", res)
|
||||||
|
}
|
||||||
|
name, _, err := ChooseDB(context.Background(), b, s.Key)
|
||||||
|
if err != nil || name != snap {
|
||||||
|
t.Fatalf("fetch-db latest = %q, %v; want the snapshot", name, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if res, err := s.Run(context.Background()); err != nil || sn.calls != 1 || res.NewestDB != snap || res.RemoteDB != 2 {
|
||||||
|
t.Fatalf("a pass that copied nothing: snapshots %d, result %+v, err %v; want no new snapshot", sn.calls, res, err)
|
||||||
|
}
|
||||||
|
|
||||||
|
next := now.Add(15 * time.Hour)
|
||||||
|
s.Now = func() time.Time { return next }
|
||||||
|
nextDaily := dbbackup.BundleName(next.Add(-time.Minute), dbbackup.LabelDaily)
|
||||||
|
writeFile(t, s.DBDir, nextDaily, 50)
|
||||||
|
res, err = s.Run(context.Background())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
if got := dbKeys(b); !slices.Equal(got, []string{daily, nextDaily}) || sn.calls != 1 || res.DBPruned != 1 {
|
||||||
|
t.Fatalf("after the next daily: bucket %v, snapshots %d, result %+v; want the snapshot pruned, both dailies kept", got, sn.calls, res)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSyncSnapshotOnlyWhenACopyIsNewer: no snapshot while the newest bundle in
|
||||||
|
// the bucket was taken after the last copy, in the same second included (bundle
|
||||||
|
// names have one-second resolution); one when a copy is newer, or when the
|
||||||
|
// bucket has no bundle at all; none on an install that never copied an archive.
|
||||||
|
func TestSyncSnapshotOnlyWhenACopyIsNewer(t *testing.T) {
|
||||||
|
stamped := func(at time.Time) string { return dbbackup.BundleName(at, dbbackup.LabelDaily) }
|
||||||
|
cases := []struct {
|
||||||
|
name string
|
||||||
|
copied time.Time
|
||||||
|
bundle string // in the bucket before the pass; "" for none
|
||||||
|
want int
|
||||||
|
wantNew string
|
||||||
|
}{
|
||||||
|
{name: "never copied", want: 0},
|
||||||
|
{name: "bundle after the copy", copied: now.Add(-time.Hour), bundle: stamped(now.Add(-time.Minute)), want: 0},
|
||||||
|
{name: "same second", copied: now.Add(400 * time.Millisecond), bundle: stamped(now), want: 0},
|
||||||
|
{name: "copy after the bundle", copied: now.Add(-time.Hour), bundle: stamped(now.Add(-2 * time.Hour)), want: 1},
|
||||||
|
{name: "no bundle", copied: now.Add(-time.Hour), want: 1},
|
||||||
|
}
|
||||||
|
for _, c := range cases {
|
||||||
|
t.Run(c.name, func(t *testing.T) {
|
||||||
|
cat := &fakeCatalog{rows: []*row{
|
||||||
|
{WorldBackup: WorldBackup{ID: "b1", Ref: "/a/alpha-1.tar.gz"}, status: "present", offsite: c.copied},
|
||||||
|
}}
|
||||||
|
if c.copied.IsZero() {
|
||||||
|
cat.rows[0].status = "deleted"
|
||||||
|
}
|
||||||
|
s, b := newSyncer(t, cat)
|
||||||
|
sn := withSnapshot(t, s)
|
||||||
|
if c.bundle != "" {
|
||||||
|
b.objs[DBKey(c.bundle)] = seal(t, []byte("bundle"), s.Key)
|
||||||
|
}
|
||||||
|
res, err := s.Run(context.Background())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("run: %v", err)
|
||||||
|
}
|
||||||
|
if sn.calls != c.want {
|
||||||
|
t.Fatalf("snapshots taken = %d, want %d (newest %s)", sn.calls, c.want, res.NewestDB)
|
||||||
|
}
|
||||||
|
if c.want == 1 && res.NewestDB != dbbackup.BundleName(now, dbbackup.LabelOffsite) {
|
||||||
|
t.Fatalf("newest bundle = %s, want the snapshot", res.NewestDB)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
// A pass that copies no bundles (-db-dir "") takes none either.
|
||||||
|
cat := &fakeCatalog{rows: []*row{
|
||||||
|
{WorldBackup: WorldBackup{ID: "b1", Ref: "/a/alpha-1.tar.gz"}, status: "present", offsite: now.Add(-time.Hour)},
|
||||||
|
}}
|
||||||
|
s, _ := newSyncer(t, cat)
|
||||||
|
s.DBDir = ""
|
||||||
|
calls := 0
|
||||||
|
s.Snapshot = func(context.Context) error { calls++; return errors.New("no bundle directory") }
|
||||||
|
if _, err := s.Run(context.Background()); err != nil || calls != 0 {
|
||||||
|
t.Fatalf("no bundle directory: snapshots %d, err %v; want none", calls, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSyncSnapshotFailureIsReported: a snapshot that fails fails the pass, which
|
||||||
|
// the watchdog reports, and the bundles already there still count.
|
||||||
|
func TestSyncSnapshotFailureIsReported(t *testing.T) {
|
||||||
|
cat := &fakeCatalog{rows: []*row{
|
||||||
|
{WorldBackup: WorldBackup{ID: "b1", Server: "alpha", Ref: "/a/alpha-1.tar.gz"}, status: "present"},
|
||||||
|
}}
|
||||||
|
s, _ := newSyncer(t, cat)
|
||||||
|
sn := withSnapshot(t, s)
|
||||||
|
sn.err = errors.New("pg_dump: connection refused")
|
||||||
|
writeFile(t, s.ArchiveDir, "alpha-1.tar.gz", 100)
|
||||||
|
daily := "felis-db-20260924T030000Z-daily.tar"
|
||||||
|
writeFile(t, s.DBDir, daily, 50)
|
||||||
|
|
||||||
|
res, err := s.Run(context.Background())
|
||||||
|
if err == nil || !strings.Contains(err.Error(), "pg_dump: connection refused") {
|
||||||
|
t.Fatalf("run error = %v, want the snapshot's failure", err)
|
||||||
|
}
|
||||||
|
if res.NewestDB != daily || res.RemoteDB != 1 || res.WorldsUploaded != 1 {
|
||||||
|
t.Fatalf("result = %+v, want the daily still counted and the archive copied", res)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSyncBucketKeepsDailiesPastSnapshots: snapshots do not count against
|
||||||
|
// DBKeep, so hourly ones leave the daily restore points alone, and only the
|
||||||
|
// newest snapshot stays, while it is the newest bundle of all.
|
||||||
|
func TestSyncBucketKeepsDailiesPastSnapshots(t *testing.T) {
|
||||||
|
s, b := newSyncer(t, &fakeCatalog{})
|
||||||
|
for _, name := range []string{
|
||||||
|
"felis-db-20260922T030000Z-daily.tar",
|
||||||
|
"felis-db-20260923T030000Z-daily.tar",
|
||||||
|
"felis-db-20260924T030000Z-daily.tar",
|
||||||
|
} {
|
||||||
|
b.objs[DBKey(name)] = []byte("daily")
|
||||||
|
}
|
||||||
|
for h := 4; h <= 9; h++ {
|
||||||
|
name := dbbackup.BundleName(time.Date(2026, 9, 24, h, 0, 0, 0, time.UTC), dbbackup.LabelOffsite)
|
||||||
|
b.objs[DBKey(name)] = []byte("snapshot")
|
||||||
|
}
|
||||||
|
b.objs[DBKey("felis-db-20260923T120000Z-offsite.tar")] = []byte("older than a daily")
|
||||||
|
|
||||||
|
res, err := s.Run(context.Background())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
want := []string{"felis-db-20260923T030000Z-daily.tar", "felis-db-20260924T030000Z-daily.tar", "felis-db-20260924T090000Z-offsite.tar"}
|
||||||
|
if got := dbKeys(b); !slices.Equal(got, want) {
|
||||||
|
t.Fatalf("bucket bundles = %v, want %v", got, want)
|
||||||
|
}
|
||||||
|
if res.RemoteDB != 3 || res.DBPruned != 7 || res.NewestDB != want[2] {
|
||||||
|
t.Fatalf("result = %+v, want 3 held, 7 pruned, the snapshot newest", res)
|
||||||
|
}
|
||||||
|
|
||||||
|
got := keptBundles([]string{"felis-db-20260925T030000Z-daily.tar", "felis-db-20260924T090000Z-offsite.tar", "felis-db-20260924T030000Z-daily.tar"}, 2)
|
||||||
|
if len(got) != 2 || got["felis-db-20260924T090000Z-offsite.tar"] {
|
||||||
|
t.Fatalf("kept = %v, want the two dailies: a snapshot older than a daily lists nothing it does not", got)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// TestSyncSweepsUnrecordedWorlds: a world object no row keeps goes once it is
|
||||||
|
// older than OrphanAfter; a younger one, one whose age the bucket does not
|
||||||
|
// report, one a row keeps however old, and anything that is not an archive
|
||||||
|
// object stay.
|
||||||
|
func TestSyncSweepsUnrecordedWorlds(t *testing.T) {
|
||||||
|
day := 24 * time.Hour
|
||||||
|
rows := func() []*row {
|
||||||
|
return []*row{
|
||||||
|
{WorldBackup: WorldBackup{ID: "kept", Ref: "/a/kept.tar.gz"}, status: "present", offsite: now.Add(-200 * day)},
|
||||||
|
{WorldBackup: WorldBackup{ID: "evicted", Ref: "/a/evicted.tar.gz"}, status: "deleted", expires: now.Add(day), offsite: now.Add(-100 * day)},
|
||||||
|
// Copied, but the run died before MarkOffsite, and the row has
|
||||||
|
// since expired: ExpiredRefs never lists it.
|
||||||
|
{WorldBackup: WorldBackup{ID: "unmarked", Ref: "/a/unmarked.tar.gz"}, status: "deleted", expires: now.Add(-day)},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
ages := map[string]time.Duration{
|
||||||
|
"worlds/kept.tar.gz.fenc": 200 * day,
|
||||||
|
"worlds/evicted.tar.gz.fenc": 100 * day,
|
||||||
|
"worlds/unmarked.tar.gz.fenc": 95 * day,
|
||||||
|
"worlds/restored-old.tar.gz.fenc": 91 * day,
|
||||||
|
"worlds/restored-new.tar.gz.fenc": 10 * day,
|
||||||
|
"worlds/no-age.tar.gz.fenc": 0,
|
||||||
|
"worlds/README": 300 * day,
|
||||||
|
}
|
||||||
|
setup := func(t *testing.T, cat *fakeCatalog) (*Syncer, *memBucket, *bytes.Buffer) {
|
||||||
|
s, b := newSyncer(t, cat)
|
||||||
|
s.OrphanAfter = 90 * day
|
||||||
|
var log bytes.Buffer
|
||||||
|
s.Log = &log
|
||||||
|
b.modified = map[string]time.Time{}
|
||||||
|
for k, age := range ages {
|
||||||
|
b.objs[k] = []byte("x")
|
||||||
|
if age > 0 {
|
||||||
|
b.modified[k] = now.Add(-age)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return s, b, &log
|
||||||
|
}
|
||||||
|
|
||||||
|
s, b, log := setup(t, &fakeCatalog{rows: rows()})
|
||||||
|
res, err := s.Run(context.Background())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatal(err)
|
||||||
|
}
|
||||||
|
removed := slices.Sorted(slices.Values(b.removed))
|
||||||
|
if want := []string{"worlds/restored-old.tar.gz.fenc", "worlds/unmarked.tar.gz.fenc"}; !slices.Equal(removed, want) {
|
||||||
|
t.Fatalf("removed %v, want %v", removed, want)
|
||||||
|
}
|
||||||
|
if res.WorldsExpired != 2 || res.RemoteWorlds != len(ages)-2 {
|
||||||
|
t.Fatalf("result = %+v, want 2 expired, %d left", res, len(ages)-2)
|
||||||
|
}
|
||||||
|
if !strings.Contains(log.String(), "2 world archives in the bucket have no backup record") {
|
||||||
|
t.Errorf("the young unrecorded archives were not reported:\n%s", log.String())
|
||||||
|
}
|
||||||
|
|
||||||
|
// A database not restored yet keeps nothing: nothing is swept on it.
|
||||||
|
s, b, _ = setup(t, &fakeCatalog{})
|
||||||
|
if _, err := s.Run(context.Background()); err != nil || len(b.removed) != 0 {
|
||||||
|
t.Fatalf("an empty catalog: removed %v, err %v; want nothing", b.removed, err)
|
||||||
|
}
|
||||||
|
s, b, _ = setup(t, &fakeCatalog{rows: rows()})
|
||||||
|
s.OrphanAfter = 0
|
||||||
|
if _, err := s.Run(context.Background()); err != nil || len(b.removed) != 0 {
|
||||||
|
t.Fatalf("no OrphanAfter: removed %v, err %v; want nothing", b.removed, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
+136
-10
@@ -12,7 +12,11 @@
|
|||||||
// deletes an idle world only after that (internal/reaper). Remote world
|
// deletes an idle world only after that (internal/reaper). Remote world
|
||||||
// objects go when their row has expired, so the bucket keeps each archive for
|
// objects go when their row has expired, so the bucket keeps each archive for
|
||||||
// the same retention the panel promises, including archives evicted early from
|
// the same retention the panel promises, including archives evicted early from
|
||||||
// the local disk to make room. Remote bundles are pruned to the newest DBKeep.
|
// the local disk to make room; an object whose row a restore did not bring
|
||||||
|
// back goes once it is older than the longest retention. Remote bundles are
|
||||||
|
// pruned to the newest DBKeep. A pass that copies archives then sends a fresh
|
||||||
|
// bundle, whose rows list them, so the newest bundle in the bucket lists every
|
||||||
|
// archive in it and a restore from it can fetch them all.
|
||||||
package offsite
|
package offsite
|
||||||
|
|
||||||
import (
|
import (
|
||||||
@@ -78,6 +82,12 @@ type Catalog interface {
|
|||||||
ExpiredRefs(ctx context.Context, now time.Time) ([]string, error)
|
ExpiredRefs(ctx context.Context, now time.Time) ([]string, error)
|
||||||
// PresentWorlds lists every present archive, for a restore of the volume.
|
// PresentWorlds lists every present archive, for a restore of the volume.
|
||||||
PresentWorlds(ctx context.Context) ([]WorldBackup, error)
|
PresentWorlds(ctx context.Context) ([]WorldBackup, error)
|
||||||
|
// NewestOffsite is the latest offsite_at of any row: when the bucket last
|
||||||
|
// gained an archive. Zero when no archive was ever copied.
|
||||||
|
NewestOffsite(ctx context.Context) (time.Time, error)
|
||||||
|
// KeptRefs lists the backup_ref of every row whose archive the bucket
|
||||||
|
// keeps: present ones, and the rest until their expires_at.
|
||||||
|
KeptRefs(ctx context.Context, now time.Time) ([]string, error)
|
||||||
}
|
}
|
||||||
|
|
||||||
// Syncer copies what is missing from the bucket and prunes what has expired.
|
// Syncer copies what is missing from the bucket and prunes what has expired.
|
||||||
@@ -89,9 +99,16 @@ type Syncer struct {
|
|||||||
// there is none yet (nothing has been archived on this install).
|
// there is none yet (nothing has been archived on this install).
|
||||||
ArchiveDir string
|
ArchiveDir string
|
||||||
// DBDir holds the database bundles; DBKeep is how many of the newest the
|
// DBDir holds the database bundles; DBKeep is how many of the newest the
|
||||||
// bucket keeps.
|
// bucket keeps, not counting the newest Snapshot bundle.
|
||||||
DBDir string
|
DBDir string
|
||||||
DBKeep int
|
DBKeep int
|
||||||
|
// Snapshot writes a bundle labelled dbbackup.LabelOffsite into DBDir. A
|
||||||
|
// pass that leaves the bucket holding an archive newer than its newest
|
||||||
|
// bundle takes one and sends it (snapshotDB); nil takes none.
|
||||||
|
Snapshot func(ctx context.Context) error
|
||||||
|
// OrphanAfter is how old a world object no row keeps (KeptRefs) may get
|
||||||
|
// before it goes: the longest retention of any archive. Zero keeps them.
|
||||||
|
OrphanAfter time.Duration
|
||||||
// Images is the platform registry whose user images are copied (images.go);
|
// Images is the platform registry whose user images are copied (images.go);
|
||||||
// nil copies none.
|
// nil copies none.
|
||||||
Images ImageSource
|
Images ImageSource
|
||||||
@@ -190,10 +207,11 @@ func (s *Syncer) logf(format string, args ...any) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Run does one pass: database bundles, world archives, registry images,
|
// Run does one pass: database bundles, world archives, a bundle listing the
|
||||||
// submission uploads, then expiry. The bundles go first: they are small, and
|
// archives just copied, registry images, submission uploads, then expiry. The
|
||||||
// every restore starts from one. A failure on one item is recorded and the
|
// bundles go first: they are small, and every restore starts from one. A
|
||||||
// pass carries on; the returned error is non-nil when anything failed.
|
// failure on one item is recorded and the pass carries on; the returned error
|
||||||
|
// is non-nil when anything failed.
|
||||||
func (s *Syncer) Run(ctx context.Context) (Result, error) {
|
func (s *Syncer) Run(ctx context.Context) (Result, error) {
|
||||||
var res Result
|
var res Result
|
||||||
fail := func(format string, args ...any) {
|
fail := func(format string, args ...any) {
|
||||||
@@ -214,9 +232,11 @@ func (s *Syncer) Run(ctx context.Context) (Result, error) {
|
|||||||
}
|
}
|
||||||
s.syncDB(ctx, &res, fail)
|
s.syncDB(ctx, &res, fail)
|
||||||
s.syncWorlds(ctx, remoteWorlds, &res, fail)
|
s.syncWorlds(ctx, remoteWorlds, &res, fail)
|
||||||
|
s.snapshotDB(ctx, &res, fail)
|
||||||
s.syncImages(ctx, &res, fail)
|
s.syncImages(ctx, &res, fail)
|
||||||
s.syncUploads(ctx, &res, fail)
|
s.syncUploads(ctx, &res, fail)
|
||||||
s.expireWorlds(ctx, remoteWorlds, &res, fail)
|
s.expireWorlds(ctx, remoteWorlds, &res, fail)
|
||||||
|
s.sweepWorlds(ctx, remoteWorlds, &res, fail)
|
||||||
|
|
||||||
for _, size := range remoteWorlds {
|
for _, size := range remoteWorlds {
|
||||||
res.RemoteWorlds++
|
res.RemoteWorlds++
|
||||||
@@ -333,10 +353,7 @@ func (s *Syncer) syncDB(ctx context.Context, res *Result, fail func(string, ...a
|
|||||||
// Bundle names start with their UTC stamp, so reversed order is newest first.
|
// Bundle names start with their UTC stamp, so reversed order is newest first.
|
||||||
ranked := slices.Sorted(maps.Keys(names))
|
ranked := slices.Sorted(maps.Keys(names))
|
||||||
slices.Reverse(ranked)
|
slices.Reverse(ranked)
|
||||||
kept := map[string]bool{}
|
kept := keptBundles(ranked, keep)
|
||||||
for _, name := range ranked[:min(keep, len(ranked))] {
|
|
||||||
kept[name] = true
|
|
||||||
}
|
|
||||||
for _, b := range local {
|
for _, b := range local {
|
||||||
if !kept[b.Name] {
|
if !kept[b.Name] {
|
||||||
continue
|
continue
|
||||||
@@ -367,6 +384,7 @@ func (s *Syncer) syncDB(ctx context.Context, res *Result, fail func(string, ...a
|
|||||||
delete(remote, key)
|
delete(remote, key)
|
||||||
res.DBPruned++
|
res.DBPruned++
|
||||||
}
|
}
|
||||||
|
res.RemoteDB, res.NewestDB = 0, ""
|
||||||
for _, name := range ranked {
|
for _, name := range ranked {
|
||||||
if _, ok := remote[DBKey(name)]; ok {
|
if _, ok := remote[DBKey(name)]; ok {
|
||||||
res.RemoteDB++
|
res.RemoteDB++
|
||||||
@@ -377,6 +395,59 @@ func (s *Syncer) syncDB(ctx context.Context, res *Result, fail func(string, ...a
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// keptBundles picks the bundles the bucket keeps from ranked, newest first:
|
||||||
|
// the newest keep that are not snapshots, and the newest snapshot while it is
|
||||||
|
// the newest of all. A pass takes a snapshot whenever it copies archives, up to
|
||||||
|
// once an hour, so counting them in keep would push the daily bundles, the
|
||||||
|
// restore points of the last fortnight, out of the bucket within a day; one
|
||||||
|
// older than the newest daily lists nothing that daily does not.
|
||||||
|
func keptBundles(ranked []string, keep int) map[string]bool {
|
||||||
|
kept := map[string]bool{}
|
||||||
|
n := 0
|
||||||
|
for i, name := range ranked {
|
||||||
|
if _, label, _ := dbbackup.ParseBundleName(name); label == dbbackup.LabelOffsite {
|
||||||
|
if i == 0 {
|
||||||
|
kept[name] = true
|
||||||
|
}
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if n++; n > keep {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
kept[name] = true
|
||||||
|
}
|
||||||
|
return kept
|
||||||
|
}
|
||||||
|
|
||||||
|
// snapshotDB takes a bundle and sends it when the bucket holds an archive
|
||||||
|
// copied after its newest bundle was taken. A restore finds archives through
|
||||||
|
// the world_backups rows of the bundle it restores: an archive with no row
|
||||||
|
// there is neither fetched back nor ever expired, and until the next daily
|
||||||
|
// bundle, up to a day later, every archive copied since the last one was in
|
||||||
|
// that state. The comparison is by second, the resolution of bundle names; a
|
||||||
|
// bundle taken in the second of the copy was taken after it.
|
||||||
|
func (s *Syncer) snapshotDB(ctx context.Context, res *Result, fail func(string, ...any)) {
|
||||||
|
if s.Snapshot == nil || s.DBDir == "" {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
copied, err := s.Catalog.NewestOffsite(ctx)
|
||||||
|
if err != nil {
|
||||||
|
fail("read when an archive was last copied: %v", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if copied.IsZero() {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if taken, _, ok := dbbackup.ParseBundleName(res.NewestDB); ok && !copied.Truncate(time.Second).After(taken) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err := s.Snapshot(ctx); err != nil {
|
||||||
|
fail("take a database bundle listing the archives just copied: %v", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
s.syncDB(ctx, res, fail)
|
||||||
|
}
|
||||||
|
|
||||||
func (s *Syncer) expireWorlds(ctx context.Context, remote map[string]int64, res *Result, fail func(string, ...any)) {
|
func (s *Syncer) expireWorlds(ctx context.Context, remote map[string]int64, res *Result, fail func(string, ...any)) {
|
||||||
refs, err := s.Catalog.ExpiredRefs(ctx, s.now())
|
refs, err := s.Catalog.ExpiredRefs(ctx, s.now())
|
||||||
if err != nil {
|
if err != nil {
|
||||||
@@ -400,6 +471,61 @@ func (s *Syncer) expireWorlds(ctx context.Context, remote map[string]int64, res
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// sweepWorlds removes world objects no row keeps once they are older than
|
||||||
|
// OrphanAfter, the longest any archive is kept: what a restore from an older
|
||||||
|
// bundle leaves in the bucket (its archives were copied after that bundle),
|
||||||
|
// and the copy of a row whose MarkOffsite failed before the row expired.
|
||||||
|
// Younger ones are kept and reported, as the reaper does on the archive
|
||||||
|
// volume. A catalog that keeps nothing is a database not restored yet, whose
|
||||||
|
// rows would keep everything, so nothing is swept on it; an object whose age
|
||||||
|
// the bucket does not report stays.
|
||||||
|
func (s *Syncer) sweepWorlds(ctx context.Context, remote map[string]int64, res *Result, fail func(string, ...any)) {
|
||||||
|
if s.OrphanAfter <= 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
refs, err := s.Catalog.KeptRefs(ctx, s.now())
|
||||||
|
if err != nil {
|
||||||
|
fail("list the world archives the bucket keeps: %v", err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if len(refs) == 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
kept := make(map[string]bool, len(refs))
|
||||||
|
for _, ref := range refs {
|
||||||
|
if key, ok := WorldKey(ref); ok {
|
||||||
|
kept[key] = true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
objs, err := s.Bucket.List(ctx, worldsDir)
|
||||||
|
if err != nil {
|
||||||
|
fail("list %s in the bucket: %v", worldsDir, err)
|
||||||
|
return
|
||||||
|
}
|
||||||
|
cutoff := s.now().Add(-s.OrphanAfter)
|
||||||
|
var young []string
|
||||||
|
for _, o := range objs {
|
||||||
|
if _, ok := remote[o.Key]; !ok || kept[o.Key] || !strings.HasSuffix(o.Key, objExt) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if o.Modified.IsZero() || o.Modified.After(cutoff) {
|
||||||
|
young = append(young, path.Base(o.Key))
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if err := s.Bucket.Remove(ctx, o.Key); err != nil {
|
||||||
|
fail("remove %s, which no backup records: %v", o.Key, err)
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
delete(remote, o.Key)
|
||||||
|
res.WorldsExpired++
|
||||||
|
s.logf("removed world archive %s: no backup records it, and it was copied %s ago, past the longest retention", path.Base(o.Key), s.now().Sub(o.Modified).Round(time.Hour))
|
||||||
|
}
|
||||||
|
if len(young) > 0 {
|
||||||
|
s.logf("%d world archives in the bucket have no backup record (a database restored from an older bundle?); each goes once it is older than %s: %s",
|
||||||
|
len(young), s.OrphanAfter, strings.Join(young[:min(len(young), 5)], ", "))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// putFile encrypts the file at p into key.
|
// putFile encrypts the file at p into key.
|
||||||
func (s *Syncer) putFile(ctx context.Context, key, p string, size int64) error {
|
func (s *Syncer) putFile(ctx context.Context, key, p string, size int64) error {
|
||||||
f, err := os.Open(p)
|
f, err := os.Open(p)
|
||||||
|
|||||||
@@ -0,0 +1,60 @@
|
|||||||
|
//go:build pgint
|
||||||
|
|
||||||
|
package pgint
|
||||||
|
|
||||||
|
import (
|
||||||
|
"context"
|
||||||
|
"slices"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"felis.lolicon.best/internal/offsite"
|
||||||
|
"felis.lolicon.best/internal/reaper"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestOffsiteCatalogNewestCopyAndKeptRefs: the two reads behind the off-site
|
||||||
|
// copy's snapshot and sweep. NewestOffsite is the latest copy of any row;
|
||||||
|
// KeptRefs keeps a present archive past its expiry (the reaper expires it, not
|
||||||
|
// the sweep) and a deleted one until its expires_at, and nothing after.
|
||||||
|
func TestOffsiteCatalogNewestCopyAndKeptRefs(t *testing.T) {
|
||||||
|
ctx := context.Background()
|
||||||
|
cat := offsite.PGCatalog{DB: db}
|
||||||
|
sfx := suffix(t)
|
||||||
|
server := "offsite-" + sfx
|
||||||
|
t.Cleanup(func() { db.ExecContext(ctx, `DELETE FROM world_backups WHERE server_name = $1`, server) })
|
||||||
|
now := time.Now().UTC().Truncate(time.Microsecond)
|
||||||
|
// Later than any copy another test records.
|
||||||
|
latest := now.Add(1000 * reaper.Day)
|
||||||
|
insert := func(id, status string, expires time.Time, offsiteAt any) string {
|
||||||
|
t.Helper()
|
||||||
|
ref := "/archives/" + id + "-" + sfx + ".tar.gz"
|
||||||
|
mustExec(t, `INSERT INTO world_backups (id, server_name, backup_ref, size_bytes, reason, status, created_at, expires_at, offsite_at)
|
||||||
|
VALUES ($1, $2, $3, 1, 'manual', $4, $5, $6, $7)`,
|
||||||
|
id+"-"+sfx, server, ref, status, now.Add(-reaper.Day), expires, offsiteAt)
|
||||||
|
return ref
|
||||||
|
}
|
||||||
|
presentPast := insert("present-past", "present", now.Add(-time.Minute), nil)
|
||||||
|
deletedLive := insert("deleted-live", "deleted", now.Add(reaper.Day), now)
|
||||||
|
deletedGone := insert("deleted-gone", "deleted", now.Add(-time.Minute), now)
|
||||||
|
expiredGone := insert("expired-gone", "expired", now.Add(-time.Minute), nil)
|
||||||
|
insert("newest-copy", "deleted", now.Add(-time.Minute), latest)
|
||||||
|
|
||||||
|
got, err := cat.NewestOffsite(ctx)
|
||||||
|
if err != nil || !got.Equal(latest) {
|
||||||
|
t.Fatalf("NewestOffsite = %v, %v; want %v", got, err, latest)
|
||||||
|
}
|
||||||
|
refs, err := cat.KeptRefs(ctx, now)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("KeptRefs: %v", err)
|
||||||
|
}
|
||||||
|
for _, want := range []string{presentPast, deletedLive} {
|
||||||
|
if !slices.Contains(refs, want) {
|
||||||
|
t.Errorf("KeptRefs lacks %s", want)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
for _, gone := range []string{deletedGone, expiredGone} {
|
||||||
|
if slices.Contains(refs, gone) {
|
||||||
|
t.Errorf("KeptRefs keeps %s, past its expiry", gone)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
Reference in new issue
Block a user