Unverified Commit 9a892541 authored by Lemon-miaow's avatar Lemon-miaow
Browse files

test(db): 恢复链在真实 PostgreSQL 上跑往返与三种失败回滚,-yes 闸门测试真正走到闸门,e2e 按文档做一次同机恢复

parent 7e627299
Loading
Loading
Loading
Loading
+4 −0
Changes for .github/workflows/ci.yml: 4 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -72,6 +72,9 @@ jobs:
  # The business stores' SQL against a real PostgreSQL (internal/pgint): the unit suites run
  # on fakes, and PGRepo drifted from them three times while those stayed green. 13 is the
  # oldest server a supported distribution installs (EL9), 18 the newest (Arch).
  # `felis db backup` and `restore` run there too, with the tools inside the service
  # container, as production runs them inside felis-postgres: the runner's own client is one
  # major version, and pg_dump refuses a newer server.
  pgint:
    runs-on: ubuntu-latest
    strategy:
@@ -102,6 +105,7 @@ jobs:
      - run: go test -race -tags pgint -count=1 ./internal/pgint/
        env:
          FELIS_TEST_PG_URL: postgres://felis:pgint@localhost:5432/felis_pgint?sslmode=disable
          FELIS_TEST_PG_EXEC: docker exec -i ${{ job.services.postgres.id }}

  shell:
    runs-on: ubuntu-latest
+9 −2
Changes for CONTRIBUTING.md: 9 added lines, 2 removed lines.
Original line number Diff line number Diff line
@@ -78,8 +78,15 @@ FELIS_TEST_PG_URL='postgres://felis:***@127.0.0.1:5432/felis_pgint?sslmode=disab
```

Run it after touching anything under `internal/api/pgrepo.go`, `internal/submit`,
or `internal/build` that speaks SQL: the fakes encode the contract, and this
suite exists to catch the drift between the fakes and the real queries.
`internal/build` or `internal/dbbackup` that speaks SQL: the fakes encode the
contract, and this suite exists to catch the drift between the fakes and the real
queries. The `felis db backup` and `restore` tests also run `pg_dump`, `pg_restore`
and `psql`, which must be the server's major version. For a server in a container,
run them in it, as production does in felis-postgres:

```bash
FELIS_TEST_PG_EXEC='docker exec -i <container>' FELIS_TEST_PG_URL=... go test -tags pgint ./internal/pgint/
```

Build the CLI:

+49 −11
Changes for cmd/felis/db_test.go: 49 added lines, 11 removed lines.
Original line number Diff line number Diff line
@@ -29,13 +29,43 @@ func TestDBUsage(t *testing.T) {
	}
}

// TestDBRestoreNeedsYes: without -yes a restore describes the bundle and stops
// before anything reaches the database, even with -force and
// -no-safety-backup, which would otherwise let the replay run at once.
func TestDBRestoreNeedsYes(t *testing.T) {
	// A bundle that does not exist fails verification (1) before -yes matters;
	// the -yes gate itself is exercised against a real bundle in internal/dbbackup
	// and on the VM. Here: the refusal path never reaches the config or database.
	dir := newPodRig(t)
	cfg := podConfig(t, dir)
	bundles := filepath.Join(dir, "bundles")
	var out, errBuf bytes.Buffer
	if code := run([]string{"db", "restore", "-dir", t.TempDir(), "missing.tar"}, &out, &errBuf); code != 1 {
		t.Fatalf("exit %d, stderr %q", code, errBuf.String())
	if code := run([]string{"db", "backup", "-config", cfg, "-dir", bundles, "-state-dir", "", "-no-servers"}, &out, &errBuf); code != 0 {
		t.Fatalf("backup: exit %d: %s", code, errBuf.String())
	}
	bundle := strings.TrimSpace(strings.TrimPrefix(out.String(), "felis db backup: wrote "))
	podRuns(t, dir)

	out.Reset()
	errBuf.Reset()
	code := run([]string{"db", "restore", "-config", cfg, "-dir", bundles, "-force", "-no-safety-backup", filepath.Base(bundle)}, &out, &errBuf)
	if code != 2 {
		t.Fatalf("exit %d, want 2; stderr %q", code, errBuf.String())
	}
	if want := filepath.Base(bundle) + " (manual, taken "; !strings.Contains(errBuf.String(), want) || !strings.Contains(errBuf.String(), "schema 3).") {
		t.Errorf("stderr %q does not describe the bundle", errBuf.String())
	}
	if !strings.Contains(errBuf.String(), "re-run with -yes") {
		t.Errorf("stderr %q does not say how to go on", errBuf.String())
	}
	if argv, err := os.ReadFile(filepath.Join(dir, "k3s.args")); err == nil {
		t.Errorf("a restore without -yes ran in the database pod:\n%s", argv)
	}

	// A bundle that does not verify is refused before -yes is weighed.
	errBuf.Reset()
	if code := run([]string{"db", "restore", "-config", cfg, "-dir", bundles, "-yes", "missing.tar"}, &out, &errBuf); code != 1 {
		t.Errorf("missing bundle: exit %d, want 1; stderr %q", code, errBuf.String())
	}
	if _, err := os.Stat(filepath.Join(dir, "k3s.args")); err == nil {
		t.Error("a missing bundle reached the database pod")
	}
}

@@ -268,6 +298,19 @@ func podRuns(t *testing.T, dir string) []string {
	return runs
}

// podConfig writes an installed host's felis.toml, [database] pointing at the
// pod, into dir.
func podConfig(t *testing.T, dir string) string {
	t.Helper()
	toml := strings.Replace(installerTOML("example.com", "127.0.0.1"),
		`url = "postgres://felis:[email protected]:5432/felis?sslmode=disable"`,
		`url = "`+podDB.URL+`"
deployment = "`+podDB.Deployment+`"`, 1)
	cfg := filepath.Join(dir, "felis.toml")
	writeTestFile(t, cfg, toml, 0o600)
	return cfg
}

func ranIn(runs []string, prefix string) bool {
	for _, r := range runs {
		if strings.HasPrefix(r, prefix) {
@@ -284,12 +327,7 @@ func ranIn(runs []string, prefix string) bool {
// line.
func TestDBBackupAndRestoreRunTheToolsInTheDatabasePod(t *testing.T) {
	dir := newPodRig(t)
	toml := strings.Replace(installerTOML("example.com", "127.0.0.1"),
		`url = "postgres://felis:[email protected]:5432/felis?sslmode=disable"`,
		`url = "`+podDB.URL+`"
deployment = "`+podDB.Deployment+`"`, 1)
	cfg := filepath.Join(dir, "felis.toml")
	writeTestFile(t, cfg, toml, 0o600)
	cfg := podConfig(t, dir)

	var out, errBuf bytes.Buffer
	bundles := filepath.Join(dir, "bundles")
+57 −2
Changes for deploy/e2e_check.sh: 57 added lines, 2 removed lines.
Original line number Diff line number Diff line
@@ -8,8 +8,9 @@
#   sudo bash deploy/e2e_check.sh upgrade   # after this commit ran over a release
#
# It asks what an operator's first minutes ask: the binary runs, the control plane and its
# database are rolled out and ready, a database backup can be taken, the panel answers on
# its NodePort, the proxy answers a Minecraft status ping, and the host timers are there.
# database are rolled out and ready, a database backup can be taken and restored, the panel
# answers on its NodePort, the proxy answers a Minecraft status ping, and the host timers
# are there.
# A rerun must also leave the proxy running (it restarts only when what it runs changed)
# and keep every earlier answer.
set -euo pipefail
@@ -50,6 +51,58 @@ check "felis-api is ready (database and cluster reachable)" \
for unit in k3s felis-velocity; do
  check "${unit} is active" systemctl is-active --quiet "$unit"
done
# pod_psql runs one statement in the database's pod, over its socket, as the felis role on
# the felis database.
pod_psql() {
  "${KUBECTL[@]}" -n felis exec -i deploy/felis-postgres -c postgres -- \
    psql -X -q -At -v ON_ERROR_STOP=1 -U felis -d felis -c "$1"
}

# restore_drill walks troubleshooting.md's "Restore on the same host": refused while the
# control plane is connected; with it scaled to 0 the bundle comes back (a row written after
# it is gone) and the database it replaced is kept; migrate up runs; the control plane serves
# again.
restore_drill() { # dir bundle
  local dir="$1" bundle="$2" out rc
  local sel="app.kubernetes.io/part-of=felis-control-plane,app.kubernetes.io/component in (api,operator)"
  if ! pod_psql "INSERT INTO platform_settings (key, value) VALUES ('e2e_restore_drill', '1')" >/dev/null; then
    fail "write a row after the bundle"
    return
  fi
  rc=0
  out="$(/usr/local/bin/felis db restore -dir "$dir" -yes "$bundle" 2>&1)" || rc=$?
  if [ "$rc" -eq 1 ] && grep -q "other clients are connected to the database" <<<"$out"; then
    pass "felis db restore refuses while the control plane is connected"
  else
    fail "felis db restore refuses while the control plane is connected (exit ${rc}): ${out}"
  fi

  "${KUBECTL[@]}" -n felis scale deployment felis-api felis-operator --replicas=0 >/dev/null
  for _ in $(seq 60); do
    [ -z "$("${KUBECTL[@]}" -n felis get pods -l "$sel" -o name)" ] && break
    sleep 2
  done
  rc=0
  out="$(/usr/local/bin/felis db restore -dir "$dir" -yes "$bundle" 2>&1)" || rc=$?
  if [ "$rc" -eq 0 ]; then
    pass "felis db restore replays the bundle with the control plane scaled to 0"
  else
    fail "felis db restore replays the bundle with the control plane scaled to 0 (exit ${rc}): ${out}"
  fi
  check "the restore dropped the row written after the bundle" \
    test "$(pod_psql "SELECT count(*) FROM platform_settings WHERE key = 'e2e_restore_drill'")" = 0
  check "the restore kept the database it replaced in a pre-restore bundle" \
    sh -c "ls '${dir}' | grep -q -- '-pre-restore\.tar\$'"
  check "felis migrate up runs on the restored database" \
    /usr/local/bin/felis migrate up -config /etc/felis/felis.host.toml
  "${KUBECTL[@]}" -n felis scale deployment felis-api felis-operator --replicas=1 >/dev/null
  for d in felis-api felis-operator; do
    check "deployment ${d} is rolled out again after the restore" "${KUBECTL[@]}" -n felis rollout status "deploy/${d}" --timeout=180s
  done
  check "felis-api is ready on the restored database" \
    curl -sf --retry 10 --retry-delay 3 --retry-all-errors -o /dev/null "http://${internal}/readyz"
}

# The database runs in k3s; a release may still run it on the host, and the upgrade moved
# it. The host has no PostgreSQL client: a bundle that verifies proves felis reaches the
# database's pod through kubectl exec, and that pg_dump there reads every table.
@@ -59,6 +112,8 @@ if [ "$phase" != release ]; then
  if out="$(/usr/local/bin/felis db backup -dir "$bundle_dir" -state-dir "" -no-servers 2>&1)"; then
    bundle="$(printf '%s\n' "$out" | sed -n 's/^felis db backup: wrote //p' | tail -n 1)"
    check "felis db backup writes a bundle that verifies" /usr/local/bin/felis db verify "$bundle"
    # A rerun keeps what the install left; the install and the upgrade restore it.
    [ "$phase" = rerun ] || restore_drill "$bundle_dir" "$bundle"
  else
    fail "felis db backup writes a bundle: ${out}"
  fi
+1 −1
Changes for docs/troubleshooting.md: 1 added line, 1 removed line.
Original line number Diff line number Diff line
@@ -2011,7 +2011,7 @@ kubectl -n felis scale deployment felis-api felis-operator --replicas=1
- The replay is one transaction: it drops everything the `felis` role owns and
  loads the dump. **Any failure rolls back and leaves the database exactly as it
  was** (`rolled back, the database is unchanged`, with the psql and pg_restore
  errors). [GO-TESTED]
  errors). [PG-TESTED]
- The database before the restore is in the `pre-restore` bundle it names;
  restoring that one undoes the restore.
- `migrate up` brings an older bundle's schema up to the running release.
Loading