feat(distributed): 支持单主控多节点部署和停服迁移
复用现有 k3s 调度和 Job 生命周期,增加 worker 接入与批准、受保护节点身份、归档传输、持久迁移锁及活动 PVC 切换;同步管理员 API、CLI、面板和隔离规则。分布式模式默认关闭,保持单机兼容。 验证:Go 全量测试与 vet;面板 874 个测试、lint/build;Linux VM 安装器测试、清单服务端 dry-run、网络命名空间防火墙实测。A/B/C 三机 WireGuard、Velocity 和迁移验收仍待完成。
This commit is contained in:
77 files changed
+5224
-73
No files matched your search
+10
-4
@@ -29,10 +29,11 @@ import (
|
||||
|
||||
// API holds the dependencies shared by every handler.
|
||||
type API struct {
|
||||
Repo Repo
|
||||
Cluster Cluster
|
||||
Internal InternalAuth
|
||||
External ExternalAuth
|
||||
Distribution Distribution
|
||||
Repo Repo
|
||||
Cluster Cluster
|
||||
Internal InternalAuth
|
||||
External ExternalAuth
|
||||
|
||||
// Builder is the image build subsystem (spec §16). It is optional: when nil
|
||||
// the /images routes report 503 rather than 404, so the admin boundary is
|
||||
@@ -721,6 +722,11 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
// Zero-Trust path, unlike the app-tier /me/servers. A path distinct from the
|
||||
// internal velocity GET /api/v1/servers on purpose: the parity test forbids one
|
||||
// {method, path} from carrying both the service and admin tiers.
|
||||
{Method: "GET", Pattern: "/api/v1/nodes", Admin: true, h: a.handleNodes},
|
||||
{Method: "GET", Pattern: "/api/v1/servers/{name}/migrations", Admin: true, h: a.handleMigrationStatus},
|
||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/migrations", Admin: true, h: a.handleMigration},
|
||||
{Method: "GET", Pattern: "/api/v1/servers/{name}/migrations/{id}", Admin: true, h: a.handleMigrationStatus},
|
||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/migrations/{id}/retry", Admin: true, h: a.handleMigrationRetry},
|
||||
{Method: "GET", Pattern: "/api/v1/fleet", Admin: true, h: a.handleFleet},
|
||||
// Image build + whitelist (spec §16, §15). Every route is admin-tier: a build
|
||||
// is build-time RCE against the cluster, so submission requires the admin
|
||||
|
||||
@@ -13,6 +13,7 @@ import (
|
||||
// patch (spec §7 PATCH /servers/{name}) — never the business-layer fields, which
|
||||
// live in Postgres (spec §22).
|
||||
type ServerInfo struct {
|
||||
NodeName string `json:"nodeName,omitempty"`
|
||||
Name string `json:"name"`
|
||||
Subdomain string `json:"subdomain"`
|
||||
Phase string `json:"phase"`
|
||||
@@ -64,6 +65,7 @@ type ServerInfo struct {
|
||||
// no free-form YAML path — every field is a typed, validated value. A created
|
||||
// server starts DesiredState=Stopped and unowned (claimed later, spec §9.3).
|
||||
type CreateServerInput struct {
|
||||
NodeName string
|
||||
Name string
|
||||
Subdomain string
|
||||
DisplayName string
|
||||
|
||||
@@ -0,0 +1,122 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
|
||||
"felis.lolicon.best/internal/distributed"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
)
|
||||
|
||||
type Distribution interface {
|
||||
Nodes(context.Context) ([]distributed.Node, error)
|
||||
ValidateNode(context.Context, string) error
|
||||
BeginMigration(context.Context, string, string, string) (distributed.Operation, error)
|
||||
Migration(context.Context, string, string) (distributed.Operation, error)
|
||||
RetryMigration(context.Context, string, string) (distributed.Operation, error)
|
||||
}
|
||||
|
||||
func (a *API) distributedReady(w http.ResponseWriter, r *http.Request) bool {
|
||||
if a.Distribution == nil {
|
||||
writeError(w, r, newError(503, "distributed_unavailable", "distributed deployment is not configured"))
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
func (a *API) handleNodes(w http.ResponseWriter, r *http.Request) {
|
||||
if !a.distributedReady(w, r) {
|
||||
return
|
||||
}
|
||||
nodes, err := a.Distribution.Nodes(r.Context())
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"nodes": nodes})
|
||||
}
|
||||
|
||||
func (a *API) handleMigration(w http.ResponseWriter, r *http.Request) {
|
||||
if !a.distributedReady(w, r) {
|
||||
return
|
||||
}
|
||||
server := r.PathValue("name")
|
||||
if err := naming.ValidateServerName(server); err != nil {
|
||||
writeError(w, r, newError(400, "bad_name", "%v", err))
|
||||
return
|
||||
}
|
||||
rec, err := a.Repo.ServerByName(r.Context(), server)
|
||||
if err != nil {
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
}
|
||||
if rec.Retire != nil {
|
||||
writeError(w, r, errServerRetiring)
|
||||
return
|
||||
}
|
||||
var body struct {
|
||||
TargetNode string `json:"targetNode"`
|
||||
}
|
||||
if err := decodeJSON(w, r, &body); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
if err := a.Distribution.ValidateNode(r.Context(), body.TargetNode); err != nil {
|
||||
writeError(w, r, newError(400, "bad_node", "%v", err))
|
||||
return
|
||||
}
|
||||
op, err := a.Distribution.BeginMigration(r.Context(), server, body.TargetNode, rec.OwnerID)
|
||||
if err != nil {
|
||||
a.writeMigrationError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusAccepted, op)
|
||||
}
|
||||
|
||||
func (a *API) handleMigrationStatus(w http.ResponseWriter, r *http.Request) {
|
||||
if !a.distributedReady(w, r) {
|
||||
return
|
||||
}
|
||||
op, err := a.Distribution.Migration(r.Context(), r.PathValue("name"), r.PathValue("id"))
|
||||
if err != nil {
|
||||
a.writeMigrationError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusOK, op)
|
||||
}
|
||||
|
||||
func (a *API) handleMigrationRetry(w http.ResponseWriter, r *http.Request) {
|
||||
if !a.distributedReady(w, r) {
|
||||
return
|
||||
}
|
||||
rec, err := a.Repo.ServerByName(r.Context(), r.PathValue("name"))
|
||||
if err != nil {
|
||||
a.writeLookupError(w, r, err)
|
||||
return
|
||||
}
|
||||
if rec.Retire != nil {
|
||||
writeError(w, r, errServerRetiring)
|
||||
return
|
||||
}
|
||||
op, err := a.Distribution.RetryMigration(r.Context(), r.PathValue("name"), r.PathValue("id"))
|
||||
if err != nil {
|
||||
a.writeMigrationError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusAccepted, op)
|
||||
}
|
||||
|
||||
func (a *API) writeMigrationError(w http.ResponseWriter, r *http.Request, err error) {
|
||||
switch {
|
||||
case errors.Is(err, distributed.ErrBusy):
|
||||
writeError(w, r, newError(409, "migration_busy", "%v", err))
|
||||
case errors.Is(err, distributed.ErrNotFound), apierrors.IsNotFound(err):
|
||||
writeError(w, r, newError(404, "not_found", "%v", err))
|
||||
case apierrors.IsConflict(err):
|
||||
writeError(w, r, newError(409, "conflict", "retry after refreshing migration status"))
|
||||
default:
|
||||
writeError(w, r, err)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/distributed"
|
||||
)
|
||||
|
||||
type fakeDistribution struct {
|
||||
calls int
|
||||
target, owner string
|
||||
}
|
||||
|
||||
func (d *fakeDistribution) Nodes(context.Context) ([]distributed.Node, error) {
|
||||
d.calls++
|
||||
return []distributed.Node{{Name: "b", Ready: true, Approved: true, Addresses: []string{}}}, nil
|
||||
}
|
||||
func (d *fakeDistribution) ValidateNode(_ context.Context, name string) error {
|
||||
if name != "b" {
|
||||
return errors.New("not approved")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
func (d *fakeDistribution) BeginMigration(_ context.Context, name, target, owner string) (distributed.Operation, error) {
|
||||
d.calls++
|
||||
d.target = target
|
||||
d.owner = owner
|
||||
return distributed.Operation{ID: "op", Server: name, State: "backing_up"}, nil
|
||||
}
|
||||
func (d *fakeDistribution) Migration(context.Context, string, string) (distributed.Operation, error) {
|
||||
d.calls++
|
||||
return distributed.Operation{ID: "op", State: "failed"}, nil
|
||||
}
|
||||
func (d *fakeDistribution) RetryMigration(context.Context, string, string) (distributed.Operation, error) {
|
||||
d.calls++
|
||||
return distributed.Operation{ID: "op", State: "restoring"}, nil
|
||||
}
|
||||
|
||||
func TestDistributedRoutesAreAdministratorOnly(t *testing.T) {
|
||||
for _, role := range []string{"user", "admin"} {
|
||||
t.Run(role, func(t *testing.T) {
|
||||
repo := newFakeRepo()
|
||||
repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner"}
|
||||
a := newTestAPI(repo, newFakeCluster())
|
||||
a.External = staticExternal{p: &Principal{UserID: "owner", Role: role, ViaAdminAccess: role == "admin"}}
|
||||
d := &fakeDistribution{}
|
||||
a.Distribution = d
|
||||
for _, tc := range []struct {
|
||||
method, path, body string
|
||||
status int
|
||||
}{
|
||||
{"GET", "/api/v1/nodes", "", 200},
|
||||
{"POST", "/api/v1/servers/survival/migrations", `{"targetNode":"b"}`, 202},
|
||||
{"GET", "/api/v1/servers/survival/migrations", "", 200},
|
||||
{"GET", "/api/v1/servers/survival/migrations/op", "", 200},
|
||||
{"POST", "/api/v1/servers/survival/migrations/op/retry", "", 202},
|
||||
} {
|
||||
w := do(a.ExternalHandler(), tc.method, tc.path, tc.body, jsonHeader)
|
||||
want := tc.status
|
||||
if role == "user" {
|
||||
want = 403
|
||||
}
|
||||
if w.Code != want {
|
||||
t.Fatalf("%s: %d %s", tc.path, w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
if role == "user" && d.calls != 0 {
|
||||
t.Fatal("owner reached cluster-wide operations")
|
||||
}
|
||||
if role == "admin" && (d.target != "b" || d.owner != "owner") {
|
||||
t.Fatal("wrong migration scope")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -406,6 +406,7 @@ type fleetServerView struct {
|
||||
// value and decodeJSON rejects unknown fields, so a caller can never smuggle
|
||||
// free-form YAML or raw CRD fields through this endpoint.
|
||||
type createServerRequest struct {
|
||||
NodeName string `json:"nodeName,omitempty"`
|
||||
Name string `json:"name"`
|
||||
Subdomain string `json:"subdomain"`
|
||||
DisplayName string `json:"displayName,omitempty"`
|
||||
@@ -445,6 +446,15 @@ func (a *API) handleCreateServer(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
if a.Distribution != nil {
|
||||
if err := a.Distribution.ValidateNode(r.Context(), body.NodeName); err != nil {
|
||||
writeError(w, r, newError(400, "bad_node", "select an approved worker: %v", err))
|
||||
return
|
||||
}
|
||||
} else if body.NodeName != "" {
|
||||
writeError(w, r, newError(400, "bad_node", "node selection requires distributed deployment"))
|
||||
return
|
||||
}
|
||||
// Server name and subdomain both obey the §22 portability rule and the
|
||||
// reservation list.
|
||||
if err := naming.ValidateServerName(body.Name); err != nil {
|
||||
@@ -572,6 +582,7 @@ func (a *API) handleCreateServer(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
|
||||
in := CreateServerInput{
|
||||
NodeName: body.NodeName,
|
||||
Name: body.Name,
|
||||
Subdomain: body.Subdomain,
|
||||
DisplayName: displayName,
|
||||
|
||||
@@ -3,11 +3,13 @@ package api
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
"felis.lolicon.best/internal/placement"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
apierrors "k8s.io/apimachinery/pkg/api/errors"
|
||||
@@ -32,8 +34,10 @@ import (
|
||||
// the direct client c, so a write never works from a copy the watch has not caught
|
||||
// up with yet.
|
||||
type K8sCluster struct {
|
||||
c client.Client
|
||||
namespace string
|
||||
distributed bool
|
||||
controller string
|
||||
c client.Client
|
||||
namespace string
|
||||
// servers serves the fleet-wide reads; nil means c.
|
||||
servers client.Reader
|
||||
// synced reports whether servers has its first full list; nil means no cache.
|
||||
@@ -101,15 +105,28 @@ func (k *K8sCluster) GetServer(ctx context.Context, name string) (*ServerInfo, e
|
||||
// and restore Jobs mount), so existence here is exactly existence at Job mount
|
||||
// time. NotFound is (false, nil): the caller refuses with a specific 409.
|
||||
func (k *K8sCluster) WorldVolumeExists(ctx context.Context, name string) (bool, error) {
|
||||
var pvc corev1.PersistentVolumeClaim
|
||||
err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: naming.WorldPVCName(name)}, &pvc)
|
||||
if apierrors.IsNotFound(err) {
|
||||
return false, nil
|
||||
}
|
||||
if err != nil {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
err := k.getServer(ctx, name, &ms)
|
||||
claim := naming.WorldPVCName(name)
|
||||
if err == nil {
|
||||
claim = ms.WorldPVC()
|
||||
} else if !errors.Is(err, ErrNotFound) {
|
||||
return false, err
|
||||
}
|
||||
return true, nil
|
||||
var pvc corev1.PersistentVolumeClaim
|
||||
if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: claim}, &pvc); err == nil {
|
||||
return true, nil
|
||||
} else if !apierrors.IsNotFound(err) {
|
||||
return false, err
|
||||
}
|
||||
if ms.Name == "" {
|
||||
var retained corev1.PersistentVolumeClaimList
|
||||
if err := k.c.List(ctx, &retained, client.InNamespace(k.namespace), client.MatchingLabels{v1alpha1.LabelServer: name}); err != nil {
|
||||
return false, err
|
||||
}
|
||||
return len(retained.Items) > 0, nil
|
||||
}
|
||||
return false, nil
|
||||
}
|
||||
|
||||
// PodImages lists the image of every container and init container of every pod
|
||||
@@ -183,6 +200,7 @@ func (k *K8sCluster) CreateServer(ctx context.Context, in CreateServerInput) err
|
||||
Namespace: k.namespace,
|
||||
},
|
||||
Spec: v1alpha1.MinecraftServerSpec{
|
||||
NodeName: in.NodeName,
|
||||
Subdomain: in.Subdomain,
|
||||
DisplayName: in.DisplayName,
|
||||
Image: in.Image,
|
||||
@@ -262,6 +280,22 @@ func (k *K8sCluster) startWith(ctx context.Context, name string, retryFailed boo
|
||||
if err := k.getServer(ctx, name, &ms); err != nil {
|
||||
return err
|
||||
}
|
||||
node := ms.Spec.NodeName
|
||||
if node == "" {
|
||||
node = ms.Status.NodeName
|
||||
}
|
||||
if node == "" {
|
||||
node = k.controller
|
||||
}
|
||||
if k.distributed && node != "" {
|
||||
var n corev1.Node
|
||||
if err := k.c.Get(ctx, types.NamespacedName{Name: node}, &n); err != nil {
|
||||
return err
|
||||
}
|
||||
if !placement.Admitted(&n, k.controller) || n.Spec.Unschedulable {
|
||||
return newError(503, "node_unavailable", "execution node is offline")
|
||||
}
|
||||
}
|
||||
kind, held, err := k.maintenanceHolder(ctx, &ms)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -335,6 +369,9 @@ func (k *K8sCluster) ReleaseMaintenance(ctx context.Context, name string) error
|
||||
}
|
||||
return err
|
||||
}
|
||||
if value := ms.Annotations[maintenance.Annotation]; strings.HasPrefix(value, maintenance.KindMigration+"@") {
|
||||
return nil
|
||||
}
|
||||
if _, ok := ms.Annotations[maintenance.Annotation]; !ok {
|
||||
return nil
|
||||
}
|
||||
@@ -445,8 +482,13 @@ func serverInfo(ms *v1alpha1.MinecraftServer) *ServerInfo {
|
||||
memStr = limit.String()
|
||||
}
|
||||
|
||||
node := ms.Spec.NodeName
|
||||
if node == "" {
|
||||
node = ms.Status.NodeName
|
||||
}
|
||||
return &ServerInfo{
|
||||
Name: ms.Name,
|
||||
NodeName: node,
|
||||
Subdomain: ms.Spec.Subdomain,
|
||||
Phase: string(ms.Status.Phase),
|
||||
Ready: ms.Status.Ready,
|
||||
@@ -483,3 +525,11 @@ func idleStopSeconds(ms *v1alpha1.MinecraftServer) int32 {
|
||||
}
|
||||
return ms.Spec.Idle.EmptySecondsBeforeStop
|
||||
}
|
||||
|
||||
func (k *K8sCluster) WithDistributed(enabled bool, controller ...string) *K8sCluster {
|
||||
k.distributed = enabled
|
||||
if len(controller) > 0 {
|
||||
k.controller = controller[0]
|
||||
}
|
||||
return k
|
||||
}
|
||||
@@ -7,9 +7,12 @@ import (
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
"felis.lolicon.best/internal/naming"
|
||||
batchv1 "k8s.io/api/batch/v1"
|
||||
corev1 "k8s.io/api/core/v1"
|
||||
"k8s.io/apimachinery/pkg/api/resource"
|
||||
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
||||
@@ -19,6 +22,37 @@ import (
|
||||
"sigs.k8s.io/controller-runtime/pkg/client/fake"
|
||||
)
|
||||
|
||||
func TestPersistentMigrationBlocksWakeAndWorldOperations(t *testing.T) {
|
||||
scheme := runtime.NewScheme()
|
||||
v1alpha1.AddToScheme(scheme)
|
||||
corev1.AddToScheme(scheme)
|
||||
batchv1.AddToScheme(scheme)
|
||||
s := &v1alpha1.MinecraftServer{ObjectMeta: metav1.ObjectMeta{Name: "survival", Namespace: "minecraft", Annotations: map[string]string{maintenance.Annotation: maintenance.LockValue(maintenance.KindMigration, time.Now().Add(-24*time.Hour))}}, Spec: v1alpha1.MinecraftServerSpec{DesiredState: v1alpha1.DesiredStopped}, Status: v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseStopped}}
|
||||
s.Spec.Storage.ClaimName = "world-survival-migrated"
|
||||
c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(s, &corev1.PersistentVolumeClaim{ObjectMeta: metav1.ObjectMeta{Name: s.WorldPVC(), Namespace: s.Namespace}}).Build()
|
||||
k := NewK8sCluster(c, s.Namespace)
|
||||
ctx := context.Background()
|
||||
if exists, err := k.WorldVolumeExists(ctx, s.Name); err != nil || !exists {
|
||||
t.Fatal("active volume ignored", err)
|
||||
}
|
||||
if err := k.SetDesiredState(ctx, s.Name, v1alpha1.DesiredRunning); !errors.Is(err, ErrMaintenanceInProgress) {
|
||||
t.Fatal("migration admitted wake", err)
|
||||
}
|
||||
for _, kind := range []string{maintenance.KindBackup, maintenance.KindFileWrite, maintenance.KindReap} {
|
||||
if err := k.AcquireMaintenance(ctx, s.Name, kind); !errors.Is(err, ErrMaintenanceInProgress) {
|
||||
t.Fatal("migration admitted", kind, err)
|
||||
}
|
||||
}
|
||||
if err := k.ReleaseMaintenance(ctx, s.Name); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var current v1alpha1.MinecraftServer
|
||||
c.Get(ctx, types.NamespacedName{Namespace: s.Namespace, Name: s.Name}, ¤t)
|
||||
if current.Annotations[maintenance.Annotation] == "" {
|
||||
t.Fatal("normal release cleared persistent migration lock")
|
||||
}
|
||||
}
|
||||
|
||||
// K8sCluster is documented as integration-tested against a live cluster rather
|
||||
// than covered by the hermetic suite, and for most of it that is the right call —
|
||||
// merge-patch semantics are not worth faking. This one test departs from it
|
||||
|
||||
@@ -202,6 +202,9 @@ func (k *K8sJobStatus) PendingRestoreChains(ctx context.Context) ([]RestoreChain
|
||||
snapshot = ChainSnapshotFailed
|
||||
}
|
||||
}
|
||||
if j.Labels["felis.lolicon.best/archive-pending"] == "true" && snapshot == ChainSnapshotSucceeded {
|
||||
snapshot = ChainSnapshotRunning
|
||||
}
|
||||
out = append(out, RestoreChain{
|
||||
Job: j.Name,
|
||||
Server: j.Labels[jobServerLabel],
|
||||
|
||||
Reference in new issue
Block a user