feat: manage distributed node operations from Owner panel
This commit is contained in:
33 files changed
+2417
-26
No files matched your search
@@ -32,6 +32,7 @@ import (
|
||||
// API holds the dependencies shared by every handler.
|
||||
type API struct {
|
||||
Distribution Distribution
|
||||
NodeControl NodeControl
|
||||
Repo Repo
|
||||
Cluster Cluster
|
||||
Internal InternalAuth
|
||||
@@ -650,6 +651,10 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
// Staff can designate their own game identity after panel setup. Players
|
||||
// retain the in-game proof flow above.
|
||||
{Method: "GET", Pattern: "/api/v1/account/link/sources", Admin: true, h: a.handleLinkSources},
|
||||
{Method: "GET", Pattern: "/api/v1/settings/node-control", Owner: true, Admin: true, h: a.handleNodeTasks},
|
||||
{Method: "POST", Pattern: "/api/v1/settings/node-control/tasks", Owner: true, Admin: true, h: a.handleStartNodeTask},
|
||||
{Method: "GET", Pattern: "/api/v1/settings/node-control/tasks/{id}", Owner: true, Admin: true, h: a.handleNodeTask},
|
||||
{Method: "POST", Pattern: "/api/v1/settings/node-control/tasks/{id}/retry", Owner: true, Admin: true, h: a.handleRetryNodeTask},
|
||||
{Method: "GET", Pattern: "/api/v1/settings/wake-policy", Owner: true, Admin: true, h: a.handleGetWakePolicy},
|
||||
{Method: "PUT", Pattern: "/api/v1/settings/wake-policy", Owner: true, Admin: true, h: a.handleSetWakePolicy},
|
||||
{Method: "GET", Pattern: "/api/v1/settings/auth-sources", Owner: true, Admin: true, h: a.handleGetAuthSources},
|
||||
|
||||
@@ -39,6 +39,13 @@ func (a *API) handleNodes(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
|
||||
func (a *API) handleMigration(w http.ResponseWriter, r *http.Request) {
|
||||
if a.NodeControl != nil {
|
||||
if err := NodeMaintenanceGuard(a.NodeControl)(r.Context()); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
if !a.distributedReady(w, r) {
|
||||
return
|
||||
}
|
||||
@@ -88,6 +95,13 @@ func (a *API) handleMigrationStatus(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
|
||||
func (a *API) handleMigrationRetry(w http.ResponseWriter, r *http.Request) {
|
||||
if a.NodeControl != nil {
|
||||
if err := NodeMaintenanceGuard(a.NodeControl)(r.Context()); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
if !a.distributedReady(w, r) {
|
||||
return
|
||||
}
|
||||
|
||||
@@ -0,0 +1,115 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
|
||||
"felis.lolicon.best/internal/nodecontrol"
|
||||
)
|
||||
|
||||
type NodeControl interface {
|
||||
List(context.Context) ([]nodecontrol.Task, error)
|
||||
Get(context.Context, string) (nodecontrol.Task, error)
|
||||
Start(context.Context, nodecontrol.Request, string) (nodecontrol.Task, error)
|
||||
}
|
||||
|
||||
func (a *API) handleNodeTasks(w http.ResponseWriter, r *http.Request) {
|
||||
if a.NodeControl == nil {
|
||||
writeJSON(w, 200, map[string]any{"available": false, "tasks": []nodecontrol.Task{}})
|
||||
return
|
||||
}
|
||||
tasks, err := a.NodeControl.List(r.Context())
|
||||
if err != nil {
|
||||
a.writeNodeControlError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, 200, map[string]any{"available": true, "tasks": tasks})
|
||||
}
|
||||
func (a *API) handleNodeTask(w http.ResponseWriter, r *http.Request) {
|
||||
if !a.nodeControlReady(w, r) {
|
||||
return
|
||||
}
|
||||
task, err := a.NodeControl.Get(r.Context(), r.PathValue("id"))
|
||||
if err != nil {
|
||||
a.writeNodeControlError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, 200, task)
|
||||
}
|
||||
func (a *API) handleStartNodeTask(w http.ResponseWriter, r *http.Request) {
|
||||
if !a.requireReauth(w, r, principalFromContext(r.Context())) || !a.nodeControlReady(w, r) {
|
||||
return
|
||||
}
|
||||
if err := requireJSONContentType(r); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
var req nodecontrol.Request
|
||||
if err := decodeJSON(w, r, &req); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
a.startNodeTask(w, r, req)
|
||||
}
|
||||
func (a *API) handleRetryNodeTask(w http.ResponseWriter, r *http.Request) {
|
||||
if !a.requireReauth(w, r, principalFromContext(r.Context())) || !a.nodeControlReady(w, r) {
|
||||
return
|
||||
}
|
||||
task, err := a.NodeControl.Get(r.Context(), r.PathValue("id"))
|
||||
if err != nil {
|
||||
a.writeNodeControlError(w, r, err)
|
||||
return
|
||||
}
|
||||
if task.State != "failed" {
|
||||
writeError(w, r, newError(409, "conflict", "only failed node tasks can be retried"))
|
||||
return
|
||||
}
|
||||
a.startNodeTask(w, r, task.Request)
|
||||
}
|
||||
func (a *API) startNodeTask(w http.ResponseWriter, r *http.Request, req nodecontrol.Request) {
|
||||
if err := req.Validate(); err != nil {
|
||||
writeError(w, r, newError(400, "bad_request", "%v", err))
|
||||
return
|
||||
}
|
||||
task, err := a.NodeControl.Start(r.Context(), req, principalFromContext(r.Context()).UserID)
|
||||
if err != nil {
|
||||
a.writeNodeControlError(w, r, err)
|
||||
return
|
||||
}
|
||||
a.audit(r, "platform.node."+req.Action, task.ID)
|
||||
writeJSON(w, http.StatusAccepted, task)
|
||||
}
|
||||
func (a *API) nodeControlReady(w http.ResponseWriter, r *http.Request) bool {
|
||||
if a.NodeControl == nil {
|
||||
a.writeNodeControlError(w, r, errors.New("node-control not configured"))
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
func (a *API) writeNodeControlError(w http.ResponseWriter, r *http.Request, err error) {
|
||||
switch {
|
||||
case errors.Is(err, nodecontrol.ErrBusy):
|
||||
writeError(w, r, newError(409, "node_operation_busy", "a node operation is already running"))
|
||||
case errors.Is(err, nodecontrol.ErrNotFound):
|
||||
writeError(w, r, newError(404, "not_found", "node task not found"))
|
||||
default:
|
||||
writeError(w, r, newError(503, "node_control_unavailable", "host node service is unavailable; inspect felis-node-control.service"))
|
||||
}
|
||||
}
|
||||
|
||||
// Fail closed for starts when the configured host service cannot report its maintenance state.
|
||||
func NodeMaintenanceGuard(control NodeControl) func(context.Context) error {
|
||||
return func(ctx context.Context) error {
|
||||
tasks, err := control.List(ctx)
|
||||
if err != nil {
|
||||
return newError(503, "node_control_unavailable", "host node service is unavailable")
|
||||
}
|
||||
for _, task := range tasks {
|
||||
if task.State == "running" {
|
||||
return newError(409, "node_operation_busy", "node maintenance is in progress")
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,122 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
"felis.lolicon.best/internal/nodecontrol"
|
||||
)
|
||||
|
||||
type fakeNodeControl struct {
|
||||
tasks []nodecontrol.Task
|
||||
calls int
|
||||
err error
|
||||
request nodecontrol.Request
|
||||
actor string
|
||||
}
|
||||
|
||||
func (f *fakeNodeControl) List(context.Context) ([]nodecontrol.Task, error) { return f.tasks, f.err }
|
||||
func (f *fakeNodeControl) Get(context.Context, string) (nodecontrol.Task, error) {
|
||||
if f.err != nil {
|
||||
return nodecontrol.Task{}, f.err
|
||||
}
|
||||
if len(f.tasks) == 0 {
|
||||
return nodecontrol.Task{}, nodecontrol.ErrNotFound
|
||||
}
|
||||
return f.tasks[0], nil
|
||||
}
|
||||
func (f *fakeNodeControl) Start(_ context.Context, r nodecontrol.Request, actor string) (nodecontrol.Task, error) {
|
||||
f.calls++
|
||||
f.request = r
|
||||
f.actor = actor
|
||||
return nodecontrol.Task{ID: "task", Request: r, State: "running"}, f.err
|
||||
}
|
||||
|
||||
const nodeTaskPath = "/api/v1/settings/node-control/tasks"
|
||||
const nodeTaskBody = `{"action":"approve","name":"worker-01","sshTarget":"worker-01","confirmMaintenance":true}`
|
||||
|
||||
func TestNodeControlOwnerBoundaryAndRetries(t *testing.T) {
|
||||
for _, role := range []string{"user", "admin", "owner"} {
|
||||
a := newTestAPI(newFakeRepo(), newFakeCluster())
|
||||
a.External = staticExternal{p: &Principal{UserID: "actor", Role: role, ViaAdminAccess: role != "user"}}
|
||||
control := &fakeNodeControl{tasks: []nodecontrol.Task{{ID: "task", State: "failed", Request: nodecontrol.Request{Action: "approve", Name: "worker-01", SSHTarget: "worker-01", ConfirmMaintenance: true}}}}
|
||||
a.NodeControl = control
|
||||
for _, tc := range []struct {
|
||||
method, path, body string
|
||||
want int
|
||||
}{{"GET", "/api/v1/settings/node-control", "", 200}, {"GET", nodeTaskPath + "/task", "", 200}, {"POST", nodeTaskPath, nodeTaskBody, 202}, {"POST", nodeTaskPath + "/task/retry", "", 202}} {
|
||||
w := do(a.ExternalHandler(), tc.method, tc.path, tc.body, jsonHeader)
|
||||
want := tc.want
|
||||
if role != "owner" {
|
||||
want = 403
|
||||
}
|
||||
if w.Code != want {
|
||||
t.Fatalf("%s %s: %d %s", role, tc.path, w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
if role != "owner" && control.calls != 0 {
|
||||
t.Fatal("nonowner reached root operations")
|
||||
}
|
||||
if role == "owner" && (control.actor != "actor" || control.request.Name != "worker-01") {
|
||||
t.Fatal("actor/scope not preserved")
|
||||
}
|
||||
}
|
||||
}
|
||||
func TestNodeControlErrorsAndStartGuard(t *testing.T) {
|
||||
a := newTestAPI(newFakeRepo(), newFakeCluster())
|
||||
a.External = staticExternal{p: &Principal{UserID: "owner", Role: "owner", ViaAdminAccess: true}}
|
||||
control := &fakeNodeControl{}
|
||||
a.NodeControl = control
|
||||
w := do(a.ExternalHandler(), "POST", nodeTaskPath, `{"action":"approve","name":"x","sshTarget":"-exec","confirmMaintenance":true}`, jsonHeader)
|
||||
if w.Code != 400 || control.calls != 0 {
|
||||
t.Fatal(w.Code, w.Body.String())
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
err error
|
||||
code int
|
||||
}{{nodecontrol.ErrBusy, 409}, {errors.New("private host detail"), 503}} {
|
||||
control.err = tc.err
|
||||
w := do(a.ExternalHandler(), "POST", nodeTaskPath, nodeTaskBody, jsonHeader)
|
||||
if w.Code != tc.code {
|
||||
t.Fatal(w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
guard := NodeMaintenanceGuard(control)
|
||||
if guard(context.Background()) == nil {
|
||||
t.Fatal("host failure permitted start")
|
||||
}
|
||||
control.err = nil
|
||||
control.tasks = []nodecontrol.Task{{State: "running"}}
|
||||
if guard(context.Background()) == nil {
|
||||
t.Fatal("maintenance permitted start")
|
||||
}
|
||||
control.tasks[0].State = "failed"
|
||||
if err := guard(context.Background()); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
control.tasks[0].State = "succeeded"
|
||||
w = do(a.ExternalHandler(), "POST", nodeTaskPath+"/task/retry", "", jsonHeader)
|
||||
if w.Code != 409 {
|
||||
t.Fatal(w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestNodeControlRequiresFreshOwnerProof(t *testing.T) {
|
||||
repo := newFakeRepo()
|
||||
a := newTestAPI(repo, newFakeCluster())
|
||||
p := &Principal{UserID: "owner", Role: "owner", ViaAdminAccess: true, ViaSession: true}
|
||||
repo.passkeyCreds["owner-key"] = PasskeyCredential{ID: "owner-key", UserID: "owner", UserVerified: true}
|
||||
a.External = staticExternal{p: p}
|
||||
control := &fakeNodeControl{}
|
||||
a.NodeControl = control
|
||||
w := do(a.ExternalHandler(), "POST", nodeTaskPath, nodeTaskBody, jsonHeader)
|
||||
if w.Code != 403 || decodeErr(t, w) != "reauth_required" || control.calls != 0 {
|
||||
t.Fatal(w.Code, w.Body.String())
|
||||
}
|
||||
p.ReauthAt = a.now()
|
||||
w = do(a.ExternalHandler(), "POST", nodeTaskPath, nodeTaskBody, jsonHeader)
|
||||
if w.Code != 202 || control.calls != 1 {
|
||||
t.Fatal(w.Code, w.Body.String())
|
||||
}
|
||||
}
|
||||
@@ -36,10 +36,12 @@ import (
|
||||
// the direct client c, so a write never works from a copy the watch has not caught
|
||||
// up with yet.
|
||||
type K8sCluster struct {
|
||||
distributed bool
|
||||
controller string
|
||||
c client.Client
|
||||
namespace string
|
||||
// NodeMaintenanceGuard checks host maintenance for all manual and scheduled starts.
|
||||
NodeMaintenanceGuard func(context.Context) error
|
||||
distributed bool
|
||||
controller string
|
||||
c client.Client
|
||||
namespace string
|
||||
// servers serves the fleet-wide reads; nil means c.
|
||||
servers client.Reader
|
||||
// synced reports whether servers has its first full list; nil means no cache.
|
||||
@@ -308,6 +310,11 @@ func (k *K8sCluster) RestartServer(ctx context.Context, name string) error {
|
||||
}
|
||||
|
||||
func (k *K8sCluster) startWith(ctx context.Context, name, annotation string) error {
|
||||
if k.NodeMaintenanceGuard != nil {
|
||||
if err := k.NodeMaintenanceGuard(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := k.getServer(ctx, name, &ms); err != nil {
|
||||
@@ -350,6 +357,11 @@ func (k *K8sCluster) startWith(ctx context.Context, name, annotation string) err
|
||||
}
|
||||
ms.Annotations[annotation] = k.clock().UTC().Format(time.RFC3339Nano)
|
||||
}
|
||||
if k.NodeMaintenanceGuard != nil {
|
||||
if err := k.NodeMaintenanceGuard(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return k.c.Patch(ctx, &ms, patch)
|
||||
})
|
||||
}
|
||||
@@ -362,6 +374,12 @@ func (k *K8sCluster) startWith(ctx context.Context, name, annotation string) err
|
||||
// checked; the caller creates its Job and then calls ReleaseMaintenance, after
|
||||
// which the Job itself is the lock.
|
||||
func (k *K8sCluster) AcquireMaintenance(ctx context.Context, name, kind string) error {
|
||||
if k.NodeMaintenanceGuard != nil {
|
||||
if err := k.NodeMaintenanceGuard(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
return retry.RetryOnConflict(retry.DefaultRetry, func() error {
|
||||
var ms v1alpha1.MinecraftServer
|
||||
if err := k.getServer(ctx, name, &ms); err != nil {
|
||||
|
||||
@@ -459,3 +459,26 @@ func TestResourcePatchesKeepWhatTheyLeaveOut(t *testing.T) {
|
||||
t.Fatalf("cleared cpu: limits %v, want the CPU limit gone and memory 8Gi kept", after.Resources.Limits)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHostNodeMaintenanceBlocksStartsAndWorldJobs(t *testing.T) {
|
||||
scheme := runtime.NewScheme()
|
||||
v1alpha1.AddToScheme(scheme)
|
||||
server := &v1alpha1.MinecraftServer{ObjectMeta: metav1.ObjectMeta{Name: "survival", Namespace: "minecraft"}, Spec: v1alpha1.MinecraftServerSpec{DesiredState: v1alpha1.DesiredStopped}, Status: v1alpha1.MinecraftServerStatus{Phase: v1alpha1.PhaseStopped}}
|
||||
c := fake.NewClientBuilder().WithScheme(scheme).WithObjects(server).Build()
|
||||
k := NewK8sCluster(c, server.Namespace)
|
||||
blocked := newError(409, "node_operation_busy", "host task running")
|
||||
k.NodeMaintenanceGuard = func(context.Context) error { return blocked }
|
||||
ctx := context.Background()
|
||||
for _, operation := range []func() error{func() error { return k.SetDesiredState(ctx, server.Name, v1alpha1.DesiredRunning) }, func() error { return k.RetryStart(ctx, server.Name) }, func() error { return k.AcquireMaintenance(ctx, server.Name, maintenance.KindBackup) }} {
|
||||
if err := operation(); err != blocked {
|
||||
t.Fatal("host maintenance bypassed", err)
|
||||
}
|
||||
}
|
||||
if err := k.SetDesiredState(ctx, server.Name, v1alpha1.DesiredStopped); err != nil {
|
||||
t.Fatal("stop entry blocked", err)
|
||||
}
|
||||
var got v1alpha1.MinecraftServer
|
||||
if err := c.Get(ctx, types.NamespacedName{Namespace: server.Namespace, Name: server.Name}, &got); err != nil || got.Spec.DesiredState != v1alpha1.DesiredStopped || len(got.Annotations) != 0 {
|
||||
t.Fatal("state changed during host maintenance", got, err)
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user