Unverified Commit a7d3ca4c authored by Lemon-miaow's avatar Lemon-miaow
Browse files

fix(panel): diagnose runtime failures and add owner recovery controls

parent d7ae1e66
Loading
Loading
Loading
Loading
+116 −0
Changes for docs/openapi.yaml: 116 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -331,6 +331,25 @@ components:
          description: Optional Yggdrasil API base for role lookup; empty infers it from the standard hasJoined suffix.
        enabled: { type: boolean }

    StartupStatus:
      type: object
      required: [stage, logsAvailable]
      properties:
        stage: { type: string, enum: [creating, scheduling, preparing, booting, failed] }
        reason: { type: string }
        message: { type: string }
        startedAt: { type: string, format: date-time }
        logsAvailable: { type: boolean }

    WakePolicySettings:
      type: object
      required: [maxRunningServers, wakeCooldownSeconds, revision, managed]
      properties:
        maxRunningServers: { type: integer, minimum: 0, maximum: 10000 }
        wakeCooldownSeconds: { type: integer, minimum: 0, maximum: 3600 }
        revision: { type: string }
        managed: { type: boolean }

    AuthSourcesSettings:
      type: object
      required: [sources, revision, managed]
@@ -551,6 +570,7 @@ components:
      description: Status projection of one server (internal/api/cluster.go ServerInfo).
      required: [name, subdomain, phase, ready, playersOnline, playersMax, idleStopSeconds]
      properties:
        startup: { $ref: '#/components/schemas/StartupStatus', description: Private runtime and startup diagnostics for authorized server managers. }
        nodeName: { type: string, description: Execution node; legacy servers report the observed node. }
        name: { type: string }
        subdomain: { type: string }
@@ -2539,6 +2559,52 @@ paths:
        '404':
          $ref: '#/components/responses/NotFound'

  /api/v1/servers/{name}/emergency-stop:
    post:
      tags: [servers]
      operationId: emergencyStop
      summary: Scale an owned game workload to zero independently of Operator and RCON.
      description: Owner only; fresh authentication and exact name confirmation required. A 202 acknowledges scaling, not process exit. Normal Pod termination grace is retained.
      x-felis-face: [external]
      x-felis-tier: owner
      security: [{ sessionCookie: [] }]
      parameters:
        - { name: name, in: path, required: true, schema: { type: string } }
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              additionalProperties: false
              required: [confirm]
              properties:
                confirm: { type: string }
      responses:
        '202':
          description: Stop intent persisted and workload scaled to zero; await Pod termination.
          content:
            application/json:
              schema:
                type: object
                required: [name, desiredState]
                properties:
                  name: { type: string }
                  desiredState: { type: string, const: Stopped }
        '400':
          $ref: '#/components/responses/BadRequest'
        '401':
          $ref: '#/components/responses/Unauthorized'
        '403':
          $ref: '#/components/responses/Forbidden'
        '404':
          $ref: '#/components/responses/NotFound'

        '409':
          $ref: '#/components/responses/Conflict'
        '503':
          $ref: '#/components/responses/ServiceUnavailable'

  /api/v1/servers/{name}/restart:
    post:
      tags: [servers]
@@ -4265,6 +4331,56 @@ paths:
        '401':
          $ref: '#/components/responses/Unauthorized'

  /api/v1/settings/wake-policy:
    get:
      tags: [account]
      operationId: getWakePolicy
      summary: Read platform startup policy (Owner).
      x-felis-face: [external]
      x-felis-tier: owner
      security: [{ sessionCookie: [] }]
      responses:
        '200':
          description: Current startup policy and revision.
          content:
            application/json:
              schema: { $ref: '#/components/schemas/WakePolicySettings' }
        '401': { $ref: '#/components/responses/Unauthorized' }
        '403': { $ref: '#/components/responses/Forbidden' }
    put:
      tags: [account]
      operationId: setWakePolicy
      summary: Save platform startup policy (Owner, fresh reauthentication).
      description: Atomically persists an override shared by all replicas. The admission limit applies to panel, in-game and scheduled starts without restarting. It does not stop existing servers or reserve memory. Wake cooldown applies to panel and in-game wakes.
      x-felis-face: [external]
      x-felis-tier: owner
      security: [{ sessionCookie: [] }]
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              required: [maxRunningServers, wakeCooldownSeconds, revision]
              properties:
                maxRunningServers: { type: integer, minimum: 0, maximum: 10000 }
                wakeCooldownSeconds: { type: integer, minimum: 0, maximum: 3600 }
                revision: { type: string }
      responses:
        '200':
          description: Saved policy and new revision.
          content:
            application/json:
              schema: { $ref: '#/components/schemas/WakePolicySettings' }
        '400': { $ref: '#/components/responses/BadRequest' }
        '401': { $ref: '#/components/responses/Unauthorized' }
        '403': { $ref: '#/components/responses/Forbidden' }
        '409':
          description: The policy changed; reload before saving.
          content:
            application/json:
              schema: { $ref: '#/components/schemas/Error' }

  /api/v1/settings/auth-sources:
    get:
      tags: [account]
+35 −1
Changes for docs/troubleshooting.md: 35 added lines, 1 removed line.
Original line number Diff line number Diff line
@@ -326,9 +326,23 @@ per-server cooldown → global running cap**. Map the API result:
| `403` | `forbidden` | `autostartPolicy=allowlist` and UUID not allowlisted, or `ownerOnly` and caller is not owner | Add the UUID / claim the server / set `autostartPolicy=public` |
| `409` | `maintenance_in_progress` | A restore, backup or file change holds the server's world volume (§3b) | Wait for the Job to finish |
| `409` | `world_reclaiming` | The idle reaper is archiving the world (§3b item 3); afterwards the server is released with an empty world | Nothing to wait for; the old world stays in the archive |
| `429` | (cooldown) | Wake retried within the 30s per-server `WakeCooldown` | Wait out the cooldown |
| `429` | (cooldown) | Wake retried within the configured per-server cooldown | Wait out the cooldown |
| `503` | `at_capacity` | Global `MaxRunningServers` cap reached | Stop another server or raise the cap |

Owners can change the running-server limit and wake cooldown in **Platform
settings**. Saved values apply on all API replicas without a restart; the running
limit also applies to scheduled starts. Zero disables the corresponding control.
The running limit is an admission check, not an atomic reservation: concurrent
requests can briefly exceed it, and lowering it never stops existing servers.

A server marked Starting need not have launched Java. Its console now reads Pod
scheduling and container state: `Unschedulable` / insufficient memory means it is
waiting for resources, while image-pull failures and container exits show their
reasons. Logs attach once the container can produce them. A status request that
cannot complete within 12 seconds marks the last view as stale, shows its last
successful read time, and retries without overlapping polls. Do not infer game
readiness from an old Starting label or a disconnected log stream.

[GO-TESTED: `handlers_internal_wake_test.go`, cooldown, running-cap shape.] The
operator's RCON probe — **not** the wake call — is the authoritative readiness
gate; the proxy polls `GET /api/v1/internal/servers/{name}/status` every ~2s and
@@ -3406,3 +3420,23 @@ PG-TESTED: `TestScheduleStoreRunCAS`, `TestDueSchedules`, `TestSchedulesFollowTh
| How long sessions, codes and audit rows are kept; export audit rows | §17 |
| Files page: a change, upload or unzip refused (`file_exists`, `bad_path`, `too_large`, `upload_staging_full`, `upload_not_found`, `volume_full`, `archive_unsafe`, `job_failed`, `files_timeout`) | §18 |
| A scheduled task shows `skipped`, `missed` or `failed`; a task switched itself off after an owner change | §19 |

### 面板无法确认状态与应急停止

控制 API 超时或失联时,最后一次读取的状态仅作参考;“启动中”不能表示
游戏进程仍然健康。控制台保留已有日志,显示状态已过期和最后读取时间,
并自动重试。首次读取失败也提供 Owner 可见的主机侧恢复步骤。

普通停止由 Operator 执行。Owner 可以在服务器控制台的“更多操作”中选择
“应急停止”,输入服务器名并重新验证身份。它先持久保存 `desiredState=Stopped`,
再通过 Kubernetes 的 scale 子资源将该 MinecraftServer 自己拥有的 StatefulSet
缩到零,绕过游戏 RCON 和 Operator,但保留正常 Pod 终止时间,不删除世界。
如果缩容失败,已保存的停止意图不会撤销;界面明确提示结果未确认。
受理请求不等于游戏已停止:状态读取独立检查 StatefulSet 副本目标和游戏 Pod,
在副本目标仍非零或游戏 Pod 仍运行/退出时显示“停止中”。

控制 API 自己离线时,浏览器不能凭空执行集群操作。Owner 可展开主机侧恢复
步骤,在部署主机依次保存停止意图、缩容并检查 Pod(命名空间不是 `minecraft`
时需要替换)。节点离线时还需检查节点和容器运行时;强制删除 Pod 不是
进程已经退出的证据。不要靠强制删除世界、Pod 或 PVC 解决控制端失联。
应急停止入口的 Kubernetes 权限需要随平台更新应用新的 RBAC。
+9 −9
Changes for internal/api/api.go: 9 added lines, 9 removed lines.
Original line number Diff line number Diff line
@@ -177,12 +177,9 @@ type API struct {
	SubmitCreateCooldown time.Duration
	SubmitUploadCooldown time.Duration

	// MaxRunningServers caps how many servers may be desired-Running cluster-wide
	// (spec §9.1: the concurrency-上限 lever hanging on the same wake chokepoint as
	// cooldown and autostartPolicy). Zero — the default — disables it: §9.2 wires
	// only autostartPolicy + cooldown as active wake gates, so this lever ships
	// inert, exactly like a zero WakeCooldown, and a deployment opts in by setting
	// a positive value. Enforced via withinRunningCap on the wake path.
	// MaxRunningServers is the default desired-Running admission limit (spec §9.1).
	// Zero disables it. A saved Owner wake policy overrides this and WakeCooldown;
	// panel, in-game and scheduled starts share withinRunningCap.
	MaxRunningServers int

	// MaxStreamsPerPrincipal caps how many concurrent Server-Sent Event streams
@@ -538,6 +535,7 @@ func (a *API) externalAPIRoutes() []apiRoute {
		// App-auth tier: operations on your own servers (spec §14).
		{Method: "POST", Pattern: "/api/v1/servers/{name}/wake", h: a.handleWake},
		{Method: "POST", Pattern: "/api/v1/servers/{name}/stop", h: a.handleStop},
		{Method: "POST", Pattern: "/api/v1/servers/{name}/emergency-stop", h: a.handleEmergencyStop, Admin: true, Owner: true},
		{Method: "POST", Pattern: "/api/v1/servers/{name}/claim", h: a.handleClaim},
		// Console write (spec §8 写=RCON): owner/admin-gated inside the handler, so
		// it sits in the app-tier block (操作自己服 → app 鉴权), not behind adminOnly.
@@ -652,6 +650,8 @@ func (a *API) externalAPIRoutes() []apiRoute {
		// Staff can designate their own game identity after panel setup. Players
		// retain the in-game proof flow above.
		{Method: "GET", Pattern: "/api/v1/account/link/sources", Admin: true, h: a.handleLinkSources},
		{Method: "GET", Pattern: "/api/v1/settings/wake-policy", Owner: true, Admin: true, h: a.handleGetWakePolicy},
		{Method: "PUT", Pattern: "/api/v1/settings/wake-policy", Owner: true, Admin: true, h: a.handleSetWakePolicy},
		{Method: "GET", Pattern: "/api/v1/settings/auth-sources", Owner: true, Admin: true, h: a.handleGetAuthSources},
		{Method: "PUT", Pattern: "/api/v1/settings/auth-sources", Owner: true, Admin: true, h: a.handleSetAuthSources},
		{Method: "POST", Pattern: "/api/v1/settings/auth-sources/test", Owner: true, Admin: true, h: a.handleTestAuthSource},
@@ -1205,8 +1205,8 @@ func (l *streamLimiter) release(key string) {
// can momentarily exceed the cap. That is acceptable because the operator
// reconcile is idempotent and the §18 reaper / §9.3 quota bound steady-state
// load; the cap exists to refuse an obvious flood, not to hold a hard ceiling.
func (a *API) withinRunningCap(ctx context.Context, info *ServerInfo) (bool, error) {
	if a.MaxRunningServers <= 0 || naming.IsSystemServer(info.Name) {
func (a *API) withinRunningCap(ctx context.Context, info *ServerInfo, cap int) (bool, error) {
	if cap <= 0 || naming.IsSystemServer(info.Name) {
		return true, nil
	}
	if info.DesiredState == string(v1alpha1.DesiredRunning) {
@@ -1222,5 +1222,5 @@ func (a *API) withinRunningCap(ctx context.Context, info *ServerInfo) (bool, err
			running++
		}
	}
	return running < a.MaxRunningServers, nil
	return running < cap, nil
}
+2 −0
Changes for internal/api/cluster.go: 2 added lines, 0 removed lines.
Original line number Diff line number Diff line
@@ -57,6 +57,8 @@ type ServerInfo struct {
	// Resources is the spec's pod resource block. It stays off the wire; a spec
	// patch reads it so the fields the admin left out keep their values.
	Resources corev1.ResourceRequirements `json:"-"`
	// Startup explains scheduling and container state independently of the CR phase.
	Startup *StartupStatus `json:"startup,omitempty"`
}

// CreateServerInput is the validated, structured create-server form (spec §15).
+92 −0
Changes for internal/api/emergency_stop.go: 92 added lines, 0 removed lines.
Original line number Diff line number Diff line
package api

import (
	"context"
	"fmt"
	"net/http"
	"time"

	"felis.lolicon.best/internal/apis/felis/v1alpha1"
	appsv1 "k8s.io/api/apps/v1"
	autoscalingv1 "k8s.io/api/autoscaling/v1"
	apierrors "k8s.io/apimachinery/pkg/api/errors"
	metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
	"k8s.io/apimachinery/pkg/types"
	"k8s.io/client-go/util/retry"
	"sigs.k8s.io/controller-runtime/pkg/client"
)

// EmergencyStop records the stop intent before scaling the owned workload. It
// bypasses RCON and operator reconciliation, retaining normal Pod shutdown grace.
func (k *K8sCluster) EmergencyStop(ctx context.Context, name string) error {
	if err := k.SetDesiredState(ctx, name, v1alpha1.DesiredStopped); err != nil {
		return err
	}
	return retry.RetryOnConflict(retry.DefaultRetry, func() error {
		var ms v1alpha1.MinecraftServer
		if err := k.getServer(ctx, name, &ms); err != nil {
			return err
		}
		if ms.Spec.DesiredState != v1alpha1.DesiredStopped {
			return newError(http.StatusConflict, "conflict", "stop intent changed; retry emergency stop")
		}
		var sts appsv1.StatefulSet
		if err := k.c.Get(ctx, types.NamespacedName{Namespace: k.namespace, Name: name}, &sts); err != nil {
			if apierrors.IsNotFound(err) {
				return nil
			}
			return err
		}
		owner := metav1.GetControllerOf(&sts)
		if owner == nil || ms.UID == "" || owner.UID != ms.UID || owner.Kind != "MinecraftServer" || owner.APIVersion != v1alpha1.GroupVersion.String() {
			return fmt.Errorf("stop intent saved, but workload ownership could not be verified")
		}
		scale := &autoscalingv1.Scale{ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: k.namespace, ResourceVersion: sts.ResourceVersion}, Spec: autoscalingv1.ScaleSpec{Replicas: 0}}
		if err := k.c.SubResource("scale").Update(ctx, &sts, client.WithSubResourceBody(scale)); err != nil {
			return fmt.Errorf("stop intent saved, but workload scale-down was not confirmed: %w", err)
		}
		return nil
	})
}

func (a *API) handleEmergencyStop(w http.ResponseWriter, r *http.Request) {
	if !a.requireReauth(w, r, principalFromContext(r.Context())) {
		return
	}
	name := r.PathValue("name")
	if err := validateManagedServerName(r, name); err != nil {
		writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name"))
		return
	}
	if err := requireJSONContentType(r); err != nil {
		writeError(w, r, err)
		return
	}
	var body struct {
		Confirm string `json:"confirm"`
	}
	if err := decodeJSON(w, r, &body); err != nil {
		writeError(w, r, err)
		return
	}
	if body.Confirm != name {
		writeError(w, r, newError(http.StatusBadRequest, "bad_request", "type the exact server name to confirm"))
		return
	}
	stopper, ok := a.Cluster.(interface {
		EmergencyStop(context.Context, string) error
	})
	if !ok {
		writeError(w, r, newError(http.StatusServiceUnavailable, "unavailable", "this cluster does not support direct emergency shutdown"))
		return
	}
	ctx, cancel := context.WithTimeout(r.Context(), 15*time.Second)
	defer cancel()
	if err := stopper.EmergencyStop(ctx, name); err != nil {
		a.audit(r, "emergency_stop.failed", name)
		a.writeLookupError(w, r, err)
		return
	}
	a.audit(r, "emergency_stop", name)
	writeJSON(w, http.StatusAccepted, map[string]any{"name": name, "desiredState": "Stopped"})
}
Loading