From c1025274e127ae8116271a3c89c963a5bbf87585 Mon Sep 17 00:00:00 2001 From: Lemon-miaow Date: Sat, 26 Sep 2026 21:41:01 +0800 Subject: [PATCH] =?UTF-8?q?fix(velocity):=20=E7=AD=89=E5=BE=85=E9=98=9F?= =?UTF-8?q?=E5=88=97=E8=B7=9F=E9=9A=8F=E6=9C=8D=E7=9A=84=E5=90=AF=E5=8A=A8?= =?UTF-8?q?=E8=BF=9B=E5=BA=A6=EF=BC=8C=E8=87=AA=E5=8A=A8=E9=87=8D=E8=AF=95?= =?UTF-8?q?=E6=9C=9F=E9=97=B4=E4=B8=80=E7=9B=B4=E7=AD=89=E5=B9=B6=E6=AF=8F?= =?UTF-8?q?=E5=88=86=E9=92=9F=E6=8A=A5=E8=BF=9B=E5=BA=A6=EF=BC=8C=E6=94=BE?= =?UTF-8?q?=E5=BC=83=E6=88=96=E8=A2=AB=E5=81=9C=E6=AD=A2=E6=97=B6=E8=AF=B4?= =?UTF-8?q?=E6=98=8E=E5=8E=9F=E5=9B=A0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/openapi.yaml | 10 ++ internal/api/cluster.go | 6 + internal/api/k8scluster.go | 2 + internal/api/k8scluster_test.go | 30 ++++ .../felis/v1alpha1/minecraftserver_types.go | 28 +++- .../apis/felis/v1alpha1/startgaveup_test.go | 44 ++++++ internal/operator/autorestart_test.go | 7 + internal/operator/reconciler.go | 6 +- panel/src/lib/openapi.gen.ts | 7 + panel/src/lib/types.ts | 5 + plugins/README.md | 2 +- .../best/lolicon/felis/link/ServerView.java | 32 +++- .../lolicon/felis/velocity/WaitingRouter.java | 148 ++++++++++++++---- .../best/lolicon/felis/velocity/Fakes.java | 23 ++- .../felis/velocity/WaitingRouterTest.java | 102 ++++++++++++ 15 files changed, 413 insertions(+), 39 deletions(-) create mode 100644 internal/apis/felis/v1alpha1/startgaveup_test.go diff --git a/docs/openapi.yaml b/docs/openapi.yaml index 1fae4c6..170192d 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -488,6 +488,16 @@ components: Present and true when the CR carries the label felis.lolicon.best/forwarding=legacy. The proxy then forwards this server's players BungeeCord-style in the handshake address instead of modern forwarding (Felis-Legacy Velocity fork only). + autoRestarts: + type: integer + format: int32 + description: Present when non-zero; how often the operator recreated the pod of this start after it timed out (at most 3). + startGaveUp: + type: boolean + description: >- + Present and true for a Failed server no automatic retry will bring up: its + start timed out with the retries spent, or its spec is invalid. A Failed + server without it is still in its restart backoff and may come up on its own. FleetServer: description: One row of the fleet-wide admin read (internal/api/handlers_user.go fleetServerView). diff --git a/internal/api/cluster.go b/internal/api/cluster.go index ec63ba9..eb0f2a4 100644 --- a/internal/api/cluster.go +++ b/internal/api/cluster.go @@ -37,6 +37,12 @@ type ServerInfo struct { // LegacyForwarding mirrors the CR's forwarding=legacy label: the proxy // forwards this server's players in the handshake address (#15). LegacyForwarding bool `json:"legacyForwarding,omitempty"` + // AutoRestarts is how often the operator has recreated the pod of this start + // after it timed out; StartGaveUp is true once no automatic retry is coming + // (v1alpha1.StartGaveUp). A Failed server without StartGaveUp is still in its + // restart backoff and may yet come up on its own. + AutoRestarts int32 `json:"autoRestarts,omitempty"` + StartGaveUp bool `json:"startGaveUp,omitempty"` } // CreateServerInput is the validated, structured create-server form (spec §15). diff --git a/internal/api/k8scluster.go b/internal/api/k8scluster.go index 0b872a8..014e7a3 100644 --- a/internal/api/k8scluster.go +++ b/internal/api/k8scluster.go @@ -447,6 +447,8 @@ func serverInfo(ms *v1alpha1.MinecraftServer) *ServerInfo { PlayerCountUnknown: ms.Status.Phase == v1alpha1.PhaseRunning && meta.IsStatusConditionFalse(ms.Status.Conditions, v1alpha1.ConditionPlayersCounted), LegacyForwarding: ms.Labels[v1alpha1.LabelForwarding] == v1alpha1.ForwardingLegacy, + AutoRestarts: ms.Status.AutoRestarts, + StartGaveUp: v1alpha1.StartGaveUp(&ms.Status), } } diff --git a/internal/api/k8scluster_test.go b/internal/api/k8scluster_test.go index ee5dbe8..f8f0ea5 100644 --- a/internal/api/k8scluster_test.go +++ b/internal/api/k8scluster_test.go @@ -290,3 +290,33 @@ func TestServerListCarriesLegacyForwarding(t *testing.T) { } } } + +// A Failed server reaches the proxy and the panel with whether the operator will +// still retry it: the proxy keeps a waiting player through the restart backoff and +// lets them go once the retries are spent. +func TestServerInfoCarriesStartGaveUp(t *testing.T) { + failed := func(name string, restarts int32) *v1alpha1.MinecraftServer { + ms := testServer(name, name) + now := metav1.Now() + ms.Status = v1alpha1.MinecraftServerStatus{ + Phase: v1alpha1.PhaseFailed, + AutoRestarts: restarts, + StartRequestedAt: &now, + Conditions: []metav1.Condition{{Type: v1alpha1.ConditionReady, Status: metav1.ConditionFalse, + Reason: v1alpha1.ReasonStartupTimeout}}, + } + return ms + } + retrying := serverInfo(failed("retrying", 1)) + if retrying.StartGaveUp || retrying.AutoRestarts != 1 { + t.Fatalf("retrying: startGaveUp=%v autoRestarts=%d, want false 1", retrying.StartGaveUp, retrying.AutoRestarts) + } + spent := serverInfo(failed("spent", v1alpha1.MaxAutoRestarts)) + b, err := json.Marshal(spent) + if err != nil { + t.Fatal(err) + } + if !spent.StartGaveUp || !strings.Contains(string(b), `"startGaveUp":true`) { + t.Fatalf("spent: startGaveUp=%v JSON %s, want true", spent.StartGaveUp, b) + } +} diff --git a/internal/apis/felis/v1alpha1/minecraftserver_types.go b/internal/apis/felis/v1alpha1/minecraftserver_types.go index a058d46..7a28eae 100644 --- a/internal/apis/felis/v1alpha1/minecraftserver_types.go +++ b/internal/apis/felis/v1alpha1/minecraftserver_types.go @@ -2,6 +2,7 @@ package v1alpha1 import ( corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" ) @@ -91,6 +92,31 @@ const ( ConditionPlayersCounted = "PlayersCounted" ) +// Ready-condition reasons of a start that timed out: the pod never passed its TCP +// readiness, or RCON never answered. Both are retried by recreating the pod, at +// most MaxAutoRestarts times with a doubling backoff. +const ( + ReasonStartupTimeout = "StartupTimeout" + ReasonReadinessTimeout = "ReadinessTimeout" +) + +// MaxAutoRestarts bounds how often the operator retries a timed-out start. +const MaxAutoRestarts = 3 + +// StartGaveUp reports a Failed server that no automatic retry will bring up: its +// start timed out with the retries spent, or it failed for a reason the operator +// never retries (an invalid spec). Only a person moves it on. A Failed server still +// inside its restart backoff has not given up: the operator recreates its pod when +// the backoff runs out, and whoever waits on it should keep waiting. +func StartGaveUp(s *MinecraftServerStatus) bool { + if s.Phase != PhaseFailed { + return false + } + c := meta.FindStatusCondition(s.Conditions, ConditionReady) + timedOut := c != nil && (c.Reason == ReasonStartupTimeout || c.Reason == ReasonReadinessTimeout) + return !timedOut || s.AutoRestarts >= MaxAutoRestarts || s.StartRequestedAt == nil +} + // +kubebuilder:object:root=true // +kubebuilder:subresource:status @@ -310,7 +336,7 @@ type MinecraftServerStatus struct { // the server becomes unoccupied. EmptySince *metav1.Time `json:"emptySince,omitempty"` // AutoRestarts counts how often the operator recreated the pod of a start - // that timed out (at most 3, with a doubling backoff); reaching Ready or + // that timed out (at most MaxAutoRestarts, with a doubling backoff); reaching Ready or // stopping resets it. AutoRestarts int32 `json:"autoRestarts,omitempty"` // LastAutoRestartAt is when the operator last recreated the pod. diff --git a/internal/apis/felis/v1alpha1/startgaveup_test.go b/internal/apis/felis/v1alpha1/startgaveup_test.go new file mode 100644 index 0000000..a93915e --- /dev/null +++ b/internal/apis/felis/v1alpha1/startgaveup_test.go @@ -0,0 +1,44 @@ +package v1alpha1_test + +import ( + "testing" + + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" +) + +// StartGaveUp tells whoever waits on a start whether to keep waiting. The +// operator's own timeout path pins the retrying cases (autorestart_test.go); +// these are the failures it never retries. +func TestStartGaveUp(t *testing.T) { + anchor := metav1.Now() + status := func(phase v1alpha1.Phase, reason string, restarts int32, anchored bool) *v1alpha1.MinecraftServerStatus { + s := &v1alpha1.MinecraftServerStatus{Phase: phase, AutoRestarts: restarts} + if reason != "" { + s.Conditions = []metav1.Condition{{Type: v1alpha1.ConditionReady, Status: metav1.ConditionFalse, Reason: reason}} + } + if anchored { + s.StartRequestedAt = &anchor + } + return s + } + cases := []struct { + name string + s *v1alpha1.MinecraftServerStatus + want bool + }{ + {"starting", status(v1alpha1.PhaseStarting, "PodNotReady", 0, true), false}, + {"timed out, retries left", status(v1alpha1.PhaseFailed, v1alpha1.ReasonStartupTimeout, 2, true), false}, + {"rcon timed out, retries left", status(v1alpha1.PhaseFailed, v1alpha1.ReasonReadinessTimeout, 0, true), false}, + {"timed out, retries spent", status(v1alpha1.PhaseFailed, v1alpha1.ReasonStartupTimeout, v1alpha1.MaxAutoRestarts, true), true}, + {"invalid spec is never retried", status(v1alpha1.PhaseFailed, "InvalidSpec", 0, true), true}, + {"no start anchor, nothing to retry from", status(v1alpha1.PhaseFailed, v1alpha1.ReasonStartupTimeout, 0, false), true}, + {"failed with no condition", status(v1alpha1.PhaseFailed, "", 0, true), true}, + } + for _, c := range cases { + if got := v1alpha1.StartGaveUp(c.s); got != c.want { + t.Errorf("%s: StartGaveUp = %v, want %v", c.name, got, c.want) + } + } +} diff --git a/internal/operator/autorestart_test.go b/internal/operator/autorestart_test.go index b03ed58..b7e491e 100644 --- a/internal/operator/autorestart_test.go +++ b/internal/operator/autorestart_test.go @@ -52,6 +52,8 @@ func TestTimedOutStartRecreatesThePodWithBackoff(t *testing.T) { for i, a := range attempts { if s := at(a[1]); s.Status.Phase != v1alpha1.PhaseFailed || !podExists() || s.Status.AutoRestarts != int32(i) { t.Fatalf("attempt %d before due: phase=%s pod=%v restarts=%d", i+1, s.Status.Phase, podExists(), s.Status.AutoRestarts) + } else if v1alpha1.StartGaveUp(&s.Status) { + t.Fatalf("attempt %d before due: reported as given up while a retry is coming", i+1) } s := at(a[2]) if podExists() { @@ -76,6 +78,9 @@ func TestTimedOutStartRecreatesThePodWithBackoff(t *testing.T) { if s.Status.Phase != v1alpha1.PhaseFailed || s.Status.AutoRestarts != 3 || !podExists() { t.Fatalf("after three attempts: phase=%s restarts=%d pod=%v", s.Status.Phase, s.Status.AutoRestarts, podExists()) } + if !v1alpha1.StartGaveUp(&s.Status) { + t.Fatal("after three attempts: not reported as given up") + } } // An RCON channel that never answers on a TCP-ready pod gets the same retry. @@ -93,6 +98,8 @@ func TestReadinessTimeoutRecreatesThePod(t *testing.T) { reconcile(t, r, "survival") if s := getServer(t, c, "survival"); s.Status.Phase != v1alpha1.PhaseFailed || s.Status.AutoRestarts != 0 { t.Fatalf("before due: phase=%s restarts=%d", s.Status.Phase, s.Status.AutoRestarts) + } else if v1alpha1.StartGaveUp(&s.Status) { + t.Fatal("before due: reported as given up while a retry is coming") } clock = base.Add(90 * time.Second) reconcile(t, r, "survival") diff --git a/internal/operator/reconciler.go b/internal/operator/reconciler.go index b0066d7..5bedfa1 100644 --- a/internal/operator/reconciler.go +++ b/internal/operator/reconciler.go @@ -58,7 +58,7 @@ const ( // maxAutoRestarts bounds how often a timed-out start is retried by // recreating its pod; autoRestartBaseBackoff is the first wait, doubling // per attempt. - maxAutoRestarts = 3 + maxAutoRestarts = v1alpha1.MaxAutoRestarts autoRestartBaseBackoff = time.Minute defaultReadinessTimeoutSec = 300 ) @@ -300,7 +300,7 @@ func (r *Reconciler) reconcileRunning(ctx context.Context, server *v1alpha1.Mine if current.Status.ReadyReplicas < 1 { r.markStarting(server, "PodNotReady", "waiting for pod TCP readiness") if r.startupTimedOut(server) { - r.markFailed(server, "StartupTimeout", "pod did not become ready within startup timeout") + r.markFailed(server, v1alpha1.ReasonStartupTimeout, "pod did not become ready within startup timeout") if err := r.recoverFailedStart(ctx, server, startupTimeout(server)); err != nil { return ctrl.Result{}, err } @@ -339,7 +339,7 @@ func (r *Reconciler) reconcileRunning(ctx context.Context, server *v1alpha1.Mine } r.markStarting(server, "RconNotReachable", err.Error()) if r.readinessTimedOut(server) { - r.markFailed(server, "ReadinessTimeout", "RCON probe did not succeed within readiness timeout") + r.markFailed(server, v1alpha1.ReasonReadinessTimeout, "RCON probe did not succeed within readiness timeout") if err := r.recoverFailedStart(ctx, server, readinessTimeout(server)); err != nil { return ctrl.Result{}, err } diff --git a/panel/src/lib/openapi.gen.ts b/panel/src/lib/openapi.gen.ts index deb815a..810cc48 100644 --- a/panel/src/lib/openapi.gen.ts +++ b/panel/src/lib/openapi.gen.ts @@ -2348,6 +2348,13 @@ export interface components { playerCountUnknown?: boolean; /** @description Present and true when the CR carries the label felis.lolicon.best/forwarding=legacy. The proxy then forwards this server's players BungeeCord-style in the handshake address instead of modern forwarding (Felis-Legacy Velocity fork only). */ legacyForwarding?: boolean; + /** + * Format: int32 + * @description Present when non-zero; how often the operator recreated the pod of this start after it timed out (at most 3). + */ + autoRestarts?: number; + /** @description Present and true for a Failed server no automatic retry will bring up: its start timed out with the retries spent, or its spec is invalid. A Failed server without it is still in its restart backoff and may come up on its own. */ + startGaveUp?: boolean; }; /** @description One row of the fleet-wide admin read (internal/api/handlers_user.go fleetServerView). */ FleetServer: components["schemas"]["ServerInfo"] & { diff --git a/panel/src/lib/types.ts b/panel/src/lib/types.ts index ba8f377..0aa0587 100644 --- a/panel/src/lib/types.ts +++ b/panel/src/lib/types.ts @@ -65,6 +65,11 @@ export interface ServerStatus { /** True when the CR is labelled forwarding=legacy: the proxy forwards this * server's players BungeeCord-style (a 1.8 backend behind ViaVersion). */ legacyForwarding?: boolean; + /** How often the operator recreated the pod of this start after it timed out. */ + autoRestarts?: number; + /** True for a Failed server no automatic retry will bring up; a Failed server + * without it is still in its restart backoff. */ + startGaveUp?: boolean; } /** WhitelistResult projects GET /servers/{name}/access/whitelist (spec §7 access). diff --git a/plugins/README.md b/plugins/README.md index 62f0814..7bf55d3 100644 --- a/plugins/README.md +++ b/plugins/README.md @@ -168,7 +168,7 @@ What it does when routing is active: | ------- | -------- | | Backend registry | Polls `GET /api/v1/servers` every 15 s and reconciles Velocity's dynamic registry. A failed poll **keeps existing registrations** — a control-plane blip never deregisters live backends. The API advertises each backend Service's host-routable ClusterIP, avoiding cluster-DNS names on the host-run proxy. | | Join (`PlayerChooseInitialServerEvent`) | Resolves `subdomain.` and remembers the target, but every fresh connection still enters `login`. When the login gate requests its post-auth lobby transfer, Velocity re-checks link status: a ready remembered target is selected immediately; an asleep target is woken and queued from the lobby. | -| Waiting queue | One scheduled drain every 2 s polls status once per distinct waited-on server; a waiter drops out on transfer, on the player leaving, or after a 120 s timeout. | +| Waiting queue | One scheduled drain every 2 s polls status once per distinct waited-on server. A waiter stays while its server is on the way up (starting, or Failed inside the operator's restart backoff), hears a progress line each minute, and drops out on transfer, on the player leaving, when the start is given up (`startGaveUp`) or the server is stopped, after 120 s without an answer from felis-api, or at a one-hour backstop. | | Wake gate | The wake is `POST /api/v1/internal/servers/{name}/wake` keyed on the player's online-mode UUID. **403** (policy refused) tells the player and stops; **429** (wake already in flight) keeps waiting. | | Server-list ping (`ProxyPingEvent`) | Answers from the cached lifecycle view with a phase-aware MOTD (online / starting / sleeping) — **read-only, never wakes** anything. Mirroring each backend's own MOTD by background-pinging ready servers is a later slice. | | Join report (`ServerConnectedEvent`) | Reports real joins to a felis backend via `POST …/join-event`, so the reaper sees activity and the player is auto-added to the server allowlist. | diff --git a/plugins/shared/src/main/java/best/lolicon/felis/link/ServerView.java b/plugins/shared/src/main/java/best/lolicon/felis/link/ServerView.java index e661e1f..88b61b8 100644 --- a/plugins/shared/src/main/java/best/lolicon/felis/link/ServerView.java +++ b/plugins/shared/src/main/java/best/lolicon/felis/link/ServerView.java @@ -35,6 +35,8 @@ public final class ServerView { private final int playersOnline; private final int playersMax; private final boolean legacyForwarding; + private final int autoRestarts; + private final boolean startGaveUp; public ServerView(String name, String subdomain, String phase, boolean ready, String autostartPolicy, String desiredState, String endpointMode, @@ -47,6 +49,14 @@ public final class ServerView { String autostartPolicy, String desiredState, String endpointMode, String endpointAddress, int playersOnline, int playersMax, boolean legacyForwarding) { + this(name, subdomain, phase, ready, autostartPolicy, desiredState, endpointMode, + endpointAddress, playersOnline, playersMax, legacyForwarding, 0, false); + } + + public ServerView(String name, String subdomain, String phase, boolean ready, + String autostartPolicy, String desiredState, String endpointMode, + String endpointAddress, int playersOnline, int playersMax, + boolean legacyForwarding, int autoRestarts, boolean startGaveUp) { this.name = name; this.subdomain = subdomain; this.phase = phase; @@ -58,6 +68,8 @@ public final class ServerView { this.playersOnline = playersOnline; this.playersMax = playersMax; this.legacyForwarding = legacyForwarding; + this.autoRestarts = autoRestarts; + this.startGaveUp = startGaveUp; } /** fromJson builds a view from a parsed felis-api object, tolerating absent fields. */ @@ -73,7 +85,9 @@ public final class ServerView { str(o, "endpointAddress"), intval(o, "playersOnline"), intval(o, "playersMax"), - bool(o, "legacyForwarding")); + bool(o, "legacyForwarding"), + intval(o, "autoRestarts"), + bool(o, "startGaveUp")); } /** @@ -124,6 +138,8 @@ public final class ServerView { b.append(",\"playersOnline\":").append(v.playersOnline); b.append(",\"playersMax\":").append(v.playersMax); b.append(",\"legacyForwarding\":").append(v.legacyForwarding); + b.append(",\"autoRestarts\":").append(v.autoRestarts); + b.append(",\"startGaveUp\":").append(v.startGaveUp); b.append('}'); } return b.append("]}").toString(); @@ -188,6 +204,20 @@ public final class ServerView { return legacyForwarding; } + /** autoRestarts is how often the operator recreated the pod of a start that timed out. */ + public int autoRestarts() { + return autoRestarts; + } + + /** + * startGaveUp is true for a Failed server no automatic retry will bring up (the + * restarts are spent, or its spec is invalid). A Failed server without it is in + * its restart backoff and may still come up on its own. + */ + public boolean startGaveUp() { + return startGaveUp; + } + private static String str(Map o, String key) { Object v = o.get(key); return v instanceof String ? (String) v : null; diff --git a/plugins/velocity/src/main/java/best/lolicon/felis/velocity/WaitingRouter.java b/plugins/velocity/src/main/java/best/lolicon/felis/velocity/WaitingRouter.java index 91486e5..0a6f90f 100644 --- a/plugins/velocity/src/main/java/best/lolicon/felis/velocity/WaitingRouter.java +++ b/plugins/velocity/src/main/java/best/lolicon/felis/velocity/WaitingRouter.java @@ -27,6 +27,7 @@ import java.util.Optional; import java.util.UUID; import java.util.concurrent.ConcurrentHashMap; import java.util.concurrent.atomic.AtomicBoolean; +import java.util.function.LongSupplier; /** * WaitingRouter implements the §11 domain-autostart routing loop and its waiting @@ -42,8 +43,10 @@ import java.util.concurrent.atomic.AtomicBoolean; * *

The queue is drained by {@link #tick()}, scheduled by the plugin on the async * pool. Each tick polls felis-api once per distinct waited-on server and, when one - * reports ready, transfers everyone waiting on it. A waiter drops out when it times - * out, when the player leaves the proxy, or on a successful transfer. + * reports ready, transfers everyone waiting on it. A waiter stays as long as its + * server is on the way up, through the operator's restart backoff, and drops out on a + * successful transfer, when the player leaves the proxy, when the start is given up or + * the server stopped, or when felis-api stops answering for the wait window. * *

Every transition out of login is checked against felis-api's link status, and * command/menu queue entries are checked the same way. The wake is then gated @@ -58,7 +61,19 @@ import java.util.concurrent.atomic.AtomicBoolean; * outage on a recent positive answer for the same UUID and otherwise fails closed. */ public final class WaitingRouter { + // A waiter stays while its server is on the way up — starting, or Failed inside the + // operator's restart backoff — and every poll that says so renews this window. It + // runs out only when felis-api stops answering or the server stops heading for + // Running. A modpack's cold start (a 300 s budget, then up to three recreated pods + // with a 1, 2, 4 min backoff) outlasts any fixed wait, and a waiter dropped early + // was never moved in when the server did come up. private static final long WAIT_TIMEOUT_MILLIS = 120_000L; + // The backstop for a start whose status never moves (an operator that is down): + // twice the default budget, 4 × 300 s plus 7 min of backoff. + private static final long MAX_WAIT_MILLIS = 60 * 60_000L; + // A long wait tells the player how it is going this often, so it is not silence. + private static final long PROGRESS_NOTICE_MILLIS = 60_000L; + private static final String DESIRED_STOPPED = "Stopped"; // How long a positive link answer can stand in for felis-api while it is down. private static final long LINK_GRACE_MILLIS = 10 * 60_000L; // The login gate re-sends its release with backoff (and during a felis-api outage @@ -86,6 +101,8 @@ public final class WaitingRouter { // felis:control face can tell the player's GUI the backend is ready. Null until // the ControlChannel is wired in at proxy init; set once, read on the tick pool. private volatile MenuTransferListener menuListener; + // The waiting queue's clock; tests move it to walk a long start. + private volatile LongSupplier clock = System::currentTimeMillis; WaitingRouter(ProxyServer proxy, Logger log, FelisApiClient api, ServerRegistry registry, FelisVelocityPlugin plugin, String loginServer, String lobbyServer) { @@ -99,6 +116,10 @@ public final class WaitingRouter { this.links = new LinkGate(api::linkStatus, LINK_GRACE_MILLIS, System::currentTimeMillis); } + void setClock(LongSupplier clock) { + this.clock = clock; + } + /** pruneLinks bounds the LinkGate's fallback records; called on the refresh loop. */ void pruneLinks() { links.prune(); @@ -399,8 +420,8 @@ public final class WaitingRouter { } private void drain() { - long now = System.currentTimeMillis(); - Map readyCache = new HashMap<>(); // one status poll per distinct server + long now = clock.getAsLong(); + Map> polled = new HashMap<>(); // one status poll per distinct server for (Map.Entry e : new ArrayList<>(waiting.entrySet())) { UUID id = e.getKey(); Waiter w = e.getValue(); @@ -411,31 +432,22 @@ public final class WaitingRouter { } Player player = po.get(); boolean zh = FelisVelocityPlugin.zh(player); - if (now > w.deadlineMillis) { - waiting.remove(id); - player.sendMessage(Component.text( - zh ? "「" + w.serverName + "」启动耗时超出预期。你可以稍后在大厅重试。" - : "« " + w.serverName + " » is taking longer than expected to start. " - + "You can try again from the lobby later.", NamedTextColor.YELLOW)); - continue; + Optional poll = polled.get(w.serverName); + if (poll == null) { + poll = poll(w.serverName); + polled.put(w.serverName, poll); } - Boolean ready = readyCache.get(w.serverName); - if (ready == null) { - try { - ServerView status = api.serverStatus(w.serverName); - ready = status.ready(); - // The registry refreshes every 15 s; a server that just came up may - // still be registered at its old address, or not at all. Register - // what this poll reports before transferring anyone to it. - if (ready) { - registry.observe(status); - } - } catch (LinkException ex) { - ready = Boolean.FALSE; // transient → keep waiting until the deadline + ServerView status = poll.orElse(null); + if (status == null || !status.ready()) { + if (status != null && !stillComing(player, zh, w, status, now)) { + waiting.remove(id); + } else if (now > w.deadlineMillis || now - w.sinceMillis > MAX_WAIT_MILLIS) { + waiting.remove(id); + player.sendMessage(Component.text( + zh ? "「" + w.serverName + "」启动耗时超出预期。你可以稍后在大厅重试。" + : "« " + w.serverName + " » is taking longer than expected to start. " + + "You can try again from the lobby later.", NamedTextColor.YELLOW)); } - readyCache.put(w.serverName, ready); - } - if (!ready) { continue; } Optional backend = registry.registered(w.serverName); @@ -470,6 +482,71 @@ public final class WaitingRouter { } } + // poll asks felis-api how one waited-on server is doing; empty when it does not + // answer, which the waiters ride out until their window closes. + private Optional poll(String serverName) { + try { + ServerView status = api.serverStatus(serverName); + // The registry refreshes every 15 s; a server that just came up may still + // be registered at its old address, or not at all. Register what this poll + // reports before transferring anyone to it. + if (status.ready()) { + registry.observe(status); + } + return Optional.of(status); + } catch (LinkException ex) { + return Optional.empty(); + } + } + + // stillComing reads a not-ready poll for one waiter. While the server is heading + // for Running it renews the waiter's window, and now and then tells the player how + // the start is going. When nothing is coming — the retries are spent, or somebody + // stopped the server — it says so and returns false. + private boolean stillComing(Player player, boolean zh, Waiter w, ServerView status, long now) { + if (status.startGaveUp()) { + player.sendMessage(Component.text( + zh ? "「" + w.serverName + "」启动失败,自动重试也已用完。服主可以在面板查看日志后重新启动。" + : "« " + w.serverName + " » failed to start and its automatic retries are spent. " + + "The owner can check its log in the panel and start it again.", + NamedTextColor.RED)); + return false; + } + if (DESIRED_STOPPED.equalsIgnoreCase(status.desiredState())) { + // felis-api reads servers from an informer cache, so the first poll after + // the wake can still show the old desired state; two in a row are a stop. + if (w.stopSeen) { + player.sendMessage(Component.text( + zh ? "「" + w.serverName + "」已被停止,不再为你排队。" + : "« " + w.serverName + " » was stopped, so you're no longer waiting for it.", + NamedTextColor.YELLOW)); + return false; + } + w.stopSeen = true; + return true; + } + w.stopSeen = false; + w.deadlineMillis = now + WAIT_TIMEOUT_MILLIS; + if (status.autoRestarts() > w.restartsSeen) { + w.restartsSeen = status.autoRestarts(); + w.noticedMillis = now; + player.sendMessage(Component.text( + zh ? "「" + w.serverName + "」启动超时,正在自动重试(第 " + w.restartsSeen + " 次)……" + : "« " + w.serverName + " » timed out starting; retrying automatically (attempt " + + w.restartsSeen + ")…", + NamedTextColor.YELLOW)); + } else if (now - w.noticedMillis >= PROGRESS_NOTICE_MILLIS) { + w.noticedMillis = now; + long minutes = (now - w.sinceMillis) / 60_000L; + player.sendMessage(Component.text( + zh ? "「" + w.serverName + "」仍在启动(已等 " + minutes + " 分钟),就绪后会自动把你传送过去。" + : "« " + w.serverName + " » is still starting (" + minutes + " min so far); " + + "you'll be moved in when it's ready.", + NamedTextColor.GRAY)); + } + return true; + } + private void authorizeAndWait(Player player, String serverName, boolean fromMenu) { UUID id = player.getUniqueId(); boolean zh = FelisVelocityPlugin.zh(player); @@ -608,8 +685,7 @@ public final class WaitingRouter { : "Starting « " + serverName + " » — you'll be moved in automatically.", NamedTextColor.GRAY)); } - waiting.put(id, new Waiter( - serverName, System.currentTimeMillis() + WAIT_TIMEOUT_MILLIS, fromMenu)); + waiting.put(id, new Waiter(serverName, clock.getAsLong(), fromMenu)); } private void logWakeFailure(Player player, String serverName, boolean zh, LinkException e) { @@ -659,13 +735,21 @@ public final class WaitingRouter { private static final class Waiter { final String serverName; - final long deadlineMillis; final boolean fromMenu; // true → notify the felis:control face on transfer + final long sinceMillis; + // Only the drain touches these, one tick at a time (the ticking flag orders + // the ticks), so they need no further synchronization. + long deadlineMillis; + long noticedMillis; + int restartsSeen; + boolean stopSeen; - Waiter(String serverName, long deadlineMillis, boolean fromMenu) { + Waiter(String serverName, long nowMillis, boolean fromMenu) { this.serverName = serverName; - this.deadlineMillis = deadlineMillis; this.fromMenu = fromMenu; + this.sinceMillis = nowMillis; + this.deadlineMillis = nowMillis + WAIT_TIMEOUT_MILLIS; + this.noticedMillis = nowMillis; } } diff --git a/plugins/velocity/test/best/lolicon/felis/velocity/Fakes.java b/plugins/velocity/test/best/lolicon/felis/velocity/Fakes.java index 3d55052..db8e387 100644 --- a/plugins/velocity/test/best/lolicon/felis/velocity/Fakes.java +++ b/plugins/velocity/test/best/lolicon/felis/velocity/Fakes.java @@ -446,6 +446,17 @@ final class Fakes { final Map ready = new ConcurrentHashMap<>(); /** address is the direct endpoint a server reports while it is ready. */ final Map address = new ConcurrentHashMap<>(); + /** + * phase, desired, restarts and gaveUp are how a not-ready server's start is going, + * as the status route reports it (phase Stopped and desiredState Running unless + * set: the wake the waiter followed has asked for Running). + */ + final Map phase = new ConcurrentHashMap<>(); + final Map desired = new ConcurrentHashMap<>(); + final Map restarts = new ConcurrentHashMap<>(); + final Set gaveUp = ConcurrentHashMap.newKeySet(); + /** statusDown makes the status route answer 500. */ + volatile boolean statusDown; /** wakeError maps a server to "status code" (e.g. "403 forbidden"). */ final Map wakeError = new ConcurrentHashMap<>(); volatile int joinStatus = 204; @@ -493,6 +504,10 @@ final class Fakes { String action = parts.length > 1 ? parts[1] : ""; switch (method + " " + action) { case "GET status": + if (statusDown) { + reply(ex, 500, "{\"error\":{\"code\":\"internal\",\"message\":\"down\"}}"); + return; + } reply(ex, 200, status(name, ready.getOrDefault(name, false))); return; case "POST wake": @@ -538,8 +553,14 @@ final class Fakes { String endpoint = up ? "\"endpointMode\":\"direct\"" + (addr == null ? "" : ",\"endpointAddress\":\"" + addr + "\"") : "\"endpointMode\":\"fallback\",\"endpointAddress\":\"login\""; + // Like the real ServerInfo, autoRestarts and startGaveUp are left out at 0/false. + int restarts = this.restarts.getOrDefault(name, 0); return "{\"name\":\"" + name + "\",\"subdomain\":\"" + name + "\",\"phase\":\"" - + (up ? "Running" : "Stopped") + "\",\"ready\":" + up + "," + endpoint + "}"; + + (up ? "Running" : phase.getOrDefault(name, "Stopped")) + "\",\"ready\":" + up + + ",\"desiredState\":\"" + desired.getOrDefault(name, "Running") + "\"" + + (restarts == 0 ? "" : ",\"autoRestarts\":" + restarts) + + (gaveUp.contains(name) ? ",\"startGaveUp\":true" : "") + + "," + endpoint + "}"; } private static void reply(HttpExchange ex, int status, String body) throws IOException { diff --git a/plugins/velocity/test/best/lolicon/felis/velocity/WaitingRouterTest.java b/plugins/velocity/test/best/lolicon/felis/velocity/WaitingRouterTest.java index 196f80b..7f17a77 100644 --- a/plugins/velocity/test/best/lolicon/felis/velocity/WaitingRouterTest.java +++ b/plugins/velocity/test/best/lolicon/felis/velocity/WaitingRouterTest.java @@ -17,6 +17,7 @@ import java.util.ArrayList; import java.util.Collections; import java.util.List; import java.util.Locale; +import java.util.concurrent.atomic.AtomicLong; /** * WaitingRouterTest drives the real WaitingRouter, ServerRegistry, FelisApiClient and @@ -70,6 +71,7 @@ public final class WaitingRouterTest { loginGate(); wakeRefusals(); queue(); + longStart(); menuAndCommands(); joins(); disconnectAndRelease(); @@ -91,6 +93,10 @@ public final class WaitingRouterTest { view("delta", false, "10.43.0.6:25565"), view("epsilon", false, "10.43.0.7:25565"), view("zeta", false, "10.43.0.8:25565"), + view("eta", false, "10.43.0.21:25565"), + view("theta", false, "10.43.0.22:25565"), + view("iota", false, "10.43.0.23:25565"), + view("kappa", false, "10.43.0.24:25565"), view("fresh", false, null))); if (withGone) { list.add(view("gone", false, "10.43.0.9:25565")); @@ -324,6 +330,102 @@ public final class WaitingRouterTest { assertEq("registered: queue empty", 0, router.waitingCount()); } + // A modpack's cold start outlasts any fixed wait: the operator gives a start 300 s, + // then recreates the pod up to three times with a 1, 2, 4 min backoff. The waiter + // follows the server's own progress instead of a clock, and hears how it is going. + private static void longStart() { + assertEq("long start: queue empty to begin with", 0, router.waitingCount()); + AtomicLong now = new AtomicLong(1_000_000_000L); + router.setClock(now::get); + try { + // Starting, then Failed inside the backoff, a recreated pod, and up at 25 min. + Fakes.FakePlayer slow = player("eta.mc.test", true); + choose(slow); + release(slow); + api.phase.put("eta", "Starting"); + advance(now, 60, 30); + assertEq("slow start: told how it is going", true, slow.said("« eta » is still starting (1 min so far)")); + api.phase.put("eta", "Failed"); // the 300 s budget ran out; backoff until 6 min + advance(now, 6 * 60, 30); + assertEq("in the backoff: still waiting well past two minutes", 1, router.waitingCount()); + api.phase.put("eta", "Starting"); + api.restarts.put("eta", 1); + advance(now, 30, 30); + assertEq("recreated pod: told", true, slow.said("« eta » timed out starting; retrying automatically (attempt 1)")); + advance(now, 18 * 60, 30); + assertEq("25 minutes in: still waiting", 1, router.waitingCount()); + assertEq("25 minutes in: never given up on", false, slow.said("taking longer than expected")); + int notices = count(slow.messages, "is still starting"); + assertEq("about one progress line a minute, not one a poll", true, notices >= 20 && notices <= 25); + api.ready.put("eta", true); + router.tick(); + assertEq("up at last: moved in", List.of("eta"), List.copyOf(slow.connects)); + assertEq("up at last: queue empty", 0, router.waitingCount()); + + // The retries are spent: nothing is coming, and the player hears why. + Fakes.FakePlayer spent = player("theta.mc.test", true); + choose(spent); + release(spent); + api.phase.put("theta", "Failed"); + api.restarts.put("theta", 3); + api.gaveUp.add("theta"); + advance(now, 2, 2); + assertEq("given up: dropped", 0, router.waitingCount()); + assertEq("given up: told", true, spent.said("« theta » failed to start and its automatic retries are spent")); + assertEq("given up: not moved", 0, spent.connects.size()); + + // Somebody stops the server. One poll can still show the desired state from + // before the wake (the api reads an informer cache); two in a row are a stop. + Fakes.FakePlayer stopped = player("iota.mc.test", true); + choose(stopped); + release(stopped); + api.desired.put("iota", "Stopped"); + advance(now, 2, 2); + assertEq("one stopped poll: still waiting", 1, router.waitingCount()); + api.desired.put("iota", "Running"); + advance(now, 2, 2); + api.desired.put("iota", "Stopped"); + advance(now, 2, 2); + assertEq("a lag blip does not count toward the stop", 1, router.waitingCount()); + advance(now, 2, 2); + assertEq("stopped: dropped", 0, router.waitingCount()); + assertEq("stopped: told", true, stopped.said("« iota » was stopped, so you're no longer waiting for it")); + + // felis-api stops answering: the waiter rides it out for the wait window only. + Fakes.FakePlayer blind = player("kappa.mc.test", true); + choose(blind); + release(blind); + api.statusDown = true; + advance(now, 110, 10); + assertEq("api down: still waiting inside the window", 1, router.waitingCount()); + advance(now, 20, 10); + assertEq("api down: dropped after the window", 0, router.waitingCount()); + assertEq("api down: told", true, blind.said("« kappa » is taking longer than expected")); + api.statusDown = false; + + // A status that never moves (an operator that is down) ends at the backstop. + Fakes.FakePlayer stuck = player("kappa.mc.test", true); + choose(stuck); + release(stuck); + api.phase.put("kappa", "Starting"); + advance(now, 59 * 60, 60); + assertEq("stuck: still waiting before the hour", 1, router.waitingCount()); + advance(now, 2 * 60, 60); + assertEq("stuck: dropped at the backstop", 0, router.waitingCount()); + assertEq("stuck: told", true, stuck.said("« kappa » is taking longer than expected")); + } finally { + router.setClock(System::currentTimeMillis); + } + } + + // advance moves the waiting queue's clock on by seconds, draining every step. + private static void advance(AtomicLong now, int seconds, int step) { + for (int s = 0; s < seconds; s += step) { + now.addAndGet(step * 1000L); + router.tick(); + } + } + private static void menuAndCommands() { List notified = Collections.synchronizedList(new ArrayList<>()); router.setMenuTransferListener((player, server) -> notified.add(player.getUsername() + "@" + server));