fix(velocity): 等待队列跟随服的启动进度,自动重试期间一直等并每分钟报进度,放弃或被停止时说明原因

This commit is contained in:
Lemon-miaow committed 2026-09-26 21:41:01 +08:00
1 parent df8b09788c
commit c1025274e1
15 files changed
+409 -35

No files matched your search

+10
View File
@@ -488,6 +488,16 @@ components:
Present and true when the CR carries the label felis.lolicon.best/forwarding=legacy.
The proxy then forwards this server's players BungeeCord-style in the handshake
address instead of modern forwarding (Felis-Legacy Velocity fork only).
autoRestarts:
type: integer
format: int32
description: Present when non-zero; how often the operator recreated the pod of this start after it timed out (at most 3).
startGaveUp:
type: boolean
description: >-
Present and true for a Failed server no automatic retry will bring up: its
start timed out with the retries spent, or its spec is invalid. A Failed
server without it is still in its restart backoff and may come up on its own.
FleetServer:
description: One row of the fleet-wide admin read (internal/api/handlers_user.go fleetServerView).
+6
View File
@@ -37,6 +37,12 @@ type ServerInfo struct {
// LegacyForwarding mirrors the CR's forwarding=legacy label: the proxy
// forwards this server's players in the handshake address (#15).
LegacyForwarding bool `json:"legacyForwarding,omitempty"`
// AutoRestarts is how often the operator has recreated the pod of this start
// after it timed out; StartGaveUp is true once no automatic retry is coming
// (v1alpha1.StartGaveUp). A Failed server without StartGaveUp is still in its
// restart backoff and may yet come up on its own.
AutoRestarts int32 `json:"autoRestarts,omitempty"`
StartGaveUp bool `json:"startGaveUp,omitempty"`
}
// CreateServerInput is the validated, structured create-server form (spec §15).
+2
View File
@@ -447,6 +447,8 @@ func serverInfo(ms *v1alpha1.MinecraftServer) *ServerInfo {
PlayerCountUnknown: ms.Status.Phase == v1alpha1.PhaseRunning &&
meta.IsStatusConditionFalse(ms.Status.Conditions, v1alpha1.ConditionPlayersCounted),
LegacyForwarding: ms.Labels[v1alpha1.LabelForwarding] == v1alpha1.ForwardingLegacy,
AutoRestarts: ms.Status.AutoRestarts,
StartGaveUp: v1alpha1.StartGaveUp(&ms.Status),
}
}
+30
View File
@@ -290,3 +290,33 @@ func TestServerListCarriesLegacyForwarding(t *testing.T) {
}
}
}
// A Failed server reaches the proxy and the panel with whether the operator will
// still retry it: the proxy keeps a waiting player through the restart backoff and
// lets them go once the retries are spent.
func TestServerInfoCarriesStartGaveUp(t *testing.T) {
failed := func(name string, restarts int32) *v1alpha1.MinecraftServer {
ms := testServer(name, name)
now := metav1.Now()
ms.Status = v1alpha1.MinecraftServerStatus{
Phase: v1alpha1.PhaseFailed,
AutoRestarts: restarts,
StartRequestedAt: &now,
Conditions: []metav1.Condition{{Type: v1alpha1.ConditionReady, Status: metav1.ConditionFalse,
Reason: v1alpha1.ReasonStartupTimeout}},
}
return ms
}
retrying := serverInfo(failed("retrying", 1))
if retrying.StartGaveUp || retrying.AutoRestarts != 1 {
t.Fatalf("retrying: startGaveUp=%v autoRestarts=%d, want false 1", retrying.StartGaveUp, retrying.AutoRestarts)
}
spent := serverInfo(failed("spent", v1alpha1.MaxAutoRestarts))
b, err := json.Marshal(spent)
if err != nil {
t.Fatal(err)
}
if !spent.StartGaveUp || !strings.Contains(string(b), `"startGaveUp":true`) {
t.Fatalf("spent: startGaveUp=%v JSON %s, want true", spent.StartGaveUp, b)
}
}
@@ -2,6 +2,7 @@ package v1alpha1
import (
corev1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/api/meta"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
@@ -91,6 +92,31 @@ const (
ConditionPlayersCounted = "PlayersCounted"
)
// Ready-condition reasons of a start that timed out: the pod never passed its TCP
// readiness, or RCON never answered. Both are retried by recreating the pod, at
// most MaxAutoRestarts times with a doubling backoff.
const (
ReasonStartupTimeout = "StartupTimeout"
ReasonReadinessTimeout = "ReadinessTimeout"
)
// MaxAutoRestarts bounds how often the operator retries a timed-out start.
const MaxAutoRestarts = 3
// StartGaveUp reports a Failed server that no automatic retry will bring up: its
// start timed out with the retries spent, or it failed for a reason the operator
// never retries (an invalid spec). Only a person moves it on. A Failed server still
// inside its restart backoff has not given up: the operator recreates its pod when
// the backoff runs out, and whoever waits on it should keep waiting.
func StartGaveUp(s *MinecraftServerStatus) bool {
if s.Phase != PhaseFailed {
return false
}
c := meta.FindStatusCondition(s.Conditions, ConditionReady)
timedOut := c != nil && (c.Reason == ReasonStartupTimeout || c.Reason == ReasonReadinessTimeout)
return !timedOut || s.AutoRestarts >= MaxAutoRestarts || s.StartRequestedAt == nil
}
// +kubebuilder:object:root=true
// +kubebuilder:subresource:status
@@ -310,7 +336,7 @@ type MinecraftServerStatus struct {
// the server becomes unoccupied.
EmptySince *metav1.Time `json:"emptySince,omitempty"`
// AutoRestarts counts how often the operator recreated the pod of a start
// that timed out (at most 3, with a doubling backoff); reaching Ready or
// that timed out (at most MaxAutoRestarts, with a doubling backoff); reaching Ready or
// stopping resets it.
AutoRestarts int32 `json:"autoRestarts,omitempty"`
// LastAutoRestartAt is when the operator last recreated the pod.
@@ -0,0 +1,44 @@
package v1alpha1_test
import (
"testing"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"felis.lolicon.best/internal/apis/felis/v1alpha1"
)
// StartGaveUp tells whoever waits on a start whether to keep waiting. The
// operator's own timeout path pins the retrying cases (autorestart_test.go);
// these are the failures it never retries.
func TestStartGaveUp(t *testing.T) {
anchor := metav1.Now()
status := func(phase v1alpha1.Phase, reason string, restarts int32, anchored bool) *v1alpha1.MinecraftServerStatus {
s := &v1alpha1.MinecraftServerStatus{Phase: phase, AutoRestarts: restarts}
if reason != "" {
s.Conditions = []metav1.Condition{{Type: v1alpha1.ConditionReady, Status: metav1.ConditionFalse, Reason: reason}}
}
if anchored {
s.StartRequestedAt = &anchor
}
return s
}
cases := []struct {
name string
s *v1alpha1.MinecraftServerStatus
want bool
}{
{"starting", status(v1alpha1.PhaseStarting, "PodNotReady", 0, true), false},
{"timed out, retries left", status(v1alpha1.PhaseFailed, v1alpha1.ReasonStartupTimeout, 2, true), false},
{"rcon timed out, retries left", status(v1alpha1.PhaseFailed, v1alpha1.ReasonReadinessTimeout, 0, true), false},
{"timed out, retries spent", status(v1alpha1.PhaseFailed, v1alpha1.ReasonStartupTimeout, v1alpha1.MaxAutoRestarts, true), true},
{"invalid spec is never retried", status(v1alpha1.PhaseFailed, "InvalidSpec", 0, true), true},
{"no start anchor, nothing to retry from", status(v1alpha1.PhaseFailed, v1alpha1.ReasonStartupTimeout, 0, false), true},
{"failed with no condition", status(v1alpha1.PhaseFailed, "", 0, true), true},
}
for _, c := range cases {
if got := v1alpha1.StartGaveUp(c.s); got != c.want {
t.Errorf("%s: StartGaveUp = %v, want %v", c.name, got, c.want)
}
}
}
+7
View File
@@ -52,6 +52,8 @@ func TestTimedOutStartRecreatesThePodWithBackoff(t *testing.T) {
for i, a := range attempts {
if s := at(a[1]); s.Status.Phase != v1alpha1.PhaseFailed || !podExists() || s.Status.AutoRestarts != int32(i) {
t.Fatalf("attempt %d before due: phase=%s pod=%v restarts=%d", i+1, s.Status.Phase, podExists(), s.Status.AutoRestarts)
} else if v1alpha1.StartGaveUp(&s.Status) {
t.Fatalf("attempt %d before due: reported as given up while a retry is coming", i+1)
}
s := at(a[2])
if podExists() {
@@ -76,6 +78,9 @@ func TestTimedOutStartRecreatesThePodWithBackoff(t *testing.T) {
if s.Status.Phase != v1alpha1.PhaseFailed || s.Status.AutoRestarts != 3 || !podExists() {
t.Fatalf("after three attempts: phase=%s restarts=%d pod=%v", s.Status.Phase, s.Status.AutoRestarts, podExists())
}
if !v1alpha1.StartGaveUp(&s.Status) {
t.Fatal("after three attempts: not reported as given up")
}
}
// An RCON channel that never answers on a TCP-ready pod gets the same retry.
@@ -93,6 +98,8 @@ func TestReadinessTimeoutRecreatesThePod(t *testing.T) {
reconcile(t, r, "survival")
if s := getServer(t, c, "survival"); s.Status.Phase != v1alpha1.PhaseFailed || s.Status.AutoRestarts != 0 {
t.Fatalf("before due: phase=%s restarts=%d", s.Status.Phase, s.Status.AutoRestarts)
} else if v1alpha1.StartGaveUp(&s.Status) {
t.Fatal("before due: reported as given up while a retry is coming")
}
clock = base.Add(90 * time.Second)
reconcile(t, r, "survival")
+3 -3
View File
@@ -58,7 +58,7 @@ const (
// maxAutoRestarts bounds how often a timed-out start is retried by
// recreating its pod; autoRestartBaseBackoff is the first wait, doubling
// per attempt.
maxAutoRestarts = 3
maxAutoRestarts = v1alpha1.MaxAutoRestarts
autoRestartBaseBackoff = time.Minute
defaultReadinessTimeoutSec = 300
)
@@ -300,7 +300,7 @@ func (r *Reconciler) reconcileRunning(ctx context.Context, server *v1alpha1.Mine
if current.Status.ReadyReplicas < 1 {
r.markStarting(server, "PodNotReady", "waiting for pod TCP readiness")
if r.startupTimedOut(server) {
r.markFailed(server, "StartupTimeout", "pod did not become ready within startup timeout")
r.markFailed(server, v1alpha1.ReasonStartupTimeout, "pod did not become ready within startup timeout")
if err := r.recoverFailedStart(ctx, server, startupTimeout(server)); err != nil {
return ctrl.Result{}, err
}
@@ -339,7 +339,7 @@ func (r *Reconciler) reconcileRunning(ctx context.Context, server *v1alpha1.Mine
}
r.markStarting(server, "RconNotReachable", err.Error())
if r.readinessTimedOut(server) {
r.markFailed(server, "ReadinessTimeout", "RCON probe did not succeed within readiness timeout")
r.markFailed(server, v1alpha1.ReasonReadinessTimeout, "RCON probe did not succeed within readiness timeout")
if err := r.recoverFailedStart(ctx, server, readinessTimeout(server)); err != nil {
return ctrl.Result{}, err
}
+7
View File
@@ -2348,6 +2348,13 @@ export interface components {
playerCountUnknown?: boolean;
/** @description Present and true when the CR carries the label felis.lolicon.best/forwarding=legacy. The proxy then forwards this server's players BungeeCord-style in the handshake address instead of modern forwarding (Felis-Legacy Velocity fork only). */
legacyForwarding?: boolean;
/**
* Format: int32
* @description Present when non-zero; how often the operator recreated the pod of this start after it timed out (at most 3).
*/
autoRestarts?: number;
/** @description Present and true for a Failed server no automatic retry will bring up: its start timed out with the retries spent, or its spec is invalid. A Failed server without it is still in its restart backoff and may come up on its own. */
startGaveUp?: boolean;
};
/** @description One row of the fleet-wide admin read (internal/api/handlers_user.go fleetServerView). */
FleetServer: components["schemas"]["ServerInfo"] & {
+5
View File
@@ -65,6 +65,11 @@ export interface ServerStatus {
/** True when the CR is labelled forwarding=legacy: the proxy forwards this
* server's players BungeeCord-style (a 1.8 backend behind ViaVersion). */
legacyForwarding?: boolean;
/** How often the operator recreated the pod of this start after it timed out. */
autoRestarts?: number;
/** True for a Failed server no automatic retry will bring up; a Failed server
* without it is still in its restart backoff. */
startGaveUp?: boolean;
}
/** WhitelistResult projects GET /servers/{name}/access/whitelist (spec §7 access).
+1 -1
View File
@@ -168,7 +168,7 @@ What it does when routing is active:
| ------- | -------- |
| Backend registry | Polls `GET /api/v1/servers` every 15 s and reconciles Velocity's dynamic registry. A failed poll **keeps existing registrations** — a control-plane blip never deregisters live backends. The API advertises each backend Service's host-routable ClusterIP, avoiding cluster-DNS names on the host-run proxy. |
| Join (`PlayerChooseInitialServerEvent`) | Resolves `subdomain.<root-domain>` and remembers the target, but every fresh connection still enters `login`. When the login gate requests its post-auth lobby transfer, Velocity re-checks link status: a ready remembered target is selected immediately; an asleep target is woken and queued from the lobby. |
| Waiting queue | One scheduled drain every 2 s polls status once per distinct waited-on server; a waiter drops out on transfer, on the player leaving, or after a 120 s timeout. |
| Waiting queue | One scheduled drain every 2 s polls status once per distinct waited-on server. A waiter stays while its server is on the way up (starting, or Failed inside the operator's restart backoff), hears a progress line each minute, and drops out on transfer, on the player leaving, when the start is given up (`startGaveUp`) or the server is stopped, after 120 s without an answer from felis-api, or at a one-hour backstop. |
| Wake gate | The wake is `POST /api/v1/internal/servers/{name}/wake` keyed on the player's online-mode UUID. **403** (policy refused) tells the player and stops; **429** (wake already in flight) keeps waiting. |
| Server-list ping (`ProxyPingEvent`) | Answers from the cached lifecycle view with a phase-aware MOTD (online / starting / sleeping) — **read-only, never wakes** anything. Mirroring each backend's own MOTD by background-pinging ready servers is a later slice. |
| Join report (`ServerConnectedEvent`) | Reports real joins to a felis backend via `POST …/join-event`, so the reaper sees activity and the player is auto-added to the server allowlist. |
@@ -35,6 +35,8 @@ public final class ServerView {
private final int playersOnline;
private final int playersMax;
private final boolean legacyForwarding;
private final int autoRestarts;
private final boolean startGaveUp;
public ServerView(String name, String subdomain, String phase, boolean ready,
String autostartPolicy, String desiredState, String endpointMode,
@@ -47,6 +49,14 @@ public final class ServerView {
String autostartPolicy, String desiredState, String endpointMode,
String endpointAddress, int playersOnline, int playersMax,
boolean legacyForwarding) {
this(name, subdomain, phase, ready, autostartPolicy, desiredState, endpointMode,
endpointAddress, playersOnline, playersMax, legacyForwarding, 0, false);
}
public ServerView(String name, String subdomain, String phase, boolean ready,
String autostartPolicy, String desiredState, String endpointMode,
String endpointAddress, int playersOnline, int playersMax,
boolean legacyForwarding, int autoRestarts, boolean startGaveUp) {
this.name = name;
this.subdomain = subdomain;
this.phase = phase;
@@ -58,6 +68,8 @@ public final class ServerView {
this.playersOnline = playersOnline;
this.playersMax = playersMax;
this.legacyForwarding = legacyForwarding;
this.autoRestarts = autoRestarts;
this.startGaveUp = startGaveUp;
}
/** fromJson builds a view from a parsed felis-api object, tolerating absent fields. */
@@ -73,7 +85,9 @@ public final class ServerView {
str(o, "endpointAddress"),
intval(o, "playersOnline"),
intval(o, "playersMax"),
bool(o, "legacyForwarding"));
bool(o, "legacyForwarding"),
intval(o, "autoRestarts"),
bool(o, "startGaveUp"));
}
/**
@@ -124,6 +138,8 @@ public final class ServerView {
b.append(",\"playersOnline\":").append(v.playersOnline);
b.append(",\"playersMax\":").append(v.playersMax);
b.append(",\"legacyForwarding\":").append(v.legacyForwarding);
b.append(",\"autoRestarts\":").append(v.autoRestarts);
b.append(",\"startGaveUp\":").append(v.startGaveUp);
b.append('}');
}
return b.append("]}").toString();
@@ -188,6 +204,20 @@ public final class ServerView {
return legacyForwarding;
}
/** autoRestarts is how often the operator recreated the pod of a start that timed out. */
public int autoRestarts() {
return autoRestarts;
}
/**
* startGaveUp is true for a Failed server no automatic retry will bring up (the
* restarts are spent, or its spec is invalid). A Failed server without it is in
* its restart backoff and may still come up on its own.
*/
public boolean startGaveUp() {
return startGaveUp;
}
private static String str(Map<?, ?> o, String key) {
Object v = o.get(key);
return v instanceof String ? (String) v : null;
@@ -27,6 +27,7 @@ import java.util.Optional;
import java.util.UUID;
import java.util.concurrent.ConcurrentHashMap;
import java.util.concurrent.atomic.AtomicBoolean;
import java.util.function.LongSupplier;
/**
* WaitingRouter implements the §11 domain-autostart routing loop and its waiting
@@ -42,8 +43,10 @@ import java.util.concurrent.atomic.AtomicBoolean;
*
* <p>The queue is drained by {@link #tick()}, scheduled by the plugin on the async
* pool. Each tick polls felis-api once per distinct waited-on server and, when one
* reports ready, transfers everyone waiting on it. A waiter drops out when it times
* out, when the player leaves the proxy, or on a successful transfer.
* reports ready, transfers everyone waiting on it. A waiter stays as long as its
* server is on the way up, through the operator's restart backoff, and drops out on a
* successful transfer, when the player leaves the proxy, when the start is given up or
* the server stopped, or when felis-api stops answering for the wait window.
*
* <p>Every transition out of login is checked against felis-api's link status, and
* command/menu queue entries are checked the same way. The wake is then gated
@@ -58,7 +61,19 @@ import java.util.concurrent.atomic.AtomicBoolean;
* outage on a recent positive answer for the same UUID and otherwise fails closed.
*/
public final class WaitingRouter {
// A waiter stays while its server is on the way up — starting, or Failed inside the
// operator's restart backoff — and every poll that says so renews this window. It
// runs out only when felis-api stops answering or the server stops heading for
// Running. A modpack's cold start (a 300 s budget, then up to three recreated pods
// with a 1, 2, 4 min backoff) outlasts any fixed wait, and a waiter dropped early
// was never moved in when the server did come up.
private static final long WAIT_TIMEOUT_MILLIS = 120_000L;
// The backstop for a start whose status never moves (an operator that is down):
// twice the default budget, 4 × 300 s plus 7 min of backoff.
private static final long MAX_WAIT_MILLIS = 60 * 60_000L;
// A long wait tells the player how it is going this often, so it is not silence.
private static final long PROGRESS_NOTICE_MILLIS = 60_000L;
private static final String DESIRED_STOPPED = "Stopped";
// How long a positive link answer can stand in for felis-api while it is down.
private static final long LINK_GRACE_MILLIS = 10 * 60_000L;
// The login gate re-sends its release with backoff (and during a felis-api outage
@@ -86,6 +101,8 @@ public final class WaitingRouter {
// felis:control face can tell the player's GUI the backend is ready. Null until
// the ControlChannel is wired in at proxy init; set once, read on the tick pool.
private volatile MenuTransferListener menuListener;
// The waiting queue's clock; tests move it to walk a long start.
private volatile LongSupplier clock = System::currentTimeMillis;
WaitingRouter(ProxyServer proxy, Logger log, FelisApiClient api, ServerRegistry registry,
FelisVelocityPlugin plugin, String loginServer, String lobbyServer) {
@@ -99,6 +116,10 @@ public final class WaitingRouter {
this.links = new LinkGate(api::linkStatus, LINK_GRACE_MILLIS, System::currentTimeMillis);
}
void setClock(LongSupplier clock) {
this.clock = clock;
}
/** pruneLinks bounds the LinkGate's fallback records; called on the refresh loop. */
void pruneLinks() {
links.prune();
@@ -399,8 +420,8 @@ public final class WaitingRouter {
}
private void drain() {
long now = System.currentTimeMillis();
Map<String, Boolean> readyCache = new HashMap<>(); // one status poll per distinct server
long now = clock.getAsLong();
Map<String, Optional<ServerView>> polled = new HashMap<>(); // one status poll per distinct server
for (Map.Entry<UUID, Waiter> e : new ArrayList<>(waiting.entrySet())) {
UUID id = e.getKey();
Waiter w = e.getValue();
@@ -411,31 +432,22 @@ public final class WaitingRouter {
}
Player player = po.get();
boolean zh = FelisVelocityPlugin.zh(player);
if (now > w.deadlineMillis) {
Optional<ServerView> poll = polled.get(w.serverName);
if (poll == null) {
poll = poll(w.serverName);
polled.put(w.serverName, poll);
}
ServerView status = poll.orElse(null);
if (status == null || !status.ready()) {
if (status != null && !stillComing(player, zh, w, status, now)) {
waiting.remove(id);
} else if (now > w.deadlineMillis || now - w.sinceMillis > MAX_WAIT_MILLIS) {
waiting.remove(id);
player.sendMessage(Component.text(
zh ? "「" + w.serverName + "」启动耗时超出预期。你可以稍后在大厅重试。"
: "« " + w.serverName + " » is taking longer than expected to start. "
+ "You can try again from the lobby later.", NamedTextColor.YELLOW));
continue;
}
Boolean ready = readyCache.get(w.serverName);
if (ready == null) {
try {
ServerView status = api.serverStatus(w.serverName);
ready = status.ready();
// The registry refreshes every 15 s; a server that just came up may
// still be registered at its old address, or not at all. Register
// what this poll reports before transferring anyone to it.
if (ready) {
registry.observe(status);
}
} catch (LinkException ex) {
ready = Boolean.FALSE; // transient → keep waiting until the deadline
}
readyCache.put(w.serverName, ready);
}
if (!ready) {
continue;
}
Optional<RegisteredServer> backend = registry.registered(w.serverName);
@@ -470,6 +482,71 @@ public final class WaitingRouter {
}
}
// poll asks felis-api how one waited-on server is doing; empty when it does not
// answer, which the waiters ride out until their window closes.
private Optional<ServerView> poll(String serverName) {
try {
ServerView status = api.serverStatus(serverName);
// The registry refreshes every 15 s; a server that just came up may still
// be registered at its old address, or not at all. Register what this poll
// reports before transferring anyone to it.
if (status.ready()) {
registry.observe(status);
}
return Optional.of(status);
} catch (LinkException ex) {
return Optional.empty();
}
}
// stillComing reads a not-ready poll for one waiter. While the server is heading
// for Running it renews the waiter's window, and now and then tells the player how
// the start is going. When nothing is coming — the retries are spent, or somebody
// stopped the server — it says so and returns false.
private boolean stillComing(Player player, boolean zh, Waiter w, ServerView status, long now) {
if (status.startGaveUp()) {
player.sendMessage(Component.text(
zh ? "「" + w.serverName + "」启动失败,自动重试也已用完。服主可以在面板查看日志后重新启动。"
: "« " + w.serverName + " » failed to start and its automatic retries are spent. "
+ "The owner can check its log in the panel and start it again.",
NamedTextColor.RED));
return false;
}
if (DESIRED_STOPPED.equalsIgnoreCase(status.desiredState())) {
// felis-api reads servers from an informer cache, so the first poll after
// the wake can still show the old desired state; two in a row are a stop.
if (w.stopSeen) {
player.sendMessage(Component.text(
zh ? "「" + w.serverName + "」已被停止,不再为你排队。"
: "« " + w.serverName + " » was stopped, so you're no longer waiting for it.",
NamedTextColor.YELLOW));
return false;
}
w.stopSeen = true;
return true;
}
w.stopSeen = false;
w.deadlineMillis = now + WAIT_TIMEOUT_MILLIS;
if (status.autoRestarts() > w.restartsSeen) {
w.restartsSeen = status.autoRestarts();
w.noticedMillis = now;
player.sendMessage(Component.text(
zh ? "「" + w.serverName + "」启动超时,正在自动重试(第 " + w.restartsSeen + " 次)……"
: "« " + w.serverName + " » timed out starting; retrying automatically (attempt "
+ w.restartsSeen + ")…",
NamedTextColor.YELLOW));
} else if (now - w.noticedMillis >= PROGRESS_NOTICE_MILLIS) {
w.noticedMillis = now;
long minutes = (now - w.sinceMillis) / 60_000L;
player.sendMessage(Component.text(
zh ? "「" + w.serverName + "」仍在启动(已等 " + minutes + " 分钟),就绪后会自动把你传送过去。"
: "« " + w.serverName + " » is still starting (" + minutes + " min so far); "
+ "you'll be moved in when it's ready.",
NamedTextColor.GRAY));
}
return true;
}
private void authorizeAndWait(Player player, String serverName, boolean fromMenu) {
UUID id = player.getUniqueId();
boolean zh = FelisVelocityPlugin.zh(player);
@@ -608,8 +685,7 @@ public final class WaitingRouter {
: "Starting « " + serverName + " » — you'll be moved in automatically.",
NamedTextColor.GRAY));
}
waiting.put(id, new Waiter(
serverName, System.currentTimeMillis() + WAIT_TIMEOUT_MILLIS, fromMenu));
waiting.put(id, new Waiter(serverName, clock.getAsLong(), fromMenu));
}
private void logWakeFailure(Player player, String serverName, boolean zh, LinkException e) {
@@ -659,13 +735,21 @@ public final class WaitingRouter {
private static final class Waiter {
final String serverName;
final long deadlineMillis;
final boolean fromMenu; // true → notify the felis:control face on transfer
final long sinceMillis;
// Only the drain touches these, one tick at a time (the ticking flag orders
// the ticks), so they need no further synchronization.
long deadlineMillis;
long noticedMillis;
int restartsSeen;
boolean stopSeen;
Waiter(String serverName, long deadlineMillis, boolean fromMenu) {
Waiter(String serverName, long nowMillis, boolean fromMenu) {
this.serverName = serverName;
this.deadlineMillis = deadlineMillis;
this.fromMenu = fromMenu;
this.sinceMillis = nowMillis;
this.deadlineMillis = nowMillis + WAIT_TIMEOUT_MILLIS;
this.noticedMillis = nowMillis;
}
}
@@ -446,6 +446,17 @@ final class Fakes {
final Map<String, Boolean> ready = new ConcurrentHashMap<>();
/** address is the direct endpoint a server reports while it is ready. */
final Map<String, String> address = new ConcurrentHashMap<>();
/**
* phase, desired, restarts and gaveUp are how a not-ready server's start is going,
* as the status route reports it (phase Stopped and desiredState Running unless
* set: the wake the waiter followed has asked for Running).
*/
final Map<String, String> phase = new ConcurrentHashMap<>();
final Map<String, String> desired = new ConcurrentHashMap<>();
final Map<String, Integer> restarts = new ConcurrentHashMap<>();
final Set<String> gaveUp = ConcurrentHashMap.newKeySet();
/** statusDown makes the status route answer 500. */
volatile boolean statusDown;
/** wakeError maps a server to "status code" (e.g. "403 forbidden"). */
final Map<String, String> wakeError = new ConcurrentHashMap<>();
volatile int joinStatus = 204;
@@ -493,6 +504,10 @@ final class Fakes {
String action = parts.length > 1 ? parts[1] : "";
switch (method + " " + action) {
case "GET status":
if (statusDown) {
reply(ex, 500, "{\"error\":{\"code\":\"internal\",\"message\":\"down\"}}");
return;
}
reply(ex, 200, status(name, ready.getOrDefault(name, false)));
return;
case "POST wake":
@@ -538,8 +553,14 @@ final class Fakes {
String endpoint = up
? "\"endpointMode\":\"direct\"" + (addr == null ? "" : ",\"endpointAddress\":\"" + addr + "\"")
: "\"endpointMode\":\"fallback\",\"endpointAddress\":\"login\"";
// Like the real ServerInfo, autoRestarts and startGaveUp are left out at 0/false.
int restarts = this.restarts.getOrDefault(name, 0);
return "{\"name\":\"" + name + "\",\"subdomain\":\"" + name + "\",\"phase\":\""
+ (up ? "Running" : "Stopped") + "\",\"ready\":" + up + "," + endpoint + "}";
+ (up ? "Running" : phase.getOrDefault(name, "Stopped")) + "\",\"ready\":" + up
+ ",\"desiredState\":\"" + desired.getOrDefault(name, "Running") + "\""
+ (restarts == 0 ? "" : ",\"autoRestarts\":" + restarts)
+ (gaveUp.contains(name) ? ",\"startGaveUp\":true" : "")
+ "," + endpoint + "}";
}
private static void reply(HttpExchange ex, int status, String body) throws IOException {
@@ -17,6 +17,7 @@ import java.util.ArrayList;
import java.util.Collections;
import java.util.List;
import java.util.Locale;
import java.util.concurrent.atomic.AtomicLong;
/**
* WaitingRouterTest drives the real WaitingRouter, ServerRegistry, FelisApiClient and
@@ -70,6 +71,7 @@ public final class WaitingRouterTest {
loginGate();
wakeRefusals();
queue();
longStart();
menuAndCommands();
joins();
disconnectAndRelease();
@@ -91,6 +93,10 @@ public final class WaitingRouterTest {
view("delta", false, "10.43.0.6:25565"),
view("epsilon", false, "10.43.0.7:25565"),
view("zeta", false, "10.43.0.8:25565"),
view("eta", false, "10.43.0.21:25565"),
view("theta", false, "10.43.0.22:25565"),
view("iota", false, "10.43.0.23:25565"),
view("kappa", false, "10.43.0.24:25565"),
view("fresh", false, null)));
if (withGone) {
list.add(view("gone", false, "10.43.0.9:25565"));
@@ -324,6 +330,102 @@ public final class WaitingRouterTest {
assertEq("registered: queue empty", 0, router.waitingCount());
}
// A modpack's cold start outlasts any fixed wait: the operator gives a start 300 s,
// then recreates the pod up to three times with a 1, 2, 4 min backoff. The waiter
// follows the server's own progress instead of a clock, and hears how it is going.
private static void longStart() {
assertEq("long start: queue empty to begin with", 0, router.waitingCount());
AtomicLong now = new AtomicLong(1_000_000_000L);
router.setClock(now::get);
try {
// Starting, then Failed inside the backoff, a recreated pod, and up at 25 min.
Fakes.FakePlayer slow = player("eta.mc.test", true);
choose(slow);
release(slow);
api.phase.put("eta", "Starting");
advance(now, 60, 30);
assertEq("slow start: told how it is going", true, slow.said("« eta » is still starting (1 min so far)"));
api.phase.put("eta", "Failed"); // the 300 s budget ran out; backoff until 6 min
advance(now, 6 * 60, 30);
assertEq("in the backoff: still waiting well past two minutes", 1, router.waitingCount());
api.phase.put("eta", "Starting");
api.restarts.put("eta", 1);
advance(now, 30, 30);
assertEq("recreated pod: told", true, slow.said("« eta » timed out starting; retrying automatically (attempt 1)"));
advance(now, 18 * 60, 30);
assertEq("25 minutes in: still waiting", 1, router.waitingCount());
assertEq("25 minutes in: never given up on", false, slow.said("taking longer than expected"));
int notices = count(slow.messages, "is still starting");
assertEq("about one progress line a minute, not one a poll", true, notices >= 20 && notices <= 25);
api.ready.put("eta", true);
router.tick();
assertEq("up at last: moved in", List.of("eta"), List.copyOf(slow.connects));
assertEq("up at last: queue empty", 0, router.waitingCount());
// The retries are spent: nothing is coming, and the player hears why.
Fakes.FakePlayer spent = player("theta.mc.test", true);
choose(spent);
release(spent);
api.phase.put("theta", "Failed");
api.restarts.put("theta", 3);
api.gaveUp.add("theta");
advance(now, 2, 2);
assertEq("given up: dropped", 0, router.waitingCount());
assertEq("given up: told", true, spent.said("« theta » failed to start and its automatic retries are spent"));
assertEq("given up: not moved", 0, spent.connects.size());
// Somebody stops the server. One poll can still show the desired state from
// before the wake (the api reads an informer cache); two in a row are a stop.
Fakes.FakePlayer stopped = player("iota.mc.test", true);
choose(stopped);
release(stopped);
api.desired.put("iota", "Stopped");
advance(now, 2, 2);
assertEq("one stopped poll: still waiting", 1, router.waitingCount());
api.desired.put("iota", "Running");
advance(now, 2, 2);
api.desired.put("iota", "Stopped");
advance(now, 2, 2);
assertEq("a lag blip does not count toward the stop", 1, router.waitingCount());
advance(now, 2, 2);
assertEq("stopped: dropped", 0, router.waitingCount());
assertEq("stopped: told", true, stopped.said("« iota » was stopped, so you're no longer waiting for it"));
// felis-api stops answering: the waiter rides it out for the wait window only.
Fakes.FakePlayer blind = player("kappa.mc.test", true);
choose(blind);
release(blind);
api.statusDown = true;
advance(now, 110, 10);
assertEq("api down: still waiting inside the window", 1, router.waitingCount());
advance(now, 20, 10);
assertEq("api down: dropped after the window", 0, router.waitingCount());
assertEq("api down: told", true, blind.said("« kappa » is taking longer than expected"));
api.statusDown = false;
// A status that never moves (an operator that is down) ends at the backstop.
Fakes.FakePlayer stuck = player("kappa.mc.test", true);
choose(stuck);
release(stuck);
api.phase.put("kappa", "Starting");
advance(now, 59 * 60, 60);
assertEq("stuck: still waiting before the hour", 1, router.waitingCount());
advance(now, 2 * 60, 60);
assertEq("stuck: dropped at the backstop", 0, router.waitingCount());
assertEq("stuck: told", true, stuck.said("« kappa » is taking longer than expected"));
} finally {
router.setClock(System::currentTimeMillis);
}
}
// advance moves the waiting queue's clock on by seconds, draining every step.
private static void advance(AtomicLong now, int seconds, int step) {
for (int s = 0; s < seconds; s += step) {
now.addAndGet(step * 1000L);
router.tick();
}
}
private static void menuAndCommands() {
List<String> notified = Collections.synchronizedList(new ArrayList<>());
router.setMenuTransferListener((player, server) -> notified.add(player.getUsername() + "@" + server));