diff --git a/cmd/felis/api.go b/cmd/felis/api.go index bb21fb6..2d5a700 100644 --- a/cmd/felis/api.go +++ b/cmd/felis/api.go @@ -370,6 +370,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int { InternalBaseURL: internalAPIBaseURL(), Submissions: submissions, Mailer: mailer, + Schedules: repo, // The external face authenticates the local session cookie the sign-in doors // mint, live once `felis breakGlass` flips local_auth_enabled on. Cloudflare // Access, when the install sits behind it, is enforced at the edge only. @@ -464,6 +465,7 @@ func cmdAPI(args []string, stdout, stderr io.Writer) int { // reconciles it, but this loop converges builds nobody is polling. go reconcileBuilds(ctx, builder, stderr) go settleRestoreChains(ctx, a, stderr) + go runSchedules(ctx, a, stderr) // A daily restore point of every world played since its last one, taken // once the server stops ([archive] scheduled_every; 0s turns it off). if backuper != nil && rcfg.ScheduledEvery > 0 { @@ -734,6 +736,24 @@ func settleRestoreChains(ctx context.Context, a *api.API, stderr io.Writer) { } } +// runSchedules runs the servers' scheduled tasks (api.API.RunSchedules). The +// interval is how late a task may start, and how often a restart or backup in +// progress checks whether it can take its next step. +func runSchedules(ctx context.Context, a *api.API, stderr io.Writer) { + t := time.NewTicker(15 * time.Second) + defer t.Stop() + for { + select { + case <-ctx.Done(): + return + case <-t.C: + if err := a.RunSchedules(ctx); err != nil { + fmt.Fprintf(stderr, "felis api: scheduled tasks: %v\n", err) + } + } + } +} + // scheduleBackups starts the scheduled backups (api.BackupScheduler). Each tick // starts at most one, so the interval also spaces the worlds that stopped at // the same time: a world that stops waits at most this long for its point to diff --git a/docs/openapi.yaml b/docs/openapi.yaml index df179f4..e071a6d 100644 --- a/docs/openapi.yaml +++ b/docs/openapi.yaml @@ -123,6 +123,8 @@ tags: description: Read (SSE) and write (RCON) server console (external face, app tier). - name: backups description: World backup listing and restore (external face, app tier). + - name: schedules + description: Scheduled tasks of your own servers (external face, app tier). - name: account description: Web side of account linking (external face, app tier). - name: admin-servers @@ -639,6 +641,68 @@ components: allOf: [{ $ref: '#/components/schemas/RetireState' }] description: Owned rows only. Present while the server is given up or being deleted. + Schedule: + type: object + description: >- + One scheduled task of a server (internal/api/schedules.go Schedule). It runs + at minute_of_day on the weekdays, or with every_minutes set at every multiple + of it since midnight on those days, in timezone. + required: [id, server, label, action, command, every_minutes, minute_of_day, weekdays, timezone, warn_minutes, enabled, next_run_at, run_state, last_run_at, last_result, last_detail, created_by, created_at] + properties: + id: { type: integer, format: int64 } + server: { type: string } + label: { type: string, description: Free text naming the task; may be empty. } + action: + type: string + enum: [command, restart, stop, start, backup] + description: >- + backup of a running server stops it, takes a backup (pruned with the daily + restore points, [archive] scheduled_keep) and starts it again; of a stopped + server it leaves the server stopped. + command: { type: string, description: The console command of a command task, without a slash; empty otherwise. } + every_minutes: + type: integer + enum: [0, 15, 30, 60, 120, 180, 240, 360, 480, 720] + description: 0 runs once a day at minute_of_day. A restart, stop, start or backup repeats at most every 60 minutes. + minute_of_day: { type: integer, minimum: 0, maximum: 1439, description: Minutes after local midnight; 0 when every_minutes is set. } + weekdays: { type: integer, minimum: 1, maximum: 127, description: 'Bitmask of the days it runs on: bit 0 Sunday to bit 6 Saturday.' } + timezone: { type: string, description: IANA zone the times are in, e.g. Asia/Shanghai. } + warn_minutes: + type: integer + enum: [0, 1, 5, 10, 15, 30] + description: How long before a restart, stop or backup the players on the server are told (say); 0 for none, and always 0 for a command or start. + enabled: { type: boolean } + next_run_at: { type: string, format: date-time, nullable: true, description: Null while disabled. } + run_state: + type: string + enum: ['', claimed, stopping, backing_up, starting] + description: What a run in progress is doing; empty when none is. + last_run_at: { type: string, format: date-time, nullable: true } + last_result: + type: string + enum: ['', ok, skipped, failed, missed] + description: >- + How the last run ended; empty before the first and during a run. missed is a + run felis-api was down for, dropped once it was 10 minutes late. + last_detail: { type: string, description: What happened, in English (a command's reply, or why the run was skipped or failed). } + created_by: { type: string } + created_at: { type: string, format: date-time } + + ScheduleInput: + type: object + required: [action, weekdays, timezone] + additionalProperties: false + properties: + label: { type: string, maxLength: 64 } + action: { type: string, enum: [command, restart, stop, start, backup] } + command: { type: string, maxLength: 1024, description: Required for a command task and refused for the others. One line; a leading slash is dropped. } + every_minutes: { type: integer, enum: [0, 15, 30, 60, 120, 180, 240, 360, 480, 720] } + minute_of_day: { type: integer, minimum: 0, maximum: 1439 } + weekdays: { type: integer, minimum: 1, maximum: 127 } + timezone: { type: string } + warn_minutes: { type: integer, enum: [0, 1, 5, 10, 15, 30] } + enabled: { type: boolean, description: Default true. } + BackupView: type: object description: One world backup (internal/api/repo.go BackupView). backup_ref is withheld (spec §286). @@ -4595,6 +4659,201 @@ paths: application/json: schema: { $ref: '#/components/schemas/Error' } + # --------------------------------------------------- scheduled tasks (app) --- + /api/v1/servers/{name}/schedules: + get: + tags: [schedules] + operationId: listServerSchedules + summary: List a server's scheduled tasks (owner-or-admin). + x-felis-face: [external] + x-felis-tier: app + security: [{ sessionCookie: [] }] + parameters: + - { name: name, in: path, required: true, schema: { type: string } } + responses: + '200': + description: The server's tasks, oldest first, and how many it may have. + content: + application/json: + schema: + type: object + required: [server, schedules, limit] + properties: + server: { type: string } + schedules: + type: array + items: { $ref: '#/components/schemas/Schedule' } + limit: { type: integer } + '400': + $ref: '#/components/responses/BadRequest' + '401': + $ref: '#/components/responses/Unauthorized' + '403': + $ref: '#/components/responses/Forbidden' + '404': + $ref: '#/components/responses/NotFound' + '503': + $ref: '#/components/responses/ServiceUnavailable' + post: + tags: [schedules] + operationId: createServerSchedule + summary: Add a scheduled task to a server (owner-or-admin). + description: >- + The task belongs to the server's current owner: once the server has another + owner felis-api disables it instead of running it, until somebody saves it + again. felis-api checks the tasks every 15 seconds; a run it was down for is + started late, up to 10 minutes, and dropped as missed after that. A command + runs only on a running server, a restart only restarts a running one, and a + start goes through the running-server cap and a pending retirement like a + wake. Audited as schedule.create; each run as schedule.run by scheduler. + x-felis-face: [external] + x-felis-tier: app + security: [{ sessionCookie: [] }] + parameters: + - { name: name, in: path, required: true, schema: { type: string } } + requestBody: + required: true + content: + application/json: + schema: { $ref: '#/components/schemas/ScheduleInput' } + responses: + '201': + description: Created. + content: + application/json: + schema: { $ref: '#/components/schemas/Schedule' } + '400': + description: A malformed body or name, or settings out of range (bad_schedule, bad_request for the command). + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + '401': + $ref: '#/components/responses/Unauthorized' + '403': + $ref: '#/components/responses/Forbidden' + '404': + $ref: '#/components/responses/NotFound' + '409': + description: The server already has 20 tasks (schedule_limit). + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + '503': + $ref: '#/components/responses/ServiceUnavailable' + /api/v1/servers/{name}/schedules/{id}: + put: + tags: [schedules] + operationId: updateServerSchedule + summary: Change a scheduled task (owner-or-admin). + description: >- + Replaces the task's settings and recomputes its next run. The task passes to + the server's current owner. Refused while a run is in progress. Audited as + schedule.update. + x-felis-face: [external] + x-felis-tier: app + security: [{ sessionCookie: [] }] + parameters: + - { name: name, in: path, required: true, schema: { type: string } } + - { name: id, in: path, required: true, schema: { type: integer, format: int64 } } + requestBody: + required: true + content: + application/json: + schema: { $ref: '#/components/schemas/ScheduleInput' } + responses: + '200': + description: Saved. + content: + application/json: + schema: { $ref: '#/components/schemas/Schedule' } + '400': + description: A malformed body, name or id, or settings out of range (bad_schedule). + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + '401': + $ref: '#/components/responses/Unauthorized' + '403': + $ref: '#/components/responses/Forbidden' + '404': + $ref: '#/components/responses/NotFound' + '409': + description: A run is in progress (schedule_running). + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + '503': + $ref: '#/components/responses/ServiceUnavailable' + delete: + tags: [schedules] + operationId: deleteServerSchedule + summary: Remove a scheduled task (owner-or-admin). + description: Refused while a run is in progress. Audited as schedule.delete. + x-felis-face: [external] + x-felis-tier: app + security: [{ sessionCookie: [] }] + parameters: + - { name: name, in: path, required: true, schema: { type: string } } + - { name: id, in: path, required: true, schema: { type: integer, format: int64 } } + responses: + '204': + description: Removed. + '400': + $ref: '#/components/responses/BadRequest' + '401': + $ref: '#/components/responses/Unauthorized' + '403': + $ref: '#/components/responses/Forbidden' + '404': + $ref: '#/components/responses/NotFound' + '409': + description: A run is in progress (schedule_running). + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + '503': + $ref: '#/components/responses/ServiceUnavailable' + /api/v1/servers/{name}/schedules/{id}/run: + post: + tags: [schedules] + operationId: runServerSchedule + summary: Run a scheduled task now (owner-or-admin). + description: >- + Starts a run at once, without the players' warning, whether the task is + enabled or not; its next scheduled run stays where it was. The answer is the + task after the run's first step: a command, stop or start has finished, and a + restart or backup goes on in the background (run_state). Audited as + schedule.run_now, and the run itself as schedule.run. + x-felis-face: [external] + x-felis-tier: app + security: [{ sessionCookie: [] }] + parameters: + - { name: name, in: path, required: true, schema: { type: string } } + - { name: id, in: path, required: true, schema: { type: integer, format: int64 } } + responses: + '202': + description: Started. + content: + application/json: + schema: { $ref: '#/components/schemas/Schedule' } + '400': + $ref: '#/components/responses/BadRequest' + '401': + $ref: '#/components/responses/Unauthorized' + '403': + $ref: '#/components/responses/Forbidden' + '404': + $ref: '#/components/responses/NotFound' + '409': + description: >- + A run is already in progress (schedule_running), or the server has another + owner since the task was saved (schedule_stale). + content: + application/json: + schema: { $ref: '#/components/schemas/Error' } + '503': + $ref: '#/components/responses/ServiceUnavailable' + # ------------------------------------------------------ users (admin tier) ---- /api/v1/users: get: diff --git a/docs/troubleshooting.md b/docs/troubleshooting.md index ba0f426..1038363 100644 --- a/docs/troubleshooting.md +++ b/docs/troubleshooting.md @@ -3163,6 +3163,65 @@ end, has not been run on a cluster.] --- +## 19. Scheduled tasks: a run is skipped, missed or failed + +A server's owner (or an admin) keeps up to 20 scheduled tasks on it, on the +panel's **Scheduled tasks** page (`/api/v1/servers/{name}/schedules`). Each one +sends a console command, restarts, stops, starts or backs up the server, once a +day at a set time or every 15 minutes to 12 hours, on the chosen weekdays, in an +IANA time zone (the browser's by default). A restart, stop or backup may warn the +players in game (`say`) 1 to 30 minutes ahead. felis-api runs the tasks itself: a +loop every 15 seconds, so a task starts within about 15 seconds of its time, and +the database hands each run to one felis-api only. + +What each action does: + +| Action | Server running | Server stopped | +| --- | --- | --- | +| command | runs it over RCON (a leading `/` is dropped) | skipped | +| restart | stops the pod, then starts it | skipped | +| stop | stops it | skipped | +| start | skipped (a start that `Failed` is started over) | starts it, as the panel's start does (running-server cap, retiring server, busy world) | +| backup | stops it, takes the same backup Job as "Back up now", then starts it again | takes the backup | + +A step has a deadline: the pod must stop within 15 minutes, the backup Job must +end within 45, and the start is given 15. A restart or backup that cannot finish +still starts the server again when it was running before, so a failed backup never +leaves a world offline. + +The row's last result says how the latest run went: + +| `last_result` | `last_detail` | What to do | +| --- | --- | --- | +| `ok` | empty, or the reply of a console command | nothing | +| `skipped` | `the server has a new owner since this schedule was saved; save it again to use it` | the server changed owner (claim, account migration, admin). The task switched itself off; the new owner reviews it and saves it to turn it back on | +| `skipped` | `the server was not running`, `the server was already stopped`, `the server was already running`, `the server has no world yet`, `the server is being given up or deleted`, `the server no longer exists`, and for a start `the cluster is at its running-server cap` or `the world is busy with …` | the run had nothing to act on; the next run tries again | +| `missed` | `felis-api was not running at the scheduled time` | felis-api was down more than 10 minutes past the time. The run is dropped, so a restart never lands hours late; the next one runs as usual | +| `failed` | `felis-api stopped in the middle of this run` | felis-api restarted during the run's first step (the claim was over 2 minutes old); check that the server is in the state you want | +| `failed` | `the server did not stop within 15 minutes` | see §1 and §2 for a pod that hangs; the backup did not run | +| `failed` | `another operation kept the world busy for 15 minutes` | a restore, file change or another backup held the world (§3b) | +| `failed` | `the backup failed: …`, `the backup did not finish within 45 minutes`, `the backup store is full; ask an administrator to free space` | §10 for backup Jobs; the Backups page shows the Job | +| `failed` | `could not start the server again: …` (after a restart or a backup) | the server stopped and could not be started again: the running-server cap, a server being given up, or a world still busy after 15 minutes of retries. Start it from the panel once the cause is gone (§1, §3b) | +| `failed` | `the server console could not be reached`, `the command failed: …` | the server's RCON (§1c) | + +"Run now" starts a run at once, including on a switched-off task, and leaves the +next planned run where it was. While a run is in progress (`run_state` is +`claimed`, `stopping`, `backing_up` or `starting`) the task cannot be edited, +deleted or run again (`409 schedule_running`). + +A time the clock skips at a daylight-saving change runs an hour early (the offset +before the jump applies); a time it repeats runs once, the first time. A time zone +the host no longer knows runs in UTC. Deleting a server drops its tasks, and a new +server of the same name starts with none. The audit actions are `schedule.create`, +`schedule.update`, `schedule.delete`, `schedule.run_now` (by a person) and +`schedule.run` (every finished run, with its result and detail). + +[GO-TESTED: `TestScheduleNextRun`, `TestScheduleInputValidation`, +`TestScheduleRunnerBackup`, `TestScheduleRunnerRestart`, `TestScheduleRunnerStaleClaim`; +PG-TESTED: `TestScheduleStoreRunCAS`, `TestDueSchedules`, `TestSchedulesFollowTheServer`] + +--- + ## Quick reference: symptom → section | Symptom | Section | @@ -3208,3 +3267,4 @@ end, has not been run on a cluster.] | `felis breakGlass` sends no code / shows `Root override`; `otp_skipped` in the audit | §17 | | How long sessions, codes and audit rows are kept; export audit rows | §17 | | Files page: a change or upload refused (`file_exists`, `bad_path`, `too_large`, `upload_staging_full`, `volume_full`, `files_timeout`) | §18 | +| A scheduled task shows `skipped`, `missed` or `failed`; a task switched itself off after an owner change | §19 | diff --git a/internal/api/api.go b/internal/api/api.go index 1658fd8..b16dcea 100644 --- a/internal/api/api.go +++ b/internal/api/api.go @@ -98,6 +98,11 @@ type API struct { FileStage *fileedit.Stage InternalBaseURL string + // Schedules stores the servers' scheduled tasks (schedules.go), which + // RunSchedules fires. Optional: when nil the schedule routes report 503 and + // RunSchedules does nothing. + Schedules ServerSchedules + // Submissions is the user-modpack approval lane (a user-directed extension over // the §16 build subsystem; see internal/submit). It is optional: when // nil the /me/submissions and /submissions routes report 503 rather than 404, so @@ -576,6 +581,15 @@ func (a *API) externalAPIRoutes() []apiRoute { {Method: "POST", Pattern: "/api/v1/servers/{name}/files/mkdir", h: a.handleMkdir}, {Method: "POST", Pattern: "/api/v1/servers/{name}/files/rename", h: a.handleRenameFile}, {Method: "PUT", Pattern: "/api/v1/servers/{name}/files/upload", h: a.handleUploadFile}, + // Scheduled tasks (handlers_schedules.go): a console command, restart, stop, + // start or backup at set times, which felis-api's runner fires. App-tier and + // owner-or-admin inside the handler, like the console and power routes they + // automate; a schedule reaches nothing its owner could not do by hand. + {Method: "GET", Pattern: "/api/v1/servers/{name}/schedules", h: a.handleListSchedules}, + {Method: "POST", Pattern: "/api/v1/servers/{name}/schedules", h: a.handleCreateSchedule}, + {Method: "PUT", Pattern: "/api/v1/servers/{name}/schedules/{id}", h: a.handleUpdateSchedule}, + {Method: "DELETE", Pattern: "/api/v1/servers/{name}/schedules/{id}", h: a.handleDeleteSchedule}, + {Method: "POST", Pattern: "/api/v1/servers/{name}/schedules/{id}/run", h: a.handleRunSchedule}, // Account linking (spec §10), web side: /start reports link status (it is the // pointer handleClaim's 412 emits), /verify consumes the in-game code and binds // the account. App-tier, not admin — linking your own account is an ordinary diff --git a/internal/api/handlers_console.go b/internal/api/handlers_console.go index e85f479..a22cdfc 100644 --- a/internal/api/handlers_console.go +++ b/internal/api/handlers_console.go @@ -3,7 +3,6 @@ package api import ( "errors" "net/http" - "strings" "felis.lolicon.best/internal/naming" ) @@ -51,23 +50,10 @@ func (a *API) handleCommand(w http.ResponseWriter, r *http.Request) { return } - // A console command is exactly one line. Trim surrounding space, strip a - // single leading '/' (players type "/say hi"; RCON wants "say hi"), then - // reject control characters so one request can never smuggle a second command - // past a newline. - command := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(body.Command), "/")) - if command == "" { - writeError(w, r, newError(http.StatusBadRequest, "bad_request", "command is required")) - return - } - if len(command) > maxConsoleCommandLen { - writeError(w, r, newError(http.StatusBadRequest, "bad_request", - "command too long (max %d bytes)", maxConsoleCommandLen)) - return - } - if strings.IndexFunc(command, func(c rune) bool { return c < 0x20 }) >= 0 { - writeError(w, r, newError(http.StatusBadRequest, "bad_request", - "command must be a single line (no control characters)")) + // A console command is exactly one line (normalizeConsoleCommand). + command, err := normalizeConsoleCommand(body.Command) + if err != nil { + writeError(w, r, err) return } diff --git a/internal/api/handlers_schedules.go b/internal/api/handlers_schedules.go new file mode 100644 index 0000000..b487469 --- /dev/null +++ b/internal/api/handlers_schedules.go @@ -0,0 +1,233 @@ +package api + +import ( + "context" + "errors" + "net/http" + "strconv" + "time" + + "felis.lolicon.best/internal/naming" +) + +// scheduleRunTimeout bounds the first step of a run somebody asked for. The +// step outlives the request, so a client that hangs up cannot cut a stop off +// halfway. +const scheduleRunTimeout = 30 * time.Second + +// scheduleServer resolves {name} for the schedule routes and applies their +// gate: 400 for a malformed name, 404 for a server that does not exist, 403 +// for a caller who neither owns it nor is an admin, 503 without a store. +func (a *API) scheduleServer(w http.ResponseWriter, r *http.Request) (*ServerRecord, bool) { + name := r.PathValue("name") + if err := naming.ValidateServerName(name); err != nil { + writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err)) + return nil, false + } + rec, err := a.Repo.ServerByName(r.Context(), name) + if err != nil { + a.writeLookupError(w, r, err) + return nil, false + } + if !a.isOwnerOrAdmin(principalFromContext(r.Context()), rec) { + writeError(w, r, errForbidden) + return nil, false + } + if a.Schedules == nil { + writeError(w, r, newError(http.StatusServiceUnavailable, "schedules_unavailable", + "scheduled tasks are not configured")) + return nil, false + } + return rec, true +} + +// scheduleID parses {id}. +func scheduleID(w http.ResponseWriter, r *http.Request) (int64, bool) { + id, err := strconv.ParseInt(r.PathValue("id"), 10, 64) + if err != nil || id <= 0 { + writeError(w, r, newError(http.StatusBadRequest, "bad_id", "invalid schedule id")) + return 0, false + } + return id, true +} + +// writeScheduleError maps the store's schedule errors. +func (a *API) writeScheduleError(w http.ResponseWriter, r *http.Request, err error) { + switch { + case errors.Is(err, ErrScheduleRunning): + writeError(w, r, newError(http.StatusConflict, "schedule_running", + "this schedule is running; try again once the run finishes")) + case errors.Is(err, ErrScheduleLimit): + writeError(w, r, newError(http.StatusConflict, "schedule_limit", + "a server can have at most %d scheduled tasks", maxSchedulesPerServer)) + default: + a.writeLookupError(w, r, err) + } +} + +// auditSchedule records a change to a schedule by the signed-in caller. +func (a *API) auditSchedule(r *http.Request, action string, s *Schedule) { + p := principalFromContext(r.Context()) + e := AuditEntry{Actor: auditActor(p), Action: action, ServerName: s.Server, + Payload: auditPayload(map[string]any{"schedule": s.ID, "action": s.Action})} + if p != nil { + e.ActorUserID = p.UserID + } + a.auditEntry(r, e) +} + +// handleListSchedules serves GET /servers/{name}/schedules. +func (a *API) handleListSchedules(w http.ResponseWriter, r *http.Request) { + rec, ok := a.scheduleServer(w, r) + if !ok { + return + } + list, err := a.Schedules.ListSchedules(r.Context(), rec.Name) + if err != nil { + writeError(w, r, err) + return + } + if list == nil { + list = []Schedule{} + } + writeJSON(w, http.StatusOK, map[string]any{"server": rec.Name, "schedules": list, "limit": maxSchedulesPerServer}) +} + +// handleCreateSchedule serves POST /servers/{name}/schedules. The schedule +// belongs to the server's current owner (see Schedule.OwnerID). +func (a *API) handleCreateSchedule(w http.ResponseWriter, r *http.Request) { + rec, ok := a.scheduleServer(w, r) + if !ok { + return + } + var in scheduleInput + if err := decodeJSON(w, r, &in); err != nil { + writeError(w, r, err) + return + } + s := &Schedule{Server: rec.Name, OwnerID: rec.OwnerID, CreatedBy: auditActor(principalFromContext(r.Context()))} + if err := in.apply(s); err != nil { + writeError(w, r, err) + return + } + s.NextRunAt = a.firstRun(s) + if err := a.Schedules.CreateSchedule(r.Context(), s, maxSchedulesPerServer); err != nil { + a.writeScheduleError(w, r, err) + return + } + a.auditSchedule(r, "schedule.create", s) + writeJSON(w, http.StatusCreated, s) +} + +// handleUpdateSchedule serves PUT /servers/{name}/schedules/{id}: new settings, +// and the schedule passes to the server's current owner. +func (a *API) handleUpdateSchedule(w http.ResponseWriter, r *http.Request) { + rec, ok := a.scheduleServer(w, r) + if !ok { + return + } + id, ok := scheduleID(w, r) + if !ok { + return + } + var in scheduleInput + if err := decodeJSON(w, r, &in); err != nil { + writeError(w, r, err) + return + } + s, err := a.Schedules.GetSchedule(r.Context(), rec.Name, id) + if err != nil { + a.writeScheduleError(w, r, err) + return + } + if err := in.apply(s); err != nil { + writeError(w, r, err) + return + } + s.OwnerID, s.NextRunAt = rec.OwnerID, a.firstRun(s) + if err := a.Schedules.UpdateSchedule(r.Context(), s); err != nil { + a.writeScheduleError(w, r, err) + return + } + a.auditSchedule(r, "schedule.update", s) + writeJSON(w, http.StatusOK, s) +} + +// handleDeleteSchedule serves DELETE /servers/{name}/schedules/{id}. +func (a *API) handleDeleteSchedule(w http.ResponseWriter, r *http.Request) { + rec, ok := a.scheduleServer(w, r) + if !ok { + return + } + id, ok := scheduleID(w, r) + if !ok { + return + } + s, err := a.Schedules.GetSchedule(r.Context(), rec.Name, id) + if err != nil { + a.writeScheduleError(w, r, err) + return + } + if err := a.Schedules.DeleteSchedule(r.Context(), rec.Name, id); err != nil { + a.writeScheduleError(w, r, err) + return + } + a.auditSchedule(r, "schedule.delete", s) + w.WriteHeader(http.StatusNoContent) +} + +// handleRunSchedule serves POST /servers/{name}/schedules/{id}/run: the run +// starts now, without the players' warning, and the next scheduled run stays +// where it is. The answer is the schedule after the run's first step; a restart +// or backup goes on in the background. +func (a *API) handleRunSchedule(w http.ResponseWriter, r *http.Request) { + rec, ok := a.scheduleServer(w, r) + if !ok { + return + } + id, ok := scheduleID(w, r) + if !ok { + return + } + s, err := a.Schedules.GetSchedule(r.Context(), rec.Name, id) + if err != nil { + a.writeScheduleError(w, r, err) + return + } + if s.OwnerID != rec.OwnerID { + writeError(w, r, newError(http.StatusConflict, "schedule_stale", + "the server has a new owner since this schedule was saved; save it again first")) + return + } + ctx, cancel := context.WithTimeout(context.WithoutCancel(r.Context()), scheduleRunTimeout) + defer cancel() + now := a.now() + claimed, err := a.Schedules.ClaimScheduleRun(ctx, id, nil, nil, now) + if err != nil { + writeError(w, r, err) + return + } + if !claimed { + a.writeScheduleError(w, r, ErrScheduleRunning) + return + } + a.auditSchedule(r, "schedule.run_now", s) + if err := a.beginRun(ctx, s); err != nil { + writeError(w, r, err) + return + } + after, err := a.Schedules.GetSchedule(ctx, rec.Name, id) + if err != nil { + writeError(w, r, err) + return + } + writeJSON(w, http.StatusAccepted, after) +} + +// firstRun is the next run of a schedule just saved, nil while it is disabled. +func (a *API) firstRun(s *Schedule) *time.Time { + if !s.Enabled { + return nil + } + return ptrTime(s.nextRun(a.now())) +} diff --git a/internal/api/openapi_parity_test.go b/internal/api/openapi_parity_test.go index 6e04657..8d6be1f 100644 --- a/internal/api/openapi_parity_test.go +++ b/internal/api/openapi_parity_test.go @@ -40,6 +40,7 @@ func TestOpenAPISchemasMatchWireStructs(t *testing.T) { "MyServerView": MyServerView{}, "AllowlistEntry": AllowlistEntry{}, "BackupView": BackupView{}, + "Schedule": Schedule{}, "Build": build.Build{}, "Image": build.Image{}, "BuildScan": buildScanView{}, diff --git a/internal/api/pgrepo.go b/internal/api/pgrepo.go index 1ce722e..ef81713 100644 --- a/internal/api/pgrepo.go +++ b/internal/api/pgrepo.go @@ -830,11 +830,11 @@ func (p *PGRepo) ServerOwners(ctx context.Context) (map[string]ServerOwnership, // a create whose CRD write failed, or one the reaper deleted between removing // its MinecraftServer and marking the row. That row starts over, and the earlier // server's other aliases and allowlist go with it; nothing of its owner, claim, -// activity clock, reaper warnings or pending retirement reaches the new server, -// so the reaper's unfinished deletion no longer applies to it. A retried create -// lands on the same fresh state. The alias subdomain is a PRIMARY KEY: bound to -// another server, it rolls the whole seed back and returns ErrConflict, letting -// the create handler answer 409 before it touches the CRD. +// activity clock, reaper warnings, pending retirement or scheduled tasks reaches +// the new server, so the reaper's unfinished deletion no longer applies to it. A +// retried create lands on the same fresh state. The alias subdomain is a PRIMARY +// KEY: bound to another server, it rolls the whole seed back and returns +// ErrConflict, letting the create handler answer 409 before it touches the CRD. // // Two creates of one name racing between the handler's cluster check and the // first CRD write can still leave the loser's alias in place of the winner's; @@ -862,6 +862,9 @@ func (p *PGRepo) SeedServer(ctx context.Context, name, subdomain string, cpuMill if _, err := tx.ExecContext(ctx, `DELETE FROM server_aliases WHERE server_name = $1`, name); err != nil { return fmt.Errorf("clear earlier aliases: %w", err) } + if _, err := tx.ExecContext(ctx, `DELETE FROM server_schedules WHERE server_name = $1`, name); err != nil { + return fmt.Errorf("clear earlier schedules: %w", err) + } if _, err := tx.ExecContext(ctx, `INSERT INTO server_aliases (subdomain, server_name) VALUES ($1, $2) ON CONFLICT DO NOTHING`, subdomain, name); err != nil { @@ -2613,7 +2616,8 @@ func (p *PGRepo) RedeemMigration(ctx context.Context, targetUserID, codeHash str } // Re-point every server the source owns to the target, collecting the names for - // the audit trail. Server ownership is the only thing that moves. + // the audit trail. Server ownership moves, and the servers' scheduled tasks + // with it: they belong to the same person. rows, err := tx.QueryContext(ctx, `UPDATE servers SET owner_id = $2, claimed_at = $3 WHERE owner_id = $1 AND deleted_at IS NULL @@ -2637,6 +2641,13 @@ func (p *PGRepo) RedeemMigration(ctx context.Context, targetUserID, codeHash str } rows.Close() + if _, err := tx.ExecContext(ctx, + `UPDATE server_schedules s SET owner_id = $2 FROM servers v + WHERE v.name = s.server_name AND v.owner_id = $2 AND s.owner_id = $1`, + sourceUserID, targetUserID); err != nil { + return "", nil, err + } + // Retire the source: revoke its live sessions and soft-delete it so it can neither // log in nor start another migration (double-spend defense). The servers just moved // away, so there is nothing left to release. diff --git a/internal/api/pgschedules.go b/internal/api/pgschedules.go new file mode 100644 index 0000000..1632be8 --- /dev/null +++ b/internal/api/pgschedules.go @@ -0,0 +1,252 @@ +package api + +import ( + "context" + "database/sql" + "errors" + "fmt" + "time" +) + +// The server_schedules store (ServerSchedules, migration 0036). + +// scheduleColumns is the SELECT list scanSchedule reads, over server_schedules +// aliased s. +const scheduleColumns = `s.id, s.server_name, COALESCE(s.owner_id, ''), s.label, s.action, s.command, + s.every_minutes, s.minute_of_day, s.weekdays, s.timezone, s.warn_minutes, s.enabled, + s.next_run_at, s.warned_for, s.run_state, s.run_resume, s.run_step_at, + s.last_run_at, s.last_result, s.last_detail, s.created_by, s.created_at` + +type rowScanner interface{ Scan(dest ...any) error } + +func scanSchedule(row rowScanner, extra ...any) (*Schedule, error) { + var s Schedule + var next, warned, step, last sql.NullTime + dest := []any{&s.ID, &s.Server, &s.OwnerID, &s.Label, &s.Action, &s.Command, + &s.EveryMinutes, &s.MinuteOfDay, &s.Weekdays, &s.Timezone, &s.WarnMinutes, &s.Enabled, + &next, &warned, &s.RunState, &s.RunResume, &step, + &last, &s.LastResult, &s.LastDetail, &s.CreatedBy, &s.CreatedAt} + if err := row.Scan(append(dest, extra...)...); err != nil { + return nil, err + } + s.NextRunAt, s.WarnedFor, s.RunStepAt, s.LastRunAt = nullTimePtr(next), nullTimePtr(warned), nullTimePtr(step), nullTimePtr(last) + return &s, nil +} + +func nullTimePtr(t sql.NullTime) *time.Time { + if !t.Valid { + return nil + } + return &t.Time +} + +// ListSchedules lists a server's schedules, oldest first. +func (p *PGRepo) ListSchedules(ctx context.Context, server string) ([]Schedule, error) { + rows, err := p.db.QueryContext(ctx, + `SELECT `+scheduleColumns+` FROM server_schedules s WHERE s.server_name = $1 ORDER BY s.id`, server) + if err != nil { + return nil, err + } + defer rows.Close() + var out []Schedule + for rows.Next() { + s, err := scanSchedule(rows) + if err != nil { + return nil, err + } + out = append(out, *s) + } + return out, rows.Err() +} + +// GetSchedule reads one schedule of server, or ErrNotFound. +func (p *PGRepo) GetSchedule(ctx context.Context, server string, id int64) (*Schedule, error) { + s, err := scanSchedule(p.db.QueryRowContext(ctx, + `SELECT `+scheduleColumns+` FROM server_schedules s WHERE s.id = $1 AND s.server_name = $2`, id, server)) + if errors.Is(err, sql.ErrNoRows) { + return nil, ErrNotFound + } + return s, err +} + +// CreateSchedule inserts s under the server row's lock, so two saves racing +// for the last free place cannot both take it. +func (p *PGRepo) CreateSchedule(ctx context.Context, s *Schedule, limit int) error { + tx, err := p.db.BeginTx(ctx, nil) + if err != nil { + return err + } + defer tx.Rollback() //nolint:errcheck // no-op after commit + + var one int + if err := tx.QueryRowContext(ctx, + `SELECT 1 FROM servers WHERE name = $1 AND deleted_at IS NULL FOR UPDATE`, s.Server).Scan(&one); err != nil { + if errors.Is(err, sql.ErrNoRows) { + return ErrNotFound + } + return err + } + var n int + if err := tx.QueryRowContext(ctx, + `SELECT count(*) FROM server_schedules WHERE server_name = $1`, s.Server).Scan(&n); err != nil { + return err + } + if n >= limit { + return ErrScheduleLimit + } + if err := tx.QueryRowContext(ctx, + `INSERT INTO server_schedules (server_name, owner_id, label, action, command, every_minutes, + minute_of_day, weekdays, timezone, warn_minutes, enabled, next_run_at, created_by) + VALUES ($1, NULLIF($2, ''), $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13) + RETURNING id, created_at`, + s.Server, s.OwnerID, s.Label, s.Action, s.Command, s.EveryMinutes, + s.MinuteOfDay, s.Weekdays, s.Timezone, s.WarnMinutes, s.Enabled, s.NextRunAt, s.CreatedBy, + ).Scan(&s.ID, &s.CreatedAt); err != nil { + return err + } + return tx.Commit() +} + +// scheduleMissing tells why a write guarded on an idle run_state touched no +// row: the schedule is gone, or it is running. +func (p *PGRepo) scheduleMissing(ctx context.Context, server string, id int64) error { + var state string + err := p.db.QueryRowContext(ctx, + `SELECT run_state FROM server_schedules WHERE id = $1 AND server_name = $2`, id, server).Scan(&state) + switch { + case errors.Is(err, sql.ErrNoRows): + return ErrNotFound + case err != nil: + return err + case state != "": + return ErrScheduleRunning + } + return fmt.Errorf("schedule %d of %s did not change", id, server) +} + +// UpdateSchedule writes s's settings, owner and next run, unless it is running. +func (p *PGRepo) UpdateSchedule(ctx context.Context, s *Schedule) error { + res, err := p.db.ExecContext(ctx, + `UPDATE server_schedules SET owner_id = NULLIF($3, ''), label = $4, action = $5, command = $6, + every_minutes = $7, minute_of_day = $8, weekdays = $9, timezone = $10, warn_minutes = $11, + enabled = $12, next_run_at = $13, warned_for = NULL, updated_at = now() + WHERE id = $1 AND server_name = $2 AND run_state = ''`, + s.ID, s.Server, s.OwnerID, s.Label, s.Action, s.Command, + s.EveryMinutes, s.MinuteOfDay, s.Weekdays, s.Timezone, s.WarnMinutes, + s.Enabled, s.NextRunAt) + if err != nil { + return err + } + if n, err := res.RowsAffected(); err != nil || n == 0 { + if err != nil { + return err + } + return p.scheduleMissing(ctx, s.Server, s.ID) + } + return nil +} + +// DeleteSchedule removes a schedule, unless it is running. +func (p *PGRepo) DeleteSchedule(ctx context.Context, server string, id int64) error { + res, err := p.db.ExecContext(ctx, + `DELETE FROM server_schedules WHERE id = $1 AND server_name = $2 AND run_state = ''`, id, server) + if err != nil { + return err + } + if n, err := res.RowsAffected(); err != nil || n == 0 { + if err != nil { + return err + } + return p.scheduleMissing(ctx, server, id) + } + return nil +} + +// DueSchedules lists the schedules the runner has to look at, runs in progress +// first, then by due time. +func (p *PGRepo) DueSchedules(ctx context.Context, horizon time.Time) ([]DueSchedule, error) { + rows, err := p.db.QueryContext(ctx, + `SELECT `+scheduleColumns+`, COALESCE(v.owner_id, '') + FROM server_schedules s JOIN servers v ON v.name = s.server_name + WHERE v.deleted_at IS NULL AND (s.run_state <> '' OR (s.enabled AND s.next_run_at <= $1)) + ORDER BY s.run_state = '', s.next_run_at, s.id`, horizon) + if err != nil { + return nil, err + } + defer rows.Close() + var out []DueSchedule + for rows.Next() { + var owner string + s, err := scanSchedule(rows, &owner) + if err != nil { + return nil, err + } + out = append(out, DueSchedule{Schedule: *s, ServerOwner: owner}) + } + return out, rows.Err() +} + +// execApplied runs a compare-and-set write and reports whether it matched. +func (p *PGRepo) execApplied(ctx context.Context, query string, args ...any) (bool, error) { + res, err := p.db.ExecContext(ctx, query, args...) + if err != nil { + return false, err + } + n, err := res.RowsAffected() + return n > 0, err +} + +// ClaimScheduleRun starts a run of a schedule without one. +func (p *PGRepo) ClaimScheduleRun(ctx context.Context, id int64, due, next *time.Time, now time.Time) (bool, error) { + if due == nil { + return p.execApplied(ctx, + `UPDATE server_schedules SET run_state = 'claimed', run_resume = false, run_step_at = $2, + last_run_at = $2, last_result = '', last_detail = '' + WHERE id = $1 AND run_state = ''`, id, now) + } + return p.execApplied(ctx, + `UPDATE server_schedules SET run_state = 'claimed', run_resume = false, run_step_at = $4, + last_run_at = $4, last_result = '', last_detail = '', next_run_at = $3, warned_for = NULL + WHERE id = $1 AND run_state = '' AND enabled AND next_run_at = $2`, id, *due, next, now) +} + +// AdvanceScheduleRun moves a run on to its next step. +func (p *PGRepo) AdvanceScheduleRun(ctx context.Context, id int64, from, to string, resume bool, result, detail string, now time.Time) (bool, error) { + return p.execApplied(ctx, + `UPDATE server_schedules SET run_state = $3, run_resume = $4, run_step_at = $7, + last_result = CASE WHEN $5::text = '' THEN last_result ELSE $5::text END, + last_detail = CASE WHEN $5::text = '' THEN last_detail ELSE $6::text END + WHERE id = $1 AND run_state = $2`, id, from, to, resume, result, detail, now) +} + +// FinishScheduleRun ends a run with its outcome. +func (p *PGRepo) FinishScheduleRun(ctx context.Context, id int64, from, result, detail string) (bool, error) { + return p.execApplied(ctx, + `UPDATE server_schedules SET run_state = '', run_resume = false, run_step_at = NULL, + last_result = $3, last_detail = $4 + WHERE id = $1 AND run_state = $2::text AND $2::text <> ''`, id, from, result, detail) +} + +// MissScheduleRun records the run due then as missed and moves on to next. +func (p *PGRepo) MissScheduleRun(ctx context.Context, id int64, due, next time.Time, detail string) (bool, error) { + return p.execApplied(ctx, + `UPDATE server_schedules SET next_run_at = $3, warned_for = NULL, + last_run_at = $2, last_result = 'missed', last_detail = $4 + WHERE id = $1 AND run_state = '' AND enabled AND next_run_at = $2`, id, due, next, detail) +} + +// WarnScheduleRun marks the run due then as warned about. +func (p *PGRepo) WarnScheduleRun(ctx context.Context, id int64, due time.Time) (bool, error) { + return p.execApplied(ctx, + `UPDATE server_schedules SET warned_for = $2 + WHERE id = $1 AND run_state = '' AND enabled AND next_run_at = $2 + AND warned_for IS DISTINCT FROM $2`, id, due) +} + +// DisableSchedule turns off an enabled, idle schedule and records why. +func (p *PGRepo) DisableSchedule(ctx context.Context, id int64, detail string) (bool, error) { + return p.execApplied(ctx, + `UPDATE server_schedules SET enabled = false, next_run_at = NULL, warned_for = NULL, + last_result = 'skipped', last_detail = $2, updated_at = now() + WHERE id = $1 AND run_state = '' AND enabled`, id, detail) +} diff --git a/internal/api/schedulerunner.go b/internal/api/schedulerunner.go new file mode 100644 index 0000000..df4a6e0 --- /dev/null +++ b/internal/api/schedulerunner.go @@ -0,0 +1,486 @@ +package api + +import ( + "context" + "errors" + "fmt" + "log" + "math" + "strings" + "time" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/maintenance" +) + +// The schedule runner. cmd/felis calls RunSchedules every few seconds; each +// call warns the players about the runs coming up, fires the runs that are +// due and moves every run in progress one step on. All of its state is in +// server_schedules, so a felis-api restart picks a restart or backup up where +// it was. +// +// A command, stop or start is one step. A restart stops the server (runStopping) +// and starts it once it is down (runStarting). A backup of a running server +// stops it, takes a backup once the world volume is free (runBackingUp), waits +// for the backup Job and starts the server again; a backup of a stopped server +// leaves it stopped. Each step gives up after its own wait, and a run that took +// the server down tries to bring it back when it gives up. + +// Run steps (Schedule.RunState). +const ( + runClaimed = "claimed" + runStopping = "stopping" + runBackingUp = "backing_up" + runStarting = "starting" +) + +const ( + // scheduleMissGrace is how late a run may still start: felis-api back from + // a short restart catches up, and a run hours late is dropped as missed. + scheduleMissGrace = 10 * time.Minute + // claimedStale is how long a run may sit in its first step before it is + // taken for one whose felis-api stopped mid-step. The first step is a few + // API calls and at most one RCON round trip. + claimedStale = 2 * time.Minute + // stopWait is how long a run waits for the server to stop and its world + // volume to come free. A pod saving a big world takes a while. + stopWait = 15 * time.Minute + // backupWait is how long a run waits for its backup Job: past the Job's own + // deadline ([archive] backup Job deadline, 30m by default). + backupWait = 45 * time.Minute + // startWait is how long a run keeps trying to start the server again while + // the cluster is at its running cap or the world volume is still busy. + startWait = 15 * time.Minute + // backupJobSkew is how much earlier than the step the backup Job's creation + // stamp may read: felis-api's clock and the API server's differ a little. + backupJobSkew = 2 * time.Minute + // scheduleWarnHorizon is the longest warning lead time, in minutes. + scheduleWarnHorizon = 30 +) + +// Audit identity of a run. Each run leaves one schedule.run row when it ends. +const ( + scheduleActor = "scheduler" + scheduleRunAction = "schedule.run" +) + +// RunSchedules warns about, fires and advances the schedules once. A schedule +// that fails is logged and left for the next call; it never holds up the rest. +func (a *API) RunSchedules(ctx context.Context) error { + if a.Schedules == nil { + return nil + } + now := a.now() + due, err := a.Schedules.DueSchedules(ctx, now.Add(scheduleWarnHorizon*time.Minute)) + if err != nil { + return fmt.Errorf("list the due schedules: %w", err) + } + for i := range due { + d := &due[i] + if err := a.tickSchedule(ctx, d, now); err != nil { + log.Printf("api: schedule %d of %s: %v", d.ID, d.Server, err) + } + } + return nil +} + +// tickSchedule does what one due schedule needs now. +func (a *API) tickSchedule(ctx context.Context, d *DueSchedule, now time.Time) error { + s := &d.Schedule + if s.RunState != "" { + return a.advanceRun(ctx, s, now) + } + // Somebody else's server now: their commands must not run on it. Saving + // the schedule again (PUT) hands it to the new owner. + if d.ServerOwner != s.OwnerID { + _, err := a.Schedules.DisableSchedule(ctx, s.ID, + "the server has a new owner since this schedule was saved; save it again to use it") + return err + } + due := *s.NextRunAt // set on every enabled schedule DueSchedules returns idle + if now.Before(due) { + return a.warnRun(ctx, s, due, now) + } + next := s.nextRun(now) + if now.Sub(due) > scheduleMissGrace { + _, err := a.Schedules.MissScheduleRun(ctx, s.ID, due, next, + "felis-api was not running at the scheduled time") + return err + } + ok, err := a.Schedules.ClaimScheduleRun(ctx, s.ID, &due, &next, now) + if err != nil || !ok { + return err + } + return a.beginRun(ctx, s) +} + +// warnRun tells the players on the server that a restart, stop or backup is +// coming, once per run, when its warning time has come. +func (a *API) warnRun(ctx context.Context, s *Schedule, due, now time.Time) error { + lead := time.Duration(s.WarnMinutes) * time.Minute + if now.Before(due.Add(-lead)) || a.Console == nil { + return nil // no warning, or not yet + } + info, err := a.Cluster.GetServer(ctx, s.Server) + if err != nil { + return err + } + if !info.Ready { + return nil + } + ok, err := a.Schedules.WarnScheduleRun(ctx, s.ID, due) + if err != nil || !ok { + return err + } + // Warned late (felis-api was restarting at the warning time): say how long + // is really left. + minutes := int(math.Ceil(due.Sub(now).Minutes())) + if _, err := a.Console.RunCommand(ctx, s.Server, "say "+scheduleWarning(s.Action, minutes)); err != nil { + return fmt.Errorf("warn the players: %w", err) + } + return nil +} + +// scheduleWarning is the in-game line announcing a run minutes ahead, in both +// panel languages. +func scheduleWarning(action string, minutes int) string { + switch action { + case ScheduleRestart: + return fmt.Sprintf("[Felis] 服务器将在 %d 分钟后重启 / Server restarts in %d min", minutes, minutes) + case ScheduleStop: + return fmt.Sprintf("[Felis] 服务器将在 %d 分钟后关闭 / Server stops in %d min", minutes, minutes) + default: + return fmt.Sprintf("[Felis] 服务器将在 %d 分钟后暂停做备份,完成后自动恢复 / Server pauses for a backup in %d min and comes back after", minutes, minutes) + } +} + +// beginRun takes the first step of a run just claimed (runClaimed). +func (a *API) beginRun(ctx context.Context, s *Schedule) error { + finish := func(result, detail string) error { return a.finishRun(ctx, s, runClaimed, result, detail) } + info, err := a.Cluster.GetServer(ctx, s.Server) + if errors.Is(err, ErrNotFound) { + return finish(ScheduleSkipped, "the server no longer exists") + } + if err != nil { + return finish(ScheduleFailed, "could not read the server: "+err.Error()) + } + running := info.DesiredState == string(v1alpha1.DesiredRunning) + + switch s.Action { + case ScheduleCommand: + if !info.Ready { + return finish(ScheduleSkipped, "the server was not running") + } + if a.Console == nil { + return finish(ScheduleFailed, "the console is not configured") + } + out, err := a.Console.RunCommand(ctx, s.Server, s.Command) + if errors.Is(err, ErrConsoleUnavailable) { + return finish(ScheduleFailed, "the server console could not be reached") + } + if err != nil { + return finish(ScheduleFailed, "the command failed: "+err.Error()) + } + return finish(ScheduleOK, stripFormatting(out)) + + case ScheduleStop: + if !running { + return finish(ScheduleSkipped, "the server was already stopped") + } + if err := a.Cluster.SetDesiredState(ctx, s.Server, v1alpha1.DesiredStopped); err != nil { + return finish(ScheduleFailed, "could not stop the server: "+err.Error()) + } + return finish(ScheduleOK, "") + + case ScheduleStart: + if running && info.Phase != string(v1alpha1.PhaseFailed) { + return finish(ScheduleSkipped, "the server was already running") + } + why, _, err := a.startScheduled(ctx, s.Server) + if err != nil { + return finish(ScheduleFailed, "could not start the server: "+err.Error()) + } + if why != "" { + return finish(ScheduleSkipped, why) + } + return finish(ScheduleOK, "") + + case ScheduleRestart: + if !running { + return finish(ScheduleSkipped, "the server was not running") + } + return a.stopForRun(ctx, s, true) + + case ScheduleBackup: + if _, ok := a.Backuper.(ScheduledBackuper); !ok { + return finish(ScheduleFailed, "backups are not configured") + } + if limit := a.BackupStoreCap; limit > 0 { + used, err := a.Repo.BackupStoreBytes(ctx) + if err != nil { + return finish(ScheduleFailed, "could not read the backup store size: "+err.Error()) + } + if used >= limit { + return finish(ScheduleFailed, "the backup store is full; ask an administrator to free space") + } + } + exists, err := a.Cluster.WorldVolumeExists(ctx, s.Server) + if err != nil { + return finish(ScheduleFailed, "could not look up the world volume: "+err.Error()) + } + if !exists { + return finish(ScheduleSkipped, "the server has no world yet") + } + if running { + return a.stopForRun(ctx, s, true) + } + // Already stopped: back it up as soon as the world volume is free, and + // leave it stopped afterwards. + if ok, err := a.Schedules.AdvanceScheduleRun(ctx, s.ID, runClaimed, runStopping, false, "", "", a.now()); err != nil || !ok { + return err + } + s.RunState, s.RunResume, s.RunStepAt = runStopping, false, ptrTime(a.now()) + return a.advanceRun(ctx, s, a.now()) + } + return finish(ScheduleFailed, "unknown action "+s.Action) +} + +// stopForRun records the stop step and then stops the server, in that order: a +// felis-api that dies between the two leaves a running server in runStopping, +// which the next call reads as someone having started it, and nothing is lost. +func (a *API) stopForRun(ctx context.Context, s *Schedule, resume bool) error { + now := a.now() + if ok, err := a.Schedules.AdvanceScheduleRun(ctx, s.ID, runClaimed, runStopping, resume, "", "", now); err != nil || !ok { + return err + } + if err := a.Cluster.SetDesiredState(ctx, s.Server, v1alpha1.DesiredStopped); err != nil { + return a.finishRun(ctx, s, runStopping, ScheduleFailed, "could not stop the server: "+err.Error()) + } + return nil +} + +// advanceRun moves a run in progress on by one step, or leaves it waiting. +func (a *API) advanceRun(ctx context.Context, s *Schedule, now time.Time) error { + waited := now.Sub(*s.RunStepAt) + switch s.RunState { + case runClaimed: + if waited < claimedStale { + return nil // its first step is running right now + } + return a.finishRun(ctx, s, runClaimed, ScheduleFailed, "felis-api stopped in the middle of this run") + case runStopping: + return a.advanceStopping(ctx, s, waited) + case runBackingUp: + return a.advanceBackingUp(ctx, s, waited) + case runStarting: + return a.advanceStarting(ctx, s, waited) + } + return a.finishRun(ctx, s, s.RunState, ScheduleFailed, "unknown run step "+s.RunState) +} + +// advanceStopping waits for the server to go down. A restart then starts it; +// a backup takes the world volume and starts the backup Job. +func (a *API) advanceStopping(ctx context.Context, s *Schedule, waited time.Duration) error { + info, err := a.Cluster.GetServer(ctx, s.Server) + if errors.Is(err, ErrNotFound) { + return a.finishRun(ctx, s, runStopping, ScheduleSkipped, "the server no longer exists") + } + if err != nil { + return err + } + // Only this run stops the server; wanted running again means a person (or + // a player's join) started it meanwhile, and their start stands. + if info.DesiredState == string(v1alpha1.DesiredRunning) { + return a.finishRun(ctx, s, runStopping, ScheduleSkipped, "someone started the server before the run finished") + } + giveUp := func(detail string) error { + if waited < stopWait { + return nil + } + if s.RunResume { + if err := a.Cluster.SetDesiredState(ctx, s.Server, v1alpha1.DesiredRunning); err == nil { + detail += "; it was started again" + } + } + return a.finishRun(ctx, s, runStopping, ScheduleFailed, detail) + } + + if s.Action == ScheduleRestart { + if info.Phase != string(v1alpha1.PhaseStopped) { + return giveUp("the server did not stop within 15 minutes") + } + if ok, err := a.Schedules.AdvanceScheduleRun(ctx, s.ID, runStopping, runStarting, s.RunResume, "", "", a.now()); err != nil || !ok { + return err + } + s.RunState, s.RunStepAt = runStarting, ptrTime(a.now()) + return a.advanceStarting(ctx, s, 0) + } + + b, ok := a.Backuper.(ScheduledBackuper) + if !ok { + return a.finishRun(ctx, s, runStopping, ScheduleFailed, "backups are not configured") + } + switch err := a.Cluster.AcquireMaintenance(ctx, s.Server, maintenance.KindBackup); { + case errors.Is(err, ErrNotStopped): + return giveUp("the server did not stop within 15 minutes") + case errors.Is(err, ErrMaintenanceInProgress): + return giveUp("another operation kept the world busy for 15 minutes") + case errors.Is(err, ErrNotFound): + return a.finishRun(ctx, s, runStopping, ScheduleSkipped, "the server no longer exists") + case err != nil: + return err + } + err = b.BackupScheduled(ctx, s.Server, s.OwnerID) + // Once the Job exists it holds the world; the lock only covered the gap. + if rerr := a.Cluster.ReleaseMaintenance(context.WithoutCancel(ctx), s.Server); rerr != nil { + log.Printf("api: release the maintenance lock on %s: %v (it lapses after %s)", s.Server, rerr, maintenance.Grace) + } + if err != nil { + detail := "could not start the backup: " + err.Error() + if s.RunResume { + if serr := a.Cluster.SetDesiredState(ctx, s.Server, v1alpha1.DesiredRunning); serr == nil { + detail += "; the server was started again" + } + } + return a.finishRun(ctx, s, runStopping, ScheduleFailed, detail) + } + log.Printf("api: schedule %d started a backup of %s", s.ID, s.Server) + _, err = a.Schedules.AdvanceScheduleRun(ctx, s.ID, runStopping, runBackingUp, s.RunResume, "", "", a.now()) + return err +} + +// advanceBackingUp waits for the backup Job and records how it ended; a run +// that stopped a running server then starts it again. +func (a *API) advanceBackingUp(ctx context.Context, s *Schedule, waited time.Duration) error { + result, detail := ScheduleOK, "" + if a.JobStatus != nil { + jobs, err := a.JobStatus.LatestJobs(ctx, s.Server) + if err != nil { + return err + } + var job *AsyncJob + for i := range jobs { + j := &jobs[i] + if j.Scheduled && !j.StartedAt.Before(s.RunStepAt.Add(-backupJobSkew)) { + job = j + break // newest first + } + } + switch { + case job == nil && waited < backupJobSkew: + return nil // not listed yet + case job == nil: + result, detail = ScheduleFailed, "the backup Job is gone before it could be checked; see the Backups page" + case job.State == "running" && waited < backupWait: + return nil + case job.State == "running": + result, detail = ScheduleFailed, "the backup did not finish within 45 minutes" + case job.State == "failed": + result, detail = ScheduleFailed, strings.TrimSpace("the backup failed: "+job.Message) + } + } + if !s.RunResume { + return a.finishRun(ctx, s, runBackingUp, result, detail) + } + if ok, err := a.Schedules.AdvanceScheduleRun(ctx, s.ID, runBackingUp, runStarting, true, result, detail, a.now()); err != nil || !ok { + return err + } + s.RunState, s.RunStepAt, s.LastResult, s.LastDetail = runStarting, ptrTime(a.now()), result, detail + return a.advanceStarting(ctx, s, 0) +} + +// advanceStarting brings the server back up after a restart's stop or a +// backup. The outcome recorded so far (the backup's) is kept when it starts. +func (a *API) advanceStarting(ctx context.Context, s *Schedule, waited time.Duration) error { + why, retry, err := a.startScheduled(ctx, s.Server) + if errors.Is(err, ErrNotFound) { + return a.finishRun(ctx, s, runStarting, ScheduleSkipped, "the server no longer exists") + } + if err != nil { + return err + } + if why == "" { + result, detail := s.LastResult, s.LastDetail + if result == "" { + result = ScheduleOK + } + return a.finishRun(ctx, s, runStarting, result, detail) + } + if retry && waited < startWait { + return nil + } + detail := "could not start the server again: " + why + if s.LastResult == ScheduleFailed { + detail = s.LastDetail + "; " + detail + } + return a.finishRun(ctx, s, runStarting, ScheduleFailed, detail) +} + +// startScheduled starts a server for a run: the wake path without the +// per-player cooldown. why names what kept it stopped ("" once it is wanted +// running), and retry says whether that may clear by itself. +func (a *API) startScheduled(ctx context.Context, name string) (why string, retry bool, err error) { + rec, err := a.Repo.ServerByName(ctx, name) + if err != nil { + return "", false, err + } + if rec.Retire != nil { + return "the server is being given up or deleted", false, nil + } + info, err := a.Cluster.GetServer(ctx, name) + if err != nil { + return "", false, err + } + ok, err := a.withinRunningCap(ctx, info) + if err != nil { + return "", false, err + } + if !ok { + return "the cluster is at its running-server cap", true, nil + } + // A start that Failed is started over, as a person's start does (handleStart). + if info.Phase == string(v1alpha1.PhaseFailed) && info.DesiredState == string(v1alpha1.DesiredRunning) { + err = a.Cluster.RetryStart(ctx, name) + } else { + err = a.Cluster.SetDesiredState(ctx, name, v1alpha1.DesiredRunning) + } + var busy *MaintenanceBusyError + if errors.As(err, &busy) { + return "the world is busy with " + maintenanceLabel(busy.Kind), true, nil + } + return "", false, err +} + +// finishRun ends a run at step from and audits it. +func (a *API) finishRun(ctx context.Context, s *Schedule, from, result, detail string) error { + detail = truncateUTF8(detail, maxScheduleDetail) + ok, err := a.Schedules.FinishScheduleRun(ctx, s.ID, from, result, detail) + if err != nil || !ok { + return err + } + a.writeAudit(ctx, AuditEntry{ + Actor: scheduleActor, Source: scheduleActor, Action: scheduleRunAction, ServerName: s.Server, + Payload: auditPayload(map[string]any{"schedule": s.ID, "action": s.Action, "result": result, "detail": detail}), + }) + return nil +} + +// stripFormatting drops Minecraft's section-sign formatting codes from a +// console reply and trims it. +func stripFormatting(s string) string { + var b strings.Builder + skip := false + for _, c := range s { + switch { + case skip: + skip = false + case c == '§': + skip = true + default: + b.WriteRune(c) + } + } + return strings.TrimSpace(b.String()) +} + +func ptrTime(t time.Time) *time.Time { return &t } diff --git a/internal/api/schedulerunner_test.go b/internal/api/schedulerunner_test.go new file mode 100644 index 0000000..28dc1b3 --- /dev/null +++ b/internal/api/schedulerunner_test.go @@ -0,0 +1,696 @@ +package api + +import ( + "errors" + "strings" + "testing" + "time" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" + "felis.lolicon.best/internal/maintenance" +) + +// outcome asserts how a schedule's last run ended and that no run is left. +func (r *schedRig) outcome(t *testing.T, id int64, result, detail string) *Schedule { + t.Helper() + s := r.st.row(t, id) + if s.RunState != "" || s.LastResult != result || s.LastDetail != detail { + t.Fatalf("run state %q, result %q %q; want done with %q %q", s.RunState, s.LastResult, s.LastDetail, result, detail) + } + return s +} + +// step asserts a run waits at step. +func (r *schedRig) step(t *testing.T, id int64, step string) { + t.Helper() + if s := r.st.row(t, id); s.RunState != step { + t.Fatalf("run state %q (result %q %q), want %q", s.RunState, s.LastResult, s.LastDetail, step) + } +} + +// runAudits lists the schedule.run rows as result:detail. +func (r *schedRig) runAudits() []string { + var out []string + for _, e := range r.repo.audits { + if e.Action == scheduleRunAction && e.Actor == scheduleActor && e.Source == scheduleActor { + out = append(out, string(e.Payload)) + } + } + return out +} + +func TestScheduleRunnerFiring(t *testing.T) { + t.Run("a due command runs, moves on a day and is audited", func(t *testing.T) { + r := newSchedRig(t) + r.con.reply = " §6There are §c2§6 players online " + id := r.schedule(ScheduleCommand, schedT0, func(s *Schedule) { s.Command = "list" }) + r.clock = schedT0.Add(20 * time.Second) + r.tick(t) + s := r.outcome(t, id, ScheduleOK, "There are 2 players online") + if !sameTime(s.NextRunAt, schedT0.AddDate(0, 0, 1)) || !sameTime(s.LastRunAt, r.clock) || r.con.gotCommand != "list" { + t.Fatalf("next %v, last %v, console %q", s.NextRunAt, s.LastRunAt, r.con.gotCommand) + } + if got := r.runAudits(); len(got) != 1 || + got[0] != `{"action":"command","detail":"There are 2 players online","result":"ok","schedule":1}` { + t.Fatalf("run audits %v", got) + } + r.tick(t) // not due again + if r.con.calls != 1 || len(r.runAudits()) != 1 { + t.Fatalf("ran again: %d console calls", r.con.calls) + } + }) + + t.Run("not due yet: nothing happens", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleCommand, schedT0.Add(time.Second)) + r.tick(t) + if s := r.st.row(t, id); r.con.calls != 0 || s.LastRunAt != nil || !sameTime(s.NextRunAt, schedT0.Add(time.Second)) { + t.Fatalf("ran early: %+v", s) + } + }) + + t.Run("a long reply is cut at 500 bytes on a rune boundary", func(t *testing.T) { + r := newSchedRig(t) + r.con.reply = "x" + strings.Repeat("猫", 200) + id := r.schedule(ScheduleCommand, schedT0) + r.tick(t) + if s := r.st.row(t, id); s.LastDetail != "x"+strings.Repeat("猫", 166) { + t.Fatalf("detail is %d bytes: %q", len(s.LastDetail), s.LastDetail) + } + }) + + t.Run("a command on a stopped server is skipped", func(t *testing.T) { + r := newSchedRig(t) + r.stopped() + id := r.schedule(ScheduleCommand, schedT0) + r.tick(t) + r.outcome(t, id, ScheduleSkipped, "the server was not running") + if r.con.calls != 0 { + t.Fatal("dialed a stopped server") + } + }) + + t.Run("an unreachable console fails the run", func(t *testing.T) { + r := newSchedRig(t) + r.con.err = ErrConsoleUnavailable + id := r.schedule(ScheduleCommand, schedT0) + r.tick(t) + r.outcome(t, id, ScheduleFailed, "the server console could not be reached") + }) + + t.Run("ten minutes late still runs; later is missed", func(t *testing.T) { + r := newSchedRig(t) + late := r.schedule(ScheduleCommand, schedT0.Add(-scheduleMissGrace)) + missed := r.schedule(ScheduleCommand, schedT0.Add(-scheduleMissGrace-time.Second)) + r.tick(t) + r.outcome(t, late, ScheduleOK, "") + s := r.outcome(t, missed, ScheduleMissed, "felis-api was not running at the scheduled time") + if r.con.calls != 1 || !sameTime(s.LastRunAt, schedT0.Add(-scheduleMissGrace-time.Second)) || + !sameTime(s.NextRunAt, time.Date(2026, 9, 29, 2, 49, 0, 0, time.UTC)) { + t.Fatalf("console calls %d, missed row %+v", r.con.calls, s) + } + }) + + t.Run("a new owner disables the schedule instead of running it", func(t *testing.T) { + r := newSchedRig(t) + r.st.owners["survival"] = "newowner" + id := r.schedule(ScheduleCommand, schedT0) + r.tick(t) + s := r.outcome(t, id, ScheduleSkipped, "the server has a new owner since this schedule was saved; save it again to use it") + if s.Enabled || s.NextRunAt != nil || r.con.calls != 0 { + t.Fatalf("still enabled or ran: %+v, %d console calls", s, r.con.calls) + } + }) + + t.Run("the schedule of a released server stops too", func(t *testing.T) { + r := newSchedRig(t) + r.st.owners["survival"] = "" + id := r.schedule(ScheduleStop, schedT0) + r.tick(t) + if s := r.st.row(t, id); s.Enabled || r.cl.desired["survival"] != "" { + t.Fatalf("ran on a released server: %+v", s) + } + }) + + t.Run("stop", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleStop, schedT0) + r.tick(t) + r.outcome(t, id, ScheduleOK, "") + if r.cl.desired["survival"] != v1alpha1.DesiredStopped { + t.Fatalf("desired %q", r.cl.desired["survival"]) + } + r2 := newSchedRig(t) + r2.stopped() + id = r2.schedule(ScheduleStop, schedT0) + r2.tick(t) + r2.outcome(t, id, ScheduleSkipped, "the server was already stopped") + if r2.cl.desired["survival"] != "" { + t.Fatal("stopped a stopped server again") + } + }) + + t.Run("a server gone since the schedule was saved", func(t *testing.T) { + r := newSchedRig(t) + delete(r.cl.byName, "survival") + id := r.schedule(ScheduleStop, schedT0) + r.tick(t) + r.outcome(t, id, ScheduleSkipped, "the server no longer exists") + }) +} + +func TestScheduleRunnerStart(t *testing.T) { + cases := []struct { + name string + setup func(r *schedRig) + result string + detail string + desired v1alpha1.DesiredState + retried bool + }{ + {"a stopped server starts", func(r *schedRig) { r.stopped() }, ScheduleOK, "", v1alpha1.DesiredRunning, false}, + {"a running server is left alone", func(*schedRig) {}, ScheduleSkipped, "the server was already running", "", false}, + {"a failed server is retried", func(r *schedRig) { r.cl.byName["survival"].Phase = string(v1alpha1.PhaseFailed) }, + ScheduleOK, "", v1alpha1.DesiredRunning, true}, + {"a failed start of a stopped server starts plainly", func(r *schedRig) { + r.stopped() + r.cl.byName["survival"].Phase = string(v1alpha1.PhaseFailed) + }, ScheduleOK, "", v1alpha1.DesiredRunning, false}, + {"the running cap holds it back", func(r *schedRig) { + r.stopped() + r.a.MaxRunningServers = 1 + r.cl.list = []ServerInfo{{Name: "other", DesiredState: string(v1alpha1.DesiredRunning)}} + }, ScheduleSkipped, "the cluster is at its running-server cap", "", false}, + {"a pending retirement holds it back", func(r *schedRig) { + r.stopped() + r.repo.byName["survival"].Retire = &RetireState{} + }, ScheduleSkipped, "the server is being given up or deleted", "", false}, + {"a busy world holds it back", func(r *schedRig) { + r.stopped() + r.cl.wakeErr["survival"] = &MaintenanceBusyError{Kind: maintenance.KindBackup} + }, ScheduleSkipped, "the world is busy with a backup", "", false}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + r := newSchedRig(t) + tc.setup(r) + id := r.schedule(ScheduleStart, schedT0) + r.tick(t) + r.outcome(t, id, tc.result, tc.detail) + if r.cl.desired["survival"] != tc.desired || (len(r.cl.retried) == 1) != tc.retried { + t.Fatalf("desired %q, retried %v", r.cl.desired["survival"], r.cl.retried) + } + }) + } +} + +func TestScheduleRunnerWarning(t *testing.T) { + t.Run("told once, at the warning time", func(t *testing.T) { + r := newSchedRig(t) + due := schedT0.Add(5 * time.Minute) + id := r.schedule(ScheduleRestart, due, func(s *Schedule) { s.WarnMinutes = 5 }) + r.clock = schedT0.Add(-time.Second) + r.tick(t) + if r.con.calls != 0 { + t.Fatal("warned early") + } + r.clock = schedT0 + r.tick(t) + r.clock = schedT0.Add(15 * time.Second) + r.tick(t) + if r.con.calls != 1 || r.con.gotCommand != "say [Felis] 服务器将在 5 分钟后重启 / Server restarts in 5 min" { + t.Fatalf("%d warnings, last %q", r.con.calls, r.con.gotCommand) + } + if s := r.st.row(t, id); !sameTime(s.WarnedFor, due) || s.LastRunAt != nil { + t.Fatalf("after the warning %+v", s) + } + r.clock = due + r.tick(t) // the run itself claims and clears the marker + if s := r.st.row(t, id); s.WarnedFor != nil || s.RunState != runStopping { + t.Fatalf("after the run started %+v", s) + } + }) + + t.Run("late warnings say how long is really left", func(t *testing.T) { + for action, want := range map[string]string{ + ScheduleStop: "say [Felis] 服务器将在 2 分钟后关闭 / Server stops in 2 min", + ScheduleBackup: "say [Felis] 服务器将在 2 分钟后暂停做备份,完成后自动恢复 / Server pauses for a backup in 2 min and comes back after", + } { + r := newSchedRig(t) + r.schedule(action, schedT0.Add(90*time.Second), func(s *Schedule) { s.WarnMinutes = 10 }) + r.tick(t) + if r.con.gotCommand != want { + t.Fatalf("%s warned %q, want %q", action, r.con.gotCommand, want) + } + } + }) + + t.Run("nobody to tell on a stopped server, none without a lead time", func(t *testing.T) { + r := newSchedRig(t) + r.stopped() + a := r.schedule(ScheduleRestart, schedT0.Add(time.Minute), func(s *Schedule) { s.WarnMinutes = 5 }) + r.tick(t) + r2 := newSchedRig(t) + r2.schedule(ScheduleRestart, schedT0.Add(time.Minute)) + r2.tick(t) + if r.con.calls != 0 || r2.con.calls != 0 || r.st.row(t, a).WarnedFor != nil { + t.Fatalf("warned: %d / %d", r.con.calls, r2.con.calls) + } + }) +} + +func TestScheduleRunnerRestart(t *testing.T) { + t.Run("stops, waits for the pod, starts", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleRestart, schedT0) + r.tick(t) + r.step(t, id, runStopping) + if r.cl.desired["survival"] != v1alpha1.DesiredStopped { + t.Fatalf("desired %q", r.cl.desired["survival"]) + } + // Still shutting down. + info := r.cl.byName["survival"] + info.DesiredState = string(v1alpha1.DesiredStopped) + r.clock = schedT0.Add(time.Minute) + r.tick(t) + r.step(t, id, runStopping) + info.Ready = false + r.tick(t) + r.step(t, id, runStopping) // not Ready, but not Stopped either + info.Phase = string(v1alpha1.PhaseStopped) + r.tick(t) + r.outcome(t, id, ScheduleOK, "") + if r.cl.desired["survival"] != v1alpha1.DesiredRunning { + t.Fatalf("desired %q", r.cl.desired["survival"]) + } + if got := r.runAudits(); len(got) != 1 || got[0] != `{"action":"restart","detail":"","result":"ok","schedule":1}` { + t.Fatalf("run audits %v", got) + } + }) + + t.Run("a stopped server is not restarted", func(t *testing.T) { + r := newSchedRig(t) + r.stopped() + id := r.schedule(ScheduleRestart, schedT0) + r.tick(t) + r.outcome(t, id, ScheduleSkipped, "the server was not running") + }) + + t.Run("someone starting it meanwhile ends the run", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleRestart, schedT0) + r.tick(t) + r.clock = schedT0.Add(30 * time.Second) + r.tick(t) // info still says desired Running: a person started it again + r.outcome(t, id, ScheduleSkipped, "someone started the server before the run finished") + }) + + t.Run("a pod that does not stop in 15 minutes: give up and start it again", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleRestart, schedT0) + r.tick(t) + r.cl.byName["survival"].DesiredState = string(v1alpha1.DesiredStopped) + r.clock = schedT0.Add(stopWait - time.Second) + r.tick(t) + r.step(t, id, runStopping) + r.clock = schedT0.Add(stopWait) + r.tick(t) + r.outcome(t, id, ScheduleFailed, "the server did not stop within 15 minutes; it was started again") + if r.cl.desired["survival"] != v1alpha1.DesiredRunning { + t.Fatalf("desired %q", r.cl.desired["survival"]) + } + }) + + t.Run("the running cap: keep trying for 15 minutes, then fail", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleRestart, schedT0) + r.tick(t) + r.stopped() + r.a.MaxRunningServers = 1 + r.cl.list = []ServerInfo{{Name: "other", DesiredState: string(v1alpha1.DesiredRunning)}} + r.clock = schedT0.Add(time.Minute) + r.tick(t) + r.step(t, id, runStarting) + r.clock = schedT0.Add(time.Minute + startWait - time.Second) + r.tick(t) + r.step(t, id, runStarting) + r.clock = schedT0.Add(time.Minute + startWait) + r.tick(t) + r.outcome(t, id, ScheduleFailed, "could not start the server again: the cluster is at its running-server cap") + }) + + t.Run("the cap clearing lets it start", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleRestart, schedT0) + r.tick(t) + r.stopped() + r.a.MaxRunningServers = 1 + r.cl.list = []ServerInfo{{Name: "other", DesiredState: string(v1alpha1.DesiredRunning)}} + r.tick(t) + r.cl.list = nil + r.clock = schedT0.Add(time.Minute) + r.tick(t) + r.outcome(t, id, ScheduleOK, "") + }) +} + +func TestScheduleRunnerBackup(t *testing.T) { + job := func(state, message string, at time.Time) AsyncJob { + return AsyncJob{Kind: "backup", State: state, Message: message, StartedAt: at, Scheduled: true} + } + + t.Run("a running server: stop, back up, start", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.step(t, id, runStopping) + r.clock = schedT0.Add(time.Minute) + r.cl.maintErr["survival"] = ErrNotStopped + r.cl.byName["survival"].DesiredState = string(v1alpha1.DesiredStopped) + r.tick(t) + r.step(t, id, runStopping) // the pod is still going down + delete(r.cl.maintErr, "survival") + r.stopped() + r.clock = schedT0.Add(2 * time.Minute) + r.tick(t) + r.step(t, id, runBackingUp) + if strings.Join(r.cl.acquired, ",") != "survival:backup" || strings.Join(r.cl.released, ",") != "survival" || + len(r.b.scheduled) != 1 || r.b.scheduled[0] != (ScheduledCandidate{Name: "survival", OwnerID: "owner1"}) { + t.Fatalf("acquired %v released %v backups %v", r.cl.acquired, r.cl.released, r.b.scheduled) + } + r.jobs.jobs = []AsyncJob{job("running", "", r.clock.Add(-time.Minute))} // clock skew + r.clock = schedT0.Add(30 * time.Minute) + r.tick(t) + r.step(t, id, runBackingUp) + r.jobs.jobs = []AsyncJob{job("succeeded", "", schedT0.Add(time.Minute))} + r.tick(t) + r.outcome(t, id, ScheduleOK, "") + if r.cl.desired["survival"] != v1alpha1.DesiredRunning || r.jobs.got != "survival" { + t.Fatalf("desired %q, jobs read for %q", r.cl.desired["survival"], r.jobs.got) + } + }) + + t.Run("a stopped server is backed up at once and left stopped", func(t *testing.T) { + r := newSchedRig(t) + r.stopped() + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.step(t, id, runBackingUp) + r.jobs.jobs = []AsyncJob{job("succeeded", "", schedT0)} + r.tick(t) + r.outcome(t, id, ScheduleOK, "") + if r.cl.desired["survival"] != "" { + t.Fatalf("desired %q, want untouched", r.cl.desired["survival"]) + } + }) + + t.Run("a failed backup is reported and the server still comes back", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.stopped() + r.tick(t) + r.jobs.jobs = []AsyncJob{job("failed", "disk full", schedT0), job("succeeded", "", schedT0.Add(-time.Hour))} + r.tick(t) + r.outcome(t, id, ScheduleFailed, "the backup failed: disk full") + if r.cl.desired["survival"] != v1alpha1.DesiredRunning { + t.Fatalf("desired %q", r.cl.desired["survival"]) + } + }) + + t.Run("an older or unscheduled Job is not this run's", func(t *testing.T) { + r := newSchedRig(t) + r.stopped() + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + manual := job("failed", "x", schedT0) + manual.Scheduled = false + r.jobs.jobs = []AsyncJob{manual, job("succeeded", "", schedT0.Add(-backupJobSkew-time.Second))} + r.clock = schedT0.Add(backupJobSkew - time.Second) + r.tick(t) + r.step(t, id, runBackingUp) + r.clock = schedT0.Add(backupJobSkew) + r.tick(t) + r.outcome(t, id, ScheduleFailed, "the backup Job is gone before it could be checked; see the Backups page") + }) + + t.Run("two scheduled Jobs since the step: the newest counts", func(t *testing.T) { + r := newSchedRig(t) + r.stopped() + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.jobs.jobs = []AsyncJob{job("succeeded", "", schedT0.Add(time.Minute)), job("failed", "disk full", schedT0)} + r.tick(t) + r.outcome(t, id, ScheduleOK, "") + }) + + t.Run("a Job running past 45 minutes", func(t *testing.T) { + r := newSchedRig(t) + r.stopped() + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.jobs.jobs = []AsyncJob{job("running", "", schedT0)} + r.clock = schedT0.Add(backupWait - time.Second) + r.tick(t) + r.step(t, id, runBackingUp) + r.clock = schedT0.Add(backupWait) + r.tick(t) + r.outcome(t, id, ScheduleFailed, "the backup did not finish within 45 minutes") + }) + + t.Run("a world kept busy for 15 minutes", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.stopped() + r.cl.maintErr["survival"] = &MaintenanceBusyError{Kind: maintenance.KindRestore} + r.clock = schedT0.Add(stopWait - time.Second) + r.tick(t) + r.step(t, id, runStopping) + r.clock = schedT0.Add(stopWait) + r.tick(t) + r.outcome(t, id, ScheduleFailed, "another operation kept the world busy for 15 minutes; it was started again") + if len(r.b.scheduled) != 0 || r.cl.desired["survival"] != v1alpha1.DesiredRunning { + t.Fatalf("backups %v, desired %q", r.b.scheduled, r.cl.desired["survival"]) + } + }) + + t.Run("a pod that does not stop in 15 minutes: no backup, started again", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.cl.byName["survival"].DesiredState = string(v1alpha1.DesiredStopped) + r.cl.maintErr["survival"] = ErrNotStopped + r.clock = schedT0.Add(stopWait) + r.tick(t) + r.outcome(t, id, ScheduleFailed, "the server did not stop within 15 minutes; it was started again") + if len(r.b.scheduled) != 0 { + t.Fatalf("backups %v", r.b.scheduled) + } + }) + + t.Run("a failed Job without a message", func(t *testing.T) { + r := newSchedRig(t) + r.stopped() + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.jobs.jobs = []AsyncJob{job("failed", "", schedT0)} + r.tick(t) + r.outcome(t, id, ScheduleFailed, "the backup failed:") + }) + + t.Run("the backup cannot start", func(t *testing.T) { + r := newSchedRig(t) + r.b.err = errors.New("quota") + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.stopped() + r.tick(t) + r.outcome(t, id, ScheduleFailed, "could not start the backup: quota; the server was started again") + if strings.Join(r.cl.released, ",") != "survival" { + t.Fatalf("released %v", r.cl.released) + } + }) + + t.Run("refused before touching the server", func(t *testing.T) { + cases := []struct { + name, result, detail string + setup func(r *schedRig) + }{ + {"no backup executor", ScheduleFailed, "backups are not configured", func(r *schedRig) { r.a.Backuper = &fakeBackuper{} }}, + {"store full", ScheduleFailed, "the backup store is full; ask an administrator to free space", func(r *schedRig) { + r.a.BackupStoreCap = 1 + r.repo.backups = append(r.repo.backups, fakeBackup{view: BackupView{Status: "present", SizeBytes: 1}}) + }}, + {"no world yet", ScheduleSkipped, "the server has no world yet", func(r *schedRig) { r.cl.noWorld["survival"] = true }}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + r := newSchedRig(t) + tc.setup(r) + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.outcome(t, id, tc.result, tc.detail) + if r.cl.desired["survival"] != "" || len(r.b.scheduled) != 0 { + t.Fatalf("desired %q, backups %v", r.cl.desired["survival"], r.b.scheduled) + } + }) + } + }) +} + +func TestScheduleRunnerStaleClaim(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleCommand, schedT0.Add(time.Hour), func(s *Schedule) { + s.RunState, s.RunStepAt = runClaimed, ptrTime(schedT0.Add(-claimedStale+time.Second)) + }) + r.tick(t) + r.step(t, id, runClaimed) + r.clock = schedT0.Add(time.Second) + r.tick(t) + r.outcome(t, id, ScheduleFailed, "felis-api stopped in the middle of this run") + if r.con.calls != 0 { + t.Fatal("re-ran the command of a stale claim") + } +} + +func TestStripFormatting(t *testing.T) { + if got := stripFormatting(" §l§aHi§r §x§§ok "); got != "Hi ok" { + t.Fatalf("stripFormatting = %q", got) + } +} + +func TestScheduleRunnerEdges(t *testing.T) { + t.Run("no store: nothing to run", func(t *testing.T) { + r := newSchedRig(t) + r.a.Schedules = nil + r.tick(t) + }) + + t.Run("no console: no warning, and a command fails", func(t *testing.T) { + r := newSchedRig(t) + r.a.Console = nil + warn := r.schedule(ScheduleRestart, schedT0.Add(time.Minute), func(s *Schedule) { s.WarnMinutes = 5 }) + cmd := r.schedule(ScheduleCommand, schedT0) + r.tick(t) + r.outcome(t, cmd, ScheduleFailed, "the console is not configured") + if s := r.st.row(t, warn); s.WarnedFor != nil { + t.Fatalf("marked warned without a console: %+v", s) + } + }) + + t.Run("a stopped server's backup gives up without starting it", func(t *testing.T) { + r := newSchedRig(t) + r.stopped() + r.cl.maintErr["survival"] = &MaintenanceBusyError{Kind: maintenance.KindFileWrite} + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.step(t, id, runStopping) + r.clock = schedT0.Add(stopWait) + r.tick(t) + r.outcome(t, id, ScheduleFailed, "another operation kept the world busy for 15 minutes") + if r.cl.desired["survival"] != "" { + t.Fatalf("desired %q, want untouched", r.cl.desired["survival"]) + } + }) + + t.Run("a stopped server's backup that cannot start leaves it stopped", func(t *testing.T) { + r := newSchedRig(t) + r.stopped() + r.b.err = errors.New("quota") + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.outcome(t, id, ScheduleFailed, "could not start the backup: quota") + if r.cl.desired["survival"] != "" { + t.Fatalf("desired %q, want untouched", r.cl.desired["survival"]) + } + }) + + t.Run("the backup executor gone mid-run (felis-api restarted without it)", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.stopped() + r.a.Backuper = &fakeBackuper{} + r.tick(t) + r.outcome(t, id, ScheduleFailed, "backups are not configured") + }) + + t.Run("the server deleted while its world is being locked", func(t *testing.T) { + r := newSchedRig(t) + r.stopped() + r.cl.maintErr["survival"] = ErrNotFound + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.outcome(t, id, ScheduleSkipped, "the server no longer exists") + }) + + t.Run("without Job status the backup counts as done", func(t *testing.T) { + r := newSchedRig(t) + r.stopped() + r.a.JobStatus = nil + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.step(t, id, runBackingUp) + r.tick(t) + r.outcome(t, id, ScheduleOK, "") + }) + + t.Run("a failed backup whose server cannot start either names both", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleBackup, schedT0) + r.tick(t) + r.stopped() + r.tick(t) + r.repo.byName["survival"].Retire = &RetireState{} + r.jobs.jobs = []AsyncJob{{Kind: "backup", State: "failed", Message: "disk full", StartedAt: schedT0, Scheduled: true}} + r.tick(t) + r.outcome(t, id, ScheduleFailed, "the backup failed: disk full; could not start the server again: the server is being given up or deleted") + }) + + t.Run("a retirement stops the restart at once", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleRestart, schedT0) + r.tick(t) + r.stopped() + r.repo.byName["survival"].Retire = &RetireState{} + r.tick(t) + r.outcome(t, id, ScheduleFailed, "could not start the server again: the server is being given up or deleted") + }) + + t.Run("the server deleted before it could start again", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleRestart, schedT0) + r.tick(t) + r.stopped() + delete(r.repo.byName, "survival") + r.tick(t) + r.outcome(t, id, ScheduleSkipped, "the server no longer exists") + }) +} + +func TestScheduleNextRunMidnightJump(t *testing.T) { + // Chile moves its clocks from 00:00 to 01:00 on 2026-09-06: that Sunday has + // no midnight, so the weekday is read at midday. + santiago := mustZone(t, "America/Santiago") + s := Schedule{MinuteOfDay: 12 * 60, Weekdays: 1, Timezone: "America/Santiago"} + after := time.Date(2026, 9, 5, 13, 0, 0, 0, santiago) + if got, want := s.nextRun(after), time.Date(2026, 9, 6, 12, 0, 0, 0, santiago); !got.Equal(want) { + t.Fatalf("nextRun = %s, want %s", got, want) + } +} + +// TestScheduleFinishLostRace: a finish that another felis-api beat to the row +// changes nothing and leaves no second schedule.run audit. +func TestScheduleFinishLostRace(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleRestart, schedT0, func(s *Schedule) { s.LastResult = ScheduleOK }) + s := r.st.row(t, id) + if err := r.a.finishRun(t.Context(), s, runStarting, ScheduleFailed, "late"); err != nil { + t.Fatal(err) + } + r.outcome(t, id, ScheduleOK, "") + if len(r.repo.audits) != 0 { + t.Fatalf("audits %+v", r.repo.audits) + } +} diff --git a/internal/api/schedules.go b/internal/api/schedules.go new file mode 100644 index 0000000..a6f0974 --- /dev/null +++ b/internal/api/schedules.go @@ -0,0 +1,293 @@ +package api + +import ( + "context" + "errors" + "net/http" + "slices" + "strings" + "time" + _ "time/tzdata" // a schedule's timezone resolves on a host without zoneinfo + "unicode/utf8" +) + +// Scheduled tasks. An owner or admin asks felis-api to act on a server at set +// times: run a console command, restart it, stop it, start it, or back it up. +// A backup of a running server stops it, takes the backup and starts it again, +// so a world kept up around the clock still gets restore points; the +// BackupScheduler only ever backs up a stopped world. The schedules live in +// server_schedules (migration 0036), and RunSchedules (schedulerunner.go) is +// the loop that fires them. + +// Schedule actions. +const ( + ScheduleCommand = "command" + ScheduleRestart = "restart" + ScheduleStop = "stop" + ScheduleStart = "start" + ScheduleBackup = "backup" +) + +// Run results (Schedule.LastResult). A run in progress has none yet. +const ( + ScheduleOK = "ok" + ScheduleSkipped = "skipped" + ScheduleFailed = "failed" + // ScheduleMissed is a run felis-api was not up for: it is dropped once it + // is scheduleMissGrace late, so a restart never lands hours after its time. + ScheduleMissed = "missed" +) + +const ( + // maxSchedulesPerServer bounds the rows one server can hold. + maxSchedulesPerServer = 20 + // maxScheduleLabel bounds a label, in characters. + maxScheduleLabel = 64 + // maxScheduleDetail bounds the stored outcome of a run (a command's reply + // can be long), in bytes. + maxScheduleDetail = 500 + // minPowerEveryMinutes is the shortest interval of an action that takes the + // server down; a command may run every scheduleEveryMinutes[0]. + minPowerEveryMinutes = 60 +) + +var ( + // scheduleEveryMinutes are the intervals a repeating schedule may use: each + // divides a day, so the runs sit at the same clock times every day. + scheduleEveryMinutes = []int{15, 30, 60, 120, 180, 240, 360, 480, 720} + // scheduleWarnMinutes are the lead times of the in-game warning. + scheduleWarnMinutes = []int{0, 1, 5, 10, 15, 30} +) + +var ( + // ErrScheduleLimit means the server already holds maxSchedulesPerServer. + ErrScheduleLimit = errors.New("schedule limit reached") + // ErrScheduleRunning means the schedule has a run in progress, which must + // finish before the schedule is changed, deleted or run again. + ErrScheduleRunning = errors.New("schedule is running") +) + +// Schedule is one server_schedules row as the owner sees it. OwnerID, the +// warning marker and the run step stay off the wire; RunState tells the panel +// what a run in progress is waiting for. +type Schedule struct { + ID int64 `json:"id"` + Server string `json:"server"` + Label string `json:"label"` + Action string `json:"action"` + // Command is the console command of a command schedule, without a slash. + Command string `json:"command"` + // EveryMinutes repeats the schedule at every multiple of it since local + // midnight; 0 runs it once a day at MinuteOfDay. Either way only on the + // Weekdays (a bitmask, bit 0 Sunday), in Timezone. + EveryMinutes int `json:"every_minutes"` + MinuteOfDay int `json:"minute_of_day"` + Weekdays int `json:"weekdays"` + Timezone string `json:"timezone"` + // WarnMinutes is how long before a restart, stop or backup the players on + // the server are told it is coming; 0 says nothing. + WarnMinutes int `json:"warn_minutes"` + Enabled bool `json:"enabled"` + // NextRunAt is nil while the schedule is disabled. + NextRunAt *time.Time `json:"next_run_at"` + // RunState is the step of a run in progress (runStopping, runBackingUp, + // runStarting, or runClaimed while its first step runs); empty otherwise. + RunState string `json:"run_state"` + LastRunAt *time.Time `json:"last_run_at"` + LastResult string `json:"last_result"` + LastDetail string `json:"last_detail"` + CreatedBy string `json:"created_by"` + CreatedAt time.Time `json:"created_at"` + + // OwnerID is the server's owner when the schedule was saved (empty for an + // unowned server). The runner fires the schedule only while it still is. + OwnerID string `json:"-"` + WarnedFor *time.Time `json:"-"` + RunResume bool `json:"-"` + RunStepAt *time.Time `json:"-"` +} + +// DueSchedule is a schedule the runner has to look at, with the server's +// current owner. +type DueSchedule struct { + Schedule + ServerOwner string +} + +// ServerSchedules stores the schedules (PGRepo). The run methods are +// compare-and-set writes that report whether they applied, so two felis-api +// processes side by side during a rollout never fire one run twice. +type ServerSchedules interface { + // ListSchedules lists a server's schedules, oldest first. + ListSchedules(ctx context.Context, server string) ([]Schedule, error) + // GetSchedule reads one schedule of server, or ErrNotFound. + GetSchedule(ctx context.Context, server string, id int64) (*Schedule, error) + // CreateSchedule inserts s and fills in its ID and CreatedAt, or returns + // ErrScheduleLimit when the server already holds limit schedules. + CreateSchedule(ctx context.Context, s *Schedule, limit int) error + // UpdateSchedule writes s's settings, owner and next run and clears its + // warning marker: ErrNotFound, or ErrScheduleRunning during a run. + UpdateSchedule(ctx context.Context, s *Schedule) error + // DeleteSchedule removes a schedule: ErrNotFound, or ErrScheduleRunning + // during a run. + DeleteSchedule(ctx context.Context, server string, id int64) error + + // DueSchedules lists the enabled schedules of live servers due by horizon + // and every schedule with a run in progress. + DueSchedules(ctx context.Context, horizon time.Time) ([]DueSchedule, error) + // ClaimScheduleRun starts a run (run state runClaimed, a fresh result) of + // a schedule without one. With due set it claims only the enabled run due + // then and moves the schedule on to next; nil due is a run on request that + // leaves the next run where it is. + ClaimScheduleRun(ctx context.Context, id int64, due, next *time.Time, now time.Time) (bool, error) + // AdvanceScheduleRun moves a run from step from to step to, recording + // resume and, when result is set, the outcome so far. + AdvanceScheduleRun(ctx context.Context, id int64, from, to string, resume bool, result, detail string, now time.Time) (bool, error) + // FinishScheduleRun ends a run at step from with its outcome. + FinishScheduleRun(ctx context.Context, id int64, from, result, detail string) (bool, error) + // MissScheduleRun records the run due then as missed and moves on to next. + MissScheduleRun(ctx context.Context, id int64, due, next time.Time, detail string) (bool, error) + // WarnScheduleRun marks the run due then as warned about, once. + WarnScheduleRun(ctx context.Context, id int64, due time.Time) (bool, error) + // DisableSchedule turns off an enabled schedule without a run in progress + // and records why. + DisableSchedule(ctx context.Context, id int64, detail string) (bool, error) +} + +// scheduleInput is the body of POST /servers/{name}/schedules and of PUT +// /servers/{name}/schedules/{id}. Enabled defaults to true. +type scheduleInput struct { + Label string `json:"label"` + Action string `json:"action"` + Command string `json:"command"` + EveryMinutes int `json:"every_minutes"` + MinuteOfDay int `json:"minute_of_day"` + Weekdays int `json:"weekdays"` + Timezone string `json:"timezone"` + WarnMinutes int `json:"warn_minutes"` + Enabled *bool `json:"enabled"` +} + +func badSchedule(format string, a ...any) error { + return newError(http.StatusBadRequest, "bad_schedule", format, a...) +} + +// apply validates in and writes it onto s. +func (in *scheduleInput) apply(s *Schedule) error { + label := strings.TrimSpace(in.Label) + if utf8.RuneCountInString(label) > maxScheduleLabel { + return badSchedule("label too long (max %d characters)", maxScheduleLabel) + } + if strings.IndexFunc(label, func(c rune) bool { return c < 0x20 || c == 0x7f }) >= 0 { + return badSchedule("label must be a single line") + } + + var command string + switch in.Action { + case ScheduleCommand: + c, err := normalizeConsoleCommand(in.Command) + if err != nil { + return err + } + command = c + case ScheduleRestart, ScheduleStop, ScheduleStart, ScheduleBackup: + if strings.TrimSpace(in.Command) != "" { + return badSchedule("only a command schedule takes a command") + } + default: + return badSchedule("action must be one of command, restart, stop, start, backup") + } + + switch { + case in.EveryMinutes == 0: + if in.MinuteOfDay < 0 || in.MinuteOfDay > 24*60-1 { + return badSchedule("minute_of_day must be between 0 and 1439") + } + case !slices.Contains(scheduleEveryMinutes, in.EveryMinutes): + return badSchedule("every_minutes must be 0 or one of 15, 30, 60, 120, 180, 240, 360, 480, 720") + case in.EveryMinutes < minPowerEveryMinutes && in.Action != ScheduleCommand: + return badSchedule("a %s can repeat at most every %d minutes", in.Action, minPowerEveryMinutes) + case in.MinuteOfDay != 0: + return badSchedule("a repeating schedule has no minute_of_day") + } + if in.Weekdays < 1 || in.Weekdays > 0x7f { + return badSchedule("weekdays must name at least one day (bits 0 to 6, Sunday first)") + } + + tz := strings.TrimSpace(in.Timezone) + if tz == "" || tz == "Local" { + return badSchedule("timezone must be an IANA zone name such as Asia/Shanghai") + } + if _, err := time.LoadLocation(tz); err != nil { + return badSchedule("unknown timezone %q", tz) + } + + if !slices.Contains(scheduleWarnMinutes, in.WarnMinutes) { + return badSchedule("warn_minutes must be one of 0, 1, 5, 10, 15, 30") + } + // A restart, stop or backup repeats at most hourly, so its warning (30 + // minutes ahead at most) always comes after the run before it. + if in.WarnMinutes > 0 && (in.Action == ScheduleCommand || in.Action == ScheduleStart) { + return badSchedule("only a restart, stop or backup warns the players") + } + + s.Label, s.Action, s.Command = label, in.Action, command + s.EveryMinutes, s.MinuteOfDay, s.Weekdays = in.EveryMinutes, in.MinuteOfDay, in.Weekdays + s.Timezone, s.WarnMinutes = tz, in.WarnMinutes + s.Enabled = in.Enabled == nil || *in.Enabled + return nil +} + +// nextRun is the first run of s strictly after after, or the zero time for a +// schedule that names no weekday. The runs are wall-clock times in s's zone: +// a time the zone skips at the start of daylight saving runs an hour early +// (time.Date reads it with the offset before the jump), and one it repeats +// runs once, the first time the clock shows it. +func (s *Schedule) nextRun(after time.Time) time.Time { + loc, err := time.LoadLocation(s.Timezone) + if err != nil { + loc = time.UTC // validated when saved; a zone later dropped runs in UTC + } + y, m, d := after.In(loc).Date() + // Eight days: today's runs may all be past, and a schedule of one weekday + // next runs a week from today. + for day := d; day <= d+7; day++ { + if s.Weekdays&(1< 0 { + for at := 0; at < 24*60; at += s.EveryMinutes { + minutes = append(minutes, at) + } + } else { + minutes = []int{s.MinuteOfDay} + } + for _, at := range minutes { + if t := time.Date(y, m, day, at/60, at%60, 0, 0, loc); t.After(after) { + return t + } + } + } + return time.Time{} +} + +// normalizeConsoleCommand makes raw one console command: surrounding space +// and a single leading slash removed (players type "/say hi"; RCON wants +// "say hi"), within maxConsoleCommandLen, and without control characters, so +// one request can never smuggle a second command past a newline. +func normalizeConsoleCommand(raw string) (string, error) { + command := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(raw), "/")) + if command == "" { + return "", newError(http.StatusBadRequest, "bad_request", "command is required") + } + if len(command) > maxConsoleCommandLen { + return "", newError(http.StatusBadRequest, "bad_request", + "command too long (max %d bytes)", maxConsoleCommandLen) + } + if strings.IndexFunc(command, func(c rune) bool { return c < 0x20 }) >= 0 { + return "", newError(http.StatusBadRequest, "bad_request", + "command must be a single line (no control characters)") + } + return command, nil +} diff --git a/internal/api/schedules_test.go b/internal/api/schedules_test.go new file mode 100644 index 0000000..31c595f --- /dev/null +++ b/internal/api/schedules_test.go @@ -0,0 +1,712 @@ +package api + +import ( + "context" + "encoding/json" + "errors" + "fmt" + "io" + "net/http" + "slices" + "strings" + "testing" + "time" + + "felis.lolicon.best/internal/apis/felis/v1alpha1" +) + +var _ ServerSchedules = (*PGRepo)(nil) + +// fakeSchedules is an in-memory ServerSchedules with the compare-and-set +// conditions of the PGRepo queries (internal/pgint/schedules_test.go runs the +// same cases against Postgres). It hands out copies, so a test sees only what +// was written through the interface. +type fakeSchedules struct { + rows map[int64]*Schedule + owners map[string]string // server -> its owner now, for DueSchedules + nextID int64 + claims int +} + +func newFakeSchedules() *fakeSchedules { + return &fakeSchedules{rows: map[int64]*Schedule{}, owners: map[string]string{}} +} + +func cloneSchedule(s *Schedule) *Schedule { + c := *s + for _, p := range []**time.Time{&c.NextRunAt, &c.WarnedFor, &c.RunStepAt, &c.LastRunAt} { + if *p != nil { + *p = ptrTime(**p) + } + } + return &c +} + +func sameTime(p *time.Time, t time.Time) bool { return p != nil && p.Equal(t) } + +// put stores s as it is and returns its id. +func (f *fakeSchedules) put(s Schedule) int64 { + f.nextID++ + s.ID = f.nextID + f.rows[s.ID] = cloneSchedule(&s) + return s.ID +} + +// row reads a schedule back for a test's assertions. +func (f *fakeSchedules) row(t *testing.T, id int64) *Schedule { + t.Helper() + r, ok := f.rows[id] + if !ok { + t.Fatalf("schedule %d is gone", id) + } + return cloneSchedule(r) +} + +func (f *fakeSchedules) ListSchedules(_ context.Context, server string) ([]Schedule, error) { + var out []Schedule + for _, r := range f.rows { + if r.Server == server { + out = append(out, *cloneSchedule(r)) + } + } + slices.SortFunc(out, func(a, b Schedule) int { return int(a.ID - b.ID) }) + return out, nil +} + +func (f *fakeSchedules) GetSchedule(_ context.Context, server string, id int64) (*Schedule, error) { + r, ok := f.rows[id] + if !ok || r.Server != server { + return nil, ErrNotFound + } + return cloneSchedule(r), nil +} + +func (f *fakeSchedules) CreateSchedule(_ context.Context, s *Schedule, limit int) error { + n := 0 + for _, r := range f.rows { + if r.Server == s.Server { + n++ + } + } + if n >= limit { + return ErrScheduleLimit + } + s.CreatedAt = time.Unix(1_700_000_000, 0) + s.ID = f.put(*s) + return nil +} + +func (f *fakeSchedules) idle(server string, id int64) (*Schedule, error) { + r, ok := f.rows[id] + switch { + case !ok || r.Server != server: + return nil, ErrNotFound + case r.RunState != "": + return nil, ErrScheduleRunning + } + return r, nil +} + +func (f *fakeSchedules) UpdateSchedule(_ context.Context, s *Schedule) error { + r, err := f.idle(s.Server, s.ID) + if err != nil { + return err + } + r.OwnerID, r.Label, r.Action, r.Command = s.OwnerID, s.Label, s.Action, s.Command + r.EveryMinutes, r.MinuteOfDay, r.Weekdays, r.Timezone = s.EveryMinutes, s.MinuteOfDay, s.Weekdays, s.Timezone + r.WarnMinutes, r.Enabled, r.NextRunAt, r.WarnedFor = s.WarnMinutes, s.Enabled, s.NextRunAt, nil + if r.NextRunAt != nil { + r.NextRunAt = ptrTime(*r.NextRunAt) + } + return nil +} + +func (f *fakeSchedules) DeleteSchedule(_ context.Context, server string, id int64) error { + if _, err := f.idle(server, id); err != nil { + return err + } + delete(f.rows, id) + return nil +} + +func (f *fakeSchedules) DueSchedules(_ context.Context, horizon time.Time) ([]DueSchedule, error) { + var out []DueSchedule + for _, r := range f.rows { + if r.RunState != "" || (r.Enabled && r.NextRunAt != nil && !r.NextRunAt.After(horizon)) { + out = append(out, DueSchedule{Schedule: *cloneSchedule(r), ServerOwner: f.owners[r.Server]}) + } + } + slices.SortFunc(out, func(a, b DueSchedule) int { + if (a.RunState == "") != (b.RunState == "") { + if a.RunState != "" { + return -1 + } + return 1 + } + if a.NextRunAt != nil && b.NextRunAt != nil && !a.NextRunAt.Equal(*b.NextRunAt) { + return a.NextRunAt.Compare(*b.NextRunAt) + } + return int(a.ID - b.ID) + }) + return out, nil +} + +func (f *fakeSchedules) ClaimScheduleRun(_ context.Context, id int64, due, next *time.Time, now time.Time) (bool, error) { + r, ok := f.rows[id] + if !ok || r.RunState != "" { + return false, nil + } + if due != nil { + if !r.Enabled || !sameTime(r.NextRunAt, *due) { + return false, nil + } + r.NextRunAt, r.WarnedFor = ptrTime(*next), nil + } + f.claims++ + r.RunState, r.RunResume, r.RunStepAt = runClaimed, false, ptrTime(now) + r.LastRunAt, r.LastResult, r.LastDetail = ptrTime(now), "", "" + return true, nil +} + +func (f *fakeSchedules) AdvanceScheduleRun(_ context.Context, id int64, from, to string, resume bool, result, detail string, now time.Time) (bool, error) { + r, ok := f.rows[id] + if !ok || r.RunState != from { + return false, nil + } + r.RunState, r.RunResume, r.RunStepAt = to, resume, ptrTime(now) + if result != "" { + r.LastResult, r.LastDetail = result, detail + } + return true, nil +} + +func (f *fakeSchedules) FinishScheduleRun(_ context.Context, id int64, from, result, detail string) (bool, error) { + r, ok := f.rows[id] + if !ok || from == "" || r.RunState != from { + return false, nil + } + r.RunState, r.RunResume, r.RunStepAt = "", false, nil + r.LastResult, r.LastDetail = result, detail + return true, nil +} + +func (f *fakeSchedules) MissScheduleRun(_ context.Context, id int64, due, next time.Time, detail string) (bool, error) { + r, ok := f.rows[id] + if !ok || r.RunState != "" || !r.Enabled || !sameTime(r.NextRunAt, due) { + return false, nil + } + r.NextRunAt, r.WarnedFor = ptrTime(next), nil + r.LastRunAt, r.LastResult, r.LastDetail = ptrTime(due), ScheduleMissed, detail + return true, nil +} + +func (f *fakeSchedules) WarnScheduleRun(_ context.Context, id int64, due time.Time) (bool, error) { + r, ok := f.rows[id] + if !ok || r.RunState != "" || !r.Enabled || !sameTime(r.NextRunAt, due) || sameTime(r.WarnedFor, due) { + return false, nil + } + r.WarnedFor = ptrTime(due) + return true, nil +} + +func (f *fakeSchedules) DisableSchedule(_ context.Context, id int64, detail string) (bool, error) { + r, ok := f.rows[id] + if !ok || r.RunState != "" || !r.Enabled { + return false, nil + } + r.Enabled, r.NextRunAt, r.WarnedFor = false, nil, nil + r.LastResult, r.LastDetail = ScheduleSkipped, detail + return true, nil +} + +func mustZone(t *testing.T, name string) *time.Location { + t.Helper() + loc, err := time.LoadLocation(name) + if err != nil { + t.Fatal(err) + } + return loc +} + +// TestScheduleNextRun pins the run times: daily and repeating rules, the +// weekday mask read in the schedule's zone, and the two daylight-saving edges. +func TestScheduleNextRun(t *testing.T) { + sh := mustZone(t, "Asia/Shanghai") + ny := mustZone(t, "America/New_York") + const everyDay, weekdaysOnly, monday, saturday = 0x7f, 0x3e, 1 << 1, 1 << 6 + cases := []struct { + name string + s Schedule + after time.Time + want time.Time + }{ + {"daily, later today", + Schedule{MinuteOfDay: 4*60 + 30, Weekdays: everyDay, Timezone: "Asia/Shanghai"}, + time.Date(2026, 9, 28, 1, 0, 0, 0, sh), time.Date(2026, 9, 28, 4, 30, 0, 0, sh)}, + {"daily, today's run is past", + Schedule{MinuteOfDay: 4 * 60, Weekdays: everyDay, Timezone: "Asia/Shanghai"}, + time.Date(2026, 9, 28, 4, 0, 0, 0, sh), time.Date(2026, 9, 29, 4, 0, 0, 0, sh)}, + {"weekday mask in the schedule's zone (Sunday 23:00 UTC is Monday in Shanghai)", + Schedule{MinuteOfDay: 9 * 60, Weekdays: monday, Timezone: "Asia/Shanghai"}, + time.Date(2026, 9, 27, 23, 0, 0, 0, time.UTC), time.Date(2026, 9, 28, 9, 0, 0, 0, sh)}, + {"one weekday, next week, across the month end", + Schedule{MinuteOfDay: 9 * 60, Weekdays: monday, Timezone: "Asia/Shanghai"}, + time.Date(2026, 9, 28, 10, 0, 0, 0, sh), time.Date(2026, 10, 5, 9, 0, 0, 0, sh)}, + {"interval: the next multiple since midnight", + Schedule{EveryMinutes: 180, Weekdays: everyDay, Timezone: "Asia/Shanghai"}, + time.Date(2026, 9, 28, 7, 10, 0, 0, sh), time.Date(2026, 9, 28, 9, 0, 0, 0, sh)}, + {"interval skips the days off (Friday night -> Monday 00:00)", + Schedule{EveryMinutes: 720, Weekdays: weekdaysOnly, Timezone: "Asia/Shanghai"}, + time.Date(2026, 10, 2, 12, 0, 0, 0, sh), time.Date(2026, 10, 5, 0, 0, 0, 0, sh)}, + {"interval, Saturday only, from the week before", + Schedule{EveryMinutes: 15, Weekdays: saturday, Timezone: "UTC"}, + time.Date(2026, 9, 26, 23, 50, 0, 0, time.UTC), time.Date(2026, 10, 3, 0, 0, 0, 0, time.UTC)}, + {"a time daylight saving skips runs an hour early", + Schedule{MinuteOfDay: 2*60 + 30, Weekdays: everyDay, Timezone: "America/New_York"}, + time.Date(2026, 3, 8, 0, 0, 0, 0, ny), time.Date(2026, 3, 8, 6, 30, 0, 0, time.UTC)}, + {"a repeated time runs once: after the first 01:30 comes tomorrow's", + Schedule{MinuteOfDay: 90, Weekdays: everyDay, Timezone: "America/New_York"}, + time.Date(2026, 11, 1, 5, 30, 0, 0, time.UTC), time.Date(2026, 11, 2, 6, 30, 0, 0, time.UTC)}, + {"a zone no longer known runs in UTC", + Schedule{MinuteOfDay: 60, Weekdays: everyDay, Timezone: "Gone/Zone"}, + time.Date(2026, 9, 28, 0, 0, 0, 0, time.UTC), time.Date(2026, 9, 28, 1, 0, 0, 0, time.UTC)}, + {"the first 01:30 on the day the clock goes back", + Schedule{MinuteOfDay: 90, Weekdays: everyDay, Timezone: "America/New_York"}, + time.Date(2026, 11, 1, 0, 0, 0, 0, ny), time.Date(2026, 11, 1, 5, 30, 0, 0, time.UTC)}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + if got := tc.s.nextRun(tc.after); !got.Equal(tc.want) { + t.Fatalf("nextRun(%s) = %s, want %s", tc.after, got, tc.want) + } + }) + } +} + +// TestScheduleInputValidation walks the settings apply refuses, each with the +// one field that makes it wrong, and what a valid body stores. +func TestScheduleInputValidation(t *testing.T) { + valid := func() scheduleInput { + return scheduleInput{Action: ScheduleRestart, MinuteOfDay: 240, Weekdays: 0x7f, Timezone: "Asia/Shanghai", WarnMinutes: 5} + } + bad := []struct { + name string + edit func(*scheduleInput) + code string + msg string + }{ + {"label too long", func(in *scheduleInput) { in.Label = strings.Repeat("猫", 65) }, "bad_schedule", "label too long"}, + {"label with a newline", func(in *scheduleInput) { in.Label = "a\nb" }, "bad_schedule", "single line"}, + {"label with DEL", func(in *scheduleInput) { in.Label = "a\x7fb" }, "bad_schedule", "single line"}, + {"unknown action", func(in *scheduleInput) { in.Action = "reboot" }, "bad_schedule", "action must be"}, + {"command on a restart", func(in *scheduleInput) { in.Command = "say hi" }, "bad_schedule", "only a command schedule"}, + {"command action without a command", func(in *scheduleInput) { in.Action, in.WarnMinutes = ScheduleCommand, 0 }, "bad_request", "command is required"}, + {"two commands in one", func(in *scheduleInput) { in.Action, in.WarnMinutes, in.Command = ScheduleCommand, 0, "say a\nop me" }, "bad_request", "single line"}, + {"minute_of_day negative", func(in *scheduleInput) { in.MinuteOfDay = -1 }, "bad_schedule", "minute_of_day must be"}, + {"minute_of_day 1440", func(in *scheduleInput) { in.MinuteOfDay = 1440 }, "bad_schedule", "minute_of_day must be"}, + {"interval off the list", func(in *scheduleInput) { in.EveryMinutes, in.MinuteOfDay = 45, 0 }, "bad_schedule", "every_minutes must be"}, + {"restart every 30 minutes", func(in *scheduleInput) { in.EveryMinutes, in.MinuteOfDay = 30, 0 }, "bad_schedule", "at most every 60 minutes"}, + {"interval with a minute_of_day", func(in *scheduleInput) { in.EveryMinutes = 60 }, "bad_schedule", "no minute_of_day"}, + {"no weekday", func(in *scheduleInput) { in.Weekdays = 0 }, "bad_schedule", "weekdays"}, + {"weekday bit 7", func(in *scheduleInput) { in.Weekdays = 0x80 }, "bad_schedule", "weekdays"}, + {"empty timezone", func(in *scheduleInput) { in.Timezone = " " }, "bad_schedule", "IANA zone"}, + {"Local", func(in *scheduleInput) { in.Timezone = "Local" }, "bad_schedule", "IANA zone"}, + {"unknown timezone", func(in *scheduleInput) { in.Timezone = "Mars/Olympus" }, "bad_schedule", "unknown timezone"}, + {"warning off the list", func(in *scheduleInput) { in.WarnMinutes = 2 }, "bad_schedule", "warn_minutes must be"}, + {"warning on a command", func(in *scheduleInput) { in.Action, in.Command = ScheduleCommand, "say hi" }, "bad_schedule", "only a restart, stop or backup warns"}, + {"warning on a start", func(in *scheduleInput) { in.Action = ScheduleStart }, "bad_schedule", "only a restart, stop or backup warns"}, + } + for _, tc := range bad { + t.Run(tc.name, func(t *testing.T) { + in := valid() + tc.edit(&in) + err := in.apply(&Schedule{}) + var he *apiError + if !errors.As(err, &he) || he.status != http.StatusBadRequest || he.code != tc.code || !strings.Contains(he.msg, tc.msg) { + t.Fatalf("apply = %v, want 400 %s containing %q", err, tc.code, tc.msg) + } + }) + } + + t.Run("valid bodies store trimmed values", func(t *testing.T) { + in := scheduleInput{Label: " 每晚 ", Action: ScheduleCommand, Command: " /say 晚安 ", EveryMinutes: 15, + Weekdays: 0x41, Timezone: " Europe/Berlin "} + var s Schedule + if err := in.apply(&s); err != nil { + t.Fatal(err) + } + want := Schedule{Label: "每晚", Action: ScheduleCommand, Command: "say 晚安", EveryMinutes: 15, + Weekdays: 0x41, Timezone: "Europe/Berlin", Enabled: true} + if s != want { + t.Fatalf("stored %+v, want %+v", s, want) + } + off := false + in = scheduleInput{Label: strings.Repeat("猫", 64), Action: ScheduleBackup, MinuteOfDay: 1439, Weekdays: 1, + Timezone: "UTC", WarnMinutes: 30, Enabled: &off} + if err := in.apply(&s); err != nil { + t.Fatal(err) + } + if s.Enabled || s.Command != "" || s.WarnMinutes != 30 || s.MinuteOfDay != 1439 { + t.Fatalf("stored %+v", s) + } + }) +} + +// schedRig is a server with a schedule store: survival, owned by owner1, +// running, at 2026-09-28 03:00 UTC. +type schedRig struct { + a *API + repo *fakeRepo + cl *fakeCluster + st *fakeSchedules + con *fakeConsole + b *fakeScheduledBackuper + jobs *fakeJobStatus + clock time.Time +} + +var schedT0 = time.Date(2026, 9, 28, 3, 0, 0, 0, time.UTC) + +func newSchedRig(t *testing.T) *schedRig { + t.Helper() + r := &schedRig{repo: newFakeRepo(), cl: newFakeCluster(), st: newFakeSchedules(), con: &fakeConsole{}, + b: &fakeScheduledBackuper{}, jobs: &fakeJobStatus{}, clock: schedT0} + r.repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"} + r.cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: string(v1alpha1.PhaseRunning), Ready: true, + DesiredState: string(v1alpha1.DesiredRunning)} + r.st.owners["survival"] = "owner1" + r.a = newTestAPI(r.repo, r.cl) + r.a.Now = func() time.Time { return r.clock } + r.a.Schedules, r.a.Console, r.a.Backuper, r.a.JobStatus = r.st, r.con, r.b, r.jobs + return r +} + +func (r *schedRig) stopped() { + info := r.cl.byName["survival"] + info.Phase, info.Ready, info.DesiredState = string(v1alpha1.PhaseStopped), false, string(v1alpha1.DesiredStopped) +} + +// schedule stores a schedule of survival owned by owner1, due at due. +func (r *schedRig) schedule(action string, due time.Time, edit ...func(*Schedule)) int64 { + s := Schedule{Server: "survival", OwnerID: "owner1", Action: action, MinuteOfDay: due.Hour()*60 + due.Minute(), + Weekdays: 0x7f, Timezone: "UTC", Enabled: true, NextRunAt: ptrTime(due), CreatedBy: "owner1@example.net"} + if action == ScheduleCommand { + s.Command = "say hi" + } + for _, e := range edit { + e(&s) + } + return r.st.put(s) +} + +func (r *schedRig) tick(t *testing.T) { + t.Helper() + if err := r.a.RunSchedules(context.Background()); err != nil { + t.Fatalf("RunSchedules: %v", err) + } +} + +var ( + schedOwner = &Principal{UserID: "owner1", Email: "owner1@example.net", Role: "user"} + schedStranger = &Principal{UserID: "other", Email: "other@example.net", Role: "user"} + schedAdmin = &Principal{UserID: "adm", Email: "adm@example.net", Role: "admin", ViaAdminAccess: true} +) + +func (r *schedRig) do(p *Principal, method, target, body string) *http.Response { + r.a.External = staticExternal{p: p} + var h map[string]string + if body != "" { + h = jsonHeader + } + return do(r.a.ExternalHandler(), method, target, body, h).Result() +} + +func decodeSchedule(t *testing.T, res *http.Response) Schedule { + t.Helper() + var s Schedule + if err := json.NewDecoder(res.Body).Decode(&s); err != nil { + t.Fatal(err) + } + return s +} + +func schedErrCode(t *testing.T, res *http.Response) string { + t.Helper() + var raw map[string]map[string]string + if err := json.NewDecoder(res.Body).Decode(&raw); err != nil { + t.Fatal(err) + } + return raw["error"]["code"] +} + +const schedBase = "/api/v1/servers/survival/schedules" + +// TestScheduleRoutesGate: every schedule route answers 400 for a bad name, 404 +// for an unknown server, 403 for a stranger and 503 without a store, before it +// reads anything else. +func TestScheduleRoutesGate(t *testing.T) { + body := `{"action":"restart","minute_of_day":240,"weekdays":127,"timezone":"UTC"}` + routes := []struct{ method, path, body string }{ + {"GET", "", ""}, {"POST", "", body}, {"PUT", "/1", body}, {"DELETE", "/1", ""}, {"POST", "/1/run", ""}, + } + for _, rt := range routes { + t.Run(rt.method+rt.path, func(t *testing.T) { + r := newSchedRig(t) + r.schedule(ScheduleRestart, schedT0.Add(time.Hour)) + check := func(p *Principal, target string, status int, code string) { + t.Helper() + res := r.do(p, rt.method, target, rt.body) + if res.StatusCode != status || schedErrCode(t, res) != code { + t.Fatalf("%s %s = %d, want %d %s", rt.method, target, res.StatusCode, status, code) + } + } + check(schedOwner, "/api/v1/servers/Bad_Name/schedules"+rt.path, 400, "bad_name") + check(schedOwner, "/api/v1/servers/nope/schedules"+rt.path, 404, "not_found") + check(schedStranger, schedBase+rt.path, 403, "forbidden") + r.a.Schedules = nil + check(schedOwner, schedBase+rt.path, 503, "schedules_unavailable") + if len(r.repo.audits) != 0 || r.con.calls != 0 { + t.Fatalf("a refused request left audits %+v / %d console calls", r.repo.audits, r.con.calls) + } + }) + } + t.Run("bad id", func(t *testing.T) { + r := newSchedRig(t) + for _, id := range []string{"0", "-1", "x"} { + res := r.do(schedOwner, "DELETE", schedBase+"/"+id, "") + if res.StatusCode != 400 || schedErrCode(t, res) != "bad_id" { + t.Fatalf("DELETE %s = %d, want 400 bad_id", id, res.StatusCode) + } + } + }) +} + +func TestScheduleCRUD(t *testing.T) { + t.Run("list: empty is [] with the limit", func(t *testing.T) { + r := newSchedRig(t) + res := r.do(schedOwner, "GET", schedBase, "") + raw, _ := io.ReadAll(res.Body) + if res.StatusCode != 200 || string(raw) != `{"limit":20,"schedules":[],"server":"survival"}`+"\n" { + t.Fatalf("GET = %d %s", res.StatusCode, raw) + } + }) + + t.Run("owner creates: next run, owner binding, audit", func(t *testing.T) { + r := newSchedRig(t) // 2026-09-28 03:00 UTC = 11:00 in Shanghai + res := r.do(schedOwner, "POST", schedBase, + `{"label":"夜间重启","action":"restart","minute_of_day":240,"weekdays":127,"timezone":"Asia/Shanghai","warn_minutes":5}`) + if res.StatusCode != 201 { + t.Fatalf("POST = %d", res.StatusCode) + } + got := decodeSchedule(t, res) + want := time.Date(2026, 9, 28, 20, 0, 0, 0, time.UTC) // 04:00 on the 29th in Shanghai + if got.ID != 1 || got.Label != "夜间重启" || !sameTime(got.NextRunAt, want) || !got.Enabled || + got.CreatedBy != "owner1@example.net" || got.LastRunAt != nil || got.RunState != "" { + t.Fatalf("created %+v", got) + } + if row := r.st.row(t, 1); row.OwnerID != "owner1" { + t.Fatalf("stored owner %q, want owner1", row.OwnerID) + } + if len(r.repo.audits) != 1 || r.repo.audits[0].Action != "schedule.create" || + r.repo.audits[0].Actor != "owner1@example.net" || r.repo.audits[0].ActorUserID != "owner1" || r.repo.audits[0].ServerName != "survival" || + string(r.repo.audits[0].Payload) != `{"action":"restart","schedule":1}` { + t.Fatalf("audits %+v", r.repo.audits) + } + }) + + t.Run("admin creates on an owned server: it belongs to the owner", func(t *testing.T) { + r := newSchedRig(t) + res := r.do(schedAdmin, "POST", schedBase, `{"action":"stop","minute_of_day":0,"weekdays":1,"timezone":"UTC","enabled":false}`) + if res.StatusCode != 201 { + t.Fatalf("POST = %d", res.StatusCode) + } + got := decodeSchedule(t, res) + if got.NextRunAt != nil || got.Enabled || got.CreatedBy != "adm@example.net" { + t.Fatalf("created %+v", got) + } + if row := r.st.row(t, got.ID); row.OwnerID != "owner1" { + t.Fatalf("stored owner %q, want owner1", row.OwnerID) + } + }) + + t.Run("bad settings and unknown fields are 400 and store nothing", func(t *testing.T) { + r := newSchedRig(t) + for body, code := range map[string]string{ + `{"action":"restart","minute_of_day":240,"weekdays":127,"timezone":"Nowhere/City"}`: "bad_schedule", + `{"action":"command","weekdays":127,"timezone":"UTC"}`: "bad_request", + `{"action":"restart","weekdays":127,"timezone":"UTC","owner_id":"me"}`: "bad_request", + } { + res := r.do(schedOwner, "POST", schedBase, body) + if res.StatusCode != 400 || schedErrCode(t, res) != code { + t.Fatalf("POST %s = %d, want 400 %s", body, res.StatusCode, code) + } + } + if len(r.st.rows) != 0 || len(r.repo.audits) != 0 { + t.Fatalf("stored %d rows, %d audits", len(r.st.rows), len(r.repo.audits)) + } + }) + + t.Run("the 21st is 409 schedule_limit", func(t *testing.T) { + r := newSchedRig(t) + for range maxSchedulesPerServer { + r.schedule(ScheduleStop, schedT0.Add(time.Hour)) + } + res := r.do(schedOwner, "POST", schedBase, `{"action":"stop","weekdays":127,"timezone":"UTC"}`) + if res.StatusCode != 409 || schedErrCode(t, res) != "schedule_limit" || len(r.st.rows) != maxSchedulesPerServer { + t.Fatalf("POST = %d, rows %d", res.StatusCode, len(r.st.rows)) + } + }) + + t.Run("update: new settings, current owner, fresh next run and warning marker", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleRestart, schedT0.Add(time.Hour), func(s *Schedule) { + s.OwnerID, s.WarnedFor = "previous", ptrTime(schedT0.Add(time.Hour)) + }) + res := r.do(schedOwner, "PUT", fmt.Sprintf("%s/%d", schedBase, id), + `{"action":"command","command":"/save-all","every_minutes":30,"weekdays":127,"timezone":"UTC"}`) + if res.StatusCode != 200 { + t.Fatalf("PUT = %d", res.StatusCode) + } + got := decodeSchedule(t, res) + row := r.st.row(t, id) + if got.Action != ScheduleCommand || got.Command != "save-all" || !sameTime(got.NextRunAt, schedT0.Add(30*time.Minute)) || + row.OwnerID != "owner1" || row.WarnedFor != nil || row.Command != "save-all" { + t.Fatalf("PUT answered %+v, stored %+v", got, row) + } + if len(r.repo.audits) != 1 || r.repo.audits[0].Action != "schedule.update" { + t.Fatalf("audits %+v", r.repo.audits) + } + }) + + t.Run("update and delete refuse a running or unknown schedule", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleRestart, schedT0.Add(time.Hour), func(s *Schedule) { s.RunState = runStopping }) + body := `{"action":"stop","weekdays":127,"timezone":"UTC"}` + for _, c := range []struct{ method, path, body, code string }{ + {"PUT", fmt.Sprintf("/%d", id), body, "schedule_running"}, + {"DELETE", fmt.Sprintf("/%d", id), "", "schedule_running"}, + {"PUT", "/99", body, "not_found"}, + {"DELETE", "/99", "", "not_found"}, + } { + res := r.do(schedOwner, c.method, schedBase+c.path, c.body) + want := 409 + if c.code == "not_found" { + want = 404 + } + if res.StatusCode != want || schedErrCode(t, res) != c.code { + t.Fatalf("%s %s = %d, want %d %s", c.method, c.path, res.StatusCode, want, c.code) + } + } + if row := r.st.row(t, id); row.Action != ScheduleRestart || len(r.repo.audits) != 0 { + t.Fatalf("refused writes changed %+v / audited %+v", row, r.repo.audits) + } + }) + + t.Run("a schedule of another server is 404", func(t *testing.T) { + r := newSchedRig(t) + r.repo.byName["creative"] = &ServerRecord{Name: "creative", OwnerID: "owner1"} + id := r.st.put(Schedule{Server: "creative", Action: ScheduleStop, Weekdays: 1, Timezone: "UTC"}) + res := r.do(schedOwner, "DELETE", fmt.Sprintf("%s/%d", schedBase, id), "") + if res.StatusCode != 404 || len(r.st.rows) != 1 { + t.Fatalf("DELETE = %d, rows %d", res.StatusCode, len(r.st.rows)) + } + }) + + t.Run("delete", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleStop, schedT0.Add(time.Hour)) + res := r.do(schedOwner, "DELETE", fmt.Sprintf("%s/%d", schedBase, id), "") + if res.StatusCode != 204 || len(r.st.rows) != 0 { + t.Fatalf("DELETE = %d, rows %d", res.StatusCode, len(r.st.rows)) + } + if len(r.repo.audits) != 1 || r.repo.audits[0].Action != "schedule.delete" || + string(r.repo.audits[0].Payload) != fmt.Sprintf(`{"action":"stop","schedule":%d}`, id) { + t.Fatalf("audits %+v", r.repo.audits) + } + }) + + t.Run("list shows the schedules oldest first", func(t *testing.T) { + r := newSchedRig(t) + a := r.schedule(ScheduleStop, schedT0.Add(2*time.Hour)) + b := r.schedule(ScheduleStart, schedT0.Add(time.Hour)) + res := r.do(schedOwner, "GET", schedBase, "") + var body struct { + Schedules []Schedule `json:"schedules"` + } + if err := json.NewDecoder(res.Body).Decode(&body); err != nil { + t.Fatal(err) + } + if len(body.Schedules) != 2 || body.Schedules[0].ID != a || body.Schedules[1].ID != b { + t.Fatalf("listed %+v", body.Schedules) + } + }) +} + +// TestScheduleRunNow: a run on request starts at once, without the warning, +// whether or not the schedule is enabled, and leaves its next run alone. +func TestScheduleRunNow(t *testing.T) { + t.Run("a command runs and the answer carries its outcome", func(t *testing.T) { + r := newSchedRig(t) + r.con.reply = "§aSaved the game" + next := schedT0.Add(5 * time.Hour) + id := r.schedule(ScheduleCommand, next, func(s *Schedule) { s.Command = "save-all" }) + res := r.do(schedOwner, "POST", fmt.Sprintf("%s/%d/run", schedBase, id), "") + if res.StatusCode != 202 { + t.Fatalf("run = %d", res.StatusCode) + } + got := decodeSchedule(t, res) + if got.LastResult != ScheduleOK || got.LastDetail != "Saved the game" || got.RunState != "" || + !sameTime(got.LastRunAt, schedT0) || !sameTime(got.NextRunAt, next) { + t.Fatalf("after the run %+v", got) + } + if r.con.gotCommand != "save-all" || r.con.calls != 1 { + t.Fatalf("console ran %q (%d calls)", r.con.gotCommand, r.con.calls) + } + var actions []string + for _, e := range r.repo.audits { + actions = append(actions, e.Action+"/"+e.Actor) + } + if strings.Join(actions, ",") != "schedule.run_now/owner1@example.net,schedule.run/scheduler" { + t.Fatalf("audits %v", actions) + } + }) + + t.Run("a disabled restart starts and goes on in the background", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleRestart, schedT0, func(s *Schedule) { s.Enabled, s.NextRunAt, s.WarnMinutes = false, nil, 5 }) + res := r.do(schedOwner, "POST", fmt.Sprintf("%s/%d/run", schedBase, id), "") + got := decodeSchedule(t, res) + if res.StatusCode != 202 || got.RunState != runStopping || got.NextRunAt != nil { + t.Fatalf("run = %d %+v", res.StatusCode, got) + } + if r.cl.desired["survival"] != v1alpha1.DesiredStopped || r.con.calls != 0 { + t.Fatalf("desired %q, %d console calls (no warning on request)", r.cl.desired["survival"], r.con.calls) + } + }) + + t.Run("a running schedule is 409 schedule_running", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleRestart, schedT0, func(s *Schedule) { s.RunState = runStarting }) + res := r.do(schedOwner, "POST", fmt.Sprintf("%s/%d/run", schedBase, id), "") + if res.StatusCode != 409 || schedErrCode(t, res) != "schedule_running" || len(r.repo.audits) != 0 { + t.Fatalf("run = %d, audits %+v", res.StatusCode, r.repo.audits) + } + }) + + t.Run("a schedule of the previous owner is 409 schedule_stale", func(t *testing.T) { + r := newSchedRig(t) + id := r.schedule(ScheduleCommand, schedT0, func(s *Schedule) { s.OwnerID = "previous" }) + res := r.do(schedAdmin, "POST", fmt.Sprintf("%s/%d/run", schedBase, id), "") + if res.StatusCode != 409 || schedErrCode(t, res) != "schedule_stale" || r.con.calls != 0 || r.st.claims != 0 { + t.Fatalf("run = %d, console %d, claims %d", res.StatusCode, r.con.calls, r.st.claims) + } + }) + + t.Run("unknown schedule is 404", func(t *testing.T) { + r := newSchedRig(t) + res := r.do(schedOwner, "POST", schedBase+"/7/run", "") + if res.StatusCode != 404 { + t.Fatalf("run = %d", res.StatusCode) + } + }) +} diff --git a/internal/pgint/schedules_test.go b/internal/pgint/schedules_test.go new file mode 100644 index 0000000..9971f29 --- /dev/null +++ b/internal/pgint/schedules_test.go @@ -0,0 +1,393 @@ +//go:build pgint + +package pgint + +import ( + "context" + "errors" + "fmt" + "strings" + "testing" + "time" + + "felis.lolicon.best/internal/api" +) + +func newSchedule(server, owner, action string, next *time.Time) *api.Schedule { + return &api.Schedule{Server: server, OwnerID: owner, Label: "每晚", Action: action, MinuteOfDay: 240, + Weekdays: 0x7f, Timezone: "Asia/Shanghai", WarnMinutes: 5, Enabled: next != nil, NextRunAt: next, + CreatedBy: "pgint"} +} + +func mustCreateSchedule(t *testing.T, s *api.Schedule) *api.Schedule { + t.Helper() + if err := repo.CreateSchedule(context.Background(), s, 20); err != nil { + t.Fatalf("CreateSchedule(%s on %s): %v", s.Action, s.Server, err) + } + return s +} + +func mustGetSchedule(t *testing.T, server string, id int64) *api.Schedule { + t.Helper() + s, err := repo.GetSchedule(context.Background(), server, id) + if err != nil { + t.Fatalf("GetSchedule(%d): %v", id, err) + } + return s +} + +// runView is the run columns of a schedule, for exact comparisons. +func runView(s *api.Schedule) string { + ts := func(p *time.Time) string { + if p == nil { + return "-" + } + return p.UTC().Format(time.RFC3339) + } + return fmt.Sprintf("state=%q resume=%v step=%s last=%s result=%q detail=%q next=%s warned=%s enabled=%v", + s.RunState, s.RunResume, ts(s.RunStepAt), ts(s.LastRunAt), s.LastResult, s.LastDetail, + ts(s.NextRunAt), ts(s.WarnedFor), s.Enabled) +} + +func applied(t *testing.T, what string, ok bool, err error, want bool) { + t.Helper() + if err != nil || ok != want { + t.Fatalf("%s = %v, %v; want applied=%v", what, ok, err, want) + } +} + +// TestScheduleStoreCRUD pins the settings side of server_schedules: a round +// trip of every column, the per-server limit, the run-state guard on changes, +// and the lookups scoped to their server. +func TestScheduleStoreCRUD(t *testing.T) { + ctx := context.Background() + u := newUser(t, "user", "sch-crud") + sfx := suffix(t) + name, other, gone := "sc-"+sfx, "sco-"+sfx, "scg-"+sfx + seedOwnedServer(t, name, u.ID, false) + seedOwnedServer(t, other, u.ID, false) + seedOwnedServer(t, gone, u.ID, true) + next := mustNow().Add(time.Hour).Truncate(time.Second) + + s := mustCreateSchedule(t, newSchedule(name, u.ID, api.ScheduleRestart, &next)) + if s.ID == 0 || time.Since(s.CreatedAt) > time.Minute { + t.Fatalf("created %+v", s) + } + settings := func(s *api.Schedule) string { + return fmt.Sprintf("%d %s owner=%q label=%q %s cmd=%q every=%d at=%d days=%d tz=%s warn=%d by=%s created=%s", + s.ID, s.Server, s.OwnerID, s.Label, s.Action, s.Command, s.EveryMinutes, s.MinuteOfDay, s.Weekdays, + s.Timezone, s.WarnMinutes, s.CreatedBy, s.CreatedAt.UTC().Format(time.RFC3339Nano)) + } + got := mustGetSchedule(t, name, s.ID) + if settings(got) != settings(s) { + t.Fatalf("read back %s\nwant %s", settings(got), settings(s)) + } + if v, want := runView(got), `state="" resume=false step=- last=- result="" detail="" next=`+next.UTC().Format(time.RFC3339)+` warned=- enabled=true`; v != want { + t.Fatalf("read back %s\nwant %s", v, want) + } + if _, err := repo.GetSchedule(ctx, other, s.ID); !errors.Is(err, api.ErrNotFound) { + t.Fatalf("GetSchedule on another server = %v, want ErrNotFound", err) + } + + unowned := mustCreateSchedule(t, newSchedule(other, "", api.ScheduleStop, nil)) + if g := mustGetSchedule(t, other, unowned.ID); g.OwnerID != "" || g.Enabled || g.NextRunAt != nil { + t.Fatalf("unowned, disabled schedule read back %+v", g) + } + if err := repo.CreateSchedule(ctx, newSchedule(gone, u.ID, api.ScheduleStop, nil), 20); !errors.Is(err, api.ErrNotFound) { + t.Fatalf("CreateSchedule on a deleted server = %v, want ErrNotFound", err) + } + if err := repo.CreateSchedule(ctx, newSchedule("nope-"+sfx, u.ID, api.ScheduleStop, nil), 20); !errors.Is(err, api.ErrNotFound) { + t.Fatalf("CreateSchedule on an unknown server = %v, want ErrNotFound", err) + } + + // The limit counts the server's own rows only. + mustCreateSchedule(t, newSchedule(name, u.ID, api.ScheduleStop, nil)) + if err := repo.CreateSchedule(ctx, newSchedule(name, u.ID, api.ScheduleStart, nil), 2); !errors.Is(err, api.ErrScheduleLimit) { + t.Fatalf("third schedule under a limit of 2 = %v, want ErrScheduleLimit", err) + } + if err := repo.CreateSchedule(ctx, newSchedule(other, u.ID, api.ScheduleStart, nil), 2); err != nil { + t.Fatalf("second schedule of the other server under a limit of 2 = %v", err) + } + + // Update writes the settings and owner and clears the warning marker. + if ok, err := repo.WarnScheduleRun(ctx, s.ID, next); err != nil || !ok { + t.Fatalf("WarnScheduleRun = %v, %v", ok, err) + } + upd := mustGetSchedule(t, name, s.ID) + upd.OwnerID, upd.Label, upd.Action, upd.Command = "", "", api.ScheduleCommand, "say hi" + upd.EveryMinutes, upd.MinuteOfDay, upd.Weekdays, upd.Timezone, upd.WarnMinutes = 30, 0, 0x41, "UTC", 0 + upd.Enabled, upd.NextRunAt = false, nil + if err := repo.UpdateSchedule(ctx, upd); err != nil { + t.Fatalf("UpdateSchedule: %v", err) + } + g := mustGetSchedule(t, name, s.ID) + if g.OwnerID != "" || g.Label != "" || g.Action != api.ScheduleCommand || g.Command != "say hi" || g.EveryMinutes != 30 || + g.MinuteOfDay != 0 || g.Weekdays != 0x41 || g.Timezone != "UTC" || g.WarnMinutes != 0 || g.Enabled || + g.NextRunAt != nil || g.WarnedFor != nil { + t.Fatalf("after update %+v", g) + } + wrong := *g + wrong.Server = other + if err := repo.UpdateSchedule(ctx, &wrong); !errors.Is(err, api.ErrNotFound) { + t.Fatalf("UpdateSchedule under another server = %v, want ErrNotFound", err) + } + + // A running schedule refuses changes and deletion; an idle one is deleted. + now := mustNow().Truncate(time.Second) + ok, err := mustClaim(t, s.ID, nil, nil, now) + applied(t, "claim", ok, err, true) + if err := repo.UpdateSchedule(ctx, g); !errors.Is(err, api.ErrScheduleRunning) { + t.Fatalf("UpdateSchedule while running = %v, want ErrScheduleRunning", err) + } + if err := repo.DeleteSchedule(ctx, name, s.ID); !errors.Is(err, api.ErrScheduleRunning) { + t.Fatalf("DeleteSchedule while running = %v, want ErrScheduleRunning", err) + } + ok, err = repo.FinishScheduleRun(ctx, s.ID, "claimed", api.ScheduleOK, "") + applied(t, "finish", ok, err, true) + if err := repo.DeleteSchedule(ctx, other, s.ID); !errors.Is(err, api.ErrNotFound) { + t.Fatalf("DeleteSchedule under another server = %v, want ErrNotFound", err) + } + if err := repo.DeleteSchedule(ctx, name, s.ID); err != nil { + t.Fatalf("DeleteSchedule: %v", err) + } + if err := repo.DeleteSchedule(ctx, name, s.ID); !errors.Is(err, api.ErrNotFound) { + t.Fatalf("second DeleteSchedule = %v, want ErrNotFound", err) + } + list, err := repo.ListSchedules(ctx, other) + if err != nil || len(list) != 2 || list[0].ID != unowned.ID || list[0].ID >= list[1].ID { + t.Fatalf("ListSchedules(other) = %+v, %v; want 2, oldest first", list, err) + } +} + +func mustClaim(t *testing.T, id int64, due, next *time.Time, now time.Time) (bool, error) { + t.Helper() + return repo.ClaimScheduleRun(context.Background(), id, due, next, now) +} + +// TestScheduleStoreRunCAS pins the compare-and-set writes of a run, which +// keep two felis-api processes from firing one run twice. +func TestScheduleStoreRunCAS(t *testing.T) { + ctx := context.Background() + u := newUser(t, "user", "sch-cas") + name := "scc-" + suffix(t) + seedOwnedServer(t, name, u.ID, false) + t0 := mustNow().Truncate(time.Second) + due, next, later := t0.Add(-time.Minute), t0.Add(23*time.Hour), t0.Add(47*time.Hour) + s := mustCreateSchedule(t, newSchedule(name, u.ID, api.ScheduleBackup, &due)) + ts := func(t time.Time) string { return t.UTC().Format(time.RFC3339) } + + // Warned once for the run due then. + ok, err := repo.WarnScheduleRun(ctx, s.ID, next) + applied(t, "warn for another due time", ok, err, false) + ok, err = repo.WarnScheduleRun(ctx, s.ID, due) + applied(t, "warn", ok, err, true) + ok, err = repo.WarnScheduleRun(ctx, s.ID, due) + applied(t, "second warn", ok, err, false) + + // A claim for another due time loses; the right one wins once. + ok, err = mustClaim(t, s.ID, &next, &later, t0) + applied(t, "claim a stale due time", ok, err, false) + ok, err = mustClaim(t, s.ID, &due, &next, t0) + applied(t, "claim", ok, err, true) + ok, err = mustClaim(t, s.ID, &due, &next, t0) + applied(t, "second claim", ok, err, false) + ok, err = mustClaim(t, s.ID, nil, nil, t0) + applied(t, "claim on request during a run", ok, err, false) + if got, want := runView(mustGetSchedule(t, name, s.ID)), + fmt.Sprintf(`state="claimed" resume=false step=%s last=%s result="" detail="" next=%s warned=- enabled=true`, ts(t0), ts(t0), ts(next)); got != want { + t.Fatalf("after claim %s\nwant %s", got, want) + } + + // Advance from the wrong step loses; the right one keeps the outcome so far + // unless it brings one. + t1 := t0.Add(time.Minute) + ok, err = repo.AdvanceScheduleRun(ctx, s.ID, "stopping", "backing_up", true, "", "", t1) + applied(t, "advance from the wrong step", ok, err, false) + ok, err = repo.AdvanceScheduleRun(ctx, s.ID, "claimed", "stopping", true, "", "", t1) + applied(t, "advance", ok, err, true) + t2 := t1.Add(time.Minute) + ok, err = repo.AdvanceScheduleRun(ctx, s.ID, "stopping", "starting", true, api.ScheduleFailed, "the backup failed: x", t2) + applied(t, "advance with an outcome", ok, err, true) + t3 := t2.Add(time.Minute) + ok, err = repo.AdvanceScheduleRun(ctx, s.ID, "starting", "starting", true, "", "", t3) + applied(t, "advance without one", ok, err, true) + if got, want := runView(mustGetSchedule(t, name, s.ID)), + fmt.Sprintf(`state="starting" resume=true step=%s last=%s result="failed" detail="the backup failed: x" next=%s warned=- enabled=true`, ts(t3), ts(t0), ts(next)); got != want { + t.Fatalf("after advancing %s\nwant %s", got, want) + } + if _, err := db.ExecContext(ctx, `UPDATE server_schedules SET run_step_at = NULL WHERE id = $1`, s.ID); err == nil || + !strings.Contains(err.Error(), "check") { + t.Fatalf("a run step without its time = %v, want a check violation", err) + } + + // Finish from the wrong step, or from none, loses. + ok, err = repo.FinishScheduleRun(ctx, s.ID, "stopping", api.ScheduleOK, "") + applied(t, "finish from the wrong step", ok, err, false) + ok, err = repo.FinishScheduleRun(ctx, s.ID, "starting", api.ScheduleFailed, "done") + applied(t, "finish", ok, err, true) + ok, err = repo.FinishScheduleRun(ctx, s.ID, "", api.ScheduleOK, "") + applied(t, "finish an idle schedule", ok, err, false) + if got, want := runView(mustGetSchedule(t, name, s.ID)), + fmt.Sprintf(`state="" resume=false step=- last=%s result="failed" detail="done" next=%s warned=- enabled=true`, ts(t0), ts(next)); got != want { + t.Fatalf("after finishing %s\nwant %s", got, want) + } + + // A run on request leaves the next run alone and works while disabled. + ok, err = mustClaim(t, s.ID, nil, nil, t3) + applied(t, "claim on request", ok, err, true) + if g := mustGetSchedule(t, name, s.ID); g.NextRunAt == nil || !g.NextRunAt.Equal(next) || g.RunState != "claimed" || g.LastResult != "" { + t.Fatalf("after a claim on request %s", runView(g)) + } + ok, err = repo.DisableSchedule(ctx, s.ID, "x") + applied(t, "disable during a run", ok, err, false) + ok, err = repo.WarnScheduleRun(ctx, s.ID, next) + applied(t, "warn during a run", ok, err, false) + ok, err = mustClaim(t, s.ID, &next, &later, t3) + applied(t, "claim the due run during a run on request", ok, err, false) + ok, err = repo.MissScheduleRun(ctx, s.ID, next, later, "x") + applied(t, "miss during a run", ok, err, false) + ok, err = repo.FinishScheduleRun(ctx, s.ID, "claimed", api.ScheduleOK, "") + applied(t, "finish the run on request", ok, err, true) + + // Missed: only the run due then, recorded at its due time. + ok, err = repo.MissScheduleRun(ctx, s.ID, due, later, "x") + applied(t, "miss a stale due time", ok, err, false) + ok, err = repo.WarnScheduleRun(ctx, s.ID, next) + applied(t, "warn the next run", ok, err, true) + ok, err = repo.MissScheduleRun(ctx, s.ID, next, later, "felis-api was down") + applied(t, "miss", ok, err, true) + if got, want := runView(mustGetSchedule(t, name, s.ID)), + fmt.Sprintf(`state="" resume=false step=- last=%s result="missed" detail="felis-api was down" next=%s warned=- enabled=true`, ts(next), ts(later)); got != want { + t.Fatalf("after a miss %s\nwant %s", got, want) + } + + // Disabled: once, and a disabled schedule is neither claimed nor missed. + ok, err = repo.DisableSchedule(ctx, s.ID, "new owner") + applied(t, "disable", ok, err, true) + ok, err = repo.DisableSchedule(ctx, s.ID, "again") + applied(t, "second disable", ok, err, false) + if got, want := runView(mustGetSchedule(t, name, s.ID)), + fmt.Sprintf(`state="" resume=false step=- last=%s result="skipped" detail="new owner" next=- warned=- enabled=false`, ts(next)); got != want { + t.Fatalf("after disabling %s\nwant %s", got, want) + } + if _, err := db.ExecContext(ctx, `UPDATE server_schedules SET next_run_at = $2 WHERE id = $1`, s.ID, later); err != nil { + t.Fatal(err) + } + ok, err = mustClaim(t, s.ID, &later, &later, t3) + applied(t, "claim a disabled schedule's due run", ok, err, false) + ok, err = repo.MissScheduleRun(ctx, s.ID, later, later, "x") + applied(t, "miss a disabled schedule's run", ok, err, false) + ok, err = repo.WarnScheduleRun(ctx, s.ID, later) + applied(t, "warn a disabled schedule's run", ok, err, false) +} + +// TestDueSchedules pins what the runner reads: enabled schedules of live +// servers due by the horizon, every run in progress, runs first, with the +// server's owner now. +func TestDueSchedules(t *testing.T) { + ctx := context.Background() + u := newUser(t, "user", "sch-due") + v := newUser(t, "user", "sch-due2") + sfx := suffix(t) + live, gone, unowned := "sdl-"+sfx, "sdg-"+sfx, "sdu-"+sfx + seedOwnedServer(t, live, v.ID, false) + seedOwnedServer(t, gone, u.ID, true) + mustExec(t, `INSERT INTO servers (name, cached_cpu_milli, cached_memory_mb, cached_storage_mb) VALUES ($1, 100, 128, 1)`, unowned) + t0 := mustNow().Truncate(time.Second) + at := func(m int) *time.Time { p := t0.Add(time.Duration(m) * time.Minute); return &p } + horizon := *at(30) + + late := mustCreateSchedule(t, newSchedule(live, u.ID, api.ScheduleStop, at(20))) + early := mustCreateSchedule(t, newSchedule(live, u.ID, api.ScheduleStart, at(-5))) + edge := mustCreateSchedule(t, newSchedule(unowned, "", api.ScheduleStop, at(30))) + mustCreateSchedule(t, newSchedule(live, u.ID, api.ScheduleStop, at(31))) // beyond the horizon + off := mustCreateSchedule(t, newSchedule(live, u.ID, api.ScheduleStop, nil)) // disabled, yet due + mustExec(t, `UPDATE server_schedules SET next_run_at = $2 WHERE id = $1`, off.ID, *at(-2)) + running := mustCreateSchedule(t, newSchedule(live, u.ID, api.ScheduleRestart, nil)) // disabled, but running + ok, err := mustClaim(t, running.ID, nil, nil, t0) + applied(t, "claim", ok, err, true) + // Deleted server: its schedules wait for SeedServer or the row's end. + mustExec(t, `INSERT INTO server_schedules (server_name, owner_id, action, timezone, next_run_at, created_by) + VALUES ($1, $2, 'stop', 'UTC', $3, 'pgint')`, gone, u.ID, *at(-1)) + + due, err := repo.DueSchedules(ctx, horizon) + if err != nil { + t.Fatal(err) + } + var got []string + for _, d := range due { + if d.Server != live && d.Server != unowned && d.Server != gone { + continue // other tests' rows + } + got = append(got, fmt.Sprintf("%d/%s/%v", d.ID, d.RunState, d.ServerOwner == v.ID)) + } + want := []string{ + fmt.Sprintf("%d/claimed/true", running.ID), + fmt.Sprintf("%d//true", early.ID), + fmt.Sprintf("%d//true", late.ID), + fmt.Sprintf("%d//false", edge.ID), + } + if strings.Join(got, " ") != strings.Join(want, " ") { + t.Fatalf("due %v, want %v", got, want) + } + for _, d := range due { + if d.ID == late.ID && d.OwnerID != u.ID { + t.Fatalf("schedule owner %q, want %q", d.OwnerID, u.ID) + } + if d.ID == edge.ID && d.ServerOwner != "" { + t.Fatalf("unowned server's owner %q", d.ServerOwner) + } + } +} + +// TestSchedulesFollowTheServer: a recreated server of the same name starts +// without the earlier one's schedules, and an account migration hands the +// migrated servers' schedules to the target. +func TestSchedulesFollowTheServer(t *testing.T) { + ctx := context.Background() + u := newUser(t, "user", "sch-seed") + name, sub := "ss-"+suffix(t), "sss-"+suffix(t) + if err := repo.SeedServer(ctx, name, sub, 100, 128, 1024); err != nil { + t.Fatal(err) + } + s := mustCreateSchedule(t, newSchedule(name, u.ID, api.ScheduleStop, nil)) + if err := repo.SeedServer(ctx, name, sub, 100, 128, 1024); err != nil { + t.Fatal(err) + } + if _, err := repo.GetSchedule(ctx, name, s.ID); !errors.Is(err, api.ErrNotFound) { + t.Fatalf("schedule of the earlier server = %v, want ErrNotFound", err) + } + + src := newUser(t, "user", "sch-src") + dst := newUser(t, "user", "sch-dst") + other := newUser(t, "user", "sch-other") + sfx := suffix(t) + mine, theirs := "sm-"+sfx, "st-"+sfx + seedOwnedServer(t, mine, src.ID, false) + seedOwnedServer(t, theirs, other.ID, false) + moved := mustCreateSchedule(t, newSchedule(mine, src.ID, api.ScheduleStop, nil)) + stale := mustCreateSchedule(t, newSchedule(mine, other.ID, api.ScheduleStop, nil)) // saved by a previous owner + kept := mustCreateSchedule(t, newSchedule(theirs, src.ID, api.ScheduleStop, nil)) // src's, on a server src lost + t0 := mustNow().Truncate(time.Second) + if err := repo.StartMigration(ctx, "schmig-"+sfx, src.ID, t0); err != nil { + t.Fatal(err) + } + if err := repo.ConfirmMigration(ctx, src.ID, "passkey", "sess", t0); err != nil { + t.Fatal(err) + } + if err := repo.IssueMigrationCode(ctx, src.ID, dst.ID, "sess", "h-sch-"+sfx, t0, t0.Add(10*time.Minute)); err != nil { + t.Fatal(err) + } + if _, _, err := repo.RedeemMigration(ctx, dst.ID, "h-sch-"+sfx, t0); err != nil { + t.Fatalf("RedeemMigration: %v", err) + } + for _, c := range []struct { + server string + id int64 + owner string + }{{mine, moved.ID, dst.ID}, {mine, stale.ID, other.ID}, {theirs, kept.ID, src.ID}} { + if g := mustGetSchedule(t, c.server, c.id); g.OwnerID != c.owner { + t.Fatalf("schedule %d owner %q, want %q", c.id, g.OwnerID, c.owner) + } + } +} diff --git a/internal/store/migrations/0036_server_schedules.sql b/internal/store/migrations/0036_server_schedules.sql new file mode 100644 index 0000000..574314d --- /dev/null +++ b/internal/store/migrations/0036_server_schedules.sql @@ -0,0 +1,46 @@ +-- Scheduled tasks (GET/POST /servers/{name}/schedules): an owner or admin asks +-- felis-api to run a console command, restart, stop, start or back up a server +-- at set times. felis-api's schedule runner reads this table every few seconds. +-- +-- A schedule fires at minute_of_day (local to timezone) on the weekdays in the +-- bitmask (bit 0 Sunday), or, with every_minutes set, at every multiple of it +-- since local midnight on those days. owner_id is the server's owner when the +-- schedule was saved (NULL for an unowned server): the runner skips and disables +-- a schedule once the server has another owner, so a new owner never inherits +-- somebody else's commands. A recreated server of the same name starts without +-- schedules (SeedServer deletes them). +-- +-- run_state is the step of a run still in progress (a restart waits for the +-- server to stop before starting it, a backup of a running server stops it, +-- backs it up and starts it again); run_step_at is when that step began, and +-- run_resume says the run starts the server again once its backup is done. +CREATE TABLE server_schedules ( + id bigserial PRIMARY KEY, + server_name text NOT NULL REFERENCES servers(name) ON DELETE CASCADE, + owner_id text REFERENCES users(id), + label text NOT NULL DEFAULT '', + action text NOT NULL CHECK (action IN ('command', 'restart', 'stop', 'start', 'backup')), + command text NOT NULL DEFAULT '', + every_minutes integer NOT NULL DEFAULT 0, + minute_of_day integer NOT NULL DEFAULT 0 CHECK (minute_of_day BETWEEN 0 AND 1439), + weekdays smallint NOT NULL DEFAULT 127 CHECK (weekdays BETWEEN 1 AND 127), + timezone text NOT NULL, + warn_minutes integer NOT NULL DEFAULT 0, + enabled boolean NOT NULL DEFAULT true, + next_run_at timestamptz, + warned_for timestamptz, + run_state text NOT NULL DEFAULT '', + run_resume boolean NOT NULL DEFAULT false, + run_step_at timestamptz, + last_run_at timestamptz, + last_result text NOT NULL DEFAULT '', + last_detail text NOT NULL DEFAULT '', + created_by text NOT NULL, + created_at timestamptz NOT NULL DEFAULT now(), + updated_at timestamptz NOT NULL DEFAULT now(), + -- Every write that starts a step stamps it; the runner times each step from it. + CHECK (run_state = '' OR run_step_at IS NOT NULL) +); +CREATE INDEX server_schedules_server ON server_schedules (server_name, id); +CREATE INDEX server_schedules_due ON server_schedules (next_run_at) WHERE enabled; +CREATE INDEX server_schedules_running ON server_schedules (id) WHERE run_state <> ''; diff --git a/panel/src/App.tsx b/panel/src/App.tsx index 6a9cf03..19d1f6e 100644 --- a/panel/src/App.tsx +++ b/panel/src/App.tsx @@ -34,6 +34,9 @@ const ServerBackups = lazyWithReload(() => const ServerFiles = lazyWithReload(() => import("@/pages/ServerFiles").then((m) => ({ default: m.ServerFiles })), ); +const ServerSchedules = lazyWithReload(() => + import("@/pages/ServerSchedules").then((m) => ({ default: m.ServerSchedules })), +); const ServerLuckPerms = lazyWithReload(() => import("@/pages/ServerLuckPerms").then((m) => ({ default: m.ServerLuckPerms })), ); @@ -107,6 +110,7 @@ export default function App() { } /> } /> } /> + } /> } /> } /> diff --git a/panel/src/i18n/index.ts b/panel/src/i18n/index.ts index a50bf7a..c9d16a2 100644 --- a/panel/src/i18n/index.ts +++ b/panel/src/i18n/index.ts @@ -13,6 +13,7 @@ import enNavigation from "./resources/en-US/navigation.json"; import enBackups from "./resources/en-US/backups.json"; import enSubmissions from "./resources/en-US/submissions.json"; import enFiles from "./resources/en-US/files.json"; +import enSchedules from "./resources/en-US/schedules.json"; import zhCommon from "./resources/zh-CN/common.json"; import zhAuth from "./resources/zh-CN/auth.json"; import zhDashboard from "./resources/zh-CN/dashboard.json"; @@ -25,6 +26,7 @@ import zhNavigation from "./resources/zh-CN/navigation.json"; import zhBackups from "./resources/zh-CN/backups.json"; import zhSubmissions from "./resources/zh-CN/submissions.json"; import zhFiles from "./resources/zh-CN/files.json"; +import zhSchedules from "./resources/zh-CN/schedules.json"; import { panelLanguage, SUPPORTED_LANGUAGES } from "./language"; // Exported so a test can boot a fresh instance with the same detection. @@ -43,6 +45,7 @@ export const i18nOptions: InitOptions = { backups: enBackups, submissions: enSubmissions, files: enFiles, + schedules: enSchedules, }, "zh-CN": { common: zhCommon, @@ -57,6 +60,7 @@ export const i18nOptions: InitOptions = { backups: zhBackups, submissions: zhSubmissions, files: zhFiles, + schedules: zhSchedules, }, }, supportedLngs: [...SUPPORTED_LANGUAGES], diff --git a/panel/src/i18n/resources/en-US/errors.json b/panel/src/i18n/resources/en-US/errors.json index 7d61020..6a4d049 100644 --- a/panel/src/i18n/resources/en-US/errors.json +++ b/panel/src/i18n/resources/en-US/errors.json @@ -118,5 +118,10 @@ "bad_request": "The request was not accepted: {{detail}}", "restore_in_progress": "Another backup is still being restored to this server's world. Try again once it finishes.", "uploads_full": "The upload store is full. An admin has to delete reviewed submissions before new uploads fit.", - "unsupported_media_type": "The request was sent in a format the server does not accept. Reload the page and try again." + "unsupported_media_type": "The request was sent in a format the server does not accept. Reload the page and try again.", + "schedules_unavailable": "Scheduled tasks aren't available right now.", + "schedule_limit": "This server already has as many scheduled tasks as it can hold. Delete one first.", + "schedule_running": "This task is running right now. Try again once the run finishes.", + "schedule_stale": "The server has a new owner since this task was saved. Save the task again before running it.", + "bad_schedule": "The task was not saved: {{detail}}" } diff --git a/panel/src/i18n/resources/en-US/schedules.json b/panel/src/i18n/resources/en-US/schedules.json new file mode 100644 index 0000000..f8c9915 --- /dev/null +++ b/panel/src/i18n/resources/en-US/schedules.json @@ -0,0 +1,159 @@ +{ + "title": "Scheduled tasks", + "subtitle": "Restart, stop, start or back up this server, or send its console a command, at set times.", + "back_to_console": "Back to console", + "not_yours_title": "No permission to manage scheduled tasks", + "not_yours_body": "Only the owner or an admin can view and change this server's scheduled tasks.", + "new_task": "New task", + "count": "{{count}} of {{limit}} tasks", + "limit_reached": "A server holds at most {{limit}} tasks. Delete one to add another.", + "unavailable_title": "Scheduled tasks aren't available", + "unavailable_body": "This felis-api runs without a schedule store, so tasks can't be listed or saved here. Ask an administrator to check the felis-api setup.", + "empty_title": "No scheduled tasks yet", + "empty_hint": "Start from a common one, or build your own with New task.", + "template_restart": "Restart every night at 04:00", + "template_backup": "Back up every night at 05:00", + "template_announce": "Announce in chat every 30 minutes", + "template_announce_command": "say Welcome to the server!", + "action_command": "Console command", + "action_command_desc": "Send one command to the server console, as if typed there. Runs only while the server is running.", + "action_restart": "Restart", + "action_restart_desc": "Stop the server and start it again. Online players are disconnected. Skipped while it is stopped.", + "action_stop": "Stop", + "action_stop_desc": "Stop the server. Online players are disconnected.", + "action_start": "Start", + "action_start_desc": "Start the server when it is stopped.", + "action_backup": "Backup", + "action_backup_desc": "Back up the world. A running server is stopped first and started again afterwards.", + "days_all": "Every day", + "days_weekdays": "Weekdays", + "days_weekend": "Weekends", + "day_sep": ", ", + "day_0": "Sun", + "day_1": "Mon", + "day_2": "Tue", + "day_3": "Wed", + "day_4": "Thu", + "day_5": "Fri", + "day_6": "Sat", + "day_chip_0": "Sun", + "day_chip_1": "Mon", + "day_chip_2": "Tue", + "day_chip_3": "Wed", + "day_chip_4": "Thu", + "day_chip_5": "Fri", + "day_chip_6": "Sat", + "day_full_0": "Sunday", + "day_full_1": "Monday", + "day_full_2": "Tuesday", + "day_full_3": "Wednesday", + "day_full_4": "Thursday", + "day_full_5": "Friday", + "day_full_6": "Saturday", + "first_day": "0", + "when_daily": "{{days}} at {{time}}", + "when_interval_days": "{{every}} · {{days}}", + "every_minutes_one": "Every minute", + "every_minutes_other": "Every {{count}} minutes", + "every_hours_one": "Every hour", + "every_hours_other": "Every {{count}} hours", + "opt_minutes_one": "{{count}} minute", + "opt_minutes_other": "{{count}} minutes", + "opt_hours_one": "{{count}} hour", + "opt_hours_other": "{{count}} hours", + "create_title": "New scheduled task", + "edit_title": "Edit scheduled task", + "dialog_desc": "Choose what runs and when.", + "create": "Create", + "save": "Save", + "field_action": "What to do", + "field_command": "Command", + "command_placeholder": "say The server restarts at 04:00", + "command_hint": "One console command. The leading / is optional.", + "field_when": "When", + "mode_daily": "Once a day", + "mode_interval": "Repeat", + "field_time": "Time", + "field_every": "Repeat every", + "interval_preview": "Counted from 00:00 each day: {{times}}", + "interval_hourly_limit": "A restart, stop, start or backup repeats at most once an hour.", + "field_days": "Days", + "field_timezone": "Time zone", + "timezone_hint": "Times are read in this zone. Your browser is in {{zone}}.", + "timezone_use_browser": "Use the browser's zone", + "field_warn": "Warn players", + "warn_none": "No warning", + "warn_option_one": "{{count}} minute before", + "warn_option_other": "{{count}} minutes before", + "warn_hint": "Posts a message in game chat ahead of the stop, in Chinese and English, while the server is running.", + "backup_note": "A backup of a running server stops it (online players are disconnected), backs up the world and starts it again, which usually takes a few minutes. A stopped server is backed up and stays stopped.", + "field_label": "Name (optional)", + "label_placeholder": "Nightly restart", + "field_enabled": "Enabled", + "enabled_hint": "Runs on schedule. Switch it off to keep the task without running it.", + "err_command_required": "Enter a command.", + "err_command_long": "The command is longer than {{max}} bytes.", + "err_time_required": "Pick a time.", + "err_days": "Pick at least one day.", + "err_timezone_required": "Enter a time zone.", + "err_timezone": "Unknown time zone. Use a name such as Asia/Shanghai.", + "err_label_long": "The name is longer than {{max}} characters.", + "next_run": "Next run {{when}}", + "next_run_off": "Off: it runs only when you press Run now", + "never_run": "Hasn't run yet", + "last_run": "Last run {{when}}", + "run_started_at": "Started {{when}}", + "result_ok": "Succeeded", + "result_skipped": "Skipped", + "result_failed": "Failed", + "result_missed": "Missed", + "reply": "Reply: {{text}}", + "run_state_claimed": "Running…", + "run_state_stopping": "Stopping the server…", + "run_state_backing_up": "Backing up…", + "run_state_starting": "Starting the server…", + "warn_badge": "Warns {{count}} min before", + "off_badge": "Off", + "tz_differs": "Times are in {{zone}}; your browser is in {{browser}}.", + "toggle_label": "Run {{name}} on schedule", + "run_now": "Run now", + "edit": "Edit", + "delete": "Delete", + "busy_hint": "This task is running. Wait for the run to finish.", + "owner_changed": "The server changed owner, so this task switched itself off. Check it, then save it or switch it on to use it again.", + "detail_not_running": "The server was not running.", + "detail_already_stopped": "The server was already stopped.", + "detail_already_running": "The server was already running.", + "detail_gone": "The server no longer exists.", + "detail_no_world": "The server has no world yet.", + "detail_missed": "felis-api was not running at the scheduled time.", + "detail_started_meanwhile": "Someone started the server before the run finished.", + "detail_interrupted": "felis-api stopped in the middle of this run.", + "detail_store_full": "The backup store is full. Ask an administrator to free space.", + "detail_console": "The server console could not be reached.", + "detail_cap": "The cluster is at its running-server cap.", + "detail_retiring": "The server is being given up or deleted.", + "run_title": "Run “{{name}}” now?", + "run_confirm": "Run now", + "run_body_command": "Sends this command to the server console now (skipped while the server is stopped):", + "run_body_restart": "Stops the server now and starts it again.", + "run_body_stop": "Stops the server now.", + "run_body_start": "Starts the server now if it is stopped.", + "run_body_backup": "Backs up the world now. A running server is stopped, backed up and started again; a stopped server stays stopped.", + "run_kick": "Online players are disconnected right away: a run started by hand skips the players' warning.", + "run_keeps_next": "The next scheduled run stays as planned.", + "delete_title": "Delete “{{name}}”?", + "delete_body": "The task is removed and won't run again.", + "delete_confirm": "Delete", + "created": "Created “{{name}}”.", + "saved": "Saved “{{name}}”.", + "deleted": "Deleted “{{name}}”.", + "run_started": "Started “{{name}}”.", + "toggled_on": "“{{name}}” is on.", + "toggled_off": "“{{name}}” is off.", + "notes_title": "How scheduled tasks run", + "note_timing": "felis-api checks every 15 seconds, so a task starts within about 15 seconds of its time. A run more than 10 minutes late (felis-api was down) is recorded as missed, and the next one runs as usual.", + "note_state": "A command, restart or stop runs only while the server is running, and a start only while it is stopped; otherwise the run is skipped.", + "note_owner": "When the server changes owner, its tasks switch off. Saving a task hands it to the current owner.", + "note_run_now": "Run now also works on a task that is off. It skips the players' warning and keeps the next scheduled run." +} diff --git a/panel/src/i18n/resources/en-US/servers.json b/panel/src/i18n/resources/en-US/servers.json index 313c144..8d79989 100644 --- a/panel/src/i18n/resources/en-US/servers.json +++ b/panel/src/i18n/resources/en-US/servers.json @@ -61,6 +61,8 @@ "backups_link_desc": "View and restore world backups for this server.", "files_link_title": "Server files", "files_link_desc": "Browse and repair files in the world volume.", + "schedules_link_title": "Scheduled tasks", + "schedules_link_desc": "Restart, back up or run commands at set times.", "system_service": "System service", "system_service_hint": "Provisioned and managed by the platform (felis systemservers); it accepts no user operations.", "players_back_to_console": "Back to console", diff --git a/panel/src/i18n/resources/zh-CN/errors.json b/panel/src/i18n/resources/zh-CN/errors.json index c521824..e8c2081 100644 --- a/panel/src/i18n/resources/zh-CN/errors.json +++ b/panel/src/i18n/resources/zh-CN/errors.json @@ -118,5 +118,10 @@ "bad_request": "请求未被接受:{{detail}}", "restore_in_progress": "这台服务器的世界正在恢复另一份备份,请等它完成后再试。", "uploads_full": "上传存储已满,需要管理员删除已审核的提交后才能继续上传。", - "unsupported_media_type": "请求格式不被服务器接受,请刷新页面后重试。" + "unsupported_media_type": "请求格式不被服务器接受,请刷新页面后重试。", + "schedules_unavailable": "计划任务当前不可用。", + "schedule_limit": "这台服务器的计划任务数量已达上限,请先删除一个。", + "schedule_running": "这个任务正在执行,请等本次执行结束后再试。", + "schedule_stale": "保存这个任务后服务器换了所有者,请先重新保存任务再执行。", + "bad_schedule": "计划任务未保存:{{detail}}" } diff --git a/panel/src/i18n/resources/zh-CN/schedules.json b/panel/src/i18n/resources/zh-CN/schedules.json new file mode 100644 index 0000000..82eef98 --- /dev/null +++ b/panel/src/i18n/resources/zh-CN/schedules.json @@ -0,0 +1,154 @@ +{ + "title": "计划任务", + "subtitle": "按设定的时间重启、关闭、启动或备份这台服务器,或向它的控制台发送命令。", + "back_to_console": "返回控制台", + "not_yours_title": "无权管理计划任务", + "not_yours_body": "只有所有者或管理员才能查看和修改该服务器的计划任务。", + "new_task": "新建任务", + "count": "{{count}} / {{limit}} 个任务", + "limit_reached": "每台服务器最多 {{limit}} 个计划任务,删除一个后才能再添加。", + "unavailable_title": "计划任务不可用", + "unavailable_body": "当前 felis-api 未配置计划任务存储,这里无法查看或保存任务。请联系管理员检查 felis-api 的配置。", + "empty_title": "还没有计划任务", + "empty_hint": "可以从常用任务开始,也可以点“新建任务”自己设置。", + "template_restart": "每天 04:00 重启", + "template_backup": "每天 05:00 备份", + "template_announce": "每 30 分钟在聊天栏发公告", + "template_announce_command": "say 欢迎来到服务器!", + "action_command": "控制台命令", + "action_command_desc": "向服务器控制台发送一条命令,效果和在控制台里输入一样。仅在服务器运行时执行。", + "action_restart": "重启", + "action_restart_desc": "关闭服务器后再启动,在线玩家会被断开。服务器已关闭时跳过。", + "action_stop": "关闭", + "action_stop_desc": "关闭服务器,在线玩家会被断开。", + "action_start": "启动", + "action_start_desc": "服务器关闭时启动它。", + "action_backup": "备份", + "action_backup_desc": "备份世界。运行中的服务器会先关闭,备份完成后再启动。", + "days_all": "每天", + "days_weekdays": "工作日", + "days_weekend": "周末", + "day_sep": "、", + "day_0": "周日", + "day_1": "周一", + "day_2": "周二", + "day_3": "周三", + "day_4": "周四", + "day_5": "周五", + "day_6": "周六", + "day_chip_0": "日", + "day_chip_1": "一", + "day_chip_2": "二", + "day_chip_3": "三", + "day_chip_4": "四", + "day_chip_5": "五", + "day_chip_6": "六", + "day_full_0": "星期日", + "day_full_1": "星期一", + "day_full_2": "星期二", + "day_full_3": "星期三", + "day_full_4": "星期四", + "day_full_5": "星期五", + "day_full_6": "星期六", + "first_day": "1", + "when_daily": "{{days}} {{time}}", + "when_interval_days": "{{every}} · {{days}}", + "every_minutes": "每 {{count}} 分钟", + "every_hours": "每 {{count}} 小时", + "opt_minutes": "{{count}} 分钟", + "opt_hours": "{{count}} 小时", + "create_title": "新建计划任务", + "edit_title": "编辑计划任务", + "dialog_desc": "选择执行什么、什么时候执行。", + "create": "创建", + "save": "保存", + "field_action": "执行什么", + "field_command": "命令", + "command_placeholder": "say 服务器将在 04:00 重启", + "command_hint": "一条控制台命令,开头的 / 可加可不加。", + "field_when": "什么时候", + "mode_daily": "每天一次", + "mode_interval": "循环执行", + "field_time": "时间", + "field_every": "间隔", + "interval_preview": "每天从 00:00 起算:{{times}}", + "interval_hourly_limit": "重启、关闭、启动和备份最多每小时执行一次。", + "field_days": "哪几天", + "field_timezone": "时区", + "timezone_hint": "按这个时区计算时间。你的浏览器时区是 {{zone}}。", + "timezone_use_browser": "改用浏览器时区", + "field_warn": "提醒玩家", + "warn_none": "不提醒", + "warn_option": "提前 {{count}} 分钟", + "warn_hint": "服务器运行时,在关闭前于游戏聊天栏发一条中英双语提醒。", + "backup_note": "备份运行中的服务器时会先关闭它(在线玩家会被断开),备份世界后再启动,通常需要几分钟。已关闭的服务器备份后保持关闭。", + "field_label": "名称(可选)", + "label_placeholder": "每晚重启", + "field_enabled": "启用", + "enabled_hint": "按计划自动执行。关闭后任务保留,但不再自动执行。", + "err_command_required": "请输入命令。", + "err_command_long": "命令超过 {{max}} 字节。", + "err_time_required": "请选择时间。", + "err_days": "至少选择一天。", + "err_timezone_required": "请输入时区。", + "err_timezone": "无法识别的时区,请填写类似 Asia/Shanghai 的名称。", + "err_label_long": "名称超过 {{max}} 个字符。", + "next_run": "下次执行:{{when}}", + "next_run_off": "已停用,只在点“立即执行”时运行", + "never_run": "尚未执行", + "last_run": "上次执行:{{when}}", + "run_started_at": "开始于:{{when}}", + "result_ok": "成功", + "result_skipped": "已跳过", + "result_failed": "失败", + "result_missed": "错过", + "reply": "回复:{{text}}", + "run_state_claimed": "执行中…", + "run_state_stopping": "正在关闭服务器…", + "run_state_backing_up": "正在备份…", + "run_state_starting": "正在启动服务器…", + "warn_badge": "提前 {{count}} 分钟提醒", + "off_badge": "已停用", + "tz_differs": "按 {{zone}} 时区计算,你的浏览器时区是 {{browser}}。", + "toggle_label": "按计划执行「{{name}}」", + "run_now": "立即执行", + "edit": "编辑", + "delete": "删除", + "busy_hint": "任务正在执行,请等本次执行结束。", + "owner_changed": "服务器更换了所有者,这个任务已自动停用。检查无误后保存或重新启用即可恢复。", + "detail_not_running": "服务器未在运行。", + "detail_already_stopped": "服务器已经是关闭状态。", + "detail_already_running": "服务器已经在运行。", + "detail_gone": "服务器已不存在。", + "detail_no_world": "服务器还没有世界存档。", + "detail_missed": "计划时间点 felis-api 未在运行。", + "detail_started_meanwhile": "执行结束前有人启动了服务器。", + "detail_interrupted": "felis-api 在执行过程中停止了。", + "detail_store_full": "备份存储已满,请联系管理员清理空间。", + "detail_console": "无法连接服务器控制台。", + "detail_cap": "集群运行中的服务器数量已达上限。", + "detail_retiring": "服务器正在被放弃或删除。", + "run_title": "立即执行「{{name}}」?", + "run_confirm": "立即执行", + "run_body_command": "立即向服务器控制台发送下面这条命令,服务器关闭时跳过:", + "run_body_restart": "立即关闭服务器,然后重新启动。", + "run_body_stop": "立即关闭服务器。", + "run_body_start": "服务器关闭时立即启动它。", + "run_body_backup": "立即备份世界。运行中的服务器会先关闭,备份后再启动;已关闭的服务器备份后保持关闭。", + "run_kick": "在线玩家会被立即断开:手动执行不发送提醒。", + "run_keeps_next": "下次计划执行时间保持不变。", + "delete_title": "删除「{{name}}」?", + "delete_body": "任务会被删除,之后不再执行。", + "delete_confirm": "删除", + "created": "已创建「{{name}}」。", + "saved": "已保存「{{name}}」。", + "deleted": "已删除「{{name}}」。", + "run_started": "已开始执行「{{name}}」。", + "toggled_on": "已启用「{{name}}」。", + "toggled_off": "已停用「{{name}}」。", + "notes_title": "计划任务如何执行", + "note_timing": "felis-api 每 15 秒检查一次,任务会在设定时间后约 15 秒内开始。晚于计划 10 分钟以上的执行(例如 felis-api 当时未运行)会记为错过,下一次照常执行。", + "note_state": "命令、重启和关闭仅在服务器运行时执行,启动仅在服务器关闭时执行,其余情况会跳过。", + "note_owner": "服务器更换所有者后,它的任务会自动停用;重新保存即可交给当前所有者。", + "note_run_now": "“立即执行”对已停用的任务同样有效,不发送提醒,也不影响下次计划执行。" +} diff --git a/panel/src/i18n/resources/zh-CN/servers.json b/panel/src/i18n/resources/zh-CN/servers.json index 970fff3..6027da4 100644 --- a/panel/src/i18n/resources/zh-CN/servers.json +++ b/panel/src/i18n/resources/zh-CN/servers.json @@ -61,6 +61,8 @@ "backups_link_desc": "查看并恢复该服务器的世界备份。", "files_link_title": "服务器文件", "files_link_desc": "浏览并修复世界卷中的文件。", + "schedules_link_title": "计划任务", + "schedules_link_desc": "定时重启、备份或执行命令。", "system_service": "系统服务", "system_service_hint": "由平台预置并管理(felis systemservers),不接受用户操作。", "players_back_to_console": "返回控制台", diff --git a/panel/src/lib/api.test.ts b/panel/src/lib/api.test.ts index ba720c2..350820c 100644 --- a/panel/src/lib/api.test.ts +++ b/panel/src/lib/api.test.ts @@ -1440,6 +1440,95 @@ describe("server file manager wire shapes", () => { }); }); +describe("scheduled task wire shapes", () => { + beforeEach(() => vi.restoreAllMocks()); + afterEach(() => vi.unstubAllGlobals()); + + function sent(fetchSpy: typeof fetch): [string, RequestInit] { + const [url, opts] = (fetchSpy as unknown as ReturnType).mock.calls[0]; + return [String(url), opts as RequestInit]; + } + + const input = { + label: "Nightly", + action: "restart" as const, + command: "", + every_minutes: 0 as const, + minute_of_day: 240, + weekdays: 127, + timezone: "Asia/Shanghai", + warn_minutes: 5 as const, + enabled: true, + }; + const saved = { ...input, id: 7, server: "survival", next_run_at: "2026-09-28T04:00:00+08:00", run_state: "" }; + + it("listSchedules GETs /servers/{name}/schedules and keeps the limit", async () => { + const fetchSpy = fakeFetch({ server: "survival", schedules: [saved], limit: 20 }); + vi.stubGlobal("fetch", fetchSpy); + expect(await api.listSchedules("survival")).toEqual({ schedules: [saved], limit: 20 }); + const [url, opts] = sent(fetchSpy); + expect(url).toBe("/servers/survival/schedules"); + expect(opts.method).toBe("GET"); + }); + + it("listSchedules reads a missing list as none", async () => { + vi.stubGlobal("fetch", fakeFetch({ server: "survival", limit: 20 })); + expect(await api.listSchedules("survival")).toEqual({ schedules: [], limit: 20 }); + }); + + it("createSchedule POSTs the input as it is", async () => { + const fetchSpy = fakeFetch(saved, { status: 201 }); + vi.stubGlobal("fetch", fetchSpy); + expect(await api.createSchedule("survival", input)).toEqual(saved); + const [url, opts] = sent(fetchSpy); + expect(url).toBe("/servers/survival/schedules"); + expect(opts.method).toBe("POST"); + expect(JSON.parse(opts.body as string)).toEqual(input); + }); + + it("updateSchedule PUTs the whole input to the schedule's id", async () => { + const fetchSpy = fakeFetch(saved); + vi.stubGlobal("fetch", fetchSpy); + expect(await api.updateSchedule("survival", 7, { ...input, enabled: false })).toEqual(saved); + const [url, opts] = sent(fetchSpy); + expect(url).toBe("/servers/survival/schedules/7"); + expect(opts.method).toBe("PUT"); + expect(JSON.parse(opts.body as string)).toEqual({ ...input, enabled: false }); + }); + + it("deleteSchedule DELETEs the schedule with no body and takes the 204", async () => { + const fetchSpy = vi.fn(async () => ({ + ok: true, + status: 204, + statusText: "No Content", + text: async () => "", + })) as unknown as typeof fetch; + vi.stubGlobal("fetch", fetchSpy); + expect(await api.deleteSchedule("survival", 7)).toBeNull(); + const [url, opts] = sent(fetchSpy); + expect(url).toBe("/servers/survival/schedules/7"); + expect(opts.method).toBe("DELETE"); + expect(opts.body).toBeUndefined(); + }); + + it("runSchedule POSTs to .../run with no body and parses the 202", async () => { + const fetchSpy = fakeFetch({ ...saved, run_state: "stopping" }, { status: 202 }); + vi.stubGlobal("fetch", fetchSpy); + expect((await api.runSchedule("survival", 7)).run_state).toBe("stopping"); + const [url, opts] = sent(fetchSpy); + expect(url).toBe("/servers/survival/schedules/7/run"); + expect(opts.method).toBe("POST"); + expect(opts.body).toBeUndefined(); + }); + + it("refuses a server name that is not one path segment before sending anything", async () => { + const fetchSpy = fakeFetch({}); + vi.stubGlobal("fetch", fetchSpy); + await expect(api.runSchedule("..", 7)).rejects.toMatchObject({ code: "bad_path_param" }); + expect(fetchSpy).not.toHaveBeenCalled(); + }); +}); + // The API's generic codes carry an English developer message ("user not found", // "invalid request"); the panel words them itself so a Chinese UI never shows it. describe("copy for the generic server codes", () => { @@ -1470,6 +1559,25 @@ describe("copy for the generic server codes", () => { expect(humanizeError({ status: 411, code: "length_required", message: "raw" })).toMatch(/did not say how large it is/); }); + it("words the scheduled tasks' refusals itself, and keeps the reason a schedule was refused", () => { + expect(humanizeError({ status: 503, code: "schedules_unavailable", message: "raw" })).toBe( + "Scheduled tasks aren't available right now.", + ); + expect(humanizeError({ status: 409, code: "schedule_limit", message: "raw" })).toBe( + "This server already has as many scheduled tasks as it can hold. Delete one first.", + ); + expect(humanizeError({ status: 409, code: "schedule_running", message: "raw" })).toBe( + "This task is running right now. Try again once the run finishes.", + ); + expect(humanizeError({ status: 409, code: "schedule_stale", message: "raw" })).toBe( + "The server has a new owner since this task was saved. Save the task again before running it.", + ); + expect( + humanizeError({ status: 400, code: "bad_schedule", message: "a restart can repeat at most every 60 minutes" }), + ).toBe("The task was not saved: a restart can repeat at most every 60 minutes"); + expect(humanizeError({ status: 400, code: "bad_schedule", message: "" })).toBe("Something went wrong."); + }); + it("reads a full upload store as full, not as an outage", () => { expect(humanizeError({ status: 507, code: "uploads_full", message: "" })).toMatch(/upload store is full/); }); diff --git a/panel/src/lib/api.ts b/panel/src/lib/api.ts index 33abf52..6b4c2f6 100644 --- a/panel/src/lib/api.ts +++ b/panel/src/lib/api.ts @@ -22,6 +22,8 @@ import type { QuotaInput, QuotaView, RetireState, + Schedule, + ScheduleInput, ServerFileEntry, ServerJob, MyServerView, @@ -681,6 +683,41 @@ export const api = rejectingSync({ urlPath`/servers/${name}/jobs`, ).then((r) => r.jobs ?? []), + // Scheduled tasks of one server (GET/POST /servers/{name}/schedules, PUT/DELETE + // .../{id}, POST .../{id}/run). Owner-or-admin gated server-side; 503 + // schedules_unavailable while the schedule store is not wired. A server holds + // at most `limit` of them (409 schedule_limit past it). + listSchedules: (name: string) => + request<{ server: string; schedules: Schedule[]; limit: number }>( + "GET", + urlPath`/servers/${name}/schedules`, + ).then((r) => ({ schedules: r.schedules ?? [], limit: r.limit })), + + // createSchedule saves a new schedule; it belongs to the server's current + // owner. Validation failures are 400 bad_request / bad_schedule with the reason + // in the message. + createSchedule: (name: string, input: ScheduleInput) => + request("POST", urlPath`/servers/${name}/schedules`, input), + + // updateSchedule replaces a schedule's settings (the whole input, not a patch) + // and passes it to the server's current owner, which is how a schedule the + // runner disabled after an owner change comes back. 409 schedule_running while + // it runs. + updateSchedule: (name: string, id: number, input: ScheduleInput) => + request("PUT", urlPath`/servers/${name}/schedules/${String(id)}`, input), + + // deleteSchedule answers 204; 409 schedule_running while it runs. + deleteSchedule: (name: string, id: number) => + request("DELETE", urlPath`/servers/${name}/schedules/${String(id)}`), + + // runSchedule runs a schedule now, without the players' warning; its next + // scheduled run stays where it is. The reply is 202 with the schedule after the + // run's first step: a restart or backup goes on in the background and shows in + // run_state. 409 schedule_running mid-run, 409 schedule_stale when the server + // changed owner since the schedule was saved. + runSchedule: (name: string, id: number) => + request("POST", urlPath`/servers/${name}/schedules/${String(id)}/run`), + // Server file manager (spec §7). Every route is owner-or-admin gated and // refuses with 409 not_stopped unless the server is fully stopped (the world // volume is RWO), so callers gate on phase === "Stopped". The path travels as a @@ -1344,10 +1381,21 @@ export function humanizeError(e: unknown): string { return t("uploads_full"); case "unsupported_media_type": return t("unsupported_media_type"); + // Scheduled tasks (internal/api/handlers_schedules.go). + case "schedules_unavailable": + return t("schedules_unavailable"); + case "schedule_limit": + return t("schedule_limit"); + case "schedule_running": + return t("schedule_running"); + case "schedule_stale": + return t("schedule_stale"); // The detail says which field was wrong; the server writes it in English, // so it rides inside a localized sentence. case "bad_request": return err.message ? t("bad_request", { detail: err.message }) : t("generic"); + case "bad_schedule": + return err.message ? t("bad_schedule", { detail: err.message }) : t("generic"); default: if (err.status === 401) return t("session_expired"); if (err.status === 403) return t("forbidden"); diff --git a/panel/src/lib/openapi.gen.ts b/panel/src/lib/openapi.gen.ts index a36cfd9..0397e49 100644 --- a/panel/src/lib/openapi.gen.ts +++ b/panel/src/lib/openapi.gen.ts @@ -1382,6 +1382,71 @@ export interface paths { patch?: never; trace?: never; }; + "/api/v1/servers/{name}/schedules": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + /** List a server's scheduled tasks (owner-or-admin). */ + get: operations["listServerSchedules"]; + put?: never; + /** + * Add a scheduled task to a server (owner-or-admin). + * @description The task belongs to the server's current owner: once the server has another owner felis-api disables it instead of running it, until somebody saves it again. felis-api checks the tasks every 15 seconds; a run it was down for is started late, up to 10 minutes, and dropped as missed after that. A command runs only on a running server, a restart only restarts a running one, and a start goes through the running-server cap and a pending retirement like a wake. Audited as schedule.create; each run as schedule.run by scheduler. + */ + post: operations["createServerSchedule"]; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/api/v1/servers/{name}/schedules/{id}": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + get?: never; + /** + * Change a scheduled task (owner-or-admin). + * @description Replaces the task's settings and recomputes its next run. The task passes to the server's current owner. Refused while a run is in progress. Audited as schedule.update. + */ + put: operations["updateServerSchedule"]; + post?: never; + /** + * Remove a scheduled task (owner-or-admin). + * @description Refused while a run is in progress. Audited as schedule.delete. + */ + delete: operations["deleteServerSchedule"]; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; + "/api/v1/servers/{name}/schedules/{id}/run": { + parameters: { + query?: never; + header?: never; + path?: never; + cookie?: never; + }; + get?: never; + put?: never; + /** + * Run a scheduled task now (owner-or-admin). + * @description Starts a run at once, without the players' warning, whether the task is enabled or not; its next scheduled run stays where it was. The answer is the task after the run's first step: a command, stop or start has finished, and a restart or backup goes on in the background (run_state). Audited as schedule.run_now, and the run itself as schedule.run. + */ + post: operations["runServerSchedule"]; + delete?: never; + options?: never; + head?: never; + patch?: never; + trace?: never; + }; "/api/v1/users": { parameters: { query?: never; @@ -2639,6 +2704,76 @@ export interface components { /** @description Owned rows only. Present while the server is given up or being deleted. */ retiring?: components["schemas"]["RetireState"]; }; + /** @description One scheduled task of a server (internal/api/schedules.go Schedule). It runs at minute_of_day on the weekdays, or with every_minutes set at every multiple of it since midnight on those days, in timezone. */ + Schedule: { + /** Format: int64 */ + id: number; + server: string; + /** @description Free text naming the task; may be empty. */ + label: string; + /** + * @description backup of a running server stops it, takes a backup (pruned with the daily restore points, [archive] scheduled_keep) and starts it again; of a stopped server it leaves the server stopped. + * @enum {string} + */ + action: "command" | "restart" | "stop" | "start" | "backup"; + /** @description The console command of a command task */ + command: string; + /** + * @description 0 runs once a day at minute_of_day. A restart, stop, start or backup repeats at most every 60 minutes. + * @enum {integer} + */ + every_minutes: 0 | 15 | 30 | 60 | 120 | 180 | 240 | 360 | 480 | 720; + /** @description Minutes after local midnight; 0 when every_minutes is set. */ + minute_of_day: number; + /** @description Bitmask of the days it runs on: bit 0 Sunday to bit 6 Saturday. */ + weekdays: number; + /** @description IANA zone the times are in */ + timezone: string; + /** + * @description How long before a restart, stop or backup the players on the server are told (say); 0 for none, and always 0 for a command or start. + * @enum {integer} + */ + warn_minutes: 0 | 1 | 5 | 10 | 15 | 30; + enabled: boolean; + /** + * Format: date-time + * @description Null while disabled. + */ + next_run_at: string | null; + /** + * @description What a run in progress is doing; empty when none is. + * @enum {string} + */ + run_state: "" | "claimed" | "stopping" | "backing_up" | "starting"; + /** Format: date-time */ + last_run_at: string | null; + /** + * @description How the last run ended; empty before the first and during a run. missed is a run felis-api was down for, dropped once it was 10 minutes late. + * @enum {string} + */ + last_result: "" | "ok" | "skipped" | "failed" | "missed"; + /** @description What happened */ + last_detail: string; + created_by: string; + /** Format: date-time */ + created_at: string; + }; + ScheduleInput: { + label?: string; + /** @enum {string} */ + action: "command" | "restart" | "stop" | "start" | "backup"; + /** @description Required for a command task and refused for the others. One line; a leading slash is dropped. */ + command?: string; + /** @enum {integer} */ + every_minutes?: 0 | 15 | 30 | 60 | 120 | 180 | 240 | 360 | 480 | 720; + minute_of_day?: number; + weekdays: number; + timezone: string; + /** @enum {integer} */ + warn_minutes?: 0 | 1 | 5 | 10 | 15 | 30; + /** @description Default true. */ + enabled?: boolean; + }; /** @description One world backup (internal/api/repo.go BackupView). backup_ref is withheld (spec §286). */ BackupView: { id: string; @@ -6640,6 +6775,206 @@ export interface operations { }; }; }; + listServerSchedules: { + parameters: { + query?: never; + header?: never; + path: { + name: string; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description The server's tasks, oldest first, and how many it may have. */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": { + server: string; + schedules: components["schemas"]["Schedule"][]; + limit: number; + }; + }; + }; + 400: components["responses"]["BadRequest"]; + 401: components["responses"]["Unauthorized"]; + 403: components["responses"]["Forbidden"]; + 404: components["responses"]["NotFound"]; + 503: components["responses"]["ServiceUnavailable"]; + }; + }; + createServerSchedule: { + parameters: { + query?: never; + header?: never; + path: { + name: string; + }; + cookie?: never; + }; + requestBody: { + content: { + "application/json": components["schemas"]["ScheduleInput"]; + }; + }; + responses: { + /** @description Created. */ + 201: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["Schedule"]; + }; + }; + /** @description A malformed body or name, or settings out of range (bad_schedule, bad_request for the command). */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["Error"]; + }; + }; + 401: components["responses"]["Unauthorized"]; + 403: components["responses"]["Forbidden"]; + 404: components["responses"]["NotFound"]; + /** @description The server already has 20 tasks (schedule_limit). */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["Error"]; + }; + }; + 503: components["responses"]["ServiceUnavailable"]; + }; + }; + updateServerSchedule: { + parameters: { + query?: never; + header?: never; + path: { + name: string; + id: number; + }; + cookie?: never; + }; + requestBody: { + content: { + "application/json": components["schemas"]["ScheduleInput"]; + }; + }; + responses: { + /** @description Saved. */ + 200: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["Schedule"]; + }; + }; + /** @description A malformed body, name or id, or settings out of range (bad_schedule). */ + 400: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["Error"]; + }; + }; + 401: components["responses"]["Unauthorized"]; + 403: components["responses"]["Forbidden"]; + 404: components["responses"]["NotFound"]; + /** @description A run is in progress (schedule_running). */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["Error"]; + }; + }; + 503: components["responses"]["ServiceUnavailable"]; + }; + }; + deleteServerSchedule: { + parameters: { + query?: never; + header?: never; + path: { + name: string; + id: number; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Removed. */ + 204: { + headers: { + [name: string]: unknown; + }; + content?: never; + }; + 400: components["responses"]["BadRequest"]; + 401: components["responses"]["Unauthorized"]; + 403: components["responses"]["Forbidden"]; + 404: components["responses"]["NotFound"]; + /** @description A run is in progress (schedule_running). */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["Error"]; + }; + }; + 503: components["responses"]["ServiceUnavailable"]; + }; + }; + runServerSchedule: { + parameters: { + query?: never; + header?: never; + path: { + name: string; + id: number; + }; + cookie?: never; + }; + requestBody?: never; + responses: { + /** @description Started. */ + 202: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["Schedule"]; + }; + }; + 400: components["responses"]["BadRequest"]; + 401: components["responses"]["Unauthorized"]; + 403: components["responses"]["Forbidden"]; + 404: components["responses"]["NotFound"]; + /** @description A run is already in progress (schedule_running), or the server has another owner since the task was saved (schedule_stale). */ + 409: { + headers: { + [name: string]: unknown; + }; + content: { + "application/json": components["schemas"]["Error"]; + }; + }; + 503: components["responses"]["ServiceUnavailable"]; + }; + }; listUsers: { parameters: { query?: { diff --git a/panel/src/lib/types.parity.ts b/panel/src/lib/types.parity.ts index 1955b99..ac21517 100644 --- a/panel/src/lib/types.parity.ts +++ b/panel/src/lib/types.parity.ts @@ -36,6 +36,8 @@ export type WireParity = [ Holds>, Holds>, Holds>, + Holds>, + Holds>, Holds>, Holds>, Holds>, diff --git a/panel/src/lib/types.ts b/panel/src/lib/types.ts index 55b419e..4593605 100644 --- a/panel/src/lib/types.ts +++ b/panel/src/lib/types.ts @@ -247,6 +247,62 @@ export interface ServerJob { scheduled?: boolean; } +// ---- Scheduled tasks (internal/api/schedules.go Schedule, scheduleInput) ---- + +export type ScheduleAction = "command" | "restart" | "stop" | "start" | "backup"; +/** 0 runs once a day at minute_of_day; the rest divide a day, so the runs sit at + * the same clock times every day. A restart, stop, start or backup repeats at + * most every 60 minutes. */ +export type ScheduleEveryMinutes = 0 | 15 | 30 | 60 | 120 | 180 | 240 | 360 | 480 | 720; +/** How long before a restart, stop or backup the players are told in game. */ +export type ScheduleWarnMinutes = 0 | 1 | 5 | 10 | 15 | 30; +/** The step of a run in progress; empty when none is. */ +export type ScheduleRunState = "" | "claimed" | "stopping" | "backing_up" | "starting"; +/** How the last run ended; empty before the first and during a run. */ +export type ScheduleResult = "" | "ok" | "skipped" | "failed" | "missed"; + +/** Schedule is one scheduled task of a server (GET /servers/{name}/schedules). + * It runs at minute_of_day on the weekdays (bit 0 Sunday … bit 6 Saturday), or + * with every_minutes set at every multiple of it since midnight on those days, + * in timezone. last_detail is the backend's English. */ +export interface Schedule { + id: number; + server: string; + label: string; + action: ScheduleAction; + /** The console command of a command task, without a slash; empty otherwise. */ + command: string; + every_minutes: ScheduleEveryMinutes; + minute_of_day: number; + weekdays: number; + timezone: string; + warn_minutes: ScheduleWarnMinutes; + enabled: boolean; + /** Null while disabled. */ + next_run_at: string | null; + run_state: ScheduleRunState; + last_run_at: string | null; + last_result: ScheduleResult; + last_detail: string; + created_by: string; + created_at: string; +} + +/** ScheduleInput is the body of a create (POST) or a save (PUT). The server trims + * the label, drops a command's leading slash, and refuses a command on any other + * action and a warning on a command or start. enabled defaults to true. */ +export interface ScheduleInput { + label?: string; + action: ScheduleAction; + command?: string; + every_minutes?: ScheduleEveryMinutes; + minute_of_day?: number; + weekdays: number; + timezone: string; + warn_minutes?: ScheduleWarnMinutes; + enabled?: boolean; +} + /** WhitelistImage is one row of GET /images (the create-form dropdown source). */ export interface WhitelistImage { image_ref: string; diff --git a/panel/src/pages/ServerConsole.test.tsx b/panel/src/pages/ServerConsole.test.tsx index 41bb4ee..8b000d3 100644 --- a/panel/src/pages/ServerConsole.test.tsx +++ b/panel/src/pages/ServerConsole.test.tsx @@ -98,6 +98,16 @@ describe("ServerConsole edit dialog", () => { }); }); +describe("ServerConsole doorways", () => { + it("links to the server's scheduled tasks while it is stopped", async () => { + calls.status.mockResolvedValue(status({ phase: "Stopped", desiredState: "Stopped" })); + renderConsole(); + + const link = await screen.findByRole("link", { name: /^Scheduled tasks/ }); + expect(link.getAttribute("href")).toBe("/servers/survival/schedules"); + }); +}); + describe("ServerConsole following the server", () => { afterEach(() => { vi.useRealTimers(); diff --git a/panel/src/pages/ServerConsole.tsx b/panel/src/pages/ServerConsole.tsx index 124f506..c3c1833 100644 --- a/panel/src/pages/ServerConsole.tsx +++ b/panel/src/pages/ServerConsole.tsx @@ -1,6 +1,6 @@ import { useState, useRef, useCallback, useId, useLayoutEffect, type KeyboardEvent } from "react"; import { Link, useParams } from "react-router-dom"; -import { Terminal, Moon, Shield, ShieldAlert, HelpCircle, Loader2, Users, Archive, FolderOpen, ChevronRight, type LucideIcon } from "lucide-react"; +import { Terminal, Moon, Shield, ShieldAlert, HelpCircle, Loader2, Users, Archive, FolderOpen, CalendarClock, ChevronRight, type LucideIcon } from "lucide-react"; import { useTranslation } from "react-i18next"; import { Card, CardContent } from "@/components/ui/card"; import { BackLink } from "@/components/BackLink"; @@ -411,6 +411,21 @@ export function ServerConsole() { + {/* Scheduled tasks — its own subpage. A schedule runs whatever phase the + server is in (a start wakes it, a backup stops it first), so the + doorway shows at every phase and the page owns its gating. */} + + +
+

{t("schedules_link_title")}

+

{t("schedules_link_desc")}

+
+ + + {/* Giving the server up (owner) or deleting it (admin) closes the sidebar. A system server is never retired, and one already on its way out shows the notice above with the way back instead. */} diff --git a/panel/src/pages/ServerSchedules.test.tsx b/panel/src/pages/ServerSchedules.test.tsx new file mode 100644 index 0000000..003037c --- /dev/null +++ b/panel/src/pages/ServerSchedules.test.tsx @@ -0,0 +1,544 @@ +// @vitest-environment jsdom +import { describe, it, expect, vi, beforeAll, beforeEach, afterEach } from "vitest"; +import { act, fireEvent, render, screen, within } from "@testing-library/react"; +import userEvent from "@testing-library/user-event"; +import { MemoryRouter, Route, Routes } from "react-router-dom"; +import i18next from "i18next"; +import { ServerSchedules } from "./ServerSchedules"; +import { STATUS_POLL_FAST_MS, STATUS_POLL_SLOW_MS } from "@/lib/hooks"; +import type { Schedule } from "@/lib/types"; + +const calls = vi.hoisted(() => ({ + status: vi.fn(), + myServers: vi.fn(), + listSchedules: vi.fn(), + createSchedule: vi.fn(), + updateSchedule: vi.fn(), + deleteSchedule: vi.fn(), + runSchedule: vi.fn(), +})); +const tier = vi.hoisted(() => ({ + loading: false, + identity: { user_id: "admin-1", email: "admin@example.test", role: "admin" }, + isAdmin: true, + isOwner: false, +})); +vi.mock("@/lib/tier", () => ({ useTier: () => tier })); +vi.mock("@/lib/config", () => ({ loadConfig: () => Promise.resolve({}) })); +vi.mock("@/lib/api", async (importOriginal) => { + const actual = await importOriginal(); + return { ...actual, api: { ...actual.api, ...calls } }; +}); + +// Radix Select opens with pointer capture and scrolls the picked item into view, +// neither of which jsdom implements. +beforeAll(() => { + Element.prototype.hasPointerCapture ??= () => false; + Element.prototype.releasePointerCapture ??= () => {}; + Element.prototype.scrollIntoView ??= () => {}; +}); + +const BROWSER = Intl.DateTimeFormat().resolvedOptions().timeZone; +const OTHER = BROWSER === "Pacific/Chatham" ? "Pacific/Kiritimati" : "Pacific/Chatham"; +const KICK = "Online players are disconnected right away: a run started by hand skips the players' warning."; + +function hoursFromNow(h: number): string { + return new Date(Date.now() + h * 3600_000 + 60_000).toISOString(); +} + +function schedule(over: Partial = {}): Schedule { + return { + id: 1, + server: "survival", + label: "", + action: "restart", + command: "", + every_minutes: 0, + minute_of_day: 240, + weekdays: 127, + timezone: BROWSER, + warn_minutes: 5, + enabled: true, + next_run_at: hoursFromNow(3), + run_state: "", + last_run_at: null, + last_result: "", + last_detail: "", + created_by: "user:u1", + created_at: "2026-09-01T00:00:00Z", + ...over, + }; +} + +function listing(schedules: Schedule[], limit = 20) { + calls.listSchedules.mockResolvedValue({ schedules, limit }); +} + +beforeEach(() => { + for (const fn of Object.values(calls)) fn.mockReset(); + tier.isAdmin = true; + calls.status.mockResolvedValue({ name: "survival", displayName: "Survival", phase: "Running" }); + calls.myServers.mockResolvedValue([]); + listing([]); +}); +afterEach(() => { + vi.useRealTimers(); + vi.restoreAllMocks(); + return i18next.changeLanguage("en-US"); +}); + +function renderPage() { + return render( + + + } /> + + , + ); +} + +async function rowOf(title: string): Promise { + return (await screen.findByText(title, { selector: "p" })).closest("li") as HTMLElement; +} + +async function openCreate() { + await userEvent.click(await screen.findByRole("button", { name: "New task" })); + return screen.findByRole("dialog"); +} + +async function pick(dialog: HTMLElement, box: string, option: string) { + await userEvent.click(within(dialog).getByRole("combobox", { name: box })); + await userEvent.click(await screen.findByRole("option", { name: option })); +} + +describe("ServerSchedules", () => { + it("shows a non-owner NotYours and never reads the list", async () => { + tier.isAdmin = false; + calls.myServers.mockResolvedValue([{ name: "survival", owned: false }]); + renderPage(); + expect(await screen.findByText("No permission to manage scheduled tasks")).toBeTruthy(); + expect(calls.listSchedules).not.toHaveBeenCalled(); + expect(screen.queryByRole("button", { name: "New task" })).toBeNull(); + }); + + it("shows an owner who is not an admin the list", async () => { + tier.isAdmin = false; + calls.myServers.mockResolvedValue([{ name: "survival", owned: true }]); + listing([schedule({ label: "Nightly" })]); + renderPage(); + expect(await rowOf("Nightly")).toBeTruthy(); + expect(calls.listSchedules).toHaveBeenCalledWith("survival"); + }); + + it("lists each task with its summary, next run and last result", async () => { + listing([ + schedule({ id: 1, label: "Nightly", last_result: "ok", last_run_at: hoursFromNow(-22) }), + schedule({ + id: 2, + action: "command", + command: "say hi", + every_minutes: 30, + minute_of_day: 0, + weekdays: 62, + timezone: OTHER, + warn_minutes: 0, + last_result: "failed", + last_run_at: hoursFromNow(-2), + last_detail: "the command failed: rcon: connection refused", + }), + schedule({ id: 3, action: "backup", weekdays: 42, minute_of_day: 330, warn_minutes: 0, enabled: false, next_run_at: null }), + ]); + renderPage(); + + const nightly = await rowOf("Nightly"); + expect(within(nightly).getByText("Every day at 04:00")).toBeTruthy(); + expect(within(nightly).getByText("Next run in 3 hours")).toBeTruthy(); + expect(within(nightly).getByText("Succeeded")).toBeTruthy(); + expect(within(nightly).getByText("Last run 22 hours ago")).toBeTruthy(); + expect(within(nightly).getByText("Warns 5 min before")).toBeTruthy(); + expect(within(nightly).getByText("Restart")).toBeTruthy(); + expect(within(nightly).queryByText(BROWSER)).toBeNull(); + + const command = await rowOf("Console command"); + expect(within(command).getByText("Every 30 minutes · Weekdays")).toBeTruthy(); + expect(within(command).getByText("say hi", { selector: "code" })).toBeTruthy(); + expect(within(command).getByText(OTHER)).toBeTruthy(); + expect(within(command).getByText("Failed")).toBeTruthy(); + expect(within(command).getByText("the command failed: rcon: connection refused")).toBeTruthy(); + expect(within(command).queryByText(/^Warns/)).toBeNull(); + + const backup = await rowOf("Backup"); + expect(within(backup).getByText("Mon, Wed, Fri at 05:30")).toBeTruthy(); + expect(within(backup).getByText("Off: it runs only when you press Run now")).toBeTruthy(); + expect(within(backup).getByText("Off")).toBeTruthy(); + expect(within(backup).getByText("Hasn't run yet")).toBeTruthy(); + + expect(screen.getByText("3 of 20 tasks")).toBeTruthy(); + }); + + it("starts the week on Monday in Chinese", async () => { + await i18next.changeLanguage("zh-CN"); + listing([schedule({ id: 1, label: "夜间重启", weekdays: 3 })]); + renderPage(); + expect(within(await rowOf("夜间重启")).getByText("周一、周日 04:00")).toBeTruthy(); + }); + + it("says in the panel's words why a run was skipped, and shows a command's reply", async () => { + listing([ + schedule({ + id: 1, + label: "Nightly", + enabled: false, + next_run_at: null, + last_result: "skipped", + last_run_at: hoursFromNow(-1), + last_detail: "the server has a new owner since this schedule was saved; save it again to use it", + }), + schedule({ + id: 2, + label: "Who", + action: "command", + command: "list", + warn_minutes: 0, + last_result: "ok", + last_run_at: hoursFromNow(-1), + last_detail: "There are 0 of a max of 20 players online", + }), + ]); + renderPage(); + const nightly = await rowOf("Nightly"); + expect(within(nightly).getByText("Skipped")).toBeTruthy(); + expect( + within(nightly).getByText( + "The server changed owner, so this task switched itself off. Check it, then save it or switch it on to use it again.", + ), + ).toBeTruthy(); + expect(within(await rowOf("Who")).getByText("Reply: There are 0 of a max of 20 players online")).toBeTruthy(); + }); + + it("creates a daily restart that warns the players", async () => { + calls.createSchedule.mockResolvedValue(schedule({ id: 9, label: "Nightly" })); + renderPage(); + const dialog = await openCreate(); + const restart = within(dialog).getByRole("radio", { name: "Restart" }) as HTMLInputElement; + expect(restart.checked).toBe(true); + // A command typed before switching back to a restart is not sent with it. + await userEvent.click(within(dialog).getByRole("radio", { name: "Console command" })); + await userEvent.type(within(dialog).getByLabelText("Command"), "say bye"); + await userEvent.click(restart); + fireEvent.change(within(dialog).getByLabelText("Time"), { target: { value: "03:30" } }); + await pick(dialog, "Warn players", "10 minutes before"); + await userEvent.type(within(dialog).getByLabelText("Name (optional)"), "Nightly"); + await userEvent.click(within(dialog).getByRole("button", { name: "Create" })); + + expect(calls.createSchedule).toHaveBeenCalledWith("survival", { + label: "Nightly", + action: "restart", + command: "", + every_minutes: 0, + minute_of_day: 210, + weekdays: 127, + timezone: BROWSER, + warn_minutes: 10, + enabled: true, + }); + expect(await screen.findByText("Created “Nightly”.")).toBeTruthy(); + expect(screen.queryByRole("dialog")).toBeNull(); + expect(calls.listSchedules).toHaveBeenCalledTimes(2); + }); + + it("creates a repeating console command as typed, on weekdays", async () => { + calls.createSchedule.mockResolvedValue(schedule({ id: 9, action: "command" })); + renderPage(); + const dialog = await openCreate(); + await userEvent.click(within(dialog).getByRole("radio", { name: "Console command" })); + await userEvent.type(within(dialog).getByLabelText("Command"), "/say hi"); + await userEvent.click(within(dialog).getByRole("radio", { name: "Repeat" })); + await pick(dialog, "Repeat every", "30 minutes"); + await userEvent.click(within(dialog).getByRole("button", { name: "Weekdays" })); + await userEvent.click(within(dialog).getByRole("button", { name: "Create" })); + + expect(calls.createSchedule).toHaveBeenCalledWith("survival", { + label: "", + action: "command", + command: "/say hi", + every_minutes: 30, + minute_of_day: 0, + weekdays: 62, + timezone: BROWSER, + warn_minutes: 0, + enabled: true, + }); + }); + + it("drops the warning for a command and keeps a server action hourly at most", async () => { + renderPage(); + const dialog = await openCreate(); + await pick(dialog, "Warn players", "10 minutes before"); + expect(within(dialog).getByRole("combobox", { name: "Warn players" }).textContent).toBe("10 minutes before"); + + await userEvent.click(within(dialog).getByRole("radio", { name: "Console command" })); + expect(within(dialog).queryByRole("combobox", { name: "Warn players" })).toBeNull(); + await userEvent.click(within(dialog).getByRole("radio", { name: "Repeat" })); + await pick(dialog, "Repeat every", "15 minutes"); + + await userEvent.click(within(dialog).getByRole("radio", { name: "Stop" })); + expect(within(dialog).getByRole("combobox", { name: "Warn players" }).textContent).toBe("No warning"); + expect(within(dialog).getByRole("combobox", { name: "Repeat every" }).textContent).toBe("1 hour"); + await userEvent.click(within(dialog).getByRole("combobox", { name: "Repeat every" })); + expect(await screen.findByRole("option", { name: "1 hour" })).toBeTruthy(); + expect(screen.queryByRole("option", { name: "30 minutes" })).toBeNull(); + }); + + it("will not save a task on no day", async () => { + renderPage(); + const dialog = await openCreate(); + for (const day of ["Sunday", "Monday", "Tuesday", "Wednesday", "Thursday", "Friday", "Saturday"]) { + await userEvent.click(within(dialog).getByRole("button", { name: day })); + } + expect(within(dialog).getByText("Pick at least one day.")).toBeTruthy(); + const create = within(dialog).getByRole("button", { name: "Create" }) as HTMLButtonElement; + expect(create.disabled).toBe(true); + await userEvent.click(create); + expect(calls.createSchedule).not.toHaveBeenCalled(); + + await userEvent.click(within(dialog).getByRole("button", { name: "Wednesday" })); + expect(within(dialog).queryByText("Pick at least one day.")).toBeNull(); + expect(create.disabled).toBe(false); + }); + + it("checks the command, name and time zone the way felis-api does", async () => { + renderPage(); + const dialog = await openCreate(); + const create = within(dialog).getByRole("button", { name: "Create" }) as HTMLButtonElement; + await userEvent.click(within(dialog).getByRole("radio", { name: "Console command" })); + const command = within(dialog).getByLabelText("Command"); + expect(within(dialog).getByText("Enter a command.")).toBeTruthy(); + // A lone slash is dropped, leaving nothing to run. + fireEvent.change(command, { target: { value: " / " } }); + expect(within(dialog).getByText("Enter a command.")).toBeTruthy(); + expect(create.disabled).toBe(true); + // The limit counts the command after its slash goes. + fireEvent.change(command, { target: { value: "/" + "x".repeat(1024) } }); + expect(create.disabled).toBe(false); + fireEvent.change(command, { target: { value: "x".repeat(1025) } }); + expect(within(dialog).getByText("The command is longer than 1024 bytes.")).toBeTruthy(); + expect(create.disabled).toBe(true); + // Bytes of UTF-8, as felis-api counts: 342 characters, 1026 bytes. + fireEvent.change(command, { target: { value: "中".repeat(342) } }); + expect(within(dialog).getByText("The command is longer than 1024 bytes.")).toBeTruthy(); + fireEvent.change(command, { target: { value: "say hi" } }); + + // Characters, not UTF-16 units, and the surrounding space does not count. + const label = within(dialog).getByLabelText("Name (optional)"); + fireEvent.change(label, { target: { value: ` ${"𠀀".repeat(64)} ` } }); + expect(create.disabled).toBe(false); + fireEvent.change(label, { target: { value: "𠀀".repeat(65) } }); + expect(within(dialog).getByText("The name is longer than 64 characters.")).toBeTruthy(); + expect(create.disabled).toBe(true); + fireEvent.change(label, { target: { value: "" } }); + + const zone = within(dialog).getByLabelText("Time zone") as HTMLInputElement; + fireEvent.change(zone, { target: { value: "Mars/Olympus" } }); + expect(within(dialog).getByText("Unknown time zone. Use a name such as Asia/Shanghai.")).toBeTruthy(); + expect(create.disabled).toBe(true); + fireEvent.change(zone, { target: { value: " " } }); + expect(within(dialog).getByText("Enter a time zone.")).toBeTruthy(); + await userEvent.click(within(dialog).getByRole("button", { name: "Use the browser's zone" })); + expect(zone.value).toBe(BROWSER); + expect(create.disabled).toBe(false); + }); + + it("keeps the dialog open with the server's reason when a save is refused", async () => { + calls.createSchedule.mockRejectedValue({ + status: 409, + code: "schedule_limit", + message: "a server can have at most 20 scheduled tasks", + }); + renderPage(); + const dialog = await openCreate(); + await userEvent.click(within(dialog).getByRole("button", { name: "Create" })); + expect((await within(dialog).findByRole("alert")).textContent).toBe( + "This server already has as many scheduled tasks as it can hold. Delete one first.", + ); + expect(screen.getByRole("dialog")).toBe(dialog); + expect((within(dialog).getByRole("button", { name: "Create" }) as HTMLButtonElement).disabled).toBe(false); + }); + + it("starts a nightly backup from its template", async () => { + calls.createSchedule.mockResolvedValue(schedule({ id: 9, action: "backup" })); + renderPage(); + await userEvent.click(await screen.findByRole("button", { name: "Back up every night at 05:00" })); + const dialog = await screen.findByRole("dialog"); + expect(within(dialog).getByText(/A backup of a running server stops it/)).toBeTruthy(); + await userEvent.click(within(dialog).getByRole("button", { name: "Create" })); + expect(calls.createSchedule).toHaveBeenCalledWith("survival", { + label: "", + action: "backup", + command: "", + every_minutes: 0, + minute_of_day: 300, + weekdays: 127, + timezone: BROWSER, + warn_minutes: 5, + enabled: true, + }); + }); + + it("edits a task and saves the whole of it", async () => { + listing([ + schedule({ id: 4, label: "Ad", action: "command", command: "say hi", every_minutes: 120, minute_of_day: 0, weekdays: 65, timezone: OTHER, warn_minutes: 0 }), + ]); + calls.updateSchedule.mockResolvedValue(schedule({ id: 4, label: "Advert" })); + renderPage(); + await userEvent.click(within(await rowOf("Ad")).getByRole("button", { name: "Edit" })); + const dialog = await screen.findByRole("dialog"); + expect(within(dialog).getByText("Edit scheduled task")).toBeTruthy(); + expect(within(dialog).getByRole("combobox", { name: "Repeat every" }).textContent).toBe("2 hours"); + await userEvent.type(within(dialog).getByLabelText("Name (optional)"), "vert"); + await userEvent.click(within(dialog).getByRole("button", { name: "Save" })); + expect(calls.updateSchedule).toHaveBeenCalledWith("survival", 4, { + label: "Advert", + action: "command", + command: "say hi", + every_minutes: 120, + minute_of_day: 0, + weekdays: 65, + timezone: OTHER, + warn_minutes: 0, + enabled: true, + }); + expect(await screen.findByText("Saved “Advert”.")).toBeTruthy(); + }); + + it("switches a task off by saving all of it with enabled flipped", async () => { + listing([ + schedule({ id: 4, label: "Nightly", minute_of_day: 330, weekdays: 42, timezone: OTHER, warn_minutes: 10 }), + ]); + let answer: (s: Schedule) => void = () => {}; + calls.updateSchedule.mockReturnValue(new Promise((resolve) => (answer = resolve))); + renderPage(); + const toggle = (await screen.findByRole("switch", { name: "Run Nightly on schedule" })) as HTMLButtonElement; + expect(toggle.getAttribute("aria-checked")).toBe("true"); + await userEvent.click(toggle); + expect(calls.updateSchedule).toHaveBeenCalledWith("survival", 4, { + label: "Nightly", + action: "restart", + command: "", + every_minutes: 0, + minute_of_day: 330, + weekdays: 42, + timezone: OTHER, + warn_minutes: 10, + enabled: false, + }); + expect(toggle.disabled).toBe(true); + await act(async () => answer(schedule({ id: 4, enabled: false }))); + expect(await screen.findByText("“Nightly” is off.")).toBeTruthy(); + }); + + it("asks before running a task now, and warns that a restart disconnects the players", async () => { + listing([ + schedule({ id: 1, label: "Nightly" }), + schedule({ id: 2, label: "Hello", action: "command", command: "say hi", warn_minutes: 0 }), + ]); + calls.runSchedule.mockResolvedValue(schedule({ id: 1, run_state: "stopping" })); + renderPage(); + await userEvent.click(within(await rowOf("Nightly")).getByRole("button", { name: "Run now" })); + const dialog = await screen.findByRole("dialog"); + expect(within(dialog).getByText("Stops the server now and starts it again.")).toBeTruthy(); + expect(within(dialog).getByText(KICK)).toBeTruthy(); + expect(calls.runSchedule).not.toHaveBeenCalled(); + await userEvent.click(within(dialog).getByRole("button", { name: "Run now" })); + expect(calls.runSchedule).toHaveBeenCalledWith("survival", 1); + expect(await screen.findByText("Started “Nightly”.")).toBeTruthy(); + + await userEvent.click(within(await rowOf("Hello")).getByRole("button", { name: "Run now" })); + const second = await screen.findByRole("dialog"); + expect(within(second).getByText("say hi", { selector: "code" })).toBeTruthy(); + expect(within(second).queryByText(KICK)).toBeNull(); + }); + + it("deletes a task once confirmed", async () => { + listing([schedule({ id: 5, label: "Old" })]); + calls.deleteSchedule.mockResolvedValue(null); + renderPage(); + await userEvent.click(within(await rowOf("Old")).getByRole("button", { name: "Delete" })); + const dialog = await screen.findByRole("dialog"); + expect(within(dialog).getByText("Delete “Old”?")).toBeTruthy(); + expect(calls.deleteSchedule).not.toHaveBeenCalled(); + await userEvent.click(within(dialog).getByRole("button", { name: "Delete" })); + expect(calls.deleteSchedule).toHaveBeenCalledWith("survival", 5); + expect(await screen.findByText("Deleted “Old”.")).toBeTruthy(); + }); + + it("locks a task while it runs, and says why", async () => { + // A claim stamps last_run_at with the run's start and clears the result. + const started = new Date(Date.now() - 5 * 60_000).toISOString(); + listing([ + schedule({ id: 1, label: "Busy", run_state: "stopping", last_run_at: started }), + schedule({ id: 2, label: "Idle" }), + ]); + renderPage(); + const busy = await rowOf("Busy"); + expect(within(busy).getByText("Stopping the server…")).toBeTruthy(); + expect(within(busy).getByText("Started 5 minutes ago")).toBeTruthy(); + expect(within(busy).queryByText("Hasn't run yet")).toBeNull(); + for (const name of ["Run now", "Edit", "Delete"]) { + const button = within(busy).getByRole("button", { name }) as HTMLButtonElement; + expect(button.disabled).toBe(true); + expect(button.parentElement?.title).toBe("This task is running. Wait for the run to finish."); + } + expect((within(busy).getByRole("switch") as HTMLButtonElement).disabled).toBe(true); + + const idle = await rowOf("Idle"); + for (const name of ["Run now", "Edit", "Delete"]) { + expect((within(idle).getByRole("button", { name }) as HTMLButtonElement).disabled).toBe(false); + } + expect((within(idle).getByRole("switch") as HTMLButtonElement).disabled).toBe(false); + }); + + it("offers no new task once the server holds its limit", async () => { + listing([schedule({ id: 1, label: "One" }), schedule({ id: 2, label: "Two" })], 2); + renderPage(); + const add = (await screen.findByRole("button", { name: "New task" })) as HTMLButtonElement; + await rowOf("One"); + expect(add.disabled).toBe(true); + expect(add.parentElement?.title).toBe("A server holds at most 2 tasks. Delete one to add another."); + }); + + it("says plainly when felis-api has no schedule store, and treats other failures as errors", async () => { + calls.listSchedules.mockRejectedValue({ status: 503, code: "schedules_unavailable", message: "not configured" }); + const first = renderPage(); + expect(await screen.findByText("Scheduled tasks aren't available")).toBeTruthy(); + expect(screen.queryByRole("button", { name: "New task" })).toBeNull(); + first.unmount(); + + calls.listSchedules.mockRejectedValue({ status: 500, code: "internal", message: "boom" }); + renderPage(); + expect(await screen.findByRole("alert")).toBeTruthy(); + expect(screen.queryByText("Scheduled tasks aren't available")).toBeNull(); + }); + + it("rereads fast while a run is in progress", async () => { + vi.useFakeTimers({ shouldAdvanceTime: true }); + listing([schedule({ id: 1, label: "Backup", action: "backup", run_state: "backing_up" })]); + renderPage(); + await screen.findByText("Backing up…"); + const reads = calls.listSchedules.mock.calls.length; + await act(() => vi.advanceTimersByTimeAsync(STATUS_POLL_FAST_MS + 100)); + expect(calls.listSchedules.mock.calls.length).toBe(reads + 1); + }); + + it("rereads slowly while nothing runs", async () => { + vi.useFakeTimers({ shouldAdvanceTime: true }); + listing([schedule({ id: 1, label: "Nightly" })]); + renderPage(); + await rowOf("Nightly"); + const reads = calls.listSchedules.mock.calls.length; + await act(() => vi.advanceTimersByTimeAsync(STATUS_POLL_FAST_MS + 100)); + expect(calls.listSchedules.mock.calls.length).toBe(reads); + await act(() => vi.advanceTimersByTimeAsync(STATUS_POLL_SLOW_MS - STATUS_POLL_FAST_MS)); + expect(calls.listSchedules.mock.calls.length).toBe(reads + 1); + }); +}); diff --git a/panel/src/pages/ServerSchedules.tsx b/panel/src/pages/ServerSchedules.tsx new file mode 100644 index 0000000..83603f3 --- /dev/null +++ b/panel/src/pages/ServerSchedules.tsx @@ -0,0 +1,1069 @@ +import { useId, useMemo, useState } from "react"; +import { useParams } from "react-router-dom"; +import { + AlertTriangle, + Archive, + CalendarClock, + CheckCircle2, + Clock, + Globe, + Info, + Loader2, + Megaphone, + MinusCircle, + Pencil, + Play, + Plus, + Power, + PowerOff, + RotateCw, + Terminal, + Trash2, + XCircle, + type LucideIcon, +} from "lucide-react"; +import { useTranslation } from "react-i18next"; +import type { TFunction } from "i18next"; +import { BackLink } from "@/components/BackLink"; +import { Badge } from "@/components/ui/badge"; +import { Button } from "@/components/ui/button"; +import { Card, CardContent } from "@/components/ui/card"; +import { Input } from "@/components/ui/input"; +import { Label } from "@/components/ui/label"; +import { Select, SelectContent, SelectItem, SelectTrigger, SelectValue } from "@/components/ui/select"; +import { Dialog, DialogContent, DialogDescription, DialogHeader, DialogTitle } from "@/components/ui/dialog"; +import { ConfirmDialog } from "@/components/ConfirmDialog"; +import { ConfirmFooter } from "@/components/ConfirmFooter"; +import { InlineError, MessageLine } from "@/components/MessageLine"; +import { PhaseBadge, shownPhase, startFailure } from "@/components/PhaseBadge"; +import { EmptyState, ErrorState, Loading, NotYours, RefreshError } from "@/components/States"; +import { PageHeader } from "@/components/PageHeader"; +import { api, humanizeError } from "@/lib/api"; +import { useAsync, usePolling, STATUS_POLL_FAST_MS, STATUS_POLL_SLOW_MS } from "@/lib/hooks"; +import { useTier } from "@/lib/tier"; +import { canManage, ownershipPending } from "@/lib/ownership"; +import { formatAbsolute, formatRelative } from "@/lib/format"; +import { cn } from "@/lib/utils"; +import type { + Schedule, + ScheduleAction, + ScheduleEveryMinutes, + ScheduleInput, + ScheduleResult, + ScheduleWarnMinutes, +} from "@/lib/types"; + +// The choices internal/api/schedules.go accepts (scheduleInput.apply). +const EVERY_OPTIONS: ScheduleEveryMinutes[] = [15, 30, 60, 120, 180, 240, 360, 480, 720]; +const WARN_OPTIONS: ScheduleWarnMinutes[] = [0, 1, 5, 10, 15, 30]; +const MAX_LABEL = 64; // runes +const MAX_COMMAND_BYTES = 1024; // maxConsoleCommandLen, after the leading "/" goes + +// Picker order: the common chores first. +const ACTIONS: ScheduleAction[] = ["restart", "backup", "command", "stop", "start"]; +const ACTION_ICONS: Record = { + command: Terminal, + restart: RotateCw, + stop: PowerOff, + start: Power, + backup: Archive, +}; +// The actions that take the server down: only these may warn the players first, +// and running one by hand disconnects them without that warning. +const WARNS = new Set(["restart", "stop", "backup"]); + +// Weekday bitmask, bit N = Go's time.Weekday N (0 = Sunday). +const ALL_DAYS = 127; +const WEEKDAYS = 62; +const WEEKEND = 65; +const DAY_PRESETS: [number, string][] = [ + [ALL_DAYS, "days_all"], + [WEEKDAYS, "days_weekdays"], + [WEEKEND, "days_weekend"], +]; + +// The fixed last_detail texts of the runner (schedulerunner.go), shown in the +// panel's language; any other detail (a console reply, an error with its +// cause) is shown as it came. +const DETAIL_KEYS: Record = { + "the server has a new owner since this schedule was saved; save it again to use it": "owner_changed", + "the server was not running": "detail_not_running", + "the server was already stopped": "detail_already_stopped", + "the server was already running": "detail_already_running", + "the server no longer exists": "detail_gone", + "the server has no world yet": "detail_no_world", + "felis-api was not running at the scheduled time": "detail_missed", + "someone started the server before the run finished": "detail_started_meanwhile", + "felis-api stopped in the middle of this run": "detail_interrupted", + "the backup store is full; ask an administrator to free space": "detail_store_full", + "the server console could not be reached": "detail_console", + "the cluster is at its running-server cap": "detail_cap", + "the server is being given up or deleted": "detail_retiring", +}; + +const RESULT_STYLE: Record, { icon: LucideIcon; className: string }> = { + ok: { icon: CheckCircle2, className: "text-emerald-600 dark:text-emerald-400" }, + skipped: { icon: MinusCircle, className: "text-muted-foreground" }, + failed: { icon: XCircle, className: "text-destructive" }, + missed: { icon: AlertTriangle, className: "text-amber-600 dark:text-amber-400" }, +}; + +/** Form is the create/edit dialog's state; toInput turns it into the wire body. */ +interface Form { + action: ScheduleAction; + command: string; + mode: "daily" | "interval"; + time: string; // HH:MM, for mode "daily" + every: ScheduleEveryMinutes; // for mode "interval" + weekdays: number; + timezone: string; + warn: ScheduleWarnMinutes; + label: string; + enabled: boolean; +} + +function browserZone(): string { + return Intl.DateTimeFormat().resolvedOptions().timeZone; +} + +function hhmm(minutes: number): string { + const pad = (n: number) => String(n).padStart(2, "0"); + return `${pad(Math.floor(minutes / 60))}:${pad(minutes % 60)}`; +} + +function newForm(over: Partial
= {}): Form { + return { + action: "restart", + command: "", + mode: "daily", + time: "04:00", + every: 60, + weekdays: ALL_DAYS, + timezone: browserZone(), + warn: 5, + label: "", + enabled: true, + ...over, + }; +} + +function fromSchedule(s: Schedule): Form { + return { + action: s.action, + command: s.command, + mode: s.every_minutes ? "interval" : "daily", + time: hhmm(s.minute_of_day), + every: s.every_minutes || 60, + weekdays: s.weekdays, + timezone: s.timezone, + warn: s.warn_minutes, + label: s.label, + enabled: s.enabled, + }; +} + +// toInput sends the fields as typed: felis-api trims them and drops a +// command's leading "/" itself, so what it stores is what it validated. +function toInput(f: Form): ScheduleInput { + const [h, m] = f.time.split(":").map(Number); + return { + label: f.label, + action: f.action, + command: f.action === "command" ? f.command : "", + every_minutes: f.mode === "interval" ? f.every : 0, + minute_of_day: f.mode === "daily" ? h * 60 + m : 0, + weekdays: f.weekdays, + timezone: f.timezone, + warn_minutes: f.warn, + enabled: f.enabled, + }; +} + +function knownZone(zone: string): boolean { + try { + new Intl.DateTimeFormat("en-US", { timeZone: zone }); + return true; + } catch { + return false; + } +} + +// zoneList is what the time zone field suggests: the browser's zone first. +function zoneList(browser: string): string[] { + let all: string[] = []; + try { + all = (Intl as unknown as { supportedValuesOf?: (key: string) => string[] }).supportedValuesOf?.("timeZone") ?? []; + } catch { + // An older browser lists nothing; the field still takes any name. + } + return [browser, ...all.filter((z) => z !== browser)]; +} + +type Field = "command" | "time" | "days" | "timezone" | "label"; + +// problems mirrors scheduleInput.apply, so a form the server would refuse +// cannot be sent. The server stays the judge: whatever slips through comes back +// as a message in the dialog. +function problems(f: Form, t: TFunction): Partial> { + const out: Partial> = {}; + if (f.action === "command") { + const command = f.command.trim().replace(/^\//, "").trim(); + if (!command) out.command = t("err_command_required"); + else if (new TextEncoder().encode(command).length > MAX_COMMAND_BYTES) { + out.command = t("err_command_long", { max: MAX_COMMAND_BYTES }); + } + } + if (f.mode === "daily" && !/^\d\d:\d\d$/.test(f.time)) out.time = t("err_time_required"); + if (f.weekdays === 0) out.days = t("err_days"); + const zone = f.timezone.trim(); + if (!zone) out.timezone = t("err_timezone_required"); + else if (!knownZone(zone)) out.timezone = t("err_timezone"); + if ([...f.label.trim()].length > MAX_LABEL) out.label = t("err_label_long", { max: MAX_LABEL }); + return out; +} + +// weekOrder is the weekdays in the locale's order (Monday first in Chinese). +function weekOrder(t: TFunction): number[] { + const first = t("first_day") === "1" ? 1 : 0; + return [0, 1, 2, 3, 4, 5, 6].map((i) => (i + first) % 7); +} + +function daysText(mask: number, t: TFunction): string { + const preset = DAY_PRESETS.find(([m]) => m === mask); + if (preset) return t(preset[1]); + return weekOrder(t) + .filter((d) => mask & (1 << d)) + .map((d) => t(`day_${d}`)) + .join(t("day_sep")); +} + +function everyText(minutes: number, t: TFunction): string { + return minutes < 60 ? t("every_minutes", { count: minutes }) : t("every_hours", { count: minutes / 60 }); +} + +function optionText(minutes: number, t: TFunction): string { + return minutes < 60 ? t("opt_minutes", { count: minutes }) : t("opt_hours", { count: minutes / 60 }); +} + +// summary is the one-line "when" of a schedule: "Every day at 04:00", +// "Every 30 minutes · Weekdays". +function summary(s: Schedule, t: TFunction): string { + const days = daysText(s.weekdays, t); + if (!s.every_minutes) return t("when_daily", { days, time: hhmm(s.minute_of_day) }); + const every = everyText(s.every_minutes, t); + return s.weekdays === ALL_DAYS ? every : t("when_interval_days", { every, days }); +} + +// runTimes previews an interval's runs: it counts from midnight in its zone. +function runTimes(every: number): string { + const times: string[] = []; + for (let m = 0; m < 24 * 60; m += every) times.push(hhmm(m)); + return times.length <= 4 ? times.join(", ") : `${times.slice(0, 3).join(", ")} … ${times[times.length - 1]}`; +} + +function titleOf(s: Schedule, t: TFunction): string { + return s.label || t(`action_${s.action}`); +} + +function FieldNote({ error, hint }: { error?: string; hint?: string }) { + if (error) return

{error}

; + return hint ?

{hint}

: null; +} + +/** ScheduleDialog creates a schedule, or edits `editing`. It stays open on a + * refusal and shows the server's reason above its buttons. */ +function ScheduleDialog({ + serverName, + editing, + initial, + onClose, + onSaved, +}: { + serverName: string; + editing: Schedule | null; + initial: Form; + onClose: () => void; + onSaved: (s: Schedule) => void; +}) { + const { t } = useTranslation("schedules"); + const { t: tc } = useTranslation("common"); + const id = useId(); + const [form, setForm] = useState(initial); + const [saving, setSaving] = useState(false); + const [error, setError] = useState(null); + const browser = browserZone(); + const zones = useMemo(() => zoneList(browser), [browser]); + const bad = problems(form, t); + const labelLength = [...form.label.trim()].length; + + function set(key: K, value: Form[K]) { + setForm((f) => ({ ...f, [key]: value })); + } + + // Only a restart, stop or backup warns, and only a command repeats more + // often than hourly: a switch clears what the new action cannot keep. + function setAction(action: ScheduleAction) { + setForm((f) => ({ + ...f, + action, + warn: WARNS.has(action) ? f.warn : 0, + every: action === "command" || f.every >= 60 ? f.every : 60, + })); + } + + async function save() { + setSaving(true); + setError(null); + try { + const input = toInput(form); + onSaved( + editing + ? await api.updateSchedule(serverName, editing.id, input) + : await api.createSchedule(serverName, input), + ); + } catch (e) { + setError(humanizeError(e)); + setSaving(false); + } + } + + const everyOptions = form.action === "command" ? EVERY_OPTIONS : EVERY_OPTIONS.filter((m) => m >= 60); + + return ( + { + if (!open && !saving) onClose(); + }} + > + + + {editing ? t("edit_title") : t("create_title")} + {t("dialog_desc")} + + +
+
+ {t("field_action")} +
+ {ACTIONS.map((a) => { + const Icon = ACTION_ICONS[a]; + const checked = form.action === a; + return ( + + ); + })} +
+
+ + {form.action === "command" && ( +
+ + set("command", e.target.value)} + placeholder={t("command_placeholder")} + autoComplete="off" + spellCheck={false} + aria-invalid={!!bad.command} + /> + +
+ )} + +
+ {t("field_when")} +
+ {(["daily", "interval"] as const).map((m) => ( + + ))} +
+ {form.mode === "daily" ? ( +
+ + set("time", e.target.value)} + aria-invalid={!!bad.time} + /> + +
+ ) : ( +
+ + +

+ {t("interval_preview", { times: runTimes(form.every) })} + {form.action !== "command" && ` ${t("interval_hourly_limit")}`} +

+
+ )} +
+ +
+ {t("field_days")} +
+ {weekOrder(t).map((d) => { + const on = (form.weekdays & (1 << d)) !== 0; + return ( + + ); + })} +
+
+ {DAY_PRESETS.map(([mask, key]) => ( + + ))} +
+ +
+ +
+ + set("timezone", e.target.value)} + autoComplete="off" + spellCheck={false} + aria-invalid={!!bad.timezone} + /> + + {zones.map((z) => ( + + {bad.timezone ? ( + + ) : ( +

{t("timezone_hint", { zone: browser })}

+ )} + {form.timezone !== browser && ( + + )} +
+ + {WARNS.has(form.action) && ( +
+ + +

{t("warn_hint")}

+
+ )} + + {form.action === "backup" && ( +

+ + {t("backup_note")} +

+ )} + +
+
+ + + {labelLength}/{MAX_LABEL} + +
+ set("label", e.target.value)} + placeholder={t("label_placeholder")} + aria-invalid={!!bad.label} + /> + +
+ + +
+ + + void save()} + disabled={Object.keys(bad).length > 0} + loading={saving} + cancelLabel={tc("cancel")} + confirmLabel={editing ? t("save") : t("create")} + confirmVariant="default" + /> +
+
+ ); +} + +/** Disabled buttons take no pointer events, so the reason rides a wrapper. */ +function Why({ why, children }: { why?: string; children: React.ReactNode }) { + return ( + + {children} + + ); +} + +/** ScheduleRow is one schedule: what, when, how the last run went, and its + * actions. A run in progress locks it (the API answers 409 schedule_running). */ +function ScheduleRow({ + s, + now, + locale, + browser, + toggling, + onToggle, + onRun, + onEdit, + onDelete, +}: { + s: Schedule; + now: number; + locale: string; + browser: string; + toggling: boolean; + onToggle: () => void; + onRun: () => void; + onEdit: () => void; + onDelete: () => void; +}) { + const { t } = useTranslation("schedules"); + const Icon = ACTION_ICONS[s.action]; + const title = titleOf(s, t); + const busy = s.run_state !== ""; + const why = busy ? t("busy_hint") : undefined; + const result = s.last_result ? RESULT_STYLE[s.last_result] : null; + const detailKey = DETAIL_KEYS[s.last_detail]; + const detail = detailKey + ? t(detailKey) + : s.last_result === "ok" && s.last_detail + ? t("reply", { text: s.last_detail }) + : s.last_detail; + + return ( +
  • +
    +
    + +
    +
    +
    +

    {title}

    + {s.label && {t(`action_${s.action}`)}} + {!s.enabled && {t("off_badge")}} + {s.warn_minutes > 0 && ( + + + {t("warn_badge", { count: s.warn_minutes })} + + )} + {busy && ( + + + {t(`run_state_${s.run_state}`)} + + )} +
    +

    + + + {summary(s, t)} + + {s.timezone !== browser && ( + + + {s.timezone} + + )} +

    + {s.action === "command" && ( + + {s.command} + + )} +
    + {s.next_run_at ? ( + + {t("next_run", { when: formatRelative(s.next_run_at, now, locale) })} + + ) : ( + {t("next_run_off")} + )} + {result && s.last_run_at ? ( + + + + {t(`result_${s.last_result}`)} + + + {t("last_run", { when: formatRelative(s.last_run_at, now, locale) })} + + + ) : s.last_run_at ? ( + // A claim sets last_run_at to its start and clears the result, so + // this is the run in progress. + + {t("run_started_at", { when: formatRelative(s.last_run_at, now, locale) })} + + ) : ( + {t("never_run")} + )} +
    + {detail && ( +

    + {detail} +

    + )} +
    +
    + +
    + + + +
    + + + + + + + + + +
    +
    +
  • + ); +} + +export function ServerSchedules() { + const { name = "" } = useParams(); + const { t, i18n } = useTranslation("schedules"); + const { isAdmin, loading: tierLoading } = useTier(); + const statusQ = useAsync(() => api.status(name), [name]); + const mineQ = useAsync(() => (isAdmin ? Promise.resolve([]) : api.myServers()), [isAdmin, name]); + // Ownership resolves from /me/servers for a non-admin; the list is read only + // once the viewer is known to own the server (the route answers 403 otherwise). + const pending = ownershipPending(tierLoading, isAdmin, mineQ.data, mineQ.error); + const owned = canManage(isAdmin, mineQ.data, name); + const listQ = useAsync(() => (owned ? api.listSchedules(name) : Promise.resolve(null)), [name, owned]); + // A run in progress steps on every few seconds (and takes the server with it), + // so both reads follow it closely; otherwise slowly, for the next runs and the + // results of runs that came due. + const running = (listQ.data?.schedules ?? []).some((s) => s.run_state !== ""); + const pollMs = running ? STATUS_POLL_FAST_MS : STATUS_POLL_SLOW_MS; + usePolling(statusQ.reload, pollMs); + usePolling(listQ.reload, pollMs); + + const [msg, setMsg] = useState<{ kind: "success" | "error"; text: string } | null>(null); + const [editor, setEditor] = useState<{ editing: Schedule | null; form: Form } | null>(null); + const [runTarget, setRunTarget] = useState(null); + const [deleteTarget, setDeleteTarget] = useState(null); + const [toggling, setToggling] = useState(null); + + function openCreate(over: Partial = {}) { + setMsg(null); + setEditor({ editing: null, form: newForm(over) }); + } + + // The switch saves the schedule as it is with `enabled` flipped: a PUT takes + // the whole input, and it also hands the schedule to the current owner. + async function toggle(s: Schedule) { + setToggling(s.id); + setMsg(null); + try { + await api.updateSchedule(name, s.id, toInput({ ...fromSchedule(s), enabled: !s.enabled })); + setMsg({ kind: "success", text: t(s.enabled ? "toggled_off" : "toggled_on", { name: titleOf(s, t) }) }); + listQ.reload(); + } catch (e) { + setMsg({ kind: "error", text: humanizeError(e) }); + } finally { + setToggling(null); + } + } + + const back = ; + + if (statusQ.loading && !statusQ.data) { + return ( + <> + {back} + + + ); + } + if (statusQ.error && !statusQ.data) { + return ( + <> + {back} + + + ); + } + if (!statusQ.data) return back; + + const now = Date.now(); + const locale = i18n.language; + const browser = browserZone(); + const list = listQ.data; + const full = !!list && list.schedules.length >= list.limit; + const templates: [string, LucideIcon, Partial][] = [ + ["template_restart", RotateCw, {}], + ["template_backup", Archive, { action: "backup", time: "05:00" }], + [ + "template_announce", + Megaphone, + { action: "command", command: t("template_announce_command"), mode: "interval", every: 30, warn: 0 }, + ], + ]; + + const header = ( + + {owned && list && ( + + + + )} + + + } + className="mb-6" + /> + ); + + let body: React.ReactNode; + if (listQ.error && !list) { + body = + (listQ.error as { code?: string }).code === "schedules_unavailable" ? ( + + + +

    {t("unavailable_title")}

    +

    {t("unavailable_body")}

    +
    +
    + ) : ( + + ); + } else if (!list) { + body = ; + } else { + body = ( + <> + {!!listQ.error && } + {list.schedules.length === 0 ? ( + +
    + {templates.map(([key, Icon, over]) => ( + + ))} +
    +
    + ) : ( + +
      + {list.schedules.map((s) => ( + void toggle(s)} + onRun={() => setRunTarget(s)} + onEdit={() => { + setMsg(null); + setEditor({ editing: s, form: fromSchedule(s) }); + }} + onDelete={() => setDeleteTarget(s)} + /> + ))} +
    +
    + )} + + +

    + + {t("notes_title")} +

    +
      +
    • {t("note_timing")}
    • +
    • {t("note_state")}
    • +
    • {t("note_owner")}
    • +
    • {t("note_run_now")}
    • +
    +
    +
    + + ); + } + + return ( + <> + {back} + {header} + {!!statusQ.error && } + {pending ? ( + + ) : mineQ.error ? ( + + ) : !owned ? ( + + ) : ( +
    +
    +

    {t("subtitle")}

    + {list && ( + + {t("count", { count: list.schedules.length, limit: list.limit })} + + )} +
    + {msg && } + {body} +
    + )} + + {editor && ( + setEditor(null)} + onSaved={(s) => { + setMsg({ kind: "success", text: t(editor.editing ? "saved" : "created", { name: titleOf(s, t) }) }); + setEditor(null); + listQ.reload(); + }} + /> + )} + + {runTarget && ( + { + if (!open) setRunTarget(null); + }} + title={t("run_title", { name: titleOf(runTarget, t) })} + description={ + <> + {t(`run_body_${runTarget.action}`)} + {runTarget.action === "command" && ( + + {runTarget.command} + + )} + {WARNS.has(runTarget.action) && ( + + + {t("run_kick")} + + )} + {t("run_keeps_next")} + + } + confirmLabel={t("run_confirm")} + confirmVariant={WARNS.has(runTarget.action) ? "destructive" : "default"} + onConfirm={async () => { + await api.runSchedule(name, runTarget.id); + setMsg({ kind: "success", text: t("run_started", { name: titleOf(runTarget, t) }) }); + listQ.reload(); + }} + /> + )} + + {deleteTarget && ( + { + if (!open) setDeleteTarget(null); + }} + title={t("delete_title", { name: titleOf(deleteTarget, t) })} + description={t("delete_body")} + confirmLabel={t("delete_confirm")} + onConfirm={async () => { + await api.deleteSchedule(name, deleteTarget.id); + setMsg({ kind: "success", text: t("deleted", { name: titleOf(deleteTarget, t) }) }); + listQ.reload(); + }} + /> + )} + + ); +}