feat(schedules): 每台服务器可设计划任务,按星期和时区定时发命令、重启、停服、开服或备份,重启和备份前可在游戏内提醒
This commit is contained in:
32 files changed
+6011
-27
No files matched your search
@@ -98,6 +98,11 @@ type API struct {
|
||||
FileStage *fileedit.Stage
|
||||
InternalBaseURL string
|
||||
|
||||
// Schedules stores the servers' scheduled tasks (schedules.go), which
|
||||
// RunSchedules fires. Optional: when nil the schedule routes report 503 and
|
||||
// RunSchedules does nothing.
|
||||
Schedules ServerSchedules
|
||||
|
||||
// Submissions is the user-modpack approval lane (a user-directed extension over
|
||||
// the §16 build subsystem; see internal/submit). It is optional: when
|
||||
// nil the /me/submissions and /submissions routes report 503 rather than 404, so
|
||||
@@ -576,6 +581,15 @@ func (a *API) externalAPIRoutes() []apiRoute {
|
||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/files/mkdir", h: a.handleMkdir},
|
||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/files/rename", h: a.handleRenameFile},
|
||||
{Method: "PUT", Pattern: "/api/v1/servers/{name}/files/upload", h: a.handleUploadFile},
|
||||
// Scheduled tasks (handlers_schedules.go): a console command, restart, stop,
|
||||
// start or backup at set times, which felis-api's runner fires. App-tier and
|
||||
// owner-or-admin inside the handler, like the console and power routes they
|
||||
// automate; a schedule reaches nothing its owner could not do by hand.
|
||||
{Method: "GET", Pattern: "/api/v1/servers/{name}/schedules", h: a.handleListSchedules},
|
||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/schedules", h: a.handleCreateSchedule},
|
||||
{Method: "PUT", Pattern: "/api/v1/servers/{name}/schedules/{id}", h: a.handleUpdateSchedule},
|
||||
{Method: "DELETE", Pattern: "/api/v1/servers/{name}/schedules/{id}", h: a.handleDeleteSchedule},
|
||||
{Method: "POST", Pattern: "/api/v1/servers/{name}/schedules/{id}/run", h: a.handleRunSchedule},
|
||||
// Account linking (spec §10), web side: /start reports link status (it is the
|
||||
// pointer handleClaim's 412 emits), /verify consumes the in-game code and binds
|
||||
// the account. App-tier, not admin — linking your own account is an ordinary
|
||||
|
||||
@@ -3,7 +3,6 @@ package api
|
||||
import (
|
||||
"errors"
|
||||
"net/http"
|
||||
"strings"
|
||||
|
||||
"felis.lolicon.best/internal/naming"
|
||||
)
|
||||
@@ -51,23 +50,10 @@ func (a *API) handleCommand(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
|
||||
// A console command is exactly one line. Trim surrounding space, strip a
|
||||
// single leading '/' (players type "/say hi"; RCON wants "say hi"), then
|
||||
// reject control characters so one request can never smuggle a second command
|
||||
// past a newline.
|
||||
command := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(body.Command), "/"))
|
||||
if command == "" {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_request", "command is required"))
|
||||
return
|
||||
}
|
||||
if len(command) > maxConsoleCommandLen {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_request",
|
||||
"command too long (max %d bytes)", maxConsoleCommandLen))
|
||||
return
|
||||
}
|
||||
if strings.IndexFunc(command, func(c rune) bool { return c < 0x20 }) >= 0 {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_request",
|
||||
"command must be a single line (no control characters)"))
|
||||
// A console command is exactly one line (normalizeConsoleCommand).
|
||||
command, err := normalizeConsoleCommand(body.Command)
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,233 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/naming"
|
||||
)
|
||||
|
||||
// scheduleRunTimeout bounds the first step of a run somebody asked for. The
|
||||
// step outlives the request, so a client that hangs up cannot cut a stop off
|
||||
// halfway.
|
||||
const scheduleRunTimeout = 30 * time.Second
|
||||
|
||||
// scheduleServer resolves {name} for the schedule routes and applies their
|
||||
// gate: 400 for a malformed name, 404 for a server that does not exist, 403
|
||||
// for a caller who neither owns it nor is an admin, 503 without a store.
|
||||
func (a *API) scheduleServer(w http.ResponseWriter, r *http.Request) (*ServerRecord, bool) {
|
||||
name := r.PathValue("name")
|
||||
if err := naming.ValidateServerName(name); err != nil {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_name", "invalid server name: %v", err))
|
||||
return nil, false
|
||||
}
|
||||
rec, err := a.Repo.ServerByName(r.Context(), name)
|
||||
if err != nil {
|
||||
a.writeLookupError(w, r, err)
|
||||
return nil, false
|
||||
}
|
||||
if !a.isOwnerOrAdmin(principalFromContext(r.Context()), rec) {
|
||||
writeError(w, r, errForbidden)
|
||||
return nil, false
|
||||
}
|
||||
if a.Schedules == nil {
|
||||
writeError(w, r, newError(http.StatusServiceUnavailable, "schedules_unavailable",
|
||||
"scheduled tasks are not configured"))
|
||||
return nil, false
|
||||
}
|
||||
return rec, true
|
||||
}
|
||||
|
||||
// scheduleID parses {id}.
|
||||
func scheduleID(w http.ResponseWriter, r *http.Request) (int64, bool) {
|
||||
id, err := strconv.ParseInt(r.PathValue("id"), 10, 64)
|
||||
if err != nil || id <= 0 {
|
||||
writeError(w, r, newError(http.StatusBadRequest, "bad_id", "invalid schedule id"))
|
||||
return 0, false
|
||||
}
|
||||
return id, true
|
||||
}
|
||||
|
||||
// writeScheduleError maps the store's schedule errors.
|
||||
func (a *API) writeScheduleError(w http.ResponseWriter, r *http.Request, err error) {
|
||||
switch {
|
||||
case errors.Is(err, ErrScheduleRunning):
|
||||
writeError(w, r, newError(http.StatusConflict, "schedule_running",
|
||||
"this schedule is running; try again once the run finishes"))
|
||||
case errors.Is(err, ErrScheduleLimit):
|
||||
writeError(w, r, newError(http.StatusConflict, "schedule_limit",
|
||||
"a server can have at most %d scheduled tasks", maxSchedulesPerServer))
|
||||
default:
|
||||
a.writeLookupError(w, r, err)
|
||||
}
|
||||
}
|
||||
|
||||
// auditSchedule records a change to a schedule by the signed-in caller.
|
||||
func (a *API) auditSchedule(r *http.Request, action string, s *Schedule) {
|
||||
p := principalFromContext(r.Context())
|
||||
e := AuditEntry{Actor: auditActor(p), Action: action, ServerName: s.Server,
|
||||
Payload: auditPayload(map[string]any{"schedule": s.ID, "action": s.Action})}
|
||||
if p != nil {
|
||||
e.ActorUserID = p.UserID
|
||||
}
|
||||
a.auditEntry(r, e)
|
||||
}
|
||||
|
||||
// handleListSchedules serves GET /servers/{name}/schedules.
|
||||
func (a *API) handleListSchedules(w http.ResponseWriter, r *http.Request) {
|
||||
rec, ok := a.scheduleServer(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
list, err := a.Schedules.ListSchedules(r.Context(), rec.Name)
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
if list == nil {
|
||||
list = []Schedule{}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]any{"server": rec.Name, "schedules": list, "limit": maxSchedulesPerServer})
|
||||
}
|
||||
|
||||
// handleCreateSchedule serves POST /servers/{name}/schedules. The schedule
|
||||
// belongs to the server's current owner (see Schedule.OwnerID).
|
||||
func (a *API) handleCreateSchedule(w http.ResponseWriter, r *http.Request) {
|
||||
rec, ok := a.scheduleServer(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
var in scheduleInput
|
||||
if err := decodeJSON(w, r, &in); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
s := &Schedule{Server: rec.Name, OwnerID: rec.OwnerID, CreatedBy: auditActor(principalFromContext(r.Context()))}
|
||||
if err := in.apply(s); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
s.NextRunAt = a.firstRun(s)
|
||||
if err := a.Schedules.CreateSchedule(r.Context(), s, maxSchedulesPerServer); err != nil {
|
||||
a.writeScheduleError(w, r, err)
|
||||
return
|
||||
}
|
||||
a.auditSchedule(r, "schedule.create", s)
|
||||
writeJSON(w, http.StatusCreated, s)
|
||||
}
|
||||
|
||||
// handleUpdateSchedule serves PUT /servers/{name}/schedules/{id}: new settings,
|
||||
// and the schedule passes to the server's current owner.
|
||||
func (a *API) handleUpdateSchedule(w http.ResponseWriter, r *http.Request) {
|
||||
rec, ok := a.scheduleServer(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
id, ok := scheduleID(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
var in scheduleInput
|
||||
if err := decodeJSON(w, r, &in); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
s, err := a.Schedules.GetSchedule(r.Context(), rec.Name, id)
|
||||
if err != nil {
|
||||
a.writeScheduleError(w, r, err)
|
||||
return
|
||||
}
|
||||
if err := in.apply(s); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
s.OwnerID, s.NextRunAt = rec.OwnerID, a.firstRun(s)
|
||||
if err := a.Schedules.UpdateSchedule(r.Context(), s); err != nil {
|
||||
a.writeScheduleError(w, r, err)
|
||||
return
|
||||
}
|
||||
a.auditSchedule(r, "schedule.update", s)
|
||||
writeJSON(w, http.StatusOK, s)
|
||||
}
|
||||
|
||||
// handleDeleteSchedule serves DELETE /servers/{name}/schedules/{id}.
|
||||
func (a *API) handleDeleteSchedule(w http.ResponseWriter, r *http.Request) {
|
||||
rec, ok := a.scheduleServer(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
id, ok := scheduleID(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
s, err := a.Schedules.GetSchedule(r.Context(), rec.Name, id)
|
||||
if err != nil {
|
||||
a.writeScheduleError(w, r, err)
|
||||
return
|
||||
}
|
||||
if err := a.Schedules.DeleteSchedule(r.Context(), rec.Name, id); err != nil {
|
||||
a.writeScheduleError(w, r, err)
|
||||
return
|
||||
}
|
||||
a.auditSchedule(r, "schedule.delete", s)
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}
|
||||
|
||||
// handleRunSchedule serves POST /servers/{name}/schedules/{id}/run: the run
|
||||
// starts now, without the players' warning, and the next scheduled run stays
|
||||
// where it is. The answer is the schedule after the run's first step; a restart
|
||||
// or backup goes on in the background.
|
||||
func (a *API) handleRunSchedule(w http.ResponseWriter, r *http.Request) {
|
||||
rec, ok := a.scheduleServer(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
id, ok := scheduleID(w, r)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
s, err := a.Schedules.GetSchedule(r.Context(), rec.Name, id)
|
||||
if err != nil {
|
||||
a.writeScheduleError(w, r, err)
|
||||
return
|
||||
}
|
||||
if s.OwnerID != rec.OwnerID {
|
||||
writeError(w, r, newError(http.StatusConflict, "schedule_stale",
|
||||
"the server has a new owner since this schedule was saved; save it again first"))
|
||||
return
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.WithoutCancel(r.Context()), scheduleRunTimeout)
|
||||
defer cancel()
|
||||
now := a.now()
|
||||
claimed, err := a.Schedules.ClaimScheduleRun(ctx, id, nil, nil, now)
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
if !claimed {
|
||||
a.writeScheduleError(w, r, ErrScheduleRunning)
|
||||
return
|
||||
}
|
||||
a.auditSchedule(r, "schedule.run_now", s)
|
||||
if err := a.beginRun(ctx, s); err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
after, err := a.Schedules.GetSchedule(ctx, rec.Name, id)
|
||||
if err != nil {
|
||||
writeError(w, r, err)
|
||||
return
|
||||
}
|
||||
writeJSON(w, http.StatusAccepted, after)
|
||||
}
|
||||
|
||||
// firstRun is the next run of a schedule just saved, nil while it is disabled.
|
||||
func (a *API) firstRun(s *Schedule) *time.Time {
|
||||
if !s.Enabled {
|
||||
return nil
|
||||
}
|
||||
return ptrTime(s.nextRun(a.now()))
|
||||
}
|
||||
@@ -40,6 +40,7 @@ func TestOpenAPISchemasMatchWireStructs(t *testing.T) {
|
||||
"MyServerView": MyServerView{},
|
||||
"AllowlistEntry": AllowlistEntry{},
|
||||
"BackupView": BackupView{},
|
||||
"Schedule": Schedule{},
|
||||
"Build": build.Build{},
|
||||
"Image": build.Image{},
|
||||
"BuildScan": buildScanView{},
|
||||
|
||||
+17
-6
@@ -830,11 +830,11 @@ func (p *PGRepo) ServerOwners(ctx context.Context) (map[string]ServerOwnership,
|
||||
// a create whose CRD write failed, or one the reaper deleted between removing
|
||||
// its MinecraftServer and marking the row. That row starts over, and the earlier
|
||||
// server's other aliases and allowlist go with it; nothing of its owner, claim,
|
||||
// activity clock, reaper warnings or pending retirement reaches the new server,
|
||||
// so the reaper's unfinished deletion no longer applies to it. A retried create
|
||||
// lands on the same fresh state. The alias subdomain is a PRIMARY KEY: bound to
|
||||
// another server, it rolls the whole seed back and returns ErrConflict, letting
|
||||
// the create handler answer 409 before it touches the CRD.
|
||||
// activity clock, reaper warnings, pending retirement or scheduled tasks reaches
|
||||
// the new server, so the reaper's unfinished deletion no longer applies to it. A
|
||||
// retried create lands on the same fresh state. The alias subdomain is a PRIMARY
|
||||
// KEY: bound to another server, it rolls the whole seed back and returns
|
||||
// ErrConflict, letting the create handler answer 409 before it touches the CRD.
|
||||
//
|
||||
// Two creates of one name racing between the handler's cluster check and the
|
||||
// first CRD write can still leave the loser's alias in place of the winner's;
|
||||
@@ -862,6 +862,9 @@ func (p *PGRepo) SeedServer(ctx context.Context, name, subdomain string, cpuMill
|
||||
if _, err := tx.ExecContext(ctx, `DELETE FROM server_aliases WHERE server_name = $1`, name); err != nil {
|
||||
return fmt.Errorf("clear earlier aliases: %w", err)
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx, `DELETE FROM server_schedules WHERE server_name = $1`, name); err != nil {
|
||||
return fmt.Errorf("clear earlier schedules: %w", err)
|
||||
}
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
`INSERT INTO server_aliases (subdomain, server_name) VALUES ($1, $2) ON CONFLICT DO NOTHING`,
|
||||
subdomain, name); err != nil {
|
||||
@@ -2613,7 +2616,8 @@ func (p *PGRepo) RedeemMigration(ctx context.Context, targetUserID, codeHash str
|
||||
}
|
||||
|
||||
// Re-point every server the source owns to the target, collecting the names for
|
||||
// the audit trail. Server ownership is the only thing that moves.
|
||||
// the audit trail. Server ownership moves, and the servers' scheduled tasks
|
||||
// with it: they belong to the same person.
|
||||
rows, err := tx.QueryContext(ctx,
|
||||
`UPDATE servers SET owner_id = $2, claimed_at = $3
|
||||
WHERE owner_id = $1 AND deleted_at IS NULL
|
||||
@@ -2637,6 +2641,13 @@ func (p *PGRepo) RedeemMigration(ctx context.Context, targetUserID, codeHash str
|
||||
}
|
||||
rows.Close()
|
||||
|
||||
if _, err := tx.ExecContext(ctx,
|
||||
`UPDATE server_schedules s SET owner_id = $2 FROM servers v
|
||||
WHERE v.name = s.server_name AND v.owner_id = $2 AND s.owner_id = $1`,
|
||||
sourceUserID, targetUserID); err != nil {
|
||||
return "", nil, err
|
||||
}
|
||||
|
||||
// Retire the source: revoke its live sessions and soft-delete it so it can neither
|
||||
// log in nor start another migration (double-spend defense). The servers just moved
|
||||
// away, so there is nothing left to release.
|
||||
|
||||
@@ -0,0 +1,252 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"database/sql"
|
||||
"errors"
|
||||
"fmt"
|
||||
"time"
|
||||
)
|
||||
|
||||
// The server_schedules store (ServerSchedules, migration 0036).
|
||||
|
||||
// scheduleColumns is the SELECT list scanSchedule reads, over server_schedules
|
||||
// aliased s.
|
||||
const scheduleColumns = `s.id, s.server_name, COALESCE(s.owner_id, ''), s.label, s.action, s.command,
|
||||
s.every_minutes, s.minute_of_day, s.weekdays, s.timezone, s.warn_minutes, s.enabled,
|
||||
s.next_run_at, s.warned_for, s.run_state, s.run_resume, s.run_step_at,
|
||||
s.last_run_at, s.last_result, s.last_detail, s.created_by, s.created_at`
|
||||
|
||||
type rowScanner interface{ Scan(dest ...any) error }
|
||||
|
||||
func scanSchedule(row rowScanner, extra ...any) (*Schedule, error) {
|
||||
var s Schedule
|
||||
var next, warned, step, last sql.NullTime
|
||||
dest := []any{&s.ID, &s.Server, &s.OwnerID, &s.Label, &s.Action, &s.Command,
|
||||
&s.EveryMinutes, &s.MinuteOfDay, &s.Weekdays, &s.Timezone, &s.WarnMinutes, &s.Enabled,
|
||||
&next, &warned, &s.RunState, &s.RunResume, &step,
|
||||
&last, &s.LastResult, &s.LastDetail, &s.CreatedBy, &s.CreatedAt}
|
||||
if err := row.Scan(append(dest, extra...)...); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
s.NextRunAt, s.WarnedFor, s.RunStepAt, s.LastRunAt = nullTimePtr(next), nullTimePtr(warned), nullTimePtr(step), nullTimePtr(last)
|
||||
return &s, nil
|
||||
}
|
||||
|
||||
func nullTimePtr(t sql.NullTime) *time.Time {
|
||||
if !t.Valid {
|
||||
return nil
|
||||
}
|
||||
return &t.Time
|
||||
}
|
||||
|
||||
// ListSchedules lists a server's schedules, oldest first.
|
||||
func (p *PGRepo) ListSchedules(ctx context.Context, server string) ([]Schedule, error) {
|
||||
rows, err := p.db.QueryContext(ctx,
|
||||
`SELECT `+scheduleColumns+` FROM server_schedules s WHERE s.server_name = $1 ORDER BY s.id`, server)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []Schedule
|
||||
for rows.Next() {
|
||||
s, err := scanSchedule(rows)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, *s)
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// GetSchedule reads one schedule of server, or ErrNotFound.
|
||||
func (p *PGRepo) GetSchedule(ctx context.Context, server string, id int64) (*Schedule, error) {
|
||||
s, err := scanSchedule(p.db.QueryRowContext(ctx,
|
||||
`SELECT `+scheduleColumns+` FROM server_schedules s WHERE s.id = $1 AND s.server_name = $2`, id, server))
|
||||
if errors.Is(err, sql.ErrNoRows) {
|
||||
return nil, ErrNotFound
|
||||
}
|
||||
return s, err
|
||||
}
|
||||
|
||||
// CreateSchedule inserts s under the server row's lock, so two saves racing
|
||||
// for the last free place cannot both take it.
|
||||
func (p *PGRepo) CreateSchedule(ctx context.Context, s *Schedule, limit int) error {
|
||||
tx, err := p.db.BeginTx(ctx, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer tx.Rollback() //nolint:errcheck // no-op after commit
|
||||
|
||||
var one int
|
||||
if err := tx.QueryRowContext(ctx,
|
||||
`SELECT 1 FROM servers WHERE name = $1 AND deleted_at IS NULL FOR UPDATE`, s.Server).Scan(&one); err != nil {
|
||||
if errors.Is(err, sql.ErrNoRows) {
|
||||
return ErrNotFound
|
||||
}
|
||||
return err
|
||||
}
|
||||
var n int
|
||||
if err := tx.QueryRowContext(ctx,
|
||||
`SELECT count(*) FROM server_schedules WHERE server_name = $1`, s.Server).Scan(&n); err != nil {
|
||||
return err
|
||||
}
|
||||
if n >= limit {
|
||||
return ErrScheduleLimit
|
||||
}
|
||||
if err := tx.QueryRowContext(ctx,
|
||||
`INSERT INTO server_schedules (server_name, owner_id, label, action, command, every_minutes,
|
||||
minute_of_day, weekdays, timezone, warn_minutes, enabled, next_run_at, created_by)
|
||||
VALUES ($1, NULLIF($2, ''), $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13)
|
||||
RETURNING id, created_at`,
|
||||
s.Server, s.OwnerID, s.Label, s.Action, s.Command, s.EveryMinutes,
|
||||
s.MinuteOfDay, s.Weekdays, s.Timezone, s.WarnMinutes, s.Enabled, s.NextRunAt, s.CreatedBy,
|
||||
).Scan(&s.ID, &s.CreatedAt); err != nil {
|
||||
return err
|
||||
}
|
||||
return tx.Commit()
|
||||
}
|
||||
|
||||
// scheduleMissing tells why a write guarded on an idle run_state touched no
|
||||
// row: the schedule is gone, or it is running.
|
||||
func (p *PGRepo) scheduleMissing(ctx context.Context, server string, id int64) error {
|
||||
var state string
|
||||
err := p.db.QueryRowContext(ctx,
|
||||
`SELECT run_state FROM server_schedules WHERE id = $1 AND server_name = $2`, id, server).Scan(&state)
|
||||
switch {
|
||||
case errors.Is(err, sql.ErrNoRows):
|
||||
return ErrNotFound
|
||||
case err != nil:
|
||||
return err
|
||||
case state != "":
|
||||
return ErrScheduleRunning
|
||||
}
|
||||
return fmt.Errorf("schedule %d of %s did not change", id, server)
|
||||
}
|
||||
|
||||
// UpdateSchedule writes s's settings, owner and next run, unless it is running.
|
||||
func (p *PGRepo) UpdateSchedule(ctx context.Context, s *Schedule) error {
|
||||
res, err := p.db.ExecContext(ctx,
|
||||
`UPDATE server_schedules SET owner_id = NULLIF($3, ''), label = $4, action = $5, command = $6,
|
||||
every_minutes = $7, minute_of_day = $8, weekdays = $9, timezone = $10, warn_minutes = $11,
|
||||
enabled = $12, next_run_at = $13, warned_for = NULL, updated_at = now()
|
||||
WHERE id = $1 AND server_name = $2 AND run_state = ''`,
|
||||
s.ID, s.Server, s.OwnerID, s.Label, s.Action, s.Command,
|
||||
s.EveryMinutes, s.MinuteOfDay, s.Weekdays, s.Timezone, s.WarnMinutes,
|
||||
s.Enabled, s.NextRunAt)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if n, err := res.RowsAffected(); err != nil || n == 0 {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return p.scheduleMissing(ctx, s.Server, s.ID)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// DeleteSchedule removes a schedule, unless it is running.
|
||||
func (p *PGRepo) DeleteSchedule(ctx context.Context, server string, id int64) error {
|
||||
res, err := p.db.ExecContext(ctx,
|
||||
`DELETE FROM server_schedules WHERE id = $1 AND server_name = $2 AND run_state = ''`, id, server)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if n, err := res.RowsAffected(); err != nil || n == 0 {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return p.scheduleMissing(ctx, server, id)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// DueSchedules lists the schedules the runner has to look at, runs in progress
|
||||
// first, then by due time.
|
||||
func (p *PGRepo) DueSchedules(ctx context.Context, horizon time.Time) ([]DueSchedule, error) {
|
||||
rows, err := p.db.QueryContext(ctx,
|
||||
`SELECT `+scheduleColumns+`, COALESCE(v.owner_id, '')
|
||||
FROM server_schedules s JOIN servers v ON v.name = s.server_name
|
||||
WHERE v.deleted_at IS NULL AND (s.run_state <> '' OR (s.enabled AND s.next_run_at <= $1))
|
||||
ORDER BY s.run_state = '', s.next_run_at, s.id`, horizon)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer rows.Close()
|
||||
var out []DueSchedule
|
||||
for rows.Next() {
|
||||
var owner string
|
||||
s, err := scanSchedule(rows, &owner)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out = append(out, DueSchedule{Schedule: *s, ServerOwner: owner})
|
||||
}
|
||||
return out, rows.Err()
|
||||
}
|
||||
|
||||
// execApplied runs a compare-and-set write and reports whether it matched.
|
||||
func (p *PGRepo) execApplied(ctx context.Context, query string, args ...any) (bool, error) {
|
||||
res, err := p.db.ExecContext(ctx, query, args...)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
n, err := res.RowsAffected()
|
||||
return n > 0, err
|
||||
}
|
||||
|
||||
// ClaimScheduleRun starts a run of a schedule without one.
|
||||
func (p *PGRepo) ClaimScheduleRun(ctx context.Context, id int64, due, next *time.Time, now time.Time) (bool, error) {
|
||||
if due == nil {
|
||||
return p.execApplied(ctx,
|
||||
`UPDATE server_schedules SET run_state = 'claimed', run_resume = false, run_step_at = $2,
|
||||
last_run_at = $2, last_result = '', last_detail = ''
|
||||
WHERE id = $1 AND run_state = ''`, id, now)
|
||||
}
|
||||
return p.execApplied(ctx,
|
||||
`UPDATE server_schedules SET run_state = 'claimed', run_resume = false, run_step_at = $4,
|
||||
last_run_at = $4, last_result = '', last_detail = '', next_run_at = $3, warned_for = NULL
|
||||
WHERE id = $1 AND run_state = '' AND enabled AND next_run_at = $2`, id, *due, next, now)
|
||||
}
|
||||
|
||||
// AdvanceScheduleRun moves a run on to its next step.
|
||||
func (p *PGRepo) AdvanceScheduleRun(ctx context.Context, id int64, from, to string, resume bool, result, detail string, now time.Time) (bool, error) {
|
||||
return p.execApplied(ctx,
|
||||
`UPDATE server_schedules SET run_state = $3, run_resume = $4, run_step_at = $7,
|
||||
last_result = CASE WHEN $5::text = '' THEN last_result ELSE $5::text END,
|
||||
last_detail = CASE WHEN $5::text = '' THEN last_detail ELSE $6::text END
|
||||
WHERE id = $1 AND run_state = $2`, id, from, to, resume, result, detail, now)
|
||||
}
|
||||
|
||||
// FinishScheduleRun ends a run with its outcome.
|
||||
func (p *PGRepo) FinishScheduleRun(ctx context.Context, id int64, from, result, detail string) (bool, error) {
|
||||
return p.execApplied(ctx,
|
||||
`UPDATE server_schedules SET run_state = '', run_resume = false, run_step_at = NULL,
|
||||
last_result = $3, last_detail = $4
|
||||
WHERE id = $1 AND run_state = $2::text AND $2::text <> ''`, id, from, result, detail)
|
||||
}
|
||||
|
||||
// MissScheduleRun records the run due then as missed and moves on to next.
|
||||
func (p *PGRepo) MissScheduleRun(ctx context.Context, id int64, due, next time.Time, detail string) (bool, error) {
|
||||
return p.execApplied(ctx,
|
||||
`UPDATE server_schedules SET next_run_at = $3, warned_for = NULL,
|
||||
last_run_at = $2, last_result = 'missed', last_detail = $4
|
||||
WHERE id = $1 AND run_state = '' AND enabled AND next_run_at = $2`, id, due, next, detail)
|
||||
}
|
||||
|
||||
// WarnScheduleRun marks the run due then as warned about.
|
||||
func (p *PGRepo) WarnScheduleRun(ctx context.Context, id int64, due time.Time) (bool, error) {
|
||||
return p.execApplied(ctx,
|
||||
`UPDATE server_schedules SET warned_for = $2
|
||||
WHERE id = $1 AND run_state = '' AND enabled AND next_run_at = $2
|
||||
AND warned_for IS DISTINCT FROM $2`, id, due)
|
||||
}
|
||||
|
||||
// DisableSchedule turns off an enabled, idle schedule and records why.
|
||||
func (p *PGRepo) DisableSchedule(ctx context.Context, id int64, detail string) (bool, error) {
|
||||
return p.execApplied(ctx,
|
||||
`UPDATE server_schedules SET enabled = false, next_run_at = NULL, warned_for = NULL,
|
||||
last_result = 'skipped', last_detail = $2, updated_at = now()
|
||||
WHERE id = $1 AND run_state = '' AND enabled`, id, detail)
|
||||
}
|
||||
@@ -0,0 +1,486 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"math"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
)
|
||||
|
||||
// The schedule runner. cmd/felis calls RunSchedules every few seconds; each
|
||||
// call warns the players about the runs coming up, fires the runs that are
|
||||
// due and moves every run in progress one step on. All of its state is in
|
||||
// server_schedules, so a felis-api restart picks a restart or backup up where
|
||||
// it was.
|
||||
//
|
||||
// A command, stop or start is one step. A restart stops the server (runStopping)
|
||||
// and starts it once it is down (runStarting). A backup of a running server
|
||||
// stops it, takes a backup once the world volume is free (runBackingUp), waits
|
||||
// for the backup Job and starts the server again; a backup of a stopped server
|
||||
// leaves it stopped. Each step gives up after its own wait, and a run that took
|
||||
// the server down tries to bring it back when it gives up.
|
||||
|
||||
// Run steps (Schedule.RunState).
|
||||
const (
|
||||
runClaimed = "claimed"
|
||||
runStopping = "stopping"
|
||||
runBackingUp = "backing_up"
|
||||
runStarting = "starting"
|
||||
)
|
||||
|
||||
const (
|
||||
// scheduleMissGrace is how late a run may still start: felis-api back from
|
||||
// a short restart catches up, and a run hours late is dropped as missed.
|
||||
scheduleMissGrace = 10 * time.Minute
|
||||
// claimedStale is how long a run may sit in its first step before it is
|
||||
// taken for one whose felis-api stopped mid-step. The first step is a few
|
||||
// API calls and at most one RCON round trip.
|
||||
claimedStale = 2 * time.Minute
|
||||
// stopWait is how long a run waits for the server to stop and its world
|
||||
// volume to come free. A pod saving a big world takes a while.
|
||||
stopWait = 15 * time.Minute
|
||||
// backupWait is how long a run waits for its backup Job: past the Job's own
|
||||
// deadline ([archive] backup Job deadline, 30m by default).
|
||||
backupWait = 45 * time.Minute
|
||||
// startWait is how long a run keeps trying to start the server again while
|
||||
// the cluster is at its running cap or the world volume is still busy.
|
||||
startWait = 15 * time.Minute
|
||||
// backupJobSkew is how much earlier than the step the backup Job's creation
|
||||
// stamp may read: felis-api's clock and the API server's differ a little.
|
||||
backupJobSkew = 2 * time.Minute
|
||||
// scheduleWarnHorizon is the longest warning lead time, in minutes.
|
||||
scheduleWarnHorizon = 30
|
||||
)
|
||||
|
||||
// Audit identity of a run. Each run leaves one schedule.run row when it ends.
|
||||
const (
|
||||
scheduleActor = "scheduler"
|
||||
scheduleRunAction = "schedule.run"
|
||||
)
|
||||
|
||||
// RunSchedules warns about, fires and advances the schedules once. A schedule
|
||||
// that fails is logged and left for the next call; it never holds up the rest.
|
||||
func (a *API) RunSchedules(ctx context.Context) error {
|
||||
if a.Schedules == nil {
|
||||
return nil
|
||||
}
|
||||
now := a.now()
|
||||
due, err := a.Schedules.DueSchedules(ctx, now.Add(scheduleWarnHorizon*time.Minute))
|
||||
if err != nil {
|
||||
return fmt.Errorf("list the due schedules: %w", err)
|
||||
}
|
||||
for i := range due {
|
||||
d := &due[i]
|
||||
if err := a.tickSchedule(ctx, d, now); err != nil {
|
||||
log.Printf("api: schedule %d of %s: %v", d.ID, d.Server, err)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// tickSchedule does what one due schedule needs now.
|
||||
func (a *API) tickSchedule(ctx context.Context, d *DueSchedule, now time.Time) error {
|
||||
s := &d.Schedule
|
||||
if s.RunState != "" {
|
||||
return a.advanceRun(ctx, s, now)
|
||||
}
|
||||
// Somebody else's server now: their commands must not run on it. Saving
|
||||
// the schedule again (PUT) hands it to the new owner.
|
||||
if d.ServerOwner != s.OwnerID {
|
||||
_, err := a.Schedules.DisableSchedule(ctx, s.ID,
|
||||
"the server has a new owner since this schedule was saved; save it again to use it")
|
||||
return err
|
||||
}
|
||||
due := *s.NextRunAt // set on every enabled schedule DueSchedules returns idle
|
||||
if now.Before(due) {
|
||||
return a.warnRun(ctx, s, due, now)
|
||||
}
|
||||
next := s.nextRun(now)
|
||||
if now.Sub(due) > scheduleMissGrace {
|
||||
_, err := a.Schedules.MissScheduleRun(ctx, s.ID, due, next,
|
||||
"felis-api was not running at the scheduled time")
|
||||
return err
|
||||
}
|
||||
ok, err := a.Schedules.ClaimScheduleRun(ctx, s.ID, &due, &next, now)
|
||||
if err != nil || !ok {
|
||||
return err
|
||||
}
|
||||
return a.beginRun(ctx, s)
|
||||
}
|
||||
|
||||
// warnRun tells the players on the server that a restart, stop or backup is
|
||||
// coming, once per run, when its warning time has come.
|
||||
func (a *API) warnRun(ctx context.Context, s *Schedule, due, now time.Time) error {
|
||||
lead := time.Duration(s.WarnMinutes) * time.Minute
|
||||
if now.Before(due.Add(-lead)) || a.Console == nil {
|
||||
return nil // no warning, or not yet
|
||||
}
|
||||
info, err := a.Cluster.GetServer(ctx, s.Server)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if !info.Ready {
|
||||
return nil
|
||||
}
|
||||
ok, err := a.Schedules.WarnScheduleRun(ctx, s.ID, due)
|
||||
if err != nil || !ok {
|
||||
return err
|
||||
}
|
||||
// Warned late (felis-api was restarting at the warning time): say how long
|
||||
// is really left.
|
||||
minutes := int(math.Ceil(due.Sub(now).Minutes()))
|
||||
if _, err := a.Console.RunCommand(ctx, s.Server, "say "+scheduleWarning(s.Action, minutes)); err != nil {
|
||||
return fmt.Errorf("warn the players: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// scheduleWarning is the in-game line announcing a run minutes ahead, in both
|
||||
// panel languages.
|
||||
func scheduleWarning(action string, minutes int) string {
|
||||
switch action {
|
||||
case ScheduleRestart:
|
||||
return fmt.Sprintf("[Felis] 服务器将在 %d 分钟后重启 / Server restarts in %d min", minutes, minutes)
|
||||
case ScheduleStop:
|
||||
return fmt.Sprintf("[Felis] 服务器将在 %d 分钟后关闭 / Server stops in %d min", minutes, minutes)
|
||||
default:
|
||||
return fmt.Sprintf("[Felis] 服务器将在 %d 分钟后暂停做备份,完成后自动恢复 / Server pauses for a backup in %d min and comes back after", minutes, minutes)
|
||||
}
|
||||
}
|
||||
|
||||
// beginRun takes the first step of a run just claimed (runClaimed).
|
||||
func (a *API) beginRun(ctx context.Context, s *Schedule) error {
|
||||
finish := func(result, detail string) error { return a.finishRun(ctx, s, runClaimed, result, detail) }
|
||||
info, err := a.Cluster.GetServer(ctx, s.Server)
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
return finish(ScheduleSkipped, "the server no longer exists")
|
||||
}
|
||||
if err != nil {
|
||||
return finish(ScheduleFailed, "could not read the server: "+err.Error())
|
||||
}
|
||||
running := info.DesiredState == string(v1alpha1.DesiredRunning)
|
||||
|
||||
switch s.Action {
|
||||
case ScheduleCommand:
|
||||
if !info.Ready {
|
||||
return finish(ScheduleSkipped, "the server was not running")
|
||||
}
|
||||
if a.Console == nil {
|
||||
return finish(ScheduleFailed, "the console is not configured")
|
||||
}
|
||||
out, err := a.Console.RunCommand(ctx, s.Server, s.Command)
|
||||
if errors.Is(err, ErrConsoleUnavailable) {
|
||||
return finish(ScheduleFailed, "the server console could not be reached")
|
||||
}
|
||||
if err != nil {
|
||||
return finish(ScheduleFailed, "the command failed: "+err.Error())
|
||||
}
|
||||
return finish(ScheduleOK, stripFormatting(out))
|
||||
|
||||
case ScheduleStop:
|
||||
if !running {
|
||||
return finish(ScheduleSkipped, "the server was already stopped")
|
||||
}
|
||||
if err := a.Cluster.SetDesiredState(ctx, s.Server, v1alpha1.DesiredStopped); err != nil {
|
||||
return finish(ScheduleFailed, "could not stop the server: "+err.Error())
|
||||
}
|
||||
return finish(ScheduleOK, "")
|
||||
|
||||
case ScheduleStart:
|
||||
if running && info.Phase != string(v1alpha1.PhaseFailed) {
|
||||
return finish(ScheduleSkipped, "the server was already running")
|
||||
}
|
||||
why, _, err := a.startScheduled(ctx, s.Server)
|
||||
if err != nil {
|
||||
return finish(ScheduleFailed, "could not start the server: "+err.Error())
|
||||
}
|
||||
if why != "" {
|
||||
return finish(ScheduleSkipped, why)
|
||||
}
|
||||
return finish(ScheduleOK, "")
|
||||
|
||||
case ScheduleRestart:
|
||||
if !running {
|
||||
return finish(ScheduleSkipped, "the server was not running")
|
||||
}
|
||||
return a.stopForRun(ctx, s, true)
|
||||
|
||||
case ScheduleBackup:
|
||||
if _, ok := a.Backuper.(ScheduledBackuper); !ok {
|
||||
return finish(ScheduleFailed, "backups are not configured")
|
||||
}
|
||||
if limit := a.BackupStoreCap; limit > 0 {
|
||||
used, err := a.Repo.BackupStoreBytes(ctx)
|
||||
if err != nil {
|
||||
return finish(ScheduleFailed, "could not read the backup store size: "+err.Error())
|
||||
}
|
||||
if used >= limit {
|
||||
return finish(ScheduleFailed, "the backup store is full; ask an administrator to free space")
|
||||
}
|
||||
}
|
||||
exists, err := a.Cluster.WorldVolumeExists(ctx, s.Server)
|
||||
if err != nil {
|
||||
return finish(ScheduleFailed, "could not look up the world volume: "+err.Error())
|
||||
}
|
||||
if !exists {
|
||||
return finish(ScheduleSkipped, "the server has no world yet")
|
||||
}
|
||||
if running {
|
||||
return a.stopForRun(ctx, s, true)
|
||||
}
|
||||
// Already stopped: back it up as soon as the world volume is free, and
|
||||
// leave it stopped afterwards.
|
||||
if ok, err := a.Schedules.AdvanceScheduleRun(ctx, s.ID, runClaimed, runStopping, false, "", "", a.now()); err != nil || !ok {
|
||||
return err
|
||||
}
|
||||
s.RunState, s.RunResume, s.RunStepAt = runStopping, false, ptrTime(a.now())
|
||||
return a.advanceRun(ctx, s, a.now())
|
||||
}
|
||||
return finish(ScheduleFailed, "unknown action "+s.Action)
|
||||
}
|
||||
|
||||
// stopForRun records the stop step and then stops the server, in that order: a
|
||||
// felis-api that dies between the two leaves a running server in runStopping,
|
||||
// which the next call reads as someone having started it, and nothing is lost.
|
||||
func (a *API) stopForRun(ctx context.Context, s *Schedule, resume bool) error {
|
||||
now := a.now()
|
||||
if ok, err := a.Schedules.AdvanceScheduleRun(ctx, s.ID, runClaimed, runStopping, resume, "", "", now); err != nil || !ok {
|
||||
return err
|
||||
}
|
||||
if err := a.Cluster.SetDesiredState(ctx, s.Server, v1alpha1.DesiredStopped); err != nil {
|
||||
return a.finishRun(ctx, s, runStopping, ScheduleFailed, "could not stop the server: "+err.Error())
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// advanceRun moves a run in progress on by one step, or leaves it waiting.
|
||||
func (a *API) advanceRun(ctx context.Context, s *Schedule, now time.Time) error {
|
||||
waited := now.Sub(*s.RunStepAt)
|
||||
switch s.RunState {
|
||||
case runClaimed:
|
||||
if waited < claimedStale {
|
||||
return nil // its first step is running right now
|
||||
}
|
||||
return a.finishRun(ctx, s, runClaimed, ScheduleFailed, "felis-api stopped in the middle of this run")
|
||||
case runStopping:
|
||||
return a.advanceStopping(ctx, s, waited)
|
||||
case runBackingUp:
|
||||
return a.advanceBackingUp(ctx, s, waited)
|
||||
case runStarting:
|
||||
return a.advanceStarting(ctx, s, waited)
|
||||
}
|
||||
return a.finishRun(ctx, s, s.RunState, ScheduleFailed, "unknown run step "+s.RunState)
|
||||
}
|
||||
|
||||
// advanceStopping waits for the server to go down. A restart then starts it;
|
||||
// a backup takes the world volume and starts the backup Job.
|
||||
func (a *API) advanceStopping(ctx context.Context, s *Schedule, waited time.Duration) error {
|
||||
info, err := a.Cluster.GetServer(ctx, s.Server)
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
return a.finishRun(ctx, s, runStopping, ScheduleSkipped, "the server no longer exists")
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// Only this run stops the server; wanted running again means a person (or
|
||||
// a player's join) started it meanwhile, and their start stands.
|
||||
if info.DesiredState == string(v1alpha1.DesiredRunning) {
|
||||
return a.finishRun(ctx, s, runStopping, ScheduleSkipped, "someone started the server before the run finished")
|
||||
}
|
||||
giveUp := func(detail string) error {
|
||||
if waited < stopWait {
|
||||
return nil
|
||||
}
|
||||
if s.RunResume {
|
||||
if err := a.Cluster.SetDesiredState(ctx, s.Server, v1alpha1.DesiredRunning); err == nil {
|
||||
detail += "; it was started again"
|
||||
}
|
||||
}
|
||||
return a.finishRun(ctx, s, runStopping, ScheduleFailed, detail)
|
||||
}
|
||||
|
||||
if s.Action == ScheduleRestart {
|
||||
if info.Phase != string(v1alpha1.PhaseStopped) {
|
||||
return giveUp("the server did not stop within 15 minutes")
|
||||
}
|
||||
if ok, err := a.Schedules.AdvanceScheduleRun(ctx, s.ID, runStopping, runStarting, s.RunResume, "", "", a.now()); err != nil || !ok {
|
||||
return err
|
||||
}
|
||||
s.RunState, s.RunStepAt = runStarting, ptrTime(a.now())
|
||||
return a.advanceStarting(ctx, s, 0)
|
||||
}
|
||||
|
||||
b, ok := a.Backuper.(ScheduledBackuper)
|
||||
if !ok {
|
||||
return a.finishRun(ctx, s, runStopping, ScheduleFailed, "backups are not configured")
|
||||
}
|
||||
switch err := a.Cluster.AcquireMaintenance(ctx, s.Server, maintenance.KindBackup); {
|
||||
case errors.Is(err, ErrNotStopped):
|
||||
return giveUp("the server did not stop within 15 minutes")
|
||||
case errors.Is(err, ErrMaintenanceInProgress):
|
||||
return giveUp("another operation kept the world busy for 15 minutes")
|
||||
case errors.Is(err, ErrNotFound):
|
||||
return a.finishRun(ctx, s, runStopping, ScheduleSkipped, "the server no longer exists")
|
||||
case err != nil:
|
||||
return err
|
||||
}
|
||||
err = b.BackupScheduled(ctx, s.Server, s.OwnerID)
|
||||
// Once the Job exists it holds the world; the lock only covered the gap.
|
||||
if rerr := a.Cluster.ReleaseMaintenance(context.WithoutCancel(ctx), s.Server); rerr != nil {
|
||||
log.Printf("api: release the maintenance lock on %s: %v (it lapses after %s)", s.Server, rerr, maintenance.Grace)
|
||||
}
|
||||
if err != nil {
|
||||
detail := "could not start the backup: " + err.Error()
|
||||
if s.RunResume {
|
||||
if serr := a.Cluster.SetDesiredState(ctx, s.Server, v1alpha1.DesiredRunning); serr == nil {
|
||||
detail += "; the server was started again"
|
||||
}
|
||||
}
|
||||
return a.finishRun(ctx, s, runStopping, ScheduleFailed, detail)
|
||||
}
|
||||
log.Printf("api: schedule %d started a backup of %s", s.ID, s.Server)
|
||||
_, err = a.Schedules.AdvanceScheduleRun(ctx, s.ID, runStopping, runBackingUp, s.RunResume, "", "", a.now())
|
||||
return err
|
||||
}
|
||||
|
||||
// advanceBackingUp waits for the backup Job and records how it ended; a run
|
||||
// that stopped a running server then starts it again.
|
||||
func (a *API) advanceBackingUp(ctx context.Context, s *Schedule, waited time.Duration) error {
|
||||
result, detail := ScheduleOK, ""
|
||||
if a.JobStatus != nil {
|
||||
jobs, err := a.JobStatus.LatestJobs(ctx, s.Server)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
var job *AsyncJob
|
||||
for i := range jobs {
|
||||
j := &jobs[i]
|
||||
if j.Scheduled && !j.StartedAt.Before(s.RunStepAt.Add(-backupJobSkew)) {
|
||||
job = j
|
||||
break // newest first
|
||||
}
|
||||
}
|
||||
switch {
|
||||
case job == nil && waited < backupJobSkew:
|
||||
return nil // not listed yet
|
||||
case job == nil:
|
||||
result, detail = ScheduleFailed, "the backup Job is gone before it could be checked; see the Backups page"
|
||||
case job.State == "running" && waited < backupWait:
|
||||
return nil
|
||||
case job.State == "running":
|
||||
result, detail = ScheduleFailed, "the backup did not finish within 45 minutes"
|
||||
case job.State == "failed":
|
||||
result, detail = ScheduleFailed, strings.TrimSpace("the backup failed: "+job.Message)
|
||||
}
|
||||
}
|
||||
if !s.RunResume {
|
||||
return a.finishRun(ctx, s, runBackingUp, result, detail)
|
||||
}
|
||||
if ok, err := a.Schedules.AdvanceScheduleRun(ctx, s.ID, runBackingUp, runStarting, true, result, detail, a.now()); err != nil || !ok {
|
||||
return err
|
||||
}
|
||||
s.RunState, s.RunStepAt, s.LastResult, s.LastDetail = runStarting, ptrTime(a.now()), result, detail
|
||||
return a.advanceStarting(ctx, s, 0)
|
||||
}
|
||||
|
||||
// advanceStarting brings the server back up after a restart's stop or a
|
||||
// backup. The outcome recorded so far (the backup's) is kept when it starts.
|
||||
func (a *API) advanceStarting(ctx context.Context, s *Schedule, waited time.Duration) error {
|
||||
why, retry, err := a.startScheduled(ctx, s.Server)
|
||||
if errors.Is(err, ErrNotFound) {
|
||||
return a.finishRun(ctx, s, runStarting, ScheduleSkipped, "the server no longer exists")
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if why == "" {
|
||||
result, detail := s.LastResult, s.LastDetail
|
||||
if result == "" {
|
||||
result = ScheduleOK
|
||||
}
|
||||
return a.finishRun(ctx, s, runStarting, result, detail)
|
||||
}
|
||||
if retry && waited < startWait {
|
||||
return nil
|
||||
}
|
||||
detail := "could not start the server again: " + why
|
||||
if s.LastResult == ScheduleFailed {
|
||||
detail = s.LastDetail + "; " + detail
|
||||
}
|
||||
return a.finishRun(ctx, s, runStarting, ScheduleFailed, detail)
|
||||
}
|
||||
|
||||
// startScheduled starts a server for a run: the wake path without the
|
||||
// per-player cooldown. why names what kept it stopped ("" once it is wanted
|
||||
// running), and retry says whether that may clear by itself.
|
||||
func (a *API) startScheduled(ctx context.Context, name string) (why string, retry bool, err error) {
|
||||
rec, err := a.Repo.ServerByName(ctx, name)
|
||||
if err != nil {
|
||||
return "", false, err
|
||||
}
|
||||
if rec.Retire != nil {
|
||||
return "the server is being given up or deleted", false, nil
|
||||
}
|
||||
info, err := a.Cluster.GetServer(ctx, name)
|
||||
if err != nil {
|
||||
return "", false, err
|
||||
}
|
||||
ok, err := a.withinRunningCap(ctx, info)
|
||||
if err != nil {
|
||||
return "", false, err
|
||||
}
|
||||
if !ok {
|
||||
return "the cluster is at its running-server cap", true, nil
|
||||
}
|
||||
// A start that Failed is started over, as a person's start does (handleStart).
|
||||
if info.Phase == string(v1alpha1.PhaseFailed) && info.DesiredState == string(v1alpha1.DesiredRunning) {
|
||||
err = a.Cluster.RetryStart(ctx, name)
|
||||
} else {
|
||||
err = a.Cluster.SetDesiredState(ctx, name, v1alpha1.DesiredRunning)
|
||||
}
|
||||
var busy *MaintenanceBusyError
|
||||
if errors.As(err, &busy) {
|
||||
return "the world is busy with " + maintenanceLabel(busy.Kind), true, nil
|
||||
}
|
||||
return "", false, err
|
||||
}
|
||||
|
||||
// finishRun ends a run at step from and audits it.
|
||||
func (a *API) finishRun(ctx context.Context, s *Schedule, from, result, detail string) error {
|
||||
detail = truncateUTF8(detail, maxScheduleDetail)
|
||||
ok, err := a.Schedules.FinishScheduleRun(ctx, s.ID, from, result, detail)
|
||||
if err != nil || !ok {
|
||||
return err
|
||||
}
|
||||
a.writeAudit(ctx, AuditEntry{
|
||||
Actor: scheduleActor, Source: scheduleActor, Action: scheduleRunAction, ServerName: s.Server,
|
||||
Payload: auditPayload(map[string]any{"schedule": s.ID, "action": s.Action, "result": result, "detail": detail}),
|
||||
})
|
||||
return nil
|
||||
}
|
||||
|
||||
// stripFormatting drops Minecraft's section-sign formatting codes from a
|
||||
// console reply and trims it.
|
||||
func stripFormatting(s string) string {
|
||||
var b strings.Builder
|
||||
skip := false
|
||||
for _, c := range s {
|
||||
switch {
|
||||
case skip:
|
||||
skip = false
|
||||
case c == '§':
|
||||
skip = true
|
||||
default:
|
||||
b.WriteRune(c)
|
||||
}
|
||||
}
|
||||
return strings.TrimSpace(b.String())
|
||||
}
|
||||
|
||||
func ptrTime(t time.Time) *time.Time { return &t }
|
||||
@@ -0,0 +1,696 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
"felis.lolicon.best/internal/maintenance"
|
||||
)
|
||||
|
||||
// outcome asserts how a schedule's last run ended and that no run is left.
|
||||
func (r *schedRig) outcome(t *testing.T, id int64, result, detail string) *Schedule {
|
||||
t.Helper()
|
||||
s := r.st.row(t, id)
|
||||
if s.RunState != "" || s.LastResult != result || s.LastDetail != detail {
|
||||
t.Fatalf("run state %q, result %q %q; want done with %q %q", s.RunState, s.LastResult, s.LastDetail, result, detail)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// step asserts a run waits at step.
|
||||
func (r *schedRig) step(t *testing.T, id int64, step string) {
|
||||
t.Helper()
|
||||
if s := r.st.row(t, id); s.RunState != step {
|
||||
t.Fatalf("run state %q (result %q %q), want %q", s.RunState, s.LastResult, s.LastDetail, step)
|
||||
}
|
||||
}
|
||||
|
||||
// runAudits lists the schedule.run rows as result:detail.
|
||||
func (r *schedRig) runAudits() []string {
|
||||
var out []string
|
||||
for _, e := range r.repo.audits {
|
||||
if e.Action == scheduleRunAction && e.Actor == scheduleActor && e.Source == scheduleActor {
|
||||
out = append(out, string(e.Payload))
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func TestScheduleRunnerFiring(t *testing.T) {
|
||||
t.Run("a due command runs, moves on a day and is audited", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.con.reply = " §6There are §c2§6 players online "
|
||||
id := r.schedule(ScheduleCommand, schedT0, func(s *Schedule) { s.Command = "list" })
|
||||
r.clock = schedT0.Add(20 * time.Second)
|
||||
r.tick(t)
|
||||
s := r.outcome(t, id, ScheduleOK, "There are 2 players online")
|
||||
if !sameTime(s.NextRunAt, schedT0.AddDate(0, 0, 1)) || !sameTime(s.LastRunAt, r.clock) || r.con.gotCommand != "list" {
|
||||
t.Fatalf("next %v, last %v, console %q", s.NextRunAt, s.LastRunAt, r.con.gotCommand)
|
||||
}
|
||||
if got := r.runAudits(); len(got) != 1 ||
|
||||
got[0] != `{"action":"command","detail":"There are 2 players online","result":"ok","schedule":1}` {
|
||||
t.Fatalf("run audits %v", got)
|
||||
}
|
||||
r.tick(t) // not due again
|
||||
if r.con.calls != 1 || len(r.runAudits()) != 1 {
|
||||
t.Fatalf("ran again: %d console calls", r.con.calls)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("not due yet: nothing happens", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleCommand, schedT0.Add(time.Second))
|
||||
r.tick(t)
|
||||
if s := r.st.row(t, id); r.con.calls != 0 || s.LastRunAt != nil || !sameTime(s.NextRunAt, schedT0.Add(time.Second)) {
|
||||
t.Fatalf("ran early: %+v", s)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a long reply is cut at 500 bytes on a rune boundary", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.con.reply = "x" + strings.Repeat("猫", 200)
|
||||
id := r.schedule(ScheduleCommand, schedT0)
|
||||
r.tick(t)
|
||||
if s := r.st.row(t, id); s.LastDetail != "x"+strings.Repeat("猫", 166) {
|
||||
t.Fatalf("detail is %d bytes: %q", len(s.LastDetail), s.LastDetail)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a command on a stopped server is skipped", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.stopped()
|
||||
id := r.schedule(ScheduleCommand, schedT0)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleSkipped, "the server was not running")
|
||||
if r.con.calls != 0 {
|
||||
t.Fatal("dialed a stopped server")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("an unreachable console fails the run", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.con.err = ErrConsoleUnavailable
|
||||
id := r.schedule(ScheduleCommand, schedT0)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "the server console could not be reached")
|
||||
})
|
||||
|
||||
t.Run("ten minutes late still runs; later is missed", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
late := r.schedule(ScheduleCommand, schedT0.Add(-scheduleMissGrace))
|
||||
missed := r.schedule(ScheduleCommand, schedT0.Add(-scheduleMissGrace-time.Second))
|
||||
r.tick(t)
|
||||
r.outcome(t, late, ScheduleOK, "")
|
||||
s := r.outcome(t, missed, ScheduleMissed, "felis-api was not running at the scheduled time")
|
||||
if r.con.calls != 1 || !sameTime(s.LastRunAt, schedT0.Add(-scheduleMissGrace-time.Second)) ||
|
||||
!sameTime(s.NextRunAt, time.Date(2026, 9, 29, 2, 49, 0, 0, time.UTC)) {
|
||||
t.Fatalf("console calls %d, missed row %+v", r.con.calls, s)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a new owner disables the schedule instead of running it", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.st.owners["survival"] = "newowner"
|
||||
id := r.schedule(ScheduleCommand, schedT0)
|
||||
r.tick(t)
|
||||
s := r.outcome(t, id, ScheduleSkipped, "the server has a new owner since this schedule was saved; save it again to use it")
|
||||
if s.Enabled || s.NextRunAt != nil || r.con.calls != 0 {
|
||||
t.Fatalf("still enabled or ran: %+v, %d console calls", s, r.con.calls)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("the schedule of a released server stops too", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.st.owners["survival"] = ""
|
||||
id := r.schedule(ScheduleStop, schedT0)
|
||||
r.tick(t)
|
||||
if s := r.st.row(t, id); s.Enabled || r.cl.desired["survival"] != "" {
|
||||
t.Fatalf("ran on a released server: %+v", s)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("stop", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleStop, schedT0)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleOK, "")
|
||||
if r.cl.desired["survival"] != v1alpha1.DesiredStopped {
|
||||
t.Fatalf("desired %q", r.cl.desired["survival"])
|
||||
}
|
||||
r2 := newSchedRig(t)
|
||||
r2.stopped()
|
||||
id = r2.schedule(ScheduleStop, schedT0)
|
||||
r2.tick(t)
|
||||
r2.outcome(t, id, ScheduleSkipped, "the server was already stopped")
|
||||
if r2.cl.desired["survival"] != "" {
|
||||
t.Fatal("stopped a stopped server again")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a server gone since the schedule was saved", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
delete(r.cl.byName, "survival")
|
||||
id := r.schedule(ScheduleStop, schedT0)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleSkipped, "the server no longer exists")
|
||||
})
|
||||
}
|
||||
|
||||
func TestScheduleRunnerStart(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
setup func(r *schedRig)
|
||||
result string
|
||||
detail string
|
||||
desired v1alpha1.DesiredState
|
||||
retried bool
|
||||
}{
|
||||
{"a stopped server starts", func(r *schedRig) { r.stopped() }, ScheduleOK, "", v1alpha1.DesiredRunning, false},
|
||||
{"a running server is left alone", func(*schedRig) {}, ScheduleSkipped, "the server was already running", "", false},
|
||||
{"a failed server is retried", func(r *schedRig) { r.cl.byName["survival"].Phase = string(v1alpha1.PhaseFailed) },
|
||||
ScheduleOK, "", v1alpha1.DesiredRunning, true},
|
||||
{"a failed start of a stopped server starts plainly", func(r *schedRig) {
|
||||
r.stopped()
|
||||
r.cl.byName["survival"].Phase = string(v1alpha1.PhaseFailed)
|
||||
}, ScheduleOK, "", v1alpha1.DesiredRunning, false},
|
||||
{"the running cap holds it back", func(r *schedRig) {
|
||||
r.stopped()
|
||||
r.a.MaxRunningServers = 1
|
||||
r.cl.list = []ServerInfo{{Name: "other", DesiredState: string(v1alpha1.DesiredRunning)}}
|
||||
}, ScheduleSkipped, "the cluster is at its running-server cap", "", false},
|
||||
{"a pending retirement holds it back", func(r *schedRig) {
|
||||
r.stopped()
|
||||
r.repo.byName["survival"].Retire = &RetireState{}
|
||||
}, ScheduleSkipped, "the server is being given up or deleted", "", false},
|
||||
{"a busy world holds it back", func(r *schedRig) {
|
||||
r.stopped()
|
||||
r.cl.wakeErr["survival"] = &MaintenanceBusyError{Kind: maintenance.KindBackup}
|
||||
}, ScheduleSkipped, "the world is busy with a backup", "", false},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
tc.setup(r)
|
||||
id := r.schedule(ScheduleStart, schedT0)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, tc.result, tc.detail)
|
||||
if r.cl.desired["survival"] != tc.desired || (len(r.cl.retried) == 1) != tc.retried {
|
||||
t.Fatalf("desired %q, retried %v", r.cl.desired["survival"], r.cl.retried)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestScheduleRunnerWarning(t *testing.T) {
|
||||
t.Run("told once, at the warning time", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
due := schedT0.Add(5 * time.Minute)
|
||||
id := r.schedule(ScheduleRestart, due, func(s *Schedule) { s.WarnMinutes = 5 })
|
||||
r.clock = schedT0.Add(-time.Second)
|
||||
r.tick(t)
|
||||
if r.con.calls != 0 {
|
||||
t.Fatal("warned early")
|
||||
}
|
||||
r.clock = schedT0
|
||||
r.tick(t)
|
||||
r.clock = schedT0.Add(15 * time.Second)
|
||||
r.tick(t)
|
||||
if r.con.calls != 1 || r.con.gotCommand != "say [Felis] 服务器将在 5 分钟后重启 / Server restarts in 5 min" {
|
||||
t.Fatalf("%d warnings, last %q", r.con.calls, r.con.gotCommand)
|
||||
}
|
||||
if s := r.st.row(t, id); !sameTime(s.WarnedFor, due) || s.LastRunAt != nil {
|
||||
t.Fatalf("after the warning %+v", s)
|
||||
}
|
||||
r.clock = due
|
||||
r.tick(t) // the run itself claims and clears the marker
|
||||
if s := r.st.row(t, id); s.WarnedFor != nil || s.RunState != runStopping {
|
||||
t.Fatalf("after the run started %+v", s)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("late warnings say how long is really left", func(t *testing.T) {
|
||||
for action, want := range map[string]string{
|
||||
ScheduleStop: "say [Felis] 服务器将在 2 分钟后关闭 / Server stops in 2 min",
|
||||
ScheduleBackup: "say [Felis] 服务器将在 2 分钟后暂停做备份,完成后自动恢复 / Server pauses for a backup in 2 min and comes back after",
|
||||
} {
|
||||
r := newSchedRig(t)
|
||||
r.schedule(action, schedT0.Add(90*time.Second), func(s *Schedule) { s.WarnMinutes = 10 })
|
||||
r.tick(t)
|
||||
if r.con.gotCommand != want {
|
||||
t.Fatalf("%s warned %q, want %q", action, r.con.gotCommand, want)
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("nobody to tell on a stopped server, none without a lead time", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.stopped()
|
||||
a := r.schedule(ScheduleRestart, schedT0.Add(time.Minute), func(s *Schedule) { s.WarnMinutes = 5 })
|
||||
r.tick(t)
|
||||
r2 := newSchedRig(t)
|
||||
r2.schedule(ScheduleRestart, schedT0.Add(time.Minute))
|
||||
r2.tick(t)
|
||||
if r.con.calls != 0 || r2.con.calls != 0 || r.st.row(t, a).WarnedFor != nil {
|
||||
t.Fatalf("warned: %d / %d", r.con.calls, r2.con.calls)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestScheduleRunnerRestart(t *testing.T) {
|
||||
t.Run("stops, waits for the pod, starts", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleRestart, schedT0)
|
||||
r.tick(t)
|
||||
r.step(t, id, runStopping)
|
||||
if r.cl.desired["survival"] != v1alpha1.DesiredStopped {
|
||||
t.Fatalf("desired %q", r.cl.desired["survival"])
|
||||
}
|
||||
// Still shutting down.
|
||||
info := r.cl.byName["survival"]
|
||||
info.DesiredState = string(v1alpha1.DesiredStopped)
|
||||
r.clock = schedT0.Add(time.Minute)
|
||||
r.tick(t)
|
||||
r.step(t, id, runStopping)
|
||||
info.Ready = false
|
||||
r.tick(t)
|
||||
r.step(t, id, runStopping) // not Ready, but not Stopped either
|
||||
info.Phase = string(v1alpha1.PhaseStopped)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleOK, "")
|
||||
if r.cl.desired["survival"] != v1alpha1.DesiredRunning {
|
||||
t.Fatalf("desired %q", r.cl.desired["survival"])
|
||||
}
|
||||
if got := r.runAudits(); len(got) != 1 || got[0] != `{"action":"restart","detail":"","result":"ok","schedule":1}` {
|
||||
t.Fatalf("run audits %v", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a stopped server is not restarted", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.stopped()
|
||||
id := r.schedule(ScheduleRestart, schedT0)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleSkipped, "the server was not running")
|
||||
})
|
||||
|
||||
t.Run("someone starting it meanwhile ends the run", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleRestart, schedT0)
|
||||
r.tick(t)
|
||||
r.clock = schedT0.Add(30 * time.Second)
|
||||
r.tick(t) // info still says desired Running: a person started it again
|
||||
r.outcome(t, id, ScheduleSkipped, "someone started the server before the run finished")
|
||||
})
|
||||
|
||||
t.Run("a pod that does not stop in 15 minutes: give up and start it again", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleRestart, schedT0)
|
||||
r.tick(t)
|
||||
r.cl.byName["survival"].DesiredState = string(v1alpha1.DesiredStopped)
|
||||
r.clock = schedT0.Add(stopWait - time.Second)
|
||||
r.tick(t)
|
||||
r.step(t, id, runStopping)
|
||||
r.clock = schedT0.Add(stopWait)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "the server did not stop within 15 minutes; it was started again")
|
||||
if r.cl.desired["survival"] != v1alpha1.DesiredRunning {
|
||||
t.Fatalf("desired %q", r.cl.desired["survival"])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("the running cap: keep trying for 15 minutes, then fail", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleRestart, schedT0)
|
||||
r.tick(t)
|
||||
r.stopped()
|
||||
r.a.MaxRunningServers = 1
|
||||
r.cl.list = []ServerInfo{{Name: "other", DesiredState: string(v1alpha1.DesiredRunning)}}
|
||||
r.clock = schedT0.Add(time.Minute)
|
||||
r.tick(t)
|
||||
r.step(t, id, runStarting)
|
||||
r.clock = schedT0.Add(time.Minute + startWait - time.Second)
|
||||
r.tick(t)
|
||||
r.step(t, id, runStarting)
|
||||
r.clock = schedT0.Add(time.Minute + startWait)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "could not start the server again: the cluster is at its running-server cap")
|
||||
})
|
||||
|
||||
t.Run("the cap clearing lets it start", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleRestart, schedT0)
|
||||
r.tick(t)
|
||||
r.stopped()
|
||||
r.a.MaxRunningServers = 1
|
||||
r.cl.list = []ServerInfo{{Name: "other", DesiredState: string(v1alpha1.DesiredRunning)}}
|
||||
r.tick(t)
|
||||
r.cl.list = nil
|
||||
r.clock = schedT0.Add(time.Minute)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleOK, "")
|
||||
})
|
||||
}
|
||||
|
||||
func TestScheduleRunnerBackup(t *testing.T) {
|
||||
job := func(state, message string, at time.Time) AsyncJob {
|
||||
return AsyncJob{Kind: "backup", State: state, Message: message, StartedAt: at, Scheduled: true}
|
||||
}
|
||||
|
||||
t.Run("a running server: stop, back up, start", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.step(t, id, runStopping)
|
||||
r.clock = schedT0.Add(time.Minute)
|
||||
r.cl.maintErr["survival"] = ErrNotStopped
|
||||
r.cl.byName["survival"].DesiredState = string(v1alpha1.DesiredStopped)
|
||||
r.tick(t)
|
||||
r.step(t, id, runStopping) // the pod is still going down
|
||||
delete(r.cl.maintErr, "survival")
|
||||
r.stopped()
|
||||
r.clock = schedT0.Add(2 * time.Minute)
|
||||
r.tick(t)
|
||||
r.step(t, id, runBackingUp)
|
||||
if strings.Join(r.cl.acquired, ",") != "survival:backup" || strings.Join(r.cl.released, ",") != "survival" ||
|
||||
len(r.b.scheduled) != 1 || r.b.scheduled[0] != (ScheduledCandidate{Name: "survival", OwnerID: "owner1"}) {
|
||||
t.Fatalf("acquired %v released %v backups %v", r.cl.acquired, r.cl.released, r.b.scheduled)
|
||||
}
|
||||
r.jobs.jobs = []AsyncJob{job("running", "", r.clock.Add(-time.Minute))} // clock skew
|
||||
r.clock = schedT0.Add(30 * time.Minute)
|
||||
r.tick(t)
|
||||
r.step(t, id, runBackingUp)
|
||||
r.jobs.jobs = []AsyncJob{job("succeeded", "", schedT0.Add(time.Minute))}
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleOK, "")
|
||||
if r.cl.desired["survival"] != v1alpha1.DesiredRunning || r.jobs.got != "survival" {
|
||||
t.Fatalf("desired %q, jobs read for %q", r.cl.desired["survival"], r.jobs.got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a stopped server is backed up at once and left stopped", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.stopped()
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.step(t, id, runBackingUp)
|
||||
r.jobs.jobs = []AsyncJob{job("succeeded", "", schedT0)}
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleOK, "")
|
||||
if r.cl.desired["survival"] != "" {
|
||||
t.Fatalf("desired %q, want untouched", r.cl.desired["survival"])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a failed backup is reported and the server still comes back", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.stopped()
|
||||
r.tick(t)
|
||||
r.jobs.jobs = []AsyncJob{job("failed", "disk full", schedT0), job("succeeded", "", schedT0.Add(-time.Hour))}
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "the backup failed: disk full")
|
||||
if r.cl.desired["survival"] != v1alpha1.DesiredRunning {
|
||||
t.Fatalf("desired %q", r.cl.desired["survival"])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("an older or unscheduled Job is not this run's", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.stopped()
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
manual := job("failed", "x", schedT0)
|
||||
manual.Scheduled = false
|
||||
r.jobs.jobs = []AsyncJob{manual, job("succeeded", "", schedT0.Add(-backupJobSkew-time.Second))}
|
||||
r.clock = schedT0.Add(backupJobSkew - time.Second)
|
||||
r.tick(t)
|
||||
r.step(t, id, runBackingUp)
|
||||
r.clock = schedT0.Add(backupJobSkew)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "the backup Job is gone before it could be checked; see the Backups page")
|
||||
})
|
||||
|
||||
t.Run("two scheduled Jobs since the step: the newest counts", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.stopped()
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.jobs.jobs = []AsyncJob{job("succeeded", "", schedT0.Add(time.Minute)), job("failed", "disk full", schedT0)}
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleOK, "")
|
||||
})
|
||||
|
||||
t.Run("a Job running past 45 minutes", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.stopped()
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.jobs.jobs = []AsyncJob{job("running", "", schedT0)}
|
||||
r.clock = schedT0.Add(backupWait - time.Second)
|
||||
r.tick(t)
|
||||
r.step(t, id, runBackingUp)
|
||||
r.clock = schedT0.Add(backupWait)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "the backup did not finish within 45 minutes")
|
||||
})
|
||||
|
||||
t.Run("a world kept busy for 15 minutes", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.stopped()
|
||||
r.cl.maintErr["survival"] = &MaintenanceBusyError{Kind: maintenance.KindRestore}
|
||||
r.clock = schedT0.Add(stopWait - time.Second)
|
||||
r.tick(t)
|
||||
r.step(t, id, runStopping)
|
||||
r.clock = schedT0.Add(stopWait)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "another operation kept the world busy for 15 minutes; it was started again")
|
||||
if len(r.b.scheduled) != 0 || r.cl.desired["survival"] != v1alpha1.DesiredRunning {
|
||||
t.Fatalf("backups %v, desired %q", r.b.scheduled, r.cl.desired["survival"])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a pod that does not stop in 15 minutes: no backup, started again", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.cl.byName["survival"].DesiredState = string(v1alpha1.DesiredStopped)
|
||||
r.cl.maintErr["survival"] = ErrNotStopped
|
||||
r.clock = schedT0.Add(stopWait)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "the server did not stop within 15 minutes; it was started again")
|
||||
if len(r.b.scheduled) != 0 {
|
||||
t.Fatalf("backups %v", r.b.scheduled)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a failed Job without a message", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.stopped()
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.jobs.jobs = []AsyncJob{job("failed", "", schedT0)}
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "the backup failed:")
|
||||
})
|
||||
|
||||
t.Run("the backup cannot start", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.b.err = errors.New("quota")
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.stopped()
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "could not start the backup: quota; the server was started again")
|
||||
if strings.Join(r.cl.released, ",") != "survival" {
|
||||
t.Fatalf("released %v", r.cl.released)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("refused before touching the server", func(t *testing.T) {
|
||||
cases := []struct {
|
||||
name, result, detail string
|
||||
setup func(r *schedRig)
|
||||
}{
|
||||
{"no backup executor", ScheduleFailed, "backups are not configured", func(r *schedRig) { r.a.Backuper = &fakeBackuper{} }},
|
||||
{"store full", ScheduleFailed, "the backup store is full; ask an administrator to free space", func(r *schedRig) {
|
||||
r.a.BackupStoreCap = 1
|
||||
r.repo.backups = append(r.repo.backups, fakeBackup{view: BackupView{Status: "present", SizeBytes: 1}})
|
||||
}},
|
||||
{"no world yet", ScheduleSkipped, "the server has no world yet", func(r *schedRig) { r.cl.noWorld["survival"] = true }},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
tc.setup(r)
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, tc.result, tc.detail)
|
||||
if r.cl.desired["survival"] != "" || len(r.b.scheduled) != 0 {
|
||||
t.Fatalf("desired %q, backups %v", r.cl.desired["survival"], r.b.scheduled)
|
||||
}
|
||||
})
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestScheduleRunnerStaleClaim(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleCommand, schedT0.Add(time.Hour), func(s *Schedule) {
|
||||
s.RunState, s.RunStepAt = runClaimed, ptrTime(schedT0.Add(-claimedStale+time.Second))
|
||||
})
|
||||
r.tick(t)
|
||||
r.step(t, id, runClaimed)
|
||||
r.clock = schedT0.Add(time.Second)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "felis-api stopped in the middle of this run")
|
||||
if r.con.calls != 0 {
|
||||
t.Fatal("re-ran the command of a stale claim")
|
||||
}
|
||||
}
|
||||
|
||||
func TestStripFormatting(t *testing.T) {
|
||||
if got := stripFormatting(" §l§aHi§r §x§§ok "); got != "Hi ok" {
|
||||
t.Fatalf("stripFormatting = %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestScheduleRunnerEdges(t *testing.T) {
|
||||
t.Run("no store: nothing to run", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.a.Schedules = nil
|
||||
r.tick(t)
|
||||
})
|
||||
|
||||
t.Run("no console: no warning, and a command fails", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.a.Console = nil
|
||||
warn := r.schedule(ScheduleRestart, schedT0.Add(time.Minute), func(s *Schedule) { s.WarnMinutes = 5 })
|
||||
cmd := r.schedule(ScheduleCommand, schedT0)
|
||||
r.tick(t)
|
||||
r.outcome(t, cmd, ScheduleFailed, "the console is not configured")
|
||||
if s := r.st.row(t, warn); s.WarnedFor != nil {
|
||||
t.Fatalf("marked warned without a console: %+v", s)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a stopped server's backup gives up without starting it", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.stopped()
|
||||
r.cl.maintErr["survival"] = &MaintenanceBusyError{Kind: maintenance.KindFileWrite}
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.step(t, id, runStopping)
|
||||
r.clock = schedT0.Add(stopWait)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "another operation kept the world busy for 15 minutes")
|
||||
if r.cl.desired["survival"] != "" {
|
||||
t.Fatalf("desired %q, want untouched", r.cl.desired["survival"])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a stopped server's backup that cannot start leaves it stopped", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.stopped()
|
||||
r.b.err = errors.New("quota")
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "could not start the backup: quota")
|
||||
if r.cl.desired["survival"] != "" {
|
||||
t.Fatalf("desired %q, want untouched", r.cl.desired["survival"])
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("the backup executor gone mid-run (felis-api restarted without it)", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.stopped()
|
||||
r.a.Backuper = &fakeBackuper{}
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "backups are not configured")
|
||||
})
|
||||
|
||||
t.Run("the server deleted while its world is being locked", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.stopped()
|
||||
r.cl.maintErr["survival"] = ErrNotFound
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleSkipped, "the server no longer exists")
|
||||
})
|
||||
|
||||
t.Run("without Job status the backup counts as done", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.stopped()
|
||||
r.a.JobStatus = nil
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.step(t, id, runBackingUp)
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleOK, "")
|
||||
})
|
||||
|
||||
t.Run("a failed backup whose server cannot start either names both", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleBackup, schedT0)
|
||||
r.tick(t)
|
||||
r.stopped()
|
||||
r.tick(t)
|
||||
r.repo.byName["survival"].Retire = &RetireState{}
|
||||
r.jobs.jobs = []AsyncJob{{Kind: "backup", State: "failed", Message: "disk full", StartedAt: schedT0, Scheduled: true}}
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "the backup failed: disk full; could not start the server again: the server is being given up or deleted")
|
||||
})
|
||||
|
||||
t.Run("a retirement stops the restart at once", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleRestart, schedT0)
|
||||
r.tick(t)
|
||||
r.stopped()
|
||||
r.repo.byName["survival"].Retire = &RetireState{}
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleFailed, "could not start the server again: the server is being given up or deleted")
|
||||
})
|
||||
|
||||
t.Run("the server deleted before it could start again", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleRestart, schedT0)
|
||||
r.tick(t)
|
||||
r.stopped()
|
||||
delete(r.repo.byName, "survival")
|
||||
r.tick(t)
|
||||
r.outcome(t, id, ScheduleSkipped, "the server no longer exists")
|
||||
})
|
||||
}
|
||||
|
||||
func TestScheduleNextRunMidnightJump(t *testing.T) {
|
||||
// Chile moves its clocks from 00:00 to 01:00 on 2026-09-06: that Sunday has
|
||||
// no midnight, so the weekday is read at midday.
|
||||
santiago := mustZone(t, "America/Santiago")
|
||||
s := Schedule{MinuteOfDay: 12 * 60, Weekdays: 1, Timezone: "America/Santiago"}
|
||||
after := time.Date(2026, 9, 5, 13, 0, 0, 0, santiago)
|
||||
if got, want := s.nextRun(after), time.Date(2026, 9, 6, 12, 0, 0, 0, santiago); !got.Equal(want) {
|
||||
t.Fatalf("nextRun = %s, want %s", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScheduleFinishLostRace: a finish that another felis-api beat to the row
|
||||
// changes nothing and leaves no second schedule.run audit.
|
||||
func TestScheduleFinishLostRace(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleRestart, schedT0, func(s *Schedule) { s.LastResult = ScheduleOK })
|
||||
s := r.st.row(t, id)
|
||||
if err := r.a.finishRun(t.Context(), s, runStarting, ScheduleFailed, "late"); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
r.outcome(t, id, ScheduleOK, "")
|
||||
if len(r.repo.audits) != 0 {
|
||||
t.Fatalf("audits %+v", r.repo.audits)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,293 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net/http"
|
||||
"slices"
|
||||
"strings"
|
||||
"time"
|
||||
_ "time/tzdata" // a schedule's timezone resolves on a host without zoneinfo
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
// Scheduled tasks. An owner or admin asks felis-api to act on a server at set
|
||||
// times: run a console command, restart it, stop it, start it, or back it up.
|
||||
// A backup of a running server stops it, takes the backup and starts it again,
|
||||
// so a world kept up around the clock still gets restore points; the
|
||||
// BackupScheduler only ever backs up a stopped world. The schedules live in
|
||||
// server_schedules (migration 0036), and RunSchedules (schedulerunner.go) is
|
||||
// the loop that fires them.
|
||||
|
||||
// Schedule actions.
|
||||
const (
|
||||
ScheduleCommand = "command"
|
||||
ScheduleRestart = "restart"
|
||||
ScheduleStop = "stop"
|
||||
ScheduleStart = "start"
|
||||
ScheduleBackup = "backup"
|
||||
)
|
||||
|
||||
// Run results (Schedule.LastResult). A run in progress has none yet.
|
||||
const (
|
||||
ScheduleOK = "ok"
|
||||
ScheduleSkipped = "skipped"
|
||||
ScheduleFailed = "failed"
|
||||
// ScheduleMissed is a run felis-api was not up for: it is dropped once it
|
||||
// is scheduleMissGrace late, so a restart never lands hours after its time.
|
||||
ScheduleMissed = "missed"
|
||||
)
|
||||
|
||||
const (
|
||||
// maxSchedulesPerServer bounds the rows one server can hold.
|
||||
maxSchedulesPerServer = 20
|
||||
// maxScheduleLabel bounds a label, in characters.
|
||||
maxScheduleLabel = 64
|
||||
// maxScheduleDetail bounds the stored outcome of a run (a command's reply
|
||||
// can be long), in bytes.
|
||||
maxScheduleDetail = 500
|
||||
// minPowerEveryMinutes is the shortest interval of an action that takes the
|
||||
// server down; a command may run every scheduleEveryMinutes[0].
|
||||
minPowerEveryMinutes = 60
|
||||
)
|
||||
|
||||
var (
|
||||
// scheduleEveryMinutes are the intervals a repeating schedule may use: each
|
||||
// divides a day, so the runs sit at the same clock times every day.
|
||||
scheduleEveryMinutes = []int{15, 30, 60, 120, 180, 240, 360, 480, 720}
|
||||
// scheduleWarnMinutes are the lead times of the in-game warning.
|
||||
scheduleWarnMinutes = []int{0, 1, 5, 10, 15, 30}
|
||||
)
|
||||
|
||||
var (
|
||||
// ErrScheduleLimit means the server already holds maxSchedulesPerServer.
|
||||
ErrScheduleLimit = errors.New("schedule limit reached")
|
||||
// ErrScheduleRunning means the schedule has a run in progress, which must
|
||||
// finish before the schedule is changed, deleted or run again.
|
||||
ErrScheduleRunning = errors.New("schedule is running")
|
||||
)
|
||||
|
||||
// Schedule is one server_schedules row as the owner sees it. OwnerID, the
|
||||
// warning marker and the run step stay off the wire; RunState tells the panel
|
||||
// what a run in progress is waiting for.
|
||||
type Schedule struct {
|
||||
ID int64 `json:"id"`
|
||||
Server string `json:"server"`
|
||||
Label string `json:"label"`
|
||||
Action string `json:"action"`
|
||||
// Command is the console command of a command schedule, without a slash.
|
||||
Command string `json:"command"`
|
||||
// EveryMinutes repeats the schedule at every multiple of it since local
|
||||
// midnight; 0 runs it once a day at MinuteOfDay. Either way only on the
|
||||
// Weekdays (a bitmask, bit 0 Sunday), in Timezone.
|
||||
EveryMinutes int `json:"every_minutes"`
|
||||
MinuteOfDay int `json:"minute_of_day"`
|
||||
Weekdays int `json:"weekdays"`
|
||||
Timezone string `json:"timezone"`
|
||||
// WarnMinutes is how long before a restart, stop or backup the players on
|
||||
// the server are told it is coming; 0 says nothing.
|
||||
WarnMinutes int `json:"warn_minutes"`
|
||||
Enabled bool `json:"enabled"`
|
||||
// NextRunAt is nil while the schedule is disabled.
|
||||
NextRunAt *time.Time `json:"next_run_at"`
|
||||
// RunState is the step of a run in progress (runStopping, runBackingUp,
|
||||
// runStarting, or runClaimed while its first step runs); empty otherwise.
|
||||
RunState string `json:"run_state"`
|
||||
LastRunAt *time.Time `json:"last_run_at"`
|
||||
LastResult string `json:"last_result"`
|
||||
LastDetail string `json:"last_detail"`
|
||||
CreatedBy string `json:"created_by"`
|
||||
CreatedAt time.Time `json:"created_at"`
|
||||
|
||||
// OwnerID is the server's owner when the schedule was saved (empty for an
|
||||
// unowned server). The runner fires the schedule only while it still is.
|
||||
OwnerID string `json:"-"`
|
||||
WarnedFor *time.Time `json:"-"`
|
||||
RunResume bool `json:"-"`
|
||||
RunStepAt *time.Time `json:"-"`
|
||||
}
|
||||
|
||||
// DueSchedule is a schedule the runner has to look at, with the server's
|
||||
// current owner.
|
||||
type DueSchedule struct {
|
||||
Schedule
|
||||
ServerOwner string
|
||||
}
|
||||
|
||||
// ServerSchedules stores the schedules (PGRepo). The run methods are
|
||||
// compare-and-set writes that report whether they applied, so two felis-api
|
||||
// processes side by side during a rollout never fire one run twice.
|
||||
type ServerSchedules interface {
|
||||
// ListSchedules lists a server's schedules, oldest first.
|
||||
ListSchedules(ctx context.Context, server string) ([]Schedule, error)
|
||||
// GetSchedule reads one schedule of server, or ErrNotFound.
|
||||
GetSchedule(ctx context.Context, server string, id int64) (*Schedule, error)
|
||||
// CreateSchedule inserts s and fills in its ID and CreatedAt, or returns
|
||||
// ErrScheduleLimit when the server already holds limit schedules.
|
||||
CreateSchedule(ctx context.Context, s *Schedule, limit int) error
|
||||
// UpdateSchedule writes s's settings, owner and next run and clears its
|
||||
// warning marker: ErrNotFound, or ErrScheduleRunning during a run.
|
||||
UpdateSchedule(ctx context.Context, s *Schedule) error
|
||||
// DeleteSchedule removes a schedule: ErrNotFound, or ErrScheduleRunning
|
||||
// during a run.
|
||||
DeleteSchedule(ctx context.Context, server string, id int64) error
|
||||
|
||||
// DueSchedules lists the enabled schedules of live servers due by horizon
|
||||
// and every schedule with a run in progress.
|
||||
DueSchedules(ctx context.Context, horizon time.Time) ([]DueSchedule, error)
|
||||
// ClaimScheduleRun starts a run (run state runClaimed, a fresh result) of
|
||||
// a schedule without one. With due set it claims only the enabled run due
|
||||
// then and moves the schedule on to next; nil due is a run on request that
|
||||
// leaves the next run where it is.
|
||||
ClaimScheduleRun(ctx context.Context, id int64, due, next *time.Time, now time.Time) (bool, error)
|
||||
// AdvanceScheduleRun moves a run from step from to step to, recording
|
||||
// resume and, when result is set, the outcome so far.
|
||||
AdvanceScheduleRun(ctx context.Context, id int64, from, to string, resume bool, result, detail string, now time.Time) (bool, error)
|
||||
// FinishScheduleRun ends a run at step from with its outcome.
|
||||
FinishScheduleRun(ctx context.Context, id int64, from, result, detail string) (bool, error)
|
||||
// MissScheduleRun records the run due then as missed and moves on to next.
|
||||
MissScheduleRun(ctx context.Context, id int64, due, next time.Time, detail string) (bool, error)
|
||||
// WarnScheduleRun marks the run due then as warned about, once.
|
||||
WarnScheduleRun(ctx context.Context, id int64, due time.Time) (bool, error)
|
||||
// DisableSchedule turns off an enabled schedule without a run in progress
|
||||
// and records why.
|
||||
DisableSchedule(ctx context.Context, id int64, detail string) (bool, error)
|
||||
}
|
||||
|
||||
// scheduleInput is the body of POST /servers/{name}/schedules and of PUT
|
||||
// /servers/{name}/schedules/{id}. Enabled defaults to true.
|
||||
type scheduleInput struct {
|
||||
Label string `json:"label"`
|
||||
Action string `json:"action"`
|
||||
Command string `json:"command"`
|
||||
EveryMinutes int `json:"every_minutes"`
|
||||
MinuteOfDay int `json:"minute_of_day"`
|
||||
Weekdays int `json:"weekdays"`
|
||||
Timezone string `json:"timezone"`
|
||||
WarnMinutes int `json:"warn_minutes"`
|
||||
Enabled *bool `json:"enabled"`
|
||||
}
|
||||
|
||||
func badSchedule(format string, a ...any) error {
|
||||
return newError(http.StatusBadRequest, "bad_schedule", format, a...)
|
||||
}
|
||||
|
||||
// apply validates in and writes it onto s.
|
||||
func (in *scheduleInput) apply(s *Schedule) error {
|
||||
label := strings.TrimSpace(in.Label)
|
||||
if utf8.RuneCountInString(label) > maxScheduleLabel {
|
||||
return badSchedule("label too long (max %d characters)", maxScheduleLabel)
|
||||
}
|
||||
if strings.IndexFunc(label, func(c rune) bool { return c < 0x20 || c == 0x7f }) >= 0 {
|
||||
return badSchedule("label must be a single line")
|
||||
}
|
||||
|
||||
var command string
|
||||
switch in.Action {
|
||||
case ScheduleCommand:
|
||||
c, err := normalizeConsoleCommand(in.Command)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
command = c
|
||||
case ScheduleRestart, ScheduleStop, ScheduleStart, ScheduleBackup:
|
||||
if strings.TrimSpace(in.Command) != "" {
|
||||
return badSchedule("only a command schedule takes a command")
|
||||
}
|
||||
default:
|
||||
return badSchedule("action must be one of command, restart, stop, start, backup")
|
||||
}
|
||||
|
||||
switch {
|
||||
case in.EveryMinutes == 0:
|
||||
if in.MinuteOfDay < 0 || in.MinuteOfDay > 24*60-1 {
|
||||
return badSchedule("minute_of_day must be between 0 and 1439")
|
||||
}
|
||||
case !slices.Contains(scheduleEveryMinutes, in.EveryMinutes):
|
||||
return badSchedule("every_minutes must be 0 or one of 15, 30, 60, 120, 180, 240, 360, 480, 720")
|
||||
case in.EveryMinutes < minPowerEveryMinutes && in.Action != ScheduleCommand:
|
||||
return badSchedule("a %s can repeat at most every %d minutes", in.Action, minPowerEveryMinutes)
|
||||
case in.MinuteOfDay != 0:
|
||||
return badSchedule("a repeating schedule has no minute_of_day")
|
||||
}
|
||||
if in.Weekdays < 1 || in.Weekdays > 0x7f {
|
||||
return badSchedule("weekdays must name at least one day (bits 0 to 6, Sunday first)")
|
||||
}
|
||||
|
||||
tz := strings.TrimSpace(in.Timezone)
|
||||
if tz == "" || tz == "Local" {
|
||||
return badSchedule("timezone must be an IANA zone name such as Asia/Shanghai")
|
||||
}
|
||||
if _, err := time.LoadLocation(tz); err != nil {
|
||||
return badSchedule("unknown timezone %q", tz)
|
||||
}
|
||||
|
||||
if !slices.Contains(scheduleWarnMinutes, in.WarnMinutes) {
|
||||
return badSchedule("warn_minutes must be one of 0, 1, 5, 10, 15, 30")
|
||||
}
|
||||
// A restart, stop or backup repeats at most hourly, so its warning (30
|
||||
// minutes ahead at most) always comes after the run before it.
|
||||
if in.WarnMinutes > 0 && (in.Action == ScheduleCommand || in.Action == ScheduleStart) {
|
||||
return badSchedule("only a restart, stop or backup warns the players")
|
||||
}
|
||||
|
||||
s.Label, s.Action, s.Command = label, in.Action, command
|
||||
s.EveryMinutes, s.MinuteOfDay, s.Weekdays = in.EveryMinutes, in.MinuteOfDay, in.Weekdays
|
||||
s.Timezone, s.WarnMinutes = tz, in.WarnMinutes
|
||||
s.Enabled = in.Enabled == nil || *in.Enabled
|
||||
return nil
|
||||
}
|
||||
|
||||
// nextRun is the first run of s strictly after after, or the zero time for a
|
||||
// schedule that names no weekday. The runs are wall-clock times in s's zone:
|
||||
// a time the zone skips at the start of daylight saving runs an hour early
|
||||
// (time.Date reads it with the offset before the jump), and one it repeats
|
||||
// runs once, the first time the clock shows it.
|
||||
func (s *Schedule) nextRun(after time.Time) time.Time {
|
||||
loc, err := time.LoadLocation(s.Timezone)
|
||||
if err != nil {
|
||||
loc = time.UTC // validated when saved; a zone later dropped runs in UTC
|
||||
}
|
||||
y, m, d := after.In(loc).Date()
|
||||
// Eight days: today's runs may all be past, and a schedule of one weekday
|
||||
// next runs a week from today.
|
||||
for day := d; day <= d+7; day++ {
|
||||
if s.Weekdays&(1<<time.Date(y, m, day, 12, 0, 0, 0, loc).Weekday()) == 0 {
|
||||
continue
|
||||
}
|
||||
var minutes []int
|
||||
if s.EveryMinutes > 0 {
|
||||
for at := 0; at < 24*60; at += s.EveryMinutes {
|
||||
minutes = append(minutes, at)
|
||||
}
|
||||
} else {
|
||||
minutes = []int{s.MinuteOfDay}
|
||||
}
|
||||
for _, at := range minutes {
|
||||
if t := time.Date(y, m, day, at/60, at%60, 0, 0, loc); t.After(after) {
|
||||
return t
|
||||
}
|
||||
}
|
||||
}
|
||||
return time.Time{}
|
||||
}
|
||||
|
||||
// normalizeConsoleCommand makes raw one console command: surrounding space
|
||||
// and a single leading slash removed (players type "/say hi"; RCON wants
|
||||
// "say hi"), within maxConsoleCommandLen, and without control characters, so
|
||||
// one request can never smuggle a second command past a newline.
|
||||
func normalizeConsoleCommand(raw string) (string, error) {
|
||||
command := strings.TrimSpace(strings.TrimPrefix(strings.TrimSpace(raw), "/"))
|
||||
if command == "" {
|
||||
return "", newError(http.StatusBadRequest, "bad_request", "command is required")
|
||||
}
|
||||
if len(command) > maxConsoleCommandLen {
|
||||
return "", newError(http.StatusBadRequest, "bad_request",
|
||||
"command too long (max %d bytes)", maxConsoleCommandLen)
|
||||
}
|
||||
if strings.IndexFunc(command, func(c rune) bool { return c < 0x20 }) >= 0 {
|
||||
return "", newError(http.StatusBadRequest, "bad_request",
|
||||
"command must be a single line (no control characters)")
|
||||
}
|
||||
return command, nil
|
||||
}
|
||||
@@ -0,0 +1,712 @@
|
||||
package api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"slices"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/apis/felis/v1alpha1"
|
||||
)
|
||||
|
||||
var _ ServerSchedules = (*PGRepo)(nil)
|
||||
|
||||
// fakeSchedules is an in-memory ServerSchedules with the compare-and-set
|
||||
// conditions of the PGRepo queries (internal/pgint/schedules_test.go runs the
|
||||
// same cases against Postgres). It hands out copies, so a test sees only what
|
||||
// was written through the interface.
|
||||
type fakeSchedules struct {
|
||||
rows map[int64]*Schedule
|
||||
owners map[string]string // server -> its owner now, for DueSchedules
|
||||
nextID int64
|
||||
claims int
|
||||
}
|
||||
|
||||
func newFakeSchedules() *fakeSchedules {
|
||||
return &fakeSchedules{rows: map[int64]*Schedule{}, owners: map[string]string{}}
|
||||
}
|
||||
|
||||
func cloneSchedule(s *Schedule) *Schedule {
|
||||
c := *s
|
||||
for _, p := range []**time.Time{&c.NextRunAt, &c.WarnedFor, &c.RunStepAt, &c.LastRunAt} {
|
||||
if *p != nil {
|
||||
*p = ptrTime(**p)
|
||||
}
|
||||
}
|
||||
return &c
|
||||
}
|
||||
|
||||
func sameTime(p *time.Time, t time.Time) bool { return p != nil && p.Equal(t) }
|
||||
|
||||
// put stores s as it is and returns its id.
|
||||
func (f *fakeSchedules) put(s Schedule) int64 {
|
||||
f.nextID++
|
||||
s.ID = f.nextID
|
||||
f.rows[s.ID] = cloneSchedule(&s)
|
||||
return s.ID
|
||||
}
|
||||
|
||||
// row reads a schedule back for a test's assertions.
|
||||
func (f *fakeSchedules) row(t *testing.T, id int64) *Schedule {
|
||||
t.Helper()
|
||||
r, ok := f.rows[id]
|
||||
if !ok {
|
||||
t.Fatalf("schedule %d is gone", id)
|
||||
}
|
||||
return cloneSchedule(r)
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) ListSchedules(_ context.Context, server string) ([]Schedule, error) {
|
||||
var out []Schedule
|
||||
for _, r := range f.rows {
|
||||
if r.Server == server {
|
||||
out = append(out, *cloneSchedule(r))
|
||||
}
|
||||
}
|
||||
slices.SortFunc(out, func(a, b Schedule) int { return int(a.ID - b.ID) })
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) GetSchedule(_ context.Context, server string, id int64) (*Schedule, error) {
|
||||
r, ok := f.rows[id]
|
||||
if !ok || r.Server != server {
|
||||
return nil, ErrNotFound
|
||||
}
|
||||
return cloneSchedule(r), nil
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) CreateSchedule(_ context.Context, s *Schedule, limit int) error {
|
||||
n := 0
|
||||
for _, r := range f.rows {
|
||||
if r.Server == s.Server {
|
||||
n++
|
||||
}
|
||||
}
|
||||
if n >= limit {
|
||||
return ErrScheduleLimit
|
||||
}
|
||||
s.CreatedAt = time.Unix(1_700_000_000, 0)
|
||||
s.ID = f.put(*s)
|
||||
return nil
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) idle(server string, id int64) (*Schedule, error) {
|
||||
r, ok := f.rows[id]
|
||||
switch {
|
||||
case !ok || r.Server != server:
|
||||
return nil, ErrNotFound
|
||||
case r.RunState != "":
|
||||
return nil, ErrScheduleRunning
|
||||
}
|
||||
return r, nil
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) UpdateSchedule(_ context.Context, s *Schedule) error {
|
||||
r, err := f.idle(s.Server, s.ID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
r.OwnerID, r.Label, r.Action, r.Command = s.OwnerID, s.Label, s.Action, s.Command
|
||||
r.EveryMinutes, r.MinuteOfDay, r.Weekdays, r.Timezone = s.EveryMinutes, s.MinuteOfDay, s.Weekdays, s.Timezone
|
||||
r.WarnMinutes, r.Enabled, r.NextRunAt, r.WarnedFor = s.WarnMinutes, s.Enabled, s.NextRunAt, nil
|
||||
if r.NextRunAt != nil {
|
||||
r.NextRunAt = ptrTime(*r.NextRunAt)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) DeleteSchedule(_ context.Context, server string, id int64) error {
|
||||
if _, err := f.idle(server, id); err != nil {
|
||||
return err
|
||||
}
|
||||
delete(f.rows, id)
|
||||
return nil
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) DueSchedules(_ context.Context, horizon time.Time) ([]DueSchedule, error) {
|
||||
var out []DueSchedule
|
||||
for _, r := range f.rows {
|
||||
if r.RunState != "" || (r.Enabled && r.NextRunAt != nil && !r.NextRunAt.After(horizon)) {
|
||||
out = append(out, DueSchedule{Schedule: *cloneSchedule(r), ServerOwner: f.owners[r.Server]})
|
||||
}
|
||||
}
|
||||
slices.SortFunc(out, func(a, b DueSchedule) int {
|
||||
if (a.RunState == "") != (b.RunState == "") {
|
||||
if a.RunState != "" {
|
||||
return -1
|
||||
}
|
||||
return 1
|
||||
}
|
||||
if a.NextRunAt != nil && b.NextRunAt != nil && !a.NextRunAt.Equal(*b.NextRunAt) {
|
||||
return a.NextRunAt.Compare(*b.NextRunAt)
|
||||
}
|
||||
return int(a.ID - b.ID)
|
||||
})
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) ClaimScheduleRun(_ context.Context, id int64, due, next *time.Time, now time.Time) (bool, error) {
|
||||
r, ok := f.rows[id]
|
||||
if !ok || r.RunState != "" {
|
||||
return false, nil
|
||||
}
|
||||
if due != nil {
|
||||
if !r.Enabled || !sameTime(r.NextRunAt, *due) {
|
||||
return false, nil
|
||||
}
|
||||
r.NextRunAt, r.WarnedFor = ptrTime(*next), nil
|
||||
}
|
||||
f.claims++
|
||||
r.RunState, r.RunResume, r.RunStepAt = runClaimed, false, ptrTime(now)
|
||||
r.LastRunAt, r.LastResult, r.LastDetail = ptrTime(now), "", ""
|
||||
return true, nil
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) AdvanceScheduleRun(_ context.Context, id int64, from, to string, resume bool, result, detail string, now time.Time) (bool, error) {
|
||||
r, ok := f.rows[id]
|
||||
if !ok || r.RunState != from {
|
||||
return false, nil
|
||||
}
|
||||
r.RunState, r.RunResume, r.RunStepAt = to, resume, ptrTime(now)
|
||||
if result != "" {
|
||||
r.LastResult, r.LastDetail = result, detail
|
||||
}
|
||||
return true, nil
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) FinishScheduleRun(_ context.Context, id int64, from, result, detail string) (bool, error) {
|
||||
r, ok := f.rows[id]
|
||||
if !ok || from == "" || r.RunState != from {
|
||||
return false, nil
|
||||
}
|
||||
r.RunState, r.RunResume, r.RunStepAt = "", false, nil
|
||||
r.LastResult, r.LastDetail = result, detail
|
||||
return true, nil
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) MissScheduleRun(_ context.Context, id int64, due, next time.Time, detail string) (bool, error) {
|
||||
r, ok := f.rows[id]
|
||||
if !ok || r.RunState != "" || !r.Enabled || !sameTime(r.NextRunAt, due) {
|
||||
return false, nil
|
||||
}
|
||||
r.NextRunAt, r.WarnedFor = ptrTime(next), nil
|
||||
r.LastRunAt, r.LastResult, r.LastDetail = ptrTime(due), ScheduleMissed, detail
|
||||
return true, nil
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) WarnScheduleRun(_ context.Context, id int64, due time.Time) (bool, error) {
|
||||
r, ok := f.rows[id]
|
||||
if !ok || r.RunState != "" || !r.Enabled || !sameTime(r.NextRunAt, due) || sameTime(r.WarnedFor, due) {
|
||||
return false, nil
|
||||
}
|
||||
r.WarnedFor = ptrTime(due)
|
||||
return true, nil
|
||||
}
|
||||
|
||||
func (f *fakeSchedules) DisableSchedule(_ context.Context, id int64, detail string) (bool, error) {
|
||||
r, ok := f.rows[id]
|
||||
if !ok || r.RunState != "" || !r.Enabled {
|
||||
return false, nil
|
||||
}
|
||||
r.Enabled, r.NextRunAt, r.WarnedFor = false, nil, nil
|
||||
r.LastResult, r.LastDetail = ScheduleSkipped, detail
|
||||
return true, nil
|
||||
}
|
||||
|
||||
func mustZone(t *testing.T, name string) *time.Location {
|
||||
t.Helper()
|
||||
loc, err := time.LoadLocation(name)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return loc
|
||||
}
|
||||
|
||||
// TestScheduleNextRun pins the run times: daily and repeating rules, the
|
||||
// weekday mask read in the schedule's zone, and the two daylight-saving edges.
|
||||
func TestScheduleNextRun(t *testing.T) {
|
||||
sh := mustZone(t, "Asia/Shanghai")
|
||||
ny := mustZone(t, "America/New_York")
|
||||
const everyDay, weekdaysOnly, monday, saturday = 0x7f, 0x3e, 1 << 1, 1 << 6
|
||||
cases := []struct {
|
||||
name string
|
||||
s Schedule
|
||||
after time.Time
|
||||
want time.Time
|
||||
}{
|
||||
{"daily, later today",
|
||||
Schedule{MinuteOfDay: 4*60 + 30, Weekdays: everyDay, Timezone: "Asia/Shanghai"},
|
||||
time.Date(2026, 9, 28, 1, 0, 0, 0, sh), time.Date(2026, 9, 28, 4, 30, 0, 0, sh)},
|
||||
{"daily, today's run is past",
|
||||
Schedule{MinuteOfDay: 4 * 60, Weekdays: everyDay, Timezone: "Asia/Shanghai"},
|
||||
time.Date(2026, 9, 28, 4, 0, 0, 0, sh), time.Date(2026, 9, 29, 4, 0, 0, 0, sh)},
|
||||
{"weekday mask in the schedule's zone (Sunday 23:00 UTC is Monday in Shanghai)",
|
||||
Schedule{MinuteOfDay: 9 * 60, Weekdays: monday, Timezone: "Asia/Shanghai"},
|
||||
time.Date(2026, 9, 27, 23, 0, 0, 0, time.UTC), time.Date(2026, 9, 28, 9, 0, 0, 0, sh)},
|
||||
{"one weekday, next week, across the month end",
|
||||
Schedule{MinuteOfDay: 9 * 60, Weekdays: monday, Timezone: "Asia/Shanghai"},
|
||||
time.Date(2026, 9, 28, 10, 0, 0, 0, sh), time.Date(2026, 10, 5, 9, 0, 0, 0, sh)},
|
||||
{"interval: the next multiple since midnight",
|
||||
Schedule{EveryMinutes: 180, Weekdays: everyDay, Timezone: "Asia/Shanghai"},
|
||||
time.Date(2026, 9, 28, 7, 10, 0, 0, sh), time.Date(2026, 9, 28, 9, 0, 0, 0, sh)},
|
||||
{"interval skips the days off (Friday night -> Monday 00:00)",
|
||||
Schedule{EveryMinutes: 720, Weekdays: weekdaysOnly, Timezone: "Asia/Shanghai"},
|
||||
time.Date(2026, 10, 2, 12, 0, 0, 0, sh), time.Date(2026, 10, 5, 0, 0, 0, 0, sh)},
|
||||
{"interval, Saturday only, from the week before",
|
||||
Schedule{EveryMinutes: 15, Weekdays: saturday, Timezone: "UTC"},
|
||||
time.Date(2026, 9, 26, 23, 50, 0, 0, time.UTC), time.Date(2026, 10, 3, 0, 0, 0, 0, time.UTC)},
|
||||
{"a time daylight saving skips runs an hour early",
|
||||
Schedule{MinuteOfDay: 2*60 + 30, Weekdays: everyDay, Timezone: "America/New_York"},
|
||||
time.Date(2026, 3, 8, 0, 0, 0, 0, ny), time.Date(2026, 3, 8, 6, 30, 0, 0, time.UTC)},
|
||||
{"a repeated time runs once: after the first 01:30 comes tomorrow's",
|
||||
Schedule{MinuteOfDay: 90, Weekdays: everyDay, Timezone: "America/New_York"},
|
||||
time.Date(2026, 11, 1, 5, 30, 0, 0, time.UTC), time.Date(2026, 11, 2, 6, 30, 0, 0, time.UTC)},
|
||||
{"a zone no longer known runs in UTC",
|
||||
Schedule{MinuteOfDay: 60, Weekdays: everyDay, Timezone: "Gone/Zone"},
|
||||
time.Date(2026, 9, 28, 0, 0, 0, 0, time.UTC), time.Date(2026, 9, 28, 1, 0, 0, 0, time.UTC)},
|
||||
{"the first 01:30 on the day the clock goes back",
|
||||
Schedule{MinuteOfDay: 90, Weekdays: everyDay, Timezone: "America/New_York"},
|
||||
time.Date(2026, 11, 1, 0, 0, 0, 0, ny), time.Date(2026, 11, 1, 5, 30, 0, 0, time.UTC)},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
if got := tc.s.nextRun(tc.after); !got.Equal(tc.want) {
|
||||
t.Fatalf("nextRun(%s) = %s, want %s", tc.after, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestScheduleInputValidation walks the settings apply refuses, each with the
|
||||
// one field that makes it wrong, and what a valid body stores.
|
||||
func TestScheduleInputValidation(t *testing.T) {
|
||||
valid := func() scheduleInput {
|
||||
return scheduleInput{Action: ScheduleRestart, MinuteOfDay: 240, Weekdays: 0x7f, Timezone: "Asia/Shanghai", WarnMinutes: 5}
|
||||
}
|
||||
bad := []struct {
|
||||
name string
|
||||
edit func(*scheduleInput)
|
||||
code string
|
||||
msg string
|
||||
}{
|
||||
{"label too long", func(in *scheduleInput) { in.Label = strings.Repeat("猫", 65) }, "bad_schedule", "label too long"},
|
||||
{"label with a newline", func(in *scheduleInput) { in.Label = "a\nb" }, "bad_schedule", "single line"},
|
||||
{"label with DEL", func(in *scheduleInput) { in.Label = "a\x7fb" }, "bad_schedule", "single line"},
|
||||
{"unknown action", func(in *scheduleInput) { in.Action = "reboot" }, "bad_schedule", "action must be"},
|
||||
{"command on a restart", func(in *scheduleInput) { in.Command = "say hi" }, "bad_schedule", "only a command schedule"},
|
||||
{"command action without a command", func(in *scheduleInput) { in.Action, in.WarnMinutes = ScheduleCommand, 0 }, "bad_request", "command is required"},
|
||||
{"two commands in one", func(in *scheduleInput) { in.Action, in.WarnMinutes, in.Command = ScheduleCommand, 0, "say a\nop me" }, "bad_request", "single line"},
|
||||
{"minute_of_day negative", func(in *scheduleInput) { in.MinuteOfDay = -1 }, "bad_schedule", "minute_of_day must be"},
|
||||
{"minute_of_day 1440", func(in *scheduleInput) { in.MinuteOfDay = 1440 }, "bad_schedule", "minute_of_day must be"},
|
||||
{"interval off the list", func(in *scheduleInput) { in.EveryMinutes, in.MinuteOfDay = 45, 0 }, "bad_schedule", "every_minutes must be"},
|
||||
{"restart every 30 minutes", func(in *scheduleInput) { in.EveryMinutes, in.MinuteOfDay = 30, 0 }, "bad_schedule", "at most every 60 minutes"},
|
||||
{"interval with a minute_of_day", func(in *scheduleInput) { in.EveryMinutes = 60 }, "bad_schedule", "no minute_of_day"},
|
||||
{"no weekday", func(in *scheduleInput) { in.Weekdays = 0 }, "bad_schedule", "weekdays"},
|
||||
{"weekday bit 7", func(in *scheduleInput) { in.Weekdays = 0x80 }, "bad_schedule", "weekdays"},
|
||||
{"empty timezone", func(in *scheduleInput) { in.Timezone = " " }, "bad_schedule", "IANA zone"},
|
||||
{"Local", func(in *scheduleInput) { in.Timezone = "Local" }, "bad_schedule", "IANA zone"},
|
||||
{"unknown timezone", func(in *scheduleInput) { in.Timezone = "Mars/Olympus" }, "bad_schedule", "unknown timezone"},
|
||||
{"warning off the list", func(in *scheduleInput) { in.WarnMinutes = 2 }, "bad_schedule", "warn_minutes must be"},
|
||||
{"warning on a command", func(in *scheduleInput) { in.Action, in.Command = ScheduleCommand, "say hi" }, "bad_schedule", "only a restart, stop or backup warns"},
|
||||
{"warning on a start", func(in *scheduleInput) { in.Action = ScheduleStart }, "bad_schedule", "only a restart, stop or backup warns"},
|
||||
}
|
||||
for _, tc := range bad {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
in := valid()
|
||||
tc.edit(&in)
|
||||
err := in.apply(&Schedule{})
|
||||
var he *apiError
|
||||
if !errors.As(err, &he) || he.status != http.StatusBadRequest || he.code != tc.code || !strings.Contains(he.msg, tc.msg) {
|
||||
t.Fatalf("apply = %v, want 400 %s containing %q", err, tc.code, tc.msg)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
t.Run("valid bodies store trimmed values", func(t *testing.T) {
|
||||
in := scheduleInput{Label: " 每晚 ", Action: ScheduleCommand, Command: " /say 晚安 ", EveryMinutes: 15,
|
||||
Weekdays: 0x41, Timezone: " Europe/Berlin "}
|
||||
var s Schedule
|
||||
if err := in.apply(&s); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
want := Schedule{Label: "每晚", Action: ScheduleCommand, Command: "say 晚安", EveryMinutes: 15,
|
||||
Weekdays: 0x41, Timezone: "Europe/Berlin", Enabled: true}
|
||||
if s != want {
|
||||
t.Fatalf("stored %+v, want %+v", s, want)
|
||||
}
|
||||
off := false
|
||||
in = scheduleInput{Label: strings.Repeat("猫", 64), Action: ScheduleBackup, MinuteOfDay: 1439, Weekdays: 1,
|
||||
Timezone: "UTC", WarnMinutes: 30, Enabled: &off}
|
||||
if err := in.apply(&s); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if s.Enabled || s.Command != "" || s.WarnMinutes != 30 || s.MinuteOfDay != 1439 {
|
||||
t.Fatalf("stored %+v", s)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// schedRig is a server with a schedule store: survival, owned by owner1,
|
||||
// running, at 2026-09-28 03:00 UTC.
|
||||
type schedRig struct {
|
||||
a *API
|
||||
repo *fakeRepo
|
||||
cl *fakeCluster
|
||||
st *fakeSchedules
|
||||
con *fakeConsole
|
||||
b *fakeScheduledBackuper
|
||||
jobs *fakeJobStatus
|
||||
clock time.Time
|
||||
}
|
||||
|
||||
var schedT0 = time.Date(2026, 9, 28, 3, 0, 0, 0, time.UTC)
|
||||
|
||||
func newSchedRig(t *testing.T) *schedRig {
|
||||
t.Helper()
|
||||
r := &schedRig{repo: newFakeRepo(), cl: newFakeCluster(), st: newFakeSchedules(), con: &fakeConsole{},
|
||||
b: &fakeScheduledBackuper{}, jobs: &fakeJobStatus{}, clock: schedT0}
|
||||
r.repo.byName["survival"] = &ServerRecord{Name: "survival", OwnerID: "owner1"}
|
||||
r.cl.byName["survival"] = &ServerInfo{Name: "survival", Phase: string(v1alpha1.PhaseRunning), Ready: true,
|
||||
DesiredState: string(v1alpha1.DesiredRunning)}
|
||||
r.st.owners["survival"] = "owner1"
|
||||
r.a = newTestAPI(r.repo, r.cl)
|
||||
r.a.Now = func() time.Time { return r.clock }
|
||||
r.a.Schedules, r.a.Console, r.a.Backuper, r.a.JobStatus = r.st, r.con, r.b, r.jobs
|
||||
return r
|
||||
}
|
||||
|
||||
func (r *schedRig) stopped() {
|
||||
info := r.cl.byName["survival"]
|
||||
info.Phase, info.Ready, info.DesiredState = string(v1alpha1.PhaseStopped), false, string(v1alpha1.DesiredStopped)
|
||||
}
|
||||
|
||||
// schedule stores a schedule of survival owned by owner1, due at due.
|
||||
func (r *schedRig) schedule(action string, due time.Time, edit ...func(*Schedule)) int64 {
|
||||
s := Schedule{Server: "survival", OwnerID: "owner1", Action: action, MinuteOfDay: due.Hour()*60 + due.Minute(),
|
||||
Weekdays: 0x7f, Timezone: "UTC", Enabled: true, NextRunAt: ptrTime(due), CreatedBy: "[email protected]"}
|
||||
if action == ScheduleCommand {
|
||||
s.Command = "say hi"
|
||||
}
|
||||
for _, e := range edit {
|
||||
e(&s)
|
||||
}
|
||||
return r.st.put(s)
|
||||
}
|
||||
|
||||
func (r *schedRig) tick(t *testing.T) {
|
||||
t.Helper()
|
||||
if err := r.a.RunSchedules(context.Background()); err != nil {
|
||||
t.Fatalf("RunSchedules: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
var (
|
||||
schedOwner = &Principal{UserID: "owner1", Email: "[email protected]", Role: "user"}
|
||||
schedStranger = &Principal{UserID: "other", Email: "[email protected]", Role: "user"}
|
||||
schedAdmin = &Principal{UserID: "adm", Email: "[email protected]", Role: "admin", ViaAdminAccess: true}
|
||||
)
|
||||
|
||||
func (r *schedRig) do(p *Principal, method, target, body string) *http.Response {
|
||||
r.a.External = staticExternal{p: p}
|
||||
var h map[string]string
|
||||
if body != "" {
|
||||
h = jsonHeader
|
||||
}
|
||||
return do(r.a.ExternalHandler(), method, target, body, h).Result()
|
||||
}
|
||||
|
||||
func decodeSchedule(t *testing.T, res *http.Response) Schedule {
|
||||
t.Helper()
|
||||
var s Schedule
|
||||
if err := json.NewDecoder(res.Body).Decode(&s); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func schedErrCode(t *testing.T, res *http.Response) string {
|
||||
t.Helper()
|
||||
var raw map[string]map[string]string
|
||||
if err := json.NewDecoder(res.Body).Decode(&raw); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return raw["error"]["code"]
|
||||
}
|
||||
|
||||
const schedBase = "/api/v1/servers/survival/schedules"
|
||||
|
||||
// TestScheduleRoutesGate: every schedule route answers 400 for a bad name, 404
|
||||
// for an unknown server, 403 for a stranger and 503 without a store, before it
|
||||
// reads anything else.
|
||||
func TestScheduleRoutesGate(t *testing.T) {
|
||||
body := `{"action":"restart","minute_of_day":240,"weekdays":127,"timezone":"UTC"}`
|
||||
routes := []struct{ method, path, body string }{
|
||||
{"GET", "", ""}, {"POST", "", body}, {"PUT", "/1", body}, {"DELETE", "/1", ""}, {"POST", "/1/run", ""},
|
||||
}
|
||||
for _, rt := range routes {
|
||||
t.Run(rt.method+rt.path, func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.schedule(ScheduleRestart, schedT0.Add(time.Hour))
|
||||
check := func(p *Principal, target string, status int, code string) {
|
||||
t.Helper()
|
||||
res := r.do(p, rt.method, target, rt.body)
|
||||
if res.StatusCode != status || schedErrCode(t, res) != code {
|
||||
t.Fatalf("%s %s = %d, want %d %s", rt.method, target, res.StatusCode, status, code)
|
||||
}
|
||||
}
|
||||
check(schedOwner, "/api/v1/servers/Bad_Name/schedules"+rt.path, 400, "bad_name")
|
||||
check(schedOwner, "/api/v1/servers/nope/schedules"+rt.path, 404, "not_found")
|
||||
check(schedStranger, schedBase+rt.path, 403, "forbidden")
|
||||
r.a.Schedules = nil
|
||||
check(schedOwner, schedBase+rt.path, 503, "schedules_unavailable")
|
||||
if len(r.repo.audits) != 0 || r.con.calls != 0 {
|
||||
t.Fatalf("a refused request left audits %+v / %d console calls", r.repo.audits, r.con.calls)
|
||||
}
|
||||
})
|
||||
}
|
||||
t.Run("bad id", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
for _, id := range []string{"0", "-1", "x"} {
|
||||
res := r.do(schedOwner, "DELETE", schedBase+"/"+id, "")
|
||||
if res.StatusCode != 400 || schedErrCode(t, res) != "bad_id" {
|
||||
t.Fatalf("DELETE %s = %d, want 400 bad_id", id, res.StatusCode)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestScheduleCRUD(t *testing.T) {
|
||||
t.Run("list: empty is [] with the limit", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
res := r.do(schedOwner, "GET", schedBase, "")
|
||||
raw, _ := io.ReadAll(res.Body)
|
||||
if res.StatusCode != 200 || string(raw) != `{"limit":20,"schedules":[],"server":"survival"}`+"\n" {
|
||||
t.Fatalf("GET = %d %s", res.StatusCode, raw)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("owner creates: next run, owner binding, audit", func(t *testing.T) {
|
||||
r := newSchedRig(t) // 2026-09-28 03:00 UTC = 11:00 in Shanghai
|
||||
res := r.do(schedOwner, "POST", schedBase,
|
||||
`{"label":"夜间重启","action":"restart","minute_of_day":240,"weekdays":127,"timezone":"Asia/Shanghai","warn_minutes":5}`)
|
||||
if res.StatusCode != 201 {
|
||||
t.Fatalf("POST = %d", res.StatusCode)
|
||||
}
|
||||
got := decodeSchedule(t, res)
|
||||
want := time.Date(2026, 9, 28, 20, 0, 0, 0, time.UTC) // 04:00 on the 29th in Shanghai
|
||||
if got.ID != 1 || got.Label != "夜间重启" || !sameTime(got.NextRunAt, want) || !got.Enabled ||
|
||||
got.CreatedBy != "[email protected]" || got.LastRunAt != nil || got.RunState != "" {
|
||||
t.Fatalf("created %+v", got)
|
||||
}
|
||||
if row := r.st.row(t, 1); row.OwnerID != "owner1" {
|
||||
t.Fatalf("stored owner %q, want owner1", row.OwnerID)
|
||||
}
|
||||
if len(r.repo.audits) != 1 || r.repo.audits[0].Action != "schedule.create" ||
|
||||
r.repo.audits[0].Actor != "[email protected]" || r.repo.audits[0].ActorUserID != "owner1" || r.repo.audits[0].ServerName != "survival" ||
|
||||
string(r.repo.audits[0].Payload) != `{"action":"restart","schedule":1}` {
|
||||
t.Fatalf("audits %+v", r.repo.audits)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("admin creates on an owned server: it belongs to the owner", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
res := r.do(schedAdmin, "POST", schedBase, `{"action":"stop","minute_of_day":0,"weekdays":1,"timezone":"UTC","enabled":false}`)
|
||||
if res.StatusCode != 201 {
|
||||
t.Fatalf("POST = %d", res.StatusCode)
|
||||
}
|
||||
got := decodeSchedule(t, res)
|
||||
if got.NextRunAt != nil || got.Enabled || got.CreatedBy != "[email protected]" {
|
||||
t.Fatalf("created %+v", got)
|
||||
}
|
||||
if row := r.st.row(t, got.ID); row.OwnerID != "owner1" {
|
||||
t.Fatalf("stored owner %q, want owner1", row.OwnerID)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("bad settings and unknown fields are 400 and store nothing", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
for body, code := range map[string]string{
|
||||
`{"action":"restart","minute_of_day":240,"weekdays":127,"timezone":"Nowhere/City"}`: "bad_schedule",
|
||||
`{"action":"command","weekdays":127,"timezone":"UTC"}`: "bad_request",
|
||||
`{"action":"restart","weekdays":127,"timezone":"UTC","owner_id":"me"}`: "bad_request",
|
||||
} {
|
||||
res := r.do(schedOwner, "POST", schedBase, body)
|
||||
if res.StatusCode != 400 || schedErrCode(t, res) != code {
|
||||
t.Fatalf("POST %s = %d, want 400 %s", body, res.StatusCode, code)
|
||||
}
|
||||
}
|
||||
if len(r.st.rows) != 0 || len(r.repo.audits) != 0 {
|
||||
t.Fatalf("stored %d rows, %d audits", len(r.st.rows), len(r.repo.audits))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("the 21st is 409 schedule_limit", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
for range maxSchedulesPerServer {
|
||||
r.schedule(ScheduleStop, schedT0.Add(time.Hour))
|
||||
}
|
||||
res := r.do(schedOwner, "POST", schedBase, `{"action":"stop","weekdays":127,"timezone":"UTC"}`)
|
||||
if res.StatusCode != 409 || schedErrCode(t, res) != "schedule_limit" || len(r.st.rows) != maxSchedulesPerServer {
|
||||
t.Fatalf("POST = %d, rows %d", res.StatusCode, len(r.st.rows))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("update: new settings, current owner, fresh next run and warning marker", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleRestart, schedT0.Add(time.Hour), func(s *Schedule) {
|
||||
s.OwnerID, s.WarnedFor = "previous", ptrTime(schedT0.Add(time.Hour))
|
||||
})
|
||||
res := r.do(schedOwner, "PUT", fmt.Sprintf("%s/%d", schedBase, id),
|
||||
`{"action":"command","command":"/save-all","every_minutes":30,"weekdays":127,"timezone":"UTC"}`)
|
||||
if res.StatusCode != 200 {
|
||||
t.Fatalf("PUT = %d", res.StatusCode)
|
||||
}
|
||||
got := decodeSchedule(t, res)
|
||||
row := r.st.row(t, id)
|
||||
if got.Action != ScheduleCommand || got.Command != "save-all" || !sameTime(got.NextRunAt, schedT0.Add(30*time.Minute)) ||
|
||||
row.OwnerID != "owner1" || row.WarnedFor != nil || row.Command != "save-all" {
|
||||
t.Fatalf("PUT answered %+v, stored %+v", got, row)
|
||||
}
|
||||
if len(r.repo.audits) != 1 || r.repo.audits[0].Action != "schedule.update" {
|
||||
t.Fatalf("audits %+v", r.repo.audits)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("update and delete refuse a running or unknown schedule", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleRestart, schedT0.Add(time.Hour), func(s *Schedule) { s.RunState = runStopping })
|
||||
body := `{"action":"stop","weekdays":127,"timezone":"UTC"}`
|
||||
for _, c := range []struct{ method, path, body, code string }{
|
||||
{"PUT", fmt.Sprintf("/%d", id), body, "schedule_running"},
|
||||
{"DELETE", fmt.Sprintf("/%d", id), "", "schedule_running"},
|
||||
{"PUT", "/99", body, "not_found"},
|
||||
{"DELETE", "/99", "", "not_found"},
|
||||
} {
|
||||
res := r.do(schedOwner, c.method, schedBase+c.path, c.body)
|
||||
want := 409
|
||||
if c.code == "not_found" {
|
||||
want = 404
|
||||
}
|
||||
if res.StatusCode != want || schedErrCode(t, res) != c.code {
|
||||
t.Fatalf("%s %s = %d, want %d %s", c.method, c.path, res.StatusCode, want, c.code)
|
||||
}
|
||||
}
|
||||
if row := r.st.row(t, id); row.Action != ScheduleRestart || len(r.repo.audits) != 0 {
|
||||
t.Fatalf("refused writes changed %+v / audited %+v", row, r.repo.audits)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a schedule of another server is 404", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.repo.byName["creative"] = &ServerRecord{Name: "creative", OwnerID: "owner1"}
|
||||
id := r.st.put(Schedule{Server: "creative", Action: ScheduleStop, Weekdays: 1, Timezone: "UTC"})
|
||||
res := r.do(schedOwner, "DELETE", fmt.Sprintf("%s/%d", schedBase, id), "")
|
||||
if res.StatusCode != 404 || len(r.st.rows) != 1 {
|
||||
t.Fatalf("DELETE = %d, rows %d", res.StatusCode, len(r.st.rows))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("delete", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleStop, schedT0.Add(time.Hour))
|
||||
res := r.do(schedOwner, "DELETE", fmt.Sprintf("%s/%d", schedBase, id), "")
|
||||
if res.StatusCode != 204 || len(r.st.rows) != 0 {
|
||||
t.Fatalf("DELETE = %d, rows %d", res.StatusCode, len(r.st.rows))
|
||||
}
|
||||
if len(r.repo.audits) != 1 || r.repo.audits[0].Action != "schedule.delete" ||
|
||||
string(r.repo.audits[0].Payload) != fmt.Sprintf(`{"action":"stop","schedule":%d}`, id) {
|
||||
t.Fatalf("audits %+v", r.repo.audits)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("list shows the schedules oldest first", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
a := r.schedule(ScheduleStop, schedT0.Add(2*time.Hour))
|
||||
b := r.schedule(ScheduleStart, schedT0.Add(time.Hour))
|
||||
res := r.do(schedOwner, "GET", schedBase, "")
|
||||
var body struct {
|
||||
Schedules []Schedule `json:"schedules"`
|
||||
}
|
||||
if err := json.NewDecoder(res.Body).Decode(&body); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(body.Schedules) != 2 || body.Schedules[0].ID != a || body.Schedules[1].ID != b {
|
||||
t.Fatalf("listed %+v", body.Schedules)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestScheduleRunNow: a run on request starts at once, without the warning,
|
||||
// whether or not the schedule is enabled, and leaves its next run alone.
|
||||
func TestScheduleRunNow(t *testing.T) {
|
||||
t.Run("a command runs and the answer carries its outcome", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
r.con.reply = "§aSaved the game"
|
||||
next := schedT0.Add(5 * time.Hour)
|
||||
id := r.schedule(ScheduleCommand, next, func(s *Schedule) { s.Command = "save-all" })
|
||||
res := r.do(schedOwner, "POST", fmt.Sprintf("%s/%d/run", schedBase, id), "")
|
||||
if res.StatusCode != 202 {
|
||||
t.Fatalf("run = %d", res.StatusCode)
|
||||
}
|
||||
got := decodeSchedule(t, res)
|
||||
if got.LastResult != ScheduleOK || got.LastDetail != "Saved the game" || got.RunState != "" ||
|
||||
!sameTime(got.LastRunAt, schedT0) || !sameTime(got.NextRunAt, next) {
|
||||
t.Fatalf("after the run %+v", got)
|
||||
}
|
||||
if r.con.gotCommand != "save-all" || r.con.calls != 1 {
|
||||
t.Fatalf("console ran %q (%d calls)", r.con.gotCommand, r.con.calls)
|
||||
}
|
||||
var actions []string
|
||||
for _, e := range r.repo.audits {
|
||||
actions = append(actions, e.Action+"/"+e.Actor)
|
||||
}
|
||||
if strings.Join(actions, ",") != "schedule.run_now/[email protected],schedule.run/scheduler" {
|
||||
t.Fatalf("audits %v", actions)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a disabled restart starts and goes on in the background", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleRestart, schedT0, func(s *Schedule) { s.Enabled, s.NextRunAt, s.WarnMinutes = false, nil, 5 })
|
||||
res := r.do(schedOwner, "POST", fmt.Sprintf("%s/%d/run", schedBase, id), "")
|
||||
got := decodeSchedule(t, res)
|
||||
if res.StatusCode != 202 || got.RunState != runStopping || got.NextRunAt != nil {
|
||||
t.Fatalf("run = %d %+v", res.StatusCode, got)
|
||||
}
|
||||
if r.cl.desired["survival"] != v1alpha1.DesiredStopped || r.con.calls != 0 {
|
||||
t.Fatalf("desired %q, %d console calls (no warning on request)", r.cl.desired["survival"], r.con.calls)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a running schedule is 409 schedule_running", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleRestart, schedT0, func(s *Schedule) { s.RunState = runStarting })
|
||||
res := r.do(schedOwner, "POST", fmt.Sprintf("%s/%d/run", schedBase, id), "")
|
||||
if res.StatusCode != 409 || schedErrCode(t, res) != "schedule_running" || len(r.repo.audits) != 0 {
|
||||
t.Fatalf("run = %d, audits %+v", res.StatusCode, r.repo.audits)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("a schedule of the previous owner is 409 schedule_stale", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
id := r.schedule(ScheduleCommand, schedT0, func(s *Schedule) { s.OwnerID = "previous" })
|
||||
res := r.do(schedAdmin, "POST", fmt.Sprintf("%s/%d/run", schedBase, id), "")
|
||||
if res.StatusCode != 409 || schedErrCode(t, res) != "schedule_stale" || r.con.calls != 0 || r.st.claims != 0 {
|
||||
t.Fatalf("run = %d, console %d, claims %d", res.StatusCode, r.con.calls, r.st.claims)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("unknown schedule is 404", func(t *testing.T) {
|
||||
r := newSchedRig(t)
|
||||
res := r.do(schedOwner, "POST", schedBase+"/7/run", "")
|
||||
if res.StatusCode != 404 {
|
||||
t.Fatalf("run = %d", res.StatusCode)
|
||||
}
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,393 @@
|
||||
//go:build pgint
|
||||
|
||||
package pgint
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"felis.lolicon.best/internal/api"
|
||||
)
|
||||
|
||||
func newSchedule(server, owner, action string, next *time.Time) *api.Schedule {
|
||||
return &api.Schedule{Server: server, OwnerID: owner, Label: "每晚", Action: action, MinuteOfDay: 240,
|
||||
Weekdays: 0x7f, Timezone: "Asia/Shanghai", WarnMinutes: 5, Enabled: next != nil, NextRunAt: next,
|
||||
CreatedBy: "pgint"}
|
||||
}
|
||||
|
||||
func mustCreateSchedule(t *testing.T, s *api.Schedule) *api.Schedule {
|
||||
t.Helper()
|
||||
if err := repo.CreateSchedule(context.Background(), s, 20); err != nil {
|
||||
t.Fatalf("CreateSchedule(%s on %s): %v", s.Action, s.Server, err)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func mustGetSchedule(t *testing.T, server string, id int64) *api.Schedule {
|
||||
t.Helper()
|
||||
s, err := repo.GetSchedule(context.Background(), server, id)
|
||||
if err != nil {
|
||||
t.Fatalf("GetSchedule(%d): %v", id, err)
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
// runView is the run columns of a schedule, for exact comparisons.
|
||||
func runView(s *api.Schedule) string {
|
||||
ts := func(p *time.Time) string {
|
||||
if p == nil {
|
||||
return "-"
|
||||
}
|
||||
return p.UTC().Format(time.RFC3339)
|
||||
}
|
||||
return fmt.Sprintf("state=%q resume=%v step=%s last=%s result=%q detail=%q next=%s warned=%s enabled=%v",
|
||||
s.RunState, s.RunResume, ts(s.RunStepAt), ts(s.LastRunAt), s.LastResult, s.LastDetail,
|
||||
ts(s.NextRunAt), ts(s.WarnedFor), s.Enabled)
|
||||
}
|
||||
|
||||
func applied(t *testing.T, what string, ok bool, err error, want bool) {
|
||||
t.Helper()
|
||||
if err != nil || ok != want {
|
||||
t.Fatalf("%s = %v, %v; want applied=%v", what, ok, err, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestScheduleStoreCRUD pins the settings side of server_schedules: a round
|
||||
// trip of every column, the per-server limit, the run-state guard on changes,
|
||||
// and the lookups scoped to their server.
|
||||
func TestScheduleStoreCRUD(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
u := newUser(t, "user", "sch-crud")
|
||||
sfx := suffix(t)
|
||||
name, other, gone := "sc-"+sfx, "sco-"+sfx, "scg-"+sfx
|
||||
seedOwnedServer(t, name, u.ID, false)
|
||||
seedOwnedServer(t, other, u.ID, false)
|
||||
seedOwnedServer(t, gone, u.ID, true)
|
||||
next := mustNow().Add(time.Hour).Truncate(time.Second)
|
||||
|
||||
s := mustCreateSchedule(t, newSchedule(name, u.ID, api.ScheduleRestart, &next))
|
||||
if s.ID == 0 || time.Since(s.CreatedAt) > time.Minute {
|
||||
t.Fatalf("created %+v", s)
|
||||
}
|
||||
settings := func(s *api.Schedule) string {
|
||||
return fmt.Sprintf("%d %s owner=%q label=%q %s cmd=%q every=%d at=%d days=%d tz=%s warn=%d by=%s created=%s",
|
||||
s.ID, s.Server, s.OwnerID, s.Label, s.Action, s.Command, s.EveryMinutes, s.MinuteOfDay, s.Weekdays,
|
||||
s.Timezone, s.WarnMinutes, s.CreatedBy, s.CreatedAt.UTC().Format(time.RFC3339Nano))
|
||||
}
|
||||
got := mustGetSchedule(t, name, s.ID)
|
||||
if settings(got) != settings(s) {
|
||||
t.Fatalf("read back %s\nwant %s", settings(got), settings(s))
|
||||
}
|
||||
if v, want := runView(got), `state="" resume=false step=- last=- result="" detail="" next=`+next.UTC().Format(time.RFC3339)+` warned=- enabled=true`; v != want {
|
||||
t.Fatalf("read back %s\nwant %s", v, want)
|
||||
}
|
||||
if _, err := repo.GetSchedule(ctx, other, s.ID); !errors.Is(err, api.ErrNotFound) {
|
||||
t.Fatalf("GetSchedule on another server = %v, want ErrNotFound", err)
|
||||
}
|
||||
|
||||
unowned := mustCreateSchedule(t, newSchedule(other, "", api.ScheduleStop, nil))
|
||||
if g := mustGetSchedule(t, other, unowned.ID); g.OwnerID != "" || g.Enabled || g.NextRunAt != nil {
|
||||
t.Fatalf("unowned, disabled schedule read back %+v", g)
|
||||
}
|
||||
if err := repo.CreateSchedule(ctx, newSchedule(gone, u.ID, api.ScheduleStop, nil), 20); !errors.Is(err, api.ErrNotFound) {
|
||||
t.Fatalf("CreateSchedule on a deleted server = %v, want ErrNotFound", err)
|
||||
}
|
||||
if err := repo.CreateSchedule(ctx, newSchedule("nope-"+sfx, u.ID, api.ScheduleStop, nil), 20); !errors.Is(err, api.ErrNotFound) {
|
||||
t.Fatalf("CreateSchedule on an unknown server = %v, want ErrNotFound", err)
|
||||
}
|
||||
|
||||
// The limit counts the server's own rows only.
|
||||
mustCreateSchedule(t, newSchedule(name, u.ID, api.ScheduleStop, nil))
|
||||
if err := repo.CreateSchedule(ctx, newSchedule(name, u.ID, api.ScheduleStart, nil), 2); !errors.Is(err, api.ErrScheduleLimit) {
|
||||
t.Fatalf("third schedule under a limit of 2 = %v, want ErrScheduleLimit", err)
|
||||
}
|
||||
if err := repo.CreateSchedule(ctx, newSchedule(other, u.ID, api.ScheduleStart, nil), 2); err != nil {
|
||||
t.Fatalf("second schedule of the other server under a limit of 2 = %v", err)
|
||||
}
|
||||
|
||||
// Update writes the settings and owner and clears the warning marker.
|
||||
if ok, err := repo.WarnScheduleRun(ctx, s.ID, next); err != nil || !ok {
|
||||
t.Fatalf("WarnScheduleRun = %v, %v", ok, err)
|
||||
}
|
||||
upd := mustGetSchedule(t, name, s.ID)
|
||||
upd.OwnerID, upd.Label, upd.Action, upd.Command = "", "", api.ScheduleCommand, "say hi"
|
||||
upd.EveryMinutes, upd.MinuteOfDay, upd.Weekdays, upd.Timezone, upd.WarnMinutes = 30, 0, 0x41, "UTC", 0
|
||||
upd.Enabled, upd.NextRunAt = false, nil
|
||||
if err := repo.UpdateSchedule(ctx, upd); err != nil {
|
||||
t.Fatalf("UpdateSchedule: %v", err)
|
||||
}
|
||||
g := mustGetSchedule(t, name, s.ID)
|
||||
if g.OwnerID != "" || g.Label != "" || g.Action != api.ScheduleCommand || g.Command != "say hi" || g.EveryMinutes != 30 ||
|
||||
g.MinuteOfDay != 0 || g.Weekdays != 0x41 || g.Timezone != "UTC" || g.WarnMinutes != 0 || g.Enabled ||
|
||||
g.NextRunAt != nil || g.WarnedFor != nil {
|
||||
t.Fatalf("after update %+v", g)
|
||||
}
|
||||
wrong := *g
|
||||
wrong.Server = other
|
||||
if err := repo.UpdateSchedule(ctx, &wrong); !errors.Is(err, api.ErrNotFound) {
|
||||
t.Fatalf("UpdateSchedule under another server = %v, want ErrNotFound", err)
|
||||
}
|
||||
|
||||
// A running schedule refuses changes and deletion; an idle one is deleted.
|
||||
now := mustNow().Truncate(time.Second)
|
||||
ok, err := mustClaim(t, s.ID, nil, nil, now)
|
||||
applied(t, "claim", ok, err, true)
|
||||
if err := repo.UpdateSchedule(ctx, g); !errors.Is(err, api.ErrScheduleRunning) {
|
||||
t.Fatalf("UpdateSchedule while running = %v, want ErrScheduleRunning", err)
|
||||
}
|
||||
if err := repo.DeleteSchedule(ctx, name, s.ID); !errors.Is(err, api.ErrScheduleRunning) {
|
||||
t.Fatalf("DeleteSchedule while running = %v, want ErrScheduleRunning", err)
|
||||
}
|
||||
ok, err = repo.FinishScheduleRun(ctx, s.ID, "claimed", api.ScheduleOK, "")
|
||||
applied(t, "finish", ok, err, true)
|
||||
if err := repo.DeleteSchedule(ctx, other, s.ID); !errors.Is(err, api.ErrNotFound) {
|
||||
t.Fatalf("DeleteSchedule under another server = %v, want ErrNotFound", err)
|
||||
}
|
||||
if err := repo.DeleteSchedule(ctx, name, s.ID); err != nil {
|
||||
t.Fatalf("DeleteSchedule: %v", err)
|
||||
}
|
||||
if err := repo.DeleteSchedule(ctx, name, s.ID); !errors.Is(err, api.ErrNotFound) {
|
||||
t.Fatalf("second DeleteSchedule = %v, want ErrNotFound", err)
|
||||
}
|
||||
list, err := repo.ListSchedules(ctx, other)
|
||||
if err != nil || len(list) != 2 || list[0].ID != unowned.ID || list[0].ID >= list[1].ID {
|
||||
t.Fatalf("ListSchedules(other) = %+v, %v; want 2, oldest first", list, err)
|
||||
}
|
||||
}
|
||||
|
||||
func mustClaim(t *testing.T, id int64, due, next *time.Time, now time.Time) (bool, error) {
|
||||
t.Helper()
|
||||
return repo.ClaimScheduleRun(context.Background(), id, due, next, now)
|
||||
}
|
||||
|
||||
// TestScheduleStoreRunCAS pins the compare-and-set writes of a run, which
|
||||
// keep two felis-api processes from firing one run twice.
|
||||
func TestScheduleStoreRunCAS(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
u := newUser(t, "user", "sch-cas")
|
||||
name := "scc-" + suffix(t)
|
||||
seedOwnedServer(t, name, u.ID, false)
|
||||
t0 := mustNow().Truncate(time.Second)
|
||||
due, next, later := t0.Add(-time.Minute), t0.Add(23*time.Hour), t0.Add(47*time.Hour)
|
||||
s := mustCreateSchedule(t, newSchedule(name, u.ID, api.ScheduleBackup, &due))
|
||||
ts := func(t time.Time) string { return t.UTC().Format(time.RFC3339) }
|
||||
|
||||
// Warned once for the run due then.
|
||||
ok, err := repo.WarnScheduleRun(ctx, s.ID, next)
|
||||
applied(t, "warn for another due time", ok, err, false)
|
||||
ok, err = repo.WarnScheduleRun(ctx, s.ID, due)
|
||||
applied(t, "warn", ok, err, true)
|
||||
ok, err = repo.WarnScheduleRun(ctx, s.ID, due)
|
||||
applied(t, "second warn", ok, err, false)
|
||||
|
||||
// A claim for another due time loses; the right one wins once.
|
||||
ok, err = mustClaim(t, s.ID, &next, &later, t0)
|
||||
applied(t, "claim a stale due time", ok, err, false)
|
||||
ok, err = mustClaim(t, s.ID, &due, &next, t0)
|
||||
applied(t, "claim", ok, err, true)
|
||||
ok, err = mustClaim(t, s.ID, &due, &next, t0)
|
||||
applied(t, "second claim", ok, err, false)
|
||||
ok, err = mustClaim(t, s.ID, nil, nil, t0)
|
||||
applied(t, "claim on request during a run", ok, err, false)
|
||||
if got, want := runView(mustGetSchedule(t, name, s.ID)),
|
||||
fmt.Sprintf(`state="claimed" resume=false step=%s last=%s result="" detail="" next=%s warned=- enabled=true`, ts(t0), ts(t0), ts(next)); got != want {
|
||||
t.Fatalf("after claim %s\nwant %s", got, want)
|
||||
}
|
||||
|
||||
// Advance from the wrong step loses; the right one keeps the outcome so far
|
||||
// unless it brings one.
|
||||
t1 := t0.Add(time.Minute)
|
||||
ok, err = repo.AdvanceScheduleRun(ctx, s.ID, "stopping", "backing_up", true, "", "", t1)
|
||||
applied(t, "advance from the wrong step", ok, err, false)
|
||||
ok, err = repo.AdvanceScheduleRun(ctx, s.ID, "claimed", "stopping", true, "", "", t1)
|
||||
applied(t, "advance", ok, err, true)
|
||||
t2 := t1.Add(time.Minute)
|
||||
ok, err = repo.AdvanceScheduleRun(ctx, s.ID, "stopping", "starting", true, api.ScheduleFailed, "the backup failed: x", t2)
|
||||
applied(t, "advance with an outcome", ok, err, true)
|
||||
t3 := t2.Add(time.Minute)
|
||||
ok, err = repo.AdvanceScheduleRun(ctx, s.ID, "starting", "starting", true, "", "", t3)
|
||||
applied(t, "advance without one", ok, err, true)
|
||||
if got, want := runView(mustGetSchedule(t, name, s.ID)),
|
||||
fmt.Sprintf(`state="starting" resume=true step=%s last=%s result="failed" detail="the backup failed: x" next=%s warned=- enabled=true`, ts(t3), ts(t0), ts(next)); got != want {
|
||||
t.Fatalf("after advancing %s\nwant %s", got, want)
|
||||
}
|
||||
if _, err := db.ExecContext(ctx, `UPDATE server_schedules SET run_step_at = NULL WHERE id = $1`, s.ID); err == nil ||
|
||||
!strings.Contains(err.Error(), "check") {
|
||||
t.Fatalf("a run step without its time = %v, want a check violation", err)
|
||||
}
|
||||
|
||||
// Finish from the wrong step, or from none, loses.
|
||||
ok, err = repo.FinishScheduleRun(ctx, s.ID, "stopping", api.ScheduleOK, "")
|
||||
applied(t, "finish from the wrong step", ok, err, false)
|
||||
ok, err = repo.FinishScheduleRun(ctx, s.ID, "starting", api.ScheduleFailed, "done")
|
||||
applied(t, "finish", ok, err, true)
|
||||
ok, err = repo.FinishScheduleRun(ctx, s.ID, "", api.ScheduleOK, "")
|
||||
applied(t, "finish an idle schedule", ok, err, false)
|
||||
if got, want := runView(mustGetSchedule(t, name, s.ID)),
|
||||
fmt.Sprintf(`state="" resume=false step=- last=%s result="failed" detail="done" next=%s warned=- enabled=true`, ts(t0), ts(next)); got != want {
|
||||
t.Fatalf("after finishing %s\nwant %s", got, want)
|
||||
}
|
||||
|
||||
// A run on request leaves the next run alone and works while disabled.
|
||||
ok, err = mustClaim(t, s.ID, nil, nil, t3)
|
||||
applied(t, "claim on request", ok, err, true)
|
||||
if g := mustGetSchedule(t, name, s.ID); g.NextRunAt == nil || !g.NextRunAt.Equal(next) || g.RunState != "claimed" || g.LastResult != "" {
|
||||
t.Fatalf("after a claim on request %s", runView(g))
|
||||
}
|
||||
ok, err = repo.DisableSchedule(ctx, s.ID, "x")
|
||||
applied(t, "disable during a run", ok, err, false)
|
||||
ok, err = repo.WarnScheduleRun(ctx, s.ID, next)
|
||||
applied(t, "warn during a run", ok, err, false)
|
||||
ok, err = mustClaim(t, s.ID, &next, &later, t3)
|
||||
applied(t, "claim the due run during a run on request", ok, err, false)
|
||||
ok, err = repo.MissScheduleRun(ctx, s.ID, next, later, "x")
|
||||
applied(t, "miss during a run", ok, err, false)
|
||||
ok, err = repo.FinishScheduleRun(ctx, s.ID, "claimed", api.ScheduleOK, "")
|
||||
applied(t, "finish the run on request", ok, err, true)
|
||||
|
||||
// Missed: only the run due then, recorded at its due time.
|
||||
ok, err = repo.MissScheduleRun(ctx, s.ID, due, later, "x")
|
||||
applied(t, "miss a stale due time", ok, err, false)
|
||||
ok, err = repo.WarnScheduleRun(ctx, s.ID, next)
|
||||
applied(t, "warn the next run", ok, err, true)
|
||||
ok, err = repo.MissScheduleRun(ctx, s.ID, next, later, "felis-api was down")
|
||||
applied(t, "miss", ok, err, true)
|
||||
if got, want := runView(mustGetSchedule(t, name, s.ID)),
|
||||
fmt.Sprintf(`state="" resume=false step=- last=%s result="missed" detail="felis-api was down" next=%s warned=- enabled=true`, ts(next), ts(later)); got != want {
|
||||
t.Fatalf("after a miss %s\nwant %s", got, want)
|
||||
}
|
||||
|
||||
// Disabled: once, and a disabled schedule is neither claimed nor missed.
|
||||
ok, err = repo.DisableSchedule(ctx, s.ID, "new owner")
|
||||
applied(t, "disable", ok, err, true)
|
||||
ok, err = repo.DisableSchedule(ctx, s.ID, "again")
|
||||
applied(t, "second disable", ok, err, false)
|
||||
if got, want := runView(mustGetSchedule(t, name, s.ID)),
|
||||
fmt.Sprintf(`state="" resume=false step=- last=%s result="skipped" detail="new owner" next=- warned=- enabled=false`, ts(next)); got != want {
|
||||
t.Fatalf("after disabling %s\nwant %s", got, want)
|
||||
}
|
||||
if _, err := db.ExecContext(ctx, `UPDATE server_schedules SET next_run_at = $2 WHERE id = $1`, s.ID, later); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
ok, err = mustClaim(t, s.ID, &later, &later, t3)
|
||||
applied(t, "claim a disabled schedule's due run", ok, err, false)
|
||||
ok, err = repo.MissScheduleRun(ctx, s.ID, later, later, "x")
|
||||
applied(t, "miss a disabled schedule's run", ok, err, false)
|
||||
ok, err = repo.WarnScheduleRun(ctx, s.ID, later)
|
||||
applied(t, "warn a disabled schedule's run", ok, err, false)
|
||||
}
|
||||
|
||||
// TestDueSchedules pins what the runner reads: enabled schedules of live
|
||||
// servers due by the horizon, every run in progress, runs first, with the
|
||||
// server's owner now.
|
||||
func TestDueSchedules(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
u := newUser(t, "user", "sch-due")
|
||||
v := newUser(t, "user", "sch-due2")
|
||||
sfx := suffix(t)
|
||||
live, gone, unowned := "sdl-"+sfx, "sdg-"+sfx, "sdu-"+sfx
|
||||
seedOwnedServer(t, live, v.ID, false)
|
||||
seedOwnedServer(t, gone, u.ID, true)
|
||||
mustExec(t, `INSERT INTO servers (name, cached_cpu_milli, cached_memory_mb, cached_storage_mb) VALUES ($1, 100, 128, 1)`, unowned)
|
||||
t0 := mustNow().Truncate(time.Second)
|
||||
at := func(m int) *time.Time { p := t0.Add(time.Duration(m) * time.Minute); return &p }
|
||||
horizon := *at(30)
|
||||
|
||||
late := mustCreateSchedule(t, newSchedule(live, u.ID, api.ScheduleStop, at(20)))
|
||||
early := mustCreateSchedule(t, newSchedule(live, u.ID, api.ScheduleStart, at(-5)))
|
||||
edge := mustCreateSchedule(t, newSchedule(unowned, "", api.ScheduleStop, at(30)))
|
||||
mustCreateSchedule(t, newSchedule(live, u.ID, api.ScheduleStop, at(31))) // beyond the horizon
|
||||
off := mustCreateSchedule(t, newSchedule(live, u.ID, api.ScheduleStop, nil)) // disabled, yet due
|
||||
mustExec(t, `UPDATE server_schedules SET next_run_at = $2 WHERE id = $1`, off.ID, *at(-2))
|
||||
running := mustCreateSchedule(t, newSchedule(live, u.ID, api.ScheduleRestart, nil)) // disabled, but running
|
||||
ok, err := mustClaim(t, running.ID, nil, nil, t0)
|
||||
applied(t, "claim", ok, err, true)
|
||||
// Deleted server: its schedules wait for SeedServer or the row's end.
|
||||
mustExec(t, `INSERT INTO server_schedules (server_name, owner_id, action, timezone, next_run_at, created_by)
|
||||
VALUES ($1, $2, 'stop', 'UTC', $3, 'pgint')`, gone, u.ID, *at(-1))
|
||||
|
||||
due, err := repo.DueSchedules(ctx, horizon)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var got []string
|
||||
for _, d := range due {
|
||||
if d.Server != live && d.Server != unowned && d.Server != gone {
|
||||
continue // other tests' rows
|
||||
}
|
||||
got = append(got, fmt.Sprintf("%d/%s/%v", d.ID, d.RunState, d.ServerOwner == v.ID))
|
||||
}
|
||||
want := []string{
|
||||
fmt.Sprintf("%d/claimed/true", running.ID),
|
||||
fmt.Sprintf("%d//true", early.ID),
|
||||
fmt.Sprintf("%d//true", late.ID),
|
||||
fmt.Sprintf("%d//false", edge.ID),
|
||||
}
|
||||
if strings.Join(got, " ") != strings.Join(want, " ") {
|
||||
t.Fatalf("due %v, want %v", got, want)
|
||||
}
|
||||
for _, d := range due {
|
||||
if d.ID == late.ID && d.OwnerID != u.ID {
|
||||
t.Fatalf("schedule owner %q, want %q", d.OwnerID, u.ID)
|
||||
}
|
||||
if d.ID == edge.ID && d.ServerOwner != "" {
|
||||
t.Fatalf("unowned server's owner %q", d.ServerOwner)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestSchedulesFollowTheServer: a recreated server of the same name starts
|
||||
// without the earlier one's schedules, and an account migration hands the
|
||||
// migrated servers' schedules to the target.
|
||||
func TestSchedulesFollowTheServer(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
u := newUser(t, "user", "sch-seed")
|
||||
name, sub := "ss-"+suffix(t), "sss-"+suffix(t)
|
||||
if err := repo.SeedServer(ctx, name, sub, 100, 128, 1024); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
s := mustCreateSchedule(t, newSchedule(name, u.ID, api.ScheduleStop, nil))
|
||||
if err := repo.SeedServer(ctx, name, sub, 100, 128, 1024); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := repo.GetSchedule(ctx, name, s.ID); !errors.Is(err, api.ErrNotFound) {
|
||||
t.Fatalf("schedule of the earlier server = %v, want ErrNotFound", err)
|
||||
}
|
||||
|
||||
src := newUser(t, "user", "sch-src")
|
||||
dst := newUser(t, "user", "sch-dst")
|
||||
other := newUser(t, "user", "sch-other")
|
||||
sfx := suffix(t)
|
||||
mine, theirs := "sm-"+sfx, "st-"+sfx
|
||||
seedOwnedServer(t, mine, src.ID, false)
|
||||
seedOwnedServer(t, theirs, other.ID, false)
|
||||
moved := mustCreateSchedule(t, newSchedule(mine, src.ID, api.ScheduleStop, nil))
|
||||
stale := mustCreateSchedule(t, newSchedule(mine, other.ID, api.ScheduleStop, nil)) // saved by a previous owner
|
||||
kept := mustCreateSchedule(t, newSchedule(theirs, src.ID, api.ScheduleStop, nil)) // src's, on a server src lost
|
||||
t0 := mustNow().Truncate(time.Second)
|
||||
if err := repo.StartMigration(ctx, "schmig-"+sfx, src.ID, t0); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := repo.ConfirmMigration(ctx, src.ID, "passkey", "sess", t0); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := repo.IssueMigrationCode(ctx, src.ID, dst.ID, "sess", "h-sch-"+sfx, t0, t0.Add(10*time.Minute)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, _, err := repo.RedeemMigration(ctx, dst.ID, "h-sch-"+sfx, t0); err != nil {
|
||||
t.Fatalf("RedeemMigration: %v", err)
|
||||
}
|
||||
for _, c := range []struct {
|
||||
server string
|
||||
id int64
|
||||
owner string
|
||||
}{{mine, moved.ID, dst.ID}, {mine, stale.ID, other.ID}, {theirs, kept.ID, src.ID}} {
|
||||
if g := mustGetSchedule(t, c.server, c.id); g.OwnerID != c.owner {
|
||||
t.Fatalf("schedule %d owner %q, want %q", c.id, g.OwnerID, c.owner)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
-- Scheduled tasks (GET/POST /servers/{name}/schedules): an owner or admin asks
|
||||
-- felis-api to run a console command, restart, stop, start or back up a server
|
||||
-- at set times. felis-api's schedule runner reads this table every few seconds.
|
||||
--
|
||||
-- A schedule fires at minute_of_day (local to timezone) on the weekdays in the
|
||||
-- bitmask (bit 0 Sunday), or, with every_minutes set, at every multiple of it
|
||||
-- since local midnight on those days. owner_id is the server's owner when the
|
||||
-- schedule was saved (NULL for an unowned server): the runner skips and disables
|
||||
-- a schedule once the server has another owner, so a new owner never inherits
|
||||
-- somebody else's commands. A recreated server of the same name starts without
|
||||
-- schedules (SeedServer deletes them).
|
||||
--
|
||||
-- run_state is the step of a run still in progress (a restart waits for the
|
||||
-- server to stop before starting it, a backup of a running server stops it,
|
||||
-- backs it up and starts it again); run_step_at is when that step began, and
|
||||
-- run_resume says the run starts the server again once its backup is done.
|
||||
CREATE TABLE server_schedules (
|
||||
id bigserial PRIMARY KEY,
|
||||
server_name text NOT NULL REFERENCES servers(name) ON DELETE CASCADE,
|
||||
owner_id text REFERENCES users(id),
|
||||
label text NOT NULL DEFAULT '',
|
||||
action text NOT NULL CHECK (action IN ('command', 'restart', 'stop', 'start', 'backup')),
|
||||
command text NOT NULL DEFAULT '',
|
||||
every_minutes integer NOT NULL DEFAULT 0,
|
||||
minute_of_day integer NOT NULL DEFAULT 0 CHECK (minute_of_day BETWEEN 0 AND 1439),
|
||||
weekdays smallint NOT NULL DEFAULT 127 CHECK (weekdays BETWEEN 1 AND 127),
|
||||
timezone text NOT NULL,
|
||||
warn_minutes integer NOT NULL DEFAULT 0,
|
||||
enabled boolean NOT NULL DEFAULT true,
|
||||
next_run_at timestamptz,
|
||||
warned_for timestamptz,
|
||||
run_state text NOT NULL DEFAULT '',
|
||||
run_resume boolean NOT NULL DEFAULT false,
|
||||
run_step_at timestamptz,
|
||||
last_run_at timestamptz,
|
||||
last_result text NOT NULL DEFAULT '',
|
||||
last_detail text NOT NULL DEFAULT '',
|
||||
created_by text NOT NULL,
|
||||
created_at timestamptz NOT NULL DEFAULT now(),
|
||||
updated_at timestamptz NOT NULL DEFAULT now(),
|
||||
-- Every write that starts a step stamps it; the runner times each step from it.
|
||||
CHECK (run_state = '' OR run_step_at IS NOT NULL)
|
||||
);
|
||||
CREATE INDEX server_schedules_server ON server_schedules (server_name, id);
|
||||
CREATE INDEX server_schedules_due ON server_schedules (next_run_at) WHERE enabled;
|
||||
CREATE INDEX server_schedules_running ON server_schedules (id) WHERE run_state <> '';
|
||||
Reference in new issue
Block a user