423 lines
14 KiB
Go
423 lines
14 KiB
Go
// Package watchdog is the consumer the platform's health signals otherwise lack:
|
||
// `felis watchdog` runs on the host from a systemd timer, checks the control
|
||
// plane, the login gate, the fleet, PostgreSQL, the database backups and the
|
||
// node, and mails the platform owners when something has stayed wrong long
|
||
// enough to matter, again every day while it stays wrong, and once more when it
|
||
// clears.
|
||
//
|
||
// It runs on the host rather than in the cluster so the failures that take the
|
||
// cluster down (k3s stopped, the API server wedged, the node out of disk) are
|
||
// still reported: the one thing it needs to send mail, the relay, it reaches
|
||
// directly, with the credentials it cached on its last good run.
|
||
//
|
||
// This file is the pure part: turning one run's findings into state changes and
|
||
// a message. The probes that produce findings live in probes.go.
|
||
package watchdog
|
||
|
||
import (
|
||
"encoding/json"
|
||
"errors"
|
||
"fmt"
|
||
"os"
|
||
"path/filepath"
|
||
"sort"
|
||
"strings"
|
||
"time"
|
||
)
|
||
|
||
// Severity orders how urgent a finding is.
|
||
type Severity string
|
||
|
||
const (
|
||
Critical Severity = "critical"
|
||
Warning Severity = "warning"
|
||
)
|
||
|
||
// Finding is one condition a probe saw on this run.
|
||
type Finding struct {
|
||
// Key identifies the condition across runs, e.g. "deployment/felis-api". A
|
||
// key is also how a condition is found again once it clears.
|
||
Key string `json:"key"`
|
||
Severity Severity `json:"severity"`
|
||
// Summary and SummaryEN name what is wrong, in Chinese and English.
|
||
Summary string `json:"summary"`
|
||
SummaryEN string `json:"summary_en"`
|
||
// Hint says where to look next (a command, a troubleshooting section).
|
||
Hint string `json:"hint,omitempty"`
|
||
// For is how long the condition must hold before it is mailed, so a rollout
|
||
// or a restart that heals itself never pages anyone.
|
||
For time.Duration `json:"for"`
|
||
// Event marks a one-off (a failed Job): mailed once, with no reminder and no
|
||
// resolved notice when it goes away.
|
||
Event bool `json:"event,omitempty"`
|
||
}
|
||
|
||
// Report is one run's observations.
|
||
type Report struct {
|
||
Findings []Finding
|
||
// Unknown lists key prefixes no probe could evaluate on this run (the API
|
||
// server being down hides every cluster check). Alerts under them keep their
|
||
// state rather than reading as cleared.
|
||
Unknown []string
|
||
}
|
||
|
||
// Alert is a finding the state file remembers across runs.
|
||
type Alert struct {
|
||
Finding
|
||
FirstSeen time.Time `json:"first_seen"`
|
||
// Notified is when the alert was last mailed, at NotifiedSeverity; zero
|
||
// while it is pending.
|
||
Notified time.Time `json:"notified,omitzero"`
|
||
NotifiedSeverity Severity `json:"notified_severity,omitempty"`
|
||
// ClearedAt is when a mailed alert was first seen gone. It is mailed as
|
||
// resolved only after staying gone for resolveAfter, so a value hovering at
|
||
// its threshold does not mail on every crossing.
|
||
ClearedAt time.Time `json:"cleared_at,omitzero"`
|
||
}
|
||
|
||
// State is what the watchdog keeps between runs.
|
||
type State struct {
|
||
Alerts map[string]*Alert `json:"alerts"`
|
||
// Recipients and SMTPPassword are cached from the last run that could read
|
||
// them (the database and the felis-smtp Secret), so an outage of either can
|
||
// still be mailed.
|
||
Recipients []string `json:"recipients,omitempty"`
|
||
SMTPPassword string `json:"smtp_password,omitempty"`
|
||
// Relay is the [smtp] relay of the last run that could read felis.toml,
|
||
// so a run of the watchdog that failed can still be mailed after the
|
||
// config stopped loading; nil when there is none.
|
||
Relay *Relay `json:"relay,omitempty"`
|
||
}
|
||
|
||
// Relay is the part of [smtp] a mail needs, besides the password.
|
||
type Relay struct {
|
||
Host string `json:"host"`
|
||
Port int `json:"port"`
|
||
From string `json:"from"`
|
||
Username string `json:"username,omitempty"`
|
||
RequireTLS bool `json:"require_tls"`
|
||
}
|
||
|
||
// Open reports whether the owners were told of a condition that still holds:
|
||
// an alert mailed (or logged) and not seen gone since, one-off events aside.
|
||
func (s *State) Open() bool {
|
||
for _, a := range s.Alerts {
|
||
if !a.Notified.IsZero() && a.ClearedAt.IsZero() && !a.Event {
|
||
return true
|
||
}
|
||
}
|
||
return false
|
||
}
|
||
|
||
const (
|
||
// remindEvery is how often a firing alert is mailed again.
|
||
remindEvery = 24 * time.Hour
|
||
// resolveAfter is how long a mailed alert must stay gone before it is
|
||
// mailed as resolved.
|
||
resolveAfter = 10 * time.Minute
|
||
)
|
||
|
||
// Plan is what one run has to tell the owners.
|
||
type Plan struct {
|
||
Firing []Alert // newly past their For, or worse than when last mailed
|
||
Reminders []Alert // still firing a day after the last mail
|
||
Resolved []Alert // gone for resolveAfter
|
||
// Active is every mailed condition still firing (one-off events aside), for
|
||
// the message's summary.
|
||
Active []Alert
|
||
}
|
||
|
||
// Empty reports whether the plan has nothing to mail.
|
||
func (p Plan) Empty() bool {
|
||
return len(p.Firing) == 0 && len(p.Reminders) == 0 && len(p.Resolved) == 0
|
||
}
|
||
|
||
// Observe folds a report into s and returns what is due to be mailed. It only
|
||
// tracks when conditions appeared and cleared; Commit records the plan as
|
||
// delivered. A run whose mail failed, or that falls in a quiet period, saves
|
||
// the state without committing, and the same alerts come due again next run.
|
||
func (s *State) Observe(r Report, now time.Time) Plan {
|
||
if s.Alerts == nil {
|
||
s.Alerts = map[string]*Alert{}
|
||
}
|
||
seen := map[string]bool{}
|
||
var plan Plan
|
||
for _, f := range r.Findings {
|
||
seen[f.Key] = true
|
||
a, ok := s.Alerts[f.Key]
|
||
if !ok {
|
||
a = &Alert{FirstSeen: now}
|
||
s.Alerts[f.Key] = a
|
||
}
|
||
a.Finding = f
|
||
a.ClearedAt = time.Time{}
|
||
notified := !a.Notified.IsZero()
|
||
switch {
|
||
case !notified && now.Sub(a.FirstSeen) >= f.For:
|
||
plan.Firing = append(plan.Firing, *a)
|
||
case notified && f.Severity == Critical && a.NotifiedSeverity != Critical:
|
||
// Worse than when it was mailed (a disk from low to nearly full):
|
||
// new news, mailed at once.
|
||
plan.Firing = append(plan.Firing, *a)
|
||
case notified && !f.Event && now.Sub(a.Notified) >= remindEvery:
|
||
plan.Reminders = append(plan.Reminders, *a)
|
||
}
|
||
}
|
||
for key, a := range s.Alerts {
|
||
if seen[key] || underAny(key, r.Unknown) {
|
||
continue
|
||
}
|
||
switch {
|
||
case a.Notified.IsZero() || a.Event:
|
||
// Never mailed, or a one-off: nothing to take back.
|
||
delete(s.Alerts, key)
|
||
case a.ClearedAt.IsZero():
|
||
a.ClearedAt = now
|
||
case now.Sub(a.ClearedAt) >= resolveAfter:
|
||
plan.Resolved = append(plan.Resolved, *a)
|
||
}
|
||
}
|
||
for _, a := range s.Alerts {
|
||
if !a.Notified.IsZero() && a.ClearedAt.IsZero() && !a.Event {
|
||
plan.Active = append(plan.Active, *a)
|
||
}
|
||
}
|
||
for _, l := range [][]Alert{plan.Firing, plan.Reminders, plan.Resolved, plan.Active} {
|
||
sortAlerts(l)
|
||
}
|
||
return plan
|
||
}
|
||
|
||
// Commit records plan as mailed at now.
|
||
func (s *State) Commit(plan Plan, now time.Time) {
|
||
for _, l := range [][]Alert{plan.Firing, plan.Reminders} {
|
||
for _, a := range l {
|
||
if cur := s.Alerts[a.Key]; cur != nil {
|
||
cur.Notified, cur.NotifiedSeverity = now, a.Severity
|
||
}
|
||
}
|
||
}
|
||
for _, a := range plan.Resolved {
|
||
delete(s.Alerts, a.Key)
|
||
}
|
||
}
|
||
|
||
func underAny(key string, prefixes []string) bool {
|
||
for _, p := range prefixes {
|
||
if strings.HasPrefix(key, p) {
|
||
return true
|
||
}
|
||
}
|
||
return false
|
||
}
|
||
|
||
// sortAlerts puts critical alerts first, then orders by key.
|
||
func sortAlerts(l []Alert) {
|
||
sort.Slice(l, func(i, j int) bool {
|
||
if (l[i].Severity == Critical) != (l[j].Severity == Critical) {
|
||
return l[i].Severity == Critical
|
||
}
|
||
return l[i].Key < l[j].Key
|
||
})
|
||
}
|
||
|
||
// Message renders the plan as one mail: a subject naming the counts and host,
|
||
// and a body listing what fired, what is still wrong and what cleared.
|
||
func (p Plan) Message(host string, now time.Time) (subject, body string) {
|
||
var parts, partsEN []string
|
||
if n := len(p.Firing) + len(p.Reminders); n > 0 {
|
||
parts = append(parts, fmt.Sprintf("%d 项异常", n))
|
||
partsEN = append(partsEN, fmt.Sprintf("%d firing", n))
|
||
}
|
||
if n := len(p.Resolved); n > 0 {
|
||
parts = append(parts, fmt.Sprintf("%d 项恢复", n))
|
||
partsEN = append(partsEN, fmt.Sprintf("%d resolved", n))
|
||
}
|
||
tag := "告警"
|
||
if len(p.Firing)+len(p.Reminders) == 0 {
|
||
tag = "恢复"
|
||
} else if hasCritical(p.Firing) || hasCritical(p.Reminders) {
|
||
tag = "严重告警"
|
||
}
|
||
subject = fmt.Sprintf("Felis %s(%s):%s · %s", tag, host, strings.Join(parts, ","), strings.Join(partsEN, ", "))
|
||
|
||
var b strings.Builder
|
||
fmt.Fprintf(&b, "Felis 平台巡检 / platform watchdog — %s, %s\r\n", host, now.UTC().Format("2006-01-02 15:04 UTC"))
|
||
section := func(title string, l []Alert, since bool) {
|
||
if len(l) == 0 {
|
||
return
|
||
}
|
||
fmt.Fprintf(&b, "\r\n%s\r\n", title)
|
||
for _, a := range l {
|
||
fmt.Fprintf(&b, "\r\n[%s] %s\r\n %s\r\n", a.Severity, a.Summary, a.SummaryEN)
|
||
if since {
|
||
fmt.Fprintf(&b, " 自 / since %s\r\n", a.FirstSeen.UTC().Format("2006-01-02 15:04 UTC"))
|
||
}
|
||
if a.Hint != "" {
|
||
fmt.Fprintf(&b, " → %s\r\n", a.Hint)
|
||
}
|
||
}
|
||
}
|
||
section("== 新出现的异常 / new ==", p.Firing, true)
|
||
section("== 仍未恢复(每日提醒)/ still firing (daily reminder) ==", p.Reminders, true)
|
||
section("== 已恢复 / resolved ==", p.Resolved, false)
|
||
if others := without(p.Active, p.Firing, p.Reminders); len(others) > 0 {
|
||
fmt.Fprintf(&b, "\r\n== 其他仍在进行的告警 / also still firing ==\r\n")
|
||
for _, a := range others {
|
||
fmt.Fprintf(&b, " [%s] %s / %s\r\n", a.Severity, a.Summary, a.SummaryEN)
|
||
}
|
||
}
|
||
// The journal rather than a -dry-run: the unit carries the flags (proxy port,
|
||
// disk paths) a bare command line would not.
|
||
fmt.Fprintf(&b, "\r\n在主机上运行 `journalctl -u felis-watchdog -n 40` 查看最近一次巡检的全部检查结果。\r\n"+
|
||
"Run `journalctl -u felis-watchdog -n 40` on the host for every check of the latest run.\r\n")
|
||
return subject, b.String()
|
||
}
|
||
|
||
func hasCritical(l []Alert) bool {
|
||
for _, a := range l {
|
||
if a.Severity == Critical {
|
||
return true
|
||
}
|
||
}
|
||
return false
|
||
}
|
||
|
||
// without returns the alerts of all not keyed in any of skip.
|
||
func without(all []Alert, skip ...[]Alert) []Alert {
|
||
keys := map[string]bool{}
|
||
for _, l := range skip {
|
||
for _, a := range l {
|
||
keys[a.Key] = true
|
||
}
|
||
}
|
||
var out []Alert
|
||
for _, a := range all {
|
||
if !keys[a.Key] {
|
||
out = append(out, a)
|
||
}
|
||
}
|
||
return out
|
||
}
|
||
|
||
// ErrBadState is a state file that is not the JSON the watchdog writes.
|
||
var ErrBadState = errors.New("watchdog: the state file is not one the watchdog wrote")
|
||
|
||
// LoadState reads the state file; a missing file is a fresh state.
|
||
func LoadState(path string) (*State, error) {
|
||
raw, err := os.ReadFile(path)
|
||
if errors.Is(err, os.ErrNotExist) {
|
||
return &State{Alerts: map[string]*Alert{}}, nil
|
||
}
|
||
if err != nil {
|
||
return nil, err
|
||
}
|
||
var s State
|
||
if err := json.Unmarshal(raw, &s); err != nil {
|
||
return nil, fmt.Errorf("%w: %s: %v", ErrBadState, path, err)
|
||
}
|
||
if s.Alerts == nil {
|
||
s.Alerts = map[string]*Alert{}
|
||
}
|
||
return &s, nil
|
||
}
|
||
|
||
// RecoverState is LoadState for a run that must go on: a state file that does
|
||
// not parse (a disk that filled mid-write, a hand edit) is renamed aside and
|
||
// the run starts from a fresh state, rather than every later run failing on
|
||
// it and mailing nothing. aside is where it went, "" when nothing was moved.
|
||
func RecoverState(path string, now time.Time) (s *State, aside string, err error) {
|
||
s, err = LoadState(path)
|
||
if !errors.Is(err, ErrBadState) {
|
||
return s, "", err
|
||
}
|
||
aside = fmt.Sprintf("%s.unreadable-%d", path, now.Unix())
|
||
if rerr := os.Rename(path, aside); rerr != nil {
|
||
return nil, "", fmt.Errorf("%w (and could not move it aside: %v)", err, rerr)
|
||
}
|
||
return &State{Alerts: map[string]*Alert{}}, aside, nil
|
||
}
|
||
|
||
// SaveState writes s atomically, readable by root only: it caches the relay
|
||
// password.
|
||
func SaveState(path string, s *State) error {
|
||
raw, err := json.MarshalIndent(s, "", " ")
|
||
if err != nil {
|
||
return err
|
||
}
|
||
if err := os.MkdirAll(filepath.Dir(path), 0o700); err != nil {
|
||
return err
|
||
}
|
||
tmp, err := os.CreateTemp(filepath.Dir(path), ".state-*")
|
||
if err != nil {
|
||
return err
|
||
}
|
||
defer os.Remove(tmp.Name())
|
||
if err := tmp.Chmod(0o600); err != nil {
|
||
tmp.Close()
|
||
return err
|
||
}
|
||
if _, err := tmp.Write(raw); err != nil {
|
||
tmp.Close()
|
||
return err
|
||
}
|
||
if err := tmp.Close(); err != nil {
|
||
return err
|
||
}
|
||
return os.Rename(tmp.Name(), path)
|
||
}
|
||
|
||
// FallbackStatePath is where a run keeps its state when the state file cannot be
|
||
// written (a full or read-only /var/lib): tmpfs, which lasts until the host
|
||
// restarts, next to the installer's quiet marker.
|
||
const FallbackStatePath = "/run/felis/watchdog-state.json"
|
||
|
||
// NewestState is the state file a run reads: fallback when a run wrote it after
|
||
// path (its save to path failed, SaveStateOr), else path.
|
||
func NewestState(path, fallback string) string {
|
||
fb, err := os.Stat(fallback) // "" is no fallback: it does not stat
|
||
if err != nil {
|
||
return path
|
||
}
|
||
if st, err := os.Stat(path); err == nil && !fb.ModTime().After(st.ModTime()) {
|
||
return path
|
||
}
|
||
return fallback
|
||
}
|
||
|
||
// SaveStateOr saves s to path and drops fallback. When path cannot be written it
|
||
// saves s to fallback instead, so the next run still knows what this one mailed
|
||
// and does not mail it again every two minutes. The error is path's either way,
|
||
// saying where the state went.
|
||
func SaveStateOr(path, fallback string, s *State) error {
|
||
err := SaveState(path, s)
|
||
if err == nil {
|
||
if fallback != "" {
|
||
os.Remove(fallback)
|
||
}
|
||
return nil
|
||
}
|
||
if fallback == "" {
|
||
return err
|
||
}
|
||
if ferr := SaveState(fallback, s); ferr != nil {
|
||
return fmt.Errorf("%w; nor in %s: %v", err, fallback, ferr)
|
||
}
|
||
return fmt.Errorf("%w; kept in %s until the host restarts", err, fallback)
|
||
}
|
||
|
||
// QuietUntil reads the maintenance marker the installer writes while it
|
||
// restarts things on purpose: a Unix timestamp, before which nothing is mailed.
|
||
// A missing or unreadable marker means no quiet period.
|
||
func QuietUntil(path string) time.Time {
|
||
raw, err := os.ReadFile(path)
|
||
if err != nil {
|
||
return time.Time{}
|
||
}
|
||
var sec int64
|
||
if _, err := fmt.Sscan(strings.TrimSpace(string(raw)), &sec); err != nil {
|
||
return time.Time{}
|
||
}
|
||
return time.Unix(sec, 0)
|
||
}
|