fix(backup): 手动备份按服冷却、每服保留上限与独立保留期,容量驱逐不再删除回收世界的唯一副本

This commit is contained in:
Lemon-miaow committed 2026-09-24 19:58:59 +08:00
1 parent f69cdec9d4
commit 9bda3a52fa
30 files changed
+801 -42

No files matched your search

+59
View File
@@ -3,6 +3,8 @@ package api
import (
"context"
"sort"
"strings"
"time"
batchv1 "k8s.io/api/batch/v1"
corev1 "k8s.io/api/core/v1"
@@ -53,9 +55,66 @@ func (k *K8sJobStatus) LatestJobs(ctx context.Context, serverName string) ([]Asy
if len(out) > 20 {
out = out[:20]
}
k.explainFailures(ctx, serverName, out)
return out, nil
}
// explainFailures replaces a failed Job's condition text ("Job has reached the
// specified backoff limit") with the error its container exited on. The
// executors set TerminationMessagePolicy FallbackToLogsOnError, so the
// terminated state carries the tail of the log, whose last line is the
// "felis backup: …" / "felis restore: …" error. The pods carry the Job's
// labels, so one list covers every Job of the server; a pod already collected
// by the Job TTL, or a list error, leaves the condition text in place.
func (k *K8sJobStatus) explainFailures(ctx context.Context, serverName string, jobs []AsyncJob) {
failed := map[string]int{}
for i, j := range jobs {
if j.State == "failed" {
failed[j.Name] = i
}
}
if len(failed) == 0 {
return
}
var pods corev1.PodList
if err := k.c.List(ctx, &pods, client.InNamespace(k.namespace),
client.MatchingLabels{jobServerLabel: serverName}); err != nil {
return
}
newest := map[string]time.Time{}
for i := range pods.Items {
pod := &pods.Items[i]
idx, ok := failed[pod.Labels["job-name"]]
if !ok {
continue
}
msg := lastTerminationLine(pod)
if msg == "" || !pod.CreationTimestamp.Time.After(newest[pod.Labels["job-name"]]) {
continue
}
newest[pod.Labels["job-name"]] = pod.CreationTimestamp.Time
jobs[idx].Message = msg
}
}
// lastTerminationLine returns the last non-empty line of the pod's terminated
// container message, capped for display.
func lastTerminationLine(pod *corev1.Pod) string {
for _, cs := range pod.Status.ContainerStatuses {
t := cs.State.Terminated
if t == nil || t.ExitCode == 0 {
continue
}
lines := strings.Split(strings.TrimSpace(t.Message), "\n")
line := strings.TrimSpace(lines[len(lines)-1])
if len(line) > 400 {
line = line[:400] + "…"
}
return line
}
return ""
}
// jobToAsyncJob projects one Job onto its kind/state/message. Complete condition →
// succeeded, Failed → failed with its reason (Job conditions carry the generic
// "backoff limit exceeded" text; the pod log holds the underlying error), anything