350 lines
15 KiB
Go
350 lines
15 KiB
Go
package fileedit
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"io"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"strings"
|
|
"testing"
|
|
"time"
|
|
|
|
batchv1 "k8s.io/api/batch/v1"
|
|
corev1 "k8s.io/api/core/v1"
|
|
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
|
"k8s.io/apimachinery/pkg/runtime"
|
|
"k8s.io/client-go/kubernetes"
|
|
"k8s.io/client-go/kubernetes/fake"
|
|
"k8s.io/client-go/rest"
|
|
k8stesting "k8s.io/client-go/testing"
|
|
)
|
|
|
|
func TestRunReadsResultBeforePodTermination(t *testing.T) {
|
|
for _, op := range []string{OpList, OpRead} {
|
|
t.Run(op, func(t *testing.T) {
|
|
closed := make(chan struct{})
|
|
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
|
switch {
|
|
case r.Method == http.MethodPost && strings.HasSuffix(r.URL.Path, "/jobs"):
|
|
w.Header().Set("Content-Type", "application/json")
|
|
io.WriteString(w, `{"apiVersion":"batch/v1","kind":"Job","metadata":{"name":"files-test"}}`)
|
|
case strings.HasSuffix(r.URL.Path, "/pods"):
|
|
if !strings.Contains(r.URL.Query().Get("labelSelector"), LabelOpID+"=") {
|
|
t.Error("pod lookup did not select this operation")
|
|
}
|
|
w.Header().Set("Content-Type", "application/json")
|
|
io.WriteString(w, `{"apiVersion":"v1","kind":"PodList","items":[{"metadata":{"name":"file-pod"},"status":{"phase":"Running","containerStatuses":[{"name":"files","state":{"running":{}}}]}}]}`)
|
|
case strings.HasSuffix(r.URL.Path, "/log"):
|
|
if r.URL.Query().Get("follow") != "true" {
|
|
t.Error("result log was not streamed")
|
|
}
|
|
io.WriteString(w, "diagnostic line\n"+ResultPrefix+`{"entries":[]}`+"\n")
|
|
w.(http.Flusher).Flush()
|
|
// The log stays open and the Pod stays Running. Run must return
|
|
// on the result line and close the stream, not await either EOF.
|
|
<-r.Context().Done()
|
|
close(closed)
|
|
default:
|
|
t.Errorf("unexpected cluster call: %s %s", r.Method, r.URL)
|
|
w.WriteHeader(http.StatusNotFound)
|
|
}
|
|
}))
|
|
defer srv.Close()
|
|
cs, err := kubernetes.NewForConfig(&rest.Config{Host: srv.URL})
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
ctx, cancel := context.WithTimeout(context.Background(), 3*time.Second)
|
|
defer cancel()
|
|
got, err := NewK8sRunner(cs).Run(ctx, testParams(op))
|
|
if err != nil || string(got) != `{"entries":[]}` {
|
|
t.Fatalf("Run = %s, %v", got, err)
|
|
}
|
|
select {
|
|
case <-closed:
|
|
case <-ctx.Done():
|
|
t.Fatal("result stream was not closed")
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
func TestAwaitPodKeepsWritesWaitingForTermination(t *testing.T) {
|
|
p := testParams(OpWrite)
|
|
pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "file-pod", Namespace: p.Namespace, Labels: filesLabels(p)},
|
|
Status: corev1.PodStatus{Phase: corev1.PodRunning, ContainerStatuses: []corev1.ContainerStatus{{Name: containerName, State: corev1.ContainerState{Running: &corev1.ContainerStateRunning{}}}}}}
|
|
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Millisecond)
|
|
defer cancel()
|
|
if got, err := NewK8sRunner(fake.NewSimpleClientset(pod)).awaitPod(ctx, p); err == nil || got != nil {
|
|
t.Fatalf("a running writer was treated as finished: %v, %v", got, err)
|
|
}
|
|
}
|
|
|
|
var (
|
|
opCreated = time.Date(2026, 9, 28, 10, 0, 0, 0, time.UTC)
|
|
opEnded = opCreated.Add(3 * time.Minute)
|
|
)
|
|
|
|
func asyncJob(conds ...batchv1.JobCondition) *batchv1.Job {
|
|
return &batchv1.Job{
|
|
ObjectMeta: metav1.ObjectMeta{
|
|
Name: "files-survival-0a",
|
|
CreationTimestamp: metav1.NewTime(opCreated),
|
|
Labels: map[string]string{LabelOpID: "0a", LabelMode: OpUnzip},
|
|
Annotations: map[string]string{AnnotationPath: "maps/world.zip"},
|
|
},
|
|
Status: batchv1.JobStatus{Conditions: conds},
|
|
}
|
|
}
|
|
|
|
func cond(typ batchv1.JobConditionType, status corev1.ConditionStatus, reason string) batchv1.JobCondition {
|
|
return batchv1.JobCondition{Type: typ, Status: status, Reason: reason, LastTransitionTime: metav1.NewTime(opEnded)}
|
|
}
|
|
|
|
// TestOpState pins how a background Job and the tail of its log read as an
|
|
// OpState: the printed result decides the outcome whatever the Job's condition
|
|
// says, and a Job that ended without one failed for the condition's reason.
|
|
func TestOpState(t *testing.T) {
|
|
progress := ProgressPrefix + `{"done":10,"total":100}` + "\n" +
|
|
"a stderr line\n" +
|
|
ProgressPrefix + `{"done":40,"total":100}` + "\n"
|
|
ok := ResultPrefix + `{"files":3,"bytes":40}` + "\n"
|
|
conflict := ResultPrefix + `{"code":"exists","conflicts":["a.txt"],"conflict_count":1}` + "\n"
|
|
complete := cond(batchv1.JobComplete, corev1.ConditionTrue, "")
|
|
backoff := cond(batchv1.JobFailed, corev1.ConditionTrue, "BackoffLimitExceeded")
|
|
killed := func(container, reason string) *corev1.Pod {
|
|
return &corev1.Pod{Status: corev1.PodStatus{Phase: corev1.PodFailed, ContainerStatuses: []corev1.ContainerStatus{{
|
|
Name: container, State: corev1.ContainerState{Terminated: &corev1.ContainerStateTerminated{ExitCode: 137, Reason: reason}},
|
|
}}}}
|
|
}
|
|
|
|
cases := []struct {
|
|
name string
|
|
conds []batchv1.JobCondition
|
|
pod *corev1.Pod
|
|
log string
|
|
state string
|
|
reason string
|
|
code string
|
|
files int
|
|
done int64
|
|
finished bool
|
|
wantResult bool
|
|
}{
|
|
{name: "running, at its latest progress", log: progress, state: OpRunning, done: 40},
|
|
{name: "running, before any progress", state: OpRunning},
|
|
{name: "a condition not yet true is still running",
|
|
conds: []batchv1.JobCondition{cond(batchv1.JobFailed, corev1.ConditionFalse, "DeadlineExceeded")},
|
|
log: progress, state: OpRunning, done: 40},
|
|
{name: "complete with a clean result",
|
|
conds: []batchv1.JobCondition{complete}, log: progress + ok,
|
|
state: OpSucceeded, files: 3, done: 40, finished: true, wantResult: true},
|
|
{name: "success criteria met before complete",
|
|
conds: []batchv1.JobCondition{cond(batchv1.JobSuccessCriteriaMet, corev1.ConditionTrue, "")}, log: ok,
|
|
state: OpSucceeded, files: 3, finished: true, wantResult: true},
|
|
{name: "killed at its deadline after printing a clean result",
|
|
conds: []batchv1.JobCondition{cond(batchv1.JobFailed, corev1.ConditionTrue, "DeadlineExceeded")}, log: ok,
|
|
state: OpSucceeded, files: 3, finished: true, wantResult: true},
|
|
{name: "complete with a refusal",
|
|
conds: []batchv1.JobCondition{complete}, log: conflict,
|
|
state: OpFailed, code: CodeExists, finished: true, wantResult: true},
|
|
{name: "killed at its deadline without a result",
|
|
conds: []batchv1.JobCondition{cond(batchv1.JobFailed, corev1.ConditionTrue, "DeadlineExceeded")}, log: progress,
|
|
state: OpFailed, reason: "DeadlineExceeded", done: 40, finished: true},
|
|
{name: "failure target before failed",
|
|
conds: []batchv1.JobCondition{cond(batchv1.JobFailureTarget, corev1.ConditionTrue, "BackoffLimitExceeded")},
|
|
state: OpFailed, reason: "BackoffLimitExceeded", finished: true},
|
|
{name: "failed without a reason",
|
|
conds: []batchv1.JobCondition{cond(batchv1.JobFailed, corev1.ConditionTrue, "")},
|
|
state: OpFailed, reason: "Failed", finished: true},
|
|
{name: "complete but its log is gone",
|
|
conds: []batchv1.JobCondition{complete},
|
|
state: OpFailed, reason: "ResultUnavailable", finished: true},
|
|
{name: "complete with a result that does not parse",
|
|
conds: []batchv1.JobCondition{complete}, log: ResultPrefix + "{\n",
|
|
state: OpFailed, reason: "ResultUnavailable", finished: true},
|
|
{name: "killed for memory without a result",
|
|
conds: []batchv1.JobCondition{backoff}, pod: killed(containerName, "OOMKilled"), log: progress,
|
|
state: OpFailed, reason: "OOMKilled", done: 40, finished: true},
|
|
{name: "killed for memory after printing a clean result",
|
|
conds: []batchv1.JobCondition{backoff}, pod: killed(containerName, "OOMKilled"), log: ok,
|
|
state: OpSucceeded, files: 3, finished: true, wantResult: true},
|
|
{name: "killed another way",
|
|
conds: []batchv1.JobCondition{backoff}, pod: killed(containerName, "Error"),
|
|
state: OpFailed, reason: "BackoffLimitExceeded", finished: true},
|
|
{name: "another container killed for memory",
|
|
conds: []batchv1.JobCondition{backoff}, pod: killed("sidecar", "OOMKilled"),
|
|
state: OpFailed, reason: "BackoffLimitExceeded", finished: true},
|
|
}
|
|
for _, tc := range cases {
|
|
t.Run(tc.name, func(t *testing.T) {
|
|
st := opState(asyncJob(tc.conds...), tc.pod, tc.log)
|
|
if st.ID != "0a" || st.Op != OpUnzip || st.Path != "maps/world.zip" || !st.Started.Equal(opCreated) {
|
|
t.Fatalf("identity = %q %q %q %v", st.ID, st.Op, st.Path, st.Started)
|
|
}
|
|
if st.State != tc.state || st.Reason != tc.reason {
|
|
t.Fatalf("state = %q reason %q, want %q reason %q", st.State, st.Reason, tc.state, tc.reason)
|
|
}
|
|
if st.Done != tc.done || (tc.done != 0 && st.Total != 100) {
|
|
t.Fatalf("progress = %d/%d, want %d/100", st.Done, st.Total, tc.done)
|
|
}
|
|
if want := map[bool]time.Time{true: opEnded}[tc.finished]; !st.Finished.Equal(want) {
|
|
t.Fatalf("finished = %v, want %v", st.Finished, want)
|
|
}
|
|
if (st.Result != nil) != tc.wantResult {
|
|
t.Fatalf("result = %+v, want one: %v", st.Result, tc.wantResult)
|
|
}
|
|
if st.Result != nil && (st.Result.Code != tc.code || st.Result.Files != tc.files) {
|
|
t.Fatalf("result = %+v, want code %q and %d files", st.Result, tc.code, tc.files)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// TestK8sRunnerOps checks what Ops lists: this server's background Jobs only,
|
|
// newest first and at most maxOps of them, with a log read for each started Pod
|
|
// and none for a Pod still waiting to run.
|
|
func TestK8sRunnerOps(t *testing.T) {
|
|
job := func(server, id string, age time.Duration, async bool) *batchv1.Job {
|
|
p := testParams(OpUnzip)
|
|
p.Server, p.OpID, p.Path, p.Async = server, id, "maps/"+id+".zip", async
|
|
j, err := FilesJob(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
j.CreationTimestamp = metav1.NewTime(opCreated.Add(-age))
|
|
return j
|
|
}
|
|
pod := func(j *batchv1.Job, phase corev1.PodPhase, age time.Duration) *corev1.Pod {
|
|
return &corev1.Pod{
|
|
ObjectMeta: metav1.ObjectMeta{
|
|
Name: fmt.Sprintf("%s-%d", j.Name, age), Namespace: "minecraft", Labels: j.Spec.Template.Labels,
|
|
CreationTimestamp: metav1.NewTime(opCreated.Add(-age)),
|
|
},
|
|
Status: corev1.PodStatus{Phase: phase},
|
|
}
|
|
}
|
|
|
|
var objs []runtime.Object
|
|
for i := range maxOps + 2 {
|
|
j := job("survival", fmt.Sprintf("%02d", i), time.Duration(i)*time.Minute, true)
|
|
objs = append(objs, j, pod(j, corev1.PodSucceeded, time.Duration(i)*time.Minute))
|
|
}
|
|
// The newest Job's newest Pod has not started, so its log is not read; the
|
|
// older Pod beside it is not the one Ops reports on.
|
|
newest := job("survival", "new", -time.Minute, true)
|
|
objs = append(objs, newest,
|
|
pod(newest, corev1.PodPending, -time.Minute), pod(newest, corev1.PodFailed, 0))
|
|
objs = append(objs,
|
|
job("creative", "other", -2*time.Minute, true),
|
|
job("survival", "sync", -3*time.Minute, false))
|
|
cs := fake.NewSimpleClientset(objs...)
|
|
|
|
ops, err := NewK8sRunner(cs).Ops(context.Background(), "minecraft", "survival")
|
|
if err != nil {
|
|
t.Fatalf("Ops: %v", err)
|
|
}
|
|
var ids []string
|
|
for _, op := range ops {
|
|
ids = append(ids, op.ID)
|
|
}
|
|
want := []string{"new", "00", "01", "02", "03", "04", "05", "06", "07", "08"}
|
|
if fmt.Sprint(ids) != fmt.Sprint(want) {
|
|
t.Fatalf("ops = %v, want %v", ids, want)
|
|
}
|
|
if ops[0].Path != "maps/new.zip" || ops[0].Op != OpUnzip {
|
|
t.Fatalf("newest op = %+v", ops[0])
|
|
}
|
|
|
|
logs := 0
|
|
for _, a := range cs.Actions() {
|
|
if a.GetVerb() == "get" && a.GetSubresource() == "log" {
|
|
logs++
|
|
opts := a.(k8stesting.GenericAction).GetValue().(*corev1.PodLogOptions)
|
|
if opts.Container != containerName || opts.TailLines == nil || *opts.TailLines != opLogLines {
|
|
t.Fatalf("log options = %+v", opts)
|
|
}
|
|
}
|
|
}
|
|
if logs != maxOps-1 {
|
|
t.Fatalf("read %d logs, want %d: one per listed op whose Pod has started", logs, maxOps-1)
|
|
}
|
|
}
|
|
|
|
// TestK8sRunnerOpsReadsAMemoryKill checks Ops hands each Job's Pod to opState: a
|
|
// Pod the kernel killed for memory is what names an op's reason OOMKilled.
|
|
func TestK8sRunnerOpsReadsAMemoryKill(t *testing.T) {
|
|
p := testParams(OpUnzip)
|
|
p.Server, p.OpID, p.Path, p.Async = "survival", "oom", "maps/tiles.zip", true
|
|
j, err := FilesJob(p)
|
|
if err != nil {
|
|
t.Fatal(err)
|
|
}
|
|
j.Status.Conditions = []batchv1.JobCondition{cond(batchv1.JobFailed, corev1.ConditionTrue, "BackoffLimitExceeded")}
|
|
pod := &corev1.Pod{
|
|
ObjectMeta: metav1.ObjectMeta{Name: j.Name + "-x", Namespace: "minecraft", Labels: j.Spec.Template.Labels},
|
|
Status: corev1.PodStatus{Phase: corev1.PodFailed, ContainerStatuses: []corev1.ContainerStatus{{
|
|
Name: containerName, State: corev1.ContainerState{Terminated: &corev1.ContainerStateTerminated{ExitCode: 137, Reason: "OOMKilled"}},
|
|
}}},
|
|
}
|
|
ops, err := NewK8sRunner(fake.NewSimpleClientset(j, pod)).Ops(context.Background(), "minecraft", "survival")
|
|
if err != nil || len(ops) != 1 || ops[0].State != OpFailed || ops[0].Reason != "OOMKilled" {
|
|
t.Fatalf("Ops = %+v, %v; want the one op failed for OOMKilled", ops, err)
|
|
}
|
|
}
|
|
|
|
// TestK8sRunnerOpsFailsLoudly checks a listing the cluster refused is an error,
|
|
// never an empty list that would read as nothing running.
|
|
func TestK8sRunnerOpsFailsLoudly(t *testing.T) {
|
|
for _, resource := range []string{"jobs", "pods"} {
|
|
t.Run(resource, func(t *testing.T) {
|
|
cs := fake.NewSimpleClientset()
|
|
cs.PrependReactor("list", resource, func(k8stesting.Action) (bool, runtime.Object, error) {
|
|
return true, nil, fmt.Errorf("forbidden")
|
|
})
|
|
ops, err := NewK8sRunner(cs).Ops(context.Background(), "minecraft", "survival")
|
|
if err == nil || ops != nil {
|
|
t.Fatalf("Ops = %v, %v; want the refusal", ops, err)
|
|
}
|
|
})
|
|
}
|
|
}
|
|
|
|
// TestK8sRunnerStart checks Start creates the rendered Job and nothing else,
|
|
// and refuses params the renderer refuses without touching the cluster.
|
|
func TestK8sRunnerStart(t *testing.T) {
|
|
cs := fake.NewSimpleClientset()
|
|
p := testParams(OpUnzip)
|
|
p.Path, p.Async = "maps/world.zip", true
|
|
if err := NewK8sRunner(cs).Start(context.Background(), p); err != nil {
|
|
t.Fatalf("Start: %v", err)
|
|
}
|
|
want, _ := FilesJob(p)
|
|
got, err := cs.BatchV1().Jobs("minecraft").Get(context.Background(), want.Name, metav1.GetOptions{})
|
|
if err != nil {
|
|
t.Fatalf("the Job was not created: %v", err)
|
|
}
|
|
if got.Labels[LabelAsync] != "true" || got.Annotations[AnnotationPath] != "maps/world.zip" {
|
|
t.Fatalf("created %+v", got.ObjectMeta)
|
|
}
|
|
if n := len(cs.Actions()); n != 2 { // the create, and this test's get
|
|
t.Fatalf("%d calls to the cluster, want the one create", n-1)
|
|
}
|
|
|
|
cs = fake.NewSimpleClientset()
|
|
p.Op = OpRead
|
|
if err := NewK8sRunner(cs).Start(context.Background(), p); err == nil || len(cs.Actions()) != 0 {
|
|
t.Fatalf("err %v after %d calls, want a refusal before any", err, len(cs.Actions()))
|
|
}
|
|
|
|
cs = fake.NewSimpleClientset()
|
|
cs.PrependReactor("create", "jobs", func(k8stesting.Action) (bool, runtime.Object, error) {
|
|
return true, nil, fmt.Errorf("quota exceeded")
|
|
})
|
|
p.Op = OpUnzip
|
|
if err := NewK8sRunner(cs).Start(context.Background(), p); err == nil {
|
|
t.Fatal("a refused create must be an error")
|
|
}
|
|
}
|