返回 DeepSeek-Reasonix
checkpoints_test.go
根目录 / cmd / e2ebench / checkpoints_test.go
1 package main
2
3 import (
4 "fmt"
5 "os"
6 "path/filepath"
7 "strings"
8 "testing"
9 "time"
10 )
11
12 func writeTaskVerify(t *testing.T, taskDir, script string) {
13 t.Helper()
14 if err := os.MkdirAll(taskDir, 0o755); err != nil {
15 t.Fatal(err)
16 }
17 if err := os.WriteFile(filepath.Join(taskDir, "verify.sh"), []byte(script), 0o644); err != nil {
18 t.Fatal(err)
19 }
20 }
21
22 func TestGradeCheckpointsFindsEarliestCorrectState(t *testing.T) {
23 requireRealBash(t)
24 taskDir := filepath.Join(t.TempDir(), "task")
25 writeTaskVerify(t, taskDir, "#!/usr/bin/env bash\ngrep -q done answer.txt\n")
26
27 snaps := t.TempDir()
28 mk := func(seq int, elapsed int64, content string) checkpoint {
29 dir := filepath.Join(snaps, filepath.Base(strings.ReplaceAll(content, " ", "-"))+"-cp")
30 dir = filepath.Join(snaps, filepath.Base(dir)+"-"+strings.ReplaceAll(content, " ", "_"))
31 if err := os.MkdirAll(dir, 0o755); err != nil {
32 t.Fatal(err)
33 }
34 if err := os.WriteFile(filepath.Join(dir, "answer.txt"), []byte(content), 0o644); err != nil {
35 t.Fatal(err)
36 }
37 return checkpoint{Seq: seq, ElapsedMs: elapsed, dir: dir}
38 }
39 checkpoints := gradeCheckpoints([]checkpoint{
40 mk(1, 18_000, "not yet"),
41 mk(2, 63_000, "done"),
42 mk(3, 128_000, "done and polished"),
43 }, taskDir)
44 if checkpoints[0].Pass || !checkpoints[1].Pass || !checkpoints[2].Pass {
45 t.Fatalf("pass flags = %+v", checkpoints)
46 }
47
48 firstMs, broke := firstCorrect(checkpoints, true)
49 if firstMs != 63_000 || broke {
50 t.Fatalf("firstCorrect = %d, %v; want 63000, false", firstMs, broke)
51 }
52
53 // A run whose final state failed after a passing snapshot is the
54 // solved-then-broke alarm.
55 firstMs, broke = firstCorrect(checkpoints, false)
56 if firstMs != 63_000 || !broke {
57 t.Fatalf("solved-then-broke = %d, %v; want 63000, true", firstMs, broke)
58 }
59
60 if ms, broke := firstCorrect([]checkpoint{{Seq: 1, ElapsedMs: 5, dir: snaps}}, true); ms != 0 || broke {
61 t.Fatalf("no passing snapshot must yield 0/false, got %d/%v", ms, broke)
62 }
63 }
64
65 func TestSnapshotterCapturesWorkspaceChanges(t *testing.T) {
66 work := t.TempDir()
67 dst := t.TempDir()
68 if err := os.WriteFile(filepath.Join(work, "code.py"), []byte("v1"), 0o644); err != nil {
69 t.Fatal(err)
70 }
71 polls := make(chan time.Time)
72 acks := make(chan struct{})
73 snap := startSnapshotterWithPoll(work, dst, time.Now(), polls, acks)
74 poll := func() {
75 t.Helper()
76 polls <- time.Time{}
77 <-acks
78 }
79 poll()
80 // The metrics sidecar updating must not trigger a snapshot on its own.
81 if err := os.WriteFile(filepath.Join(work, ".run-metrics.json"), []byte("{}"), 0o644); err != nil {
82 t.Fatal(err)
83 }
84 poll()
85 if err := os.WriteFile(filepath.Join(work, "code.py"), []byte("v2 changed"), 0o644); err != nil {
86 t.Fatal(err)
87 }
88 poll()
89 taken := snap.halt()
90
91 if len(taken) != 1 {
92 t.Fatalf("snapshots = %d (%+v), want exactly 1 (the code change; metrics writes excluded)", len(taken), taken)
93 }
94 data, err := os.ReadFile(filepath.Join(taken[0].dir, "code.py"))
95 if err != nil || string(data) != "v2 changed" {
96 t.Fatalf("snapshot content = %q, %v", data, err)
97 }
98 if _, err := os.Stat(filepath.Join(taken[0].dir, ".run-metrics.json")); !os.IsNotExist(err) {
99 t.Fatal("metrics sidecar must be stripped from snapshots")
100 }
101 }
102
103 func TestKPILineIncludesTTFCS(t *testing.T) {
104 r := result{task: task{ID: "a"}, Passed: true, WallMs: 142_000, Attempt: 1, TTCSMs: 142_000}
105 r.FirstCorrectMs = 63_000
106 r.PostSolveWasteMs = 79_000
107 got := renderBody([]result{r})
108 for _, want := range []string{
109 "**TTFCS median** 1m03s",
110 "**post-solve waste median** 1m19s",
111 } {
112 if !strings.Contains(got, want) {
113 t.Fatalf("KPI line missing %q:\n%s", want, got)
114 }
115 }
116 }
117
118 func TestSolveProfileTriage(t *testing.T) {
119 cp := []checkpoint{{Seq: 1, ElapsedMs: 1000}}
120 early := result{Passed: true, WallMs: 140_000, FirstCorrectMs: 55_000, PostSolveWasteMs: 85_000, Checkpoints: cp}
121 late := result{Passed: true, WallMs: 140_000, FirstCorrectMs: 132_000, PostSolveWasteMs: 8_000, Checkpoints: cp}
122 never := result{Passed: false, Checkpoints: cp}
123 broke := result{Passed: false, SolvedThenBroken: true, Checkpoints: cp}
124 finalOnly := result{Passed: true, WallMs: 30_000, Checkpoints: cp}
125 off := result{Passed: true}
126
127 for want, r := range map[string]result{
128 "early_correct": early, "late_correct": late, "never_correct": never,
129 "solved_then_broke": broke, "": off,
130 } {
131 if got := solveProfile(r); got != want {
132 t.Fatalf("solveProfile = %q, want %q", got, want)
133 }
134 }
135 if got := solveProfile(finalOnly); got != "late_correct" {
136 t.Fatalf("final-only pass = %q, want late_correct", got)
137 }
138
139 line := renderSolveProfiles([]result{early, late, never, broke})
140 for _, want := range []string{
141 "**early_correct** 1 (median waste 1m25s)",
142 "**late_correct** 1",
143 "**never_correct** 1",
144 "**solved_then_broke** 1",
145 } {
146 if !strings.Contains(line, want) {
147 t.Fatalf("solve profile line missing %q:\n%s", want, line)
148 }
149 }
150 if renderSolveProfiles([]result{off}) != "" {
151 t.Fatal("uncheckpointed suites must not render the line")
152 }
153 }
154
155 func TestCorrectBoundaryMetrics(t *testing.T) {
156 cps := []checkpoint{
157 {Seq: 1, ElapsedMs: 10, Pass: false},
158 {Seq: 2, ElapsedMs: 20, Pass: false},
159 {Seq: 3, ElapsedMs: 30, Pass: true},
160 {Seq: 4, ElapsedMs: 40, Pass: false},
161 {Seq: 5, ElapsedMs: 50, Pass: true},
162 }
163 if got := mutationsBeforeCorrect(cps); got != 2 {
164 t.Fatalf("mutations before correct = %d, want 2", got)
165 }
166 if !regressedAfterCorrect(cps) {
167 t.Fatal("PASS→FAIL→PASS must count as a regression even though it was repaired")
168 }
169 if regressedAfterCorrect(cps[:3]) {
170 t.Fatal("no regression before the first failure-after-pass")
171 }
172 if got := mutationsBeforeCorrect(cps[:2]); got != 2 {
173 t.Fatalf("all-failing run: mutations = %d, want len", got)
174 }
175 }
176
177 func TestRoundsSplitAt(t *testing.T) {
178 path := filepath.Join(t.TempDir(), "split.trajectory.jsonl")
179 lines := []string{
180 `{"seq":1,"ts":1000,"event":{"kind":"turn_started"}}`,
181 `{"seq":2,"ts":2000,"event":{"kind":"tool_dispatch","tool":{"id":"a","name":"write_file"}}}`,
182 `{"seq":3,"ts":2100,"event":{"kind":"tool_result","tool":{"id":"a","name":"write_file","durationMs":100}}}`,
183 `{"seq":4,"ts":3000,"event":{"kind":"tool_dispatch","tool":{"id":"b","name":"bash"}}}`,
184 `{"seq":5,"ts":3200,"event":{"kind":"tool_result","tool":{"id":"b","name":"bash","readOnly":true,"durationMs":200,"execution":{"verification":"passed"}}}}`,
185 `{"seq":6,"ts":5000,"event":{"kind":"tool_dispatch","tool":{"id":"c","name":"bash"}}}`,
186 `{"seq":7,"ts":5200,"event":{"kind":"tool_result","tool":{"id":"c","name":"bash","readOnly":true,"durationMs":200,"execution":{"verification":"passed"}}}}`,
187 `{"seq":8,"ts":6000,"event":{"kind":"tool_dispatch","tool":{"id":"d","name":"read_file","readOnly":true}}}`,
188 `{"seq":9,"ts":6100,"event":{"kind":"tool_result","tool":{"id":"d","name":"read_file","readOnly":true,"durationMs":100}}}`,
189 `{"seq":10,"ts":7000,"event":{"kind":"turn_done"}}`,
190 }
191 if err := writeLines(path, lines); err != nil {
192 t.Fatal(err)
193 }
194 split := splitAtCorrect(path, 4000)
195 if split.RoundsBefore != 2 || split.RoundsAfter != 2 || split.VerifyAfter != 1 {
196 t.Fatalf("split = %+v, want 2 rounds before, 2 after, 1 verification after", split)
197 }
198 if split.CallsBefore != 2 || split.CallsAfter != 2 {
199 t.Fatalf("calls = %d/%d, want 2/2", split.CallsBefore, split.CallsAfter)
200 }
201 if split.MutationsAfter != 0 {
202 t.Fatalf("read-only tail must count no mutations, got %d", split.MutationsAfter)
203 }
204 }
205
206 func TestComputeStopEvalCurveAndHarmfulContinuations(t *testing.T) {
207 cps := []checkpoint{
208 {Seq: 1, ElapsedMs: 8_000, Pass: false},
209 {Seq: 2, ElapsedMs: 18_000, Pass: true},
210 {Seq: 3, ElapsedMs: 28_000, Pass: true},
211 {Seq: 4, ElapsedMs: 38_000, Pass: false}, // the "improvement" that broke it
212 {Seq: 5, ElapsedMs: 48_000, Pass: true},
213 }
214 rounds := []int64{10_000, 20_000, 30_000, 40_000, 50_000}
215 eval := computeStopEval(cps, rounds)
216 want := []bool{false, true, true, false, true}
217 for i, pass := range want {
218 if eval.Curve[i] != pass {
219 t.Fatalf("curve = %v, want %v", eval.Curve, want)
220 }
221 }
222 if eval.FirstStoppableRound != 2 {
223 t.Fatalf("first stoppable = %d, want 2", eval.FirstStoppableRound)
224 }
225 if eval.HarmfulContinuation != 1 {
226 t.Fatalf("harmful continuations = %d, want 1 (round 4 destroyed a passing state)", eval.HarmfulContinuation)
227 }
228 if eval.ContinuationsPast != 3 {
229 t.Fatalf("continuations past stoppable = %d, want 3", eval.ContinuationsPast)
230 }
231
232 if computeStopEval(nil, rounds) != nil || computeStopEval(cps, nil) != nil {
233 t.Fatal("missing inputs must yield no eval")
234 }
235 // A boundary before any snapshot grades as the seed: fail.
236 early := computeStopEval(cps, []int64{1_000})
237 if early.Curve[0] || early.FirstStoppableRound != 0 {
238 t.Fatalf("pre-snapshot boundary must fail: %+v", early)
239 }
240 }
241
242 func TestOverthinkingDamageRateInKPIAndCompare(t *testing.T) {
243 damaged := result{task: task{ID: "a"}, Passed: true, WallMs: 60_000, Attempt: 1, TTCSMs: 60_000}
244 damaged.FirstCorrectMs = 20_000
245 damaged.PostSolveWasteMs = 40_000
246 damaged.RegressedAfterCorrect = true
247 clean := result{task: task{ID: "b"}, Passed: true, WallMs: 30_000, Attempt: 1, TTCSMs: 30_000}
248 clean.FirstCorrectMs = 25_000
249 clean.PostSolveWasteMs = 5_000
250
251 got := renderBody([]result{damaged, clean})
252 if !strings.Contains(got, "**overthinking damage** 50%") {
253 t.Fatalf("KPI line missing damage rate:\n%s", got)
254 }
255 }
256
257 func TestFirstUsefulMutationApproximatesTTFUM(t *testing.T) {
258 seed := t.TempDir()
259 final := t.TempDir()
260 snaps := t.TempDir()
261 write := func(dir, name, content string) {
262 t.Helper()
263 if err := os.WriteFile(filepath.Join(dir, name), []byte(content), 0o644); err != nil {
264 t.Fatal(err)
265 }
266 }
267 write(seed, "util.py", "v0")
268 write(seed, "keep.py", "same")
269 write(final, "util.py", "final fix")
270 write(final, "keep.py", "same")
271 write(final, "helper.py", "created")
272 write(final, "verify.sh", "grader")
273
274 mk := func(seq int, elapsed int64, utilContent string) checkpoint {
275 dir := filepath.Join(snaps, fmt.Sprintf("%03d", seq))
276 if err := os.MkdirAll(dir, 0o755); err != nil {
277 t.Fatal(err)
278 }
279 write(dir, "util.py", utilContent)
280 return checkpoint{Seq: seq, ElapsedMs: elapsed, dir: dir}
281 }
282 cps := []checkpoint{
283 mk(1, 10_000, "wrong attempt"),
284 mk(2, 20_000, "final fix"), // part of the final solution appears
285 mk(3, 30_000, "final fix"),
286 }
287 if got := firstUsefulMutation(cps, seed, final); got != 20_000 {
288 t.Fatalf("TTFUM = %d, want 20000", got)
289 }
290
291 // A created file reaching its final content also counts.
292 write(filepath.Join(snaps, "001"), "helper.py", "created")
293 if got := firstUsefulMutation(cps, seed, final); got != 10_000 {
294 t.Fatalf("TTFUM with created file = %d, want 10000", got)
295 }
296
297 // Unchanged and harness files are never solution files.
298 files := solutionFiles(seed, final)
299 if _, ok := files["keep.py"]; ok {
300 t.Fatal("unchanged file counted as solution")
301 }
302 if _, ok := files["verify.sh"]; ok {
303 t.Fatal("grader counted as solution")
304 }
305 if len(files) != 2 {
306 t.Fatalf("solution files = %v", files)
307 }
308 }
309
310 func TestDiagnosisAppliesThePriorityTree(t *testing.T) {
311 cps := []checkpoint{{Seq: 1, ElapsedMs: 1000}}
312 wasteful := result{task: task{ID: "a"}, Passed: true, WallMs: 100_000, Checkpoints: cps,
313 FirstCorrectMs: 30_000, PostSolveWasteMs: 70_000, FirstUsefulMs: 20_000}
314 explorer := result{task: task{ID: "b"}, Passed: true, WallMs: 100_000, Checkpoints: cps,
315 FirstCorrectMs: 90_000, PostSolveWasteMs: 10_000, FirstUsefulMs: 85_000}
316 never := result{task: task{ID: "c"}, Checkpoints: cps}
317 broken := result{task: task{ID: "d"}, Passed: true, WallMs: 50_000, Checkpoints: cps,
318 FirstCorrectMs: 10_000, PostSolveWasteMs: 5_000, RegressedAfterCorrect: true}
319
320 if got := renderDiagnosis([]result{wasteful}); !strings.Contains(got, "post-solve waste dominates") {
321 t.Fatalf("waste verdict missing: %s", got)
322 }
323 if got := renderDiagnosis([]result{explorer}); !strings.Contains(got, "exploration dominates") {
324 t.Fatalf("exploration verdict missing: %s", got)
325 }
326 if got := renderDiagnosis([]result{never, never, wasteful}); !strings.Contains(got, "never-correct dominates") {
327 t.Fatalf("capability verdict must outrank latency: %s", got)
328 }
329 if got := renderDiagnosis([]result{broken, wasteful, wasteful}); !strings.Contains(got, "correct→incorrect regressions") {
330 t.Fatalf("damage verdict must outrank waste-vs-exploration: %s", got)
331 }
332 if renderDiagnosis([]result{{task: task{ID: "x"}}}) != "" {
333 t.Fatal("uncheckpointed suites carry no diagnosis")
334 }
335 }
336
336 lines GO