Skip to content

Commit ff45329

Browse files
fix(evals): contain completion-harness task paths and validate timeout (#1049)
* fix(evals): contain completion-harness task paths and validate timeout Confine custom task-set fixture/script/verify references to the harness directory, mark the version-controlled grading trust boundary, reject degenerate timeout values, and re-record the baseline against the current tree. * fix(evals): align baseline and totals with gate-suspension signal Carry the gate-suspension total through schema, aggregation, and summary so the checked-in baseline validates. Add a regression test asserting the frozen baseline against the current report schema. * fix(evals): label estimated turns and failed tool calls honestly (#1062)
1 parent 4827e63 commit ff45329

5 files changed

Lines changed: 426 additions & 51 deletions

File tree

‎evals/completion/baseline-2026-09-14.json‎

Lines changed: 22 additions & 21 deletions
Original file line numberDiff line numberDiff line change
@@ -1,9 +1,9 @@
11
{
22
"harness": "completion-baseline",
33
"version": 1,
4-
"startedAt": "2026-09-14T07:30:17.369Z",
5-
"finishedAt": "2026-09-14T07:30:19.083Z",
6-
"commitSha": "b509ed3edb0db5368eb190528d3beaee8895837b",
4+
"startedAt": "2026-09-14T20:11:37.538Z",
5+
"finishedAt": "2026-09-14T20:11:41.269Z",
6+
"commitSha": "b1bace7e4a6c469050df048d53e7d1dea4b89ab6",
77
"provider": "stub-scripted",
88
"model": "completion-baseline-v1",
99
"repeats": 2,
@@ -25,8 +25,8 @@
2525
"doomLoopInterventions": 0,
2626
"thrashInterventions": 0,
2727
"gateSuspensions": 0,
28-
"agentDurationMs": 163,
29-
"verifyDurationMs": 149,
28+
"agentDurationMs": 567,
29+
"verifyDurationMs": 196,
3030
"verifyExitCode": 0,
3131
"overBudget": false
3232
},
@@ -45,8 +45,8 @@
4545
"doomLoopInterventions": 0,
4646
"thrashInterventions": 0,
4747
"gateSuspensions": 0,
48-
"agentDurationMs": 62,
49-
"verifyDurationMs": 141,
48+
"agentDurationMs": 167,
49+
"verifyDurationMs": 151,
5050
"verifyExitCode": 0,
5151
"overBudget": false
5252
},
@@ -65,8 +65,8 @@
6565
"doomLoopInterventions": 0,
6666
"thrashInterventions": 0,
6767
"gateSuspensions": 0,
68-
"agentDurationMs": 131,
69-
"verifyDurationMs": 135,
68+
"agentDurationMs": 72,
69+
"verifyDurationMs": 188,
7070
"verifyExitCode": 0,
7171
"overBudget": false
7272
},
@@ -85,8 +85,8 @@
8585
"doomLoopInterventions": 0,
8686
"thrashInterventions": 0,
8787
"gateSuspensions": 0,
88-
"agentDurationMs": 66,
89-
"verifyDurationMs": 127,
88+
"agentDurationMs": 68,
89+
"verifyDurationMs": 164,
9090
"verifyExitCode": 0,
9191
"overBudget": false
9292
},
@@ -105,8 +105,8 @@
105105
"doomLoopInterventions": 0,
106106
"thrashInterventions": 0,
107107
"gateSuspensions": 0,
108-
"agentDurationMs": 30,
109-
"verifyDurationMs": 72,
108+
"agentDurationMs": 25,
109+
"verifyDurationMs": 99,
110110
"verifyExitCode": 1,
111111
"overBudget": false
112112
},
@@ -125,8 +125,8 @@
125125
"doomLoopInterventions": 0,
126126
"thrashInterventions": 0,
127127
"gateSuspensions": 0,
128-
"agentDurationMs": 22,
129-
"verifyDurationMs": 65,
128+
"agentDurationMs": 27,
129+
"verifyDurationMs": 86,
130130
"verifyExitCode": 1,
131131
"overBudget": false
132132
},
@@ -145,8 +145,8 @@
145145
"doomLoopInterventions": 1,
146146
"thrashInterventions": 0,
147147
"gateSuspensions": 0,
148-
"agentDurationMs": 95,
149-
"verifyDurationMs": 16,
148+
"agentDurationMs": 318,
149+
"verifyDurationMs": 30,
150150
"verifyExitCode": 2,
151151
"overBudget": false,
152152
"error": "reactor error: Doom loop detected: an identical tool batch (read_file) executed 3 times consecutively"
@@ -166,8 +166,8 @@
166166
"doomLoopInterventions": 1,
167167
"thrashInterventions": 0,
168168
"gateSuspensions": 0,
169-
"agentDurationMs": 65,
170-
"verifyDurationMs": 17,
169+
"agentDurationMs": 332,
170+
"verifyDurationMs": 52,
171171
"verifyExitCode": 2,
172172
"overBudget": false,
173173
"error": "reactor error: Doom loop detected: an identical tool batch (read_file) executed 3 times consecutively"
@@ -179,10 +179,11 @@
179179
"completedRuns": 4,
180180
"completionRate": 0.5,
181181
"meanTurnsToCompletion": 3,
182-
"meanAgentDurationMs": 79.25,
182+
"meanAgentDurationMs": 197,
183183
"totalRetries": 0,
184184
"totalCompactionEvents": 0,
185185
"totalDoomLoopInterventions": 2,
186-
"totalThrashInterventions": 0
186+
"totalThrashInterventions": 0,
187+
"totalGateSuspensions": 0
187188
}
188189
}

‎evals/completion/lib.test.ts‎

Lines changed: 168 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -1,11 +1,18 @@
11
import { describe, expect, test } from "bun:test";
2+
import { readFile } from "node:fs/promises";
3+
import { dirname, join } from "node:path";
4+
import { fileURLToPath } from "node:url";
25
import {
36
CompletionReport,
7+
REPORT_VERSION,
8+
assertTaskSetContained,
49
computeTotals,
510
formatSummary,
611
isCompletedRun,
12+
migrateLegacyCompletionReport,
713
parseResponderScript,
814
parseTaskSetFile,
15+
resolveTaskRelativePath,
916
TaskResult,
1017
type CompletionReport as CompletionReportType,
1118
type TaskResult as TaskResultType,
@@ -20,9 +27,9 @@ const result = (overrides: Partial<TaskResultType> = {}): TaskResultType =>
2027
completed: true,
2128
runStatus: "completed",
2229
turnsUsed: 3,
30+
turnsEstimated: false,
2331
toolCallCount: 2,
2432
failedToolCalls: 0,
25-
retryCount: 0,
2633
compactionEvents: 0,
2734
doomLoopInterventions: 0,
2835
thrashInterventions: 0,
@@ -47,18 +54,19 @@ describe("completion predicate", () => {
4754
describe("completion totals", () => {
4855
test("mixed outcomes yield a fractional rate and completion-only turn mean", () => {
4956
const totals = computeTotals([
50-
result({ turnsUsed: 3, agentDurationMs: 1000 }),
57+
result({ turnsUsed: 3, agentDurationMs: 1000, verifyDurationMs: 100 }),
5158
result({
5259
taskId: "stall-read",
5360
profile: "stall",
5461
completed: false,
5562
runStatus: "failed",
5663
turnsUsed: 4,
5764
toolCallCount: 4,
58-
retryCount: 1,
65+
failedToolCalls: 1,
5966
doomLoopInterventions: 1,
6067
verifyExitCode: 1,
6168
agentDurationMs: 3000,
69+
verifyDurationMs: 300,
6270
}),
6371
]);
6472
expect(totals.tasksTotal).toBe(2);
@@ -67,7 +75,8 @@ describe("completion totals", () => {
6775
expect(totals.completionRate).toBe(0.5);
6876
expect(totals.meanTurnsToCompletion).toBe(3);
6977
expect(totals.meanAgentDurationMs).toBe(2000);
70-
expect(totals.totalRetries).toBe(1);
78+
expect(totals.meanVerifyDurationMs).toBe(200);
79+
expect(totals.totalFailedToolCalls).toBe(1);
7180
expect(totals.totalDoomLoopInterventions).toBe(1);
7281
});
7382

@@ -76,6 +85,8 @@ describe("completion totals", () => {
7685
expect(totals.completionRate).toBe(0);
7786
expect(totals.meanTurnsToCompletion).toBe(0);
7887
expect(totals.meanAgentDurationMs).toBe(0);
88+
expect(totals.meanVerifyDurationMs).toBe(0);
89+
expect(totals.totalFailedToolCalls).toBe(0);
7990
});
8091

8192
test("aggregates gate suspensions across runs", () => {
@@ -117,15 +128,68 @@ describe("boundary parsing", () => {
117128
});
118129
});
119130

131+
describe("task path containment", () => {
132+
const root = "/repo/evals/completion";
133+
134+
test("keeps relative fixture paths inside the harness directory", () => {
135+
expect(resolveTaskRelativePath(root, "tasks/sum-fix/fixture")).toBe(
136+
"/repo/evals/completion/tasks/sum-fix/fixture",
137+
);
138+
});
139+
140+
test("rejects absolute escapes", () => {
141+
expect(() => resolveTaskRelativePath(root, "/etc/passwd")).toThrow(
142+
/escapes the harness directory/,
143+
);
144+
});
145+
146+
test("rejects dot-dot escapes outside the harness directory", () => {
147+
expect(() => resolveTaskRelativePath(root, "../capability/lib.ts")).toThrow(
148+
/escapes the harness directory/,
149+
);
150+
expect(() =>
151+
resolveTaskRelativePath(root, "tasks/../../package.json"),
152+
).toThrow(/escapes the harness directory/);
153+
});
154+
155+
test("rejects an empty reference", () => {
156+
expect(() => resolveTaskRelativePath(root, " ")).toThrow(
157+
/escapes the harness directory/,
158+
);
159+
});
160+
161+
test("assertTaskSetContained rejects a task set with an escaped grader", () => {
162+
const taskSet = parseTaskSetFile({
163+
version: 1,
164+
note: "x",
165+
tasks: [
166+
{
167+
id: "evil",
168+
title: "evil",
169+
profile: "solve",
170+
prompt: "p",
171+
fixture: "tasks/sum-fix/fixture",
172+
script: "tasks/sum-fix/script.json",
173+
verify: "/tmp/evil.sh",
174+
maxTurns: 1,
175+
},
176+
],
177+
});
178+
expect(() => assertTaskSetContained(taskSet, root)).toThrow(
179+
/escapes the harness directory/,
180+
);
181+
});
182+
});
183+
120184
describe("human summary", () => {
121185
test("names the harness, provenance, rate, and every run", () => {
122186
const results = [
123187
result(),
124-
result({ taskId: "stall-read", completed: false }),
188+
result({ taskId: "stall-read", completed: false, turnsEstimated: true }),
125189
];
126190
const report: CompletionReportType = CompletionReport.assert({
127191
harness: "completion-baseline",
128-
version: 1,
192+
version: REPORT_VERSION,
129193
startedAt: "2026-09-14T00:00:00.000Z",
130194
finishedAt: "2026-09-14T00:01:00.000Z",
131195
commitSha: "deadbeef",
@@ -141,8 +205,106 @@ describe("human summary", () => {
141205
expect(summary).toContain("completion rate 50.0% (1/2 runs)");
142206
expect(summary).toContain("deadbeef");
143207
expect(summary).toContain("stub-scripted");
208+
expect(summary).toContain("mean verify time 300ms");
209+
expect(summary).toContain("failed tool calls 0");
210+
expect(summary).not.toContain("retries");
144211
expect(summary).toContain("version-endpoint r0: complete");
145212
expect(summary).toContain("stall-read r0: incomplete");
213+
expect(summary).toContain("turns 3 (estimated)");
214+
expect(summary).toContain("verify 300ms");
215+
});
216+
});
217+
218+
describe("legacy v1 migration", () => {
219+
const legacyResult = (overrides: Record<string, unknown> = {}) => ({
220+
taskId: "version-endpoint",
221+
title: "Add GET /version",
222+
profile: "solve",
223+
repeat: 0,
224+
completed: true,
225+
runStatus: "completed",
226+
turnsUsed: 3,
227+
toolCallCount: 2,
228+
failedToolCalls: 0,
229+
retryCount: 0,
230+
compactionEvents: 0,
231+
doomLoopInterventions: 0,
232+
thrashInterventions: 0,
233+
gateSuspensions: 0,
234+
agentDurationMs: 1200,
235+
verifyDurationMs: 300,
236+
verifyExitCode: 0,
237+
overBudget: false,
238+
...overrides,
239+
});
240+
241+
const legacyReport = (results: Record<string, unknown>[]) => ({
242+
harness: "completion-baseline",
243+
version: 1,
244+
startedAt: "2026-09-14T00:00:00.000Z",
245+
finishedAt: "2026-09-14T00:01:00.000Z",
246+
commitSha: "deadbeef",
247+
provider: "stub-scripted",
248+
model: "completion-baseline-v1",
249+
repeats: 1,
250+
taskSetVersion: 1,
251+
taskIds: ["version-endpoint"],
252+
results,
253+
totals: {
254+
tasksTotal: 1,
255+
runsTotal: results.length,
256+
completedRuns: 1,
257+
completionRate: 1,
258+
meanTurnsToCompletion: 3,
259+
meanAgentDurationMs: 1200,
260+
totalRetries: 0,
261+
totalCompactionEvents: 0,
262+
totalDoomLoopInterventions: 0,
263+
totalThrashInterventions: 0,
264+
totalGateSuspensions: 0,
265+
},
266+
});
267+
268+
test("drops the mislabeled retry count and aggregates verify durations", () => {
269+
const migrated = migrateLegacyCompletionReport(
270+
legacyReport([legacyResult(), legacyResult({ repeat: 1 })]),
271+
);
272+
expect(migrated.version).toBe(REPORT_VERSION);
273+
expect(migrated.totals.totalFailedToolCalls).toBe(0);
274+
expect(migrated.totals.meanVerifyDurationMs).toBe(300);
275+
for (const migratedResult of migrated.results) {
276+
expect("retryCount" in migratedResult).toBe(false);
277+
expect("turnsEstimated" in migratedResult).toBe(false);
278+
}
279+
});
280+
281+
test("rejects a legacy result whose retry count is not a failure count", () => {
282+
expect(() =>
283+
migrateLegacyCompletionReport(
284+
legacyReport([legacyResult({ retryCount: 2, failedToolCalls: 1 })]),
285+
),
286+
).toThrow(/mislabels retries/);
287+
});
288+
289+
test("rejects a non-v1 report", () => {
290+
const payload = legacyReport([legacyResult()]);
291+
payload.version = REPORT_VERSION;
292+
expect(() => migrateLegacyCompletionReport(payload)).toThrow(
293+
/Only v1 reports can be migrated/,
294+
);
295+
});
296+
});
297+
298+
describe("checked-in baseline", () => {
299+
test("the frozen v1 baseline migrates to the current report schema", async () => {
300+
const dir = dirname(fileURLToPath(import.meta.url));
301+
const raw = await readFile(join(dir, "baseline-2026-09-14.json"), "utf8");
302+
const report = migrateLegacyCompletionReport(JSON.parse(raw));
303+
expect(report.totals.runsTotal).toBe(report.results.length);
304+
expect(report.totals.runsTotal).toBe(8);
305+
expect(report.totals.completionRate).toBe(0.5);
306+
expect(report.totals.totalFailedToolCalls).toBe(0);
307+
expect(report.totals.meanVerifyDurationMs).toBe(120.75);
146308
});
147309

148310
test("prints the aggregated gate suspensions", () => {

0 commit comments

Comments
 (0)