11import { describe , expect , test } from "bun:test" ;
2+ import { readFile } from "node:fs/promises" ;
3+ import { dirname , join } from "node:path" ;
4+ import { fileURLToPath } from "node:url" ;
25import {
36 CompletionReport ,
7+ REPORT_VERSION ,
8+ assertTaskSetContained ,
49 computeTotals ,
510 formatSummary ,
611 isCompletedRun ,
12+ migrateLegacyCompletionReport ,
713 parseResponderScript ,
814 parseTaskSetFile ,
15+ resolveTaskRelativePath ,
916 TaskResult ,
1017 type CompletionReport as CompletionReportType ,
1118 type TaskResult as TaskResultType ,
@@ -20,9 +27,9 @@ const result = (overrides: Partial<TaskResultType> = {}): TaskResultType =>
2027 completed : true ,
2128 runStatus : "completed" ,
2229 turnsUsed : 3 ,
30+ turnsEstimated : false ,
2331 toolCallCount : 2 ,
2432 failedToolCalls : 0 ,
25- retryCount : 0 ,
2633 compactionEvents : 0 ,
2734 doomLoopInterventions : 0 ,
2835 thrashInterventions : 0 ,
@@ -47,18 +54,19 @@ describe("completion predicate", () => {
4754describe ( "completion totals" , ( ) => {
4855 test ( "mixed outcomes yield a fractional rate and completion-only turn mean" , ( ) => {
4956 const totals = computeTotals ( [
50- result ( { turnsUsed : 3 , agentDurationMs : 1000 } ) ,
57+ result ( { turnsUsed : 3 , agentDurationMs : 1000 , verifyDurationMs : 100 } ) ,
5158 result ( {
5259 taskId : "stall-read" ,
5360 profile : "stall" ,
5461 completed : false ,
5562 runStatus : "failed" ,
5663 turnsUsed : 4 ,
5764 toolCallCount : 4 ,
58- retryCount : 1 ,
65+ failedToolCalls : 1 ,
5966 doomLoopInterventions : 1 ,
6067 verifyExitCode : 1 ,
6168 agentDurationMs : 3000 ,
69+ verifyDurationMs : 300 ,
6270 } ) ,
6371 ] ) ;
6472 expect ( totals . tasksTotal ) . toBe ( 2 ) ;
@@ -67,7 +75,8 @@ describe("completion totals", () => {
6775 expect ( totals . completionRate ) . toBe ( 0.5 ) ;
6876 expect ( totals . meanTurnsToCompletion ) . toBe ( 3 ) ;
6977 expect ( totals . meanAgentDurationMs ) . toBe ( 2000 ) ;
70- expect ( totals . totalRetries ) . toBe ( 1 ) ;
78+ expect ( totals . meanVerifyDurationMs ) . toBe ( 200 ) ;
79+ expect ( totals . totalFailedToolCalls ) . toBe ( 1 ) ;
7180 expect ( totals . totalDoomLoopInterventions ) . toBe ( 1 ) ;
7281 } ) ;
7382
@@ -76,6 +85,8 @@ describe("completion totals", () => {
7685 expect ( totals . completionRate ) . toBe ( 0 ) ;
7786 expect ( totals . meanTurnsToCompletion ) . toBe ( 0 ) ;
7887 expect ( totals . meanAgentDurationMs ) . toBe ( 0 ) ;
88+ expect ( totals . meanVerifyDurationMs ) . toBe ( 0 ) ;
89+ expect ( totals . totalFailedToolCalls ) . toBe ( 0 ) ;
7990 } ) ;
8091
8192 test ( "aggregates gate suspensions across runs" , ( ) => {
@@ -117,15 +128,68 @@ describe("boundary parsing", () => {
117128 } ) ;
118129} ) ;
119130
131+ describe ( "task path containment" , ( ) => {
132+ const root = "/repo/evals/completion" ;
133+
134+ test ( "keeps relative fixture paths inside the harness directory" , ( ) => {
135+ expect ( resolveTaskRelativePath ( root , "tasks/sum-fix/fixture" ) ) . toBe (
136+ "/repo/evals/completion/tasks/sum-fix/fixture" ,
137+ ) ;
138+ } ) ;
139+
140+ test ( "rejects absolute escapes" , ( ) => {
141+ expect ( ( ) => resolveTaskRelativePath ( root , "/etc/passwd" ) ) . toThrow (
142+ / e s c a p e s t h e h a r n e s s d i r e c t o r y / ,
143+ ) ;
144+ } ) ;
145+
146+ test ( "rejects dot-dot escapes outside the harness directory" , ( ) => {
147+ expect ( ( ) => resolveTaskRelativePath ( root , "../capability/lib.ts" ) ) . toThrow (
148+ / e s c a p e s t h e h a r n e s s d i r e c t o r y / ,
149+ ) ;
150+ expect ( ( ) =>
151+ resolveTaskRelativePath ( root , "tasks/../../package.json" ) ,
152+ ) . toThrow ( / e s c a p e s t h e h a r n e s s d i r e c t o r y / ) ;
153+ } ) ;
154+
155+ test ( "rejects an empty reference" , ( ) => {
156+ expect ( ( ) => resolveTaskRelativePath ( root , " " ) ) . toThrow (
157+ / e s c a p e s t h e h a r n e s s d i r e c t o r y / ,
158+ ) ;
159+ } ) ;
160+
161+ test ( "assertTaskSetContained rejects a task set with an escaped grader" , ( ) => {
162+ const taskSet = parseTaskSetFile ( {
163+ version : 1 ,
164+ note : "x" ,
165+ tasks : [
166+ {
167+ id : "evil" ,
168+ title : "evil" ,
169+ profile : "solve" ,
170+ prompt : "p" ,
171+ fixture : "tasks/sum-fix/fixture" ,
172+ script : "tasks/sum-fix/script.json" ,
173+ verify : "/tmp/evil.sh" ,
174+ maxTurns : 1 ,
175+ } ,
176+ ] ,
177+ } ) ;
178+ expect ( ( ) => assertTaskSetContained ( taskSet , root ) ) . toThrow (
179+ / e s c a p e s t h e h a r n e s s d i r e c t o r y / ,
180+ ) ;
181+ } ) ;
182+ } ) ;
183+
120184describe ( "human summary" , ( ) => {
121185 test ( "names the harness, provenance, rate, and every run" , ( ) => {
122186 const results = [
123187 result ( ) ,
124- result ( { taskId : "stall-read" , completed : false } ) ,
188+ result ( { taskId : "stall-read" , completed : false , turnsEstimated : true } ) ,
125189 ] ;
126190 const report : CompletionReportType = CompletionReport . assert ( {
127191 harness : "completion-baseline" ,
128- version : 1 ,
192+ version : REPORT_VERSION ,
129193 startedAt : "2026-09-14T00:00:00.000Z" ,
130194 finishedAt : "2026-09-14T00:01:00.000Z" ,
131195 commitSha : "deadbeef" ,
@@ -141,8 +205,106 @@ describe("human summary", () => {
141205 expect ( summary ) . toContain ( "completion rate 50.0% (1/2 runs)" ) ;
142206 expect ( summary ) . toContain ( "deadbeef" ) ;
143207 expect ( summary ) . toContain ( "stub-scripted" ) ;
208+ expect ( summary ) . toContain ( "mean verify time 300ms" ) ;
209+ expect ( summary ) . toContain ( "failed tool calls 0" ) ;
210+ expect ( summary ) . not . toContain ( "retries" ) ;
144211 expect ( summary ) . toContain ( "version-endpoint r0: complete" ) ;
145212 expect ( summary ) . toContain ( "stall-read r0: incomplete" ) ;
213+ expect ( summary ) . toContain ( "turns 3 (estimated)" ) ;
214+ expect ( summary ) . toContain ( "verify 300ms" ) ;
215+ } ) ;
216+ } ) ;
217+
218+ describe ( "legacy v1 migration" , ( ) => {
219+ const legacyResult = ( overrides : Record < string , unknown > = { } ) => ( {
220+ taskId : "version-endpoint" ,
221+ title : "Add GET /version" ,
222+ profile : "solve" ,
223+ repeat : 0 ,
224+ completed : true ,
225+ runStatus : "completed" ,
226+ turnsUsed : 3 ,
227+ toolCallCount : 2 ,
228+ failedToolCalls : 0 ,
229+ retryCount : 0 ,
230+ compactionEvents : 0 ,
231+ doomLoopInterventions : 0 ,
232+ thrashInterventions : 0 ,
233+ gateSuspensions : 0 ,
234+ agentDurationMs : 1200 ,
235+ verifyDurationMs : 300 ,
236+ verifyExitCode : 0 ,
237+ overBudget : false ,
238+ ...overrides ,
239+ } ) ;
240+
241+ const legacyReport = ( results : Record < string , unknown > [ ] ) => ( {
242+ harness : "completion-baseline" ,
243+ version : 1 ,
244+ startedAt : "2026-09-14T00:00:00.000Z" ,
245+ finishedAt : "2026-09-14T00:01:00.000Z" ,
246+ commitSha : "deadbeef" ,
247+ provider : "stub-scripted" ,
248+ model : "completion-baseline-v1" ,
249+ repeats : 1 ,
250+ taskSetVersion : 1 ,
251+ taskIds : [ "version-endpoint" ] ,
252+ results,
253+ totals : {
254+ tasksTotal : 1 ,
255+ runsTotal : results . length ,
256+ completedRuns : 1 ,
257+ completionRate : 1 ,
258+ meanTurnsToCompletion : 3 ,
259+ meanAgentDurationMs : 1200 ,
260+ totalRetries : 0 ,
261+ totalCompactionEvents : 0 ,
262+ totalDoomLoopInterventions : 0 ,
263+ totalThrashInterventions : 0 ,
264+ totalGateSuspensions : 0 ,
265+ } ,
266+ } ) ;
267+
268+ test ( "drops the mislabeled retry count and aggregates verify durations" , ( ) => {
269+ const migrated = migrateLegacyCompletionReport (
270+ legacyReport ( [ legacyResult ( ) , legacyResult ( { repeat : 1 } ) ] ) ,
271+ ) ;
272+ expect ( migrated . version ) . toBe ( REPORT_VERSION ) ;
273+ expect ( migrated . totals . totalFailedToolCalls ) . toBe ( 0 ) ;
274+ expect ( migrated . totals . meanVerifyDurationMs ) . toBe ( 300 ) ;
275+ for ( const migratedResult of migrated . results ) {
276+ expect ( "retryCount" in migratedResult ) . toBe ( false ) ;
277+ expect ( "turnsEstimated" in migratedResult ) . toBe ( false ) ;
278+ }
279+ } ) ;
280+
281+ test ( "rejects a legacy result whose retry count is not a failure count" , ( ) => {
282+ expect ( ( ) =>
283+ migrateLegacyCompletionReport (
284+ legacyReport ( [ legacyResult ( { retryCount : 2 , failedToolCalls : 1 } ) ] ) ,
285+ ) ,
286+ ) . toThrow ( / m i s l a b e l s r e t r i e s / ) ;
287+ } ) ;
288+
289+ test ( "rejects a non-v1 report" , ( ) => {
290+ const payload = legacyReport ( [ legacyResult ( ) ] ) ;
291+ payload . version = REPORT_VERSION ;
292+ expect ( ( ) => migrateLegacyCompletionReport ( payload ) ) . toThrow (
293+ / O n l y v 1 r e p o r t s c a n b e m i g r a t e d / ,
294+ ) ;
295+ } ) ;
296+ } ) ;
297+
298+ describe ( "checked-in baseline" , ( ) => {
299+ test ( "the frozen v1 baseline migrates to the current report schema" , async ( ) => {
300+ const dir = dirname ( fileURLToPath ( import . meta. url ) ) ;
301+ const raw = await readFile ( join ( dir , "baseline-2026-09-14.json" ) , "utf8" ) ;
302+ const report = migrateLegacyCompletionReport ( JSON . parse ( raw ) ) ;
303+ expect ( report . totals . runsTotal ) . toBe ( report . results . length ) ;
304+ expect ( report . totals . runsTotal ) . toBe ( 8 ) ;
305+ expect ( report . totals . completionRate ) . toBe ( 0.5 ) ;
306+ expect ( report . totals . totalFailedToolCalls ) . toBe ( 0 ) ;
307+ expect ( report . totals . meanVerifyDurationMs ) . toBe ( 120.75 ) ;
146308 } ) ;
147309
148310 test ( "prints the aggregated gate suspensions" , ( ) => {
0 commit comments