From 4f8ab20ec89b0bd5b46129ff5b4d5ada1ec7c8e2 Mon Sep 17 00:00:00 2001 From: Bai Li Date: Thu, 24 Sep 2026 14:02:40 -0700 Subject: [PATCH 1/5] feat(evalboard): headline agent time per passed task, not full task wall clock The Time tab's headline divided each task's full duration by the passes, so sandbox setup, pre_run, grading and cleanup counted as agent speed. That share is ~4% of claude-code's wall clock and ~20% of codex's, so it skewed every cross-harness comparison and diluted slowdowns on short tasks. The headline now sums the agent's turns only; every past run.json already records them, so the whole history switches at once. The within-expected-time ratios are unchanged: they compare against the runner-stamped expected_seconds, which is still drawn from full durations. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../__tests__/efficiency-charts.test.tsx | 2 +- evalboard/app/_overview/efficiency-charts.tsx | 6 ++--- evalboard/lib/__tests__/overview.test.ts | 27 ++++++++++++------- evalboard/lib/__tests__/runs.test.ts | 19 +++++++++++++ evalboard/lib/overview.ts | 7 +++-- evalboard/lib/runs.ts | 12 +++++++++ 6 files changed, 58 insertions(+), 15 deletions(-) diff --git a/evalboard/app/_overview/__tests__/efficiency-charts.test.tsx b/evalboard/app/_overview/__tests__/efficiency-charts.test.tsx index 0e33607a..14276709 100644 --- a/evalboard/app/_overview/__tests__/efficiency-charts.test.tsx +++ b/evalboard/app/_overview/__tests__/efficiency-charts.test.tsx @@ -32,7 +32,7 @@ describe("EfficiencyCharts", () => { test("opens on the wall-clock metric", () => { renderCharts(); expect(screen.getByRole("heading")).toHaveTextContent( - "Time per Passed Task", + "Agent Time per Passed Task", ); expect( screen.getByRole("tab", { name: "Time" }), diff --git a/evalboard/app/_overview/efficiency-charts.tsx b/evalboard/app/_overview/efficiency-charts.tsx index 002e74b9..095fbc2c 100644 --- a/evalboard/app/_overview/efficiency-charts.tsx +++ b/evalboard/app/_overview/efficiency-charts.tsx @@ -40,10 +40,10 @@ const TABS: Array<{ { key: "time", label: "Time", - heading: "Time per Passed Task", - title: "Total wall clock ÷ tasks passed. Every task's seconds count, failures included; only passes count in the denominator, so a run that fails more reads slower.", + heading: "Agent Time per Passed Task", + title: "Agent wall clock ÷ tasks passed. Counts the agent's turns only: sandbox setup, pre_run, grading and cleanup are excluded. Every task's agent seconds count, failures included; only passes count in the denominator, so a run that fails more reads slower.", blurb: (scoped) => - "Seconds of every task that ran ÷ the number that passed · hover a point for the share within 2× expected" + + "Agent seconds of every task that ran (setup and grading excluded) ÷ the number that passed · hover a point for the share within 2× expected" + (scoped ? " · scoped to the active filter" : ""), render: (props) => , }, diff --git a/evalboard/lib/__tests__/overview.test.ts b/evalboard/lib/__tests__/overview.test.ts index 38b20899..72e13a66 100644 --- a/evalboard/lib/__tests__/overview.test.ts +++ b/evalboard/lib/__tests__/overview.test.ts @@ -304,12 +304,12 @@ describe("timePerPassedTaskForTasks", () => { // divided real seconds by tasks that never ran. A codex nightly whose // runner block said 3m12s rendered as 1m17s on the front page. const executed = [ - task({ durationSeconds: 100 }), - task({ durationSeconds: 300 }), + task({ agentSeconds: 100 }), + task({ agentSeconds: 300 }), ]; const carried = [ - task({ durationSeconds: 0, matureSkipped: true }), - task({ durationSeconds: 0, matureSkipped: true }), + task({ agentSeconds: 0, matureSkipped: true }), + task({ agentSeconds: 0, matureSkipped: true }), ]; expect(timePerPassedTaskForTasks([...executed, ...carried])).toBe( timePerPassedTaskForTasks(executed), @@ -332,8 +332,8 @@ describe("timePerPassedTaskForTasks", () => { test("divides all seconds that ran by the number that passed", () => { expect( timePerPassedTaskForTasks([ - task({ durationSeconds: 100 }), - task({ durationSeconds: 300 }), + task({ agentSeconds: 100 }), + task({ agentSeconds: 300 }), ]), ).toBe(200); }); @@ -343,8 +343,8 @@ describe("timePerPassedTaskForTasks", () => { // says so: 400 seconds over the single pass. expect( timePerPassedTaskForTasks([ - task({ durationSeconds: 100 }), - task({ status: "FAILURE", durationSeconds: 300 }), + task({ agentSeconds: 100 }), + task({ status: "FAILURE", agentSeconds: 300 }), ]), ).toBe(400); }); @@ -352,7 +352,7 @@ describe("timePerPassedTaskForTasks", () => { test("null when nothing passed", () => { expect( timePerPassedTaskForTasks([ - task({ status: "FAILURE", durationSeconds: 300 }), + task({ status: "FAILURE", agentSeconds: 300 }), ]), ).toBeNull(); }); @@ -360,6 +360,15 @@ describe("timePerPassedTaskForTasks", () => { test("null when no duration was recorded", () => { expect(timePerPassedTaskForTasks([task({})])).toBeNull(); }); + + test("counts agent seconds, not the row's full duration", () => { + expect( + timePerPassedTaskForTasks([ + task({ durationSeconds: 130, agentSeconds: 100 }), + task({ durationSeconds: 330, agentSeconds: 300 }), + ]), + ).toBe(200); + }); }); describe("collectPipelineRuns", () => { diff --git a/evalboard/lib/__tests__/runs.test.ts b/evalboard/lib/__tests__/runs.test.ts index d7736f8d..2c1b90e2 100644 --- a/evalboard/lib/__tests__/runs.test.ts +++ b/evalboard/lib/__tests__/runs.test.ts @@ -11,6 +11,7 @@ import { test, } from "vitest"; import { + agentSecondsFromRaw, aggregateSubAgentUsage, type ArtifactRef, clearRunCacheDir, @@ -500,6 +501,24 @@ describe("visibleTurnsFromRaw (historical backfill)", () => { }); }); +describe("agentSecondsFromRaw", () => { + test("sums every turn, leaving out setup and grading", () => { + expect( + agentSecondsFromRaw({ + duration: 130, + iterations: [{ duration_seconds: 40 }, { duration_seconds: 60 }], + }), + ).toBe(100); + }); + + test("null when no turn recorded a duration", () => { + expect(agentSecondsFromRaw({ duration: 24 })).toBeNull(); + expect( + agentSecondsFromRaw({ iterations: [{ duration_seconds: null }] }), + ).toBeNull(); + }); +}); + describe("extractComponentShas", () => { const baseEnv = { git_commit: "a1b2c3d4e5f6a1b2c3d4e5f6a1b2c3d4e5f6a1b2", diff --git a/evalboard/lib/overview.ts b/evalboard/lib/overview.ts index bd32866c..968b68c5 100644 --- a/evalboard/lib/overview.ts +++ b/evalboard/lib/overview.ts @@ -131,12 +131,15 @@ export function withinExpectedTimeRateForTasks( return eligible > 0 ? (within / eligible) * 100 : null; } -// Seconds of every task that ran, over the number that passed. Mirrors the +// Agent seconds of every task that ran, over the number that passed. Mirrors the // runner's headline (timing.py::time_per_passed_task) so a filtered front-page // view and the block stamped into run.json compute the same thing on the same // rows. The Slack rollup does not report this yet — the metric is being watched // on the dashboard first. // +// Agent time, not the row's `duration`: setup and grading are eval scaffolding +// no skill can change, and their share differs by harness. +// // Mature-skipped rows leave BOTH sides. They are carried-forward passes with no // duration, so counting them only in the denominator divides real seconds by a // task count that never ran: on a codex nightly that reported 3m12s, including @@ -147,7 +150,7 @@ export function timePerPassedTaskForTasks( const executed = tasks.filter((t) => !t.matureSkipped); const passed = executed.filter((t) => isPassStatus(t.status)).length; if (!passed) return null; - const total = executed.reduce((a, t) => a + (t.durationSeconds ?? 0), 0); + const total = executed.reduce((a, t) => a + (t.agentSeconds ?? 0), 0); return total > 0 ? total / passed : null; } diff --git a/evalboard/lib/runs.ts b/evalboard/lib/runs.ts index 47b06470..23e679d6 100644 --- a/evalboard/lib/runs.ts +++ b/evalboard/lib/runs.ts @@ -503,6 +503,7 @@ export interface RawTaskResult { status?: string; weighted_score?: number; duration?: number; + iterations?: { duration_seconds?: number | null }[] | null; total_cost_usd?: number; input_tokens?: number | null; output_tokens?: number | null; @@ -1141,6 +1142,9 @@ export interface RunOverviewTask { skill: string | null; totalCostUsd: number | null; durationSeconds: number | null; + // The agent's turns alone: `durationSeconds` minus sandbox setup, pre_run, + // grading and cleanup. Optional so test factories that predate it stay valid. + agentSeconds?: number | null; weightedScore: number | null; actualCommands: number | null; totalTurns: number | null; @@ -1257,6 +1261,13 @@ function mostCommonAgentType(rows: RawTaskResult[]): string | null { return best; } +export function agentSecondsFromRaw(t: RawTaskResult): number | null { + const seconds = (t.iterations ?? []) + .map((i) => i.duration_seconds) + .filter((d): d is number => typeof d === "number" && d > 0); + return seconds.length ? seconds.reduce((a, d) => a + d, 0) : null; +} + export async function readRunOverview( id: string, source: Source = DEFAULT_SOURCE, @@ -1275,6 +1286,7 @@ export async function readRunOverview( skill: deriveSkill(t.task_path, tags), totalCostUsd: t.total_cost_usd ?? null, durationSeconds: t.duration ?? null, + agentSeconds: agentSecondsFromRaw(t), weightedScore: t.weighted_score ?? null, actualCommands: t.actual_commands ?? null, totalTurns: t.total_turns ?? null, From b5750a89537144114b96f3a8b37a7d49316cd473 Mon Sep 17 00:00:00 2001 From: Bai Li Date: Thu, 24 Sep 2026 14:20:01 -0700 Subject: [PATCH 2/5] feat(evalboard): show agent time on the run page and filter it by variant The run page's Time card summed each task's full duration, so on a two-arm run it compared the arms' setup and grading along with the agents: the 2026-09-24 flow v1 vs v2 run read v2 as ~49 minutes faster where the agent gap is ~9. The card is now "Agent time", from the same per-turn seconds as the overview headline. Multi-arm runs also get a Variant selector (all, or one arm) that scopes the cards and the task grid to that arm, kept in the URL as ?variant=. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../[id]/__tests__/run-view.render.test.tsx | 81 ++++++++++-- .../app/runs/[id]/__tests__/run-view.test.ts | 29 ++--- evalboard/app/runs/[id]/run-view.tsx | 117 +++++++++++++++--- evalboard/lib/runs.ts | 4 + 4 files changed, 191 insertions(+), 40 deletions(-) diff --git a/evalboard/app/runs/[id]/__tests__/run-view.render.test.tsx b/evalboard/app/runs/[id]/__tests__/run-view.render.test.tsx index 4b4ed252..dcd6add7 100644 --- a/evalboard/app/runs/[id]/__tests__/run-view.render.test.tsx +++ b/evalboard/app/runs/[id]/__tests__/run-view.render.test.tsx @@ -1,15 +1,20 @@ -import { describe, expect, test, vi } from "vitest"; -import { render, screen } from "@testing-library/react"; +import { afterEach, describe, expect, test, vi } from "vitest"; +import { fireEvent, render, screen } from "@testing-library/react"; import type { TaskResultSummary } from "@/lib/runs"; // RunView reads the URL via next/navigation hooks; stub them so it renders in // jsdom. The filter state we don't exercise here just resolves to "no filter". +const nav = vi.hoisted(() => ({ search: new URLSearchParams() })); vi.mock("next/navigation", () => ({ useRouter: () => ({ replace: vi.fn(), push: vi.fn() }), usePathname: () => "/runs/r1", - useSearchParams: () => new URLSearchParams(), + useSearchParams: () => nav.search, })); +afterEach(() => { + nav.search = new URLSearchParams(); +}); + const { RunView } = await import("../run-view"); function row( @@ -23,6 +28,7 @@ function row( status: "SUCCESS", weightedScore: 1.0, durationSeconds: 1.0, + agentSeconds: 1.0, totalCostUsd: 0.1, actualCommands: null, totalTurns: null, @@ -90,10 +96,10 @@ describe("RunView — multi-variant runs replace the pooled pass rate", () => { // 50%, which describes neither configuration; the whole point of the tile // change is that this number no longer appears anywhere on the page. const AB = [ - row("X", { variantId: "A", status: "SUCCESS", totalCostUsd: 0.1, durationSeconds: 1 }), - row("Y", { variantId: "A", status: "SUCCESS", totalCostUsd: 0.1, durationSeconds: 1 }), - row("X", { variantId: "B", status: "FAILURE", totalCostUsd: 0.3, durationSeconds: 5 }), - row("Y", { variantId: "B", status: "FAILURE", totalCostUsd: 0.3, durationSeconds: 5 }), + row("X", { variantId: "A", status: "SUCCESS", totalCostUsd: 0.1, agentSeconds: 1 }), + row("Y", { variantId: "A", status: "SUCCESS", totalCostUsd: 0.1, agentSeconds: 1 }), + row("X", { variantId: "B", status: "FAILURE", totalCostUsd: 0.3, agentSeconds: 5 }), + row("Y", { variantId: "B", status: "FAILURE", totalCostUsd: 0.3, agentSeconds: 5 }), ]; test("each arm gets its own rate and the blended rate is gone", () => { @@ -111,6 +117,8 @@ describe("RunView — multi-variant runs replace the pooled pass rate", () => { // Pooled totals stay: a run's cost is real however many arms produced it. expect(screen.getByText("$0.80")).toBeInTheDocument(); + expect(screen.getByText("Agent time")).toBeInTheDocument(); + // Agent seconds (1+1+5+5), not the rows' full durations (4 x 1s). expect(screen.getByText("12s")).toBeInTheDocument(); // ...with the per-arm split replacing p50/p90, which would describe a // pooled population that does not exist. @@ -136,3 +144,62 @@ describe("RunView — multi-variant runs replace the pooled pass rate", () => { expect(screen.getAllByText(/p50/).length).toBe(2); }); }); + +describe("RunView — variant filter", () => { + const AB = [ + row("X", { variantId: "A", status: "SUCCESS" }), + row("Y", { variantId: "A", status: "SUCCESS" }), + row("X", { variantId: "B", status: "FAILURE" }), + row("Y", { variantId: "B", status: "FAILURE" }), + ]; + + test("offered on a multi-arm run only", () => { + const { unmount } = render( + , + ); + const group = screen.getByRole("group", { name: "Filter by variant" }); + expect(group).toHaveTextContent("allAB"); + expect(screen.getByRole("button", { name: "all" })).toHaveAttribute( + "aria-pressed", + "true", + ); + unmount(); + + render( + , + ); + expect(screen.queryByRole("group", { name: "Filter by variant" })).toBeNull(); + }); + + test("a selected arm scopes the tiles and the grid to its rows", () => { + nav.search = new URLSearchParams("variant=B"); + render(); + + expect(screen.getByRole("button", { name: "B" })).toHaveAttribute( + "aria-pressed", + "true", + ); + expect(screen.getByText("0%")).toBeInTheDocument(); + expect(screen.queryByText("100%")).toBeNull(); + expect(screen.queryByText(/arms · spread/)).toBeNull(); + expect(screen.getByText("2 / 4")).toBeInTheDocument(); + }); + + test("an unknown arm in the URL is ignored", () => { + nav.search = new URLSearchParams("variant=nope"); + render(); + expect(screen.getByText(/2 arms · spread 100 pts/)).toBeInTheDocument(); + }); + + test("picking an arm writes it to the URL", () => { + const replace = vi.spyOn(window.history, "replaceState"); + render(); + fireEvent.click(screen.getByRole("button", { name: "B" })); + expect(replace).toHaveBeenLastCalledWith(null, "", "/runs/r1?variant=B"); + replace.mockRestore(); + }); +}); diff --git a/evalboard/app/runs/[id]/__tests__/run-view.test.ts b/evalboard/app/runs/[id]/__tests__/run-view.test.ts index 3b1c45ae..7da6b837 100644 --- a/evalboard/app/runs/[id]/__tests__/run-view.test.ts +++ b/evalboard/app/runs/[id]/__tests__/run-view.test.ts @@ -17,6 +17,7 @@ function row( status: "SUCCESS", weightedScore: 1.0, durationSeconds: 1.0, + agentSeconds: 1.0, totalCostUsd: 0.1, actualCommands: null, totalTurns: null, @@ -36,14 +37,14 @@ function row( } describe("computeRunMetrics — mature-skipped exclusion", () => { - test("mature rows count as passes but are excluded from cost/duration totals and percentiles", () => { + test("mature rows count as passes but are excluded from cost/agent-time totals and percentiles", () => { const m = computeRunMetrics([ - row("ran-a", { totalCostUsd: 0.1, durationSeconds: 1.0 }), - row("ran-b", { totalCostUsd: 0.3, durationSeconds: 3.0 }), + row("ran-a", { totalCostUsd: 0.1, agentSeconds: 1.0 }), + row("ran-b", { totalCostUsd: 0.3, agentSeconds: 3.0 }), // Carried-forward mature row: 0 cost / 0 duration, never ran. row("mature", { totalCostUsd: 0, - durationSeconds: 0, + agentSeconds: 0, matureSkipped: true, }), ]); @@ -56,19 +57,19 @@ describe("computeRunMetrics — mature-skipped exclusion", () => { // Totals reflect only the two tasks that actually ran (0.1 + 0.3, 1 + 3). expect(m.cost).toBeCloseTo(0.4, 10); - expect(m.duration).toBeCloseTo(4.0, 10); + expect(m.agentSeconds).toBeCloseTo(4.0, 10); // Percentiles are taken over [0.1, 0.3] / [1, 3] only. If the mature 0s // leaked in, the p50 would be dragged to 0.1 / 1.0 (median of three). expect(m.costP50).toBeCloseTo(0.2, 10); expect(m.costP90).toBeCloseTo(0.28, 10); - expect(m.durationP50).toBeCloseTo(2.0, 10); + expect(m.agentSecondsP50).toBeCloseTo(2.0, 10); }); test("an all-mature run reports passes but null cost/duration (renders as —)", () => { const m = computeRunMetrics([ - row("m1", { totalCostUsd: 0, durationSeconds: 0, matureSkipped: true }), - row("m2", { totalCostUsd: 0, durationSeconds: 0, matureSkipped: true }), + row("m1", { totalCostUsd: 0, agentSeconds: 0, matureSkipped: true }), + row("m2", { totalCostUsd: 0, agentSeconds: 0, matureSkipped: true }), ]); expect(m.total).toBe(2); @@ -77,8 +78,8 @@ describe("computeRunMetrics — mature-skipped exclusion", () => { expect(m.cost).toBeNull(); expect(m.costP50).toBeNull(); expect(m.costP90).toBeNull(); - expect(m.duration).toBeNull(); - expect(m.durationP50).toBeNull(); + expect(m.agentSeconds).toBeNull(); + expect(m.agentSecondsP50).toBeNull(); }); }); @@ -153,13 +154,13 @@ describe("computeVariantMetrics", () => { // finding, and pooling would erase it. test("cost and duration are per arm, not pooled", () => { const rows = computeVariantMetrics([ - row("A", { variantId: "a", totalCostUsd: 1, durationSeconds: 10 }), - row("A", { variantId: "b", totalCostUsd: 3, durationSeconds: 30 }), + row("A", { variantId: "a", totalCostUsd: 1, agentSeconds: 10 }), + row("A", { variantId: "b", totalCostUsd: 3, agentSeconds: 30 }), ]); expect(rows[0].metrics.cost).toBeCloseTo(1, 10); expect(rows[1].metrics.cost).toBeCloseTo(3, 10); - expect(rows[0].metrics.duration).toBeCloseTo(10, 10); - expect(rows[1].metrics.duration).toBeCloseTo(30, 10); + expect(rows[0].metrics.agentSeconds).toBeCloseTo(10, 10); + expect(rows[1].metrics.agentSeconds).toBeCloseTo(30, 10); }); // Rows with no variant_id belong to the default arm, so a run that mixes diff --git a/evalboard/app/runs/[id]/run-view.tsx b/evalboard/app/runs/[id]/run-view.tsx index 68398baf..15b23b62 100644 --- a/evalboard/app/runs/[id]/run-view.tsx +++ b/evalboard/app/runs/[id]/run-view.tsx @@ -61,9 +61,9 @@ export interface RunMetrics { cost: number | null; costP50: number | null; costP90: number | null; - duration: number | null; - durationP50: number | null; - durationP90: number | null; + agentSeconds: number | null; + agentSecondsP50: number | null; + agentSecondsP90: number | null; } // Aggregate run-level metrics from a set of task rows. Status categorization is @@ -82,9 +82,9 @@ export function computeRunMetrics(tasks: TaskResultSummary[]): RunMetrics { let errored = 0; let ungraded = 0; let cost = 0; - let durationSum = 0; + let agentSum = 0; const costSamples: number[] = []; - const durSamples: number[] = []; + const agentSamples: number[] = []; for (const t of tasks) { // A `switch` with an assertNever default, not an if/else chain with a // catch-all `else failed++`. That chain is what made adding "ungraded" @@ -115,9 +115,9 @@ export function computeRunMetrics(tasks: TaskResultSummary[]): RunMetrics { cost += t.totalCostUsd; costSamples.push(t.totalCostUsd); } - if (t.durationSeconds != null) { - durationSum += t.durationSeconds; - durSamples.push(t.durationSeconds); + if (t.agentSeconds != null) { + agentSum += t.agentSeconds; + agentSamples.push(t.agentSeconds); } } const graded = total - ungraded; @@ -147,9 +147,9 @@ export function computeRunMetrics(tasks: TaskResultSummary[]): RunMetrics { cost: costSamples.length ? cost : null, costP50: percentile(costSamples, 0.5), costP90: percentile(costSamples, 0.9), - duration: durSamples.length ? durationSum : null, - durationP50: percentile(durSamples, 0.5), - durationP90: percentile(durSamples, 0.9), + agentSeconds: agentSamples.length ? agentSum : null, + agentSecondsP50: percentile(agentSamples, 0.5), + agentSecondsP90: percentile(agentSamples, 0.9), }; } @@ -208,6 +208,46 @@ function Metric({ ); } +function VariantFilter({ + variants, + selected, + onSelect, +}: { + variants: string[]; + selected: string | null; + onSelect: (variant: string | null) => void; +}) { + return ( +
+ + Variant + + {[null, ...variants].map((v) => { + const active = v === selected; + return ( + + ); + })} +
+ ); +} + // Replaces the pooled number inside the Pass rate tile: a blended rate averages // configurations that were deliberately made to differ, and moves when the arms // are merely reordered. Spend and time keep their pooled totals instead, since a @@ -329,6 +369,13 @@ export function RunView({ const q = searchParams.get("q") ?? ""; const [showAllTags, setShowAllTags] = useState(false); + const allVariants = useMemo(() => variantsOf(tasks), [tasks]); + const rawVariant = searchParams.get("variant"); + const selectedVariant = + allVariants.length > 1 && rawVariant && allVariants.includes(rawVariant) + ? rawVariant + : null; + // Commit a filter change to the URL WITHOUT a server round-trip. // // Every reader of `tags` / `rtags` / `q` on this page is client-side (see @@ -388,11 +435,22 @@ export function RunView({ [selectedReviewSet, selectedReviewTags, updateParam], ); + const selectVariant = useCallback( + (variant: string | null) => { + const params = new URLSearchParams(window.location.search); + if (variant == null) params.delete("variant"); + else params.set("variant", variant); + setSearchParams(params); + }, + [setSearchParams], + ); + const clearAll = useCallback(() => { const params = new URLSearchParams(window.location.search); params.delete("q"); params.delete("tags"); params.delete("rtags"); + params.delete("variant"); setSearchParams(params); }, [setSearchParams]); @@ -427,6 +485,11 @@ export function RunView({ const filtered = useMemo(() => { let arr = tasks; + if (selectedVariant != null) { + arr = arr.filter( + (t) => (t.variantId ?? DEFAULT_VARIANT_ID) === selectedVariant, + ); + } if (selectedTags.length > 0) { // Match on either real tags or the derived skill, so the same // `tags` URL param works for both rails. Robust to new runs where @@ -459,9 +522,17 @@ export function RunView({ }); } return arr; - }, [tasks, selectedTags, selectedReviewTags, reviewsByTask, qLower]); + }, [ + tasks, + selectedVariant, + selectedTags, + selectedReviewTags, + reviewsByTask, + qLower, + ]); const isFiltered = + selectedVariant != null || selectedTags.length > 0 || selectedReviewTags.length > 0 || qLower.length > 0; @@ -676,23 +747,31 @@ export function RunView({ } /> - m.duration != null - ? fmtDuration(m.duration) + m.agentSeconds != null + ? fmtDuration(m.agentSeconds) : null, ) - : metrics.durationP50 != null && - metrics.durationP90 != null - ? `p50 ${fmtDuration(metrics.durationP50)} · p90 ${fmtDuration(metrics.durationP90)}` + : metrics.agentSecondsP50 != null && + metrics.agentSecondsP90 != null + ? `p50 ${fmtDuration(metrics.agentSecondsP50)} · p90 ${fmtDuration(metrics.agentSecondsP90)}` : undefined } /> + {allVariants.length > 1 && ( + + )} + {/* The colored skill/review/tag filter rail (+ its color legend) is an internal-only surface — see lib/edition.ts. The public OSS edition hides it; per-task chips on the grid below still render diff --git a/evalboard/lib/runs.ts b/evalboard/lib/runs.ts index 23e679d6..f93f7aab 100644 --- a/evalboard/lib/runs.ts +++ b/evalboard/lib/runs.ts @@ -102,6 +102,9 @@ export interface TaskResultSummary { status: string | null; weightedScore: number | null; durationSeconds: number | null; + // The agent's turns alone, without setup and grading. Optional so test + // factories that predate it stay valid. + agentSeconds?: number | null; totalCostUsd: number | null; actualCommands: number | null; totalTurns: number | null; @@ -892,6 +895,7 @@ export function toTaskRow(t: RawTaskResult): TaskResultSummary { status: t.status ?? null, weightedScore: t.weighted_score ?? null, durationSeconds: t.duration ?? null, + agentSeconds: agentSecondsFromRaw(t), totalCostUsd: t.total_cost_usd ?? null, actualCommands: t.actual_commands ?? null, totalTurns: t.total_turns ?? null, From 5fa1260170dc395936626ceaa40e57a38b0e1c47 Mon Sep 17 00:00:00 2001 From: Bai Li Date: Thu, 24 Sep 2026 14:20:01 -0700 Subject: [PATCH 3/5] feat(evalboard): split the task timeline into agent time and eval overhead The task page's timing strip listed setup, the turn buckets, grading and one task-wide residual side by side. It now reads as a hierarchy: task total, then agent time (startup, generation, tool exec, teardown, unaccounted) and eval overhead (setup, grading, other), with each group's cells summing to its header. Runs without per-turn durations keep the single task-wide residual. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../__tests__/message-timeline.test.tsx | 53 ++++ .../app/runs/[id]/[...task]/_sections.tsx | 264 ++++++++++-------- evalboard/app/runs/[id]/[...task]/page.tsx | 1 + .../lib/__tests__/no-zero-coalesce.test.ts | 14 +- 4 files changed, 211 insertions(+), 121 deletions(-) diff --git a/evalboard/app/runs/[id]/[...task]/__tests__/message-timeline.test.tsx b/evalboard/app/runs/[id]/[...task]/__tests__/message-timeline.test.tsx index e8a3b709..099f1a2a 100644 --- a/evalboard/app/runs/[id]/[...task]/__tests__/message-timeline.test.tsx +++ b/evalboard/app/runs/[id]/[...task]/__tests__/message-timeline.test.tsx @@ -753,6 +753,59 @@ describe("MessageTimelineSection — Unaccounted cell", () => { }); }); +describe("MessageTimelineSection — agent time vs eval overhead", () => { + function cell(label: string): HTMLElement { + const parent = screen.getByText(label).parentElement as HTMLElement; + return parent.children[1] as HTMLElement; + } + + const message = () => makeMessage({ generationMs: 4000, textMs: 4000 }); + + function renderSplit() { + return render( + , + ); + } + + test("the task total splits into agent time and eval overhead", () => { + renderSplit(); + expect(cell("Task total").textContent).toBe("10.0s"); + expect(cell("Agent time").textContent).toBe("6.0s (60%)"); + expect(cell("Eval overhead").textContent).toBe("4.0s (40%)"); + }); + + test("each group's cells sum to its header", () => { + renderSplit(); + // 6s agent = 1s startup + 4s generation + 0.5s teardown + 0.5s unaccounted. + expect(cell("Unaccounted").textContent).toBe("500ms (8%)"); + // 4s overhead = 2.5s setup + 1s grading + 0.5s other. + expect(cell("Other").textContent).toBe("500ms"); + }); + + test("without per-turn durations the residual spans the whole task", () => { + render( + , + ); + expect(cell("Agent time").textContent).toBe("—"); + expect(cell("Eval overhead").textContent).toBe("—"); + expect(cell("Other").textContent).toBe("—"); + expect(cell("Unaccounted").textContent).toBe("2.5s (25%)"); + }); +}); + describe("MessageTimelineSection — a row's EXEC cell", () => { function span(start: number, end: number) { return { diff --git a/evalboard/app/runs/[id]/[...task]/_sections.tsx b/evalboard/app/runs/[id]/[...task]/_sections.tsx index 2cced093..6b0a7983 100644 --- a/evalboard/app/runs/[id]/[...task]/_sections.tsx +++ b/evalboard/app/runs/[id]/[...task]/_sections.tsx @@ -267,6 +267,37 @@ function fmtMs(ms: number | null): string { return `${sign}${Math.round(abs)}ms`; } +function StripCell({ + label, + title, + valueClass = "text-gray-900 font-medium", + children, +}: { + label: string; + title?: string; + valueClass?: string; + children: ReactNode; +}) { + return ( +
+
+ {label} +
+
{children}
+
+ ); +} + +function ShareOf({ share }: { share: number | null }) { + if (share == null) return null; + return ( + + {" "} + ({Math.round(share * 100)}%) + + ); +} + function fmtTokens(n: number | null): string { if (n == null) return "—"; if (n === 0) return "0"; @@ -318,6 +349,7 @@ export function MessageTimelineSection({ subAgentUsageByToolId = {}, impactByIndex, taskDurationSeconds, + agentSeconds, harnessStartupMs, harnessTeardownMs, storedToolMs, @@ -337,6 +369,9 @@ export function MessageTimelineSection({ // tool execution do NOT account for. Null/absent on a run predating // duration capture — the cell then renders "—" rather than a fake residual. taskDurationSeconds?: number | null; + // The agent's turns, summed. Absent on a run without per-turn durations, + // and the strip then falls back to one task-wide residual. + agentSeconds?: number | null; // The turn-level head and tail, summed over the task's turns: wall clock // before the first generation window opened and after the last one closed. // Turn-scoped, so they cannot be derived from the per-message stream the @@ -438,29 +473,35 @@ export function MessageTimelineSection({ const attributableGenMs = totalGenMs - mixedMs; const thinkingShare = attributableGenMs > 0 ? thinkingMs / attributableGenMs : 0; - // Wall clock the agent stream does not explain, AFTER every named bucket. - // Startup and teardown are subtracted because they are measured intervals, - // not residual — leaving them in reported a harness's CLI boot as - // unexplained time. `?? 0` subtracts only what was actually measured, so an - // older run with neither field keeps exactly its previous number. - // Negative means generation and tool execution overlapped, which is a real - // signal — never clamped. + // With per-turn durations the strip splits into Agent time and Eval + // overhead, and each group's cells sum to its header. Without them, one + // residual over the whole task, exactly as older runs published it. + // Negative residuals mean buckets overlapped and are never clamped. const taskMs = taskDurationSeconds != null ? taskDurationSeconds * 1000 : null; + const agentMs = agentSeconds != null ? agentSeconds * 1000 : null; + const overheadMs = + taskMs != null && agentMs != null ? taskMs - agentMs : null; + const agentBucketsMs = + totalGenMs + + (toolExecMs ?? 0) + + (harnessStartupMs ?? 0) + + (harnessTeardownMs ?? 0); + const phaseMs = (setupMs ?? 0) + (gradingMs ?? 0); const unaccountedMs = - taskMs != null - ? taskMs - - totalGenMs - - (toolExecMs ?? 0) - - (harnessStartupMs ?? 0) - - (harnessTeardownMs ?? 0) - - (setupMs ?? 0) - - (gradingMs ?? 0) - : null; + agentMs != null + ? agentMs - agentBucketsMs + : taskMs != null + ? taskMs - agentBucketsMs - phaseMs + : null; + const unaccountedBase = agentMs ?? taskMs; const unaccountedShare = - taskMs != null && taskMs > 0 && unaccountedMs != null - ? unaccountedMs / taskMs + unaccountedBase != null && unaccountedBase > 0 && unaccountedMs != null + ? unaccountedMs / unaccountedBase : null; + const otherOverheadMs = overheadMs != null ? overheadMs - phaseMs : null; + const shareOfTask = (ms: number | null) => + taskMs != null && taskMs > 0 && ms != null ? ms / taskMs : null; return (
@@ -473,107 +514,103 @@ export function MessageTimelineSection({

MIXED = multiple block types · red = slow (gen ≥10s, tool ≥5s)

- {/* TWO LEVELS, two rows. The top row's five time cells — Startup, - Generation, Tool exec, Teardown, Unaccounted — sum to the task's - wall clock; the bottom row splits Generation alone and sums to - IT. Rendering the split as a sub-cell of one top-row cell put - both sums on one line, where nothing said which total each part - belonged to. The time cells are ordered as the turn runs. */} + {/* Three levels: Task total splits into Agent time and Eval + overhead, each group's cells below sum to its header, and the + Generation split sums to Generation. */}
-
+
+ {messageCount} + + {fmtMs(taskMs)} + + + {fmtMs(agentMs)} + + + + {fmtMs(overheadMs)} + + + 0 + ? "text-red-700 font-medium" + : "text-gray-900 font-medium" + } + > + {slowGen} gen · {slowTool} tool + +
+
-
- Messages -
-
{messageCount}
-
-
-
- Setup -
-
- {fmtMs(setupMs ?? null)} -
-
-
-
- Startup -
-
- {fmtMs(harnessStartupMs ?? null)} +
+ Agent time breakdown
-
-
-
- Generation -
-
- {fmtMs(totalGenMs)} -
-
-
-
- Tool exec -
-
- {fmtMs(toolExecMs ?? null)} -
-
-
-
- Teardown -
-
- {fmtMs(harnessTeardownMs ?? null)} -
-
-
-
- Grading -
-
- {fmtMs(gradingMs ?? null)} -
-
-
-
- Unaccounted -
-
= 0.25 - ? "text-red-700 font-medium" - : "text-gray-900 font-medium" - } - > - {fmtMs(unaccountedMs)} - {unaccountedShare != null && ( - - {" "} - ({Math.round(unaccountedShare * 100)}%) - - )} +
+ + {fmtMs(harnessStartupMs ?? null)} + + + {fmtMs(totalGenMs)} + + + {fmtMs(toolExecMs ?? null)} + + + {fmtMs(harnessTeardownMs ?? null)} + + = 0.25 + ? "text-red-700 font-medium" + : "text-gray-900 font-medium" + } + > + {fmtMs(unaccountedMs)} + +
-
- Slow events +
+ Eval overhead breakdown
-
0 - ? "text-red-700 font-medium" - : "text-gray-900 font-medium" - } - > - {slowGen} gen · {slowTool} tool +
+ + {fmtMs(setupMs ?? null)} + + + {fmtMs(gradingMs ?? null)} + + + {fmtMs(otherOverheadMs)} +
@@ -858,6 +895,7 @@ export function CostExplorerSection({ tokens, recordedCostUsd, taskDurationSeconds, + agentSeconds, harnessStartupMs, harnessTeardownMs, storedToolMs, @@ -870,6 +908,7 @@ export function CostExplorerSection({ recordedCostUsd: number | null; // Forwarded verbatim to the timeline's Unaccounted cell. taskDurationSeconds?: number | null; + agentSeconds?: number | null; // Forwarded verbatim to the timeline's Startup/Teardown cells. harnessStartupMs?: number | null; harnessTeardownMs?: number | null; @@ -916,6 +955,7 @@ export function CostExplorerSection({ subAgentUsageByToolId={subAgentUsageByToolId} impactByIndex={impactByIndex} taskDurationSeconds={taskDurationSeconds} + agentSeconds={agentSeconds} harnessStartupMs={harnessStartupMs} harnessTeardownMs={harnessTeardownMs} storedToolMs={storedToolMs} diff --git a/evalboard/app/runs/[id]/[...task]/page.tsx b/evalboard/app/runs/[id]/[...task]/page.tsx index 760dcf53..d2ea563b 100644 --- a/evalboard/app/runs/[id]/[...task]/page.tsx +++ b/evalboard/app/runs/[id]/[...task]/page.tsx @@ -365,6 +365,7 @@ export default async function TaskPage({ tokens={task.tokens} recordedCostUsd={task.totalCostUsd} taskDurationSeconds={task.durationSeconds} + agentSeconds={task.agentSeconds} harnessStartupMs={task.harnessStartupMs} setupMs={task.setupMs} gradingMs={task.gradingMs} diff --git a/evalboard/lib/__tests__/no-zero-coalesce.test.ts b/evalboard/lib/__tests__/no-zero-coalesce.test.ts index b65fab91..08d22f15 100644 --- a/evalboard/lib/__tests__/no-zero-coalesce.test.ts +++ b/evalboard/lib/__tests__/no-zero-coalesce.test.ts @@ -85,24 +85,20 @@ const ALLOWED = new Map([ "A threshold comparison: an untimed call is not a slow call, so 0 answers the question asked.", ], [ - "(toolExecMs ?? 0) -", + "(toolExecMs ?? 0) +", "The residual's tool half, and the reason the cell itself now renders a dash: a turn with no BOUNDED span measured no tool time, so its time belongs IN the residual rather than being subtracted as a zero.", ], [ - "(harnessStartupMs ?? 0) -", + "(harnessStartupMs ?? 0) +", "The residual. Subtracting only what was measured is the whole point; an unmeasured head leaves its time IN the residual rather than silently claiming it.", ], [ - "(harnessTeardownMs ?? 0) -", + "(harnessTeardownMs ?? 0);", "The residual's tail half: subtracting only what was measured leaves unmeasured time IN the residual.", ], [ - "(setupMs ?? 0) -", - "Same residual rule: a run predating the field leaves its setup time IN the residual rather than having it silently subtracted as zero.", - ], - [ - "(gradingMs ?? 0)", - "Same residual rule, and it is also the ungraded case — `coder-eval execute` grades nothing, so there is no grading time to subtract.", + "const phaseMs = (setupMs ?? 0) + (gradingMs ?? 0);", + "Same residual rule: a run predating the fields leaves setup and grading time IN the residual, and `coder-eval execute` grades nothing, so there is no grading time to subtract.", ], [ "const slowExec = (execMs ?? 0) >= SLOW_TOOL_MS;", From 17d677a32cdbf2e8bc5d62a00353119806b4ebfc Mon Sep 17 00:00:00 2001 From: Bai Li Date: Thu, 24 Sep 2026 14:27:12 -0700 Subject: [PATCH 4/5] feat(evalboard): simplify the task timing strip into one bar and two groups Co-Authored-By: Claude Opus 5.5 (1M context) --- .../__tests__/message-timeline.test.tsx | 16 +- .../app/runs/[id]/[...task]/_sections.tsx | 324 ++++++++++-------- 2 files changed, 200 insertions(+), 140 deletions(-) diff --git a/evalboard/app/runs/[id]/[...task]/__tests__/message-timeline.test.tsx b/evalboard/app/runs/[id]/[...task]/__tests__/message-timeline.test.tsx index 099f1a2a..ebbfaf4d 100644 --- a/evalboard/app/runs/[id]/[...task]/__tests__/message-timeline.test.tsx +++ b/evalboard/app/runs/[id]/[...task]/__tests__/message-timeline.test.tsx @@ -544,9 +544,8 @@ describe("MessageTimelineSection — Unaccounted cell", () => { expect(cell("Unaccounted").textContent).toBe("5.0s (50%)"); }); - test("the four pre-existing cells still render their values", () => { + test("the pre-existing cells still render their values", () => { renderStrip(10); - expect(cell("Messages").textContent).toBe("1"); expect(cell("Generation").textContent).toBe("4.0s"); expect(cell("Tool exec").textContent).toBe("1.0s"); expect(cell("Slow events").textContent).toBe("0 gen · 0 tool"); @@ -790,6 +789,19 @@ describe("MessageTimelineSection — agent time vs eval overhead", () => { expect(cell("Other").textContent).toBe("500ms"); }); + test("a residual that rounds to 0% of agent time is hidden", () => { + render( + , + ); + expect(screen.queryByText("Unaccounted")).toBeNull(); + }); + test("without per-turn durations the residual spans the whole task", () => { render( -
+
+
+ {swatch && ( +
-
{children}
+
+ {children} +
+
+ ); +} + +function SplitCell({ + label, + ms, + totalMs, + valueClass = "text-gray-800 font-medium", +}: { + label: string; + ms: number; + totalMs: number; + valueClass?: string; +}) { + return ( +
+
{label}
+
+ {fmtMs(ms)} + {totalMs > 0 && } +
+
+ ); +} + +// Widths are flex-grow weights, so negative residuals (overlap) and unmeasured +// buckets simply drop out instead of skewing the rest. +function TimeBar({ + groups, +}: { + groups: { label: string; ms: number | null; swatch: string }[][]; +}) { + const shown = groups + .map((g) => g.filter((s): s is { label: string; ms: number; swatch: string } => s.ms != null && s.ms > 0)) + .filter((g) => g.length > 0); + if (shown.length === 0) return null; + return ( +
diff --git a/evalboard/lib/__tests__/overview.test.ts b/evalboard/lib/__tests__/overview.test.ts index 72e13a66..11381650 100644 --- a/evalboard/lib/__tests__/overview.test.ts +++ b/evalboard/lib/__tests__/overview.test.ts @@ -60,55 +60,55 @@ describe("summarizeListing", () => { tasksSucceeded: 0, tasksRun: 0, tasksGraded: 0, - durationSeconds: null, - durationPartial: false, + agentSeconds: null, + agentPartial: false, }); }); - test("sums cost, duration, and task counts across runs", () => { + test("sums cost, agent time, and task counts across runs", () => { const t = summarizeListing([ row({ tasksSucceeded: 8, tasksRun: 10, totalCostUsd: 1.5, - taskDurationSeconds: 120, + agentSeconds: 120, }), row({ tasksSucceeded: 3, tasksRun: 5, totalCostUsd: 2.25, - taskDurationSeconds: 60, + agentSeconds: 60, }), ]); expect(t.costUsd).toBeCloseTo(3.75); - expect(t.durationSeconds).toBe(180); + expect(t.agentSeconds).toBe(180); expect(t.tasksSucceeded).toBe(11); expect(t.tasksRun).toBe(15); expect(t.costPartial).toBe(false); - expect(t.durationPartial).toBe(false); + expect(t.agentPartial).toBe(false); }); - test("flags partial when a run lacks cost or duration", () => { - // One run recorded cost/duration, one didn't: sum reflects only the + test("flags partial when a run lacks cost or agent time", () => { + // One run recorded cost/agent time, one didn't: sum reflects only the // recorded run and the *Partial flags say so. const t = summarizeListing([ - row({ totalCostUsd: 4, taskDurationSeconds: 30 }), - row({ totalCostUsd: null, taskDurationSeconds: null }), + row({ totalCostUsd: 4, agentSeconds: 30 }), + row({ totalCostUsd: null, agentSeconds: null }), ]); expect(t.costUsd).toBe(4); expect(t.costPartial).toBe(true); - expect(t.durationSeconds).toBe(30); - expect(t.durationPartial).toBe(true); + expect(t.agentSeconds).toBe(30); + expect(t.agentPartial).toBe(true); }); test("all-missing cost stays null, not zero", () => { // A window where no run has a cost must read "—", not "$0.00" — the - // sum is unknown, not zero. Same for duration. + // sum is unknown, not zero. Same for agent time. const t = summarizeListing([row({}), row({})]); expect(t.costUsd).toBeNull(); expect(t.costPartial).toBe(false); - expect(t.durationSeconds).toBeNull(); - expect(t.durationPartial).toBe(false); + expect(t.agentSeconds).toBeNull(); + expect(t.agentPartial).toBe(false); }); }); @@ -1276,6 +1276,16 @@ describe("projectRunRow", () => { ); expect(scoped?.tasks).toHaveLength(2); }); + + test("sums agent time over the tasks that ran, whole or filtered", () => { + const r = run("r", [ + task({ taskId: "a", tags: ["keep"], agentSeconds: 30 }), + task({ taskId: "b", agentSeconds: 12 }), + task({ taskId: "c", tags: ["keep"], agentSeconds: 99, matureSkipped: true }), + ]); + expect(scopeRunTasks(r, null, null)?.agentSeconds).toBe(42); + expect(scopeRunTasks(r, "keep", null)?.agentSeconds).toBe(30); + }); }); }); diff --git a/evalboard/lib/overview.ts b/evalboard/lib/overview.ts index 968b68c5..4b925049 100644 --- a/evalboard/lib/overview.ts +++ b/evalboard/lib/overview.ts @@ -194,6 +194,8 @@ export interface RunListingRow { tasksExecuted: number; totalCostUsd: number | null; taskDurationSeconds: number | null; + // Optional so existing test factories stay valid. + agentSeconds?: number | null; // Run-level harness (coder-eval AgentKind) for the Harness column; null on // legacy runs that predate the RunConfig stamp / carry no agent_config.type. // Optional so existing test factories stay valid. @@ -211,8 +213,8 @@ export interface RunListingTotals { tasksSucceeded: number; tasksRun: number; tasksGraded: number; // the pass-rate denominator; see RunListingRow.tasksGraded - durationSeconds: number | null; // null when no matched run recorded a duration - durationPartial: boolean; + agentSeconds: number | null; // null when no matched run recorded agent time + agentPartial: boolean; } // Sum a matched-run slice into a window rollup. Pure over RunListingRow[] so it @@ -224,8 +226,8 @@ export function summarizeListing(rows: RunListingRow[]): RunListingTotals { let tasksSucceeded = 0; let tasksRun = 0; let tasksGraded = 0; - let durationSeconds = 0; - let durationRuns = 0; + let agentSeconds = 0; + let agentRuns = 0; for (const r of rows) { tasksSucceeded += r.tasksSucceeded; tasksRun += r.tasksRun; @@ -234,9 +236,9 @@ export function summarizeListing(rows: RunListingRow[]): RunListingTotals { costUsd += r.totalCostUsd; costRuns += 1; } - if (r.taskDurationSeconds != null) { - durationSeconds += r.taskDurationSeconds; - durationRuns += 1; + if (r.agentSeconds != null) { + agentSeconds += r.agentSeconds; + agentRuns += 1; } } return { @@ -245,8 +247,8 @@ export function summarizeListing(rows: RunListingRow[]): RunListingTotals { tasksSucceeded, tasksRun, tasksGraded, - durationSeconds: durationRuns > 0 ? durationSeconds : null, - durationPartial: durationRuns > 0 && durationRuns < rows.length, + agentSeconds: agentRuns > 0 ? agentSeconds : null, + agentPartial: agentRuns > 0 && agentRuns < rows.length, }; } @@ -1078,6 +1080,18 @@ export interface ScopedRun { // second count over `tasks` — which drops rows with no task_id — could // disagree with the duration's own denominator. tasksExecuted: number; + agentSeconds: number | null; +} + +function sumAgentSeconds(tasks: RunOverviewTask[]): number | null { + let sum = 0; + let any = false; + for (const t of tasks) { + if (t.matureSkipped || t.agentSeconds == null) continue; + sum += t.agentSeconds; + any = true; + } + return any ? sum : null; } export function scopeRunTasks( @@ -1092,6 +1106,7 @@ export function scopeRunTasks( taskDurationSeconds: overview.taskDurationSeconds, // ?? for an overview built before the field existed (test factories). tasksExecuted: overview.tasksExecuted ?? overview.tasks.length, + agentSeconds: sumAgentSeconds(overview.tasks), }; if (tag == null && needle == null) return wholeRun; @@ -1141,6 +1156,7 @@ export function scopeRunTasks( totalCostUsd: costHasAny ? costSum : null, taskDurationSeconds: durAllPresent ? durSum : null, tasksExecuted: matching.filter((t) => !t.matureSkipped).length, + agentSeconds: sumAgentSeconds(matching), }; } @@ -1158,6 +1174,7 @@ function rowFromScoped( tasksExecuted: scoped.tasksExecuted, totalCostUsd: scoped.totalCostUsd, taskDurationSeconds: scoped.taskDurationSeconds, + agentSeconds: scoped.agentSeconds, harness: harness ?? null, }; }