diff --git a/evalboard/app/_overview/__tests__/efficiency-charts.test.tsx b/evalboard/app/_overview/__tests__/efficiency-charts.test.tsx index 0e33607a7..14276709c 100644 --- a/evalboard/app/_overview/__tests__/efficiency-charts.test.tsx +++ b/evalboard/app/_overview/__tests__/efficiency-charts.test.tsx @@ -32,7 +32,7 @@ describe("EfficiencyCharts", () => { test("opens on the wall-clock metric", () => { renderCharts(); expect(screen.getByRole("heading")).toHaveTextContent( - "Time per Passed Task", + "Agent Time per Passed Task", ); expect( screen.getByRole("tab", { name: "Time" }), diff --git a/evalboard/app/_overview/efficiency-charts.tsx b/evalboard/app/_overview/efficiency-charts.tsx index 002e74b9c..095fbc2c9 100644 --- a/evalboard/app/_overview/efficiency-charts.tsx +++ b/evalboard/app/_overview/efficiency-charts.tsx @@ -40,10 +40,10 @@ const TABS: Array<{ { key: "time", label: "Time", - heading: "Time per Passed Task", - title: "Total wall clock ÷ tasks passed. Every task's seconds count, failures included; only passes count in the denominator, so a run that fails more reads slower.", + heading: "Agent Time per Passed Task", + title: "Agent wall clock ÷ tasks passed. Counts the agent's turns only: sandbox setup, pre_run, grading and cleanup are excluded. Every task's agent seconds count, failures included; only passes count in the denominator, so a run that fails more reads slower.", blurb: (scoped) => - "Seconds of every task that ran ÷ the number that passed · hover a point for the share within 2× expected" + + "Agent seconds of every task that ran (setup and grading excluded) ÷ the number that passed · hover a point for the share within 2× expected" + (scoped ? " · scoped to the active filter" : ""), render: (props) => , }, diff --git a/evalboard/app/_overview/window-summary.tsx b/evalboard/app/_overview/window-summary.tsx index e67ace75b..b95c95c1e 100644 --- a/evalboard/app/_overview/window-summary.tsx +++ b/evalboard/app/_overview/window-summary.tsx @@ -36,7 +36,7 @@ function Tile({ // Front-page window rollup: total spend + shape of the runs in scope. Every // tile is summed over the same set — `runCount` and `totals` both come out of // getOverview's single pass, so the Runs tile can never disagree with the -// Cost/Tasks/Pass/Compute tiles or with the charts. The totals are scoped to +// Cost/Tasks/Pass/Agent time tiles or with the charts. The totals are scoped to // matching tasks whenever a filter is active. // // This describes the window the charts plot, NOT however far the run table below @@ -91,9 +91,9 @@ export function WindowSummary({ valueClass={passClass(pct)} /> diff --git a/evalboard/app/runs/[id]/[...task]/__tests__/message-timeline.test.tsx b/evalboard/app/runs/[id]/[...task]/__tests__/message-timeline.test.tsx index e8a3b7095..ebbfaf4da 100644 --- a/evalboard/app/runs/[id]/[...task]/__tests__/message-timeline.test.tsx +++ b/evalboard/app/runs/[id]/[...task]/__tests__/message-timeline.test.tsx @@ -544,9 +544,8 @@ describe("MessageTimelineSection — Unaccounted cell", () => { expect(cell("Unaccounted").textContent).toBe("5.0s (50%)"); }); - test("the four pre-existing cells still render their values", () => { + test("the pre-existing cells still render their values", () => { renderStrip(10); - expect(cell("Messages").textContent).toBe("1"); expect(cell("Generation").textContent).toBe("4.0s"); expect(cell("Tool exec").textContent).toBe("1.0s"); expect(cell("Slow events").textContent).toBe("0 gen · 0 tool"); @@ -753,6 +752,72 @@ describe("MessageTimelineSection — Unaccounted cell", () => { }); }); +describe("MessageTimelineSection — agent time vs eval overhead", () => { + function cell(label: string): HTMLElement { + const parent = screen.getByText(label).parentElement as HTMLElement; + return parent.children[1] as HTMLElement; + } + + const message = () => makeMessage({ generationMs: 4000, textMs: 4000 }); + + function renderSplit() { + return render( + , + ); + } + + test("the task total splits into agent time and eval overhead", () => { + renderSplit(); + expect(cell("Task total").textContent).toBe("10.0s"); + expect(cell("Agent time").textContent).toBe("6.0s (60%)"); + expect(cell("Eval overhead").textContent).toBe("4.0s (40%)"); + }); + + test("each group's cells sum to its header", () => { + renderSplit(); + // 6s agent = 1s startup + 4s generation + 0.5s teardown + 0.5s unaccounted. + expect(cell("Unaccounted").textContent).toBe("500ms (8%)"); + // 4s overhead = 2.5s setup + 1s grading + 0.5s other. + expect(cell("Other").textContent).toBe("500ms"); + }); + + test("a residual that rounds to 0% of agent time is hidden", () => { + render( + , + ); + expect(screen.queryByText("Unaccounted")).toBeNull(); + }); + + test("without per-turn durations the residual spans the whole task", () => { + render( + , + ); + expect(cell("Agent time").textContent).toBe("—"); + expect(cell("Eval overhead").textContent).toBe("—"); + expect(cell("Other").textContent).toBe("—"); + expect(cell("Unaccounted").textContent).toBe("2.5s (25%)"); + }); +}); + describe("MessageTimelineSection — a row's EXEC cell", () => { function span(start: number, end: number) { return { diff --git a/evalboard/app/runs/[id]/[...task]/_sections.tsx b/evalboard/app/runs/[id]/[...task]/_sections.tsx index 2cced0936..83fbc12d3 100644 --- a/evalboard/app/runs/[id]/[...task]/_sections.tsx +++ b/evalboard/app/runs/[id]/[...task]/_sections.tsx @@ -267,6 +267,114 @@ function fmtMs(ms: number | null): string { return `${sign}${Math.round(abs)}ms`; } +const SWATCH = { + startup: "bg-indigo-300", + generation: "bg-studio-blue", + tool: "bg-sky-300", + teardown: "bg-indigo-200", + unaccounted: "bg-amber-300", + setup: "bg-gray-400", + grading: "bg-gray-300", + other: "bg-gray-200", +}; + +function StripCell({ + label, + title, + swatch, + strong = false, + valueClass, + children, +}: { + label: string; + title?: string; + swatch?: string; + strong?: boolean; + valueClass?: string; + children: ReactNode; +}) { + return ( +
+
+ {swatch && ( +
+
+ {children} +
+
+ ); +} + +function SplitCell({ + label, + ms, + totalMs, + valueClass = "text-gray-800 font-medium", +}: { + label: string; + ms: number; + totalMs: number; + valueClass?: string; +}) { + return ( +
+
{label}
+
+ {fmtMs(ms)} + {totalMs > 0 && } +
+
+ ); +} + +// Widths are flex-grow weights, so negative residuals (overlap) and unmeasured +// buckets simply drop out instead of skewing the rest. +function TimeBar({ + groups, +}: { + groups: { label: string; ms: number | null; swatch: string }[][]; +}) { + const shown = groups + .map((g) => g.filter((s): s is { label: string; ms: number; swatch: string } => s.ms != null && s.ms > 0)) + .filter((g) => g.length > 0); + if (shown.length === 0) return null; + return ( +