= 0.25
- ? "text-red-700 font-medium"
- : "text-gray-900 font-medium"
- }
+ {/* Task total splits into Agent time and Eval overhead, each
+ group's items sum to its header, and the Generation split sums
+ to Generation. */}
+
+
+
+ {fmtMs(taskMs)}
+
+ 0
+ ? "text-red-700 font-medium"
+ : "text-gray-900 font-medium"
+ }
+ >
+ {slowGen} gen · {slowTool} tool
+
+
+ {taskMs != null && taskMs > 0 && (
+
+ )}
+
+
+
- {fmtMs(unaccountedMs)}
- {unaccountedShare != null && (
-
- {" "}
- ({Math.round(unaccountedShare * 100)}%)
-
+ {fmtMs(agentMs)}
+
+
+
+
+ {fmtMs(harnessStartupMs ?? null)}
+
+
+ {fmtMs(totalGenMs)}
+
+
+ {fmtMs(toolExecMs ?? null)}
+
+
+ {fmtMs(harnessTeardownMs ?? null)}
+
+ {showUnaccounted && (
+ = 0.25
+ ? "text-red-700 font-medium"
+ : "text-gray-900 font-medium"
+ }
+ >
+ {fmtMs(unaccountedMs)}
+
+
)}
-
-
-
- Slow events
-
0
- ? "text-red-700 font-medium"
- : "text-gray-900 font-medium"
- }
+ className="flex flex-wrap items-baseline gap-x-3 gap-y-1 border-l-2 border-gray-200 pl-2 text-[10px]"
+ title="how the Generation figure above divides across block kinds. Within an emission that mixed kinds the split is apportioned by content size — an estimate for those emissions, not a measurement. 'unsplit' is time in an emission with no apportionable content at all, so it belongs to no kind."
>
- {slowGen} gen · {slowTool} tool
-
-
-
-
-
- Generation split
-
-
0 ? "grid-cols-4" : "grid-cols-3")
- }
- >
-
-
thinking
-
Generation split
+
= 0.4
- ? "text-red-700 font-medium tabular-nums"
- : "text-gray-800 font-medium tabular-nums"
+ ? "text-red-700 font-medium"
+ : "text-gray-800 font-medium"
}
- >
- {fmtMs(thinkingMs)}
- {totalGenMs > 0 && (
-
- {" "}
- ({Math.round((thinkingMs / totalGenMs) * 100)}%)
-
- )}
-
-
-
+ />
{/* "tool args", never "tool": this is time the model
- spent WRITING a tool call, and the Tool exec cell
- one row up is time the tool spent RUNNING. The
- bare word named both. */}
-
tool args
-
- {fmtMs(toolGenMs)}
- {totalGenMs > 0 && (
-
- {" "}
- ({Math.round((toolGenMs / totalGenMs) * 100)}%)
-
- )}
-
+ spent WRITING a tool call, and the Tool exec item
+ above is time the tool spent RUNNING. */}
+
+
+ {/* NOT "mixed": the legend above already uses MIXED
+ for a message carrying multiple block types. This
+ is the leftover no kind claimed. Backed by
+ MessageEvent.mixedGenMs. */}
+ {mixedMs > 0 && (
+
+ )}
-
-
text
-
- {fmtMs(textMs)}
- {totalGenMs > 0 && (
-
- {" "}
- ({Math.round((textMs / totalGenMs) * 100)}%)
-
- )}
-
+
+
+
+ {fmtMs(overheadMs)}
+
+
+
+
+ {fmtMs(setupMs ?? null)}
+
+
+ {fmtMs(gradingMs ?? null)}
+
+
+ {fmtMs(otherOverheadMs)}
+
- {mixedMs > 0 && (
-
- {/* NOT "mixed": the legend above already uses
- MIXED for a message carrying multiple block
- types, which is ~93% of Delegate's rows and
- the very case this cell is usually EMPTY
- for. This is the leftover no kind claimed.
- Backed by MessageEvent.mixedGenMs. */}
-
unsplit
-
- {fmtMs(mixedMs)}
- {totalGenMs > 0 && (
-
- {" "}
- ({Math.round((mixedMs / totalGenMs) * 100)}%)
-
- )}
-
-
- )}
@@ -858,6 +943,7 @@ export function CostExplorerSection({
tokens,
recordedCostUsd,
taskDurationSeconds,
+ agentSeconds,
harnessStartupMs,
harnessTeardownMs,
storedToolMs,
@@ -870,6 +956,7 @@ export function CostExplorerSection({
recordedCostUsd: number | null;
// Forwarded verbatim to the timeline's Unaccounted cell.
taskDurationSeconds?: number | null;
+ agentSeconds?: number | null;
// Forwarded verbatim to the timeline's Startup/Teardown cells.
harnessStartupMs?: number | null;
harnessTeardownMs?: number | null;
@@ -916,6 +1003,7 @@ export function CostExplorerSection({
subAgentUsageByToolId={subAgentUsageByToolId}
impactByIndex={impactByIndex}
taskDurationSeconds={taskDurationSeconds}
+ agentSeconds={agentSeconds}
harnessStartupMs={harnessStartupMs}
harnessTeardownMs={harnessTeardownMs}
storedToolMs={storedToolMs}
diff --git a/evalboard/app/runs/[id]/[...task]/page.tsx b/evalboard/app/runs/[id]/[...task]/page.tsx
index 760dcf53d..d2ea563b1 100644
--- a/evalboard/app/runs/[id]/[...task]/page.tsx
+++ b/evalboard/app/runs/[id]/[...task]/page.tsx
@@ -365,6 +365,7 @@ export default async function TaskPage({
tokens={task.tokens}
recordedCostUsd={task.totalCostUsd}
taskDurationSeconds={task.durationSeconds}
+ agentSeconds={task.agentSeconds}
harnessStartupMs={task.harnessStartupMs}
setupMs={task.setupMs}
gradingMs={task.gradingMs}
diff --git a/evalboard/app/runs/[id]/__tests__/run-view.render.test.tsx b/evalboard/app/runs/[id]/__tests__/run-view.render.test.tsx
index 4b4ed252e..dcd6add7e 100644
--- a/evalboard/app/runs/[id]/__tests__/run-view.render.test.tsx
+++ b/evalboard/app/runs/[id]/__tests__/run-view.render.test.tsx
@@ -1,15 +1,20 @@
-import { describe, expect, test, vi } from "vitest";
-import { render, screen } from "@testing-library/react";
+import { afterEach, describe, expect, test, vi } from "vitest";
+import { fireEvent, render, screen } from "@testing-library/react";
import type { TaskResultSummary } from "@/lib/runs";
// RunView reads the URL via next/navigation hooks; stub them so it renders in
// jsdom. The filter state we don't exercise here just resolves to "no filter".
+const nav = vi.hoisted(() => ({ search: new URLSearchParams() }));
vi.mock("next/navigation", () => ({
useRouter: () => ({ replace: vi.fn(), push: vi.fn() }),
usePathname: () => "/runs/r1",
- useSearchParams: () => new URLSearchParams(),
+ useSearchParams: () => nav.search,
}));
+afterEach(() => {
+ nav.search = new URLSearchParams();
+});
+
const { RunView } = await import("../run-view");
function row(
@@ -23,6 +28,7 @@ function row(
status: "SUCCESS",
weightedScore: 1.0,
durationSeconds: 1.0,
+ agentSeconds: 1.0,
totalCostUsd: 0.1,
actualCommands: null,
totalTurns: null,
@@ -90,10 +96,10 @@ describe("RunView — multi-variant runs replace the pooled pass rate", () => {
// 50%, which describes neither configuration; the whole point of the tile
// change is that this number no longer appears anywhere on the page.
const AB = [
- row("X", { variantId: "A", status: "SUCCESS", totalCostUsd: 0.1, durationSeconds: 1 }),
- row("Y", { variantId: "A", status: "SUCCESS", totalCostUsd: 0.1, durationSeconds: 1 }),
- row("X", { variantId: "B", status: "FAILURE", totalCostUsd: 0.3, durationSeconds: 5 }),
- row("Y", { variantId: "B", status: "FAILURE", totalCostUsd: 0.3, durationSeconds: 5 }),
+ row("X", { variantId: "A", status: "SUCCESS", totalCostUsd: 0.1, agentSeconds: 1 }),
+ row("Y", { variantId: "A", status: "SUCCESS", totalCostUsd: 0.1, agentSeconds: 1 }),
+ row("X", { variantId: "B", status: "FAILURE", totalCostUsd: 0.3, agentSeconds: 5 }),
+ row("Y", { variantId: "B", status: "FAILURE", totalCostUsd: 0.3, agentSeconds: 5 }),
];
test("each arm gets its own rate and the blended rate is gone", () => {
@@ -111,6 +117,8 @@ describe("RunView — multi-variant runs replace the pooled pass rate", () => {
// Pooled totals stay: a run's cost is real however many arms produced it.
expect(screen.getByText("$0.80")).toBeInTheDocument();
+ expect(screen.getByText("Agent time")).toBeInTheDocument();
+ // Agent seconds (1+1+5+5), not the rows' full durations (4 x 1s).
expect(screen.getByText("12s")).toBeInTheDocument();
// ...with the per-arm split replacing p50/p90, which would describe a
// pooled population that does not exist.
@@ -136,3 +144,62 @@ describe("RunView — multi-variant runs replace the pooled pass rate", () => {
expect(screen.getAllByText(/p50/).length).toBe(2);
});
});
+
+describe("RunView — variant filter", () => {
+ const AB = [
+ row("X", { variantId: "A", status: "SUCCESS" }),
+ row("Y", { variantId: "A", status: "SUCCESS" }),
+ row("X", { variantId: "B", status: "FAILURE" }),
+ row("Y", { variantId: "B", status: "FAILURE" }),
+ ];
+
+ test("offered on a multi-arm run only", () => {
+ const { unmount } = render(
+
,
+ );
+ const group = screen.getByRole("group", { name: "Filter by variant" });
+ expect(group).toHaveTextContent("allAB");
+ expect(screen.getByRole("button", { name: "all" })).toHaveAttribute(
+ "aria-pressed",
+ "true",
+ );
+ unmount();
+
+ render(
+
,
+ );
+ expect(screen.queryByRole("group", { name: "Filter by variant" })).toBeNull();
+ });
+
+ test("a selected arm scopes the tiles and the grid to its rows", () => {
+ nav.search = new URLSearchParams("variant=B");
+ render(
);
+
+ expect(screen.getByRole("button", { name: "B" })).toHaveAttribute(
+ "aria-pressed",
+ "true",
+ );
+ expect(screen.getByText("0%")).toBeInTheDocument();
+ expect(screen.queryByText("100%")).toBeNull();
+ expect(screen.queryByText(/arms · spread/)).toBeNull();
+ expect(screen.getByText("2 / 4")).toBeInTheDocument();
+ });
+
+ test("an unknown arm in the URL is ignored", () => {
+ nav.search = new URLSearchParams("variant=nope");
+ render(
);
+ expect(screen.getByText(/2 arms · spread 100 pts/)).toBeInTheDocument();
+ });
+
+ test("picking an arm writes it to the URL", () => {
+ const replace = vi.spyOn(window.history, "replaceState");
+ render(
);
+ fireEvent.click(screen.getByRole("button", { name: "B" }));
+ expect(replace).toHaveBeenLastCalledWith(null, "", "/runs/r1?variant=B");
+ replace.mockRestore();
+ });
+});
diff --git a/evalboard/app/runs/[id]/__tests__/run-view.test.ts b/evalboard/app/runs/[id]/__tests__/run-view.test.ts
index 3b1c45aec..7da6b8379 100644
--- a/evalboard/app/runs/[id]/__tests__/run-view.test.ts
+++ b/evalboard/app/runs/[id]/__tests__/run-view.test.ts
@@ -17,6 +17,7 @@ function row(
status: "SUCCESS",
weightedScore: 1.0,
durationSeconds: 1.0,
+ agentSeconds: 1.0,
totalCostUsd: 0.1,
actualCommands: null,
totalTurns: null,
@@ -36,14 +37,14 @@ function row(
}
describe("computeRunMetrics — mature-skipped exclusion", () => {
- test("mature rows count as passes but are excluded from cost/duration totals and percentiles", () => {
+ test("mature rows count as passes but are excluded from cost/agent-time totals and percentiles", () => {
const m = computeRunMetrics([
- row("ran-a", { totalCostUsd: 0.1, durationSeconds: 1.0 }),
- row("ran-b", { totalCostUsd: 0.3, durationSeconds: 3.0 }),
+ row("ran-a", { totalCostUsd: 0.1, agentSeconds: 1.0 }),
+ row("ran-b", { totalCostUsd: 0.3, agentSeconds: 3.0 }),
// Carried-forward mature row: 0 cost / 0 duration, never ran.
row("mature", {
totalCostUsd: 0,
- durationSeconds: 0,
+ agentSeconds: 0,
matureSkipped: true,
}),
]);
@@ -56,19 +57,19 @@ describe("computeRunMetrics — mature-skipped exclusion", () => {
// Totals reflect only the two tasks that actually ran (0.1 + 0.3, 1 + 3).
expect(m.cost).toBeCloseTo(0.4, 10);
- expect(m.duration).toBeCloseTo(4.0, 10);
+ expect(m.agentSeconds).toBeCloseTo(4.0, 10);
// Percentiles are taken over [0.1, 0.3] / [1, 3] only. If the mature 0s
// leaked in, the p50 would be dragged to 0.1 / 1.0 (median of three).
expect(m.costP50).toBeCloseTo(0.2, 10);
expect(m.costP90).toBeCloseTo(0.28, 10);
- expect(m.durationP50).toBeCloseTo(2.0, 10);
+ expect(m.agentSecondsP50).toBeCloseTo(2.0, 10);
});
test("an all-mature run reports passes but null cost/duration (renders as —)", () => {
const m = computeRunMetrics([
- row("m1", { totalCostUsd: 0, durationSeconds: 0, matureSkipped: true }),
- row("m2", { totalCostUsd: 0, durationSeconds: 0, matureSkipped: true }),
+ row("m1", { totalCostUsd: 0, agentSeconds: 0, matureSkipped: true }),
+ row("m2", { totalCostUsd: 0, agentSeconds: 0, matureSkipped: true }),
]);
expect(m.total).toBe(2);
@@ -77,8 +78,8 @@ describe("computeRunMetrics — mature-skipped exclusion", () => {
expect(m.cost).toBeNull();
expect(m.costP50).toBeNull();
expect(m.costP90).toBeNull();
- expect(m.duration).toBeNull();
- expect(m.durationP50).toBeNull();
+ expect(m.agentSeconds).toBeNull();
+ expect(m.agentSecondsP50).toBeNull();
});
});
@@ -153,13 +154,13 @@ describe("computeVariantMetrics", () => {
// finding, and pooling would erase it.
test("cost and duration are per arm, not pooled", () => {
const rows = computeVariantMetrics([
- row("A", { variantId: "a", totalCostUsd: 1, durationSeconds: 10 }),
- row("A", { variantId: "b", totalCostUsd: 3, durationSeconds: 30 }),
+ row("A", { variantId: "a", totalCostUsd: 1, agentSeconds: 10 }),
+ row("A", { variantId: "b", totalCostUsd: 3, agentSeconds: 30 }),
]);
expect(rows[0].metrics.cost).toBeCloseTo(1, 10);
expect(rows[1].metrics.cost).toBeCloseTo(3, 10);
- expect(rows[0].metrics.duration).toBeCloseTo(10, 10);
- expect(rows[1].metrics.duration).toBeCloseTo(30, 10);
+ expect(rows[0].metrics.agentSeconds).toBeCloseTo(10, 10);
+ expect(rows[1].metrics.agentSeconds).toBeCloseTo(30, 10);
});
// Rows with no variant_id belong to the default arm, so a run that mixes
diff --git a/evalboard/app/runs/[id]/run-view.tsx b/evalboard/app/runs/[id]/run-view.tsx
index 68398baf9..15b23b620 100644
--- a/evalboard/app/runs/[id]/run-view.tsx
+++ b/evalboard/app/runs/[id]/run-view.tsx
@@ -61,9 +61,9 @@ export interface RunMetrics {
cost: number | null;
costP50: number | null;
costP90: number | null;
- duration: number | null;
- durationP50: number | null;
- durationP90: number | null;
+ agentSeconds: number | null;
+ agentSecondsP50: number | null;
+ agentSecondsP90: number | null;
}
// Aggregate run-level metrics from a set of task rows. Status categorization is
@@ -82,9 +82,9 @@ export function computeRunMetrics(tasks: TaskResultSummary[]): RunMetrics {
let errored = 0;
let ungraded = 0;
let cost = 0;
- let durationSum = 0;
+ let agentSum = 0;
const costSamples: number[] = [];
- const durSamples: number[] = [];
+ const agentSamples: number[] = [];
for (const t of tasks) {
// A `switch` with an assertNever default, not an if/else chain with a
// catch-all `else failed++`. That chain is what made adding "ungraded"
@@ -115,9 +115,9 @@ export function computeRunMetrics(tasks: TaskResultSummary[]): RunMetrics {
cost += t.totalCostUsd;
costSamples.push(t.totalCostUsd);
}
- if (t.durationSeconds != null) {
- durationSum += t.durationSeconds;
- durSamples.push(t.durationSeconds);
+ if (t.agentSeconds != null) {
+ agentSum += t.agentSeconds;
+ agentSamples.push(t.agentSeconds);
}
}
const graded = total - ungraded;
@@ -147,9 +147,9 @@ export function computeRunMetrics(tasks: TaskResultSummary[]): RunMetrics {
cost: costSamples.length ? cost : null,
costP50: percentile(costSamples, 0.5),
costP90: percentile(costSamples, 0.9),
- duration: durSamples.length ? durationSum : null,
- durationP50: percentile(durSamples, 0.5),
- durationP90: percentile(durSamples, 0.9),
+ agentSeconds: agentSamples.length ? agentSum : null,
+ agentSecondsP50: percentile(agentSamples, 0.5),
+ agentSecondsP90: percentile(agentSamples, 0.9),
};
}
@@ -208,6 +208,46 @@ function Metric({
);
}
+function VariantFilter({
+ variants,
+ selected,
+ onSelect,
+}: {
+ variants: string[];
+ selected: string | null;
+ onSelect: (variant: string | null) => void;
+}) {
+ return (
+
+
+ Variant
+
+ {[null, ...variants].map((v) => {
+ const active = v === selected;
+ return (
+ onSelect(v)}
+ className={`font-mono text-xs px-2 py-0.5 rounded border transition-colors ${
+ active
+ ? "bg-studio-blue text-white border-studio-blue"
+ : "bg-white text-gray-600 border-gray-200 hover:bg-gray-50"
+ }`}
+ >
+ {v ?? "all"}
+
+ );
+ })}
+
+ );
+}
+
// Replaces the pooled number inside the Pass rate tile: a blended rate averages
// configurations that were deliberately made to differ, and moves when the arms
// are merely reordered. Spend and time keep their pooled totals instead, since a
@@ -329,6 +369,13 @@ export function RunView({
const q = searchParams.get("q") ?? "";
const [showAllTags, setShowAllTags] = useState(false);
+ const allVariants = useMemo(() => variantsOf(tasks), [tasks]);
+ const rawVariant = searchParams.get("variant");
+ const selectedVariant =
+ allVariants.length > 1 && rawVariant && allVariants.includes(rawVariant)
+ ? rawVariant
+ : null;
+
// Commit a filter change to the URL WITHOUT a server round-trip.
//
// Every reader of `tags` / `rtags` / `q` on this page is client-side (see
@@ -388,11 +435,22 @@ export function RunView({
[selectedReviewSet, selectedReviewTags, updateParam],
);
+ const selectVariant = useCallback(
+ (variant: string | null) => {
+ const params = new URLSearchParams(window.location.search);
+ if (variant == null) params.delete("variant");
+ else params.set("variant", variant);
+ setSearchParams(params);
+ },
+ [setSearchParams],
+ );
+
const clearAll = useCallback(() => {
const params = new URLSearchParams(window.location.search);
params.delete("q");
params.delete("tags");
params.delete("rtags");
+ params.delete("variant");
setSearchParams(params);
}, [setSearchParams]);
@@ -427,6 +485,11 @@ export function RunView({
const filtered = useMemo(() => {
let arr = tasks;
+ if (selectedVariant != null) {
+ arr = arr.filter(
+ (t) => (t.variantId ?? DEFAULT_VARIANT_ID) === selectedVariant,
+ );
+ }
if (selectedTags.length > 0) {
// Match on either real tags or the derived skill, so the same
// `tags` URL param works for both rails. Robust to new runs where
@@ -459,9 +522,17 @@ export function RunView({
});
}
return arr;
- }, [tasks, selectedTags, selectedReviewTags, reviewsByTask, qLower]);
+ }, [
+ tasks,
+ selectedVariant,
+ selectedTags,
+ selectedReviewTags,
+ reviewsByTask,
+ qLower,
+ ]);
const isFiltered =
+ selectedVariant != null ||
selectedTags.length > 0 ||
selectedReviewTags.length > 0 ||
qLower.length > 0;
@@ -676,23 +747,31 @@ export function RunView({
}
/>
- m.duration != null
- ? fmtDuration(m.duration)
+ m.agentSeconds != null
+ ? fmtDuration(m.agentSeconds)
: null,
)
- : metrics.durationP50 != null &&
- metrics.durationP90 != null
- ? `p50 ${fmtDuration(metrics.durationP50)} · p90 ${fmtDuration(metrics.durationP90)}`
+ : metrics.agentSecondsP50 != null &&
+ metrics.agentSecondsP90 != null
+ ? `p50 ${fmtDuration(metrics.agentSecondsP50)} · p90 ${fmtDuration(metrics.agentSecondsP90)}`
: undefined
}
/>
+ )}
+
{/* The colored skill/review/tag filter rail (+ its color legend)
is an internal-only surface — see lib/edition.ts. The public OSS
edition hides it; per-task chips on the grid below still render
diff --git a/evalboard/lib/__tests__/no-zero-coalesce.test.ts b/evalboard/lib/__tests__/no-zero-coalesce.test.ts
index b65fab91b..08d22f15e 100644
--- a/evalboard/lib/__tests__/no-zero-coalesce.test.ts
+++ b/evalboard/lib/__tests__/no-zero-coalesce.test.ts
@@ -85,24 +85,20 @@ const ALLOWED = new Map