Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 23 additions & 1 deletion docs/REPORT_SCHEMA.md
Original file line number Diff line number Diff line change
Expand Up @@ -75,7 +75,10 @@ publishing different numbers for the same run.

Each entry is an **untyped dict** (a denormalization, not a Pydantic model) with keys
including: `task_id`, `replicate_index`, `variant_id`, `status`
([`FinalStatus`](#finalstatus)), `weighted_score`, `duration`, `iteration_count`,
([`FinalStatus`](#finalstatus)), `weighted_score`, `duration` (END-TO-END seconds:
setup + the agent's turns + grading), the timing split (`agent_wall_ms`, `setup_ms`,
`grading_ms` — see [Agent wall vs end-to-end](#agent-wall-vs-end-to-end)), the four
turn buckets (`startup_ms`, `generation_ms`, `tool_ms`, `teardown_ms`), `iteration_count`,
`tags`, `task_path`, `model_used`, `reference_similarity`, the token buckets
(`input_tokens` = uncached input, `output_tokens`, `cache_creation_input_tokens`,
`cache_read_input_tokens`, `total_tokens`), the cost fields
Expand All @@ -89,6 +92,25 @@ including: `task_id`, `replicate_index`, `variant_id`, `status`
turn digest (`{iteration, duration_seconds, command_count, assistant_turn_count,
crashed, crash_reason}`) — the full transcript is in `task.json`.

### Agent wall vs end-to-end

`duration` is the task's whole wall clock. It includes `setup_ms` (sandbox, agent
start, `pre_run`) and `grading_ms` (every success check). A checker that runs a live
command can spend tens of seconds there, so `duration` is not the agent's time.

| Key | Type | Meaning |
| --- | --- | --- |
| `agent_wall_ms` | `float \| None` | Sum of the positive `iterations[].duration_seconds`, in ms. The agent's own turns. `None` when no turn was timed. |
| `setup_ms` | `float \| None` | Copied from `task.json`. |
| `grading_ms` | `float \| None` | Copied from `task.json`. `None` on an ungraded row (`coder-eval execute`) and on a detached grade. |

Example (calculator, one run): `duration` 101.3 s = `agent_wall_ms` 55,657 +
`setup_ms` 13,678 + `grading_ms` 27,657 + about 4 s not in a named phase.

A `run.json` written before these keys existed has none of them. Readers derive
agent wall from the row's `iterations[].duration_seconds` with the same rule, and show
grading as unknown, never `0`.

### Missing cost is never fatal

Pricing degrades; the evaluation does not. A model absent from the rate card, a turn
Expand Down
60 changes: 53 additions & 7 deletions evalboard/app/runs/[id]/__tests__/task-grid.test.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -52,10 +52,12 @@ function cellFor(taskId: string, index: number): HTMLElement {
return cells[index]!;
}

// Layout: Task, Status, Score, Duration, vs Exp, Cost, Turns, then the tokens
// Layout: Task, Status, Score, End-to-end, Agent, Grading, vs Exp, Cost, Turns, then the tokens
const durationCellFor = (taskId: string) => cellFor(taskId, 3);
const vsExpCellFor = (taskId: string) => cellFor(taskId, 4);
const turnsCellFor = (taskId: string) => cellFor(taskId, 6);
const agentCellFor = (taskId: string) => cellFor(taskId, 4);
const gradingCellFor = (taskId: string) => cellFor(taskId, 5);
const vsExpCellFor = (taskId: string) => cellFor(taskId, 6);
const turnsCellFor = (taskId: string) => cellFor(taskId, 8);

describe("TaskGrid — mature rows", () => {
test("opens a popover linking to the run where it last executed", () => {
Expand Down Expand Up @@ -195,7 +197,7 @@ describe("TaskGrid — vs Expected column", () => {
.slice(1)
.map((tr) => within(tr).getAllByRole("cell")[0].textContent);

fireEvent.click(screen.getByRole("button", { name: /^Duration$/ }));
fireEvent.click(screen.getByRole("button", { name: /^End-to-end$/ }));
expect(order()[0]).toMatch(/long/i);

fireEvent.click(screen.getByRole("button", { name: /^vs Expected$/ }));
Expand Down Expand Up @@ -262,7 +264,9 @@ describe("TaskGrid — Turns column", () => {
"Task",
"Status",
"Score",
"Duration",
"End-to-end",
"Agent",
"Grading",
"vs Expected",
"Cost",
"Turns",
Expand All @@ -273,7 +277,9 @@ describe("TaskGrid — Turns column", () => {
"Task",
"Status",
"Score",
"Duration",
"End-to-end",
"Agent",
"Grading",
"vs Expected",
"Cost",
"Turns",
Expand All @@ -300,7 +306,15 @@ describe("TaskGrid — column tooltips", () => {
);
expect(header("vs Expected")).toHaveAttribute(
"title",
expect.stringContaining("Duration ÷"),
expect.stringContaining("End-to-end ÷"),
);
expect(header("End-to-end")).toHaveAttribute(
"title",
expect.stringContaining("+ grading"),
);
expect(header("Agent")).toHaveAttribute(
"title",
expect.stringContaining("Excludes sandbox setup"),
);
expect(header("Turns")).toHaveAttribute(
"title",
Expand Down Expand Up @@ -595,3 +609,35 @@ describe("TaskGrid — default ordering keeps a task's arms together", () => {
expect(order[2]).toMatch(/zzz/i);
});
});

describe("TaskGrid — end-to-end vs agent vs grading (#212)", () => {
test("splits a slow checker out of the agent's time", () => {
render(
<TaskGrid
sourceId="skills"
runId="r1"
tasks={[
row("calc", 3, null, {
durationSeconds: 101.25,
agentSeconds: 55.66,
gradingSeconds: 27.66,
}),
]}
/>,
);
expect(durationCellFor("calc")).toHaveTextContent("1m41s");
expect(agentCellFor("calc")).toHaveTextContent("55.7s");
expect(gradingCellFor("calc")).toHaveTextContent("27.7s");
});

test("an unmeasured grading time renders as a dash, never 0", () => {
render(
<TaskGrid
sourceId="skills"
runId="r1"
tasks={[row("old", 3, null, { durationSeconds: 10 })]}
/>,
);
expect(gradingCellFor("old")).toHaveTextContent("—");
});
});
40 changes: 36 additions & 4 deletions evalboard/app/runs/[id]/task-grid.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,8 @@ type SortKey =
| "status"
| "score"
| "duration"
| "agent"
| "grading"
| "vsExp"
| "cost"
| "turns"
Expand All @@ -54,7 +56,10 @@ type SortKey =
const COLUMN_HELP: Partial<Record<SortKey, string>> = {
...TOKEN_COLUMN_HELP,
turns: "Visible turns: one per tool call plus one for the final reply. Tinted against the task's hand-written expected_turns budget (yellow past 1.25×, red past 1.5×); untinted when the task declares none.",
vsExp: "Duration ÷ the time this task is expected to need. The expected time is derived per task, per harness by the eval runner (its fastest passing run, or p10 once there are ten) and stamped into the run — never hand-written. Past 2× counts as slow; a task its harness has never passed shows —.",
vsExp: "End-to-end ÷ the time this task is expected to need. The expected time is derived per task, per harness by the eval runner (its fastest passing run, or p10 once there are ten) and stamped into the run — never hand-written. Past 2× counts as slow; a task its harness has never passed shows —.",
duration: "End-to-end wall clock for this task: sandbox setup + the agent's turns + grading. A checker that runs a live command can add tens of seconds here, so this is not the agent's time — see Agent.",
agent: "Agent wall clock: the agent's turns, summed. Excludes sandbox setup, pre_run and grading.",
grading: "Time spent checking success criteria (run.json grading_ms). — when the run did not record it.",
cost: "Total billed cost for this task, reported by the SDK (summed across turns).",
variant: "Experiment arm this row was produced by. A run declaring `variants:` executes every task once per arm and keeps each arm's output in its own subtree, so the same task appears once per arm and the two rows are separate measurements — never collapsed together.",
};
Expand Down Expand Up @@ -302,6 +307,8 @@ const DEFAULT_DIR: Record<SortKey, "asc" | "desc"> = {
status: "asc",
score: "desc",
duration: "desc",
agent: "desc",
grading: "desc",
vsExp: "desc",
cost: "desc",
turns: "desc",
Expand Down Expand Up @@ -347,6 +354,15 @@ function compare(
(a.durationSeconds ?? -Infinity) -
(b.durationSeconds ?? -Infinity)
);
case "agent":
return (
(a.agentSeconds ?? -Infinity) - (b.agentSeconds ?? -Infinity)
);
case "grading":
return (
(a.gradingSeconds ?? -Infinity) -
(b.gradingSeconds ?? -Infinity)
);
case "vsExp":
return (
(timeRatio(a.durationSeconds, a.expectedSeconds) ?? -Infinity) -
Expand Down Expand Up @@ -390,7 +406,9 @@ const COLUMNS: Array<{
{ key: "variant", header: "Variant" },
{ key: "status", header: "Status" },
{ key: "score", header: "Score", align: "right" },
{ key: "duration", header: "Duration", align: "right" },
{ key: "duration", header: "End-to-end", align: "right" },
{ key: "agent", header: "Agent", align: "right" },
{ key: "grading", header: "Grading", align: "right" },
{ key: "vsExp", header: "vs Expected", align: "right" },
{ key: "cost", header: "Cost", align: "right" },
{ key: "turns", header: "Turns", align: "right" },
Expand Down Expand Up @@ -804,6 +822,12 @@ export function TaskGrid({
<td className="py-3 px-4 text-right tabular-nums text-gray-700">
{fmtTableDuration(t.durationSeconds)}
</td>
<td className="py-3 px-4 text-right tabular-nums text-gray-700">
{fmtTableDuration(t.agentSeconds ?? null)}
</td>
<td className="py-3 px-4 text-right tabular-nums text-gray-700">
{fmtTableDuration(t.gradingSeconds ?? null)}
</td>
<td
className={`py-3 px-4 text-right tabular-nums font-medium ${timeCellClasses(timeTint)}`}
title={expectedTimeTitle(t.expectedSeconds)}
Expand Down Expand Up @@ -958,7 +982,7 @@ export function TaskGrid({
reviewSelectedSet={reviewSelectedSet}
onToggleReviewTag={onToggleReviewTag}
/>
<dl className="grid grid-cols-4 gap-2 pt-1 text-xs">
<dl className="grid grid-cols-3 gap-2 pt-1 text-xs">
<Stat
label="Score"
value={
Expand All @@ -968,7 +992,7 @@ export function TaskGrid({
}
/>
<Stat
label="Duration"
label="End-to-end"
value={fmtTableDuration(t.durationSeconds)}
sub={
timeRatioValue != null
Expand All @@ -978,6 +1002,14 @@ export function TaskGrid({
subClass={timeCellClasses(timeTint)}
title={expectedTimeTitle(t.expectedSeconds)}
/>
<Stat
label="Agent"
value={fmtTableDuration(t.agentSeconds ?? null)}
/>
<Stat
label="Grading"
value={fmtTableDuration(t.gradingSeconds ?? null)}
/>
<Stat
label="Cost"
value={fmtCost(t.totalCostUsd)}
Expand Down
34 changes: 34 additions & 0 deletions evalboard/lib/__tests__/runs.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -517,6 +517,40 @@ describe("agentSecondsFromRaw", () => {
agentSecondsFromRaw({ iterations: [{ duration_seconds: null }] }),
).toBeNull();
});

test("prefers the stored agent_wall_ms over the iterations sum (#212)", () => {
expect(
agentSecondsFromRaw({
duration: 101.25,
agent_wall_ms: 55657,
iterations: [{ duration_seconds: 1 }],
}),
).toBeCloseTo(55.657, 6);
});

test("a stored null stays null rather than falling back", () => {
expect(
agentSecondsFromRaw({
agent_wall_ms: null,
iterations: [{ duration_seconds: 5 }],
}),
).toBeNull();
});
});

describe("toTaskRow grading split (#212)", () => {
test("carries grading_ms as seconds", () => {
expect(
toTaskRow({ task_id: "x", grading_ms: 27657 }).gradingSeconds,
).toBeCloseTo(27.657, 6);
});

test("null on a run.json that predates the key, never 0", () => {
expect(toTaskRow({ task_id: "x" }).gradingSeconds).toBeNull();
expect(
toTaskRow({ task_id: "x", grading_ms: null }).gradingSeconds,
).toBeNull();
});
});

describe("extractComponentShas", () => {
Expand Down
18 changes: 18 additions & 0 deletions evalboard/lib/runs.ts
Original file line number Diff line number Diff line change
Expand Up @@ -105,6 +105,9 @@ export interface TaskResultSummary {
// The agent's turns alone, without setup and grading. Optional so test
// factories that predate it stay valid.
agentSeconds?: number | null;
// Time spent checking success criteria (run.json `grading_ms`). Null when
// not measured or on a run.json that predates the key — never 0.
gradingSeconds?: number | null;
totalCostUsd: number | null;
actualCommands: number | null;
totalTurns: number | null;
Expand Down Expand Up @@ -505,8 +508,14 @@ export interface RawTaskResult {
replicate_index?: number | null;
status?: string;
weighted_score?: number;
// END-TO-END seconds: setup + the agent's turns + grading.
duration?: number;
iterations?: { duration_seconds?: number | null }[] | null;
// The split of `duration` (#212). Absent on run.json written before it;
// null = not measured (grading_ms is null on an ungraded or detached grade).
agent_wall_ms?: number | null;
setup_ms?: number | null;
grading_ms?: number | null;
total_cost_usd?: number;
input_tokens?: number | null;
output_tokens?: number | null;
Expand Down Expand Up @@ -896,6 +905,8 @@ export function toTaskRow(t: RawTaskResult): TaskResultSummary {
weightedScore: t.weighted_score ?? null,
durationSeconds: t.duration ?? null,
agentSeconds: agentSecondsFromRaw(t),
gradingSeconds:
typeof t.grading_ms === "number" ? t.grading_ms / 1000 : null,
totalCostUsd: t.total_cost_usd ?? null,
actualCommands: t.actual_commands ?? null,
totalTurns: t.total_turns ?? null,
Expand Down Expand Up @@ -1265,7 +1276,14 @@ function mostCommonAgentType(rows: RawTaskResult[]): string | null {
return best;
}

// Prefers the row's stored `agent_wall_ms` (written by the runner's
// result_metrics.agent_wall_ms with this same rule), so the board and run.md
// agree by construction. The iterations sum is the legacy path for a run.json
// written before the key existed. A stored null stays null.
export function agentSecondsFromRaw(t: RawTaskResult): number | null {
if (t.agent_wall_ms !== undefined) {
return typeof t.agent_wall_ms === "number" ? t.agent_wall_ms / 1000 : null;
}
const seconds = (t.iterations ?? [])
.map((i) => i.duration_seconds)
.filter((d): d is number => typeof d === "number" && d > 0);
Expand Down
27 changes: 24 additions & 3 deletions src/coder_eval/reports/html.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,7 @@
from ..analysis import calculate_command_statistics
from ..durations import format_ms
from ..models import JUDGE_CRITERION_TYPES, FinalStatus, eval_result_total_cost, sum_costs
from ..result_metrics import expected_turns_overage, turn_time_buckets
from ..result_metrics import agent_wall_ms, expected_turns_overage, turn_time_buckets
from ..stats import stddev, welch_t_test
from .helpers import (
collect_variant_series,
Expand Down Expand Up @@ -979,11 +979,24 @@ def _render_generation_metrics(result: EvaluationResult) -> str:
+ "scripts/timing/decompose_run.py reports, and the two are not comparable. Negative means "
+ "generation and tool execution overlapped."
)
# `Total Latency` is END-TO-END. Agent Wall / Setup / Grading split it, so a
# checker that runs a live command is not read as agent time (#212). Each
# unmeasured one renders as a dash (CE058).
agent_wall = format_ms(agent_wall_ms(result))
setup = format_ms(result.setup_ms)
grading = format_ms(result.grading_ms)
return f"""
<h2>Generation Metrics</h2>
<div class="card">
<div class="grid">
<div class="stat"><div class="label">Total Latency</div><div class="value">{total_latency}</div></div>
<div class="stat" title="setup + the agent's turns + grading">
<div class="label">Total Latency (end-to-end)</div><div class="value">{total_latency}</div>
</div>
<div class="stat" title="the agent's turns, summed">
<div class="label">Agent Wall</div><div class="value">{agent_wall}</div>
</div>
<div class="stat"><div class="label">Setup</div><div class="value">{setup}</div></div>
<div class="stat"><div class="label">Grading</div><div class="value">{grading}</div></div>
<div class="stat"><div class="label">Turns</div><div class="value">{num_turns}</div></div>
<div class="stat"><div class="label">Assistant Turns</div><div class="value">{asst_turns}</div></div>
<div class="stat"><div class="label">Avg Turn Latency</div><div class="value">{avg_latency}</div></div>
Expand Down Expand Up @@ -1223,12 +1236,20 @@ def _render_variant_generation_metrics(eval_results: list[EvaluationResult]) ->
avg_turn = (sum(per_turn_latencies) / len(per_turn_latencies)) if per_turn_latencies else 0.0
total_latency_fmt = _esc(_format_duration(total_duration))
avg_turn_fmt = _esc(_format_duration(avg_turn))
walls = [ms for r in eval_results if (ms := agent_wall_ms(r)) is not None]
gradings = [r.grading_ms for r in eval_results if r.grading_ms is not None]
total_agent_fmt = _esc(format_ms(sum(walls) if walls else None))
total_grading_fmt = _esc(format_ms(sum(gradings) if gradings else None))
return f"""
<h2>Generation Metrics</h2>
<div class="card">
<div class="grid">
<div class="stat"><div class="label">Tasks</div><div class="value">{total_tasks}</div></div>
<div class="stat"><div class="label">Total Latency</div><div class="value">{total_latency_fmt}</div></div>
<div class="stat" title="setup + the agent's turns + grading, summed over tasks">
<div class="label">Total Latency (end-to-end)</div><div class="value">{total_latency_fmt}</div>
</div>
<div class="stat"><div class="label">Total Agent Wall</div><div class="value">{total_agent_fmt}</div></div>
<div class="stat"><div class="label">Total Grading</div><div class="value">{total_grading_fmt}</div></div>
<div class="stat"><div class="label">Turns</div><div class="value">{total_turns}</div></div>
<div class="stat"><div class="label">Assistant Turns</div><div class="value">{total_asst}</div></div>
<div class="stat"><div class="label">Avg Turn Latency</div><div class="value">{avg_turn_fmt}</div></div>
Expand Down
Loading
Loading