Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion evals/browser/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -53,7 +53,10 @@ success rubric. The judge sees the task, worker result, and coordinator response
a plausible but incomplete answer does not count. Agent time is measured from durable
`message.received` to the terminal `message.completed` event. LLM cost sums
`usage.costUsd` from every completed model step; a `~` prefix means at least one
step did not report cost.
step did not report cost. Cost comparisons require complete measurements for
both variants of a shared passing task; incomplete costs stay marked `~` and
do not contribute to cost deltas. Duration comparisons still include those
shared passing tasks.

## Two-revision A/B

Expand Down
6 changes: 5 additions & 1 deletion evals/browser/dashboard/lib/benchmark-comparison.ts
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
interface ComparableTask {
readonly costComplete: boolean;
readonly costUsd: number | null;
readonly durationMs: number | null;
readonly id: string;
Expand All @@ -13,7 +14,10 @@ export function compareBenchmarkTasks(
return { cost: null, time: null };
}
return {
cost: improvementRatio(baseline.costUsd, candidate.costUsd),
cost:
baseline.costComplete && candidate.costComplete
? improvementRatio(baseline.costUsd, candidate.costUsd)
: null,
time: improvementRatio(baseline.durationMs, candidate.durationMs),
};
}
Expand Down
180 changes: 175 additions & 5 deletions evals/browser/tests/browser-benchmark-comparison.test.ts
Original file line number Diff line number Diff line change
@@ -1,4 +1,11 @@
import { execFileSync } from "node:child_process";
import { mkdtempSync, rmSync, writeFileSync } from "node:fs";
import { tmpdir } from "node:os";
import { join } from "node:path";
import type { MessageStreamEvent } from "eve/client";
import { describe, expect, it } from "vitest";
import { browserBenchmarkSchema } from "../benchmark-schema";
import { measureWorkerTask } from "../worker-events";
import {
averageBenchmarkImprovement,
compareBenchmarkTasks,
Expand All @@ -8,8 +15,15 @@ describe("browser benchmark comparison", () => {
it("reports positive improvement when the candidate is faster and cheaper", () => {
expect(
compareBenchmarkTasks(
{ costUsd: 2, durationMs: 10_000, id: "task", success: true },
{
costComplete: true,
costUsd: 2,
durationMs: 10_000,
id: "task",
success: true,
},
{
costComplete: true,
costUsd: 1.5,
durationMs: 8_000,
id: "task",
Expand All @@ -23,25 +37,51 @@ describe("browser benchmark comparison", () => {
expect(
averageBenchmarkImprovement(
[
{ costUsd: 2, durationMs: 10_000, id: "one", success: true },
{ costUsd: 1, durationMs: 20_000, id: "two", success: true },
{
costComplete: true,
costUsd: 2,
durationMs: 10_000,
id: "one",
success: true,
},
{
costComplete: true,
costUsd: 1,
durationMs: 20_000,
id: "two",
success: true,
},
],
[
{ costUsd: 1, durationMs: 5_000, id: "one", success: true },
{ costUsd: 2, durationMs: 30_000, id: "two", success: true },
{
costComplete: true,
costUsd: 1,
durationMs: 5_000,
id: "one",
success: true,
},
{
costComplete: true,
costUsd: 2,
durationMs: 30_000,
id: "two",
success: true,
},
]
)
).toEqual({ cost: 0.25, time: 0 });
});

it("excludes pairs unless both variants passed", () => {
const baseline = {
costComplete: true,
costUsd: 2,
durationMs: 10_000,
id: "task",
success: true,
};
const candidate = {
costComplete: true,
costUsd: 1,
durationMs: 5_000,
id: "task",
Expand All @@ -57,4 +97,134 @@ describe("browser benchmark comparison", () => {
time: null,
});
});
it("excludes incomplete measured costs without discarding duration comparisons", () => {
const baseline = measuredTask(2, true);
const candidate = measuredTask(1, false);
expect(candidate).toMatchObject({ costComplete: false, costUsd: 1 });
expect(compareBenchmarkTasks(baseline, candidate)).toEqual({
cost: null,
time: 0,
});
expect(compareBenchmarkTasks(candidate, baseline)).toEqual({
cost: null,
time: 0,
});
expect(averageBenchmarkImprovement([baseline], [candidate])).toEqual({
cost: null,
time: 0,
});
});

it("marks partial CLI costs and compares only shared complete measurements", () => {
const directory = mkdtempSync(join(tmpdir(), "benchmark-costs-"));
try {
const baselinePath = join(directory, "baseline.json");
const candidatePath = join(directory, "candidate.json");
writeFileSync(
baselinePath,
JSON.stringify(benchmark("baseline", measuredTask(2, true)))
);
writeFileSync(
candidatePath,
JSON.stringify(benchmark("candidate", measuredTask(1, false)))
);
const partial = execFileSync(
process.execPath,
[
"--experimental-strip-types",
"scripts/compare-browser-benchmarks.ts",
baselinePath,
candidatePath,
],
{ encoding: "utf8", timeout: 10_000 }
);
expect(partial).toContain("~$1.000000");
expect(partial).toContain(
"Comparable LLM cost (0 shared complete measurements): — → — (—)"
);
expect(partial).not.toContain("(-50.0%)");
writeFileSync(
candidatePath,
JSON.stringify(benchmark("candidate", measuredTask(1, true)))
);
const complete = execFileSync(
process.execPath,
[
"--experimental-strip-types",
"scripts/compare-browser-benchmarks.ts",
baselinePath,
candidatePath,
],
{ encoding: "utf8", timeout: 10_000 }
);
expect(complete).toContain(
"Comparable LLM cost (1 shared complete measurements): $2.000000 → $1.000000 ($-1.000000 (-50.0%))"
);
expect(complete).not.toContain("~$");
} finally {
rmSync(directory, { recursive: true, force: true });
}
});
});

function measuredTask(costUsd: number, complete: boolean) {
const events: MessageStreamEvent[] = [
{
data: {
finishReason: "stop",
sequence: 0,
stepIndex: 0,
turnId: "turn",
usage: { costUsd },
},
meta: { at: "2026-10-07T06:00:00.000Z", id: "measured" },
type: "step.completed",
},
];
if (!complete)
events.push({
data: {
finishReason: "stop",
sequence: 1,
stepIndex: 1,
turnId: "turn",
usage: {},
},
meta: { at: "2026-10-07T06:00:01.000Z", id: "unmeasured" },
type: "step.completed",
});
return { ...measureWorkerTask(events, 1000), id: "task", success: true };
}

function benchmark(label: string, task: ReturnType<typeof measuredTask>) {
return browserBenchmarkSchema.parse({
completedAt: "2026-10-07T06:00:01.000Z",
gitSha: null,
label,
startedAt: "2026-10-07T06:00:00.000Z",
summary: {
costComplete: task.costComplete,
failed: 0,
passed: 1,
medianDurationMs: 1000,
p95DurationMs: 1000,
successRate: 1,
totalModelSteps: task.modelSteps,
totalCostUsd: task.costUsd,
},
target: { kind: "local", url: "http://127.0.0.1:9" },
tasks: [
{
...task,
error: null,
evalDurationMs: 1000,
name: "measured cost",
sessionId: "fixture",
status: "completed",
terminalMessage: "done",
verdict: "passed",
},
],
version: 1,
});
}
23 changes: 14 additions & 9 deletions scripts/compare-browser-benchmarks.ts
Original file line number Diff line number Diff line change
Expand Up @@ -29,11 +29,14 @@ const baselineComparableDurations = comparablePairs.map(
const candidateComparableDurations = comparablePairs.map(
(pair) => pair.candidate.durationMs
);
const costComparablePairs = comparablePairs.filter(
(pair) => pair.baseline.costComplete && pair.candidate.costComplete
);
const baselineComparableCost = sumNullable(
comparablePairs.map((pair) => pair.baseline.costUsd)
costComparablePairs.map((pair) => pair.baseline.costUsd)
);
const candidateComparableCost = sumNullable(
comparablePairs.map((pair) => pair.candidate.costUsd)
costComparablePairs.map((pair) => pair.candidate.costUsd)
);

console.log("");
Expand Down Expand Up @@ -64,9 +67,9 @@ for (const pair of pairs) {
comparable
? formatDelta(pair.baseline.durationMs, pair.candidate.durationMs, "ms")
: "—",
formatCost(pair.baseline.costUsd),
formatCost(pair.candidate.costUsd),
comparable
formatCost(pair.baseline.costUsd, pair.baseline.costComplete),
formatCost(pair.candidate.costUsd, pair.candidate.costComplete),
comparable && pair.baseline.costComplete && pair.candidate.costComplete
? formatNullableDelta(
pair.baseline.costUsd,
pair.candidate.costUsd,
Expand All @@ -88,10 +91,10 @@ console.log(
`Comparable P95: ${formatOptionalDuration(percentile(baselineComparableDurations, 0.95))} → ${formatOptionalDuration(percentile(candidateComparableDurations, 0.95))} (${formatNullableDelta(percentile(baselineComparableDurations, 0.95), percentile(candidateComparableDurations, 0.95), "ms")})`
);
console.log(
`Comparable LLM cost: ${formatCost(baselineComparableCost)} → ${formatCost(candidateComparableCost)} (${formatNullableDelta(baselineComparableCost, candidateComparableCost, "$")})`
`Comparable LLM cost (${String(costComparablePairs.length)} shared complete measurements): ${formatCost(baselineComparableCost, true)} → ${formatCost(candidateComparableCost, true)} (${formatNullableDelta(baselineComparableCost, candidateComparableCost, "$")})`
);
console.log(
`Total LLM spend: ${formatCost(baseline.summary.totalCostUsd)} → ${formatCost(candidate.summary.totalCostUsd)}`
`Total LLM spend: ${formatCost(baseline.summary.totalCostUsd, baseline.summary.costComplete)} → ${formatCost(candidate.summary.totalCostUsd, candidate.summary.costComplete)}`
);
console.log(
`Judge score: ${formatScore(baseline.summary.meanJudgeScore)} → ${formatScore(candidate.summary.meanJudgeScore)}`
Expand Down Expand Up @@ -140,8 +143,10 @@ function formatOptionalDuration(milliseconds: number | null) {
return milliseconds === null ? "—" : formatDuration(milliseconds);
}

function formatCost(costUsd: number | null) {
return costUsd === null ? "—" : `$${costUsd.toFixed(6)}`;
function formatCost(costUsd: number | null, complete: boolean) {
return costUsd === null
? "—"
: `${complete ? "" : "~"}$${costUsd.toFixed(6)}`;
}

function formatRate(rate: number) {
Expand Down