From e303ad513b21e3b9468172f73499717e73c2a2ab Mon Sep 17 00:00:00 2001 From: Rudy Celekli Date: Wed, 7 Oct 2026 02:38:04 -0400 Subject: [PATCH] fix: compare only complete browser benchmark costs Signed-off-by: Rudy Celekli --- evals/browser/README.md | 5 +- .../dashboard/lib/benchmark-comparison.ts | 6 +- .../browser-benchmark-comparison.test.ts | 180 +++++++++++++++++- scripts/compare-browser-benchmarks.ts | 23 ++- 4 files changed, 198 insertions(+), 16 deletions(-) diff --git a/evals/browser/README.md b/evals/browser/README.md index 7678280b..1eff6e93 100644 --- a/evals/browser/README.md +++ b/evals/browser/README.md @@ -53,7 +53,10 @@ success rubric. The judge sees the task, worker result, and coordinator response a plausible but incomplete answer does not count. Agent time is measured from durable `message.received` to the terminal `message.completed` event. LLM cost sums `usage.costUsd` from every completed model step; a `~` prefix means at least one -step did not report cost. +step did not report cost. Cost comparisons require complete measurements for +both variants of a shared passing task; incomplete costs stay marked `~` and +do not contribute to cost deltas. Duration comparisons still include those +shared passing tasks. ## Two-revision A/B diff --git a/evals/browser/dashboard/lib/benchmark-comparison.ts b/evals/browser/dashboard/lib/benchmark-comparison.ts index 6d14aa6b..b4682b9b 100644 --- a/evals/browser/dashboard/lib/benchmark-comparison.ts +++ b/evals/browser/dashboard/lib/benchmark-comparison.ts @@ -1,4 +1,5 @@ interface ComparableTask { + readonly costComplete: boolean; readonly costUsd: number | null; readonly durationMs: number | null; readonly id: string; @@ -13,7 +14,10 @@ export function compareBenchmarkTasks( return { cost: null, time: null }; } return { - cost: improvementRatio(baseline.costUsd, candidate.costUsd), + cost: + baseline.costComplete && candidate.costComplete + ? improvementRatio(baseline.costUsd, candidate.costUsd) + : null, time: improvementRatio(baseline.durationMs, candidate.durationMs), }; } diff --git a/evals/browser/tests/browser-benchmark-comparison.test.ts b/evals/browser/tests/browser-benchmark-comparison.test.ts index 3569e328..149ee2ef 100644 --- a/evals/browser/tests/browser-benchmark-comparison.test.ts +++ b/evals/browser/tests/browser-benchmark-comparison.test.ts @@ -1,4 +1,11 @@ +import { execFileSync } from "node:child_process"; +import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import type { MessageStreamEvent } from "eve/client"; import { describe, expect, it } from "vitest"; +import { browserBenchmarkSchema } from "../benchmark-schema"; +import { measureWorkerTask } from "../worker-events"; import { averageBenchmarkImprovement, compareBenchmarkTasks, @@ -8,8 +15,15 @@ describe("browser benchmark comparison", () => { it("reports positive improvement when the candidate is faster and cheaper", () => { expect( compareBenchmarkTasks( - { costUsd: 2, durationMs: 10_000, id: "task", success: true }, { + costComplete: true, + costUsd: 2, + durationMs: 10_000, + id: "task", + success: true, + }, + { + costComplete: true, costUsd: 1.5, durationMs: 8_000, id: "task", @@ -23,12 +37,36 @@ describe("browser benchmark comparison", () => { expect( averageBenchmarkImprovement( [ - { costUsd: 2, durationMs: 10_000, id: "one", success: true }, - { costUsd: 1, durationMs: 20_000, id: "two", success: true }, + { + costComplete: true, + costUsd: 2, + durationMs: 10_000, + id: "one", + success: true, + }, + { + costComplete: true, + costUsd: 1, + durationMs: 20_000, + id: "two", + success: true, + }, ], [ - { costUsd: 1, durationMs: 5_000, id: "one", success: true }, - { costUsd: 2, durationMs: 30_000, id: "two", success: true }, + { + costComplete: true, + costUsd: 1, + durationMs: 5_000, + id: "one", + success: true, + }, + { + costComplete: true, + costUsd: 2, + durationMs: 30_000, + id: "two", + success: true, + }, ] ) ).toEqual({ cost: 0.25, time: 0 }); @@ -36,12 +74,14 @@ describe("browser benchmark comparison", () => { it("excludes pairs unless both variants passed", () => { const baseline = { + costComplete: true, costUsd: 2, durationMs: 10_000, id: "task", success: true, }; const candidate = { + costComplete: true, costUsd: 1, durationMs: 5_000, id: "task", @@ -57,4 +97,134 @@ describe("browser benchmark comparison", () => { time: null, }); }); + it("excludes incomplete measured costs without discarding duration comparisons", () => { + const baseline = measuredTask(2, true); + const candidate = measuredTask(1, false); + expect(candidate).toMatchObject({ costComplete: false, costUsd: 1 }); + expect(compareBenchmarkTasks(baseline, candidate)).toEqual({ + cost: null, + time: 0, + }); + expect(compareBenchmarkTasks(candidate, baseline)).toEqual({ + cost: null, + time: 0, + }); + expect(averageBenchmarkImprovement([baseline], [candidate])).toEqual({ + cost: null, + time: 0, + }); + }); + + it("marks partial CLI costs and compares only shared complete measurements", () => { + const directory = mkdtempSync(join(tmpdir(), "benchmark-costs-")); + try { + const baselinePath = join(directory, "baseline.json"); + const candidatePath = join(directory, "candidate.json"); + writeFileSync( + baselinePath, + JSON.stringify(benchmark("baseline", measuredTask(2, true))) + ); + writeFileSync( + candidatePath, + JSON.stringify(benchmark("candidate", measuredTask(1, false))) + ); + const partial = execFileSync( + process.execPath, + [ + "--experimental-strip-types", + "scripts/compare-browser-benchmarks.ts", + baselinePath, + candidatePath, + ], + { encoding: "utf8", timeout: 10_000 } + ); + expect(partial).toContain("~$1.000000"); + expect(partial).toContain( + "Comparable LLM cost (0 shared complete measurements): — → — (—)" + ); + expect(partial).not.toContain("(-50.0%)"); + writeFileSync( + candidatePath, + JSON.stringify(benchmark("candidate", measuredTask(1, true))) + ); + const complete = execFileSync( + process.execPath, + [ + "--experimental-strip-types", + "scripts/compare-browser-benchmarks.ts", + baselinePath, + candidatePath, + ], + { encoding: "utf8", timeout: 10_000 } + ); + expect(complete).toContain( + "Comparable LLM cost (1 shared complete measurements): $2.000000 → $1.000000 ($-1.000000 (-50.0%))" + ); + expect(complete).not.toContain("~$"); + } finally { + rmSync(directory, { recursive: true, force: true }); + } + }); }); + +function measuredTask(costUsd: number, complete: boolean) { + const events: MessageStreamEvent[] = [ + { + data: { + finishReason: "stop", + sequence: 0, + stepIndex: 0, + turnId: "turn", + usage: { costUsd }, + }, + meta: { at: "2026-10-07T06:00:00.000Z", id: "measured" }, + type: "step.completed", + }, + ]; + if (!complete) + events.push({ + data: { + finishReason: "stop", + sequence: 1, + stepIndex: 1, + turnId: "turn", + usage: {}, + }, + meta: { at: "2026-10-07T06:00:01.000Z", id: "unmeasured" }, + type: "step.completed", + }); + return { ...measureWorkerTask(events, 1000), id: "task", success: true }; +} + +function benchmark(label: string, task: ReturnType) { + return browserBenchmarkSchema.parse({ + completedAt: "2026-10-07T06:00:01.000Z", + gitSha: null, + label, + startedAt: "2026-10-07T06:00:00.000Z", + summary: { + costComplete: task.costComplete, + failed: 0, + passed: 1, + medianDurationMs: 1000, + p95DurationMs: 1000, + successRate: 1, + totalModelSteps: task.modelSteps, + totalCostUsd: task.costUsd, + }, + target: { kind: "local", url: "http://127.0.0.1:9" }, + tasks: [ + { + ...task, + error: null, + evalDurationMs: 1000, + name: "measured cost", + sessionId: "fixture", + status: "completed", + terminalMessage: "done", + verdict: "passed", + }, + ], + version: 1, + }); +} diff --git a/scripts/compare-browser-benchmarks.ts b/scripts/compare-browser-benchmarks.ts index e2962e9d..4d04507f 100644 --- a/scripts/compare-browser-benchmarks.ts +++ b/scripts/compare-browser-benchmarks.ts @@ -29,11 +29,14 @@ const baselineComparableDurations = comparablePairs.map( const candidateComparableDurations = comparablePairs.map( (pair) => pair.candidate.durationMs ); +const costComparablePairs = comparablePairs.filter( + (pair) => pair.baseline.costComplete && pair.candidate.costComplete +); const baselineComparableCost = sumNullable( - comparablePairs.map((pair) => pair.baseline.costUsd) + costComparablePairs.map((pair) => pair.baseline.costUsd) ); const candidateComparableCost = sumNullable( - comparablePairs.map((pair) => pair.candidate.costUsd) + costComparablePairs.map((pair) => pair.candidate.costUsd) ); console.log(""); @@ -64,9 +67,9 @@ for (const pair of pairs) { comparable ? formatDelta(pair.baseline.durationMs, pair.candidate.durationMs, "ms") : "—", - formatCost(pair.baseline.costUsd), - formatCost(pair.candidate.costUsd), - comparable + formatCost(pair.baseline.costUsd, pair.baseline.costComplete), + formatCost(pair.candidate.costUsd, pair.candidate.costComplete), + comparable && pair.baseline.costComplete && pair.candidate.costComplete ? formatNullableDelta( pair.baseline.costUsd, pair.candidate.costUsd, @@ -88,10 +91,10 @@ console.log( `Comparable P95: ${formatOptionalDuration(percentile(baselineComparableDurations, 0.95))} → ${formatOptionalDuration(percentile(candidateComparableDurations, 0.95))} (${formatNullableDelta(percentile(baselineComparableDurations, 0.95), percentile(candidateComparableDurations, 0.95), "ms")})` ); console.log( - `Comparable LLM cost: ${formatCost(baselineComparableCost)} → ${formatCost(candidateComparableCost)} (${formatNullableDelta(baselineComparableCost, candidateComparableCost, "$")})` + `Comparable LLM cost (${String(costComparablePairs.length)} shared complete measurements): ${formatCost(baselineComparableCost, true)} → ${formatCost(candidateComparableCost, true)} (${formatNullableDelta(baselineComparableCost, candidateComparableCost, "$")})` ); console.log( - `Total LLM spend: ${formatCost(baseline.summary.totalCostUsd)} → ${formatCost(candidate.summary.totalCostUsd)}` + `Total LLM spend: ${formatCost(baseline.summary.totalCostUsd, baseline.summary.costComplete)} → ${formatCost(candidate.summary.totalCostUsd, candidate.summary.costComplete)}` ); console.log( `Judge score: ${formatScore(baseline.summary.meanJudgeScore)} → ${formatScore(candidate.summary.meanJudgeScore)}` @@ -140,8 +143,10 @@ function formatOptionalDuration(milliseconds: number | null) { return milliseconds === null ? "—" : formatDuration(milliseconds); } -function formatCost(costUsd: number | null) { - return costUsd === null ? "—" : `$${costUsd.toFixed(6)}`; +function formatCost(costUsd: number | null, complete: boolean) { + return costUsd === null + ? "—" + : `${complete ? "" : "~"}$${costUsd.toFixed(6)}`; } function formatRate(rate: number) {