From 71a78cd93b0d6de6b780f9cff4eb495d45187ddc Mon Sep 17 00:00:00 2001 From: chaofengw Date: Fri, 9 Oct 2026 03:53:29 +0000 Subject: [PATCH] fix(aiperf): report benchmark timings directly Report Native and TRTMC benchmark p50 as measurements while retaining work differences and partial coverage as diagnostics. Apply accuracy capacity exclusions to both timing sides and refresh saved reports without inference or changes to accuracy gates. Signed-off-by: chaofengw --- apps/aiperf_qual/README.md | 36 ++--- apps/aiperf_qual/tests/test_benchmark_perf.py | 124 ++++++++++++++++++ apps/aiperf_qual/tests/test_execution.py | 3 +- .../trtmc_aiperf_qual/accuracy_recovery.py | 7 +- .../trtmc_aiperf_qual/benchmark_perf.py | 62 +++++++++ apps/aiperf_qual/trtmc_aiperf_qual/cli.py | 11 ++ .../trtmc_aiperf_qual/execution.py | 28 +++- apps/aiperf_qual/trtmc_aiperf_qual/judge.py | 8 +- apps/aiperf_qual/trtmc_aiperf_qual/report.py | 10 +- .../trtmc_aiperf_qual/report_html.py | 8 +- apps/aiperf_qual/trtmc_aiperf_qual/runner.py | 2 +- 11 files changed, 266 insertions(+), 33 deletions(-) create mode 100644 apps/aiperf_qual/tests/test_benchmark_perf.py create mode 100644 apps/aiperf_qual/trtmc_aiperf_qual/benchmark_perf.py diff --git a/apps/aiperf_qual/README.md b/apps/aiperf_qual/README.md index 501eb48a1b..2bf4374dd2 100644 --- a/apps/aiperf_qual/README.md +++ b/apps/aiperf_qual/README.md @@ -9,9 +9,9 @@ It does not use `qualification_tests/benchmark_qualification`. (`noninferiority.py`): `pass` when TRTMC's regression is shown to be below the benchmark's margin, `fail` when it is shown to exceed it, `inconclusive` otherwise. Tasks without a gold set compare outputs with the native model (conversion parity). Random-weight test models are Perf only (`accuracy_source: none`). -- **Perf**: TRTMC must be faster than the native model (eager) at the candidate's precision: the speedup's - 90% interval lies above 1.05 x (1 + guard) (the 5% margin widened by the largest server-instance and order effect - the order check measured, `guard_percent`), on every timed request, with the same work on both sides. +- **Perf**: report Native and TRTMC task-call p50 from the same benchmark responses used for Acc, + at the recorded effective precision. Timing is a measurement, not an additional acceptance gate. + Output-length differences, execution conditions, and partial coverage remain explicit observations. ## Design @@ -69,9 +69,9 @@ A suite with `base: catalog` overrides the profile's catalog request with its da - Every AIPerf run has a deadline: three times the profile's seconds in the run's ledger (`run-all --ledger`, at least ten minutes), else 12 hours; a GPU phase that fails before producing its result runs once more. - `summary` reports one result per model, worst first: White (no verdict: an error or a failed build; or no valid - comparison: the native model below a benchmark's floor, a Task without an Acc check, timings that cannot be - compared), Red (Acc or Perf worse than native beyond its margin), Yellow (Perf about equal to native, which counts - as a pass, or an Acc difference not shown either way), Green (quality passes with valid comparable timings). Perf is reported on the quality dataset. + comparison: the native model below a benchmark's floor or a Task without an Acc check), Red (Acc worse than + native beyond its margin), Yellow (an Acc difference not shown either way), Green (quality passes with available + benchmark timings). Historical fixed-workload reports retain their original performance lights. ### Performance @@ -92,20 +92,22 @@ conversion parity only. The family's checks keep their documented coverage and t Natural evaluation workloads (`both`) provide timings from the **same outputs used for quality**. Their paired geometric speedup and total-time ratio are descriptive; the 90% interval is across dataset units, -with seeds clustered by problem, not a repeated-run stability interval. Every requested response must be -present, valid, paired, warmed, and at matching effective precision and declared task-call boundaries. -Actual work is compared per sample, so different samples may have different lengths. An unmatched -workload reports its natural-task time ratio and the reason equal-work acceleration is unavailable; -matched subsets never hide failures or shorter outputs. `max_tokens` alone is not work evidence. No -mandatory second, forced-length suite is added for variable-output families. +with seeds clustered by problem, not a repeated-run stability interval. The report shows each side's p50, +effective precision, successful timing coverage, and observed work differences. +Generation length and work comparability do not decide whether benchmark timings are measured. +Inputs excluded by Acc as exceeding bundle capacity are excluded from both timing sides too; their +count and the original attempted request counts remain visible. Other failed or missing requests +remain in coverage and produce a `partial` timing measurement when both sides have timings. +A missing workload or a side without any valid timing is an error. `max_tokens` alone is not work evidence. +No mandatory second, forced-length suite is added for variable-output families. Models with quality benchmarks run **only their required quality workloads**. For example, Qwen uses `mmlu-0shot`, and image/video models use their configured quality datasets. Catalog, near-capacity, and informational replay checks are not extra default workloads. Each side answers each selected problem once, with excluded warmup. The same profiling responses supply Acc, Native/TRTMC task-call p50, and AIPerf client metrics. Multiple required benchmarks remain separate, labelled datasets. -Dataset timing completeness and work comparability are checked; timing results are measurements, -not repeated-run performance acceptance gates. Quality thresholds remain unchanged. Failed or +Dataset timing completeness and work comparability are recorded as observations; timing results +are measurements, not repeated-run performance acceptance gates. Quality thresholds remain unchanged. Failed or unpaired responses remain visible in each side's timing coverage rather than disappearing into a matched subset. Models explicitly lacking a quality benchmark retain one configured performance workload and conversion-parity evidence. The explicit `order-check` diagnostic and historical @@ -113,8 +115,10 @@ fixed-workload reports keep their original statistics. Timing and generation pha Configuration now uses a flat `performance` policy and opt-in `service_metrics`; reports use `performance` and `service_metrics` under schema `trtmc.qualification/v2`. Earlier tiered configurations and reports -are normalized on read, including rejudge, without maintaining another execution path. Rejudge never -promotes descriptive dataset results to an acceptance gate. `torch.compile` remains an optional labelled +are normalized on read, including rejudge, without maintaining another execution path. Rejudge without an +environment refreshes benchmark timings from saved execution records while +preserving the recorded Acc entries and gates. It never promotes descriptive dataset results to +an acceptance gate. `torch.compile` remains an optional labelled reference. Service metrics (client latency, throughput, load sweeps) are opt-in and do not affect the verdict; the prototype's single execution lane and buffered SSE do not measure token TTFT/ITL. diff --git a/apps/aiperf_qual/tests/test_benchmark_perf.py b/apps/aiperf_qual/tests/test_benchmark_perf.py new file mode 100644 index 0000000000..f9f197f0e3 --- /dev/null +++ b/apps/aiperf_qual/tests/test_benchmark_perf.py @@ -0,0 +1,124 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +import copy +import json +from types import SimpleNamespace + +import pytest + +from trtmc_aiperf_qual import benchmark_perf, execution, judge + + +def row(index, ms, tokens=1, text="C", **extra): + return {"metadata": {"session_num": index, "conversation_id": f"session_{index:06d}", + "benchmark_phase": "profiling"}, "status": 200, + "payload": {"prompt": str(index)}, "responses": [{"text": json.dumps({ + "trtmc_timing": {"model_call_ms": ms}, + "trtmc_observation": {"output_tokens": tokens, "text": text}})}], **extra} + + +def rejected(index): + return row(index, None, status=422, error={"message": json.dumps({"error": { + "code": "backend_rejected_request", "message": "prompt exceeds the prefill profile"}})}) + + +def capture(out, candidate, native, accuracy=None): + evidence = execution.Session(out, {}, lambda: 0, + lambda obs: judge.work_signature("generate", obs), + lambda c, n: judge.work_check({"work": [c]}, {"work": [n]}) is None) + for side, records in (("candidate", candidate), ("reference", native)): + directory = out / side + directory.mkdir(parents=True) + (directory / "profile_export_raw.jsonl").write_text("".join(json.dumps(r) + "\n" for r in records)) + evidence.record(SimpleNamespace(directory=directory, exit_code=0, raw_records=lambda: records), { + "name": "mmlu-0shot", "role": "both", "warmup": 1, "gpu_busy_percent": 0, + "expected_requests": len(records), "identity": { + "side": side, "precision": "fp16", "concurrency": 1, "timing_scope": "task-call-wall"}}) + quality = accuracy or [{"suite": "mmlu-0shot", "source": "absolute", "status": "pass"}] + return {"model": "demo", "performance_source": "quality", "accuracy": quality, + "performance": evidence.natural_performance(quality), + "execution": {"records": str(out / "execution.jsonl")}} + + +def verdict(report): + return judge.verdict(report, expected_suites=["mmlu-0shot"], expected_modes=0) + + +def test_different_answer_lengths_are_descriptive_not_a_second_performance_gate(tmp_path): + report = capture(tmp_path, [row(0, 40, 4, "C. i")], [row(0, 20, 2)]) + perf, = report["performance"] + assert verdict(report) == {"acc": "pass", "perf": "measured", "category": "measured", "lights": {}} + assert not perf["comparable"] and perf["matched_pairs"] == 0 + assert perf["candidate"]["p50_ms"] == 40 and perf["reference"]["p50_ms"] == 20 + assert not perf["gate"] and perf["measurement_status"] == "measured" + + +def test_capacity_exclusions_use_the_accuracy_scope_on_both_sides(tmp_path): + report = capture(tmp_path, [row(0, 40), rejected(1)], [row(0, 20), row(1, 200)], [ + {"suite": "mmlu-0shot", "source": "absolute", "status": "pass", "out_of_capacity": 1}]) + perf, = report["performance"] + assert verdict(report)["perf"] == "measured" + assert perf["complete"] and perf["out_of_capacity"] == 1 + assert perf["reference"]["p50_ms"] == 20 + assert all(perf[s]["requests"] == perf[s]["valid_requests"] == 1 for s in ("candidate", "reference")) + assert all(perf[s]["attempted_requests"] == 2 for s in ("candidate", "reference")) + + +def test_partial_failed_workload_reports_coverage_and_available_timings(tmp_path): + report = capture(tmp_path, [row(0, 40), row(1, None, status=500)], [row(0, 20), row(1, 200)]) + perf, = report["performance"] + assert verdict(report)["perf"] == "partial" and verdict(report)["category"] == "measured" + assert not perf["complete"] and perf["candidate"]["valid_requests"] == 1 + assert perf["reference"]["p50_ms"] == 110 + assert perf["measurement_status"] == "partial" + + +def test_no_candidate_timing_remains_an_error(tmp_path): + report = capture(tmp_path, [row(0, None)], [row(0, 20)]) + assert verdict(report)["perf"] == "error" + assert report["performance"][0]["measurement_status"] == "unavailable" + + +def test_refresh_old_exports_preserves_accuracy_and_raw_responses(tmp_path): + report = capture(tmp_path, [row(0, 40), rejected(1)], [row(0, 20), row(1, 200)], [ + {"suite": "mmlu-0shot", "source": "absolute", "status": "pass", "out_of_capacity": 1, + "gate": {"margin": 1.0}, "metrics": {"trtmc_score": 80, "native_score": 80}}]) + path = tmp_path / "execution.jsonl" + old = [json.loads(line) for line in path.read_text().splitlines()] + for batch in old: + for r in batch["records"]: + r.pop("capacity_rejection") + path.write_text("".join(json.dumps(batch) + "\n" for batch in old)) + before = {p: p.read_bytes() for p in tmp_path.rglob("*.jsonl")} + quality = copy.deepcopy(report["accuracy"]) + updated = benchmark_perf.refresh(tmp_path, report) + assert updated["accuracy"] == quality + assert updated["performance"][0]["reference"]["p50_ms"] == 20 + assert verdict(updated)["perf"] == "measured" + assert all(p.read_bytes() == data for p, data in before.items()) + assert benchmark_perf.refresh(tmp_path, updated) == updated + + +def test_capacity_count_mismatch_is_not_silently_filtered(tmp_path): + with pytest.raises(ValueError, match="capacity rejections"): + capture(tmp_path, [row(0, 40), rejected(1)], [row(0, 20), row(1, 200)], [ + {"suite": "mmlu-0shot", "status": "pass", "out_of_capacity": 2}]) + + +def test_capacity_excluded_problem_leaves_all_seed_repetitions(tmp_path): + capture(tmp_path, [row(0, 40), rejected(1)], [row(0, 20), row(1, 200)], [ + {"suite": "mmlu-0shot", "status": "pass", "out_of_capacity": 1}]) + batches = [json.loads(line) for line in (tmp_path / "execution.jsonl").read_text().splitlines()] + more = copy.deepcopy(batches) + for batch in more: + for r in batch["records"]: + r["request_sha"] += "-seed-two" + r["capacity_rejection"] = None + r["valid"] = True + r["model_call_ms"] = 80 + perf = execution.paired_dataset("mmlu-0shot", [batches[0], more[0]], [batches[1], more[1]], + lambda c, n: True, capacity_exclusions=1) + assert perf["complete"] and perf["pairs"] == 2 and perf["out_of_capacity"] == 1 + assert all(perf[s]["requests"] == 2 and perf[s]["attempted_requests"] == 4 + for s in ("candidate", "reference")) diff --git a/apps/aiperf_qual/tests/test_execution.py b/apps/aiperf_qual/tests/test_execution.py index 93924f1c61..2322616c41 100644 --- a/apps/aiperf_qual/tests/test_execution.py +++ b/apps/aiperf_qual/tests/test_execution.py @@ -268,7 +268,8 @@ def test_quality_measurement_verdict_does_not_claim_repeated_performance_gate(tm result = judge.verdict(base, expected_suites=["evaluation"], expected_modes=0) assert result == {"acc": "pass", "perf": "measured", "lights": {}, "category": "measured"} item["complete"] = False - assert judge.verdict(base, expected_suites=["evaluation"], expected_modes=0)["category"] == "error" + partial = judge.verdict(base, expected_suites=["evaluation"], expected_modes=0) + assert partial["perf"] == "partial" and partial["category"] == "measured" item["complete"] = True base["accuracy"].append({"suite": "another-required-dataset", "source": "absolute", "status": "pass"}) assert judge.verdict(base, expected_suites=["evaluation"], expected_modes=0)["perf"] == "error" diff --git a/apps/aiperf_qual/trtmc_aiperf_qual/accuracy_recovery.py b/apps/aiperf_qual/trtmc_aiperf_qual/accuracy_recovery.py index b65ed60e61..771f9c08bc 100644 --- a/apps/aiperf_qual/trtmc_aiperf_qual/accuracy_recovery.py +++ b/apps/aiperf_qual/trtmc_aiperf_qual/accuracy_recovery.py @@ -130,6 +130,8 @@ def aligned_batches(recorded: list[dict]) -> list[dict]: if unit is None: raise ValueError("original evaluation unit is missing from the saved execution batch") rows.append({**row, "sample_id": identity, "unit_id": unit}) + if batch["identity"].get("side") == "candidate": + rows[-1]["capacity_rejection"] = absolute.capacity_rejection(source) aligned.append({**batch, "records": rows, "alignment": ALIGNMENT}) return aligned @@ -181,13 +183,14 @@ def recover(out: Path, model: dict, report: dict, archive: SelectionArchive) -> {"work": [mine] if mine is not None else []}, {"work": [theirs] if theirs is not None else []}) is None session = execution.Session(out, {}, lambda: None, lambda value: value, same_work, batches=aligned) performance = [item for item in report.get("performance", []) if item.get("kind") != "natural_dataset"] - performance.extend(session.natural_performance()) + accuracy = [updates.get(entry["suite"], entry) for entry in report.get("accuracy", [])] + performance.extend(session.natural_performance(accuracy)) # Validate the whole report before publishing any corrected evidence. for path, grades in pending_exports: partial = path.with_suffix(".tmp") partial.write_text("".join(json.dumps(row, ensure_ascii=False) + "\n" for row in grades)) partial.replace(path) - result = {**report, "accuracy": [updates.get(entry["suite"], entry) for entry in report.get("accuracy", [])], + result = {**report, "accuracy": accuracy, "performance": performance} result["accuracy_alignment"] = {"version": ALIGNMENT, "suites": sorted(updates), "runs": evidence, "original_responses_reused": True} diff --git a/apps/aiperf_qual/trtmc_aiperf_qual/benchmark_perf.py b/apps/aiperf_qual/trtmc_aiperf_qual/benchmark_perf.py new file mode 100644 index 0000000000..29e0418de5 --- /dev/null +++ b/apps/aiperf_qual/trtmc_aiperf_qual/benchmark_perf.py @@ -0,0 +1,62 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +"""Refresh benchmark timing reports from saved responses, without inference or regrading.""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from . import absolute, execution, judge +from .aiperf_runner import AiperfRun + + +def refresh(out: Path, report: dict) -> dict: + if report.get("performance_source") != "quality": + return report + recorded_path = Path((report.get("execution") or {}).get("records", "execution.jsonl")) + path = out / recorded_path.name + if not path.is_file(): + if not report.get("execution") and not any(item.get("out_of_capacity") for item in report.get("accuracy", [])): + # Older aggregate-only reports can reapply the measurement verdict, + # but cannot reconstruct a different sample selection. + return report + raise ValueError(f"{out}: recorded benchmark execution is unavailable") + batches, superseded = [], set() + for line in path.read_text().split("\n"): + if not line.strip(): + continue + batch = json.loads(line) + if batch.get("event") == "supersede_failed_attempt": + superseded.update(batch["batch_ids"]) + elif "records" in batch: + batches.append(batch) + batches = [batch for batch in batches + if not batch.get("superseded") and batch["batch_id"] not in superseded] + excluded = {item["suite"] for item in report.get("accuracy", []) + if item.get("out_of_capacity") and item.get("status") != "error"} + # Older execution exports omit rejection reasons. Read their original raw + # error records only when accuracy explicitly excluded capacity rejections. + raw = {} + for batch in batches: + if batch["workload"] not in excluded or batch["identity"].get("side") != "candidate": + continue + for row in batch["records"]: + if row["output_valid"] or "capacity_rejection" in row: + continue + ref = row["output_ref"] + directory = ref["aiperf_run"] + if directory not in raw: + raw[directory] = AiperfRun(Path(directory), batch["aiperf_exit"], []).raw_records() + source = raw[directory][ref["record_index"]] + payload = source.get("payload") or {} + sha = hashlib.sha256(json.dumps(payload, sort_keys=True, separators=(",", ":")).encode()).hexdigest() + if row["request_sha"] != sha or str(row["sample_id"]) != source["metadata"].get("conversation_id"): + raise ValueError("saved capacity rejection does not match its execution identity") + row["capacity_rejection"] = absolute.capacity_rejection(source) + same_work = lambda mine, theirs: judge.work_check( # noqa: E731 + {"work": [mine] if mine is not None else []}, {"work": [theirs] if theirs is not None else []}) is None + evidence = execution.Session(out, {}, lambda: None, lambda value: value, same_work, batches=batches) + performance = [item for item in report.get("performance", []) if item.get("kind") != "natural_dataset"] + return {**report, "performance": [*performance, *evidence.natural_performance(report.get("accuracy", []))]} diff --git a/apps/aiperf_qual/trtmc_aiperf_qual/cli.py b/apps/aiperf_qual/trtmc_aiperf_qual/cli.py index 9780455d49..5a03ec8523 100644 --- a/apps/aiperf_qual/trtmc_aiperf_qual/cli.py +++ b/apps/aiperf_qual/trtmc_aiperf_qual/cli.py @@ -183,6 +183,17 @@ def rejudge_reports(outs: Sequence[Path], environment=None, *, selection_cache: continue result = compat.report(json.loads(path.read_text())) model = compat.configuration(json.loads((out / "model.json").read_text())) + if selection_cache is None and environment is None and result.get("performance_source") == "quality": + from .benchmark_perf import refresh + + result = refresh(out, result) + result["verdict"] = judge.verdict(result, expected_suites=list(expected_suites(model)), expected_modes=0) + preserve_original(out) + result["rejudged"] = {"time": time.time(), "original": ORIGINAL_REPORT, + "benchmark_timings_refreshed": True, "accuracy_contract_preserved": True} + write_report(out, result) + print(json.dumps({"out": str(out), **result["verdict"]})) + continue if selection_cache is not None: preserve_original(out) result = recover(out, model, result, archive) diff --git a/apps/aiperf_qual/trtmc_aiperf_qual/execution.py b/apps/aiperf_qual/trtmc_aiperf_qual/execution.py index ff10d97cf1..9527e77186 100644 --- a/apps/aiperf_qual/trtmc_aiperf_qual/execution.py +++ b/apps/aiperf_qual/trtmc_aiperf_qual/execution.py @@ -139,6 +139,8 @@ class Session: batches: list[dict[str, Any]] = field(default_factory=list) def record(self, run: Any, metadata: Mapping[str, Any]) -> None: + from .absolute import capacity_rejection + identity = dict(metadata["identity"]) units = metadata.get("units") rows = [] @@ -169,6 +171,7 @@ def record(self, run: Any, metadata: Mapping[str, Any]) -> None: rows.append({"sample_id": sample, "unit_id": str(unit), "request_sha": request_key, "model_call_ms": ms if valid_time else None, "valid": bool(valid and valid_time), "output_valid": bool(valid), "work": work, + "capacity_rejection": capacity_rejection(raw) if identity.get("side") == "candidate" else None, "request_problems": list(self.request_problems(payload)), "output_ref": {"aiperf_run": str(run.directory), "record_index": ordinal, "request_id": body.get("request_id") or body.get("id")}}) @@ -182,12 +185,15 @@ def record(self, run: Any, metadata: Mapping[str, Any]) -> None: with (self.out / "execution.jsonl").open("a") as handle: handle.write(json.dumps(batch, default=str) + "\n") - def natural_performance(self) -> list[dict[str, Any]]: + def natural_performance(self, accuracy: Sequence[Mapping[str, Any]] = ()) -> list[dict[str, Any]]: grouped: dict[str, dict[str, list[dict[str, Any]]]] = defaultdict(lambda: defaultdict(list)) for batch in self.batches: if batch["role"] in ("quality", "both") and not batch.get("superseded"): grouped[batch["workload"]][batch["identity"].get("side", "unknown")].append(batch) - return [paired_dataset(name, sides.get("candidate", []), sides.get("reference", []), self.same_work) + excluded = {item["suite"]: int(item["out_of_capacity"]) for item in accuracy + if item.get("out_of_capacity") and item.get("status") != "error"} + return [paired_dataset(name, sides.get("candidate", []), sides.get("reference", []), self.same_work, + excluded.get(name, 0)) for name, sides in grouped.items()] @@ -201,7 +207,7 @@ def session(value: Session) -> Iterator[Session]: def paired_dataset(name: str, candidate: Sequence[Mapping[str, Any]], reference: Sequence[Mapping[str, Any]], - same_work: Callable[[Any, Any], bool]) -> dict[str, Any]: + same_work: Callable[[Any, Any], bool], capacity_exclusions: int = 0) -> dict[str, Any]: """Describe the complete natural workload; never promote a matched subset to a gate. The interval describes variation across paired evaluation units, not repeated @@ -209,7 +215,13 @@ def paired_dataset(name: str, candidate: Sequence[Mapping[str, Any]], reference: Formal fixed-workload gates retain their existing repeated-run statistic. """ sides = (candidate, reference) - rows = [[row for batch in batches for row in batch["records"]] for batches in sides] + all_rows = [[row for batch in batches for row in batch["records"]] for batches in sides] + excluded = {str(row["sample_id"]) for row in all_rows[0] + if row.get("capacity_rejection")} if capacity_exclusions else set() + if len(excluded) != capacity_exclusions: + raise ValueError(f"{name}: recorded capacity rejections do not match the accuracy exclusions") + rows = [[row for row in values if str(row["sample_id"]) not in excluded] + for values in all_rows] indexed = [{(str(row["sample_id"]), row["request_sha"]): row for row in values} for values in rows] reasons = [] complete = all(rows) and all(len(index) == len(values) for index, values in zip(indexed, rows)) @@ -248,18 +260,22 @@ def paired_dataset(name: str, candidate: Sequence[Mapping[str, Any]], reference: result: dict[str, Any] = {"request": name, "reference_mode": "eager", "kind": "natural_dataset", "gate": False, "timing_contract": TIMING_CONTRACT, "pairs": len(pairs), "matched_pairs": matched, "complete": bool(complete and exported), - "comparable": bool(pairs and not reasons), "light": "white" if reasons else "informational", + "comparable": bool(pairs and not reasons), "light": "informational", + "out_of_capacity": len(excluded), "reasons": list(dict.fromkeys(reasons)), "notes": [ "Shared quality outputs; not an additional performance gate.", "Interval across evaluation units, not repeated-run timing stability."], "candidate": {}, "reference": {}} # Report each side's entire valid workload, including unpaired responses. # Pair filtering is only for comparability, never for the displayed timings. - for side, values, precision in zip(("candidate", "reference"), rows, precisions): + for side, values, original, precision in zip(("candidate", "reference"), rows, all_rows, precisions): times = [row["model_call_ms"] for row in values if row["valid"]] result[side] = {"p50_ms": statistics.median(times) if times else None, "total_ms": sum(times), "requests": len(values), "valid_requests": len(times), + "attempted_requests": len(original), "precision": next(iter(precision)) if len(precision) == 1 else None} + timed = all(result[side]["p50_ms"] is not None for side in ("candidate", "reference")) + result["measurement_status"] = "unavailable" if not timed else "measured" if result["complete"] else "partial" if not pairs: return result mine = [row["model_call_ms"] for row, _ in pairs] diff --git a/apps/aiperf_qual/trtmc_aiperf_qual/judge.py b/apps/aiperf_qual/trtmc_aiperf_qual/judge.py index 77400546d3..b2e9d07b32 100644 --- a/apps/aiperf_qual/trtmc_aiperf_qual/judge.py +++ b/apps/aiperf_qual/trtmc_aiperf_qual/judge.py @@ -52,12 +52,14 @@ def verdict(result: Mapping[str, Any], *, expected_suites: Sequence[str], expect names = {item.get("request") for item in datasets} required = set(result.get("performance_expected", [item.get("suite") for item in accuracy if item.get("source") == "absolute"])) - perf = ("error" if not datasets or not required <= names or any(not item.get("complete") for item in datasets) - else "measured" if all(item.get("comparable") for item in datasets) else "not-comparable") + unavailable = any(any((item.get(side) or {}).get("p50_ms") is None + for side in ("candidate", "reference")) for item in datasets) + perf = ("error" if not datasets or not required <= names or unavailable else + "partial" if any(not item.get("complete") for item in datasets) else "measured") # A dataset measurement is not a repeated-run performance acceptance test. category = ("error" if "error" in (acc, perf) else "acc-issue" if acc == "fail" else "acc-inconclusive" if acc == "inconclusive" else "not-comparable" if acc == "not-comparable" else - "perf-inconclusive" if perf == "not-comparable" else "measured") + "measured") return {"acc": acc, "perf": perf, "lights": {}, "category": category} qualifying = [item["light"] for item in result.get("performance", []) if item.get("reference_mode") == QUALIFYING_MODE and item.get("gate", True)] diff --git a/apps/aiperf_qual/trtmc_aiperf_qual/report.py b/apps/aiperf_qual/trtmc_aiperf_qual/report.py index 94692ffd77..d321c10932 100644 --- a/apps/aiperf_qual/trtmc_aiperf_qual/report.py +++ b/apps/aiperf_qual/trtmc_aiperf_qual/report.py @@ -157,10 +157,14 @@ def write_report(out: Path, result: Mapping[str, Any]) -> tuple[Path, Path]: cand, ref = item.get("candidate", {}), item.get("reference", {}) if item.get("kind") == "natural_dataset": notes = "; ".join([*item.get("reasons", []), *item.get("notes", []), "informational; no gate"]) - lines.append(f"| {item['request']} (shared quality outputs) | {item['light']} | " + coverage = "; ".join(f"{label} {side.get('valid_requests')}/{side.get('requests')} timed" + for label, side in (("Native", ref), ("TRTMC", cand))) + if item.get("out_of_capacity"): + coverage += f"; {item['out_of_capacity']} capacity rejections excluded on both sides" + lines.append(f"| {item['request']} (shared quality outputs) | {item.get('measurement_status', 'measured')} | " f"{_fmt(cand.get('p50_ms'))} | — | {_fmt(ref.get('p50_ms'))} | — | {notes} |") - lines += ["", f"{item.get('matched_pairs')}/{item.get('pairs')} paired responses have matching work. " - "Task-call timings reuse the quality outputs; differing work prevents an equal-work comparison.", ""] + lines += ["", f"{coverage}. {item.get('matched_pairs')}/{item.get('pairs')} paired responses have matching work. " + "Task-call timings reuse the benchmark outputs; work differences are informational.", ""] continue unit = " per audio second" if cand.get("unit") or ref.get("unit") else "" lines.append(f"| {item['reference_mode']}{' ' + item['request'] if item.get('request') else ''} | {item['light']} | {_fmt(cand.get('p50_ms'))}{unit} | " diff --git a/apps/aiperf_qual/trtmc_aiperf_qual/report_html.py b/apps/aiperf_qual/trtmc_aiperf_qual/report_html.py index 1ad9f0f693..480b33ba9d 100644 --- a/apps/aiperf_qual/trtmc_aiperf_qual/report_html.py +++ b/apps/aiperf_qual/trtmc_aiperf_qual/report_html.py @@ -231,12 +231,18 @@ def _performance(items: Sequence[Mapping[str, Any]]) -> str: for item in items: candidate, reference = item.get("candidate", {}), item.get("reference", {}) light = item.get("light", "") + if item.get("kind") == "natural_dataset": + light = item.get("measurement_status", "measured") color = LIGHT_COLORS.get(light, "#6e7781") reasons = "; ".join([*item.get("reasons", []), *item.get("notes", [])]) unit = " per audio second" if candidate.get("unit") or reference.get("unit") else "" if item.get("kind") == "natural_dataset": reasons += (f" · shared quality outputs; {item.get('matched_pairs')}/{item.get('pairs')} " - "paired responses have matching work; informational, no gate") + "paired responses have matching work; informational, no gate" + f" · Native {reference.get('valid_requests')}/{reference.get('requests')} timed" + f" · TRTMC {candidate.get('valid_requests')}/{candidate.get('requests')} timed") + if item.get("out_of_capacity"): + reasons += f" · {item['out_of_capacity']} capacity rejections excluded on both sides" else: reasons += " · repeated fixed-workload gate" rows.append(f"{_e(item.get('request') or item.get('reference_mode'))} dict[ with execution.session(evidence): result = _qualify(model, environment, out) result["schema_version"] = execution.SCHEMA - result["performance"].extend(evidence.natural_performance()) + result["performance"].extend(evidence.natural_performance(result["accuracy"])) if quality_only(model): result["performance_source"] = "quality" result["performance_expected"] = [item["suite"] for item in model.get("absolute", [])]