Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 20 additions & 1 deletion src/format_bench/report.py
Original file line number Diff line number Diff line change
Expand Up @@ -59,7 +59,8 @@ def _fair(manifest: dict, results: dict) -> list[str]:
item
for item in formats
if item["comparability"] == Comparability.FULL_COMPARABLE
and item["state"] == ExecutionState.BENCHMARKED
and item["state"]
in {ExecutionState.BENCHMARKED, ExecutionState.REPORTED}
]
output.extend(["", "## Storage Ordering", ""])
if not manifest["rankable"]:
Expand Down Expand Up @@ -184,6 +185,22 @@ def _prompt(results: dict) -> list[str]:
]


def _report_observations(manifest: dict, results: dict) -> None:
for entry in manifest.get("formats", []):
if entry.get("state") == ExecutionState.BENCHMARKED:
entry["state"] = transition(
ExecutionState.BENCHMARKED, ExecutionState.REPORTED
)
Comment thread
Anionix marked this conversation as resolved.
for observation in results.get("results", {}).values():
if (
isinstance(observation, dict)
and observation.get("state") == ExecutionState.BENCHMARKED
):
observation["state"] = transition(
ExecutionState.BENCHMARKED, ExecutionState.REPORTED
)


def render_report(run_dir: Path) -> Path:
manifest = json.loads((run_dir / "manifest.json").read_text(encoding="utf-8"))
results = json.loads((run_dir / "results.json").read_text(encoding="utf-8"))
Expand All @@ -192,6 +209,8 @@ def render_report(run_dir: Path) -> Path:
raise ValueError("report requires benchmarked or reported manifest and results")
if manifest["dataset_id"] != results["dataset_id"]:
raise ValueError("manifest and results dataset mismatch")
# Project observation transitions into the report; persist only after it exists.
_report_observations(manifest, results)
profile = results["profile"]
sections = {
"fair": lambda: _fair(manifest, results),
Expand Down
53 changes: 52 additions & 1 deletion tests/test_report.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,7 @@ def test_prompt_report_is_deterministic_and_includes_exact_tokens(tmp_path: Path
},
"results": {
"prompt_v1": {
"state": "BENCHMARKED",
"metrics": {
"corpus": {
"compact_tsv": {
Expand Down Expand Up @@ -50,7 +51,9 @@ def test_prompt_report_is_deterministic_and_includes_exact_tokens(tmp_path: Path
assert "| compact_tsv | 10 | 4 | 2 | 16 | 3 | 4 |" in first
assert "Direct token counts for binary formats are N/A." in first
assert json.loads((tmp_path / "manifest.json").read_text())["state"] == "REPORTED"
assert json.loads((tmp_path / "results.json").read_text())["state"] == "REPORTED"
reported = json.loads((tmp_path / "results.json").read_text())
assert reported["state"] == "REPORTED"
assert reported["results"]["prompt_v1"]["state"] == "REPORTED"
assert render_report(tmp_path).read_text() == first


Expand Down Expand Up @@ -105,3 +108,51 @@ def test_fair_report_includes_normalized_result_hash(tmp_path: Path) -> None:
report = render_report(tmp_path).read_text()

assert "| csv | read_all | 1 | 1 | 2 | 1 | 4 | abc123 | 100 |" in report
reported = json.loads((tmp_path / "manifest.json").read_text())
assert reported["formats"][0]["state"] == "REPORTED"
assert render_report(tmp_path).read_text() == report


def test_claim_report_preserves_terminal_observations(tmp_path: Path) -> None:
manifest = {"state": "BENCHMARKED", "dataset_id": "fixture", "formats": []}
results = {
"state": "BENCHMARKED",
"dataset_id": "fixture",
"run_id": "claims-fixture",
"profile": "claims",
"environment": {
"git_commit": "abc",
"flake_lock_sha256": "def",
"platform": "test-os",
"machine": "test-cpu",
"python": "3.12.0",
},
"results": {
"measured": {
"comparability": "FULL_COMPARABLE",
"state": "BENCHMARKED",
"failure_reason": None,
},
"unsupported": {
"comparability": "ADAPTED",
"state": "UNSUPPORTED",
"failure_reason": "missing dependency",
},
"negative_research": {
"partial": {
"comparability": "PARTIAL",
"state": "FAILED",
"attempts": [{"result": "failed build"}],
}
},
},
}
(tmp_path / "manifest.json").write_text(json.dumps(manifest))
(tmp_path / "results.json").write_text(json.dumps(results))

render_report(tmp_path)

reported = json.loads((tmp_path / "results.json").read_text())["results"]
assert reported["measured"]["state"] == "REPORTED"
assert reported["unsupported"]["state"] == "UNSUPPORTED"
assert reported["negative_research"]["partial"]["state"] == "FAILED"