Skip to content

Commit 43b85ca

Browse files
committed
vim(test[harness]): Add tests for multi-ref compare and delta computation
why: Validate the extracted _compute_deltas() helper and the new compare_multi() code path with both unit and integration tests. what: - Add test_compute_deltas_classifies_scenarios unit test with synthetic BenchmarkReport objects testing noise/faster/slower classification - Add test_compare_multi_two_refs_produces_pairwise_report smoke test - Assert compare_multi_refs MCP tool is registered
1 parent d0f1b43 commit 43b85ca

2 files changed

Lines changed: 118 additions & 2 deletions

File tree

‎tests/pytest/test_compare_smoke.py‎

Lines changed: 46 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -7,7 +7,7 @@
77

88
import pytest
99

10-
from libtestvim import CompareSpec
10+
from libtestvim import CompareSpec, MultiCompareReport
1111
from libtestvim.gitmeta import detect_origin_default_ref
1212

1313

@@ -73,3 +73,48 @@ def test_compare_default_ref_runs_correctness_and_cleans_up_worktrees(
7373
assert not report.artifact_bundle.root.is_relative_to(repo_root / "artifacts" / "vim")
7474
assert (report.artifact_bundle.root / "compare.json").is_file()
7575
assert worktrees_after == worktrees_before
76+
77+
78+
@pytest.mark.benchmark
79+
def test_compare_multi_two_refs_produces_pairwise_report(
80+
repo_root: Path,
81+
vim_harness,
82+
) -> None:
83+
if shutil.which(vim_harness.spec.hyperfine_bin) is None:
84+
pytest.skip("hyperfine is not installed")
85+
86+
benchmark = replace(
87+
vim_harness.default_benchmark_spec(
88+
label="compare-multi-smoke",
89+
emit_bundle=True,
90+
append_history=False,
91+
),
92+
warmup_runs=0,
93+
timed_runs=1,
94+
)
95+
worktrees_before = _worktree_roots(repo_root)
96+
97+
report = vim_harness.compare_multi(
98+
("HEAD", "HEAD"),
99+
CompareSpec(
100+
emit_bundle=True,
101+
run_correctness=False,
102+
benchmark=benchmark,
103+
),
104+
)
105+
106+
worktrees_after = _worktree_roots(repo_root)
107+
108+
assert isinstance(report, MultiCompareReport)
109+
assert report.refs == ("HEAD", "HEAD")
110+
assert len(report.commits) == 2
111+
assert len(report.pairwise) == 1
112+
pair = report.pairwise[0]
113+
assert pair.base_ref == "HEAD"
114+
assert pair.target_ref == "HEAD"
115+
assert len(pair.scenarios) == 5
116+
for delta in pair.scenarios:
117+
assert delta.classification == "noise"
118+
assert report.artifact_bundle is not None
119+
assert (report.artifact_bundle.root / "multi-compare.json").is_file()
120+
assert worktrees_after == worktrees_before

‎tests/pytest/test_libtestvim_interfaces.py‎

Lines changed: 72 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -55,7 +55,7 @@ def test_mcp_tool_schemas_and_resources(vim_harness) -> None:
5555
async def scenario() -> None:
5656
async with fastmcp.Client(build_mcp_server(vim_harness)) as client:
5757
tools = {tool.name: tool for tool in await client.list_tools()}
58-
assert {"probe_vim", "run_vim", "benchmark_vim", "compare_refs"} <= tools.keys()
58+
assert {"probe_vim", "run_vim", "benchmark_vim", "compare_refs", "compare_multi_refs"} <= tools.keys()
5959

6060
probe_tool = tools["probe_vim"]
6161
assert probe_tool.outputSchema is not None
@@ -138,6 +138,77 @@ async def scenario() -> None:
138138
run_async(scenario())
139139

140140

141+
def test_compute_deltas_classifies_scenarios(vim_harness) -> None:
142+
from pathlib import Path
143+
144+
from libtestvim.models import (
145+
BenchmarkReport,
146+
BenchmarkScenarioResult,
147+
CapabilityReport,
148+
EnvironmentReport,
149+
GitRepoState,
150+
)
151+
152+
cap = CapabilityReport(
153+
vim_bin="vim",
154+
version_line="VIM 9.1",
155+
version_output="",
156+
feature_flags=(),
157+
supports_profile=False,
158+
supports_channel_log=False,
159+
supports_startuptime=False,
160+
supports_getscriptinfo=False,
161+
)
162+
env = EnvironmentReport(
163+
repo_root=Path("."),
164+
vimrc_path=Path("vimrc"),
165+
plugin_root=None,
166+
fzf_root=None,
167+
artifact_root=Path("."),
168+
fixture_root=None,
169+
tools=(),
170+
git=GitRepoState(root=Path("."), branch="main", commit="abc", is_dirty=False),
171+
)
172+
173+
def make_report(*scenarios: tuple[str, float]) -> BenchmarkReport:
174+
return BenchmarkReport(
175+
version=1,
176+
run_id="test",
177+
generated_at="2026-01-01T00:00:00+00:00",
178+
label="test",
179+
capability=cap,
180+
environment=env,
181+
plugin_manifest=(),
182+
scenarios=tuple(
183+
BenchmarkScenarioResult(
184+
name=name,
185+
command="vim",
186+
mean_seconds=mean,
187+
min_seconds=mean,
188+
max_seconds=mean,
189+
)
190+
for name, mean in scenarios
191+
),
192+
)
193+
194+
base = make_report(("fast", 0.100), ("slow", 0.200), ("same", 0.150))
195+
target = make_report(("fast", 0.080), ("slow", 0.220), ("same", 0.152))
196+
197+
deltas = vim_harness._compute_deltas(base, target)
198+
199+
assert len(deltas) == 3
200+
by_name = {d.name: d for d in deltas}
201+
202+
assert by_name["fast"].classification == "faster"
203+
assert by_name["fast"].percent_delta < -5
204+
205+
assert by_name["slow"].classification == "slower"
206+
assert by_name["slow"].percent_delta > 5
207+
208+
assert by_name["same"].classification == "noise"
209+
assert abs(by_name["same"].percent_delta) < 5
210+
211+
141212
def test_run_result_schema_is_object() -> None:
142213
schema = json_schema(RunResult)
143214
assert isinstance(schema, dict)

0 commit comments

Comments
 (0)