From 687009f4c0c7a77cad83bbc0bf9b56d9cb8dbd5d Mon Sep 17 00:00:00 2001 From: Emre Date: Sun, 13 Sep 2026 17:34:56 +0300 Subject: [PATCH] Exercise the CUDA comparability warning README and CONTRIBUTING both promise that a CUDA run prints a warning that its numbers are not bit-comparable with CPU runs. The line existed in `train` and no test had ever executed it. A promise nothing exercises is a promise nobody has checked. No GPU is needed and none is wanted. torch.cuda.is_available is pinned True so resolve_device and describe_environment run for real and produce a genuine cuda stamp, and train_baseline is replaced so nothing reaches for a CUDA context this machine does not have -- mocking availability is not the same as having a device, and manual_seed_all and Adam would find that out. What is under test is the CLI branch that reads result.environment["device"], which is what decides whether a CUDA user is told anything at all. Both halves of the branch are covered. A warning printed unconditionally would satisfy the positive test while telling every CPU user their numbers are not comparable, so the CPU case asserts the warning is absent. Mutations run, each red on the test that should catch it: deleting the warning fails the CUDA test; printing it unconditionally fails the CPU test; ignoring --device and always passing cpu fails the CUDA test. The assertions compare with all whitespace removed. rich wraps at a console width this test cannot set, and it strips the space it wrapped on, which turns "Do not mix" into "Do notmix" -- measured while writing this, not assumed. `_flat` alone was not enough. --- tests/test_cli_output.py | 90 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 90 insertions(+) diff --git a/tests/test_cli_output.py b/tests/test_cli_output.py index cfce6a1..48d34f6 100644 --- a/tests/test_cli_output.py +++ b/tests/test_cli_output.py @@ -286,3 +286,93 @@ def test_a_single_pair_reports_that_it_estimated_nothing() -> None: def test_a_measured_grid_reports_its_standard_error_and_sample_size() -> None: assert _uncertainty(_stats(stderr=0.019, n=15)) == "+/- 1.9 (standard error, n=15)" + + +# -------------------------------------------------------------------------- +# The CUDA comparability warning +# -------------------------------------------------------------------------- + + +def _squeezed(text: str) -> str: + """All whitespace removed, so an assertion cannot depend on where rich wrapped. + + `_flat` drops newlines but rich strips the space it wrapped on, which turns + "Do not mix" into "Do notmix" -- measured, not assumed. Removing every + space makes the comparison independent of the console width this test has + no way to set. + """ + return "".join(text.split()) + + +def _cuda_result(monkeypatch: pytest.MonkeyPatch, device: str): + """A `TrainingResult` stamped for `device`, with no CUDA context touched. + + `torch.cuda.is_available` is pinned True so `resolve_device` and + `describe_environment` run for real and produce a genuine `cuda` stamp. The + training call itself is replaced: mocking availability is not the same as + having a GPU, and `manual_seed_all` and Adam would reach for a context this + machine does not have. + """ + torch = pytest.importorskip("torch") + from iqforge.training import TrainingResult, describe_environment, resolve_device + + monkeypatch.setattr(torch.cuda, "is_available", lambda: True) + monkeypatch.setattr(torch.cuda, "get_device_name", lambda *a, **k: "mocked-gpu") + + environment = describe_environment(resolve_device(device)) + seen: dict[str, object] = {} + + def fake_train_baseline(dataset, **kwargs): + seen["device_choice"] = kwargs.get("device_choice") + return TrainingResult( + parameters=13_490, + test_accuracy=0.5, + test_per_class={"a": 0.5, "b": 0.5}, + classes=["a", "b"], + environment=environment, + ) + + monkeypatch.setattr("iqforge.training.train_baseline", fake_train_baseline) + return environment, seen + + +def test_a_cuda_run_warns_that_its_numbers_are_not_comparable( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """The warning line existed but had never been executed by any test. + + It is the only thing that tells a CUDA user their numbers cannot be put + next to a CPU table, and `README`/`CONTRIBUTING` both promise it is + printed. A promise nothing exercises is a promise nobody has checked. + """ + environment, seen = _cuda_result(monkeypatch, "cuda") + assert environment["device"] == "cuda" + + result = runner.invoke(app, ["train", str(tmp_path), "--device", "cuda"]) + + assert result.exit_code == 0, result.output + assert seen["device_choice"] == "cuda" + squeezed = _squeezed(result.output) + assert _squeezed("trained on CUDA") in squeezed + assert _squeezed("these numbers are NOT bit-comparable with CPU runs") in squeezed + assert _squeezed("Do not mix devices inside one paired experiment") in squeezed + + +def test_a_cpu_run_prints_no_comparability_warning( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """The other half of the branch: a GPU being present must not trigger it. + + Without this, a warning printed unconditionally would satisfy the test + above while telling every CPU user their numbers are not comparable. + """ + environment, seen = _cuda_result(monkeypatch, "cpu") + assert environment["device"] == "cpu" + + result = runner.invoke(app, ["train", str(tmp_path)]) + + assert result.exit_code == 0, result.output + assert seen["device_choice"] == "cpu" + squeezed = _squeezed(result.output) + assert _squeezed("trained on CUDA") not in squeezed + assert "bit-comparable" not in squeezed