From cb9d65212c89978fad9f4196b7091102007d82ee Mon Sep 17 00:00:00 2001 From: Aaron Miller Date: Tue, 25 Aug 2026 15:26:00 +0100 Subject: [PATCH 1/2] =?UTF-8?q?test(bench):=20=F0=9F=A7=AA=20record=20the?= =?UTF-8?q?=20geometry=20a=20scaling=20rung=20is=20identified=20by?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A benchmark artifact recorded `ranks` and rank 0's hostname, which is not enough to say which rung of a node-scaling ladder it came from -- every consumer had to trust the launcher's filename instead of the artifact. Records `nodes` and `ranks_per_node`, derived on the ranks themselves from the distinct hostnames rather than from the environment, and `partitions_env` under a name that does not claim to be the effective value, because no binding exposes the count the engine actually resolved. Adds `memhwm_max` alongside the summed `memhwm`: the sum is a job total, and having only the sum has inverted per-rank memory readings before. Also records the calibrated Hubbard sizing in the bench README, including that both nominal size axes saturate at the default tolerance -- measured, four points on the lattice axis and six on the tolerance axis. Assisted-by: claude-code:claude-opus-5 --- benches/README.md | 18 ++++++++++++++++++ benches/conftest.py | 34 +++++++++++++++++++++++++++++----- 2 files changed, 47 insertions(+), 5 deletions(-) diff --git a/benches/README.md b/benches/README.md index bab14a8f..713a33d4 100644 --- a/benches/README.md +++ b/benches/README.md @@ -24,3 +24,21 @@ run's artifacts into `REPORT.md` and Bencher Metric Format JSON. Cross-engine comparisons against other propagation libraries live in [`../packages/bench-third-party`](../packages/bench-third-party), a standalone uv project with its own lockfile. + +## Fixed-model sizing + +At the default `--hubbard-lower-atol 1e-4`, both nominal Hubbard size axes are +saturated -- `--hubbard-cutoff` is flat from 10, `--hubbard-num-sites` is flat +from 60 -- so `--hubbard-lower-atol` is the only axis that reaches 100M terms. +Measured at `cutoff=10`: `lower_atol` 1e-4 / 5e-5 / 2.5e-5 / 1.25e-5 gives +1,887,255 / 7,156,480 / 26,607,878 / 96,981,051 terms, i.e. terms scale +roughly as `lower_atol^-1.9`. + +`--hubbard-observable-site` (default 46) is a lattice position, not a +constant: sweeping `--hubbard-num-sites` without scaling it as 46/60 of the +lattice slides the observable off-centre and changes the light cone. + +`build_graph` extends the graph, so Hubbard's 29 Trotter steps retain 29 +layer-sets and exceed 229 GiB at 23.9M terms, where `propagate` runs the same +model to 97M terms in well under 2 GiB. Size a graph-holding benchmark from a +graph measurement, never from a `propagate` measurement. diff --git a/benches/conftest.py b/benches/conftest.py index 2204fb33..e6284a69 100644 --- a/benches/conftest.py +++ b/benches/conftest.py @@ -84,6 +84,15 @@ def _size() -> int: return 1 if MPI is None else MPI.COMM_WORLD.Get_size() +def _nodes() -> tuple[int, int]: + """Return (distinct hosts, ranks per host). Collective; serial returns ``(1, 1)``.""" + if MPI is None or MPI.COMM_WORLD.Get_size() == 1: + return 1, 1 + hosts = MPI.COMM_WORLD.allgather(socket.gethostname()) + n = len(set(hosts)) + return n, _size() // n + + def _reduce_sum(comm: Any, value: int) -> int: """Sum ``value`` across ranks. Collective; a serial run returns ``value``.""" if comm is not None and comm.Get_size() > 1: @@ -132,6 +141,7 @@ def _spread(comm: Any, value: int) -> dict[str, int]: "meta": {}, # run configuration (ranks, threads, host, ...) "params": {}, # resolved random-problem hyperparameters "memhwm": {}, # node id -> summed peak RSS, whole test, setup() included + "memhwm_max": {}, # node id -> worst-rank peak RSS, whole test, setup() included "opsize": {}, # picture / model / node id -> {"terms": n} "memrest": {}, # picture / model -> resting RSS bytes "membase": {}, # fixed model -> resting RSS bytes before the model is built @@ -191,16 +201,18 @@ def _core_md5() -> str: return "unavailable" -def _meta() -> dict[str, Any]: +def _meta(nodes: int, ranks_per_node: int) -> dict[str, Any]: """Return this run's configuration metadata for the report.""" try: nanobind_backend_version = version("nanobind-backend") except PackageNotFoundError: nanobind_backend_version = "not installed" - return { + meta = { "label": os.environ.get("monoprop_BENCH_LABEL", "?"), # noqa: SIM112 "ranks": _size(), + "nodes": nodes, + "ranks_per_node": ranks_per_node, "monoprop_threads": os.environ.get("monoprop_NUM_THREADS", "default"), # noqa: SIM112 "cpu_count_logical": psutil.cpu_count(logical=True), "cpu_count_physical": psutil.cpu_count(logical=False), @@ -218,6 +230,12 @@ def _meta() -> dict[str, Any]: # Filled by _record_placement: the threads exist only once a propagator does. "pinning": {}, } + # No nanobind property exposes the engine's resolved partition count (see bindings/binder.h), + # so record the requested env var under its own name rather than claim it is the effective value. + partitions_env = os.environ.get("monoprop_PARTITIONS") # noqa: SIM112 + if partitions_env is not None: + meta["partitions_env"] = partitions_env + return meta def _params(config: pytest.Config) -> dict[str, Any]: @@ -235,8 +253,9 @@ def pytest_configure(config: pytest.Config) -> None: Every rank runs the whole session, so rank 0 alone prints, writes and records. ``trylast`` so the terminal reporter exists before non-root ranks unregister it. """ + nodes, ranks_per_node = _nodes() # collective; every rank must call this if _rank() == 0: - _RESULTS["meta"] = _meta() + _RESULTS["meta"] = _meta(nodes, ranks_per_node) _RESULTS["params"] = _params(config) return @@ -450,17 +469,22 @@ def _do(propagator: Any) -> int: @pytest.fixture(autouse=True) def record_memory(request: pytest.FixtureRequest, bench_comm: Any) -> Iterator[None]: - """Record ``memhwm``: peak RSS over the whole test, summed over ranks. + """Record ``memhwm`` (summed peak RSS) and ``memhwm_max`` (the worst rank's peak RSS). It spans ``setup``, so it predicts an OOM kill but is the wrong number for comparing - operations -- ``opmemdelta`` is that. The reduce is collective; only rank 0 records. + operations -- ``opmemdelta`` is that. The sum alone has inverted per-rank readings here + before, so ``memhwm_max`` is recorded alongside it, never in place of it. Both reduces + are collective; only rank 0 records. """ with HighWaterMark() as window: yield key = request.node.nodeid.split("/")[-1] hwm = _reduce_sum(bench_comm, window.peak_bytes) + hwm_max = _reduce_max(bench_comm, window.peak_bytes) if hwm: _record("memhwm", key, hwm) + if hwm_max: + _record("memhwm_max", key, hwm_max) @pytest.fixture(scope="session", params=["heisenberg", "schrodinger"]) From e46f7909953728f318de97f4918082fd6c0751da Mon Sep 17 00:00:00 2001 From: Aaron Miller Date: Tue, 25 Aug 2026 17:26:24 +0100 Subject: [PATCH 2/2] =?UTF-8?q?test(bench-tools):=20=F0=9F=93=8A=20surface?= =?UTF-8?q?=20the=20scaling=20geometry=20in=20the=20report?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The previous commit records four keys that no reader surfaced: `nodes`, `ranks_per_node`, `partitions_env` and `memhwm_max`. report.py's docstring asks for schema changes in both readers, so this is the other half. `partitions_env` is labelled "Partitions (requested)". It is the environment variable, not a resolved count -- no property exposes the count the engine actually chose -- and a bare "Partitions" column would claim otherwise. `memhwm` is now labelled "summed across ranks" rather than left bare, and `memhwm_max` gets its own table labelled "max across ranks". The two differ by the rank count, so the old unqualified heading was the ambiguity. Absent keys render as the file's existing em-dash, verified against a fixture that predates the keys: absent stays visibly absent rather than becoming 0. Co-Authored-By: Claude Opus 5 --- .../src/monoprop_bench_tools/report.py | 25 ++++++++++++++--- .../monoprop-bench-tools/tests/test_report.py | 27 ++++++++++++++++++- 2 files changed, 47 insertions(+), 5 deletions(-) diff --git a/packages/monoprop-bench-tools/src/monoprop_bench_tools/report.py b/packages/monoprop-bench-tools/src/monoprop_bench_tools/report.py index c941afe5..30410117 100644 --- a/packages/monoprop-bench-tools/src/monoprop_bench_tools/report.py +++ b/packages/monoprop-bench-tools/src/monoprop_bench_tools/report.py @@ -204,6 +204,9 @@ def _config_table(labels: list[str], results: dict[str, dict]) -> list[str]: "nanobind", "Backend", "Ranks", + "Nodes", + "Ranks/node", + "Partitions (requested)", "monoprop threads", "CPUs (logical/physical)", "Host", @@ -215,6 +218,9 @@ def _config_table(labels: list[str], results: dict[str, dict]) -> list[str]: str(metas[label].get("nanobind_version", "—")), str(metas[label].get("nanobind_backend_version", "—")), str(metas[label].get("ranks", "—")), + str(metas[label].get("nodes", "—")), + str(metas[label].get("ranks_per_node", "—")), + str(metas[label].get("partitions_env", "—")), str(metas[label].get("monoprop_threads", "default")), _fmt_cpus(metas[label]), str(metas[label].get("hostname", "—")), @@ -266,16 +272,18 @@ def build_report(results_dir: Path) -> str: def sec(name: str) -> dict[str, dict]: return {lbl: results.get(lbl, {}).get(name, {}) for lbl in labels} - params, opsize, memrest, memory = ( + params, opsize, memrest, memory, memory_max = ( sec("params"), sec("opsize"), sec("memrest"), sec("memhwm"), + sec("memhwm_max"), ) all_ops = sorted( {op for table in timings.values() for op in table} | {op for table in memory.values() for op in table} + | {op for table in memory_max.values() for op in table} ) if not labels or not all_ops: return ( @@ -304,7 +312,7 @@ def ops_section(name: str, picture: str) -> list[str]: level=3, ), *_section( - "Memory (peak RSS)", + "Memory (peak RSS, summed across ranks)", "", "Operation", ops, @@ -312,6 +320,15 @@ def ops_section(name: str, picture: str) -> list[str]: lambda lbl, op: _fmt_mem(memory.get(lbl, {}).get(op)), level=3, ), + *_section( + "Memory (peak RSS, max across ranks)", + "", + "Operation", + ops, + labels, + lambda lbl, op: _fmt_mem(memory_max.get(lbl, {}).get(op)), + level=3, + ), ] lines = [ @@ -320,8 +337,8 @@ def ops_section(name: str, picture: str) -> list[str]: f"Run labels: **{', '.join(labels)}**. Times are the mean over rounds; " "memory is the kernel's exact peak resident footprint (`VmHWM`) during each " "operation, measured from a window reset and settled per operation. Under MPI " - "it is summed across ranks, so ranks peaking at different moments are counted " - "together: an upper bound on the job total.", + "the summed figure counts ranks peaking at different moments together (an " + "upper bound on the job total); the max figure is the single worst rank.", "", *_config_table(labels, results), *_section( diff --git a/packages/monoprop-bench-tools/tests/test_report.py b/packages/monoprop-bench-tools/tests/test_report.py index b90f400f..0603050e 100644 --- a/packages/monoprop-bench-tools/tests/test_report.py +++ b/packages/monoprop-bench-tools/tests/test_report.py @@ -137,6 +137,26 @@ def test_build_report_includes_runtime_provenance(tmp_path: Path) -> None: assert "| np1 | 3.13.2 | 3.2.0 | 2.4.0 |" in md +def test_build_report_includes_node_placement(tmp_path: Path) -> None: + _write_timings(tmp_path) + _write_results( + tmp_path, + meta={"nodes": 4, "ranks_per_node": 8, "partitions_env": "16"}, + ) + md = _collapse(report.build_report(tmp_path)) + + assert "| Nodes | Ranks/node | Partitions (requested) |" in md + assert "| 4 | 8 | 16 |" in md + + +def test_build_report_omits_node_placement_when_absent(tmp_path: Path) -> None: + _write_timings(tmp_path) + _write_results(tmp_path, meta={"ranks": 8}) + md = _collapse(report.build_report(tmp_path)) + + assert "| 8 | — | — | — | default |" in md + + def test_fmt_config_formats_floats_compactly() -> None: assert report._fmt_config(1e-5) == "1e-05" assert report._fmt_config(1.0) == "1" @@ -196,12 +216,17 @@ def test_build_report_includes_memory(tmp_path: Path) -> None: "bench_random.py::test_random_energy[heisenberg]": 52428800, "bench_random.py::test_random_energy[schrodinger]": 104857600, }, + memhwm_max={ + "bench_random.py::test_random_energy[heisenberg]": 41943040, + }, ) md = _collapse(report.build_report(tmp_path)) - assert "Memory (peak RSS)" in md + assert "Memory (peak RSS, summed across ranks)" in md + assert "Memory (peak RSS, max across ranks)" in md assert "50.00 MiB" in md assert "100.00 MiB" in md + assert "40.00 MiB" in md def test_build_report_includes_resting(tmp_path: Path) -> None: