From 485d3a7a4ae6f8919f6760a295c0fcca1266b8ad Mon Sep 17 00:00:00 2001 From: fishidaho Date: Thu, 17 Sep 2026 16:11:29 -0700 Subject: [PATCH 1/2] Assert memory ceilings with pytest-memray instead of tracemalloc The two memory assertions in the suite reset `tracemalloc`'s peak around the operation and compared the delta to a hand-computed bound. That works, but the bound was rebuilt inline at each site, the setup had to be fenced off by hand, and the number it compared against moved with the runner's core count: the accumulator is `nthreads * n_minor * width * 8` bytes, so the same assertion meant something different on a 4-core runner than on a 96-core one. `pytest-memray` states the same property as a mark. It is what anndata uses for this, it intercepts the allocator rather than Python's allocation hooks, and a ceiling is enforced by having the plugin installed -- no `--memray` flag, so CI needs no change. Three conventions keep the ceilings meaningful: - `pinned_threads` (new, in conftest) fixes numba's thread count, so the expected allocation is a property of the code and not of the machine. Measured at 4 threads: 20 KB for a minor-axis sum, 40 KB for the extrema kernels, 132 KB for the misaligned matmul. - Inputs and JIT warm-up move into module-scoped fixtures. A mark measures the test body only -- verified -- so the array under test no longer has to be fenced off by resetting a counter. - The ceiling rides on each `pytest.param`, never on `request.applymarker`. The plugin reads the marker at collection time, so a marker applied from the body is silently ignored and the test asserts nothing. Every ceiling here was checked by tightening it to 1 KB and confirming it fails. Adds one `limit_leaks` test: a cache that grows with use is invisible to a ceiling, since each pass alone stays well under it, and that is the shape of the bug d286cf3 fixed. Confirmed it catches a deliberately retained result. `benchmarks/` keeps `tracemalloc`: it runs outside pytest, where marks do not apply, and it records numbers rather than bounding them. Both READMEs now say which tool owns which question, including that neither sees RSS -- so neither catches the allocator fragmentation behind #49's 140 GB, which is diagnosable but not reliably gateable. Co-Authored-By: Claude Opus 5 (1M context) --- benchmarks/README.md | 29 +++++++ pyproject.toml | 1 + tests/conftest.py | 26 +++++++ tests/test_minor_axis_matmul.py | 67 ++++++++++++----- tests/test_reduction_memory.py | 74 ++++++++++++------ uv.lock | 129 +++++++++++++++++++++++++++++++- 6 files changed, 279 insertions(+), 47 deletions(-) diff --git a/benchmarks/README.md b/benchmarks/README.md index 66669e1..c5c887f 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -55,3 +55,32 @@ merge. On a 4M-nonzero array, for results of length `n_minor`: Write a function returning `{metric: value}` in `cases.py`, decorated with `@fast` (runs on every PR, keep it under a minute) or `@slow`. Add any new gated metric to `margins` in `baselines.json`, then `--record`. + +## Memory: what belongs here and what belongs in the tests + +Memory shows up in both places, measuring different things, and the split is +deliberate. + +**The test suite owns the ceilings.** `pytest-memray` marks +(`limit_memory`, `limit_leaks`) assert that an operation's allocation is +bounded by its accumulator rather than by `nnz` -- a property that either +holds or does not, with no baseline to record. `memray` intercepts the +allocator itself, so it sees numba's allocations, including the thread-local +accumulators inside a `parallel=True` kernel; `tracemalloc` sees those too, +but only as Python-level allocations, and neither sees resident-set effects +(see below). Those tests pin `numba.set_num_threads` so the expected number +is a property of the code rather than of the runner, and build their inputs +in fixtures, since a mark measures only the test body. + +**The benchmarks own the numbers.** `peak_alloc_mb` records what an operation +allocated so a change in it is visible over time and against a baseline. It +stays on `tracemalloc`: these run outside pytest, where the marks do not +apply. + +Neither measures RSS, and so neither catches allocator *fragmentation* -- many +variably-sized alloc/free cycles driving the resident set far above the live +set, which is what #49 hit at ~1,150 chunks per pass. Both tools report the +live high-water mark, which stays small throughout such a run. That failure +mode is real but is not reliably gateable: RSS moves with the allocator, the +runner and the thread count. Diagnose it with `/proc/self/status` `VmHWM` +around a workload when it is suspected, rather than asserting on it in CI. diff --git a/pyproject.toml b/pyproject.toml index d59859a..3b8a2e4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -46,6 +46,7 @@ dev = [ "ty>=0.0.1a1", "hypothesis>=6.100", "codespell>=2.3", + "pytest-memray>=1.11.0", ] [build-system] diff --git a/tests/conftest.py b/tests/conftest.py index a2d81fd..01af3db 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -36,3 +36,29 @@ def csc(dense) -> sp.csc_array: @pytest.fixture def csr(dense) -> sp.csr_array: return sp.csr_array(dense) + + +#: Threads the memory-limited tests run their kernels with. +#: +#: The thread-local accumulators those tests bound are +#: ``nthreads * n_minor * width * 8`` bytes, so their size follows +#: ``numba.get_num_threads()`` -- which is a property of the machine, not of +#: the code under test. Left free, the same assertion would mean something +#: different on a 4-core runner than on a 96-core one, and a limit loose +#: enough for the widest machine would be too loose to catch a regression on +#: any of them. Pinning it makes the expected allocation a number the test +#: can actually state. +MEMORY_TEST_THREADS = 4 + + +@pytest.fixture +def pinned_threads(): + """Pin numba's thread count so accumulator sizes are machine-independent.""" + import numba + + previous = numba.get_num_threads() + numba.set_num_threads(MEMORY_TEST_THREADS) + try: + yield MEMORY_TEST_THREADS + finally: + numba.set_num_threads(previous) diff --git a/tests/test_minor_axis_matmul.py b/tests/test_minor_axis_matmul.py index b558cb4..4e9f4b4 100644 --- a/tests/test_minor_axis_matmul.py +++ b/tests/test_minor_axis_matmul.py @@ -9,7 +9,6 @@ from __future__ import annotations -import tracemalloc import numpy as np import pytest @@ -93,29 +92,40 @@ def test_no_second_copy_of_the_array_is_built(vcls, rng): assert nv._arr is v -def test_misaligned_matmul_peak_is_bounded_by_the_accumulator_budget(rng): - """Peak memory tracks the (fixed) accumulator budget, not the size of the array.""" - dense = rng.integers(1, 5, size=(1200, 400)).astype(np.float64) +@pytest.fixture(scope="module") +def misaligned_setup(): + """A 1_200 x 400 VCSC view and a width-2 operand, JIT already warmed. + + In a fixture so that neither the array nor the one-off compilation counts + against the ceiling below -- `limit_memory` measures the test body only. + """ + rng = np.random.default_rng(0) + dense = rng.integers(1, 5, size=(1_200, 400)).astype(np.float64) v = VCSCArray.from_scipy(sp.csc_array(dense)) - nnz_bytes = v.nnz * v.indices.dtype.itemsize B = rng.normal(size=(dense.shape[1], 2)) + v.normalized() @ B # warm up the JIT + return v, B, dense + + +# 1_200 x 400 with no zeros is 480_000 nonzeros: 3.8 MB of values and 1.9 MB of +# indices. The accumulator block is `MEMORY_TEST_THREADS * 1_200 * 2 * 8` = +# 77 KB, plus the 19 KB output. A ceiling of 256 KB is ~2x that and ~15x under +# the values array, so it fails if this path ever goes back to building a +# second copy of the array and passes on any runner. +@pytest.mark.limit_memory("256 KB") +def test_misaligned_matmul_peak_is_bounded_by_the_accumulator_budget( + pinned_threads, misaligned_setup +): + """Peak memory tracks the (fixed) accumulator budget, not the size of the array.""" + v, B, _ = misaligned_setup + out = v.normalized() @ B + assert out.shape == (1_200, 2) - v.normalized() @ B # warm up the JIT before measuring - - nv = v.normalized() - tracemalloc.start() - try: - before = tracemalloc.get_traced_memory()[0] - tracemalloc.reset_peak() - out = nv @ B - peak = tracemalloc.get_traced_memory()[1] - finally: - tracemalloc.stop() - # Accumulator block is nthreads * n_rows * width * 8 bytes -- independent - # of nnz, so it stays far below a per-nonzero cost for this shape. - assert peak - before < nnz_bytes - np.testing.assert_allclose(out, _reference(dense) @ B, atol=1e-7) +def test_misaligned_matmul_result_is_still_correct(misaligned_setup): + """The bounded-memory path above must also produce the right numbers.""" + v, B, dense = misaligned_setup + np.testing.assert_allclose(v.normalized() @ B, _reference(dense) @ B, atol=1e-7) def test_accumulator_threads_degrades_to_one_for_a_huge_output_axis(): @@ -123,3 +133,20 @@ def test_accumulator_threads_degrades_to_one_for_a_huge_output_axis(): huge_axis = 2_000_000 wide_b = 200 # e.g. rank + oversampling in a randomized SVD assert accumulator_threads(huge_axis, bytes_per_element=8 * wide_b) == 1 + + +# `limit_leaks` fails when any single call stack still holds memory once the +# body returns, which is the shape of a cache that grows with use rather than +# of a big one-off allocation -- the bug d286cf3 fixed, where the normalization +# cache pinned O(nnz) duals. `limit_memory` cannot see it: each pass on its own +# stays under any sane ceiling, and only the accumulation across passes is +# wrong. It traces native stacks for every allocation and so is markedly +# slower than the ceilings above, which is why there is one of these and not +# one per operation. +@pytest.mark.limit_leaks("128 KB") +def test_repeated_matmul_on_one_view_retains_nothing(pinned_threads, misaligned_setup): + """Iterating on a view must not accumulate: every pass frees what it took.""" + v, B, _ = misaligned_setup + nv = v.normalized() + for _ in range(8): + nv @ B diff --git a/tests/test_reduction_memory.py b/tests/test_reduction_memory.py index 52a0239..d0b57e8 100644 --- a/tests/test_reduction_memory.py +++ b/tests/test_reduction_memory.py @@ -1,6 +1,19 @@ -from __future__ import annotations +"""Reductions: correctness on the minor axis, and the cost of getting there. + +The memory assertions here are `pytest-memray` ceilings rather than measured +numbers. What is being claimed is structural -- a reduction producing an +`n_minor`-sized result must not allocate anything that grows with `nnz` -- so +a ceiling well under nnz-scale states it directly, where a recorded figure +would only show it drifting. + +Two conventions make the ceilings mean the same thing on every machine: +`pinned_threads` fixes the thread count the accumulators are sized by, and the +arrays are built in module-scoped fixtures because a `limit_memory` mark +measures the test body alone. See `benchmarks/README.md` for how this divides +with the benchmark suite, which records memory rather than bounding it. +""" -import tracemalloc +from __future__ import annotations import numba import numpy as np @@ -78,31 +91,44 @@ def test_accumulator_block_stays_within_budget(n_minor, bytes_per_element): assert nthreads * n_minor * bytes_per_element <= _ACCUMULATOR_BUDGET_BYTES +@pytest.fixture(scope="module") +def reduction_array(): + """A 2_000 x 500 array, with every reduction's JIT already warmed. + + Built in a fixture rather than in the test body because ``limit_memory`` + measures only the body -- so the array itself, and the one-off compilation + of the kernels that touch it, stay out of the number being bounded. + """ + rng = np.random.default_rng(0) + dense = rng.integers(1, 5, size=(2_000, 500)).astype(np.float64) + v = VCSRArray.from_scipy(sp.csr_array(dense)) + for warm in (v.sum, v.max, v.getnnz): + warm(axis=0) + return v + + +# Reducing 2_000 x 500 over the minor axis touches 1e6 nonzeros: 8 MB of +# values and 4 MB of indices. The accumulator block is +# `MEMORY_TEST_THREADS * 500 * 8` = 16 KB (32 KB for the extrema kernels, which +# carry two). The ceilings below sit two orders of magnitude under anything +# nnz-sized and roughly 2x over the block, so they catch a reduction that +# starts scaling with nnz without tripping on allocator noise. +# +# The ceiling rides on each `pytest.param` rather than being applied inside the +# test: `pytest-memray` reads the marker when the test is collected, so a +# marker added from the body (`request.applymarker`) is never seen and the +# test silently asserts nothing. @pytest.mark.parametrize( ("label", "call"), [ - ("sum", lambda v: v.sum(axis=0)), - ("max", lambda v: v.max(axis=0)), - ("getnnz", lambda v: v.getnnz(axis=0)), + pytest.param("sum", lambda v: v.sum(axis=0), marks=pytest.mark.limit_memory("64 KB")), + pytest.param("max", lambda v: v.max(axis=0), marks=pytest.mark.limit_memory("96 KB")), + pytest.param("getnnz", lambda v: v.getnnz(axis=0), marks=pytest.mark.limit_memory("64 KB")), ], ) -def test_minor_axis_reductions_allocate_nothing_nnz_sized(label, call): +def test_minor_axis_reductions_allocate_nothing_nnz_sized( + pinned_threads, reduction_array, label, call +): """An n_minor-sized result must not cost nnz-sized scratch.""" - rng = np.random.default_rng(0) - n_rows, n_cols = 2_000, 500 - dense = rng.integers(1, 5, size=(n_rows, n_cols)).astype(np.float64) - v = VCSRArray.from_scipy(sp.csr_array(dense)) - - call(v) # warm up the JIT before measuring - - tracemalloc.start() - try: - before = tracemalloc.get_traced_memory()[0] - tracemalloc.reset_peak() - out = call(v) - peak = tracemalloc.get_traced_memory()[1] - finally: - tracemalloc.stop() - - assert peak - before < 1 << 20, f"{label} allocated {(peak - before) / 1e6:.1f} MB" - assert out.shape == (n_cols,) + out = call(reduction_array) + assert out.shape == (500,) diff --git a/uv.lock b/uv.lock index 06e08d6..ea0f251 100644 --- a/uv.lock +++ b/uv.lock @@ -590,6 +590,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/41/5b/058db09c45ba58a7321bdf2294cae651b37d6fec68117265af90cde043b0/legacy_api_wrap-1.5-py3-none-any.whl", hash = "sha256:5a8ea50e3e3bcbcdec3447b77034fd0d32cb2cf4089db799238708e4d7e0098d", size = 10182, upload-time = "2025-11-03T13:21:11.102Z" }, ] +[[package]] +name = "linkify-it-py" +version = "2.2.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/45/98/7a1a5f31fd5c7ba93e963b168e244b8e3dd705b3d2a718e3c3307583bf57/linkify_it_py-2.2.0.tar.gz", hash = "sha256:907acd2d17ac1fbb9ddb62c8957ccbd6158cac602231a15c3b0cd1e215f03cee", size = 32939, upload-time = "2026-08-29T07:07:08.305Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/13/d4/1152d1c7ab42d8b908be64fd200ddc870dc9d4925e951198702084aa1a7d/linkify_it_py-2.2.0-py3-none-any.whl", hash = "sha256:3adc40eb5af300b2605fcfdb968c24e1d780a90f1f2221af7c15e5111e94d443", size = 21971, upload-time = "2026-08-29T07:07:07.164Z" }, +] + [[package]] name = "llvmlite" version = "0.49.0" @@ -627,6 +636,11 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/b3/81/4da04ced5a082363ecfa159c010d200ecbd959ae410c10c0264a38cac0f5/markdown_it_py-4.2.0-py3-none-any.whl", hash = "sha256:9f7ebbcd14fe59494226453aed97c1070d83f8d24b6fc3a3bcf9a38092641c4a", size = 91687, upload-time = "2026-05-07T12:08:27.182Z" }, ] +[package.optional-dependencies] +linkify = [ + { name = "linkify-it-py" }, +] + [[package]] name = "markupsafe" version = "3.0.3" @@ -711,6 +725,61 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/b3/38/89ba8ad64ae25be8de66a6d463314cf1eb366222074cfda9ee839c56a4b4/mdurl-0.1.2-py3-none-any.whl", hash = "sha256:84008a41e51615a49fc9966191ff91509e3c40b939176e643fd50a5c2196b8f8", size = 9979, upload-time = "2022-08-14T12:40:09.779Z" }, ] +[[package]] +name = "memray" +version = "1.20.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "jinja2" }, + { name = "rich" }, + { name = "textual" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/99/3b/8f9736cbf698e62cb7efb0e1715a30d6d06b3cffd980b435209eb522d5f8/memray-1.20.0.tar.gz", hash = "sha256:ce1f1d900948d57d7db5b5d8d81f4ebfcb798eff674493503ce5bfaafcea84dc", size = 2416480, upload-time = "2026-08-07T20:16:20.295Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/83/39/3038dd5a10a1512ab7a0ad30c09e13ea1ad26e09ee70d319d037dafbe63c/memray-1.20.0-cp312-cp312-macosx_10_14_x86_64.whl", hash = "sha256:f5d6980770a2d5d1a4f3e8d04d6f92309119671c1f830959a8bea868fab0b01f", size = 2226078, upload-time = "2026-08-07T20:14:46.831Z" }, + { url = "https://files.pythonhosted.org/packages/fe/c9/b5b7e0ed44298d521ac0d6d334fae7d934f7a7b63be5c52933ebfcf7403d/memray-1.20.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:7bcd5094cbb9bd1d11a5dfcc523daaa99f743e645a8b0b0805beaa2bbddbf356", size = 2194301, upload-time = "2026-08-07T20:14:48.123Z" }, + { url = "https://files.pythonhosted.org/packages/c5/6d/a1afee889818ba9633638707bea93ef0db644dd4ad6be13bcd4bcaa3b739/memray-1.20.0-cp312-cp312-manylinux2014_i686.manylinux_2_17_i686.whl", hash = "sha256:f6653ac2fbd49e66e2883575fa8c5dde127f4699205415d1170a8cf7d1db1c86", size = 9873569, upload-time = "2026-08-07T20:14:49.563Z" }, + { url = "https://files.pythonhosted.org/packages/3a/b6/807399cc9e4e733b15206be18244a34b5c7606c47a803cca2c55062ce487/memray-1.20.0-cp312-cp312-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:57981abd8c526d4375c7f640a9d27d5e319b888e79b51688a8fae2f947fe8294", size = 10141659, upload-time = "2026-08-07T20:14:51.588Z" }, + { url = "https://files.pythonhosted.org/packages/e7/3e/ae31597901b5e3bffeca5c5db916f3ad6d1a45ec1547f72821025bd98501/memray-1.20.0-cp312-cp312-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2a6d63a3df7eb96809b82cbd9445033ec1d51574be0f9b3eddf4d64632af7162", size = 9555506, upload-time = "2026-08-07T20:14:53.683Z" }, + { url = "https://files.pythonhosted.org/packages/a0/f3/54f218a7c7d604f6b11cf23ae74988bc7749e861e7e91dab5ee615d3293a/memray-1.20.0-cp312-cp312-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:68dca0caa0b1d1b8474ee54efb98eeb33f548d28bb246113caac45125da7b9c3", size = 9794360, upload-time = "2026-08-07T20:14:55.627Z" }, + { url = "https://files.pythonhosted.org/packages/c2/8e/b35e61137dac319394809cabfb5405c98bade1f129368aef2d81990706e1/memray-1.20.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:b40bf09977f84afe815cca1f34f8a29040a9e757b55467d513ecb9262e6bd809", size = 12438571, upload-time = "2026-08-07T20:14:57.725Z" }, + { url = "https://files.pythonhosted.org/packages/25/51/f9f775da2a08e6bf877b38c9ad83ae140852e2f98a23685f6a04ff4bda2a/memray-1.20.0-cp313-cp313-macosx_10_14_x86_64.whl", hash = "sha256:359824cc26c9a208e83ab058850e9bf16592d7928307e8c2dc6c7b55e6b6cfa6", size = 2225759, upload-time = "2026-08-07T20:14:59.731Z" }, + { url = "https://files.pythonhosted.org/packages/bb/f2/351f0d534b8df45ac3cea2c90a14050351304f47a019db5ceab3363aa090/memray-1.20.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:60f3bf8762adc15a2214f44ad631cddb9e4fa4f4a0dda6aa0ccac6ea71239283", size = 2193524, upload-time = "2026-08-07T20:15:01.186Z" }, + { url = "https://files.pythonhosted.org/packages/9a/45/b6fe8e3120a28013709b34de654ce90b0fec1c89dec4392cd1f3d8d8a1c5/memray-1.20.0-cp313-cp313-manylinux2014_i686.manylinux_2_17_i686.whl", hash = "sha256:546a60027fc9c8a5fbeea62dce45e830d78769a45192098469ee39b2d75c97fc", size = 9871149, upload-time = "2026-08-07T20:15:02.675Z" }, + { url = "https://files.pythonhosted.org/packages/24/8b/e59d5144428046bd43f6f7fbfd04cbbebdba94ca42798fc370abd9b833da/memray-1.20.0-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:d2f3f681ce8acd713a7216d6c362387e70122527e741685e82f85cfdf6aedf0b", size = 10129435, upload-time = "2026-08-07T20:15:04.566Z" }, + { url = "https://files.pythonhosted.org/packages/3b/91/1aa88c9f9dda541265aac80bcdddffcc641374b8fe05a8caa6b3f02245bf/memray-1.20.0-cp313-cp313-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:97539fd71c565c55e208df9ea28ac158cd156b8c33564853685688595c5a991b", size = 9554666, upload-time = "2026-08-07T20:15:06.516Z" }, + { url = "https://files.pythonhosted.org/packages/a1/a6/d31b07fa7aa757c3d2cf9bc4850eb4cbfbbd3bff1f6dd710620248b57739/memray-1.20.0-cp313-cp313-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:3c3aec7a4d0b0d0eb55d72e815a492ff060a3a46b014f6ae216ae68a006ea304", size = 9791047, upload-time = "2026-08-07T20:15:08.63Z" }, + { url = "https://files.pythonhosted.org/packages/14/0e/65901e28faeb25a27655a43dfdf0e4a36dae5b12f3ee66d7ad849ae8754c/memray-1.20.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:0c18705b70d20161616d3fc9d0b0c7796211b3ab16499dfb29c8f3ab7ece46c2", size = 12431537, upload-time = "2026-08-07T20:15:10.817Z" }, + { url = "https://files.pythonhosted.org/packages/6e/63/7976092700eecf4e36ae5793020d18f3feab96aabc2f50d8c7669f751326/memray-1.20.0-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:90fc0ebfc6a264fe3b77d5c263678eaf4a47a0f4288e1b6a04ae3439743cd4be", size = 2227009, upload-time = "2026-08-07T20:15:12.733Z" }, + { url = "https://files.pythonhosted.org/packages/af/1a/0f44b1b626169d40defaa2ae9286a70d40653ac26865f0604a882a10ffb2/memray-1.20.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:d1503ac18928d2493662af95935b9ccaadbf9f2579d86f6757304c23291a077b", size = 2195287, upload-time = "2026-08-07T20:15:14.107Z" }, + { url = "https://files.pythonhosted.org/packages/ff/09/45b74d8750ddccb9ad76a8ed639b46b2c3990f99dbe77f800f6f524257fe/memray-1.20.0-cp314-cp314-manylinux2014_i686.manylinux_2_17_i686.whl", hash = "sha256:d92692fb5266aca4322e2001e18810207b170dfc7ce2e8f0570c7ee181574063", size = 9871353, upload-time = "2026-08-07T20:15:15.595Z" }, + { url = "https://files.pythonhosted.org/packages/54/b6/d0b4eb7324bf47b52165aea947bb8942807a40684a64f9decdf4b2a34387/memray-1.20.0-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:37bc107afee942f162d30160dfd7f28c2b989251c574570f2f920f5bbbb323e4", size = 10112228, upload-time = "2026-08-07T20:15:18.15Z" }, + { url = "https://files.pythonhosted.org/packages/96/18/b70db3a617ba0be54730a0ddf72822df62551d503bba5c131afea1b47da1/memray-1.20.0-cp314-cp314-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:83cfa504ed4e1061b85af4b3c40228616a852851ebf5968ff1c3d85e363f8d24", size = 9550151, upload-time = "2026-08-07T20:15:20.623Z" }, + { url = "https://files.pythonhosted.org/packages/b9/35/3a972ce61e83648f970dbff8d54dcb0a6a54ab298e283cce139c40c8615b/memray-1.20.0-cp314-cp314-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:5588174a24ac150100faa7ee60b535b0f09eae82bf672133443425a00704b769", size = 9776012, upload-time = "2026-08-07T20:15:22.822Z" }, + { url = "https://files.pythonhosted.org/packages/6b/d3/7d59a88509a301aa02c033b95460266553a689fc99f5b1b0d0f138675fc9/memray-1.20.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:2f51218744259983ad0ca66366ca2ff80158513132ca9bd835b1874a453199dd", size = 12425490, upload-time = "2026-08-07T20:15:24.977Z" }, + { url = "https://files.pythonhosted.org/packages/27/71/3560aa204af1771cb25ab91489c48643dfe1ca0ce93783fb70ad677ca058/memray-1.20.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:610643cfa186f7958a476e5589f6e8a1b0b7b3008ab10250a0f02575a692ef1d", size = 2239309, upload-time = "2026-08-07T20:15:27.025Z" }, + { url = "https://files.pythonhosted.org/packages/02/e4/f3981abfecc1572fd24274386f3cbbc36b861704a3e3a10aebe2775aa1a0/memray-1.20.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:420bc47528fedf1a92832290c7a7b8a940c2e30bd9337a85811da18a27e07220", size = 2213281, upload-time = "2026-08-07T20:15:28.629Z" }, + { url = "https://files.pythonhosted.org/packages/72/ee/a04a21557e50d76eef59f0fd4d5f293a2334142540a7180bb627d8ef5b73/memray-1.20.0-cp314-cp314t-manylinux2014_i686.manylinux_2_17_i686.whl", hash = "sha256:59e44042f2495698f6a17fe0a831051ea515242e567f08457ee8546283372993", size = 9836577, upload-time = "2026-08-07T20:15:30.102Z" }, + { url = "https://files.pythonhosted.org/packages/31/66/13ec31746d61855bbe05a090b26a0a9aa6fd741bb3a9a2c0122061d3b985/memray-1.20.0-cp314-cp314t-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:ffd3c7c53419465922be92654c17ebff577708fa1f74168dbef9eb5efa6bcb02", size = 10087796, upload-time = "2026-08-07T20:15:32.042Z" }, + { url = "https://files.pythonhosted.org/packages/f2/37/27da92dcca4c19ca19fdd5c43de474755eac9d16255a0d8f2104d5976998/memray-1.20.0-cp314-cp314t-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:ac9a5dd9bef7f44331d98174625167bf0a8f61785ed05440d99f581c93790acf", size = 9602483, upload-time = "2026-08-07T20:15:34.148Z" }, + { url = "https://files.pythonhosted.org/packages/9c/cc/6b682e4287dfb4bba6d0e0fd8e685fc3b1f0b769707f09c6305217cf8e1e/memray-1.20.0-cp314-cp314t-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:70010acd3d07dbf41aaca0787de3c81c5d8e22e07a6c0fdf04d66c4a376c0193", size = 9758476, upload-time = "2026-08-07T20:15:36.179Z" }, + { url = "https://files.pythonhosted.org/packages/13/c3/ffbba95b0c468d831c63aa6e7dc9e7bb21074d1899c310604567d10ca204/memray-1.20.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:3270aad45dfa0bc7d32bb86b352aefabee765a8658aace8fb9dc8112a07f2e3c", size = 12388833, upload-time = "2026-08-07T20:15:38.049Z" }, + { url = "https://files.pythonhosted.org/packages/42/d7/e76c9efcc466273294a5ec41a518de77c8a2e9ee6ac05049f69fff087bc5/memray-1.20.0-cp315-cp315-macosx_10_15_x86_64.whl", hash = "sha256:a6b79c356f7caeed51cbd25dee85485b6210803199f769da8bc3d061bd0e0ada", size = 2227064, upload-time = "2026-08-07T20:15:40.018Z" }, + { url = "https://files.pythonhosted.org/packages/14/78/5c481eecbbe2b954631dfa9fea383971aa34d186743f74279df644287649/memray-1.20.0-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:a5fc5251b02d99a98eb5874cabfcb74b492ab28e1d5c8b55d481ba0d5e273bf0", size = 2195169, upload-time = "2026-08-07T20:15:41.498Z" }, + { url = "https://files.pythonhosted.org/packages/6c/2e/6713da0bec71d2074c6c6a2269b9c69a209090591f277a6db1982cc580a9/memray-1.20.0-cp315-cp315-manylinux2014_i686.manylinux_2_17_i686.whl", hash = "sha256:08b4d1e17219caabc2a01aa17e8e602b9db67a6aad3ba88b134dfc22bace420d", size = 9874606, upload-time = "2026-08-07T20:15:43.27Z" }, + { url = "https://files.pythonhosted.org/packages/f1/40/451ac23a92d9cd0e11901c00377589ba64b0b8c5cfd595ae403dd14f1a88/memray-1.20.0-cp315-cp315-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:b71fe846cbabce7f33017ac94ba2f157b39e22dbc4f723aad0407dd38520f855", size = 10114282, upload-time = "2026-08-07T20:15:45.26Z" }, + { url = "https://files.pythonhosted.org/packages/a6/88/a426f7229b8e72b5d45a51781018d6c696330633bee149a5b2effbc7d426/memray-1.20.0-cp315-cp315-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:ab77856a592b04586ffb4a87e26d4de440ee3aba6facbaeabdb20855950fdebc", size = 9553111, upload-time = "2026-08-07T20:15:47.15Z" }, + { url = "https://files.pythonhosted.org/packages/82/24/cae0361d483c6816ad4e2607aaa8864902c6d7cd808dafc6c0ad9c5f99c2/memray-1.20.0-cp315-cp315-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:27dc015d1bfb9e43dc0030b412ba435ccff89cd5b7b28d7da9c8e630e7cfe3ec", size = 9778397, upload-time = "2026-08-07T20:15:49.684Z" }, + { url = "https://files.pythonhosted.org/packages/8f/a1/9be6ff8023790e2e89f6882add459e6aceb45ea7d8170f460e3431c95fd2/memray-1.20.0-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:25e4a802488d54030c30a6bdb969fd79f701228db35e6a834eac75259c69018b", size = 12429015, upload-time = "2026-08-07T20:15:51.769Z" }, + { url = "https://files.pythonhosted.org/packages/b0/fa/af4f576cce0516e8eec2d25ed83106bbf1da866a6d275b9ea6a6d64efa81/memray-1.20.0-cp315-cp315t-macosx_10_15_x86_64.whl", hash = "sha256:9d0c12433fde039a594b6bc254758c236db5532a4ee2ed758b9ec73f8fc78f62", size = 2239298, upload-time = "2026-08-07T20:15:53.916Z" }, + { url = "https://files.pythonhosted.org/packages/f0/6c/71d39e6ebd1ee3ebb3c0037c1b0c32460bd413623fc67ac50ebfb4ee936a/memray-1.20.0-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:cdf08e64a9bcb598ae76ca467273b580b4658de49b73bd78dd8345a736d78816", size = 2213507, upload-time = "2026-08-07T20:15:55.473Z" }, + { url = "https://files.pythonhosted.org/packages/28/7d/607c4dda5f3752d0696bc2c1f5f24970542bb82ccc337244b2b6a7831885/memray-1.20.0-cp315-cp315t-manylinux2014_i686.manylinux_2_17_i686.whl", hash = "sha256:7e106a6dfd3823194a5ecd0b147ea7cffae900e422402cfac9ff4b2c774b4695", size = 9836320, upload-time = "2026-08-07T20:15:57.098Z" }, + { url = "https://files.pythonhosted.org/packages/2c/bf/b7bff1759a95230bf4191b7ee4a66867fd4e78e02849fa7032464edf853d/memray-1.20.0-cp315-cp315t-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:b9ec82a31783a6e468135a6f8a7a996fbc31530a5b725faffb5a8be7e2431a2b", size = 10082346, upload-time = "2026-08-07T20:15:59.545Z" }, + { url = "https://files.pythonhosted.org/packages/85/26/0873448e1445a67679a7168a32890dd75b0d63dfe7f07bca3a58d11fe72b/memray-1.20.0-cp315-cp315t-manylinux_2_26_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c8c6ee6c3df06abf2d4230d0f2c0440e91e0c8fd51c1fe9b32599485254d1f56", size = 9593794, upload-time = "2026-08-07T20:16:01.611Z" }, + { url = "https://files.pythonhosted.org/packages/89/c6/2eb1717d80308240aa41d97915fa5fb372c1f71ef68c844f162350295a4f/memray-1.20.0-cp315-cp315t-manylinux_2_26_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:734a082db18947f8afa609b2b4daca20a8b45100a3e6b6c0e4321d59ed781288", size = 9749626, upload-time = "2026-08-07T20:16:03.523Z" }, + { url = "https://files.pythonhosted.org/packages/43/50/76598b0e0bf727f4bcdd1a65ac50d609d4952c7d736c27da6a4196468b61/memray-1.20.0-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:c19934e6b713dbbf35f15d3c84f289e048be613e85039913a500393f10272b9f", size = 12377327, upload-time = "2026-08-07T20:16:05.636Z" }, +] + [[package]] name = "myst-parser" version = "5.1.0" @@ -921,6 +990,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/d2/cf/6a51b2c38980e04c279fd2fa908a1b0982064e860444acfca4ec2e2c8359/pandas-3.0.5-cp314-cp314t-win_arm64.whl", hash = "sha256:3c5015fd1730fbf883647e88068176c839c102cea883ba1769a6f4593bfc1f8c", size = 9509776, upload-time = "2026-07-22T22:19:26.694Z" }, ] +[[package]] +name = "platformdirs" +version = "4.11.9" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/58/b9/8adc4e1b422b27fd88540ec7bf1f406f77ef393ec070e26fc430e914cde8/platformdirs-4.11.9.tar.gz", hash = "sha256:e2c66a8d384596cd98e3c4aea2d761df7bac95d9d8a2cc3946daa8cdafdaebc1", size = 38345, upload-time = "2026-09-16T13:31:45.259Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/f3/94/803ba86705257d7eedddac4b02eb88a7483b1600e9200c1fefc6f1a9a3ff/platformdirs-4.11.9-py3-none-any.whl", hash = "sha256:0a3958f58a9e30321eaef0a424dd0b77cce242886b36b8aa992f9731ef2d59c1", size = 24472, upload-time = "2026-09-16T13:31:44.049Z" }, +] + [[package]] name = "pluggy" version = "1.6.0" @@ -1073,6 +1151,19 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/9d/7a/d968e294073affff457b041c2be9868a40c1c71f4a35fcc1e45e5493067b/pytest_cov-7.1.0-py3-none-any.whl", hash = "sha256:a0461110b7865f9a271aa1b51e516c9a95de9d696734a2f71e3e78f46e1d4678", size = 22876, upload-time = "2026-03-21T20:11:14.438Z" }, ] +[[package]] +name = "pytest-memray" +version = "1.11.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "memray" }, + { name = "pytest" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/71/67/4d8d29f9edc0bd3a4ee2458197bc72cb8bac34beb5e3fee2d1a542c3252c/pytest_memray-1.11.0.tar.gz", hash = "sha256:d40a7914f63cf53d9cd3d1ed5bdaeb9039dde4821d80054fd678f3d4f2c1f13d", size = 246383, upload-time = "2026-09-17T16:42:52.198Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/d5/93/880c259592072b43e416cca0ecd0ed8d9b6413700554997c78b39c946f39/pytest_memray-1.11.0-py3-none-any.whl", hash = "sha256:33e45b77e984459c3792ef111c2142d9ca960c874c075f62349fdcee91509de6", size = 20344, upload-time = "2026-09-17T16:42:50.73Z" }, +] + [[package]] name = "python-dateutil" version = "2.9.0.post0" @@ -1155,6 +1246,19 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/a0/f4/c67b0b3f1b9245e8d266f0f112c500d50e5b4e83cb6f3b71b6528104182a/requests-2.34.2-py3-none-any.whl", hash = "sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0", size = 73075, upload-time = "2026-05-14T19:25:26.443Z" }, ] +[[package]] +name = "rich" +version = "15.0.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "markdown-it-py" }, + { name = "pygments" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/c0/8f/0722ca900cc807c13a6a0c696dacf35430f72e0ec571c4275d2371fca3e9/rich-15.0.0.tar.gz", hash = "sha256:edd07a4824c6b40189fb7ac9bc4c52536e9780fbbfbddf6f1e2502c31b068c36", size = 230680, upload-time = "2026-04-12T08:24:00.75Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/82/3b/64d4899d73f91ba49a8c18a8ff3f0ea8f1c1d75481760df8c68ef5235bf5/rich-15.0.0-py3-none-any.whl", hash = "sha256:33bd4ef74232fb73fe9279a257718407f169c09b78a87ad3d296f548e27de0bb", size = 310654, upload-time = "2026-04-12T08:24:02.83Z" }, +] + [[package]] name = "roman-numerals" version = "4.1.0" @@ -1430,6 +1534,23 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/52/a7/d2782e4e3f77c8450f727ba74a8f12756d5ba823d81b941f1b04da9d033a/sphinxcontrib_serializinghtml-2.0.0-py3-none-any.whl", hash = "sha256:6e2cb0eef194e10c27ec0023bfeb25badbbb5868244cf5bc5bdc04e4464bf331", size = 92072, upload-time = "2024-07-29T01:10:08.203Z" }, ] +[[package]] +name = "textual" +version = "8.2.8" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "markdown-it-py", extra = ["linkify"] }, + { name = "mdit-py-plugins" }, + { name = "platformdirs" }, + { name = "pygments" }, + { name = "rich" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/00/21/39a76b01bd5eea82a04baaca7580e105d8c59450df03998345bb2cfb307b/textual-8.2.8.tar.gz", hash = "sha256:3f106a9fbc73e39dd266c9712432087de78a6d644084c7c241d6a25c3169115b", size = 1860502, upload-time = "2026-06-30T06:51:24.495Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fb/be/35261223d9416a0751cdff1c7b4a6f881387218a12d439fe22fefebc8c04/textual-8.2.8-py3-none-any.whl", hash = "sha256:267375fd402dc8d981457212efa71f0e3365fd17bba144ba9bb3ed7563cb374a", size = 731418, upload-time = "2026-06-30T06:51:26.364Z" }, +] + [[package]] name = "ty" version = "0.0.80" @@ -1496,7 +1617,7 @@ wheels = [ [[package]] name = "vsparse" -version = "0.3.0" +version = "0.4.0" source = { editable = "." } dependencies = [ { name = "anndata" }, @@ -1516,10 +1637,11 @@ docs = [ [package.dev-dependencies] dev = [ - { name = "hypothesis" }, { name = "codespell" }, + { name = "hypothesis" }, { name = "pytest" }, { name = "pytest-cov" }, + { name = "pytest-memray" }, { name = "ruff" }, { name = "ty" }, ] @@ -1540,10 +1662,11 @@ provides-extras = ["docs"] [package.metadata.requires-dev] dev = [ - { name = "hypothesis", specifier = ">=6.100" }, { name = "codespell", specifier = ">=2.3" }, + { name = "hypothesis", specifier = ">=6.100" }, { name = "pytest", specifier = ">=8.0" }, { name = "pytest-cov", specifier = ">=5.0" }, + { name = "pytest-memray", specifier = ">=1.11.0" }, { name = "ruff", specifier = ">=0.6" }, { name = "ty", specifier = ">=0.0.1a1" }, ] From 4d04032dba037f9bd3c72c84fd4b5f262ae1e11c Mon Sep 17 00:00:00 2001 From: fishidaho Date: Thu, 17 Sep 2026 16:36:18 -0700 Subject: [PATCH 2/2] Retire the benchmark suite's memory metrics in favour of the ceilings With `pytest-memray` asserting the memory properties directly, recording them here as well left two mechanisms for one claim, and the weaker one gated. The claims were never about a number. "A minor-axis reduction must not allocate anything nnz-sized" either holds or it does not; a recorded figure shows it drifting only after the fact, against a baseline that has to be re-recorded whenever the bound legitimately moves -- and the checked-in ceilings were already stale, generous figures taken before the memory fixes landed, with a table in the README tracking how far off they were. Worse, the number was never portable: `peak_alloc_mb` measured a thread-local accumulator sized by the runner's core count, so the same code produced a different figure on every machine. A ceiling with a pinned thread count says the thing that is actually meant. Removed: the five memory-only cases, `peak_alloc_mb_view` and its companions from the 30 normalized-view cases, `matmul_peak_alloc_mb` from the slow case, and `peak_alloc_mb` from the harness. 24 gated cases remain. Nothing loses coverage. Two gaps were filled first rather than dropped: `minor_selection_peak_mb` had no test counterpart, so `test_minor_axis_selection_allocates_no_nnz_sized_scratch` covers it -- with a ceiling above the output, since a selection's result legitimately grows with what was selected, and below output-plus-an-nnz-sized-temporary. And the 30 per-recipe view cases became a parametrization of the existing matmul ceiling over every recipe; measured, all five allocate the same 132 KB. Layout size and throughput stay. `bytes_per_nonzero` is the compression format's core claim and has no other guard, and the scipy-relative timings have no equivalent in the test suite. Only the half memray replaces is gone. Co-Authored-By: Claude Opus 5 (1M context) --- benchmarks/README.md | 59 ++++++----------- benchmarks/baselines.json | 109 ++++++-------------------------- benchmarks/cases.py | 89 ++------------------------ benchmarks/harness.py | 20 ------ tests/conftest.py | 13 +--- tests/test_minor_axis_matmul.py | 43 ++++--------- tests/test_reduction_memory.py | 63 +++++++++--------- 7 files changed, 94 insertions(+), 302 deletions(-) diff --git a/benchmarks/README.md b/benchmarks/README.md index c5c887f..7fede32 100644 --- a/benchmarks/README.md +++ b/benchmarks/README.md @@ -17,11 +17,10 @@ Exits nonzero if a gated metric exceeds its ceiling. Correctness tests do not catch cost regressions, so the suite measures two things that move silently: -**Layout size and memory allocated by an operation.** Both deterministic and -comparable across machines, so they are gated tightly. Memory uses -`tracemalloc` rather than `ru_maxrss`, which is a process-lifetime high-water -mark and reports zero for an operation staying under the peak set while -building its input. +**Layout size.** `bytes_per_nonzero` and `indices_bytes_per_nonzero` -- what +the compressed format actually costs per stored value. Deterministic and +comparable across machines, so they are gated tightly, and nothing else in +the project guards them. **Throughput relative to scipy**, never absolute seconds. The same work is timed through scipy in the same process and the ratio recorded, which cancels @@ -38,17 +37,7 @@ leak between them otherwise. metrics named in `margins` are gated; anything else a case returns is recorded for context. -The checked-in ceilings were recorded before the memory fixes landed, so the -memory ones are deliberately generous and should be re-recorded as those -merge. On a 4M-nonzero array, for results of length `n_minor`: - -| metric | recorded | with the fix | -|---|---|---| -| `minor_sum_peak_mb` | 66 MB | 0.8 MB | -| `minor_extrema_peak_mb` | 66 MB | 1.6 MB | -| `minor_getnnz_peak_mb` | 32 MB | 0.8 MB | -| `minor_selection_peak_mb` | 62 MB | 4.8 MB | -| `misaligned_matmul_peak_mb` | 204 MB | 115 MB | +Memory is no longer measured here; see below. ## Adding a case @@ -56,31 +45,21 @@ Write a function returning `{metric: value}` in `cases.py`, decorated with `@fast` (runs on every PR, keep it under a minute) or `@slow`. Add any new gated metric to `margins` in `baselines.json`, then `--record`. -## Memory: what belongs here and what belongs in the tests - -Memory shows up in both places, measuring different things, and the split is -deliberate. +## Memory is a test, not a benchmark -**The test suite owns the ceilings.** `pytest-memray` marks -(`limit_memory`, `limit_leaks`) assert that an operation's allocation is -bounded by its accumulator rather than by `nnz` -- a property that either -holds or does not, with no baseline to record. `memray` intercepts the -allocator itself, so it sees numba's allocations, including the thread-local -accumulators inside a `parallel=True` kernel; `tracemalloc` sees those too, -but only as Python-level allocations, and neither sees resident-set effects -(see below). Those tests pin `numba.set_num_threads` so the expected number -is a property of the code rather than of the runner, and build their inputs -in fixtures, since a mark measures only the test body. +Memory is not measured here. `pytest-memray` ceilings in `tests/` assert it +instead, because the claims are structural: "a minor-axis reduction must not +allocate anything nnz-sized" either holds or it does not. A ceiling fails the +moment it stops being true, where a recorded number only shows it drifting, +against a baseline needing a re-record whenever the bound legitimately moves. -**The benchmarks own the numbers.** `peak_alloc_mb` records what an operation -allocated so a change in it is visible over time and against a baseline. It -stays on `tracemalloc`: these run outside pytest, where the marks do not -apply. +Those tests pin `numba.set_num_threads`, so a ceiling means the same thing on +a 4-core runner and a 96-core one. A recorded figure could not: the +accumulators being bounded are sized by the runner's core count. -Neither measures RSS, and so neither catches allocator *fragmentation* -- many +Neither tool sees RSS, so neither catches allocator *fragmentation* -- many variably-sized alloc/free cycles driving the resident set far above the live -set, which is what #49 hit at ~1,150 chunks per pass. Both tools report the -live high-water mark, which stays small throughout such a run. That failure -mode is real but is not reliably gateable: RSS moves with the allocator, the -runner and the thread count. Diagnose it with `/proc/self/status` `VmHWM` -around a workload when it is suspected, rather than asserting on it in CI. +set. Both report the live high-water mark, which stays small throughout such a +run. That failure mode is real but not reliably gateable, since RSS moves with +the allocator, the runner and the thread count; diagnose it with +`/proc/self/status` `VmHWM` around a workload when it is suspected. diff --git a/benchmarks/baselines.json b/benchmarks/baselines.json index 251c274..31b3717 100644 --- a/benchmarks/baselines.json +++ b/benchmarks/baselines.json @@ -1,12 +1,10 @@ { - "comment": "Ceilings a metric must stay under, regenerated with `python -m benchmarks.run --set fast --record`. Only metrics listed in `margins` are gated; the rest are recorded by cases for context. Margins are the slack applied to a measured value when recording: tight for deterministic layout/memory numbers, loose for timing ratios, which vary with the runner.", + "comment": "Ceilings a metric must stay under, regenerated with `python -m benchmarks.run --set fast --record`. Only metrics listed in `margins` are gated; the rest are recorded by cases for context. Margins are the slack applied to a measured value when recording: tight for deterministic layout numbers, loose for timing ratios, which vary with the runner.", "margins": { "bytes_per_nonzero": 1.1, "indices_bytes_per_nonzero": 1.1, "vs_scipy_ratio": 1.1, - "peak_alloc_mb": 2.0, "time_ratio_vs_scipy": 4.0, - "peak_alloc_mb_view": 2.0, "time_ratio_view_over_materialize": 4.0, "cpu_ratio_1t_cp10k_log1p": 1.5, "cpu_ratio_1t_parafac2": 1.5, @@ -20,136 +18,71 @@ "indices_bytes_per_nonzero": 4.4, "vs_scipy_ratio": 0.4751 }, - "minor_sum_peak_mb": { - "peak_alloc_mb": 132.5131 - }, - "misaligned_matmul_peak_mb": { - "peak_alloc_mb": 407.2383 - }, "matvec_vs_scipy": { "time_ratio_vs_scipy": 1.0 }, "matmat_vs_scipy": { "time_ratio_vs_scipy": 1.0 }, - "minor_extrema_peak_mb": { - "peak_alloc_mb": 132.5453 - }, - "minor_getnnz_peak_mb": { - "peak_alloc_mb": 64.0324 - }, - "minor_selection_peak_mb": { - "peak_alloc_mb": 124.5457 - }, "normalized_cp10k_log1p_matmat_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 2.7262 + "time_ratio_view_over_materialize": 0.3 }, "normalized_cp10k_log1p_matvec_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 0.3858 + "time_ratio_view_over_materialize": 0.3 }, "normalized_cp10k_log1p_rmatmat_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 3.6521 + "time_ratio_view_over_materialize": 0.3 }, "normalized_cp10k_log1p_rmatvec_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 0.1321 + "time_ratio_view_over_materialize": 0.3 }, "normalized_parafac2_matmat_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 2.7262 + "time_ratio_view_over_materialize": 0.3 }, "normalized_parafac2_matvec_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 0.3858 + "time_ratio_view_over_materialize": 0.3 }, "normalized_parafac2_rmatmat_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 3.6521 + "time_ratio_view_over_materialize": 0.3 }, "normalized_parafac2_rmatvec_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 0.1321 + "time_ratio_view_over_materialize": 0.3 }, "normalized_pearson_matmat_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 2.7262 + "time_ratio_view_over_materialize": 0.3 }, "normalized_pearson_matvec_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 0.3858 + "time_ratio_view_over_materialize": 0.3 }, "normalized_pearson_rmatmat_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 3.6521 + "time_ratio_view_over_materialize": 0.3 }, "normalized_pearson_rmatvec_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 0.1321 + "time_ratio_view_over_materialize": 0.3 }, "normalized_raw_matmat_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 2.7262 + "time_ratio_view_over_materialize": 0.3 }, "normalized_raw_matvec_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 0.3858 + "time_ratio_view_over_materialize": 0.3 }, "normalized_raw_rmatmat_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 3.6521 + "time_ratio_view_over_materialize": 0.3 }, "normalized_raw_rmatvec_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 0.1321 + "time_ratio_view_over_materialize": 0.3 }, "normalized_scanpy_matmat_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 2.7262 + "time_ratio_view_over_materialize": 0.3 }, "normalized_scanpy_matvec_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 0.3858 + "time_ratio_view_over_materialize": 0.3 }, "normalized_scanpy_rmatmat_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 3.6521 + "time_ratio_view_over_materialize": 0.3 }, "normalized_scanpy_rmatvec_vs_materialize": { - "time_ratio_view_over_materialize": 0.3, - "peak_alloc_mb_view": 0.1321 - }, - "normalized_cp10k_log1p_matmat_vs_sparse": { - "peak_alloc_mb_view": 2.7262 - }, - "normalized_cp10k_log1p_matvec_vs_sparse": { - "peak_alloc_mb_view": 0.3858 - }, - "normalized_parafac2_matmat_vs_sparse": { - "peak_alloc_mb_view": 2.7262 - }, - "normalized_parafac2_matvec_vs_sparse": { - "peak_alloc_mb_view": 0.3858 - }, - "normalized_pearson_matmat_vs_sparse": { - "peak_alloc_mb_view": 2.7262 - }, - "normalized_pearson_matvec_vs_sparse": { - "peak_alloc_mb_view": 0.3858 - }, - "normalized_raw_matmat_vs_sparse": { - "peak_alloc_mb_view": 2.7262 - }, - "normalized_raw_matvec_vs_sparse": { - "peak_alloc_mb_view": 0.3858 - }, - "normalized_scanpy_matmat_vs_sparse": { - "peak_alloc_mb_view": 2.7262 - }, - "normalized_scanpy_matvec_vs_sparse": { - "peak_alloc_mb_view": 0.3858 + "time_ratio_view_over_materialize": 0.3 }, "normalized_cpu_vs_sparse_1t": { "cpu_ratio_1t_cp10k_log1p": 7.815, diff --git a/benchmarks/cases.py b/benchmarks/cases.py index f813ab0..47443a8 100644 --- a/benchmarks/cases.py +++ b/benchmarks/cases.py @@ -9,7 +9,6 @@ best_cpu_time, best_time, integer_counts_csr, - peak_alloc_mb, ratio_vs_scipy, ) @@ -46,75 +45,6 @@ def layout_bytes_per_nonzero() -> dict[str, float]: } -# -- memory ceilings --------------------------------------------------------- - - -@fast -def minor_sum_peak_mb() -> dict[str, float]: - """Memory allocated by a minor-axis sum.""" - from vsparse import VCSRArray - - v = VCSRArray.from_scipy(integer_counts_csr(40_000, 2_000, density=0.05)) - nnz_mb = v.nnz * 8 / 1e6 - return { - "peak_alloc_mb": peak_alloc_mb(lambda: v.sum(axis=0)), - "expanded_nnz_mb": nnz_mb, # what a per-nonzero temporary would cost - } - - -@fast -def misaligned_matmul_peak_mb() -> dict[str, float]: - """Memory allocated by the matmul direction the storage isn't aligned for.""" - from vsparse import VCSCArray - - v = VCSCArray.from_scipy(integer_counts_csr(40_000, 2_000, density=0.05)) - rng = np.random.default_rng(0) - B = rng.normal(size=(v.shape[1], 4)) - array_mb = (v.values.nbytes + v.value_ptr.nbytes + v.indices.nbytes) / 1e6 - - return { - "peak_alloc_mb": peak_alloc_mb(lambda: v.normalized() @ B), - "array_mb": array_mb, # what a full second copy would cost - } - - -@fast -def minor_extrema_peak_mb() -> dict[str, float]: - """Memory allocated by a minor-axis max/min.""" - from vsparse import VCSRArray - - v = VCSRArray.from_scipy(integer_counts_csr(40_000, 2_000, density=0.05)) - return { - "peak_alloc_mb": peak_alloc_mb(lambda: v.max(axis=0)), - "expanded_nnz_mb": v.nnz * 8 / 1e6, - } - - -@fast -def minor_getnnz_peak_mb() -> dict[str, float]: - """Memory allocated by a per-minor-index stored-element count.""" - from vsparse import VCSRArray - - v = VCSRArray.from_scipy(integer_counts_csr(40_000, 2_000, density=0.05)) - return { - "peak_alloc_mb": peak_alloc_mb(lambda: v.getnnz(axis=0)), - "indices_nnz_mb": v.nnz * 8 / 1e6, - } - - -@fast -def minor_selection_peak_mb() -> dict[str, float]: - """Memory allocated by a minor-axis selection.""" - from vsparse import VCSRArray - - v = VCSRArray.from_scipy(integer_counts_csr(40_000, 2_000, density=0.05)) - cols = np.arange(0, v.shape[1], 2) - return { - "peak_alloc_mb": peak_alloc_mb(lambda: v[:, cols]), - "indices_nnz_mb": v.nnz * 8 / 1e6, - } - - # -- throughput, relative to scipy ------------------------------------------- @@ -144,12 +74,12 @@ def matmat_vs_scipy() -> dict[str, float]: # -- normalized views (issue #40 recipes): view-op vs materialize-then-op --- # -# For every recipe, the view-based matmul/matvec should cost less, both in -# time and in peak allocation, than fully materializing the (dense, -# implicit-zero-filling) normalized matrix and multiplying that -- the whole -# point of a *view*. ``time_ratio_view_over_materialize`` < 1 and -# ``peak_alloc_mb_view`` < ``peak_alloc_mb_materialize`` are the expectation -# for every case below. +# For every recipe, the view-based matmul/matvec should be faster than fully +# materializing the (dense, implicit-zero-filling) normalized matrix and +# multiplying that -- the whole point of a *view*. +# ``time_ratio_view_over_materialize`` < 1 is the expectation for every case +# below. The matching memory claim is asserted as a ceiling in +# ``tests/test_minor_axis_matmul.py`` rather than recorded here. def _normalized_bench(recipe: str, *, vector: bool) -> Callable[[], dict[str, float]]: @@ -170,8 +100,6 @@ def via_materialize() -> np.ndarray: return { "time_ratio_view_over_materialize": best_time(via_view) / best_time(via_materialize), - "peak_alloc_mb_view": peak_alloc_mb(via_view), - "peak_alloc_mb_materialize": peak_alloc_mb(via_materialize), } bench.__name__ = f"normalized_{recipe}_{'matvec' if vector else 'matmat'}_vs_materialize" @@ -196,8 +124,6 @@ def via_materialize() -> np.ndarray: return { "time_ratio_view_over_materialize": best_time(via_view) / best_time(via_materialize), - "peak_alloc_mb_view": peak_alloc_mb(via_view), - "peak_alloc_mb_materialize": peak_alloc_mb(via_materialize), } bench.__name__ = f"normalized_{recipe}_{'rmatvec' if vector else 'rmatmat'}_vs_materialize" @@ -278,8 +204,6 @@ def via_sparse() -> np.ndarray: # every case doubled the suite's runtime for a number nothing gates. return { "wall_ratio_view_over_sparse": best_time(via_view) / best_time(via_sparse), - "peak_alloc_mb_view": peak_alloc_mb(via_view), - "peak_alloc_mb_sparse_delta": peak_alloc_mb(lambda: _sparse_delta(nv, mat)), } bench.__name__ = f"normalized_{recipe}_{'matvec' if vector else 'matmat'}_vs_sparse" @@ -368,7 +292,6 @@ def large_layout_and_matmul() -> dict[str, float]: return { "bytes_per_nonzero": stored / v.nnz, "time_ratio_vs_scipy": ratio_vs_scipy(lambda: v @ B, lambda: mat @ B), - "matmul_peak_alloc_mb": peak_alloc_mb(lambda: v @ B), } diff --git a/benchmarks/harness.py b/benchmarks/harness.py index ada58dc..4f2f7f9 100644 --- a/benchmarks/harness.py +++ b/benchmarks/harness.py @@ -1,7 +1,6 @@ from __future__ import annotations import time -import tracemalloc from collections.abc import Callable from typing import Any @@ -9,25 +8,6 @@ import scipy.sparse as sp -def peak_alloc_mb(fn: Callable[[], Any]) -> float: - """Peak memory allocated during ``fn``, in MB. - - Not RSS, which is a process-lifetime high-water mark and so reports zero - for anything staying under the peak set while building its input. numpy - allocations are traced, numba's internal ones are not. - """ - fn() # JIT compile / warm caches outside the measurement - tracemalloc.start() - try: - before = tracemalloc.get_traced_memory()[0] - tracemalloc.reset_peak() - fn() - peak = tracemalloc.get_traced_memory()[1] - finally: - tracemalloc.stop() - return max(0.0, (peak - before) / 1e6) - - def best_time(fn: Callable[[], Any], repeat: int = 7) -> float: """Best wall-clock time over ``repeat`` runs, in seconds. Warms up first.""" fn() # JIT compile / allocate caches outside the measurement diff --git a/tests/conftest.py b/tests/conftest.py index 01af3db..7e0cce0 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -38,16 +38,9 @@ def csr(dense) -> sp.csr_array: return sp.csr_array(dense) -#: Threads the memory-limited tests run their kernels with. -#: -#: The thread-local accumulators those tests bound are -#: ``nthreads * n_minor * width * 8`` bytes, so their size follows -#: ``numba.get_num_threads()`` -- which is a property of the machine, not of -#: the code under test. Left free, the same assertion would mean something -#: different on a 4-core runner than on a 96-core one, and a limit loose -#: enough for the widest machine would be too loose to catch a regression on -#: any of them. Pinning it makes the expected allocation a number the test -#: can actually state. +#: Threads the memory-limited tests run their kernels with. Their accumulators +#: are ``nthreads * n_minor * width * 8`` bytes, so an unpinned count would make +#: the same ceiling mean something different on every machine. MEMORY_TEST_THREADS = 4 diff --git a/tests/test_minor_axis_matmul.py b/tests/test_minor_axis_matmul.py index 4e9f4b4..f8977e0 100644 --- a/tests/test_minor_axis_matmul.py +++ b/tests/test_minor_axis_matmul.py @@ -14,7 +14,7 @@ import pytest import scipy.sparse as sp -from vsparse import VCSCArray, VCSRArray +from vsparse import RECIPES, VCSCArray, VCSRArray from vsparse._ops import accumulator_threads @@ -94,31 +94,27 @@ def test_no_second_copy_of_the_array_is_built(vcls, rng): @pytest.fixture(scope="module") def misaligned_setup(): - """A 1_200 x 400 VCSC view and a width-2 operand, JIT already warmed. - - In a fixture so that neither the array nor the one-off compilation counts - against the ceiling below -- `limit_memory` measures the test body only. - """ + """A 1_200 x 400 VCSC view and a width-2 operand, JIT warmed, built outside the body.""" rng = np.random.default_rng(0) dense = rng.integers(1, 5, size=(1_200, 400)).astype(np.float64) v = VCSCArray.from_scipy(sp.csc_array(dense)) B = rng.normal(size=(dense.shape[1], 2)) - v.normalized() @ B # warm up the JIT + for warm in RECIPES: + v.normalized(warm) @ B # warm up the JIT for every recipe return v, B, dense -# 1_200 x 400 with no zeros is 480_000 nonzeros: 3.8 MB of values and 1.9 MB of -# indices. The accumulator block is `MEMORY_TEST_THREADS * 1_200 * 2 * 8` = -# 77 KB, plus the 19 KB output. A ceiling of 256 KB is ~2x that and ~15x under -# the values array, so it fails if this path ever goes back to building a -# second copy of the array and passes on any runner. +# 480_000 nonzeros: 3.8 MB of values, 1.9 MB of indices. The accumulator block +# is `MEMORY_TEST_THREADS * 1_200 * 2 * 8` = 77 KB plus a 19 KB output, so this +# ceiling is ~2x that and ~15x under the values array. +@pytest.mark.parametrize("recipe", sorted(RECIPES)) @pytest.mark.limit_memory("256 KB") def test_misaligned_matmul_peak_is_bounded_by_the_accumulator_budget( - pinned_threads, misaligned_setup + pinned_threads, misaligned_setup, recipe ): - """Peak memory tracks the (fixed) accumulator budget, not the size of the array.""" + """Peak memory tracks the accumulator budget, not the size of the array.""" v, B, _ = misaligned_setup - out = v.normalized() @ B + out = v.normalized(recipe) @ B assert out.shape == (1_200, 2) @@ -133,20 +129,3 @@ def test_accumulator_threads_degrades_to_one_for_a_huge_output_axis(): huge_axis = 2_000_000 wide_b = 200 # e.g. rank + oversampling in a randomized SVD assert accumulator_threads(huge_axis, bytes_per_element=8 * wide_b) == 1 - - -# `limit_leaks` fails when any single call stack still holds memory once the -# body returns, which is the shape of a cache that grows with use rather than -# of a big one-off allocation -- the bug d286cf3 fixed, where the normalization -# cache pinned O(nnz) duals. `limit_memory` cannot see it: each pass on its own -# stays under any sane ceiling, and only the accumulation across passes is -# wrong. It traces native stacks for every allocation and so is markedly -# slower than the ceilings above, which is why there is one of these and not -# one per operation. -@pytest.mark.limit_leaks("128 KB") -def test_repeated_matmul_on_one_view_retains_nothing(pinned_threads, misaligned_setup): - """Iterating on a view must not accumulate: every pass frees what it took.""" - v, B, _ = misaligned_setup - nv = v.normalized() - for _ in range(8): - nv @ B diff --git a/tests/test_reduction_memory.py b/tests/test_reduction_memory.py index d0b57e8..64c35bd 100644 --- a/tests/test_reduction_memory.py +++ b/tests/test_reduction_memory.py @@ -1,16 +1,10 @@ -"""Reductions: correctness on the minor axis, and the cost of getting there. - -The memory assertions here are `pytest-memray` ceilings rather than measured -numbers. What is being claimed is structural -- a reduction producing an -`n_minor`-sized result must not allocate anything that grows with `nnz` -- so -a ceiling well under nnz-scale states it directly, where a recorded figure -would only show it drifting. - -Two conventions make the ceilings mean the same thing on every machine: -`pinned_threads` fixes the thread count the accumulators are sized by, and the -arrays are built in module-scoped fixtures because a `limit_memory` mark -measures the test body alone. See `benchmarks/README.md` for how this divides -with the benchmark suite, which records memory rather than bounding it. +"""Minor-axis reductions: correctness, and ceilings on what they allocate. + +The ``pytest-memray`` ceilings here assert a structural property -- a reduction +producing an ``n_minor``-sized result must not allocate anything that grows +with ``nnz``. They rely on ``pinned_threads`` to fix the thread count the +accumulators are sized by, and on module-scoped fixtures for the inputs, since +a ``limit_memory`` mark measures the test body alone. """ from __future__ import annotations @@ -93,12 +87,7 @@ def test_accumulator_block_stays_within_budget(n_minor, bytes_per_element): @pytest.fixture(scope="module") def reduction_array(): - """A 2_000 x 500 array, with every reduction's JIT already warmed. - - Built in a fixture rather than in the test body because ``limit_memory`` - measures only the body -- so the array itself, and the one-off compilation - of the kernels that touch it, stay out of the number being bounded. - """ + """A 2_000 x 500 array with every reduction's JIT warmed, built outside the body.""" rng = np.random.default_rng(0) dense = rng.integers(1, 5, size=(2_000, 500)).astype(np.float64) v = VCSRArray.from_scipy(sp.csr_array(dense)) @@ -107,17 +96,13 @@ def reduction_array(): return v -# Reducing 2_000 x 500 over the minor axis touches 1e6 nonzeros: 8 MB of -# values and 4 MB of indices. The accumulator block is -# `MEMORY_TEST_THREADS * 500 * 8` = 16 KB (32 KB for the extrema kernels, which -# carry two). The ceilings below sit two orders of magnitude under anything -# nnz-sized and roughly 2x over the block, so they catch a reduction that -# starts scaling with nnz without tripping on allocator noise. +# 1e6 nonzeros here: 8 MB of values, 4 MB of indices. The accumulator block is +# `MEMORY_TEST_THREADS * 500 * 8` = 16 KB, doubled for the extrema kernels, +# which carry two. The ceilings sit ~2x over that and orders of magnitude under +# nnz-scale. # -# The ceiling rides on each `pytest.param` rather than being applied inside the -# test: `pytest-memray` reads the marker when the test is collected, so a -# marker added from the body (`request.applymarker`) is never seen and the -# test silently asserts nothing. +# Marks go on each `pytest.param`: pytest-memray reads them at collection, so +# one applied from the test body is never seen and asserts nothing. @pytest.mark.parametrize( ("label", "call"), [ @@ -132,3 +117,23 @@ def test_minor_axis_reductions_allocate_nothing_nnz_sized( """An n_minor-sized result must not cost nnz-sized scratch.""" out = call(reduction_array) assert out.shape == (500,) + + +@pytest.fixture(scope="module") +def selection_array(): + """A 2_000 x 500 array with the selection kernels' JIT warmed.""" + rng = np.random.default_rng(0) + dense = rng.integers(1, 5, size=(2_000, 500)).astype(np.float64) + v = VCSRArray.from_scipy(sp.csr_array(dense)) + v[:, np.arange(0, 500, 2)] + return v + + +# A selection's result does grow with what was selected: half of 1e6 nonzeros +# is ~2.3 MB of output, which is the answer rather than scratch. The ceiling +# sits above that and well below output-plus-an-nnz-sized-temporary. +@pytest.mark.limit_memory("4 MB") +def test_minor_axis_selection_allocates_no_nnz_sized_scratch(pinned_threads, selection_array): + """A selection may pay for its output, but not for a copy of the input.""" + out = selection_array[:, np.arange(0, 500, 2)] + assert out.shape == (2_000, 250)