Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
58 changes: 58 additions & 0 deletions CLAUDE.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,58 @@
# CLAUDE.md

Guidance for AI coding agents (and human contributors) working in this repository.

## Maintainability checks for new/changed code

This project's dev dependency group (`uv sync --group dev`) includes several
static-analysis tools that are not wired into pre-commit/CI, so they won't
run automatically. Whenever you add or substantially modify Python code in
`src/vsparse/`, run the relevant tool(s) below yourself and address anything
they flag before considering the change done.

### Dead code -- vulture

```sh
uv run vulture src/vsparse/
```

Flags functions, variables, and imports that appear to be unused. Before
deleting a reported item, confirm with `grep`/`git grep` that it really has
no callers (vulture's confidence score is a heuristic, and a few things --
e.g. manually-invoked one-off scripts -- are meant to have no in-repo
callers). If something is a deliberate exception, prefer leaving a short
comment explaining why over silencing the tool.

### Duplicated code -- jscpd

```sh
npx jscpd src/vsparse/ --min-lines 5 --min-tokens 50
```

(No install needed beyond Node/npx; jscpd is not a Python dependency.) Flags
copy-pasted blocks. If a new module duplicates an existing block of 10+
lines, prefer extracting a shared helper instead of copy-pasting.

### Cyclomatic complexity -- radon + xenon

```sh
uv run radon cc src/vsparse/ -n C -s # list functions ranked C or worse
uv run radon mi src/vsparse/ -s # maintainability index per file
uv run xenon --max-absolute B --max-modules A --max-average A src/vsparse/
```

`xenon` exits non-zero and prints every function/module exceeding the given
rank thresholds (A best -- F worst). Treat a new function ranked C or worse
as a signal to break it up: extract the branchy/loop-heavy interior into one
or more named helper functions (as opposed to introducing more parameters or
flags to the same function). A handful of pre-existing functions still
exceed these thresholds; it's fine to leave those alone unless you're
already modifying them, but don't add new ones.

## Why these aren't pre-commit hooks

`ruff` and `codespell` are fast and have an unambiguous pass/fail; they run
on every commit. `vulture`, `jscpd`, and `xenon` are noisier and require
judgment calls (a flagged function may be an intentional exception), so
they're kept as tools to run and reason about manually rather than hard
commit gates.
6 changes: 2 additions & 4 deletions benchmarks/baselines.json
Original file line number Diff line number Diff line change
Expand Up @@ -95,12 +95,10 @@
"time_ratio_vs_scipy": 1.2909
},
"misaligned_matmat_vs_scipy": {
"time_ratio_vs_scipy": 2.0436,
"peak_alloc_mb": 38.4009
"time_ratio_vs_scipy": 2.0436
},
"misaligned_rmatmat_vs_scipy": {
"time_ratio_vs_scipy": 1.0817,
"peak_alloc_mb": 20.2251
"time_ratio_vs_scipy": 1.0817
}
}
}
10 changes: 2 additions & 8 deletions benchmarks/cases.py
Original file line number Diff line number Diff line change
Expand Up @@ -308,10 +308,7 @@ def misaligned_matmat_vs_scipy() -> dict[str, float]:
v = VCSCArray.from_scipy(mat)
csc = sp.csc_array(mat)
B = np.random.default_rng(0).normal(size=(mat.shape[1], 8))
return {
"time_ratio_vs_scipy": ratio_vs_scipy(lambda: v @ B, lambda: csc @ B),
"peak_alloc_mb": peak_alloc_mb(lambda: v @ B),
}
return {"time_ratio_vs_scipy": ratio_vs_scipy(lambda: v @ B, lambda: csc @ B)}


@fast
Expand All @@ -325,10 +322,7 @@ def misaligned_rmatmat_vs_scipy() -> dict[str, float]:
v = VCSRArray.from_scipy(mat)
csr = sp.csr_array(mat)
B = np.random.default_rng(0).normal(size=(8, mat.shape[0]))
return {
"time_ratio_vs_scipy": ratio_vs_scipy(lambda: B @ v, lambda: B @ csr),
"peak_alloc_mb": peak_alloc_mb(lambda: B @ v),
}
return {"time_ratio_vs_scipy": ratio_vs_scipy(lambda: B @ v, lambda: B @ csr)}


# -- larger, for the scheduled job -------------------------------------------
Expand Down
3 changes: 3 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,9 @@ dev = [
"hypothesis>=6.100",
"codespell>=2.3",
"pytest-memray>=1.11.0",
"vulture>=2.16",
"radon>=6.0.1",
"xenon>=0.9.3",
]

[build-system]
Expand Down
140 changes: 80 additions & 60 deletions src/vsparse/_anndata_class.py
Original file line number Diff line number Diff line change
Expand Up @@ -70,6 +70,17 @@ def _subset_2d(v: Any, oidx: Any, vidx: Any) -> Any:
return np.asarray(v)[oidx][:, vidx]


def _coerce_vcs(value: Any, name: str) -> Any:
"""Coerce a raw scipy/ndarray ``X``/``raw_X`` value to VCSC/VCSR, validating anything else."""
if value is not None and not isinstance(value, _VCS_TYPES):
if sp.issparse(value) or isinstance(value, np.ndarray):
vcls = VCSCArray if isinstance(value, sp.csc_array | sp.csc_matrix) else VCSRArray
value = vcls.from_scipy(value)
else:
_check_vcs_type(value, name)
return value


def _copy_value(v: Any) -> Any:
"""A deep-enough copy of one obs/var/obsm/varm/obsp/varp/layers value.

Expand All @@ -80,6 +91,23 @@ def _copy_value(v: Any) -> Any:
return None if v is None else v.copy()


def _filtered(mapping: Mapping[Any, Any]) -> dict[Any, Any]:
"""A plain dict of ``mapping``, dropping the ``None`` key anndata sometimes carries."""
return {k: v for k, v in mapping.items() if k is not None}


def _map_subset_1d(mapping: Mapping[Any, Any], idx: Any) -> dict[Any, Any]:
return {k: _subset_1d(v, idx) for k, v in mapping.items() if k is not None}


def _map_subset_2d(mapping: Mapping[Any, Any], oidx: Any, vidx: Any) -> dict[Any, Any]:
return {k: _subset_2d(v, oidx, vidx) for k, v in mapping.items() if k is not None}


def _map_copy(mapping: Mapping[Any, Any]) -> dict[Any, Any]:
return {k: _copy_value(v) for k, v in mapping.items() if k is not None}


def _check_vcs_type(value: Any, name: str) -> None:
if value is not None and not isinstance(value, _VCS_TYPES):
raise TypeError(
Expand Down Expand Up @@ -146,12 +174,7 @@ def X(self) -> _AnyVCS | None:

@X.setter
def X(self, value: Any) -> None:
if value is not None and not isinstance(value, _VCS_TYPES):
if sp.issparse(value) or isinstance(value, np.ndarray):
vcls = VCSCArray if isinstance(value, sp.csc_array | sp.csc_matrix) else VCSRArray
value = vcls.from_scipy(value)
else:
_check_vcs_type(value, "X")
value = _coerce_vcs(value, "X")
if (
value is not None
and hasattr(self, "_obs")
Expand All @@ -168,13 +191,7 @@ def raw_X(self) -> _AnyVCS | None:

@raw_X.setter
def raw_X(self, value: Any) -> None:
if value is not None and not isinstance(value, _VCS_TYPES):
if sp.issparse(value) or isinstance(value, np.ndarray):
vcls = VCSCArray if isinstance(value, sp.csc_array | sp.csc_matrix) else VCSRArray
value = vcls.from_scipy(value)
else:
_check_vcs_type(value, "raw_X")
self._vcs_raw_X = value
self._vcs_raw_X = _coerce_vcs(value, "raw_X")

# -- indexing / view creation ---------------------------------------------

Expand Down Expand Up @@ -219,11 +236,11 @@ def __getitem__(self, index: Any) -> VCSCAnnData: # ty: ignore[invalid-method-o
obs=obs,
var=var,
uns=uns,
obsm={k: _subset_1d(v, oidx) for k, v in self.obsm.items() if k is not None},
varm={k: _subset_1d(v, vidx) for k, v in self.varm.items() if k is not None},
obsp={k: _subset_2d(v, oidx, oidx) for k, v in self.obsp.items() if k is not None},
varp={k: _subset_2d(v, vidx, vidx) for k, v in self.varp.items() if k is not None},
layers={k: _subset_2d(v, oidx, vidx) for k, v in self.layers.items() if k is not None},
obsm=_map_subset_1d(self.obsm, oidx),
varm=_map_subset_1d(self.varm, vidx),
obsp=_map_subset_2d(self.obsp, oidx, oidx),
varp=_map_subset_2d(self.varp, vidx, vidx),
layers=_map_subset_2d(self.layers, oidx, vidx),
)

def copy(self) -> VCSCAnnData: # ty: ignore[invalid-method-override]
Expand All @@ -244,11 +261,11 @@ def copy(self) -> VCSCAnnData: # ty: ignore[invalid-method-override]
obs=cast(pd.DataFrame, self.obs).copy(),
var=cast(pd.DataFrame, self.var).copy(),
uns=_copy.deepcopy(dict(self.uns)),
obsm={k: _copy_value(v) for k, v in self.obsm.items() if k is not None},
varm={k: _copy_value(v) for k, v in self.varm.items() if k is not None},
obsp={k: _copy_value(v) for k, v in self.obsp.items() if k is not None},
varp={k: _copy_value(v) for k, v in self.varp.items() if k is not None},
layers={k: _copy_value(v) for k, v in self.layers.items() if k is not None},
obsm=_map_copy(self.obsm),
varm=_map_copy(self.varm),
obsp=_map_copy(self.obsp),
varp=_map_copy(self.varp),
layers=_map_copy(self.layers),
)

def to_memory(self, *, copy: bool = False) -> VCSCAnnData:
Expand All @@ -271,6 +288,33 @@ def to_memory(self, *, copy: bool = False) -> VCSCAnnData:

# -- normalization ----------------------------------------------------------

def _normalized_from_stored(self, recipe: Recipe) -> Any:
"""Rebuild a normalized view from ``obs``/``varm``/``uns`` if they match ``recipe``, else ``None``."""
stored = self.uns.get(_VSPARSE_UNS_KEY)
if not (
stored is not None
and stored.get("recipe") == recipe.name
and _VSPARSE_OBS_A in self.obs
and len(self.obs[_VSPARSE_OBS_A]) == self.n_obs
and _VSPARSE_VARM_B in self.varm
and _VSPARSE_VARM_C in self.varm
and _VSPARSE_VARM_S in self.varm
and len(self.varm[_VSPARSE_VARM_B]) == self.n_vars
):
return None
nview_cls = (
VCSCArrayNormalized if isinstance(self._vcs_X, VCSCArray) else VCSRArrayNormalized
)
return nview_cls.from_stats(
self._vcs_X,
recipe,
a=np.asarray(self.obs[_VSPARSE_OBS_A], dtype=np.float64),
b=np.asarray(self.varm[_VSPARSE_VARM_B], dtype=np.float64).reshape(-1),
c=np.asarray(self.varm[_VSPARSE_VARM_C], dtype=np.float64).reshape(-1),
s=np.asarray(self.varm[_VSPARSE_VARM_S], dtype=np.float64).reshape(-1),
stale=bool(stored.get("stale", False)),
)

def normalized(self, view: str | Recipe = DEFAULT_RECIPE, *, recalculate: bool = True) -> Any:
"""A normalized view of ``X`` -- see :meth:`vsparse._base._VCSBase.normalized`.

Expand Down Expand Up @@ -299,31 +343,8 @@ def normalized(self, view: str | Recipe = DEFAULT_RECIPE, *, recalculate: bool =
cached = cache.get(recipe, self._vcs_X)
if cached is not None:
return cached
stored = self.uns.get(_VSPARSE_UNS_KEY)
if (
stored is not None
and stored.get("recipe") == recipe.name
and _VSPARSE_OBS_A in self.obs
and len(self.obs[_VSPARSE_OBS_A]) == self.n_obs
and _VSPARSE_VARM_B in self.varm
and _VSPARSE_VARM_C in self.varm
and _VSPARSE_VARM_S in self.varm
and len(self.varm[_VSPARSE_VARM_B]) == self.n_vars
):
nview_cls = (
VCSCArrayNormalized
if isinstance(self._vcs_X, VCSCArray)
else VCSRArrayNormalized
)
nview = nview_cls.from_stats(
self._vcs_X,
recipe,
a=np.asarray(self.obs[_VSPARSE_OBS_A], dtype=np.float64),
b=np.asarray(self.varm[_VSPARSE_VARM_B], dtype=np.float64).reshape(-1),
c=np.asarray(self.varm[_VSPARSE_VARM_C], dtype=np.float64).reshape(-1),
s=np.asarray(self.varm[_VSPARSE_VARM_S], dtype=np.float64).reshape(-1),
stale=bool(stored.get("stale", False)),
)
nview = self._normalized_from_stored(recipe)
if nview is not None:
cache.put(recipe, nview)
return nview

Expand Down Expand Up @@ -360,11 +381,11 @@ def from_anndata(
obs=cast(pd.DataFrame, adata.obs).copy(),
var=cast(pd.DataFrame, adata.var).copy(),
uns=adata.uns,
obsm={k: v for k, v in adata.obsm.items() if k is not None},
varm={k: v for k, v in adata.varm.items() if k is not None},
obsp={k: v for k, v in adata.obsp.items() if k is not None},
varp={k: v for k, v in adata.varp.items() if k is not None},
layers={k: v for k, v in adata.layers.items() if k is not None},
obsm=_filtered(adata.obsm),
varm=_filtered(adata.varm),
obsp=_filtered(adata.obsp),
varp=_filtered(adata.varp),
layers=_filtered(adata.layers),
)

def to_anndata(self) -> ad.AnnData:
Expand All @@ -376,11 +397,11 @@ def to_anndata(self) -> ad.AnnData:
obs=obs.copy(),
var=var.copy(),
uns=self.uns,
obsm=cast(Any, {k: v for k, v in self.obsm.items() if k is not None}),
varm=cast(Any, {k: v for k, v in self.varm.items() if k is not None}),
obsp=cast(Any, {k: v for k, v in self.obsp.items() if k is not None}),
varp=cast(Any, {k: v for k, v in self.varp.items() if k is not None}),
layers=cast(Any, {k: v for k, v in self.layers.items() if k is not None}),
obsm=cast(Any, _filtered(self.obsm)),
varm=cast(Any, _filtered(self.varm)),
obsp=cast(Any, _filtered(self.obsp)),
varp=cast(Any, _filtered(self.varp)),
layers=cast(Any, _filtered(self.layers)),
)
if self._vcs_raw_X is not None:
out.raw = ad.AnnData(X=self._vcs_raw_X.to_scipy(), obs=obs.copy(), var=var.copy())
Expand Down Expand Up @@ -416,8 +437,7 @@ def _write_group(
for key in _DF_KEYS:
ad.io.write_elem(g, key, getattr(self, key), dataset_kwargs=dataset_kwargs)
for key in _MAPPING_KEYS:
mapping = {k: v for k, v in getattr(self, key).items() if k is not None}
ad.io.write_elem(g, key, mapping, dataset_kwargs=dataset_kwargs)
ad.io.write_elem(g, key, _filtered(getattr(self, key)), dataset_kwargs=dataset_kwargs)
g.attrs["encoding-type"] = "anndata"
g.attrs["encoding-version"] = "0.1.0"

Expand Down
18 changes: 0 additions & 18 deletions src/vsparse/_base.py
Original file line number Diff line number Diff line change
Expand Up @@ -551,24 +551,6 @@ def _select_major(self, key: Any) -> _VCSBase:
new_shape = (n_minor, idx.shape[0]) if self._format == "csc" else (idx.shape[0], n_minor)
return type(self)(new_shape, new_major_ptr, new_values, new_value_ptr, new_indices)

def _major_range(self, start: int, stop: int) -> _VCSBase:
"""The contiguous major-slice range ``[start, stop)``, without copying values.

``values``/``indices`` come back as views; only the two pointer
arrays are rebuilt, rebased to the new start.
"""
u0, u1 = int(self.major_ptr[start]), int(self.major_ptr[stop])
k0, k1 = int(self.value_ptr[u0]), int(self.value_ptr[u1])
n_sel = stop - start
new_shape = (self.n_minor, n_sel) if self._format == "csc" else (n_sel, self.n_minor)
return type(self)(
new_shape,
self.major_ptr[start : stop + 1] - u0,
self.values[u0:u1],
self.value_ptr[u0 : u1 + 1] - k0,
self.indices[k0:k1],
)

def _select_minor(self, key: Any) -> _VCSBase:
"""Select along the minor axis (rows for VCSC, columns for VCSR).

Expand Down
Loading
Loading