From d445ff3da3bcb5e53662cbca612f2f58d42d311c Mon Sep 17 00:00:00 2001 From: wpbonelli Date: Tue, 6 Oct 2026 06:12:17 -0700 Subject: [PATCH 1/6] feat(dfns): expose shape element parsing Add ShapeRef, parse_shape_element and resolve_shape_ref, and rebuild the Array/List shape validators on them so there's one implementation of the shape grammar. Also export split_bound. Co-Authored-By: Claude Opus 5.5 --- autotest/dfns/test_schema_dimensions.py | 150 ++++++++++ docs/md/dfns.md | 39 +++ docs/md/dfnspec.md | 2 + modflow_devtools/dfns/__init__.py | 8 + modflow_devtools/dfns/schema.py | 358 +++++++++++++----------- 5 files changed, 398 insertions(+), 159 deletions(-) diff --git a/autotest/dfns/test_schema_dimensions.py b/autotest/dfns/test_schema_dimensions.py index bb42ff2e..72685119 100644 --- a/autotest/dfns/test_schema_dimensions.py +++ b/autotest/dfns/test_schema_dimensions.py @@ -16,6 +16,7 @@ Package, Record, RuntimeDim, + ShapeRef, String, _names_in_expr, _resolve_derived_dims, @@ -23,6 +24,8 @@ _validate_list_shape_element, _validate_shape_element, _validate_sum_call, + parse_shape_element, + resolve_shape_ref, split_bound, ) @@ -531,6 +534,153 @@ def test_validate_shape_element_fk_block_mismatch(): _validate_shape_element("packagedata.nlakeconn(lakeno)", arr, lak, enc, known) +@pytest.mark.parametrize( + "element,expected", + [ + ("nvert", ShapeRef("dim", "nvert")), + ("nseg-1", ShapeRef("dim", "nseg", offset=-1)), + ("ncol + 1", ShapeRef("dim", "ncol", offset=1)), + ("<=maxats", ShapeRef("dim", "maxats", bound="<=")), + (">= nper", ShapeRef("dim", "nper", bound=">=")), + ("= nper")) == ">=nper" + assert str(parse_shape_element("gwf-x.block.col(fk)")) == "gwf-x.block.col(fk)" + + +def test_resolve_shape_ref_dim(): + arr = Array(name="arr", dtype="double", shape=[]) + ref = resolve_shape_ref(parse_shape_element("<=nseg-1"), arr, known_dims={"nseg"}) + assert ref == ShapeRef("dim", "nseg", offset=-1, bound="<=") + + +def test_resolve_shape_ref_sibling(): + arr = Array(name="icvert", dtype="integer", shape=[]) + enc = Record(name="item", fields={"ncvert": Integer(name="ncvert"), "icvert": arr}) + ref = resolve_shape_ref( + parse_shape_element("ncvert"), arr, known_dims=set(), enclosing_record=enc + ) + assert ref.kind == "sibling" + assert ref.name == "ncvert" + + +def test_resolve_shape_ref_dim_shadows_sibling(): + """A name that is both a dim and a sibling Integer resolves to the dim.""" + arr = Array(name="icvert", dtype="integer", shape=[]) + enc = Record(name="item", fields={"ncvert": Integer(name="ncvert"), "icvert": arr}) + ref = resolve_shape_ref( + parse_shape_element("ncvert"), arr, known_dims={"ncvert"}, enclosing_record=enc + ) + assert ref.kind == "dim" + + +def test_resolve_shape_ref_non_integer_sibling(): + arr = Array(name="vals", dtype="double", shape=[]) + enc = Record(name="item", fields={"n": String(name="n"), "vals": arr}) + with pytest.raises(ValueError, match="does not resolve to a known dim"): + resolve_shape_ref(parse_shape_element("n"), arr, known_dims=set(), enclosing_record=enc) + + +def test_resolve_shape_ref_unknown_dim(): + arr = Array(name="arr", dtype="double", shape=[]) + with pytest.raises(ValueError, match="'nseg' does not resolve to a known dim"): + resolve_shape_ref(parse_shape_element("nseg-1"), arr, known_dims=set()) + + +def test_resolve_shape_ref_lookup(): + arr, enc, pkg, known = _lookup_ctx() + ref = parse_shape_element("packagedata.nlakeconn(lakeno)") + resolved = resolve_shape_ref(ref, arr, known_dims=known, component=pkg, enclosing_record=enc) + assert resolved == ref + + +def _cross_component_lookup_ctx(): + """An array in another component sized by a column of gwf-lak's packagedata.""" + _arr, _enc, lak, _known = _lookup_ctx() + gwf = Model(name="gwf-nam", blocks=None) + spec = Dfns(components={"gwf-nam": gwf, "gwf-lak": lak}) + fk_lakeno = Integer(name="lakeno", fk="packagedata.lakeno") + arr = Array(name="outflow", dtype="double", shape=[]) + enc = Record(name="item", fields={"lakeno": fk_lakeno, "outflow": arr}) + other = Package(name="gwf-other", parent="gwf-nam", blocks=None) + return arr, enc, other, spec + + +def test_resolve_shape_ref_cross_component_lookup(): + arr, enc, other, spec = _cross_component_lookup_ctx() + ref = parse_shape_element("gwf-lak.packagedata.nlakeconn(lakeno)") + resolved = resolve_shape_ref( + ref, arr, known_dims=set(), component=other, enclosing_record=enc, spec=spec + ) + assert resolved == ref + + +def test_resolve_shape_ref_cross_component_lookup_requires_spec(): + arr, enc, other, _spec = _cross_component_lookup_ctx() + ref = parse_shape_element("gwf-lak.packagedata.nlakeconn(lakeno)") + with pytest.raises(ValueError, match="requires a Dfns spec"): + resolve_shape_ref(ref, arr, known_dims=set(), component=other, enclosing_record=enc) + + +def test_resolve_shape_ref_cross_component_lookup_unknown_component(): + arr, enc, other, spec = _cross_component_lookup_ctx() + ref = parse_shape_element("gwf-nope.packagedata.nlakeconn(lakeno)") + with pytest.raises(ValueError, match="not found in spec"): + resolve_shape_ref( + ref, arr, known_dims=set(), component=other, enclosing_record=enc, spec=spec + ) + + +def test_resolve_shape_ref_list_lookup(): + lst = List(name="stress_period_data", item=Record(name="item", fields={})) + with pytest.raises(ValueError, match="invalid shape element"): + resolve_shape_ref(parse_shape_element("packagedata.ncon(ifno)"), lst, known_dims=set()) + + +def test_validate_list_shape_element_lookup(): + lst = List(name="stress_period_data", item=Record(name="item", fields={})) + with pytest.raises(ValueError, match="invalid shape element"): + _validate_list_shape_element("packagedata.ncon(ifno)", lst, {"maxbound"}) + + def test_local_dims(): block = _dim_block("nlay", "nrow", "ncol") pkg = Package( diff --git a/docs/md/dfns.md b/docs/md/dfns.md index b9baaecd..cc56daf3 100644 --- a/docs/md/dfns.md +++ b/docs/md/dfns.md @@ -203,6 +203,45 @@ Available field types: See [DFN specification](dfnspec.md) for full attribute documentation. +### Parsing shape elements + +`Array.shape` and `List.shape` elements are strings in a small grammar (see [Dimensions](dfnspec.md#dimensions) and [Bounds](dfnspec.md#bounds)). `parse_shape_element` parses one into a `ShapeRef`: + +```python +from modflow_devtools.dfns import ShapeRef, parse_shape_element + +parse_shape_element("nseg-1") +# ShapeRef(kind="dim", name="nseg", offset=-1) +parse_shape_element("<=maxats") +# ShapeRef(kind="dim", name="maxats", bound="<=") +parse_shape_element("packagedata.ncon(ifno)") +# ShapeRef(kind="lookup", name="ncon", block="packagedata", fk_field="ifno") +``` + +`kind` is `"dim"` for a name, with an optional arithmetic `offset`, and `"lookup"` for a row-level column lookup, which also sets `component` (for a cross-component lookup), `block` and `fk_field`. `bound` is the inequality operator, or `None` for an exact extent. `str(ref)` gives the element back in canonical form. A malformed element raises `ValueError`. + +Parsing can't tell whether a name means a dim or a sibling Integer in the same record (e.g. `ncvert` sizing `icvert`). `resolve_shape_ref` decides, and checks the reference against its context: + +```python +from modflow_devtools.dfns import resolve_shape_ref + +sfr = spec.components["gwf-sfr"] +connectiondata = sfr.blocks["connectiondata"].fields["connectiondata"] +ic = connectiondata.item.fields["ic"] +resolve_shape_ref( + parse_shape_element(ic.shape[0]), + ic, + known_dims=spec.input_dims("gwf-sfr"), + component=sfr, + enclosing_record=connectiondata.item, + spec=spec, +) +``` + +A name in `known_dims` stays a `"dim"`. Otherwise, a name that is an Integer subfield of `enclosing_record` becomes a `"sibling"`. A lookup must be in an array inside a record, and name an Integer column in a list block, selected by a sibling whose `fk` references that block. Anything that doesn't resolve raises `ValueError`. Loading `Dfns` validates every shape this way. + +`split_bound` splits off just the bound: `split_bound("<=maxbound")` returns `("<=", "maxbound")`. + ### Rendering block templates `Block.render()` produces a `BEGIN/END` template string showing the structure of a block — the same format used in the MODFLOW 6 user guide and in tooling such as IDE hover text: diff --git a/docs/md/dfnspec.md b/docs/md/dfnspec.md index f37e44bd..90d38cde 100644 --- a/docs/md/dfnspec.md +++ b/docs/md/dfnspec.md @@ -758,6 +758,8 @@ A shape expression gives an exact extent. Prefixed with one of the inequality op An unprefixed extent is exact, so a DFN must mark every bound. A consumer may reject input that doesn't satisfy the relation: too many or too few rows for an exact extent, too many for an upper bound, too few for a lower bound. +`modflow_devtools.dfns` parses shape expressions with `parse_shape_element` and resolves them against their component with `resolve_shape_ref`; see [Parsing shape elements](dfns.md#parsing-shape-elements). + #### Dimension scope Dimensions may specify a `scope` attribute controlling which other components can inherit the dimension: diff --git a/modflow_devtools/dfns/__init__.py b/modflow_devtools/dfns/__init__.py index 46a64c94..a29fe8f0 100644 --- a/modflow_devtools/dfns/__init__.py +++ b/modflow_devtools/dfns/__init__.py @@ -29,9 +29,13 @@ Record, RuntimeDim, Scalar, + ShapeRef, Simulation, String, Union, + parse_shape_element, + resolve_shape_ref, + split_bound, ) # Experimental API warning @@ -73,9 +77,13 @@ "RemoteDfnRegistry", "RuntimeDim", "Scalar", + "ShapeRef", "Simulation", "String", "Union", "fetch_dfns", "migrate", + "parse_shape_element", + "resolve_shape_ref", + "split_bound", ] diff --git a/modflow_devtools/dfns/schema.py b/modflow_devtools/dfns/schema.py index ac785f7a..8e78eacf 100644 --- a/modflow_devtools/dfns/schema.py +++ b/modflow_devtools/dfns/schema.py @@ -1,6 +1,7 @@ import ast import re from collections.abc import Callable, Mapping +from dataclasses import dataclass, replace from os import PathLike from pathlib import Path from typing import Annotated, Any, Literal, cast @@ -997,7 +998,17 @@ class Package(ComponentBase): _LEN_CALL_RE = re.compile(r"^len\([A-Za-z_]\w*\)$") _LOOKUP_RE = re.compile(r"^(?:([\w-]+)\.)?(\w+)\.(\w+)\((\w+)\)$") _BOUND_RE = re.compile(r"^[<>]=?") -_ARITH_RE = re.compile(r"^([A-Za-z_]\w*)\s*[+-]\s*\d+$") +_ARITH_RE = re.compile(r"^([A-Za-z_]\w*)\s*([+-])\s*(\d+)$") + +_SHAPE_FORMS = ( + "must be a dim reference (^[A-Za-z_]\\w*$), an arithmetic offset " + "(dim [+-] integer), or a row-level lookup (block.column(fk_field)), " + "optionally prefixed by a bound (<, <=, >, >=)" +) +_LIST_SHAPE_FORMS = ( + "must be a dim reference (^[A-Za-z_]\\w*$), an arithmetic offset " + "(dim [+-] integer), optionally prefixed by a bound (<, <=, >, >=)" +) def split_bound(element: str) -> "tuple[str | None, str]": @@ -1013,6 +1024,79 @@ def split_bound(element: str) -> "tuple[str | None, str]": return None, element +@dataclass(frozen=True) +class ShapeRef: + """ + A parsed shape element; see :func:`parse_shape_element`. + + ``kind`` is ``"dim"`` for a name reference (optionally with an arithmetic + offset) and ``"lookup"`` for a row-level column lookup. Parsing can't tell + a dim from a sibling field; :func:`resolve_shape_ref` narrows a ``"dim"`` + naming an Integer subfield of the enclosing record to ``"sibling"``. + """ + + kind: Literal["dim", "sibling", "lookup"] + # the dim, sibling field, or looked-up column + name: str + # arithmetic offset: "nseg-1" -> -1 + offset: int = 0 + # "<", "<=", ">" or ">=", or None for an exact extent + bound: str | None = None + # lookup only: "[component.]block.column(fk_field)" + component: str | None = None + block: str | None = None + fk_field: str | None = None + + def __str__(self) -> str: + if self.kind == "lookup": + expr = f"{self.block}.{self.name}({self.fk_field})" + if self.component is not None: + expr = f"{self.component}.{expr}" + else: + expr = self.name + if self.offset: + expr += f"{self.offset:+d}" + return f"{self.bound or ''}{expr}" + + +class _ShapeSyntaxError(ValueError): + """A malformed shape element; ``reason`` says why.""" + + def __init__(self, element: str, reason: str): + super().__init__(f"invalid shape element {element!r}: {reason}") + self.reason = reason + + +def parse_shape_element(element: str) -> ShapeRef: + """ + Parse one element of an ``Array.shape`` or ``List.shape``. + + Purely syntactic: a name parses as a ``"dim"`` whether it names a dim or a + sibling field, and nothing is checked against a component. Use + :func:`resolve_shape_ref` for that. Raises ``ValueError`` if the element + is malformed. + """ + bound, core = split_bound(element) + if bound is not None and split_bound(core)[0] is not None: + raise _ShapeSyntaxError(element, "at most one bound operator is allowed") + if _DIM_RE.fullmatch(core): + return ShapeRef("dim", core, bound=bound) + if m := _ARITH_RE.fullmatch(core): + name, sign, n = m.groups() + return ShapeRef("dim", name, offset=int(sign + n), bound=bound) + if m := _LOOKUP_RE.fullmatch(core): + component_ref, block_name, col_name, fk_field_name = m.groups() + return ShapeRef( + "lookup", + col_name, + bound=bound, + component=component_ref, + block=block_name, + fk_field=fk_field_name, + ) + raise _ShapeSyntaxError(element, _SHAPE_FORMS) + + def _find_list_in_block(component: "ComponentBase", block_name: str) -> "List | None": """Return the first List field in the named block, or None.""" block = (component.blocks or {}).get(block_name) @@ -1024,146 +1108,126 @@ def _find_list_in_block(component: "ComponentBase", block_name: str) -> "List | return None -def _validate_shape_element( - element: str, - array_field: "Array", - component: "ComponentBase", - enclosing_record: "Record | None", +def resolve_shape_ref( + ref: ShapeRef, + field: "Array | List", + *, known_dims: set[str], + component: "ComponentBase | None" = None, + enclosing_record: "Record | None" = None, spec: "Dfns | None" = None, -) -> None: +) -> ShapeRef: """ - Validate one element of an Array.shape list. - - Valid forms: - - Dim reference ``^[A-Za-z_]\\w*$`` - Must resolve in the 3-level scope: explicit → derived → grid dims. - - Row-level column lookup ``^(\\w+)\\.(\\w+)\\((\\w+)\\)$`` - Structural checks (see plan §Shape element parsing). - - Arithmetic offset (dim [+-] integer) - - Any of the above prefixed with a bound operator (<, >, <=, >=) - - Raises ValueError on any violation. + Resolve a parsed element of ``field``'s shape against its context. + + A name resolves to a dim in ``known_dims`` (the dims visible to the + component, e.g. ``Dfns.input_dims()``), or failing that to an Integer + subfield of ``enclosing_record``, in which case the result's ``kind`` is + ``"sibling"``. A lookup must be in an array inside a record, and name an + Integer column of the list in a block of ``component`` (or, given + ``spec``, of another component), selected by a sibling whose ``fk`` + references that block. Lists aren't inside records, so a list's shape may + only reference dims. + + Returns ``ref`` with ``kind`` narrowed. Raises ``ValueError`` if it + doesn't resolve. """ - # a bound (<, >, <=, >=) may prefix any expression valid on its own - op, core = split_bound(element) - if op is not None: - if split_bound(core)[0] is not None: - raise ValueError( - f"Array {array_field.name!r} has invalid shape element {element!r}: " - f"at most one bound operator is allowed" - ) - _validate_shape_element(core, array_field, component, enclosing_record, known_dims, spec) - return - - if _DIM_RE.fullmatch(element): - if element in known_dims: - return - # Per-row varying shape: a sibling Integer with dimension="record" supplies - # an inline count on the same line. - if enclosing_record is not None: - sibling = enclosing_record.fields.get(element) - if isinstance(sibling, Integer): - return + where = f"{type(field).__name__} {field.name!r} shape element {str(ref)!r}" + + if ref.kind != "lookup": + if ref.name in known_dims: + return replace(ref, kind="dim") + # Per-row varying shape: a sibling Integer supplies an inline count on + # the same line. + if enclosing_record is not None and isinstance( + enclosing_record.fields.get(ref.name), Integer + ): + return replace(ref, kind="sibling") + if str(ref) != ref.name: + where += f": {ref.name!r}" + raise ValueError(f"{where} does not resolve to a known dim (explicit, derived, or grid)") + + if isinstance(field, List): raise ValueError( - f"Array {array_field.name!r} shape element {element!r} " - f"does not resolve to a known dim " - f"(explicit, derived, or grid)" + f"List {field.name!r} has invalid shape element {str(ref)!r}: {_LIST_SHAPE_FORMS}" ) - if m := _LOOKUP_RE.fullmatch(element): - component_ref, block_name, col_name, fk_field_name = m.groups() - - # array must be a subfield of a record, not a top-level block field - if enclosing_record is None: - raise ValueError( - f"Array {array_field.name!r} shape element {element!r} is a " - f"row-level lookup but the array is not inside a record" - ) - - # Resolve target component (cross-component reference or local) - if component_ref is not None: - if spec is None: - raise ValueError( - f"Array {array_field.name!r} shape element {element!r}: " - f"cross-component reference requires a Dfns spec" - ) - target = spec.components.get(component_ref) - if target is None: - raise ValueError( - f"Array {array_field.name!r} shape element {element!r}: " - f"component {component_ref!r} not found in spec" - ) - else: - target = component # type: ignore - - # block_name must identify a list block in the target component - list_field = _find_list_in_block(target, block_name) # type: ignore - if list_field is None: - where = f"component {component_ref!r}" if component_ref else "this component" - raise ValueError( - f"Array {array_field.name!r} shape element {element!r}: " - f"{block_name!r} is not a list block in {where}" - ) + # array must be a subfield of a record, not a top-level block field + if enclosing_record is None: + raise ValueError(f"{where} is a row-level lookup but the array is not inside a record") + + # Resolve target component (cross-component reference or local) + target: "ComponentBase | None" + if ref.component is not None: + if spec is None: + raise ValueError(f"{where}: cross-component reference requires a Dfns spec") + target = spec.components.get(ref.component) + if target is None: + raise ValueError(f"{where}: component {ref.component!r} not found in spec") + elif component is None: + raise ValueError(f"{where}: row-level lookup requires a component") + else: + target = component - # col_name must be an Integer field in the list's item record - item = list_field.item - item_fields: dict = item.fields if isinstance(item, Record) else item.arms - col_field = item_fields.get(col_name) - if col_field is None: - raise ValueError( - f"Array {array_field.name!r} shape element {element!r}: " - f"{col_name!r} is not a field in {list_field.name!r} item" - ) - if not isinstance(col_field, Integer): - raise ValueError( - f"Array {array_field.name!r} shape element {element!r}: " - f"{col_name!r} is {type(col_field).__name__}, must be Integer" - ) + # block must identify a list block in the target component + list_field = _find_list_in_block(target, ref.block) # type: ignore + if list_field is None: + loc = f"component {ref.component!r}" if ref.component else "this component" + raise ValueError(f"{where}: {ref.block!r} is not a list block in {loc}") - # fk_field_name must be a sibling field in the enclosing record - fk_field = enclosing_record.fields.get(fk_field_name) - if fk_field is None: - raise ValueError( - f"Array {array_field.name!r} shape element {element!r}: " - f"{fk_field_name!r} is not a sibling field in the enclosing record" - ) + # the column must be an Integer field in the list's item record + item = list_field.item + item_fields: dict = item.fields if isinstance(item, Record) else item.arms + col_field = item_fields.get(ref.name) + if col_field is None: + raise ValueError(f"{where}: {ref.name!r} is not a field in {list_field.name!r} item") + if not isinstance(col_field, Integer): + raise ValueError(f"{where}: {ref.name!r} is {type(col_field).__name__}, must be Integer") - # fk_field.fk must be set and its block portion must match block_name - fk = getattr(fk_field, "fk", None) - if fk is None: - raise ValueError( - f"Array {array_field.name!r} shape element {element!r}: " - f"{fk_field_name!r}.fk is not set" - ) - fk_block = fk.split(".")[0] if "." in fk else fk - if fk_block != block_name: - raise ValueError( - f"Array {array_field.name!r} shape element {element!r}: " - f"{fk_field_name!r}.fk = {fk!r} does not reference block {block_name!r}" - ) - return + # the fk field must be a sibling field in the enclosing record + fk_field = enclosing_record.fields.get(ref.fk_field) # type: ignore + if fk_field is None: + raise ValueError( + f"{where}: {ref.fk_field!r} is not a sibling field in the enclosing record" + ) - # validate simple integer arithmetic - if m := _ARITH_RE.fullmatch(element): - dim_name = m.group(1) - if dim_name in known_dims: - return - if enclosing_record is not None: - sibling = enclosing_record.fields.get(dim_name) - if isinstance(sibling, Integer): - return + # its fk must be set and its block portion must match the lookup's block + fk = getattr(fk_field, "fk", None) + if fk is None: + raise ValueError(f"{where}: {ref.fk_field!r}.fk is not set") + fk_block = fk.split(".")[0] if "." in fk else fk + if fk_block != ref.block: raise ValueError( - f"Array {array_field.name!r} shape element {element!r}: " - f"{dim_name!r} does not resolve to a known dim " - f"(explicit, derived, or grid)" + f"{where}: {ref.fk_field!r}.fk = {fk!r} does not reference block {ref.block!r}" ) + return ref - raise ValueError( - f"Array {array_field.name!r} has invalid shape element {element!r}: " - f"must be a dim reference (^[A-Za-z_]\\w*$), an arithmetic offset " - f"(dim [+-] integer), or a row-level lookup (block.column(fk_field)), " - f"optionally prefixed by a bound (<, <=, >, >=)" + +def _validate_shape_element( + element: str, + array_field: "Array", + component: "ComponentBase", + enclosing_record: "Record | None", + known_dims: set[str], + spec: "Dfns | None" = None, +) -> None: + """ + Validate one element of an Array.shape: it must parse (see + `parse_shape_element`) and resolve (see `resolve_shape_ref`). + + Raises ValueError on any violation. + """ + try: + ref = parse_shape_element(element) + except _ShapeSyntaxError as e: + raise ValueError(f"Array {array_field.name!r} has {e}") from None + resolve_shape_ref( + ref, + array_field, + known_dims=known_dims, + component=component, + enclosing_record=enclosing_record, + spec=spec, ) @@ -1181,38 +1245,14 @@ def _validate_list_shape_element( - Arithmetic offset (dim [+-] integer) - Any of the above prefixed with a bound operator (<, >, <=, >=) """ - op, core = split_bound(element) - if op is not None: - if split_bound(core)[0] is not None: - raise ValueError( - f"List {list_field.name!r} has invalid shape element {element!r}: " - f"at most one bound operator is allowed" - ) - _validate_list_shape_element(core, list_field, known_dims) - return - - if _DIM_RE.fullmatch(element): - if element not in known_dims: - raise ValueError( - f"List {list_field.name!r} shape element {element!r} " - f"does not resolve to a known dim" - ) - return - - if m := _ARITH_RE.fullmatch(element): - dim_name = m.group(1) - if dim_name not in known_dims: - raise ValueError( - f"List {list_field.name!r} shape element {element!r}: " - f"{dim_name!r} does not resolve to a known dim" - ) - return - - raise ValueError( - f"List {list_field.name!r} has invalid shape element {element!r}: " - f"must be a dim reference (^[A-Za-z_]\\w*$), an arithmetic offset " - f"(dim [+-] integer), optionally prefixed by a bound (<, <=, >, >=)" - ) + try: + ref = parse_shape_element(element) + except _ShapeSyntaxError as e: + reason = _LIST_SHAPE_FORMS if e.reason == _SHAPE_FORMS else e.reason + raise ValueError( + f"List {list_field.name!r} has invalid shape element {element!r}: {reason}" + ) from None + resolve_shape_ref(ref, list_field, known_dims=known_dims) def _validate_fk_fields(component: "ComponentBase", spec: "Dfns") -> None: From ad66f0aac4e314964eb7311dfcc6fc1d90f0fdc6 Mon Sep 17 00:00:00 2001 From: wpbonelli Date: Tue, 6 Oct 2026 06:37:10 -0700 Subject: [PATCH 2/6] feat(dfns): classify shape siblings at parse time, evaluate dims parse_shape_element takes the enclosing record and returns "sibling" for an Integer subfield of it; loading rejects a name that's both a dim and a sibling, so no resolver is needed. Drop resolve_shape_ref and fold the Array/List shape validators into one check. Add evaluate_dim, which evaluates a dim value expression given the dim expressions and an input lookup. Require an input dim's bare field name to be the dim's own name, and document input vs derived dims in the spec. Co-Authored-By: Claude Opus 5.5 --- autotest/dfns/test_schema_dimensions.py | 187 ++++++++++------ docs/md/dfns.md | 41 ++-- docs/md/dfnspec.md | 13 +- modflow_devtools/dfns/__init__.py | 4 +- modflow_devtools/dfns/schema.py | 273 ++++++++++++++---------- 5 files changed, 313 insertions(+), 205 deletions(-) diff --git a/autotest/dfns/test_schema_dimensions.py b/autotest/dfns/test_schema_dimensions.py index 72685119..0a472028 100644 --- a/autotest/dfns/test_schema_dimensions.py +++ b/autotest/dfns/test_schema_dimensions.py @@ -4,7 +4,7 @@ import pytest -from autotest.dfns.test_schema import _pkg +from autotest.dfns.test_schema import _DEV3_SNAPSHOT_DIR, _pkg from modflow_devtools.dfns.schema import ( Array, Block, @@ -24,8 +24,8 @@ _validate_list_shape_element, _validate_shape_element, _validate_sum_call, + evaluate_dim, parse_shape_element, - resolve_shape_ref, split_bound, ) @@ -558,9 +558,7 @@ def test_validate_shape_element_fk_block_mismatch(): ], ) def test_parse_shape_element(element, expected): - ref = parse_shape_element(element) - assert ref == expected - assert parse_shape_element(str(ref)) == ref + assert parse_shape_element(element) == expected @pytest.mark.parametrize( @@ -580,56 +578,32 @@ def test_parse_shape_element_invalid(element, match): parse_shape_element(element) -def test_shape_ref_str(): - assert str(parse_shape_element("ncol + 1")) == "ncol+1" - assert str(parse_shape_element(">= nper")) == ">=nper" - assert str(parse_shape_element("gwf-x.block.col(fk)")) == "gwf-x.block.col(fk)" +def test_parse_shape_element_sibling(): + ncvert = Integer(name="ncvert") + arr = Array(name="icvert", dtype="integer", shape=["ncvert"]) + enc = Record(name="item", fields={"ncvert": ncvert, "icvert": arr}) + assert parse_shape_element("ncvert", enc) == ShapeRef("sibling", "ncvert") + assert parse_shape_element("ncvert") == ShapeRef("dim", "ncvert") + assert parse_shape_element("nvert", enc) == ShapeRef("dim", "nvert") -def test_resolve_shape_ref_dim(): - arr = Array(name="arr", dtype="double", shape=[]) - ref = resolve_shape_ref(parse_shape_element("<=nseg-1"), arr, known_dims={"nseg"}) - assert ref == ShapeRef("dim", "nseg", offset=-1, bound="<=") +def test_parse_shape_element_non_integer_not_sibling(): + arr = Array(name="vals", dtype="double", shape=["n"]) + enc = Record(name="item", fields={"n": String(name="n"), "vals": arr}) + assert parse_shape_element("n", enc).kind == "dim" -def test_resolve_shape_ref_sibling(): +def test_validate_shape_element_sibling(): arr = Array(name="icvert", dtype="integer", shape=[]) enc = Record(name="item", fields={"ncvert": Integer(name="ncvert"), "icvert": arr}) - ref = resolve_shape_ref( - parse_shape_element("ncvert"), arr, known_dims=set(), enclosing_record=enc - ) - assert ref.kind == "sibling" - assert ref.name == "ncvert" + _validate_shape_element("ncvert", arr, _pkg("test"), enc, set()) -def test_resolve_shape_ref_dim_shadows_sibling(): - """A name that is both a dim and a sibling Integer resolves to the dim.""" +def test_validate_shape_element_sibling_shadows_dim(): arr = Array(name="icvert", dtype="integer", shape=[]) enc = Record(name="item", fields={"ncvert": Integer(name="ncvert"), "icvert": arr}) - ref = resolve_shape_ref( - parse_shape_element("ncvert"), arr, known_dims={"ncvert"}, enclosing_record=enc - ) - assert ref.kind == "dim" - - -def test_resolve_shape_ref_non_integer_sibling(): - arr = Array(name="vals", dtype="double", shape=[]) - enc = Record(name="item", fields={"n": String(name="n"), "vals": arr}) - with pytest.raises(ValueError, match="does not resolve to a known dim"): - resolve_shape_ref(parse_shape_element("n"), arr, known_dims=set(), enclosing_record=enc) - - -def test_resolve_shape_ref_unknown_dim(): - arr = Array(name="arr", dtype="double", shape=[]) - with pytest.raises(ValueError, match="'nseg' does not resolve to a known dim"): - resolve_shape_ref(parse_shape_element("nseg-1"), arr, known_dims=set()) - - -def test_resolve_shape_ref_lookup(): - arr, enc, pkg, known = _lookup_ctx() - ref = parse_shape_element("packagedata.nlakeconn(lakeno)") - resolved = resolve_shape_ref(ref, arr, known_dims=known, component=pkg, enclosing_record=enc) - assert resolved == ref + with pytest.raises(ValueError, match="both a dim and a sibling field"): + _validate_shape_element("ncvert", arr, _pkg("test"), enc, {"ncvert"}) def _cross_component_lookup_ctx(): @@ -644,40 +618,29 @@ def _cross_component_lookup_ctx(): return arr, enc, other, spec -def test_resolve_shape_ref_cross_component_lookup(): +def test_validate_shape_element_cross_component_lookup(): arr, enc, other, spec = _cross_component_lookup_ctx() - ref = parse_shape_element("gwf-lak.packagedata.nlakeconn(lakeno)") - resolved = resolve_shape_ref( - ref, arr, known_dims=set(), component=other, enclosing_record=enc, spec=spec - ) - assert resolved == ref + elem = "gwf-lak.packagedata.nlakeconn(lakeno)" + _validate_shape_element(elem, arr, other, enc, set(), spec) -def test_resolve_shape_ref_cross_component_lookup_requires_spec(): +def test_validate_shape_element_cross_component_lookup_requires_spec(): arr, enc, other, _spec = _cross_component_lookup_ctx() - ref = parse_shape_element("gwf-lak.packagedata.nlakeconn(lakeno)") with pytest.raises(ValueError, match="requires a Dfns spec"): - resolve_shape_ref(ref, arr, known_dims=set(), component=other, enclosing_record=enc) + _validate_shape_element("gwf-lak.packagedata.nlakeconn(lakeno)", arr, other, enc, set()) -def test_resolve_shape_ref_cross_component_lookup_unknown_component(): +def test_validate_shape_element_cross_component_lookup_unknown_component(): arr, enc, other, spec = _cross_component_lookup_ctx() - ref = parse_shape_element("gwf-nope.packagedata.nlakeconn(lakeno)") with pytest.raises(ValueError, match="not found in spec"): - resolve_shape_ref( - ref, arr, known_dims=set(), component=other, enclosing_record=enc, spec=spec + _validate_shape_element( + "gwf-nope.packagedata.nlakeconn(lakeno)", arr, other, enc, set(), spec ) -def test_resolve_shape_ref_list_lookup(): - lst = List(name="stress_period_data", item=Record(name="item", fields={})) - with pytest.raises(ValueError, match="invalid shape element"): - resolve_shape_ref(parse_shape_element("packagedata.ncon(ifno)"), lst, known_dims=set()) - - def test_validate_list_shape_element_lookup(): lst = List(name="stress_period_data", item=Record(name="item", fields={})) - with pytest.raises(ValueError, match="invalid shape element"): + with pytest.raises(ValueError, match="not inside a record"): _validate_list_shape_element("packagedata.ncon(ifno)", lst, {"maxbound"}) @@ -1131,3 +1094,97 @@ def test_input_dims_visible_to_model_component_itself(): gwf = Model(name="gwf-nam", parent="sim-nam", blocks=None) spec = Dfns(components={"gwf-nam": gwf, "gwf-dis": dis}) assert "nodesuser" in spec.input_dims("gwf-nam") + + +def test_input_dim_must_name_its_own_field(): + pkg = Package( + name="test", + blocks={"dimensions": _dim_block("nlay")}, + dims={"nlayers": InputDim(value="nlay", scope="component")}, + ) + with pytest.raises(ValueError, match="must be the dim's own name"): + Dfns(components={"test": pkg}) + + +_DIS_DIMS = { + "nlay": "nlay", + "nrow": "nrow", + "ncol": "ncol", + "ncpl": "nrow * ncol", + "nodes": "nlay * ncpl", + "ncelldim": "3", + "naux": "len(auxiliary)", + "nconn": "sum(packagedata.nlakeconn)", + "njas": "(nja - nodes) / 2", +} +_DIS_INPUTS = { + "nlay": 2, + "nrow": 3, + "ncol": 4, + "nja": 30, + "auxiliary": ["temp", "conc"], + "packagedata.nlakeconn": [1, 2, 3], +} + + +@pytest.mark.parametrize( + "name,expected", + [ + ("nlay", 2), + ("ncpl", 12), + ("nodes", 24), + ("ncelldim", 3), + ("naux", 2), + ("nconn", 6), + ("njas", 3), + ("nja", 30), # not a dim: an input field + ], +) +def test_evaluate_dim(name, expected): + assert evaluate_dim(name, _DIS_DIMS, _DIS_INPUTS.get) == expected + + +@pytest.mark.parametrize("name", ["nlay", "nodes", "naux", "nconn"]) +def test_evaluate_dim_unset_input(name): + assert evaluate_dim(name, _DIS_DIMS, {}.get) is None + + +def test_evaluate_dim_inexact_division(): + with pytest.raises(ValueError, match="not an integer"): + evaluate_dim("njas", _DIS_DIMS, {**_DIS_INPUTS, "nja": 31}.get) + + +def test_evaluate_dim_cycle(): + with pytest.raises(ValueError, match="cycle"): + evaluate_dim("a", {"a": "b + 1", "b": "a * 2"}, {}.get) + + +@pytest.mark.parametrize("expr", ["nlay ** 2", "max(nlay, 1)", "nlay if nlay else 1", "1.5"]) +def test_evaluate_dim_unsupported(expr): + with pytest.raises(ValueError, match="unsupported"): + evaluate_dim("d", {"d": expr}, {"nlay": 2}.get) + + +class _One(int): + """1, usable as an Integer field, a self-sizing array, or a list column.""" + + def __new__(cls): + return super().__new__(cls, 1) + + def __len__(self): + return 1 + + def __iter__(self): + return iter([1]) + + +def test_evaluate_dim_snapshot_dims(): + """Every input dim in the current DFNs evaluates, given its inputs.""" + spec = Dfns.load(_DEV3_SNAPSHOT_DIR) + dis = spec.components["gwf-dis"] + dims = {n: d.value for n, d in dis.dims.items()} + assert evaluate_dim("nodes", dims, {"nlay": 2, "nrow": 3, "ncol": 4}.get) == 24 + for component in spec.components.values(): + dims = {n: d.value for n, d in (component.dims or {}).items()} + for name in dims: + assert evaluate_dim(name, dims, lambda _: _One()) is not None diff --git a/docs/md/dfns.md b/docs/md/dfns.md index cc56daf3..ddc34f35 100644 --- a/docs/md/dfns.md +++ b/docs/md/dfns.md @@ -208,7 +208,7 @@ See [DFN specification](dfnspec.md) for full attribute documentation. `Array.shape` and `List.shape` elements are strings in a small grammar (see [Dimensions](dfnspec.md#dimensions) and [Bounds](dfnspec.md#bounds)). `parse_shape_element` parses one into a `ShapeRef`: ```python -from modflow_devtools.dfns import ShapeRef, parse_shape_element +from modflow_devtools.dfns import parse_shape_element parse_shape_element("nseg-1") # ShapeRef(kind="dim", name="nseg", offset=-1) @@ -218,30 +218,37 @@ parse_shape_element("packagedata.ncon(ifno)") # ShapeRef(kind="lookup", name="ncon", block="packagedata", fk_field="ifno") ``` -`kind` is `"dim"` for a name, with an optional arithmetic `offset`, and `"lookup"` for a row-level column lookup, which also sets `component` (for a cross-component lookup), `block` and `fk_field`. `bound` is the inequality operator, or `None` for an exact extent. `str(ref)` gives the element back in canonical form. A malformed element raises `ValueError`. +`bound` is the inequality operator, or `None` for an exact extent. A `"lookup"` also sets `block`, `fk_field` and, for a cross-component lookup, `component`. -Parsing can't tell whether a name means a dim or a sibling Integer in the same record (e.g. `ncvert` sizing `icvert`). `resolve_shape_ref` decides, and checks the reference against its context: +For an inline array, pass the record it's in: a name that is an Integer subfield of the record is a `"sibling"`, an inline count on the same row, rather than a `"dim"`: ```python -from modflow_devtools.dfns import resolve_shape_ref - -sfr = spec.components["gwf-sfr"] -connectiondata = sfr.blocks["connectiondata"].fields["connectiondata"] -ic = connectiondata.item.fields["ic"] -resolve_shape_ref( - parse_shape_element(ic.shape[0]), - ic, - known_dims=spec.input_dims("gwf-sfr"), - component=sfr, - enclosing_record=connectiondata.item, - spec=spec, -) +cell2d = spec.components["gwf-disv"].blocks["cell2d"].fields["cell2d"].item +parse_shape_element("ncvert", cell2d) +# ShapeRef(kind="sibling", name="ncvert") ``` -A name in `known_dims` stays a `"dim"`. Otherwise, a name that is an Integer subfield of `enclosing_record` becomes a `"sibling"`. A lookup must be in an array inside a record, and name an Integer column in a list block, selected by a sibling whose `fk` references that block. Anything that doesn't resolve raises `ValueError`. Loading `Dfns` validates every shape this way. +A malformed element raises `ValueError`. Whether a reference resolves (the dim exists, the lookup's column and foreign key line up) is checked when `Dfns` loads. Loading also rejects a name that is both a dim and a sibling, so the record is all the context `parse_shape_element` needs. `split_bound` splits off just the bound: `split_bound("<=maxbound")` returns `("<=", "maxbound")`. +### Evaluating dims + +A dim's `value` is an expression over the component's input (see [`dims`](dfnspec.md#dims-inputdim)). `evaluate_dim` evaluates one, given the dim value expressions and a function looking up input field values: + +```python +from modflow_devtools.dfns import evaluate_dim + +dis = spec.components["gwf-dis"] +dims = {name: dim.value for name, dim in dis.dims.items()} +evaluate_dim("nodes", dims, {"nlay": 2, "nrow": 3, "ncol": 4}.get) +# 24 +``` + +A name that is another dim is evaluated in turn. Any other name, including an input dim's own name or a name not in `dims`, goes to `lookup`. `sum(list.column)` passes `lookup` the dotted path and expects the column's values back, so how rows are stored is up to the caller. The result is `None` if an input it depends on isn't set. + +It works on the expression strings alone, without a loaded `Dfns`. + ### Rendering block templates `Block.render()` produces a `BEGIN/END` template string showing the structure of a block — the same format used in the MODFLOW 6 user guide and in tooling such as IDE hover text: diff --git a/docs/md/dfnspec.md b/docs/md/dfnspec.md index 90d38cde..50574aaa 100644 --- a/docs/md/dfnspec.md +++ b/docs/md/dfnspec.md @@ -620,11 +620,14 @@ dims: `InputDim` entries may be used in the `shape` expression of `list` and `array` input fields, and in memory variable shapes. -The `value` attribute defines the dimension source as a Python expression. Three forms are distinguished: +The `value` attribute defines the dimension as a Python expression over the component's input: -- **Bare identifier** `nlay` — backed by an integer field of that name in this component. The dimension takes the runtime value of that field. -- **`len(name)`** — backed by a self-sizing array field of that name. The dimension equals the runtime length of the array. -- **Arithmetic expression** `nlay * nrow * ncol` — derived from other dims. May not use bare field names; all operands must be declared dimensions. +- **Bare identifier** `nlay` — an integer field in this component, which must have the dimension's own name. The dimension takes the runtime value of that field. +- **`len(name)`** — the runtime length of a self-sizing array field. +- **`sum(list.column)`** — the sum of an integer column over a list's rows (`sum(packagedata.nlakeconn)`). +- **Arithmetic expression** `nlay * nrow * ncol` — integer literals and other dims, combined with `+`, `-`, `*`, `/` (exact) and `//`. May not use bare field names; all operands must be declared dimensions. + +A dimension whose value is its own name is an **input dimension**: it *is* the field, so a consumer may set the field from data the dimension sizes (e.g. `numalphaj` from the width of `cellidsj`). Every other dimension is **derived**: a function of the input, which a consumer can evaluate and check data against, but not set. `modflow_devtools.dfns` evaluates any dimension with `evaluate_dim`; see [Evaluating dims](dfns.md#evaluating-dims). #### `runtime_dims` (RuntimeDim) @@ -758,7 +761,7 @@ A shape expression gives an exact extent. Prefixed with one of the inequality op An unprefixed extent is exact, so a DFN must mark every bound. A consumer may reject input that doesn't satisfy the relation: too many or too few rows for an exact extent, too many for an upper bound, too few for a lower bound. -`modflow_devtools.dfns` parses shape expressions with `parse_shape_element` and resolves them against their component with `resolve_shape_ref`; see [Parsing shape elements](dfns.md#parsing-shape-elements). +`modflow_devtools.dfns` parses shape expressions with `parse_shape_element`; see [Parsing shape elements](dfns.md#parsing-shape-elements). #### Dimension scope diff --git a/modflow_devtools/dfns/__init__.py b/modflow_devtools/dfns/__init__.py index a29fe8f0..a7bb01d7 100644 --- a/modflow_devtools/dfns/__init__.py +++ b/modflow_devtools/dfns/__init__.py @@ -33,8 +33,8 @@ Simulation, String, Union, + evaluate_dim, parse_shape_element, - resolve_shape_ref, split_bound, ) @@ -81,9 +81,9 @@ "Simulation", "String", "Union", + "evaluate_dim", "fetch_dfns", "migrate", "parse_shape_element", - "resolve_shape_ref", "split_bound", ] diff --git a/modflow_devtools/dfns/schema.py b/modflow_devtools/dfns/schema.py index 8e78eacf..bb2cc424 100644 --- a/modflow_devtools/dfns/schema.py +++ b/modflow_devtools/dfns/schema.py @@ -1,7 +1,7 @@ import ast import re from collections.abc import Callable, Mapping -from dataclasses import dataclass, replace +from dataclasses import dataclass from os import PathLike from pathlib import Path from typing import Annotated, Any, Literal, cast @@ -705,7 +705,11 @@ def _resolve_derived_dims(component: "ComponentBase", known_dims: set[str]) -> l _validate_sum_call(node, component, value) if _DIM_RE.fullmatch(value): - # Bare identifier: must name an Integer field in this component + # Bare identifier: an input dim, the Integer field of its own name + if value != name: + raise ValueError( + f"dims {name!r}: a bare field name must be the dim's own name, got {value!r}" + ) field = component.get_fields().get(value) if field is None: raise ValueError(f"dims {name!r}: field {value!r} not found in component") @@ -748,6 +752,87 @@ def _resolve_derived_dims(component: "ComponentBase", known_dims: set[str]) -> l return order +def evaluate_dim( + name: str, + dims: Mapping[str, str], + lookup: Callable[[str], Any], +) -> int | None: + """ + Evaluate a dim, given its component's dim ``value`` expressions. + + ``dims`` maps dim names to value expressions (``{"nodes": "nlay * ncpl", + ...}``, as in ``InputDim.value``). ``lookup`` gets an input field's value + by name, or by dotted path for ``sum(list.column)`` (an iterable of the + column's values), and returns None if the field isn't set. + + A name in an expression that is another dim is evaluated in turn; any + other name, like an input dim's own name or one not in ``dims``, is an + input field, passed to ``lookup``. Expressions may use integer literals, + ``+``, ``-``, ``*``, ``/`` (exact), ``//``, ``len()`` and ``sum()``. + + Returns None if an input it depends on isn't set. Raises ValueError for a + malformed expression, a cycle, or an inexact ``/``. + """ + + def dim(n: str, seen: frozenset[str]) -> int | None: + expr = dims.get(n, n) + if expr == n: + v = lookup(n) + return None if v is None else int(v) + if n in seen: + raise ValueError(f"cycle in dims at {n!r}") + try: + tree = ast.parse(expr, mode="eval") + except SyntaxError as e: + raise ValueError(f"invalid dim {n!r}: {expr!r}: {e}") from e + return ev(tree.body, seen | {n}, expr) + + def path(node: ast.expr, expr: str) -> str: + if isinstance(node, ast.Name): + return node.id + if isinstance(node, ast.Attribute): + return f"{path(node.value, expr)}.{node.attr}" + raise ValueError(f"unsupported argument in {expr!r}: {ast.unparse(node)!r}") + + def ev(node: ast.expr, seen: frozenset[str], expr: str) -> int | None: + if isinstance(node, ast.Constant) and type(node.value) is int: + return node.value + if isinstance(node, ast.Name): + return dim(node.id, seen) + if isinstance(node, ast.UnaryOp) and isinstance(node.op, ast.USub): + v = ev(node.operand, seen, expr) + return None if v is None else -v + if isinstance(node, ast.BinOp) and isinstance( + node.op, (ast.Add, ast.Sub, ast.Mult, ast.Div, ast.FloorDiv) + ): + a, b = ev(node.left, seen, expr), ev(node.right, seen, expr) + if a is None or b is None: + return None + if isinstance(node.op, ast.Add): + return a + b + if isinstance(node.op, ast.Sub): + return a - b + if isinstance(node.op, ast.Mult): + return a * b + if isinstance(node.op, ast.Div) and a % b: + raise ValueError(f"{a} / {b} is not an integer in {expr!r}") + return a // b + if ( + isinstance(node, ast.Call) + and isinstance(node.func, ast.Name) + and node.func.id in ("len", "sum") + and len(node.args) == 1 + and not node.keywords + ): + v = lookup(path(node.args[0], expr)) + if v is None: + return None + return len(v) if node.func.id == "len" else int(sum(v)) + raise ValueError(f"unsupported expression in {expr!r}: {ast.unparse(node)!r}") + + return dim(name, frozenset()) + + class BlockHeader(BaseModel): """A repeating block's header field, plus whether a missing occurrence means "reuse the prior occurrence's values" (`period`, e.g.). Nested here rather @@ -1005,10 +1090,6 @@ class Package(ComponentBase): "(dim [+-] integer), or a row-level lookup (block.column(fk_field)), " "optionally prefixed by a bound (<, <=, >, >=)" ) -_LIST_SHAPE_FORMS = ( - "must be a dim reference (^[A-Za-z_]\\w*$), an arithmetic offset " - "(dim [+-] integer), optionally prefixed by a bound (<, <=, >, >=)" -) def split_bound(element: str) -> "tuple[str | None, str]": @@ -1029,10 +1110,9 @@ class ShapeRef: """ A parsed shape element; see :func:`parse_shape_element`. - ``kind`` is ``"dim"`` for a name reference (optionally with an arithmetic - offset) and ``"lookup"`` for a row-level column lookup. Parsing can't tell - a dim from a sibling field; :func:`resolve_shape_ref` narrows a ``"dim"`` - naming an Integer subfield of the enclosing record to ``"sibling"``. + ``kind`` is ``"dim"`` for a dim reference, ``"sibling"`` for a reference + to an Integer subfield of the same record (an inline count on the same + row), and ``"lookup"`` for a row-level column lookup. """ kind: Literal["dim", "sibling", "lookup"] @@ -1047,43 +1127,24 @@ class ShapeRef: block: str | None = None fk_field: str | None = None - def __str__(self) -> str: - if self.kind == "lookup": - expr = f"{self.block}.{self.name}({self.fk_field})" - if self.component is not None: - expr = f"{self.component}.{expr}" - else: - expr = self.name - if self.offset: - expr += f"{self.offset:+d}" - return f"{self.bound or ''}{expr}" - - -class _ShapeSyntaxError(ValueError): - """A malformed shape element; ``reason`` says why.""" - def __init__(self, element: str, reason: str): - super().__init__(f"invalid shape element {element!r}: {reason}") - self.reason = reason - - -def parse_shape_element(element: str) -> ShapeRef: +def parse_shape_element(element: str, record: "Record | None" = None) -> ShapeRef: """ Parse one element of an ``Array.shape`` or ``List.shape``. - Purely syntactic: a name parses as a ``"dim"`` whether it names a dim or a - sibling field, and nothing is checked against a component. Use - :func:`resolve_shape_ref` for that. Raises ``ValueError`` if the element - is malformed. + ``record`` is the record the array is a subfield of, if any. A name that + is an Integer subfield of it is a ``"sibling"`` (cell2d's ``ncvert``, + counting ``icvert``), otherwise a ``"dim"``. Loading rejects a name that + is both, so the record is all the context needed to tell them apart. + + Nothing else is checked against a component; that happens when ``Dfns`` + loads. Raises ``ValueError`` if the element is malformed. """ bound, core = split_bound(element) if bound is not None and split_bound(core)[0] is not None: - raise _ShapeSyntaxError(element, "at most one bound operator is allowed") - if _DIM_RE.fullmatch(core): - return ShapeRef("dim", core, bound=bound) - if m := _ARITH_RE.fullmatch(core): - name, sign, n = m.groups() - return ShapeRef("dim", name, offset=int(sign + n), bound=bound) + raise ValueError( + f"invalid shape element {element!r}: at most one bound operator is allowed" + ) if m := _LOOKUP_RE.fullmatch(core): component_ref, block_name, col_name, fk_field_name = m.groups() return ShapeRef( @@ -1094,7 +1155,15 @@ def parse_shape_element(element: str) -> ShapeRef: block=block_name, fk_field=fk_field_name, ) - raise _ShapeSyntaxError(element, _SHAPE_FORMS) + if _DIM_RE.fullmatch(core): + name, offset = core, 0 + elif m := _ARITH_RE.fullmatch(core): + name, sign, n = m.groups() + offset = int(sign + n) + else: + raise ValueError(f"invalid shape element {element!r}: {_SHAPE_FORMS}") + sibling = record is not None and isinstance(record.fields.get(name), Integer) + return ShapeRef("sibling" if sibling else "dim", name, offset=offset, bound=bound) def _find_list_in_block(component: "ComponentBase", block_name: str) -> "List | None": @@ -1108,69 +1177,70 @@ def _find_list_in_block(component: "ComponentBase", block_name: str) -> "List | return None -def resolve_shape_ref( - ref: ShapeRef, +def _check_shape_element( + element: str, field: "Array | List", - *, known_dims: set[str], component: "ComponentBase | None" = None, enclosing_record: "Record | None" = None, spec: "Dfns | None" = None, -) -> ShapeRef: +) -> None: """ - Resolve a parsed element of ``field``'s shape against its context. - - A name resolves to a dim in ``known_dims`` (the dims visible to the - component, e.g. ``Dfns.input_dims()``), or failing that to an Integer - subfield of ``enclosing_record``, in which case the result's ``kind`` is - ``"sibling"``. A lookup must be in an array inside a record, and name an - Integer column of the list in a block of ``component`` (or, given - ``spec``, of another component), selected by a sibling whose ``fk`` - references that block. Lists aren't inside records, so a list's shape may - only reference dims. - - Returns ``ref`` with ``kind`` narrowed. Raises ``ValueError`` if it - doesn't resolve. + Check that a shape element parses and resolves in its context: a dim in + ``known_dims``, a sibling that isn't also a dim, or a lookup of an Integer + column in a list block (in ``component``, or with ``spec`` another + component), selected by a sibling whose ``fk`` references that block. + + Raises ValueError on any violation. """ - where = f"{type(field).__name__} {field.name!r} shape element {str(ref)!r}" + try: + ref = parse_shape_element(element, enclosing_record) + _check_shape_ref(element, ref, known_dims, component, enclosing_record, spec) + except ValueError as e: + raise ValueError(f"{type(field).__name__} {field.name!r} has {e}") from None - if ref.kind != "lookup": - if ref.name in known_dims: - return replace(ref, kind="dim") - # Per-row varying shape: a sibling Integer supplies an inline count on - # the same line. - if enclosing_record is not None and isinstance( - enclosing_record.fields.get(ref.name), Integer - ): - return replace(ref, kind="sibling") - if str(ref) != ref.name: - where += f": {ref.name!r}" - raise ValueError(f"{where} does not resolve to a known dim (explicit, derived, or grid)") - if isinstance(field, List): - raise ValueError( - f"List {field.name!r} has invalid shape element {str(ref)!r}: {_LIST_SHAPE_FORMS}" - ) +def _check_shape_ref( + element: str, + ref: ShapeRef, + known_dims: set[str], + component: "ComponentBase | None", + enclosing_record: "Record | None", + spec: "Dfns | None", +) -> None: + where = f"invalid shape element {element!r}" + + if ref.kind == "dim": + if ref.name not in known_dims: + raise ValueError( + f"{where}: {ref.name!r} does not resolve to a known dim " + f"(explicit, derived, or grid)" + ) + return + + if ref.kind == "sibling": + if ref.name in known_dims: + raise ValueError(f"{where}: {ref.name!r} is both a dim and a sibling field") + return - # array must be a subfield of a record, not a top-level block field + # a lookup selects a row by a sibling field, so the array must be in a record if enclosing_record is None: - raise ValueError(f"{where} is a row-level lookup but the array is not inside a record") + raise ValueError(f"{where}: a row-level lookup, but the field is not inside a record") # Resolve target component (cross-component reference or local) - target: "ComponentBase | None" + target: ComponentBase | None if ref.component is not None: if spec is None: raise ValueError(f"{where}: cross-component reference requires a Dfns spec") target = spec.components.get(ref.component) if target is None: raise ValueError(f"{where}: component {ref.component!r} not found in spec") - elif component is None: - raise ValueError(f"{where}: row-level lookup requires a component") else: target = component + assert target is not None and ref.block is not None and ref.fk_field is not None # block must identify a list block in the target component - list_field = _find_list_in_block(target, ref.block) # type: ignore + list_field = _find_list_in_block(target, ref.block) if list_field is None: loc = f"component {ref.component!r}" if ref.component else "this component" raise ValueError(f"{where}: {ref.block!r} is not a list block in {loc}") @@ -1185,7 +1255,7 @@ def resolve_shape_ref( raise ValueError(f"{where}: {ref.name!r} is {type(col_field).__name__}, must be Integer") # the fk field must be a sibling field in the enclosing record - fk_field = enclosing_record.fields.get(ref.fk_field) # type: ignore + fk_field = enclosing_record.fields.get(ref.fk_field) if fk_field is None: raise ValueError( f"{where}: {ref.fk_field!r} is not a sibling field in the enclosing record" @@ -1200,7 +1270,6 @@ def resolve_shape_ref( raise ValueError( f"{where}: {ref.fk_field!r}.fk = {fk!r} does not reference block {ref.block!r}" ) - return ref def _validate_shape_element( @@ -1211,24 +1280,8 @@ def _validate_shape_element( known_dims: set[str], spec: "Dfns | None" = None, ) -> None: - """ - Validate one element of an Array.shape: it must parse (see - `parse_shape_element`) and resolve (see `resolve_shape_ref`). - - Raises ValueError on any violation. - """ - try: - ref = parse_shape_element(element) - except _ShapeSyntaxError as e: - raise ValueError(f"Array {array_field.name!r} has {e}") from None - resolve_shape_ref( - ref, - array_field, - known_dims=known_dims, - component=component, - enclosing_record=enclosing_record, - spec=spec, - ) + """Validate one element of an Array.shape (see `_check_shape_element`).""" + _check_shape_element(element, array_field, known_dims, component, enclosing_record, spec) def _validate_list_shape_element( @@ -1237,22 +1290,10 @@ def _validate_list_shape_element( known_dims: set[str], ) -> None: """ - Validate one element of a List.shape. - - Valid forms are a strict subset of array shape forms — no row-level lookup - and no intra-record sibling reference, since lists are not inside records: - - Plain dim reference - - Arithmetic offset (dim [+-] integer) - - Any of the above prefixed with a bound operator (<, >, <=, >=) + Validate one element of a List.shape. A list isn't inside a record, so + its shape may only reference dims. """ - try: - ref = parse_shape_element(element) - except _ShapeSyntaxError as e: - reason = _LIST_SHAPE_FORMS if e.reason == _SHAPE_FORMS else e.reason - raise ValueError( - f"List {list_field.name!r} has invalid shape element {element!r}: {reason}" - ) from None - resolve_shape_ref(ref, list_field, known_dims=known_dims) + _check_shape_element(element, list_field, known_dims) def _validate_fk_fields(component: "ComponentBase", spec: "Dfns") -> None: From b37f2731877f1a08356a89c6d299765392fe523c Mon Sep 17 00:00:00 2001 From: wpbonelli Date: Tue, 6 Oct 2026 06:49:10 -0700 Subject: [PATCH 3/6] feat(dfns): evaluate shape expressions, keep the shape parser private Consumers evaluate shapes rather than classify them: evaluate_dim now takes any dimension expression (a dim, or a shape expression without its bound), with a select callback for row-level lookups. The shape parser and ShapeRef become private, used only by validation. Loading now requires an inline array whose size varies by row (a sibling or lookup shape) to be the rightmost field in its record, as for a self-sizing array, so consumers needn't check. Co-Authored-By: Claude Opus 5.5 --- autotest/dfns/test_schema_dimensions.py | 116 ++++++++++++++++++++---- docs/md/dfns.md | 44 +++------ docs/md/dfnspec.md | 6 +- modflow_devtools/dfns/__init__.py | 4 - modflow_devtools/dfns/schema.py | 73 +++++++++++---- 5 files changed, 168 insertions(+), 75 deletions(-) diff --git a/autotest/dfns/test_schema_dimensions.py b/autotest/dfns/test_schema_dimensions.py index 0a472028..a512c6b0 100644 --- a/autotest/dfns/test_schema_dimensions.py +++ b/autotest/dfns/test_schema_dimensions.py @@ -1,6 +1,7 @@ """Tests for DFN schema array shape expressions and dimension resolution""" import ast +from collections import ChainMap import pytest @@ -9,6 +10,7 @@ Array, Block, Dfns, + Double, InputDim, Integer, List, @@ -16,16 +18,16 @@ Package, Record, RuntimeDim, - ShapeRef, String, _names_in_expr, + _parse_shape_element, _resolve_derived_dims, + _ShapeRef, _validate_len_call, _validate_list_shape_element, _validate_shape_element, _validate_sum_call, evaluate_dim, - parse_shape_element, split_bound, ) @@ -537,28 +539,28 @@ def test_validate_shape_element_fk_block_mismatch(): @pytest.mark.parametrize( "element,expected", [ - ("nvert", ShapeRef("dim", "nvert")), - ("nseg-1", ShapeRef("dim", "nseg", offset=-1)), - ("ncol + 1", ShapeRef("dim", "ncol", offset=1)), - ("<=maxats", ShapeRef("dim", "maxats", bound="<=")), - (">= nper", ShapeRef("dim", "nper", bound=">=")), - ("= nper", _ShapeRef("dim", "nper", bound=">=")), + (" l def evaluate_dim( - name: str, + expr: str, dims: Mapping[str, str], lookup: Callable[[str], Any], + select: Callable[[str, Any], Any] | None = None, ) -> int | None: """ - Evaluate a dim, given its component's dim ``value`` expressions. - - ``dims`` maps dim names to value expressions (``{"nodes": "nlay * ncpl", - ...}``, as in ``InputDim.value``). ``lookup`` gets an input field's value - by name, or by dotted path for ``sum(list.column)`` (an iterable of the - column's values), and returns None if the field isn't set. - - A name in an expression that is another dim is evaluated in turn; any - other name, like an input dim's own name or one not in ``dims``, is an - input field, passed to ``lookup``. Expressions may use integer literals, - ``+``, ``-``, ``*``, ``/`` (exact), ``//``, ``len()`` and ``sum()``. + Evaluate a dimension expression: a dim's name or ``value``, or an + ``Array``/``List`` shape expression (without its bound; see + :func:`split_bound`). + + ``dims`` maps the component's dim names to their value expressions + (``{"nodes": "nlay * ncpl", ...}``, as in ``InputDim.value``). ``lookup`` + gets an input field's value by name, or by dotted path for + ``sum(list.column)`` (an iterable of the column's values), and returns None + if the field isn't set. To evaluate an inline array's shape, ``lookup`` + should see the fields of the array's row too. + + A name that is a dim is evaluated in turn; any other name, like an input + dim's own name, a field in the row, or a name not in ``dims``, is passed to + ``lookup``. Expressions may use integer literals, ``+``, ``-``, ``*``, + ``/`` (exact), ``//``, ``len()`` and ``sum()``. + + A row-level lookup, ``[component.]block.column(fk_field)``, needs + ``select``: given the path ``"[component.]block.column"`` and the row's + ``fk_field`` value, it returns that column's value in the referenced row. Returns None if an input it depends on isn't set. Raises ValueError for a malformed expression, a cycle, or an inexact ``/``. @@ -830,7 +839,22 @@ def ev(node: ast.expr, seen: frozenset[str], expr: str) -> int | None: return len(v) if node.func.id == "len" else int(sum(v)) raise ValueError(f"unsupported expression in {expr!r}: {ast.unparse(node)!r}") - return dim(name, frozenset()) + if _BOUND_RE.match(expr): + raise ValueError(f"{expr!r} is bounded; split off the bound with split_bound") + if m := _LOOKUP_RE.fullmatch(expr): + component_ref, block_name, col_name, fk_field_name = m.groups() + if select is None: + raise ValueError(f"{expr!r} is a row-level lookup, which needs select") + key = lookup(fk_field_name) + if key is None: + return None + v = select(".".join(filter(None, (component_ref, block_name, col_name))), key) + return None if v is None else int(v) + try: + tree = ast.parse(expr, mode="eval") + except SyntaxError as e: + raise ValueError(f"invalid expression {expr!r}: {e}") from e + return ev(tree.body, frozenset(), expr) class BlockHeader(BaseModel): @@ -1106,9 +1130,9 @@ def split_bound(element: str) -> "tuple[str | None, str]": @dataclass(frozen=True) -class ShapeRef: +class _ShapeRef: """ - A parsed shape element; see :func:`parse_shape_element`. + A parsed shape element; see :func:`_parse_shape_element`. ``kind`` is ``"dim"`` for a dim reference, ``"sibling"`` for a reference to an Integer subfield of the same record (an inline count on the same @@ -1128,7 +1152,7 @@ class ShapeRef: fk_field: str | None = None -def parse_shape_element(element: str, record: "Record | None" = None) -> ShapeRef: +def _parse_shape_element(element: str, record: "Record | None" = None) -> _ShapeRef: """ Parse one element of an ``Array.shape`` or ``List.shape``. @@ -1147,7 +1171,7 @@ def parse_shape_element(element: str, record: "Record | None" = None) -> ShapeRe ) if m := _LOOKUP_RE.fullmatch(core): component_ref, block_name, col_name, fk_field_name = m.groups() - return ShapeRef( + return _ShapeRef( "lookup", col_name, bound=bound, @@ -1163,7 +1187,7 @@ def parse_shape_element(element: str, record: "Record | None" = None) -> ShapeRe else: raise ValueError(f"invalid shape element {element!r}: {_SHAPE_FORMS}") sibling = record is not None and isinstance(record.fields.get(name), Integer) - return ShapeRef("sibling" if sibling else "dim", name, offset=offset, bound=bound) + return _ShapeRef("sibling" if sibling else "dim", name, offset=offset, bound=bound) def _find_list_in_block(component: "ComponentBase", block_name: str) -> "List | None": @@ -1191,18 +1215,27 @@ def _check_shape_element( column in a list block (in ``component``, or with ``spec`` another component), selected by a sibling whose ``fk`` references that block. + A sibling or lookup varies by row, so the array must be the rightmost + field in its record, like a self-sizing array: a reader can't know where + fields after it start. + Raises ValueError on any violation. """ try: - ref = parse_shape_element(element, enclosing_record) + ref = _parse_shape_element(element, enclosing_record) _check_shape_ref(element, ref, known_dims, component, enclosing_record, spec) + if ref.kind != "dim" and list(enclosing_record.fields)[-1] != field.name: # type: ignore + raise ValueError( + f"invalid shape element {element!r}: it varies by row, so " + f"{field.name!r} must be the rightmost field in its record" + ) except ValueError as e: raise ValueError(f"{type(field).__name__} {field.name!r} has {e}") from None def _check_shape_ref( element: str, - ref: ShapeRef, + ref: _ShapeRef, known_dims: set[str], component: "ComponentBase | None", enclosing_record: "Record | None", From f36eb387645637ae8d71d6cecd7326fb0d4f67a8 Mon Sep 17 00:00:00 2001 From: wpbonelli Date: Tue, 6 Oct 2026 07:24:14 -0700 Subject: [PATCH 4/6] feat(dfns): add solve_dim, the inverse of evaluate_dim Given the extent data has, solve_dim returns the unset input that determines it and its value (nseg-1 with 3 values gives nseg=4), so a consumer can fill in counts the user left out. Only names, through dims, and +/- of known values are undone; anything else gives None, and the data can only be checked. Shares the expression walk with evaluate_dim. Co-Authored-By: Claude Opus 5.5 --- autotest/dfns/test_schema_dimensions.py | 63 ++++++++++ docs/md/dfns.md | 17 ++- docs/md/dfnspec.md | 2 +- modflow_devtools/dfns/__init__.py | 2 + modflow_devtools/dfns/schema.py | 161 ++++++++++++++++-------- 5 files changed, 190 insertions(+), 55 deletions(-) diff --git a/autotest/dfns/test_schema_dimensions.py b/autotest/dfns/test_schema_dimensions.py index a512c6b0..125f34a2 100644 --- a/autotest/dfns/test_schema_dimensions.py +++ b/autotest/dfns/test_schema_dimensions.py @@ -28,6 +28,7 @@ _validate_shape_element, _validate_sum_call, evaluate_dim, + solve_dim, split_bound, ) @@ -1270,3 +1271,65 @@ def shapes(field): _bound, expr = split_bound(element) value = evaluate_dim(expr, dims, lambda _: _One(), lambda _p, _k: 1) assert value is not None, (component.name, element) + + +@pytest.mark.parametrize( + "expr,inputs,value,expected", + [ + ("ncvert", {}, 5, ("ncvert", 5)), # a field in the row + ("numalphaj", {}, 3, ("numalphaj", 3)), # an input dim + ("nseg-1", {}, 3, ("nseg", 4)), + ("1 + nseg", {}, 3, ("nseg", 2)), + ("10 - nseg", {}, 3, ("nseg", 7)), + ("-nseg", {}, -3, ("nseg", 3)), + ("nlay", {}, 2, ("nlay", 2)), + ("nlayp", {}, 3, ("nlay", 2)), # through a derived dim + ("nseg-1", {"nseg": 4}, 3, None), # nothing unset + ("auxiliary", {}, 2, None), # len() can't be undone + ("nconn", {}, 6, None), # nor sum() + ("ncpl", {"nrow": 3}, 12, None), # nor * + ("nlay + nseg", {}, 5, None), # two unset inputs + ("packagedata.ncon(ifno)", {"ifno": 1}, 2, None), # nor a row-level lookup + ], +) +def test_solve_dim(expr, inputs, value, expected): + dims = { + "numalphaj": "numalphaj", + "nseg": "nseg", + "nlay": "nlay", + "nlayp": "nlay + 1", + "ncpl": "nrow * ncol", + "auxiliary": "len(auxiliary)", + "nconn": "sum(packagedata.nlakeconn)", + } + assert solve_dim(expr, dims, inputs.get, value) == expected + + +def test_solve_dim_snapshot_shapes(): + """Solving any shape in the current DFNs for its sole unset input, then + evaluating it with that input set, gives back the extent.""" + spec = Dfns.load(_DEV3_SNAPSHOT_DIR) + solved = set() + for component in spec.components.values(): + dims = {n: d.value for n, d in (component.dims or {}).items()} + + def shapes(field): + if isinstance(field, (Array, List)): + yield from field.shape + if isinstance(field, List): + yield from shapes(field.item) + children = getattr(field, "fields", None) or getattr(field, "arms", None) or {} + for child in children.values(): + yield from shapes(child) + + for block in (component.blocks or {}).values(): + for field in block.fields.values(): + for element in shapes(field): + _bound, expr = split_bound(element) + if (solution := solve_dim(expr, dims, {}.get, 7)) is None: + continue + name, n = solution + assert evaluate_dim(expr, dims, {name: n}.get) == 7, (component.name, expr) + solved.add(expr) + assert {"ncvert", "numalphaj", "nseg-1", "maxbound"} <= solved + assert "auxiliary" not in solved diff --git a/docs/md/dfns.md b/docs/md/dfns.md index 311706f9..0daf2848 100644 --- a/docs/md/dfns.md +++ b/docs/md/dfns.md @@ -229,7 +229,22 @@ evaluate_dim("packagedata.ncon(ifno)", dims, row.get, select) A bounded shape expression (`"<=maxbound"`) is a relation, not a value. Split off the bound with `split_bound` (`("<=", "maxbound")`) and evaluate the rest. -It works on the expression strings alone, without a loaded `Dfns`. +`solve_dim` is the inverse: given the extent data actually has, it returns the unset input that extent determines, and that input's value. Use it to fill in counts the user left out: + +```python +from modflow_devtools.dfns import solve_dim + +solve_dim("nseg-1", dims, {}.get, 3) # pxdp has 3 values +# ("nseg", 4) +solve_dim("ncvert", {}, {}.get, 5) # icvert has 5 values +# ("ncvert", 5) +solve_dim("auxiliary", dims, {}.get, 2) # len(auxiliary): can't set names from a count +# None +``` + +Only names, through dims, and adding or subtracting known values can be undone. The result is `None` if the expression has no unset input, more than one, or one under anything else (`len()`, `sum()`, `*`, a row-level lookup). In that case the data can only be checked, by evaluating the extent once its inputs are set. + +Both functions work on the expression strings alone, without a loaded `Dfns`. ### Rendering block templates diff --git a/docs/md/dfnspec.md b/docs/md/dfnspec.md index ba036629..53adf80a 100644 --- a/docs/md/dfnspec.md +++ b/docs/md/dfnspec.md @@ -627,7 +627,7 @@ The `value` attribute defines the dimension as a Python expression over the comp - **`sum(list.column)`** — the sum of an integer column over a list's rows (`sum(packagedata.nlakeconn)`). - **Arithmetic expression** `nlay * nrow * ncol` — integer literals and other dims, combined with `+`, `-`, `*`, `/` (exact) and `//`. May not use bare field names; all operands must be declared dimensions. -A dimension whose value is its own name is an **input dimension**: it *is* the field, so a consumer may set the field from data the dimension sizes (e.g. `numalphaj` from the width of `cellidsj`). Every other dimension is **derived**: a function of the input, which a consumer can evaluate and check data against, but not set. `modflow_devtools.dfns` evaluates any dimension with `evaluate_dim`; see [Evaluating dimensions and shapes](dfns.md#evaluating-dimensions-and-shapes). +A dimension whose value is its own name is an **input dimension**: it *is* the field, so a consumer may set the field from data the dimension sizes (e.g. `numalphaj` from the width of `cellidsj`). Every other dimension is **derived**: a function of the input, which a consumer can evaluate and check data against, but not set. More generally, a consumer may set an unset input from data whenever an extent determines it uniquely: `nseg` from the width of an array with shape `["nseg-1"]`, or a row's `ncvert` from the length of its `icvert`. `modflow_devtools.dfns` evaluates any dimension or extent with `evaluate_dim`, and finds the input an extent determines with `solve_dim`; see [Evaluating dimensions and shapes](dfns.md#evaluating-dimensions-and-shapes). #### `runtime_dims` (RuntimeDim) diff --git a/modflow_devtools/dfns/__init__.py b/modflow_devtools/dfns/__init__.py index 645c7cd4..caa6c247 100644 --- a/modflow_devtools/dfns/__init__.py +++ b/modflow_devtools/dfns/__init__.py @@ -33,6 +33,7 @@ String, Union, evaluate_dim, + solve_dim, split_bound, ) @@ -81,5 +82,6 @@ "evaluate_dim", "fetch_dfns", "migrate", + "solve_dim", "split_bound", ] diff --git a/modflow_devtools/dfns/schema.py b/modflow_devtools/dfns/schema.py index 0845473a..b9362ea4 100644 --- a/modflow_devtools/dfns/schema.py +++ b/modflow_devtools/dfns/schema.py @@ -752,69 +752,53 @@ def _resolve_derived_dims(component: "ComponentBase", known_dims: set[str]) -> l return order -def evaluate_dim( - expr: str, - dims: Mapping[str, str], - lookup: Callable[[str], Any], - select: Callable[[str, Any], Any] | None = None, -) -> int | None: - """ - Evaluate a dimension expression: a dim's name or ``value``, or an - ``Array``/``List`` shape expression (without its bound; see - :func:`split_bound`). +class _DimExprs: + """A component's dim value expressions, evaluated (and solved) over input + field values from ``lookup``; see :func:`evaluate_dim`.""" - ``dims`` maps the component's dim names to their value expressions - (``{"nodes": "nlay * ncpl", ...}``, as in ``InputDim.value``). ``lookup`` - gets an input field's value by name, or by dotted path for - ``sum(list.column)`` (an iterable of the column's values), and returns None - if the field isn't set. To evaluate an inline array's shape, ``lookup`` - should see the fields of the array's row too. + def __init__(self, dims: Mapping[str, str], lookup: Callable[[str], Any]): + self.dims = dims + self.lookup = lookup - A name that is a dim is evaluated in turn; any other name, like an input - dim's own name, a field in the row, or a name not in ``dims``, is passed to - ``lookup``. Expressions may use integer literals, ``+``, ``-``, ``*``, - ``/`` (exact), ``//``, ``len()`` and ``sum()``. - - A row-level lookup, ``[component.]block.column(fk_field)``, needs - ``select``: given the path ``"[component.]block.column"`` and the row's - ``fk_field`` value, it returns that column's value in the referenced row. - - Returns None if an input it depends on isn't set. Raises ValueError for a - malformed expression, a cycle, or an inexact ``/``. - """ - - def dim(n: str, seen: frozenset[str]) -> int | None: - expr = dims.get(n, n) - if expr == n: - v = lookup(n) - return None if v is None else int(v) - if n in seen: - raise ValueError(f"cycle in dims at {n!r}") + def parse(self, expr: str) -> ast.expr: + if _BOUND_RE.match(expr): + raise ValueError(f"{expr!r} is bounded; split off the bound with split_bound") try: - tree = ast.parse(expr, mode="eval") + return ast.parse(expr, mode="eval").body except SyntaxError as e: - raise ValueError(f"invalid dim {n!r}: {expr!r}: {e}") from e - return ev(tree.body, seen | {n}, expr) + raise ValueError(f"invalid expression {expr!r}: {e}") from e - def path(node: ast.expr, expr: str) -> str: + def dim(self, name: str, seen: frozenset[str]) -> "tuple[str, ast.expr] | None": + """A dim's value expression and its tree, or None for an input.""" + expr = self.dims.get(name, name) + if expr == name: + return None + if name in seen: + raise ValueError(f"cycle in dims at {name!r}") + return expr, self.parse(expr) + + def path(self, node: ast.expr, expr: str) -> str: if isinstance(node, ast.Name): return node.id if isinstance(node, ast.Attribute): - return f"{path(node.value, expr)}.{node.attr}" + return f"{self.path(node.value, expr)}.{node.attr}" raise ValueError(f"unsupported argument in {expr!r}: {ast.unparse(node)!r}") - def ev(node: ast.expr, seen: frozenset[str], expr: str) -> int | None: + def evaluate(self, node: ast.expr, seen: frozenset[str], expr: str) -> int | None: if isinstance(node, ast.Constant) and type(node.value) is int: return node.value if isinstance(node, ast.Name): - return dim(node.id, seen) + if (d := self.dim(node.id, seen)) is None: + v = self.lookup(node.id) + return None if v is None else int(v) + return self.evaluate(d[1], seen | {node.id}, d[0]) if isinstance(node, ast.UnaryOp) and isinstance(node.op, ast.USub): - v = ev(node.operand, seen, expr) + v = self.evaluate(node.operand, seen, expr) return None if v is None else -v if isinstance(node, ast.BinOp) and isinstance( node.op, (ast.Add, ast.Sub, ast.Mult, ast.Div, ast.FloorDiv) ): - a, b = ev(node.left, seen, expr), ev(node.right, seen, expr) + a, b = self.evaluate(node.left, seen, expr), self.evaluate(node.right, seen, expr) if a is None or b is None: return None if isinstance(node.op, ast.Add): @@ -833,14 +817,62 @@ def ev(node: ast.expr, seen: frozenset[str], expr: str) -> int | None: and len(node.args) == 1 and not node.keywords ): - v = lookup(path(node.args[0], expr)) + v = self.lookup(self.path(node.args[0], expr)) if v is None: return None return len(v) if node.func.id == "len" else int(sum(v)) raise ValueError(f"unsupported expression in {expr!r}: {ast.unparse(node)!r}") - if _BOUND_RE.match(expr): - raise ValueError(f"{expr!r} is bounded; split off the bound with split_bound") + def solve( + self, node: ast.expr, value: int, seen: frozenset[str], expr: str + ) -> "tuple[str, int] | None": + if isinstance(node, ast.Name): + if (d := self.dim(node.id, seen)) is None: + return (node.id, value) if self.lookup(node.id) is None else None + return self.solve(d[1], value, seen | {node.id}, d[0]) + if isinstance(node, ast.UnaryOp) and isinstance(node.op, ast.USub): + return self.solve(node.operand, -value, seen, expr) + if isinstance(node, ast.BinOp) and isinstance(node.op, (ast.Add, ast.Sub)): + a, b = self.evaluate(node.left, seen, expr), self.evaluate(node.right, seen, expr) + add = isinstance(node.op, ast.Add) + if a is None and b is not None: + return self.solve(node.left, value - b if add else value + b, seen, expr) + if b is None and a is not None: + return self.solve(node.right, value - a if add else a - value, seen, expr) + # anything else (len(), sum(), *, /, ...) can't be undone + return None + + +def evaluate_dim( + expr: str, + dims: Mapping[str, str], + lookup: Callable[[str], Any], + select: Callable[[str, Any], Any] | None = None, +) -> int | None: + """ + Evaluate a dimension expression: a dim's name or ``value``, or an + ``Array``/``List`` shape expression (without its bound; see + :func:`split_bound`). + + ``dims`` maps the component's dim names to their value expressions + (``{"nodes": "nlay * ncpl", ...}``, as in ``InputDim.value``). ``lookup`` + gets an input field's value by name, or by dotted path for + ``sum(list.column)`` (an iterable of the column's values), and returns None + if the field isn't set. To evaluate an inline array's shape, ``lookup`` + should see the fields of the array's row too. + + A name that is a dim is evaluated in turn; any other name, like an input + dim's own name, a field in the row, or a name not in ``dims``, is passed to + ``lookup``. Expressions may use integer literals, ``+``, ``-``, ``*``, + ``/`` (exact), ``//``, ``len()`` and ``sum()``. + + A row-level lookup, ``[component.]block.column(fk_field)``, needs + ``select``: given the path ``"[component.]block.column"`` and the row's + ``fk_field`` value, it returns that column's value in the referenced row. + + Returns None if an input it depends on isn't set. Raises ValueError for a + malformed expression, a cycle, or an inexact ``/``. + """ if m := _LOOKUP_RE.fullmatch(expr): component_ref, block_name, col_name, fk_field_name = m.groups() if select is None: @@ -850,11 +882,34 @@ def ev(node: ast.expr, seen: frozenset[str], expr: str) -> int | None: return None v = select(".".join(filter(None, (component_ref, block_name, col_name))), key) return None if v is None else int(v) - try: - tree = ast.parse(expr, mode="eval") - except SyntaxError as e: - raise ValueError(f"invalid expression {expr!r}: {e}") from e - return ev(tree.body, frozenset(), expr) + exprs = _DimExprs(dims, lookup) + return exprs.evaluate(exprs.parse(expr), frozenset(), expr) + + +def solve_dim( + expr: str, + dims: Mapping[str, str], + lookup: Callable[[str], Any], + value: int, +) -> tuple[str, int] | None: + """ + The inverse of :func:`evaluate_dim`: the unset input, and its value, that + makes ``expr`` evaluate to ``value``. Arguments are as for ``evaluate_dim``. + + E.g. an inline array with shape ``["nseg-1"]`` and 3 values, with ``nseg`` + unset, gives ``("nseg", 4)``; one with shape ``["ncvert"]`` and 5 values, + with the row's ``ncvert`` unset, gives ``("ncvert", 5)``. + + Only names, through dims, and ``+``/``-`` of known values can be undone. + Returns None if ``expr`` has no unset input, more than one, or one under + anything else, like ``len()``, ``sum()``, ``*`` or a row-level lookup, and + the caller can only check ``evaluate_dim`` against ``value`` once its + inputs are set. + """ + if _LOOKUP_RE.fullmatch(expr): + return None + exprs = _DimExprs(dims, lookup) + return exprs.solve(exprs.parse(expr), value, frozenset(), expr) class BlockHeader(BaseModel): From c8f927f179667df8076baabbae70b5ec2a39f683 Mon Sep 17 00:00:00 2001 From: wpbonelli Date: Tue, 6 Oct 2026 07:28:20 -0700 Subject: [PATCH 5/6] refactor(dfns): rename evaluate_dim/solve_dim to dim_value/dim_input The nouns say what each returns (a dim's value, or the input a length determines), where the verbs read as near-synonyms. dim_input's target is the keyword-only `length`, and the spec says "length" for the size along one axis (was "extent"). dims and lookup are optional: no dims, nothing set. Co-Authored-By: Claude Opus 5.5 --- autotest/dfns/test_schema_dimensions.py | 77 ++++++++++++++----------- docs/md/dfns.md | 24 ++++---- docs/md/dfnspec.md | 12 ++-- modflow_devtools/dfns/__init__.py | 8 +-- modflow_devtools/dfns/schema.py | 47 ++++++++------- 5 files changed, 90 insertions(+), 78 deletions(-) diff --git a/autotest/dfns/test_schema_dimensions.py b/autotest/dfns/test_schema_dimensions.py index 125f34a2..250016e1 100644 --- a/autotest/dfns/test_schema_dimensions.py +++ b/autotest/dfns/test_schema_dimensions.py @@ -27,8 +27,8 @@ _validate_list_shape_element, _validate_shape_element, _validate_sum_call, - evaluate_dim, - solve_dim, + dim_input, + dim_value, split_bound, ) @@ -1143,29 +1143,29 @@ def test_input_dim_must_name_its_own_field(): ("nja", 30), # not a dim: an input field ], ) -def test_evaluate_dim(name, expected): - assert evaluate_dim(name, _DIS_DIMS, _DIS_INPUTS.get) == expected +def test_dim_value(name, expected): + assert dim_value(name, _DIS_DIMS, _DIS_INPUTS.get) == expected @pytest.mark.parametrize("name", ["nlay", "nodes", "naux", "nconn"]) -def test_evaluate_dim_unset_input(name): - assert evaluate_dim(name, _DIS_DIMS, {}.get) is None +def test_dim_value_unset_input(name): + assert dim_value(name, _DIS_DIMS, {}.get) is None -def test_evaluate_dim_inexact_division(): +def test_dim_value_inexact_division(): with pytest.raises(ValueError, match="not an integer"): - evaluate_dim("njas", _DIS_DIMS, {**_DIS_INPUTS, "nja": 31}.get) + dim_value("njas", _DIS_DIMS, {**_DIS_INPUTS, "nja": 31}.get) -def test_evaluate_dim_cycle(): +def test_dim_value_cycle(): with pytest.raises(ValueError, match="cycle"): - evaluate_dim("a", {"a": "b + 1", "b": "a * 2"}, {}.get) + dim_value("a", {"a": "b + 1", "b": "a * 2"}, {}.get) @pytest.mark.parametrize("expr", ["nlay ** 2", "max(nlay, 1)", "nlay if nlay else 1", "1.5"]) -def test_evaluate_dim_unsupported(expr): +def test_dim_value_unsupported(expr): with pytest.raises(ValueError, match="unsupported"): - evaluate_dim("d", {"d": expr}, {"nlay": 2}.get) + dim_value("d", {"d": expr}, {"nlay": 2}.get) class _One(int): @@ -1181,16 +1181,16 @@ def __iter__(self): return iter([1]) -def test_evaluate_dim_snapshot_dims(): +def test_dim_value_snapshot_dims(): """Every input dim in the current DFNs evaluates, given its inputs.""" spec = Dfns.load(_DEV3_SNAPSHOT_DIR) dis = spec.components["gwf-dis"] dims = {n: d.value for n, d in dis.dims.items()} - assert evaluate_dim("nodes", dims, {"nlay": 2, "nrow": 3, "ncol": 4}.get) == 24 + assert dim_value("nodes", dims, {"nlay": 2, "nrow": 3, "ncol": 4}.get) == 24 for component in spec.components.values(): dims = {n: d.value for n, d in (component.dims or {}).items()} for name in dims: - assert evaluate_dim(name, dims, lambda _: _One()) is not None + assert dim_value(name, dims, lambda _: _One()) is not None def test_row_varying_array_must_be_rightmost(): @@ -1220,13 +1220,13 @@ def test_row_varying_lookup_must_be_rightmost(): ("ncvert", 5), # a field in the row ], ) -def test_evaluate_dim_shape_expression(expr, expected): +def test_dim_value_shape_expression(expr, expected): package = {"nseg": 4, "auxiliary": ["temp", "conc"]} row = {"ncvert": 5} - assert evaluate_dim(expr, _WEL_DIMS, ChainMap(row, package).get) == expected + assert dim_value(expr, _WEL_DIMS, ChainMap(row, package).get) == expected -def test_evaluate_dim_row_lookup(): +def test_dim_value_row_lookup(): packagedata = {1: {"ncon": 2}, 2: {"ncon": 3}} calls = [] @@ -1234,23 +1234,23 @@ def select(path, key): calls.append((path, key)) return packagedata[key]["ncon"] - assert evaluate_dim("packagedata.ncon(ifno)", {}, {"ifno": 2}.get, select) == 3 - assert evaluate_dim("gwf-x.packagedata.ncon(ifno)", {}, {"ifno": 1}.get, select) == 2 + assert dim_value("packagedata.ncon(ifno)", {}, {"ifno": 2}.get, select) == 3 + assert dim_value("gwf-x.packagedata.ncon(ifno)", {}, {"ifno": 1}.get, select) == 2 assert calls == [("packagedata.ncon", 2), ("gwf-x.packagedata.ncon", 1)] - assert evaluate_dim("packagedata.ncon(ifno)", {}, {}.get, select) is None + assert dim_value("packagedata.ncon(ifno)", {}, {}.get, select) is None -def test_evaluate_dim_row_lookup_needs_select(): +def test_dim_value_row_lookup_needs_select(): with pytest.raises(ValueError, match="needs select"): - evaluate_dim("packagedata.ncon(ifno)", {}, {"ifno": 1}.get) + dim_value("packagedata.ncon(ifno)", {}, {"ifno": 1}.get) -def test_evaluate_dim_bounded(): +def test_dim_value_bounded(): with pytest.raises(ValueError, match="split_bound"): - evaluate_dim("<=maxbound", _WEL_DIMS, {"maxbound": 3}.get) + dim_value("<=maxbound", _WEL_DIMS, {"maxbound": 3}.get) -def test_evaluate_dim_snapshot_shapes(): +def test_dim_value_snapshot_shapes(): """Every array and list shape in the current DFNs evaluates, given its inputs.""" spec = Dfns.load(_DEV3_SNAPSHOT_DIR) @@ -1269,12 +1269,12 @@ def shapes(field): for field in block.fields.values(): for element in shapes(field): _bound, expr = split_bound(element) - value = evaluate_dim(expr, dims, lambda _: _One(), lambda _p, _k: 1) + value = dim_value(expr, dims, lambda _: _One(), lambda _p, _k: 1) assert value is not None, (component.name, element) @pytest.mark.parametrize( - "expr,inputs,value,expected", + "expr,inputs,length,expected", [ ("ncvert", {}, 5, ("ncvert", 5)), # a field in the row ("numalphaj", {}, 3, ("numalphaj", 3)), # an input dim @@ -1292,7 +1292,7 @@ def shapes(field): ("packagedata.ncon(ifno)", {"ifno": 1}, 2, None), # nor a row-level lookup ], ) -def test_solve_dim(expr, inputs, value, expected): +def test_dim_input(expr, inputs, length, expected): dims = { "numalphaj": "numalphaj", "nseg": "nseg", @@ -1302,12 +1302,12 @@ def test_solve_dim(expr, inputs, value, expected): "auxiliary": "len(auxiliary)", "nconn": "sum(packagedata.nlakeconn)", } - assert solve_dim(expr, dims, inputs.get, value) == expected + assert dim_input(expr, dims, inputs.get, length=length) == expected -def test_solve_dim_snapshot_shapes(): +def test_dim_input_snapshot_shapes(): """Solving any shape in the current DFNs for its sole unset input, then - evaluating it with that input set, gives back the extent.""" + evaluating it with that input set, gives back the length.""" spec = Dfns.load(_DEV3_SNAPSHOT_DIR) solved = set() for component in spec.components.values(): @@ -1326,10 +1326,19 @@ def shapes(field): for field in block.fields.values(): for element in shapes(field): _bound, expr = split_bound(element) - if (solution := solve_dim(expr, dims, {}.get, 7)) is None: + if (solution := dim_input(expr, dims, length=7)) is None: continue name, n = solution - assert evaluate_dim(expr, dims, {name: n}.get) == 7, (component.name, expr) + assert dim_value(expr, dims, {name: n}.get) == 7, (component.name, expr) solved.add(expr) assert {"ncvert", "numalphaj", "nseg-1", "maxbound"} <= solved assert "auxiliary" not in solved + + +def test_dim_defaults(): + """Without dims there are none; without a lookup no input is set.""" + assert dim_value("2 * 3") == 6 + assert dim_value("nlay") is None + assert dim_value("nlayp", {"nlayp": "nlay + 1"}) is None + assert dim_input("ncvert", length=5) == ("ncvert", 5) + assert dim_input("nlayp", {"nlayp": "nlay + 1"}, length=3) == ("nlay", 2) diff --git a/docs/md/dfns.md b/docs/md/dfns.md index 0daf2848..c99c7662 100644 --- a/docs/md/dfns.md +++ b/docs/md/dfns.md @@ -205,44 +205,44 @@ See [DFN specification](dfnspec.md) for full attribute documentation. ### Evaluating dimensions and shapes -A dim's `value` and an `Array`/`List` shape expression are both expressions over the component's input (see [`dims`](dfnspec.md#dims-inputdim) and [Dimensions](dfnspec.md#dimensions)). `evaluate_dim` evaluates either one, given the component's dim value expressions and a function looking up input field values: +A dim's `value` and an `Array`/`List` shape expression are both expressions over the component's input (see [`dims`](dfnspec.md#dims-inputdim) and [Dimensions](dfnspec.md#dimensions)). `dim_value` evaluates either one, given the component's dim value expressions and a function looking up input field values: ```python -from modflow_devtools.dfns import evaluate_dim, split_bound +from modflow_devtools.dfns import dim_value, split_bound dis = spec.components["gwf-dis"] dims = {name: dim.value for name, dim in dis.dims.items()} -evaluate_dim("nodes", dims, {"nlay": 2, "nrow": 3, "ncol": 4}.get) +dim_value("nodes", dims, {"nlay": 2, "nrow": 3, "ncol": 4}.get) # 24 ``` -A name that is a dim is evaluated in turn. Any other name, including an input dim's own name, goes to `lookup`. `sum(list.column)` passes `lookup` the dotted path and expects the column's values back, so how rows are stored is up to the caller. The result is `None` if an input it depends on isn't set. +A name that is a dim is evaluated in turn. Any other name, including an input dim's own name, goes to `lookup`. `sum(list.column)` passes `lookup` the dotted path and expects the column's values back, so how rows are stored is up to the caller. The result is `None` if an input it depends on isn't set. `dims` and `lookup` are optional: without them, there are no dims and no input is set. For an inline array's shape, `lookup` should also see the fields of the array's row, since a shape may name one (cell2d's `icvert` has shape `["ncvert"]`). A row-level lookup like `packagedata.ncon(ifno)` also needs `select`, which gets the path (`"packagedata.ncon"`) and this row's `ifno` value, and returns the value from the referenced row: ```python from collections import ChainMap -evaluate_dim("ncvert", {}, ChainMap(row, package).get) -evaluate_dim("packagedata.ncon(ifno)", dims, row.get, select) +dim_value("ncvert", lookup=ChainMap(row, package).get) +dim_value("packagedata.ncon(ifno)", dims, row.get, select) ``` A bounded shape expression (`"<=maxbound"`) is a relation, not a value. Split off the bound with `split_bound` (`("<=", "maxbound")`) and evaluate the rest. -`solve_dim` is the inverse: given the extent data actually has, it returns the unset input that extent determines, and that input's value. Use it to fill in counts the user left out: +`dim_input` is the inverse: given the length the data actually has along a dimension (how many values), it returns the unset input that length determines, and that input's value. Use it to fill in counts the user left out: ```python -from modflow_devtools.dfns import solve_dim +from modflow_devtools.dfns import dim_input -solve_dim("nseg-1", dims, {}.get, 3) # pxdp has 3 values +dim_input("nseg-1", dims, length=3) # pxdp has 3 values # ("nseg", 4) -solve_dim("ncvert", {}, {}.get, 5) # icvert has 5 values +dim_input("ncvert", length=5) # icvert has 5 values # ("ncvert", 5) -solve_dim("auxiliary", dims, {}.get, 2) # len(auxiliary): can't set names from a count +dim_input("auxiliary", dims, length=2) # len(auxiliary): can't set names from a count # None ``` -Only names, through dims, and adding or subtracting known values can be undone. The result is `None` if the expression has no unset input, more than one, or one under anything else (`len()`, `sum()`, `*`, a row-level lookup). In that case the data can only be checked, by evaluating the extent once its inputs are set. +Only names, through dims, and adding or subtracting known values can be undone. The result is `None` if the expression has no unset input, more than one, or one under anything else (`len()`, `sum()`, `*`, a row-level lookup). In that case the data can only be checked, by evaluating the length once its inputs are set. Both functions work on the expression strings alone, without a loaded `Dfns`. diff --git a/docs/md/dfnspec.md b/docs/md/dfnspec.md index 53adf80a..4719f9ca 100644 --- a/docs/md/dfnspec.md +++ b/docs/md/dfnspec.md @@ -489,7 +489,7 @@ An array appearing as a subfield of a record is called an **inline array**. Inli ###### `shape` -`[string] (default: [])`. The array's shape, as a list of shape expressions, one per dimension. An empty list means the array is 1-dimensional and **self-sizing** (see above). Each extent is exact unless its expression is prefixed with an inequality operator (e.g. `"<=n"`); see [Bounds](#bounds). +`[string] (default: [])`. The array's shape, as a list of shape expressions, one per dimension. An empty list means the array is 1-dimensional and **self-sizing** (see above). Each length is exact unless its expression is prefixed with an inequality operator (e.g. `"<=n"`); see [Bounds](#bounds). Dimensions are listed fastest-varying first, i.e. in the order elements are read from (or written to) the input file. A 3D grid array is `["ncol", "nrow", "nlay"]`, and a cellid array's `ncelldim` axis comes first (see [`cellid`](#cellid)). @@ -590,7 +590,7 @@ END PERIOD An **empty `shape`** means the list length is unconstrained at schema-definition time. This is the correct representation for any list whose length is determined at runtime, or by another component, rather than by a declared dimension. -A **non-empty `shape`** (exactly one element) relates the list's row count to a declared dimension. A bare expression means *exactly* that many rows (`["n"]`). An expression prefixed with an inequality operator is a bound (`["<=n"]`: at most `n` rows); see [Bounds](#bounds). Whether an extent is exact or a bound must be declared; it can't be inferred from the dimension's name. +A **non-empty `shape`** (exactly one element) relates the list's row count to a declared dimension. A bare expression means *exactly* that many rows (`["n"]`). An expression prefixed with an inequality operator is a bound (`["<=n"]`: at most `n` rows); see [Bounds](#bounds). Whether a length is exact or a bound must be declared; it can't be inferred from the dimension's name. **Note:** Some non-period lists in stress-type packages (e.g. `utl-spc`) also reference `maxbound`; these follow the same rule — `maxbound` must be an explicitly declared dimension for the shape to be meaningful. @@ -627,7 +627,7 @@ The `value` attribute defines the dimension as a Python expression over the comp - **`sum(list.column)`** — the sum of an integer column over a list's rows (`sum(packagedata.nlakeconn)`). - **Arithmetic expression** `nlay * nrow * ncol` — integer literals and other dims, combined with `+`, `-`, `*`, `/` (exact) and `//`. May not use bare field names; all operands must be declared dimensions. -A dimension whose value is its own name is an **input dimension**: it *is* the field, so a consumer may set the field from data the dimension sizes (e.g. `numalphaj` from the width of `cellidsj`). Every other dimension is **derived**: a function of the input, which a consumer can evaluate and check data against, but not set. More generally, a consumer may set an unset input from data whenever an extent determines it uniquely: `nseg` from the width of an array with shape `["nseg-1"]`, or a row's `ncvert` from the length of its `icvert`. `modflow_devtools.dfns` evaluates any dimension or extent with `evaluate_dim`, and finds the input an extent determines with `solve_dim`; see [Evaluating dimensions and shapes](dfns.md#evaluating-dimensions-and-shapes). +A dimension whose value is its own name is an **input dimension**: it *is* the field, so a consumer may set the field from data the dimension sizes (e.g. `numalphaj` from the width of `cellidsj`). Every other dimension is **derived**: a function of the input, which a consumer can evaluate and check data against, but not set. More generally, a consumer may set an unset input from data whenever a length determines it uniquely: `nseg` from the width of an array with shape `["nseg-1"]`, or a row's `ncvert` from the length of its `icvert`. `modflow_devtools.dfns` evaluates any dimension or shape expression with `dim_value`, and finds the input a length determines with `dim_input`; see [Evaluating dimensions and shapes](dfns.md#evaluating-dimensions-and-shapes). #### `runtime_dims` (RuntimeDim) @@ -757,11 +757,11 @@ connectiondata: #### Bounds -A shape expression gives an exact extent. Prefixed with one of the inequality operators `<`, `<=`, `>` or `>=`, it gives a bound on the extent instead: `"<=n"` means at most `n`. Any shape expression may be bounded, including a derived dimension, arithmetic offset, or row-level column lookup (`"<=block.column(fk_field)"`), with at most one operator per extent. Bounds apply to `list` and `array` shapes. Memory variable shapes can't be bounded; MF6 allocates memory arrays at a definite size. +A shape expression gives an exact length. Prefixed with one of the inequality operators `<`, `<=`, `>` or `>=`, it gives a bound on the length instead: `"<=n"` means at most `n`. Any shape expression may be bounded, including a derived dimension, arithmetic offset, or row-level column lookup (`"<=block.column(fk_field)"`), with at most one operator per length. Bounds apply to `list` and `array` shapes. Memory variable shapes can't be bounded; MF6 allocates memory arrays at a definite size. -An unprefixed extent is exact, so a DFN must mark every bound. A consumer may reject input that doesn't satisfy the relation: too many or too few rows for an exact extent, too many for an upper bound, too few for a lower bound. +An unprefixed length is exact, so a DFN must mark every bound. A consumer may reject input that doesn't satisfy the relation: too many or too few rows for an exact length, too many for an upper bound, too few for a lower bound. -`modflow_devtools.dfns` evaluates shape expressions with `evaluate_dim`; see [Evaluating dimensions and shapes](dfns.md#evaluating-dimensions-and-shapes). +`modflow_devtools.dfns` evaluates shape expressions with `dim_value`; see [Evaluating dimensions and shapes](dfns.md#evaluating-dimensions-and-shapes). #### Dimension scope diff --git a/modflow_devtools/dfns/__init__.py b/modflow_devtools/dfns/__init__.py index caa6c247..1c6ee547 100644 --- a/modflow_devtools/dfns/__init__.py +++ b/modflow_devtools/dfns/__init__.py @@ -32,8 +32,8 @@ Simulation, String, Union, - evaluate_dim, - solve_dim, + dim_input, + dim_value, split_bound, ) @@ -79,9 +79,9 @@ "Simulation", "String", "Union", - "evaluate_dim", + "dim_input", + "dim_value", "fetch_dfns", "migrate", - "solve_dim", "split_bound", ] diff --git a/modflow_devtools/dfns/schema.py b/modflow_devtools/dfns/schema.py index b9362ea4..7e53b6ec 100644 --- a/modflow_devtools/dfns/schema.py +++ b/modflow_devtools/dfns/schema.py @@ -754,11 +754,11 @@ def _resolve_derived_dims(component: "ComponentBase", known_dims: set[str]) -> l class _DimExprs: """A component's dim value expressions, evaluated (and solved) over input - field values from ``lookup``; see :func:`evaluate_dim`.""" + field values from ``lookup``; see :func:`dim_value`.""" - def __init__(self, dims: Mapping[str, str], lookup: Callable[[str], Any]): - self.dims = dims - self.lookup = lookup + def __init__(self, dims: Mapping[str, str] | None, lookup: Callable[[str], Any] | None): + self.dims = dims or {} + self.lookup = lookup or (lambda _: None) def parse(self, expr: str) -> ast.expr: if _BOUND_RE.match(expr): @@ -843,10 +843,10 @@ def solve( return None -def evaluate_dim( +def dim_value( expr: str, - dims: Mapping[str, str], - lookup: Callable[[str], Any], + dims: Mapping[str, str] | None = None, + lookup: Callable[[str], Any] | None = None, select: Callable[[str, Any], Any] | None = None, ) -> int | None: """ @@ -859,7 +859,8 @@ def evaluate_dim( gets an input field's value by name, or by dotted path for ``sum(list.column)`` (an iterable of the column's values), and returns None if the field isn't set. To evaluate an inline array's shape, ``lookup`` - should see the fields of the array's row too. + should see the fields of the array's row too. Without ``dims`` there are + no dims; without ``lookup`` no input is set. A name that is a dim is evaluated in turn; any other name, like an input dim's own name, a field in the row, or a name not in ``dims``, is passed to @@ -873,28 +874,30 @@ def evaluate_dim( Returns None if an input it depends on isn't set. Raises ValueError for a malformed expression, a cycle, or an inexact ``/``. """ + exprs = _DimExprs(dims, lookup) if m := _LOOKUP_RE.fullmatch(expr): component_ref, block_name, col_name, fk_field_name = m.groups() if select is None: raise ValueError(f"{expr!r} is a row-level lookup, which needs select") - key = lookup(fk_field_name) + key = exprs.lookup(fk_field_name) if key is None: return None v = select(".".join(filter(None, (component_ref, block_name, col_name))), key) return None if v is None else int(v) - exprs = _DimExprs(dims, lookup) return exprs.evaluate(exprs.parse(expr), frozenset(), expr) -def solve_dim( +def dim_input( expr: str, - dims: Mapping[str, str], - lookup: Callable[[str], Any], - value: int, + dims: Mapping[str, str] | None = None, + lookup: Callable[[str], Any] | None = None, + *, + length: int, ) -> tuple[str, int] | None: """ - The inverse of :func:`evaluate_dim`: the unset input, and its value, that - makes ``expr`` evaluate to ``value``. Arguments are as for ``evaluate_dim``. + The inverse of :func:`dim_value`: the unset input, and its value, that + gives ``expr`` the given ``length``: the number of values the data has + along that dimension. Other arguments are as for ``dim_value``. E.g. an inline array with shape ``["nseg-1"]`` and 3 values, with ``nseg`` unset, gives ``("nseg", 4)``; one with shape ``["ncvert"]`` and 5 values, @@ -903,13 +906,13 @@ def solve_dim( Only names, through dims, and ``+``/``-`` of known values can be undone. Returns None if ``expr`` has no unset input, more than one, or one under anything else, like ``len()``, ``sum()``, ``*`` or a row-level lookup, and - the caller can only check ``evaluate_dim`` against ``value`` once its + the caller can only check ``dim_value`` against ``length`` once its inputs are set. """ if _LOOKUP_RE.fullmatch(expr): return None exprs = _DimExprs(dims, lookup) - return exprs.solve(exprs.parse(expr), value, frozenset(), expr) + return exprs.solve(exprs.parse(expr), length, frozenset(), expr) class BlockHeader(BaseModel): @@ -1175,9 +1178,9 @@ def split_bound(element: str) -> "tuple[str | None, str]": """ Split a shape element into its bound operator and the expression it bounds. - A bare element (``"nper"``) is an exact extent; an element prefixed with an - inequality operator (``"<=maxbound"``) bounds the extent instead. Returns - ``(operator, expression)``, with ``operator`` ``None`` for an exact extent. + A bare element (``"nper"``) is an exact length; an element prefixed with an + inequality operator (``"<=maxbound"``) bounds the length instead. Returns + ``(operator, expression)``, with ``operator`` ``None`` for an exact length. """ if m := _BOUND_RE.match(element): return m.group(), element[m.end() :].strip() @@ -1199,7 +1202,7 @@ class _ShapeRef: name: str # arithmetic offset: "nseg-1" -> -1 offset: int = 0 - # "<", "<=", ">" or ">=", or None for an exact extent + # "<", "<=", ">" or ">=", or None for an exact length bound: str | None = None # lookup only: "[component.]block.column(fk_field)" component: str | None = None From c4d69dbdbac32946107aaf930ff3b41189baab81 Mon Sep 17 00:00:00 2001 From: wpbonelli Date: Tue, 6 Oct 2026 07:46:04 -0700 Subject: [PATCH 6/6] docs(dfns): describe dim_input as finding the input to sync Co-Authored-By: Claude Opus 5.5 --- docs/md/dfns.md | 6 +++--- modflow_devtools/dfns/schema.py | 22 +++++++++------------- 2 files changed, 12 insertions(+), 16 deletions(-) diff --git a/docs/md/dfns.md b/docs/md/dfns.md index c99c7662..df3e737e 100644 --- a/docs/md/dfns.md +++ b/docs/md/dfns.md @@ -229,7 +229,7 @@ dim_value("packagedata.ncon(ifno)", dims, row.get, select) A bounded shape expression (`"<=maxbound"`) is a relation, not a value. Split off the bound with `split_bound` (`("<=", "maxbound")`) and evaluate the rest. -`dim_input` is the inverse: given the length the data actually has along a dimension (how many values), it returns the unset input that length determines, and that input's value. Use it to fill in counts the user left out: +`dim_input` is the inverse. Rather than plug inputs in, it finds the input to plug in so the expression gives the data's length, i.e. the input to sync to the data: ```python from modflow_devtools.dfns import dim_input @@ -238,11 +238,11 @@ dim_input("nseg-1", dims, length=3) # pxdp has 3 values # ("nseg", 4) dim_input("ncvert", length=5) # icvert has 5 values # ("ncvert", 5) -dim_input("auxiliary", dims, length=2) # len(auxiliary): can't set names from a count +dim_input("auxiliary", dims, length=2) # len(auxiliary): no input to sync # None ``` -Only names, through dims, and adding or subtracting known values can be undone. The result is `None` if the expression has no unset input, more than one, or one under anything else (`len()`, `sum()`, `*`, a row-level lookup). In that case the data can only be checked, by evaluating the length once its inputs are set. +It returns `None` when there's nothing to sync: no unset input, more than one, or one it can't solve for (under `len()`, `sum()`, `*` or a row-level lookup). Then the data can only be checked with `dim_value`. Both functions work on the expression strings alone, without a loaded `Dfns`. diff --git a/modflow_devtools/dfns/schema.py b/modflow_devtools/dfns/schema.py index 7e53b6ec..949701f9 100644 --- a/modflow_devtools/dfns/schema.py +++ b/modflow_devtools/dfns/schema.py @@ -895,19 +895,15 @@ def dim_input( length: int, ) -> tuple[str, int] | None: """ - The inverse of :func:`dim_value`: the unset input, and its value, that - gives ``expr`` the given ``length``: the number of values the data has - along that dimension. Other arguments are as for ``dim_value``. - - E.g. an inline array with shape ``["nseg-1"]`` and 3 values, with ``nseg`` - unset, gives ``("nseg", 4)``; one with shape ``["ncvert"]`` and 5 values, - with the row's ``ncvert`` unset, gives ``("ncvert", 5)``. - - Only names, through dims, and ``+``/``-`` of known values can be undone. - Returns None if ``expr`` has no unset input, more than one, or one under - anything else, like ``len()``, ``sum()``, ``*`` or a row-level lookup, and - the caller can only check ``dim_value`` against ``length`` once its - inputs are set. + The inverse of :func:`dim_value`: rather than plug inputs into ``expr``, + find the input to plug in so that ``expr`` gives ``length``, the number of + values the data has. Returns ``(name, value)``, the input to sync to the + data, or None if there is none: no unset input, more than one, or one + that can't be solved for (under ``len()``, ``sum()``, ``*`` or a + row-level lookup). Then the data can only be checked with ``dim_value``. + + E.g. ``"nseg-1"`` with 3 values gives ``("nseg", 4)``. Other arguments are + as for ``dim_value``. """ if _LOOKUP_RE.fullmatch(expr): return None