From cdc358a15f2a711585479cda91b215755be361f6 Mon Sep 17 00:00:00 2001 From: thodson-usgs Date: Tue, 29 Sep 2026 09:31:17 -0500 Subject: [PATCH] fix(waterdata): send the qualifier filter as a JSON array The Water Data API parses qualifier as a JSON array, so it responded with HTTP 400 to every value the getters sent (qualifier=ICE, or ICE,ESTIMATED for a list). _get_args now encodes qualifier as one compact JSON array: "ICE" becomes ["ICE"] and a list becomes one array. This also covers qualifier passed to get_peaks through **queryables. A string that starts with "[" is sent unchanged, and an empty list is still dropped from the request. The service matches the whole list, in order: "ICE" does not match an observation qualified ["ESTIMATED", "ICE"]. It responds with 400 to the CQL2 array operators, so there is no "contains" match; the qualifier docstrings now state the exact-match rule. Co-Authored-By: Claude Opus 5.5 (1M context) --- NEWS.md | 2 ++ dataretrieval/waterdata/measurements.py | 2 ++ dataretrieval/waterdata/time_series.py | 8 +++++ dataretrieval/waterdata/utils.py | 25 ++++++++++++++-- tests/waterdata_test.py | 40 +++++++++++++++++++++++++ 5 files changed, 75 insertions(+), 2 deletions(-) diff --git a/NEWS.md b/NEWS.md index 11657fb2..c48e2337 100644 --- a/NEWS.md +++ b/NEWS.md @@ -2,6 +2,8 @@ **10/01/2026:** **Bug fix:** `waterdata.get_nearest_continuous` turned a missing target (`NaT`, `None`, `NaN`, or `""`) into a `(time >= 'nan' AND time <= 'nan')` clause and sent it to the service, including when only one entry of a list was missing. It now raises `ValueError` before any request, naming how many targets are missing and the position of the first. Targets pandas cannot parse now raise a `ValueError` that names `targets`, rather than pandas' message, which named no argument and suggested a `format=` the getter does not accept. **Bug fix:** the same getter's `window` was not validated: a missing window (`None`, `'NaT'`) failed with an `AttributeError` while the filter was built, and a negative one (`'-PT5M'`) inverted every bound, so the query matched nothing and returned an empty frame indistinguishable from a gap in the data. Both now raise `ValueError` naming `window`; a zero window remains an exact-match query. **Behavior change:** a bare-number `window` now raises `TypeError`; pandas read `window=450` as 450 nanoseconds, so it matched almost nothing. Pass a duration such as `window='PT450S'`. Numeric `targets` raise `TypeError` for the same reason: `targets=[1.5e9]` meant as epoch seconds was read as 1970-01-01T00:00:01.5Z. **Behavior change:** calls that previously sent a malformed filter now fail locally, so a caller passing targets with gaps must drop or fill them first. +**09/29/2026:** **Bug fix:** the `qualifier` filter of the `waterdata` getters (`get_daily()`, `get_continuous()`, `get_latest_daily()`, `get_latest_continuous()`, `get_field_measurements()`, and `get_peaks()` through `**queryables`) responded with HTTP 400 to every value, because the service parses the value as a JSON array. It is now sent as one: `qualifier="ICE"` sends `["ICE"]`, and a list is sent as one array. The service matches the whole list, in order, so `"ICE"` does not match an observation qualified `["ESTIMATED", "ICE"]`. A value that is already a JSON array is sent unchanged. + **09/22/2026:** `waterdata.get_continuous()` accepts `method_category`, a column USGS added to the `continuous` collection in September 2026: the RLMS method category code (`STNRD`, `LMTUS`, `EXPER` or `UNKWN`) for the method in effect over an observation's interval. It is returned on every record and is null for time series that have not been categorized. It could already be passed through `**queryables`; it is now a documented parameter. `get_latest_continuous()` is unchanged, because `latest-continuous` does not have the field. **09/22/2026:** The Water Data OGC getters now request **v1** of the Water Data APIs (`api.waterdata.usgs.gov/ogcapi/v1`), [released September 2026](https://waterdata.usgs.gov/blog/api-v1-release/); v0 stays online until June 2027. **New:** `WaterdataConfiguration(api_version="v0")`, or `api_version = "v0"` in the `[waterdata]` table of the configuration file, pins v0 during the transition. It changes only the version segment of the OGC path; Samples, Statistics and STAC are versioned separately and have no v1. **Behavior change:** `waterdata.get_time_series_metadata()` returns `begin` and `end` in UTC with a time zone, and no longer returns `begin_utc`, `end_utc`, `state_name` or `hydrologic_unit_code`. **Deprecation:** passing `begin_utc`, `end_utc`, `state`, `state_name` or `hydrologic_unit_code` to that getter, as a filter or in `properties`, emits a `DeprecationWarning` and sends the call to v0; this may be removed on or after 2027-06-01. Use `begin`, `end`, and `get_combined_metadata()` instead. **Behavior change:** `waterdata.get_field_measurements()` returns `time` as a date rather than a datetime, parsed to a tz-naive midnight timestamp as `get_daily()` already does; the time of day is in `time_of_day`. The `field-measurements-metadata` collection has no `time` field and is unaffected. **Bug fix:** `construction_date`, from `waterdata.get_monitoring_locations()` and `waterdata.get_combined_metadata()`, keeps every value. The service records it at day, month, or year precision (`19950812`, `199508`, `2005`); parsing it as a datetime turned the month-precision values into `NaT` and raised pandas' "Could not infer format" `UserWarning`. **Behavior change:** the column now holds those strings as sent rather than datetimes. diff --git a/dataretrieval/waterdata/measurements.py b/dataretrieval/waterdata/measurements.py index 2b978ab9..0e95986b 100644 --- a/dataretrieval/waterdata/measurements.py +++ b/dataretrieval/waterdata/measurements.py @@ -95,6 +95,8 @@ def get_field_measurements( qualifier : string or iterable of strings, optional Any qualifiers associated with an observation, for instance whether a sensor may have been impacted by ice or whether values were estimated. + Matches observations whose qualifiers are exactly these, in this order: + ``"ICE"`` matches ``["ICE"]`` but not ``["ESTIMATED", "ICE"]``. value : string or iterable of strings, optional The value of the observation. Values are transmitted as strings in the JSON response format to preserve precision. diff --git a/dataretrieval/waterdata/time_series.py b/dataretrieval/waterdata/time_series.py index 587eb8dc..8eec3aa4 100644 --- a/dataretrieval/waterdata/time_series.py +++ b/dataretrieval/waterdata/time_series.py @@ -112,6 +112,8 @@ def get_daily( qualifier : string or iterable of strings, optional Any qualifiers associated with an observation, for instance whether a sensor may have been impacted by ice or whether values were estimated. + Matches observations whose qualifiers are exactly these, in this order: + ``"ICE"`` matches ``["ICE"]`` but not ``["ESTIMATED", "ICE"]``. value : string or iterable of strings, optional The value of the observation. Values are transmitted as strings in the JSON response format to preserve precision. @@ -347,6 +349,8 @@ def get_continuous( qualifier : string or iterable of strings, optional Any qualifiers associated with an observation, for instance whether a sensor may have been impacted by ice or whether values were estimated. + Matches observations whose qualifiers are exactly these, in this order: + ``"ICE"`` matches ``["ICE"]`` but not ``["ESTIMATED", "ICE"]``. method_category : string or iterable of strings, optional The RLMS method category code for the method in effect over the observation's interval: "STNRD" (standardized, with known uncertainty @@ -549,6 +553,8 @@ def get_latest_continuous( qualifier : string or iterable of strings, optional Any qualifiers associated with an observation, for instance whether a sensor may have been impacted by ice or whether values were estimated. + Matches observations whose qualifiers are exactly these, in this order: + ``"ICE"`` matches ``["ICE"]`` but not ``["ESTIMATED", "ICE"]``. value : string or iterable of strings, optional The value of the observation. Values are transmitted as strings in the JSON response format to preserve precision. @@ -762,6 +768,8 @@ def get_latest_daily( qualifier : string or iterable of strings, optional Any qualifiers associated with an observation, for instance whether a sensor may have been impacted by ice or whether values were estimated. + Matches observations whose qualifiers are exactly these, in this order: + ``"ICE"`` matches ``["ICE"]`` but not ``["ESTIMATED", "ICE"]``. value : string or iterable of strings, optional The value of the observation. Values are transmitted as strings in the JSON response format to preserve precision. diff --git a/dataretrieval/waterdata/utils.py b/dataretrieval/waterdata/utils.py index 616a9957..13ddadb7 100644 --- a/dataretrieval/waterdata/utils.py +++ b/dataretrieval/waterdata/utils.py @@ -16,6 +16,7 @@ from __future__ import annotations import functools +import json from collections.abc import Callable, Mapping from typing import TYPE_CHECKING, Any, TypeVar @@ -128,6 +129,21 @@ ) +#: Parameters the service parses as a JSON array and matches against the whole +#: list, in order: ``qualifier=ICE`` responds with 400, and ``qualifier=["ICE"]`` +#: does not match ``["ESTIMATED", "ICE"]`` (v0 and v1, probed 2026-09-29). +_JSON_ARRAY_PARAMS = frozenset({"qualifier"}) + + +def _as_json_array(value: Any) -> str: + """*value* as a compact JSON array. A string that starts with ``[`` is sent as + is, so a caller who worked around the 400 keeps the same request.""" + if isinstance(value, str) and value.lstrip().startswith("["): + return value + items = list(value) if isinstance(value, (list, tuple)) else [value] + return json.dumps(items, separators=(",", ":")) + + def _flatten_queryables(local_vars: dict[str, Any]) -> dict[str, Any]: """Merge a getter's ``**queryables`` passthrough kwargs into ``local_vars``. @@ -161,12 +177,17 @@ def _get_args( Adds the Water Data API's extra no-normalize params (numeric params such as ``water_year``, ``thresholds``, ``boundingBox``) so they keep their element types. Also flattens any ``**queryables`` passthrough (see - :func:`_flatten_queryables`). + :func:`_flatten_queryables`) and encodes :data:`_JSON_ARRAY_PARAMS`. """ _flatten_queryables(local_vars) - return prepare_request_args( + args = prepare_request_args( local_vars, exclude, extra_no_normalize=_NO_NORMALIZE_PARAMS ) + for name in _JSON_ARRAY_PARAMS & args.keys(): + # An empty list is dropped from the request downstream, as for any filter. + if args[name] != []: + args[name] = _as_json_array(args[name]) + return args def _with_state(local_vars: dict[str, Any], *, to: str, into: str) -> dict[str, Any]: diff --git a/tests/waterdata_test.py b/tests/waterdata_test.py index bae73590..b94dfc1d 100644 --- a/tests/waterdata_test.py +++ b/tests/waterdata_test.py @@ -797,6 +797,46 @@ def test_get_daily_value_is_float_when_every_value_is_whole(httpx_mock): assert df["value"].dtype == "float64" +@pytest.mark.parametrize( + ("getter", "collection", "qualifier", "sent"), + [ + (get_daily, "daily", "ICE", ['["ICE"]']), + (get_daily, "daily", ["ESTIMATED", "ICE"], ['["ESTIMATED","ICE"]']), + (get_daily, "daily", '["ESTIMATED", "ICE"]', ['["ESTIMATED", "ICE"]']), + (get_daily, "daily", [], None), + (get_peaks, "peaks", "DIFFDATUM", ['["DIFFDATUM"]']), + ], + ids=["string", "list", "json-literal", "empty-list", "peaks-queryables"], +) +def test_qualifier_is_sent_as_a_json_array( + httpx_mock, getter, collection, qualifier, sent +): + """The service parses ``qualifier`` as JSON and responds with 400 to a bare + word.""" + _mock_items(httpx_mock, collection) + + getter(monitoring_location_id="USGS-05427718", qualifier=qualifier) + + assert _sent(httpx_mock, collection)[0].get("qualifier") == sent + + +@pytest.mark.live +def test_qualifier_filter_matches_the_whole_list_in_order(): + kwargs = { + "monitoring_location_id": "USGS-05420500", + "parameter_code": "00060", + "statistic_id": "00003", + "time": "2022-12-01/2023-03-31", + "skip_geometry": True, + } + both, _ = get_daily(qualifier=["ESTIMATED", "ICE"], **kwargs) + reversed_, _ = get_daily(qualifier=["ICE", "ESTIMATED"], **kwargs) + + assert len(both) > 0 + assert all(q == ["ESTIMATED", "ICE"] for q in both["qualifier"]) + assert reversed_.empty + + def test_get_daily_sends_date_only_time_interval(httpx_mock): """The Water Data dialect marks ``daily`` date-only, so an open-ended interval goes out as ``2025-01-01/..`` with no time component."""