From 3af4af5f8800cafe2ed49ae75da4551791887c62 Mon Sep 17 00:00:00 2001 From: thodson-usgs Date: Tue, 22 Sep 2026 21:29:42 -0500 Subject: [PATCH] fix(waterdata): accept the continuous method_category queryable USGS added method_category to the continuous collection in September 2026. get_continuous now takes it as a named parameter and documents it as a returned column. Add live tests that each Water Data family's default API version matches the one waterdata/endpoints.py requests. --- NEWS.md | 2 + dataretrieval/waterdata/time_series.py | 10 ++- tests/contracts/README.md | 4 +- tests/data/waterdata_ogc_fixtures.json | 18 +++-- tests/waterdata_endpoints_test.py | 108 +++++++++++++++++++++++++ tests/waterdata_test.py | 15 ++++ 6 files changed, 146 insertions(+), 11 deletions(-) create mode 100644 tests/waterdata_endpoints_test.py diff --git a/NEWS.md b/NEWS.md index 61d8713c..11657fb2 100644 --- a/NEWS.md +++ b/NEWS.md @@ -2,6 +2,8 @@ **10/01/2026:** **Bug fix:** `waterdata.get_nearest_continuous` turned a missing target (`NaT`, `None`, `NaN`, or `""`) into a `(time >= 'nan' AND time <= 'nan')` clause and sent it to the service, including when only one entry of a list was missing. It now raises `ValueError` before any request, naming how many targets are missing and the position of the first. Targets pandas cannot parse now raise a `ValueError` that names `targets`, rather than pandas' message, which named no argument and suggested a `format=` the getter does not accept. **Bug fix:** the same getter's `window` was not validated: a missing window (`None`, `'NaT'`) failed with an `AttributeError` while the filter was built, and a negative one (`'-PT5M'`) inverted every bound, so the query matched nothing and returned an empty frame indistinguishable from a gap in the data. Both now raise `ValueError` naming `window`; a zero window remains an exact-match query. **Behavior change:** a bare-number `window` now raises `TypeError`; pandas read `window=450` as 450 nanoseconds, so it matched almost nothing. Pass a duration such as `window='PT450S'`. Numeric `targets` raise `TypeError` for the same reason: `targets=[1.5e9]` meant as epoch seconds was read as 1970-01-01T00:00:01.5Z. **Behavior change:** calls that previously sent a malformed filter now fail locally, so a caller passing targets with gaps must drop or fill them first. +**09/22/2026:** `waterdata.get_continuous()` accepts `method_category`, a column USGS added to the `continuous` collection in September 2026: the RLMS method category code (`STNRD`, `LMTUS`, `EXPER` or `UNKWN`) for the method in effect over an observation's interval. It is returned on every record and is null for time series that have not been categorized. It could already be passed through `**queryables`; it is now a documented parameter. `get_latest_continuous()` is unchanged, because `latest-continuous` does not have the field. + **09/22/2026:** The Water Data OGC getters now request **v1** of the Water Data APIs (`api.waterdata.usgs.gov/ogcapi/v1`), [released September 2026](https://waterdata.usgs.gov/blog/api-v1-release/); v0 stays online until June 2027. **New:** `WaterdataConfiguration(api_version="v0")`, or `api_version = "v0"` in the `[waterdata]` table of the configuration file, pins v0 during the transition. It changes only the version segment of the OGC path; Samples, Statistics and STAC are versioned separately and have no v1. **Behavior change:** `waterdata.get_time_series_metadata()` returns `begin` and `end` in UTC with a time zone, and no longer returns `begin_utc`, `end_utc`, `state_name` or `hydrologic_unit_code`. **Deprecation:** passing `begin_utc`, `end_utc`, `state`, `state_name` or `hydrologic_unit_code` to that getter, as a filter or in `properties`, emits a `DeprecationWarning` and sends the call to v0; this may be removed on or after 2027-06-01. Use `begin`, `end`, and `get_combined_metadata()` instead. **Behavior change:** `waterdata.get_field_measurements()` returns `time` as a date rather than a datetime, parsed to a tz-naive midnight timestamp as `get_daily()` already does; the time of day is in `time_of_day`. The `field-measurements-metadata` collection has no `time` field and is unaffected. **Bug fix:** `construction_date`, from `waterdata.get_monitoring_locations()` and `waterdata.get_combined_metadata()`, keeps every value. The service records it at day, month, or year precision (`19950812`, `199508`, `2005`); parsing it as a datetime turned the month-precision values into `NaT` and raised pandas' "Could not infer format" `UserWarning`. **Behavior change:** the column now holds those strings as sent rather than datetimes. **09/09/2026:** **Bug fix:** code and identifier columns keep their leading zeros. A bare `pandas.read_csv` infers a zero-padded code as a number, so `waterdata.get_samples()` returned parameter code `00060` as `60` and HUC12 `070700050502` as `70700050502`, and `nwis.get_info()` returned `huc_cd` `02060005` as `2060005`. One rule now decides what a code column is — a name ending in `code`, the RDB abbreviation `_cd`, or a name containing `identifier`, `huc`, or `fips` — and every delimited response is parsed through it: the Samples and WQP CSV readers, `rdb.read_rdb` (which reads the names from the RDB header rather than the caller listing them), and the Water Use CSV pages. **Behavior change:** these columns now hold strings. `waterdata.get_samples()`: `USGSpcode`, `Location_HUCEightDigitCode`, `Location_HUCTwelveDigitCode`, `SampleCollectionMethod_Identifier` (`get_samples_summary()` shares the parse; no column in its current profile was affected). `nwis.get_info()`, `nwis.what_sites()`, and `nwis.get_record(service="site")`: `huc_cd`, `state_cd`, `county_cd`, `district_cd`. A comparison against a number — `df["USGSpcode"] == 60` — or a merge onto a numeric key now matches nothing instead of raising, so compare against the padded string (`== "00060"`) or call `.astype(int)` where the number is what you want. **Behavior change:** a count whose name reads as an identifier is numeric again. WQP's `AlternateLocation_IdentifierCount` has been read as text since 05/31/2026 because "Identifier" appears in its name; a name ending in `count` is now excluded from the rule, so the same column has one dtype in every service that reports it. Measurement columns are unchanged, and the `waterdata` OGC getters and `ngwmn` were never affected: their JSON responses deliver codes as strings and numeric coercion there is limited to a fixed list of measurement columns. **Correction to the 1.2.0 notes:** the same fix was applied to the nine `wqp` getters on 05/31/2026 and never recorded here — `wqp.get_results()` and the `what_*` getters have returned HUCs, parameter codes, and FIPS codes as strings since that release. diff --git a/dataretrieval/waterdata/time_series.py b/dataretrieval/waterdata/time_series.py index 6f97b208..587eb8dc 100644 --- a/dataretrieval/waterdata/time_series.py +++ b/dataretrieval/waterdata/time_series.py @@ -267,6 +267,7 @@ def get_continuous( approval_status: str | Iterable[str] | None = None, unit_of_measure: str | Iterable[str] | None = None, qualifier: str | Iterable[str] | None = None, + method_category: str | Iterable[str] | None = None, value: str | Iterable[str] | None = None, last_modified: str | Iterable[str] | None = None, time: str | Iterable[str] | None = None, @@ -316,7 +317,8 @@ def get_continuous( The columns to return from the query. Available options are: geometry, id, time_series_id, monitoring_location_id, parameter_code, statistic_id, time, value, - unit_of_measure, approval_status, qualifier, last_modified + unit_of_measure, approval_status, qualifier, method_category, + last_modified time_series_id : string or iterable of strings, optional A unique identifier representing a single time series, corresponding to the id field in the time-series-metadata endpoint. @@ -345,6 +347,12 @@ def get_continuous( qualifier : string or iterable of strings, optional Any qualifiers associated with an observation, for instance whether a sensor may have been impacted by ice or whether values were estimated. + method_category : string or iterable of strings, optional + The RLMS method category code for the method in effect over the + observation's interval: "STNRD" (standardized, with known uncertainty + and full QA/QC), "LMTUS" (limited use: a modified or externally + sourced method), "EXPER" (experimental), or "UNKWN" (unknown). + Null for time series that have not been categorized. value : string or iterable of strings, optional The value of the observation. Values are transmitted as strings in the JSON response format to preserve precision. diff --git a/tests/contracts/README.md b/tests/contracts/README.md index c985f9da..3599c99c 100644 --- a/tests/contracts/README.md +++ b/tests/contracts/README.md @@ -9,8 +9,8 @@ The suite uses four dependency-oriented layers without moving established tests: `wqp_test.py`, `nldi_test.py`, `streamstats_test.py`): service request construction, response parsing, and documented protocol behavior. - **Component** (`transport_test.py`, `waterdata_chunking_test.py`, - `waterdata_queryables_test.py`, `rdb_test.py`, `_csv_test.py`): one internal - responsibility in isolation. + `waterdata_queryables_test.py`, `waterdata_endpoints_test.py`, `rdb_test.py`, + `_csv_test.py`): one internal responsibility in isolation. - **Cross-component** (`architecture_test.py`, `headers_host_scoping_test.py`, `waterdata_progress_test.py`): dependency fitness functions and behavior that spans adapters, OGC, transport, or security boundaries. diff --git a/tests/data/waterdata_ogc_fixtures.json b/tests/data/waterdata_ogc_fixtures.json index 20731948..c93bfcdd 100644 --- a/tests/data/waterdata_ogc_fixtures.json +++ b/tests/data/waterdata_ogc_fixtures.json @@ -265,18 +265,19 @@ ], "type": "Point" }, - "id": "1f6dacef-9405-4e72-a755-6d3ff6121051", + "id": "c7ab17e0-14cb-4998-b532-95353f75e6ee", "properties": { "approval_status": "Approved", - "last_modified": "2025-08-28T11:10:36.529563+00:00", + "last_modified": "2026-01-08T21:21:09.233391+00:00", + "method_category": "UNKWN", "monitoring_location_id": "USGS-06904500", "parameter_code": "00065", "qualifier": null, "statistic_id": "00011", - "time": "2025-01-01T00:00:00+00:00", + "time": "2025-09-22T16:00:00+00:00", "time_series_id": "b36569a0067443ac9425e850a3ac7baa", "unit_of_measure": "ft", - "value": "1.76" + "value": "0.39" }, "type": "Feature" }, @@ -288,18 +289,19 @@ ], "type": "Point" }, - "id": "0be08516-f3a0-4a60-b3bf-41fba6b0a3ea", + "id": "f76707d5-2a10-4de0-99c3-65b63b4a92db", "properties": { "approval_status": "Approved", - "last_modified": "2025-08-28T11:10:36.529563+00:00", + "last_modified": "2026-01-08T21:21:09.233391+00:00", + "method_category": "UNKWN", "monitoring_location_id": "USGS-06904500", "parameter_code": "00065", "qualifier": null, "statistic_id": "00011", - "time": "2025-01-01T00:15:00+00:00", + "time": "2025-09-22T16:15:00+00:00", "time_series_id": "b36569a0067443ac9425e850a3ac7baa", "unit_of_measure": "ft", - "value": "1.76" + "value": "0.39" }, "type": "Feature" } diff --git a/tests/waterdata_endpoints_test.py b/tests/waterdata_endpoints_test.py new file mode 100644 index 00000000..6d3e8959 --- /dev/null +++ b/tests/waterdata_endpoints_test.py @@ -0,0 +1,108 @@ +"""Live monitors for the API version each Water Data family serves. + +``waterdata/endpoints.py`` puts a version in the OGC, STAC and statistics +paths. An old version keeps responding after USGS publishes a new one, so no +other test fails when that happens. (Samples and NGWMN have no version segment.) + +Two conditions are checked, in separate tests for OGC and STAC and in one +test for statistics: + +- **A new version is available**: the family's default is not the version the + package requests. +- **The version the package requests stopped working**: it no longer returns + data. + +The OGC and STAC roots publish a ``self`` link naming their default version, +so the version is read from it. OGC cannot be probed, because it responds with +200 and an empty body for any version segment (checked 2026-09-22). Statistics +has no root document, so it is probed; a missing statistics version responds +with 404. +""" + +import re + +import httpx +import pytest + +from dataretrieval.waterdata import endpoints + +#: The families whose root document names the default version and that have a +#: ``/collections`` endpoint, with the function that builds the requested URL. +_FAMILIES = { + "ogcapi": endpoints.ogc_api_url, + "stac": endpoints.ratings_catalog_url, +} + +#: A version segment anywhere in a path: ``/v0``, ``/v12/``, ``/v1?f=json``. +_VERSION_RE = re.compile(r"/(v\d+)(?=[/?#]|$)") + +#: Fail a hung request before the scheduled job's own timeout does. +_TIMEOUT = 60 + + +def _split_version(url: str) -> tuple[str, str]: + """Split *url* into its unversioned root and its version segment.""" + match = _VERSION_RE.search(url) + assert match is not None, f"no version segment in {url!r}" + return url[: match.start()] + "/", match.group(1) + + +def _served_version(root: str) -> str: + """The version in *root*'s ``self`` link, e.g. ``.../ogcapi/v1?f=json``.""" + response = httpx.get(root, timeout=_TIMEOUT, follow_redirects=True) + response.raise_for_status() + links = response.json().get("links") or [] + self_links = [link["href"] for link in links if link.get("rel") == "self"] + assert self_links, f"{root} published no self link: {links}" + return _split_version(self_links[0])[1] + + +@pytest.mark.live +@pytest.mark.parametrize("family", sorted(_FAMILIES)) +def test_service_default_is_the_version_this_package_requests(family): + root, requested = _split_version(_FAMILIES[family]()) + served = _served_version(root) + + assert served == requested, ( + f"the {family} API now serves {served} by default; this package requests " + f"{requested}. Move the pin in waterdata/endpoints.py to {served}, after " + f"checking the {served} release notes for dropped or renamed fields." + ) + + +@pytest.mark.live +def test_statistics_has_published_no_version_beyond_the_one_we_request(): + """``/statistics/vN`` responds with 404 even when vN exists, so the probe + requests ``/statistics/vN/docs``.""" + url = endpoints.statistics_api_url() + root, current = _split_version(url) + following = f"v{int(current.removeprefix('v')) + 1}" + + assert httpx.get(f"{url}/docs", timeout=_TIMEOUT).status_code == 200, ( + f"the statistics service stopped serving {current}, which this package " + "requests; check what replaced it." + ) + + probe = httpx.get(f"{root}{following}/docs", timeout=_TIMEOUT) + assert probe.status_code == 404, ( + f"the statistics service now responds to {following}/docs with HTTP " + f"{probe.status_code}; check whether the package should move to " + f"{following}, and whether the service publishes a root document from " + "which the version can be read instead of probed." + ) + + +@pytest.mark.live +@pytest.mark.parametrize("family", sorted(_FAMILIES)) +def test_the_version_this_package_requests_still_returns_data(family): + """Checked on content, not status: OGC responds with 200 and an empty body + for a version that does not exist.""" + url = _FAMILIES[family]() + response = httpx.get(f"{url}/collections", timeout=_TIMEOUT) + response.raise_for_status() + assert response.content and response.json().get("collections"), ( + f"{url}/collections returned no collections, so the " + f"{_split_version(url)[1]} {family} API this package requests has stopped " + "serving. Move the pin in waterdata/endpoints.py to the version the " + "service now serves." + ) diff --git a/tests/waterdata_test.py b/tests/waterdata_test.py index ffb56af5..bae73590 100644 --- a/tests/waterdata_test.py +++ b/tests/waterdata_test.py @@ -918,6 +918,21 @@ def test_get_continuous(httpx_mock): assert "continuous_id" in df.columns assert df["time"].dtype.name.startswith("datetime64[") assert "UTC" in df["time"].dtype.name + # A code column stays the string the service sent. + assert df["method_category"].tolist() == ["UNKWN", "UNKWN"] + + +def test_get_continuous_sends_method_category(httpx_mock): + """The named ``method_category`` parameter is sent in the request as a filter.""" + _mock_items(httpx_mock, "continuous") + + get_continuous( + monitoring_location_id="USGS-06904500", + method_category=["STNRD", "LMTUS"], + ) + + qs = _sent(httpx_mock, "continuous")[0] + assert qs["method_category"] == ["STNRD,LMTUS"] def test_get_latest_continuous(httpx_mock):