diff --git a/NEWS.md b/NEWS.md index 11657fb2..579ff061 100644 --- a/NEWS.md +++ b/NEWS.md @@ -2,6 +2,8 @@ **10/01/2026:** **Bug fix:** `waterdata.get_nearest_continuous` turned a missing target (`NaT`, `None`, `NaN`, or `""`) into a `(time >= 'nan' AND time <= 'nan')` clause and sent it to the service, including when only one entry of a list was missing. It now raises `ValueError` before any request, naming how many targets are missing and the position of the first. Targets pandas cannot parse now raise a `ValueError` that names `targets`, rather than pandas' message, which named no argument and suggested a `format=` the getter does not accept. **Bug fix:** the same getter's `window` was not validated: a missing window (`None`, `'NaT'`) failed with an `AttributeError` while the filter was built, and a negative one (`'-PT5M'`) inverted every bound, so the query matched nothing and returned an empty frame indistinguishable from a gap in the data. Both now raise `ValueError` naming `window`; a zero window remains an exact-match query. **Behavior change:** a bare-number `window` now raises `TypeError`; pandas read `window=450` as 450 nanoseconds, so it matched almost nothing. Pass a duration such as `window='PT450S'`. Numeric `targets` raise `TypeError` for the same reason: `targets=[1.5e9]` meant as epoch seconds was read as 1970-01-01T00:00:01.5Z. **Behavior change:** calls that previously sent a malformed filter now fail locally, so a caller passing targets with gaps must drop or fill them first. +**09/22/2026:** Twenty columns that the Water Data OGC collections return are now named, documented parameters of the getter that returns them; before, they could be passed only through `**queryables`. `get_field_measurements()`: `control_condition`, `day`, `field_measurements_series_id`, `measurement_rated`, `month`, `reading_type`, `time_of_day`, `year`. `get_peaks()`: `qualifier`, `time_of_day`, `value`. `get_monitoring_locations()`: `revision_created`, `revision_modified`, `revision_note`. `get_combined_metadata()`: `data_gap_interval`, `reading_type`. `get_time_series_metadata()`: `data_gap_interval`, `parameter_description`. `get_field_measurements_metadata()`: `reading_type`. `get_channel()`: `channel_location_direction`. Existing calls send the same request as before. + **09/22/2026:** `waterdata.get_continuous()` accepts `method_category`, a column USGS added to the `continuous` collection in September 2026: the RLMS method category code (`STNRD`, `LMTUS`, `EXPER` or `UNKWN`) for the method in effect over an observation's interval. It is returned on every record and is null for time series that have not been categorized. It could already be passed through `**queryables`; it is now a documented parameter. `get_latest_continuous()` is unchanged, because `latest-continuous` does not have the field. **09/22/2026:** The Water Data OGC getters now request **v1** of the Water Data APIs (`api.waterdata.usgs.gov/ogcapi/v1`), [released September 2026](https://waterdata.usgs.gov/blog/api-v1-release/); v0 stays online until June 2027. **New:** `WaterdataConfiguration(api_version="v0")`, or `api_version = "v0"` in the `[waterdata]` table of the configuration file, pins v0 during the transition. It changes only the version segment of the OGC path; Samples, Statistics and STAC are versioned separately and have no v1. **Behavior change:** `waterdata.get_time_series_metadata()` returns `begin` and `end` in UTC with a time zone, and no longer returns `begin_utc`, `end_utc`, `state_name` or `hydrologic_unit_code`. **Deprecation:** passing `begin_utc`, `end_utc`, `state`, `state_name` or `hydrologic_unit_code` to that getter, as a filter or in `properties`, emits a `DeprecationWarning` and sends the call to v0; this may be removed on or after 2027-06-01. Use `begin`, `end`, and `get_combined_metadata()` instead. **Behavior change:** `waterdata.get_field_measurements()` returns `time` as a date rather than a datetime, parsed to a tz-naive midnight timestamp as `get_daily()` already does; the time of day is in `time_of_day`. The `field-measurements-metadata` collection has no `time` field and is unaffected. **Bug fix:** `construction_date`, from `waterdata.get_monitoring_locations()` and `waterdata.get_combined_metadata()`, keeps every value. The service records it at day, month, or year precision (`19950812`, `199508`, `2005`); parsing it as a datetime turned the month-precision values into `NaT` and raised pandas' "Could not infer format" `UserWarning`. **Behavior change:** the column now holds those strings as sent rather than datetimes. diff --git a/dataretrieval/waterdata/measurements.py b/dataretrieval/waterdata/measurements.py index 2b978ab9..409bd3eb 100644 --- a/dataretrieval/waterdata/measurements.py +++ b/dataretrieval/waterdata/measurements.py @@ -26,6 +26,14 @@ def get_field_measurements( monitoring_location_id: str | Iterable[str] | None = None, parameter_code: str | Iterable[str] | None = None, observing_procedure_code: str | Iterable[str] | None = None, + control_condition: str | Iterable[str] | None = None, + day: int | list[int] | None = None, + field_measurements_series_id: str | Iterable[str] | None = None, + measurement_rated: str | Iterable[str] | None = None, + month: int | list[int] | None = None, + reading_type: str | Iterable[str] | None = None, + time_of_day: str | Iterable[str] | None = None, + year: int | list[int] | None = None, properties: str | Iterable[str] | None = None, field_visit_id: str | Iterable[str] | None = None, approval_status: str | Iterable[str] | None = None, @@ -69,6 +77,29 @@ def get_field_measurements( observing_procedure_code : string or iterable of strings, optional A short code corresponding to the observing procedure for the field measurement. + control_condition : string or iterable of strings, optional + The state of the control feature at the time of observation. + day : integer or list of integers, optional + The day of the month the field measurement was taken. If null, the day is + unknown. + field_measurements_series_id : string or iterable of strings, optional + A unique identifier representing a single collection series, corresponding to + the id field in the field-measurements-metadata endpoint. A collection series + is the set of field measurements at one monitoring location for a single + parameter code using a single reading type. + measurement_rated : string or iterable of strings, optional + A qualitative estimate of the quality of a measurement. + month : integer or list of integers, optional + The calendar month the field measurement was taken. If null, the month is + unknown. + reading_type : string or iterable of strings, optional + Distinguishes field-measurement readings from measurements. Readings have a + value of ReferencePrimary; measurements are Discharge or MeanGageHeight. + time_of_day : string or iterable of strings, optional + The time of day the field measurement was taken. If null, the time of day is + unknown. The time column holds only the date. + year : integer or list of integers, optional + The calendar year the field measurement was taken. properties : string or iterable of strings, optional The columns to return from the query. See the field-measurements schema in the OpenAPI reference for the available @@ -240,6 +271,9 @@ def get_peaks( month: int | list[int] | None = None, day: int | list[int] | None = None, peak_since: int | list[int] | None = None, + qualifier: str | Iterable[str] | None = None, + time_of_day: str | Iterable[str] | None = None, + value: str | Iterable[str] | None = None, properties: str | Iterable[str] | None = None, skip_geometry: bool | None = None, bbox: list[float] | None = None, @@ -288,6 +322,15 @@ def get_peaks( peak_since : int or list of ints, optional Filter on the year since which the peak value has been the record (the API serves this field as an integer; many rows are ``null``). + qualifier : string or iterable of strings, optional + Any qualifiers associated with a peak, for instance whether a sensor may have + been impacted by ice or whether the value was estimated. + time_of_day : string or iterable of strings, optional + The time of day a peak occurred. If null, the time of day is unknown, as is + common for historical peaks recorded only to the day. + value : string or iterable of strings, optional + The value of the peak. Values are transmitted as strings in the JSON response + to preserve precision. properties : string or iterable of strings, optional Subset of columns to return. Defaults to every available property. skip_geometry : boolean, optional @@ -393,6 +436,7 @@ def get_channel( measurement_type: str | Iterable[str] | None = None, last_modified: str | Iterable[str] | None = None, channel_measurement_type: str | Iterable[str] | None = None, + channel_location_direction: str | Iterable[str] | None = None, properties: str | Iterable[str] | None = None, skip_geometry: bool | None = None, bbox: list[float] | None = None, @@ -494,6 +538,8 @@ def get_channel( Water Data APIs use camelCase "skipGeometry" in CQL2 queries. channel_measurement_type : string or iterable of strings, optional The channel measurement type. + channel_location_direction : string or iterable of strings, optional + Location of the measurement from the gage. properties : string or iterable of strings, optional The columns to return from the query. Available options are: geometry, channel_measurements_id, monitoring_location_id, @@ -503,8 +549,8 @@ def get_channel( channel_location_distance, channel_location_distance_unit, channel_stability, channel_material, channel_evenness, horizontal_velocity_description, vertical_velocity_description, longitudinal_velocity_description, - measurement_type, last_modified, channel_measurement_type. The default - (None) returns all columns. + measurement_type, last_modified, channel_measurement_type, + channel_location_direction. The default (None) returns all columns. bbox : list of numbers, optional Only features whose geometry intersects the bounding box are selected. The bounding box is provided as four or six numbers, depending on diff --git a/dataretrieval/waterdata/metadata.py b/dataretrieval/waterdata/metadata.py index 9f085e7e..be23470f 100644 --- a/dataretrieval/waterdata/metadata.py +++ b/dataretrieval/waterdata/metadata.py @@ -67,6 +67,9 @@ def get_monitoring_locations( well_constructed_depth: str | Iterable[str] | None = None, hole_constructed_depth: str | Iterable[str] | None = None, depth_source_code: str | Iterable[str] | None = None, + revision_created: str | Iterable[str] | None = None, + revision_modified: str | Iterable[str] | None = None, + revision_note: str | Iterable[str] | None = None, properties: str | Iterable[str] | None = None, skip_geometry: bool | None = None, bbox: list[float] | None = None, @@ -253,6 +256,14 @@ def get_monitoring_locations( A code indicating the source of water-level data. A `list of codes `_ is available. + revision_created : string or iterable of strings, optional + The date a revision statement was created. + revision_modified : string or iterable of strings, optional + The most recent date a revision statement was modified. + revision_note : string or iterable of strings, optional + Text explaining a revision to this location's approved data. Revisions are + also flagged by revision qualifier codes in the data. Explanations from + before 2017 may not be online but can be requested. properties : string or iterable of strings, optional The columns to return from the query. Available options are: geometry, id, agency_code, agency_name, @@ -268,7 +279,8 @@ def get_monitoring_locations( contributing_drainage_area, time_zone_abbreviation, uses_daylight_savings, construction_date, aquifer_code, national_aquifer_code, aquifer_type_code, well_constructed_depth, - hole_constructed_depth, depth_source_code. + hole_constructed_depth, depth_source_code, revision_created, + revision_modified, revision_note. bbox : list of numbers, optional Only features whose geometry intersects the bounding box are selected. The bounding box is provided as four or six numbers, depending on @@ -395,6 +407,8 @@ def get_time_series_metadata( monitoring_location_id: str | Iterable[str] | None = None, parameter_code: str | Iterable[str] | None = None, parameter_name: str | Iterable[str] | None = None, + data_gap_interval: str | Iterable[str] | None = None, + parameter_description: str | Iterable[str] | None = None, properties: str | Iterable[str] | None = None, statistic_id: str | Iterable[str] | None = None, hydrologic_unit_code: str | Iterable[str] | None = None, @@ -448,6 +462,12 @@ def get_time_series_metadata( available at https://help.waterdata.usgs.gov/codes-and-parameters/parameters. parameter_name : string or iterable of strings, optional A human-understandable name corresponding to parameter_code. + data_gap_interval : string or iterable of strings, optional + The time interval threshold used for gap detection in the time series, as an + ISO 8601 duration. + parameter_description : string or iterable of strings, optional + A description of what the parameter code represents, as used by WDFN and other + USGS data dissemination products. properties : string or iterable of strings, optional The columns to return from the query. Available options are: begin, computation_identifier, @@ -708,6 +728,8 @@ def get_combined_metadata( well_constructed_depth: str | Iterable[str] | None = None, hole_constructed_depth: str | Iterable[str] | None = None, depth_source_code: str | Iterable[str] | None = None, + data_gap_interval: str | Iterable[str] | None = None, + reading_type: str | Iterable[str] | None = None, properties: str | Iterable[str] | None = None, skip_geometry: bool | None = None, bbox: list[float] | None = None, @@ -799,6 +821,12 @@ def get_combined_metadata( altitude, vertical/horizontal datum, drainage area, aquifer, well construction, …); see :func:`get_monitoring_locations` for descriptions of each. + data_gap_interval : string or iterable of strings, optional + The time interval threshold used for gap detection in the time series, as an + ISO 8601 duration. + reading_type : string or iterable of strings, optional + Distinguishes field-measurement readings from measurements. Readings have a + value of ReferencePrimary; measurements are Discharge or MeanGageHeight. properties : string or iterable of strings, optional Subset of columns to return. Defaults to every available property. @@ -918,6 +946,7 @@ def get_field_measurements_metadata( begin: str | Iterable[str] | None = None, end: str | Iterable[str] | None = None, last_modified: str | Iterable[str] | None = None, + reading_type: str | Iterable[str] | None = None, properties: str | Iterable[str] | None = None, skip_geometry: bool | None = None, bbox: list[float] | None = None, @@ -960,6 +989,9 @@ def get_field_measurements_metadata( interval (``"start/end"``, optionally half-bounded with ``..``), or an ISO 8601 duration (e.g. ``"P1M"``, ``"PT36H"``). See :func:`get_time_series_metadata` for the full grammar. + reading_type : string or iterable of strings, optional + Distinguishes field-measurement readings from measurements. Readings have a + value of ReferencePrimary; measurements are Discharge or MeanGageHeight. properties : string or iterable of strings, optional Subset of columns to return. Defaults to every available property. skip_geometry : boolean, optional diff --git a/tests/contracts/README.md b/tests/contracts/README.md index 3599c99c..c8d24e1d 100644 --- a/tests/contracts/README.md +++ b/tests/contracts/README.md @@ -9,8 +9,9 @@ The suite uses four dependency-oriented layers without moving established tests: `wqp_test.py`, `nldi_test.py`, `streamstats_test.py`): service request construction, response parsing, and documented protocol behavior. - **Component** (`transport_test.py`, `waterdata_chunking_test.py`, - `waterdata_queryables_test.py`, `waterdata_endpoints_test.py`, `rdb_test.py`, - `_csv_test.py`): one internal responsibility in isolation. + `waterdata_queryables_test.py`, `waterdata_endpoints_test.py`, + `waterdata_properties_test.py`, `rdb_test.py`, `_csv_test.py`): one internal + responsibility in isolation. - **Cross-component** (`architecture_test.py`, `headers_host_scoping_test.py`, `waterdata_progress_test.py`): dependency fitness functions and behavior that spans adapters, OGC, transport, or security boundaries. diff --git a/tests/waterdata_properties_test.py b/tests/waterdata_properties_test.py new file mode 100644 index 00000000..666d0629 --- /dev/null +++ b/tests/waterdata_properties_test.py @@ -0,0 +1,73 @@ +"""Live monitor: the columns a getter documents are the ones its collection has. + +Some getters list their returned columns in the ``properties`` docstring, under +"Available options are:". The list is hand-written, so it goes stale when USGS +adds or removes a field. It is checked here rather than generated, because a +generated list would make the docs build depend on the live service and would +not appear in ``help()``. +""" + +import inspect +import re + +import pytest + +from dataretrieval import waterdata +from dataretrieval.ogc.schema import _check_ogc_requests +from dataretrieval.waterdata.endpoints import ogc_api_url + +#: Getters that list their returned columns, by collection. Written out rather +#: than discovered, so a getter that loses its list fails instead of being +#: skipped. get_channel is left out: its list names the output column +#: channel_measurements_id where the schema has id. +_DOCUMENTED = { + "daily": waterdata.get_daily, + "continuous": waterdata.get_continuous, + "latest-continuous": waterdata.get_latest_continuous, + "latest-daily": waterdata.get_latest_daily, + "monitoring-locations": waterdata.get_monitoring_locations, + "time-series-metadata": waterdata.get_time_series_metadata, +} + +#: The label may be split across two lines. The list ends at the next unindented +#: line of the dedented docstring, which is the next numpydoc parameter. +_COLUMNS_RE = re.compile(r"Available\s+options\s+are:(.*?)(?=\n\S|\Z)", re.S) + +#: Some schemas list ``id`` and some do not, but every getter accepts it, so it +#: is excluded from both sides. +_ALWAYS_REQUESTABLE = {"id"} + + +def _documented_properties(getter) -> set[str]: + """The column names *getter* lists in its ``properties`` docstring.""" + match = _COLUMNS_RE.search(inspect.getdoc(getter) or "") + assert match, f"{getter.__name__} no longer documents its columns" + return {n.strip().rstrip(".") for n in match.group(1).split(",") if n.strip()} + + +def _schema_properties(collection: str) -> set[str]: + """The columns *collection* publishes in its OGC schema document.""" + body, _ = _check_ogc_requests(collection, "schema", base_url=ogc_api_url()) + properties = body.get("properties") + assert properties, f"{collection} published no schema properties" + return set(properties) + + +@pytest.mark.live +@pytest.mark.parametrize("collection", sorted(_DOCUMENTED)) +def test_documented_columns_match_the_collection_schema(collection): + """A getter's documented column list matches what the collection publishes. + + For an added field, consider a named parameter too; a removed field is a + breaking change and belongs in NEWS. + """ + getter = _DOCUMENTED[collection] + documented = _documented_properties(getter) - _ALWAYS_REQUESTABLE + published = _schema_properties(collection) - _ALWAYS_REQUESTABLE + + assert documented == published, ( + f"{getter.__name__} documents the wrong columns for {collection}: " + f"missing={sorted(published - documented)}, " + f"stale={sorted(documented - published)}. Edit the 'Available options " + "are:' list in its properties docstring." + ) diff --git a/tests/waterdata_test.py b/tests/waterdata_test.py index bae73590..343cf063 100644 --- a/tests/waterdata_test.py +++ b/tests/waterdata_test.py @@ -1300,6 +1300,72 @@ def test_v0_routing_does_not_write_on_the_callers_configuration(httpx_mock): assert other.startswith(f"{_OGC_BASE}/collections/daily") +#: Named parameters for returned columns, by getter and collection, with a value +#: to send. +_NEWLY_NAMED = { + (get_channel, "channel-measurements"): { + "channel_location_direction": "left bank", + }, + (get_field_measurements, "field-measurements"): { + "control_condition": "Clear", + "day": 5, + "field_measurements_series_id": "abc123", + "measurement_rated": "Good", + "month": 3, + "reading_type": "Discharge", + "time_of_day": "17:30:00", + "year": 2020, + }, + (get_peaks, "peaks"): { + "qualifier": "Bd", + "time_of_day": "17:30:00", + "value": "847000", + }, + (get_combined_metadata, "combined-metadata"): { + "data_gap_interval": "P1D", + "reading_type": "Discharge", + }, + (get_monitoring_locations, "monitoring-locations"): { + "revision_created": "2024-01-01", + "revision_modified": "2024-01-01", + "revision_note": "corrected", + }, + (get_time_series_metadata, "time-series-metadata"): { + "data_gap_interval": "P1D", + "parameter_description": "Discharge", + "statistics_begin": "2000", + }, + (get_field_measurements_metadata, "field-measurements-metadata"): { + "reading_type": "Discharge", + }, +} + + +@pytest.mark.parametrize( + ("getter", "collection", "parameter", "value"), + [ + pytest.param( + getter, collection, parameter, value, id=f"{getter.__name__}-{parameter}" + ) + for (getter, collection), params in _NEWLY_NAMED.items() + for parameter, value in params.items() + ], +) +def test_newly_named_columns_reach_the_request( + httpx_mock, getter, collection, parameter, value +): + """A newly named parameter is sent to the service. + + A call whose parameter is accepted but not forwarded returns unfiltered + results instead of failing. + """ + _mock_items(httpx_mock, collection) + + getter(monitoring_location_id="USGS-05427718", **{parameter: value}) + + assert _sent(httpx_mock, collection)[0][parameter] == [str(value)] + + def test_get_combined_metadata(httpx_mock): _mock_items(httpx_mock, "combined-metadata")