Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 5 additions & 0 deletions NEWS.md
Original file line number Diff line number Diff line change
@@ -1,3 +1,8 @@
**10/03/2026:** `waterdata.get_field_measurements()` and
`waterdata.get_field_measurements_metadata()` accept `sublocation_identifier`, a
field USGS added to the `field-measurements` and `field-measurements-metadata`
collections at the end of September 2026: an optional identifier for where at a monitoring location a measurement was recorded (for example `Primary`, `UPSTREAM`, or `BRIDGE`).

**10/01/2026:** **Behavior change:** the numeric columns of the `waterdata` OGC getters and `ngwmn` — `value`, `altitude`, `drainage_area`, and the other measurement columns — are always `float64`. They were `int64` whenever every value in a response happened to be whole, so the same column changed dtype between calls, breaking concatenation, schema checks, and typed stores (#428). **Behavior change:** a value in a numeric or datetime column that cannot be parsed still becomes `NaN`/`NaT`, but now emits a `UserWarning` naming the column and the count, so a parse failure is distinguishable from a missing value. Repeat the call with `convert_type=False` to see the raw values. Expect it from `waterdata.get_monitoring_locations()`: its `construction_date` holds partial dates (`1925`, `194805`, `19100000`) that a datetime column cannot represent, and they were already being dropped without notice — 7,306 of 24,089 values in a query for Maryland groundwater sites.

**10/01/2026:** **Bug fix:** `waterdata.get_nearest_continuous` turned a missing target (`NaT`, `None`, `NaN`, or `""`) into a `(time >= 'nan' AND time <= 'nan')` clause and sent it to the service, including when only one entry of a list was missing. It now raises `ValueError` before any request, naming how many targets are missing and the position of the first. Targets pandas cannot parse now raise a `ValueError` that names `targets`, rather than pandas' message, which named no argument and suggested a `format=` the getter does not accept. **Bug fix:** the same getter's `window` was not validated: a missing window (`None`, `'NaT'`) failed with an `AttributeError` while the filter was built, and a negative one (`'-PT5M'`) inverted every bound, so the query matched nothing and returned an empty frame indistinguishable from a gap in the data. Both now raise `ValueError` naming `window`; a zero window remains an exact-match query. **Behavior change:** a bare-number `window` now raises `TypeError`; pandas read `window=450` as 450 nanoseconds, so it matched almost nothing. Pass a duration such as `window='PT450S'`. Numeric `targets` raise `TypeError` for the same reason: `targets=[1.5e9]` meant as epoch seconds was read as 1970-01-01T00:00:01.5Z. **Behavior change:** calls that previously sent a malformed filter now fail locally, so a caller passing targets with gaps must drop or fill them first.
Expand Down
4 changes: 4 additions & 0 deletions dataretrieval/waterdata/measurements.py
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,7 @@ def get_field_measurements(
observing_procedure: str | Iterable[str] | None = None,
vertical_datum: str | Iterable[str] | None = None,
measuring_agency: str | Iterable[str] | None = None,
sublocation_identifier: str | Iterable[str] | None = None,
skip_geometry: bool | None = None,
time: str | Iterable[str] | None = None,
bbox: list[float] | None = None,
Expand Down Expand Up @@ -124,6 +125,9 @@ def get_field_measurements(
monitoring location.
measuring_agency : string or iterable of strings, optional
The agency performing the measurement.
sublocation_identifier : string or iterable of strings, optional
The sublocation where the measurement was recorded, such as
"UPSTREAM".
skip_geometry : boolean, optional
If True, the response omits the geometry of each feature and the
returned object is a data frame with no spatial information. The USGS
Expand Down
4 changes: 4 additions & 0 deletions dataretrieval/waterdata/metadata.py
Original file line number Diff line number Diff line change
Expand Up @@ -918,6 +918,7 @@ def get_field_measurements_metadata(
begin: str | Iterable[str] | None = None,
end: str | Iterable[str] | None = None,
last_modified: str | Iterable[str] | None = None,
sublocation_identifier: str | Iterable[str] | None = None,
properties: str | Iterable[str] | None = None,
skip_geometry: bool | None = None,
bbox: list[float] | None = None,
Expand Down Expand Up @@ -960,6 +961,9 @@ def get_field_measurements_metadata(
interval (``"start/end"``, optionally half-bounded with ``..``),
or an ISO 8601 duration (e.g. ``"P1M"``, ``"PT36H"``). See
:func:`get_time_series_metadata` for the full grammar.
sublocation_identifier : string or iterable of strings, optional
The sublocation where the series' measurements are recorded, such
as "UPSTREAM".
properties : string or iterable of strings, optional
Subset of columns to return. Defaults to every available property.
skip_geometry : boolean, optional
Expand Down
8 changes: 6 additions & 2 deletions tests/data/waterdata_ogc_fixtures.json
Original file line number Diff line number Diff line change
Expand Up @@ -388,6 +388,7 @@
"parameter_code": "00065",
"qualifier": null,
"reading_type": "ReferencePrimary",
"sublocation_identifier": null,
"time": "2014-06-19",
"time_of_day": "16:00:27+00:00",
"unit_of_measure": "ft",
Expand Down Expand Up @@ -422,6 +423,7 @@
"parameter_code": "00060",
"qualifier": null,
"reading_type": "Discharge",
"sublocation_identifier": null,
"time": "2023-11-01",
"time_of_day": "14:49:30+00:00",
"unit_of_measure": "ft^3/s",
Expand Down Expand Up @@ -455,7 +457,8 @@
"parameter_code": "00065",
"parameter_description": "Gage height, feet",
"parameter_name": "Gage height",
"reading_type": "ReferencePrimary"
"reading_type": "ReferencePrimary",
"sublocation_identifier": null
},
"type": "Feature"
},
Expand All @@ -477,7 +480,8 @@
"parameter_code": "00060",
"parameter_description": "Discharge, cubic feet per second",
"parameter_name": "Discharge",
"reading_type": "Discharge"
"reading_type": "Discharge",
"sublocation_identifier": null
},
"type": "Feature"
}
Expand Down
4 changes: 3 additions & 1 deletion tests/data/waterdata_queryables.json
Original file line number Diff line number Diff line change
Expand Up @@ -250,6 +250,7 @@
"site_type_code",
"state_code",
"state_name",
"sublocation_identifier",
"time",
"time_of_day",
"time_zone_abbreviation",
Expand All @@ -272,7 +273,8 @@
"parameter_code",
"parameter_description",
"parameter_name",
"reading_type"
"reading_type",
"sublocation_identifier"
],
"latest-continuous": [
"agency_code",
Expand Down
31 changes: 31 additions & 0 deletions tests/waterdata_test.py
Original file line number Diff line number Diff line change
Expand Up @@ -1218,6 +1218,37 @@ def test_field_measurements_time_is_a_date_with_time_of_day_alongside(httpx_mock
assert df["time_of_day"].tolist() == ["16:00:27+00:00", "14:49:30+00:00"]


@pytest.mark.parametrize(
("getter", "collection", "value", "sent"),
[
(
get_field_measurements,
"field-measurements",
["Primary", "UPSTREAM"],
["Primary,UPSTREAM"],
),
(
get_field_measurements_metadata,
"field-measurements-metadata",
"UPSTREAM",
["UPSTREAM"],
),
],
ids=["measurements-list", "metadata-string"],
)
def test_sublocation_identifier_is_sent_as_a_filter(
httpx_mock, getter, collection, value, sent
):
"""The named ``sublocation_identifier`` parameter is sent as a filter and
returned as a column."""
_mock_items(httpx_mock, collection)

df, _ = getter(monitoring_location_id="USGS-05427718", sublocation_identifier=value)

assert _sent(httpx_mock, collection)[0]["sublocation_identifier"] == sent
assert "sublocation_identifier" in df.columns


def test_get_field_measurements_metadata(httpx_mock):
_mock_items(httpx_mock, "field-measurements-metadata")

Expand Down
Loading