diff --git a/CHANGELOG.md b/CHANGELOG.md index b3bf88710e..4bf5b2b34e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -114,6 +114,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Dependencies +- Drops `onnx`, `torchvision` and `pandas` from the required dependencies. + `pandas` is now optional via `datapipes-extras` or `model-extras`. + ## [2.2.1] - 2026-XX-YY ### Fixed diff --git a/physicsnemo/core/version_check.py b/physicsnemo/core/version_check.py index b99874001b..b6e93473d9 100644 --- a/physicsnemo/core/version_check.py +++ b/physicsnemo/core/version_check.py @@ -214,6 +214,11 @@ def _format_install_hint( "h5py", group="datapipes-extras", ), + "pandas": _format_install_hint( + "pandas", + group="datapipes-extras", + direct_hint="pip install pandas (also included in [model-extras])", + ), "netCDF4": _format_install_hint( "netCDF4", group="model-extras", diff --git a/physicsnemo/datapipes/gnn/drivaernet_dataset.py b/physicsnemo/datapipes/gnn/drivaernet_dataset.py index 09d4d165da..eb42196650 100644 --- a/physicsnemo/datapipes/gnn/drivaernet_dataset.py +++ b/physicsnemo/datapipes/gnn/drivaernet_dataset.py @@ -18,7 +18,6 @@ from pathlib import Path from typing import Iterable -import pandas as pd import torch from torch import Tensor from torch.utils.data import Dataset @@ -28,6 +27,8 @@ from physicsnemo.datapipes.meta import DatapipeMetaData from physicsnemo.nn.module.gnn_layers.utils import PyGData +pd = OptionalImport("pandas") + # Lazy imports for optional dependencies pyg = OptionalImport("torch_geometric") pv = OptionalImport("pyvista") diff --git a/physicsnemo/datapipes/healpix/coupledtimeseries_dataset.py b/physicsnemo/datapipes/healpix/coupledtimeseries_dataset.py index be375fe315..05ab2ef79c 100644 --- a/physicsnemo/datapipes/healpix/coupledtimeseries_dataset.py +++ b/physicsnemo/datapipes/healpix/coupledtimeseries_dataset.py @@ -23,7 +23,6 @@ from typing import Optional, Sequence, Union import numpy as np -import pandas as pd import torch from omegaconf import DictConfig, OmegaConf @@ -34,6 +33,8 @@ from . import couplers from .timeseries_dataset import TimeSeriesDataset +pd = OptionalImport("pandas") + xr = OptionalImport("xarray") logger = logging.getLogger(__name__) diff --git a/physicsnemo/datapipes/healpix/couplers.py b/physicsnemo/datapipes/healpix/couplers.py index a8ef9470bc..d6bb82b44c 100644 --- a/physicsnemo/datapipes/healpix/couplers.py +++ b/physicsnemo/datapipes/healpix/couplers.py @@ -20,11 +20,12 @@ from typing import Sequence import numpy as np -import pandas as pd import torch as th from physicsnemo.core.version_check import OptionalImport +pd = OptionalImport("pandas") + xr = OptionalImport("xarray") logger = logging.getLogger(__name__) @@ -46,7 +47,7 @@ def __init__( presteps: int = 0, input_time_dim: int = 2, output_time_dim: int = 2, - input_times: Sequence = [pd.Timedelta("24h"), pd.Timedelta("48h")], + input_times: Sequence = ["24h", "48h"], prepared_coupled_data=True, ): """ @@ -68,8 +69,8 @@ def __init__( output_time_dim: int, optional number of output times for each model step, default 2 input_times: Sequence, optional - sequence of pandas Timedelta objects that indicate which times are to be coupled, - default [pd.Timedelta("24h"), pd.Timedelta("48h")] + sequence of pandas Timedelta objects (or strings accepted by ``pd.Timedelta``) + that indicate which times are to be coupled, default ["24h", "48h"] prepared_coupled_data: boolean, optional If True assumes data in dataset has been prepared approiately for training: averages have already been calculated so that each time step denotes @@ -289,7 +290,7 @@ def __init__( input_time_dim: int = 2, output_time_dim: int = 2, averaging_window: str = "24h", - input_times: Sequence = [pd.Timedelta("24h"), pd.Timedelta("48h")], + input_times: Sequence = ["24h", "48h"], prepared_coupled_data=True, ): """ @@ -313,8 +314,8 @@ def __init__( averaging_window: str, optional period over which coupled data is averaged before sent back to model, default "24h" input_times: Sequence, optional - sequence of pandas Timedelta objects that indicate which times are to be coupled, - default [pd.Timedelta("24h"), pd.Timedelta("48h")] + sequence of pandas Timedelta objects (or strings accepted by ``pd.Timedelta``) + that indicate which times are to be coupled, default ["24h", "48h"] prepared_coupled_data: boolean, optional If True assumes data in dataset has been prepared approiately for training: averages have already been calculated so that each time step denotes diff --git a/physicsnemo/datapipes/healpix/timeseries_dataset.py b/physicsnemo/datapipes/healpix/timeseries_dataset.py index 3fb8884034..bdeeda563e 100644 --- a/physicsnemo/datapipes/healpix/timeseries_dataset.py +++ b/physicsnemo/datapipes/healpix/timeseries_dataset.py @@ -23,7 +23,6 @@ from typing import Optional, Sequence, Union import numpy as np -import pandas as pd import torch from omegaconf import DictConfig, OmegaConf @@ -32,6 +31,8 @@ from physicsnemo.datapipes.meta import DatapipeMetaData from physicsnemo.utils.insolation import insolation +pd = OptionalImport("pandas") + xr = OptionalImport("xarray") logger = logging.getLogger(__name__) diff --git a/physicsnemo/experimental/datapipes/healda/configs/sensors.py b/physicsnemo/experimental/datapipes/healda/configs/sensors.py index 4ff3733057..1f2918adbe 100644 --- a/physicsnemo/experimental/datapipes/healda/configs/sensors.py +++ b/physicsnemo/experimental/datapipes/healda/configs/sensors.py @@ -24,7 +24,9 @@ from dataclasses import dataclass, field import numpy as np -import pandas as pd +from physicsnemo.core.version_check import OptionalImport + +pd = OptionalImport("pandas") # Recipe-side directory containing per-sensor `*_normalizations.csv` files # (and the ERA5 stats CSV consumed by `loaders.era5`). When unset, sensor diff --git a/physicsnemo/experimental/datapipes/healda/dataset.py b/physicsnemo/experimental/datapipes/healda/dataset.py index 6c0e77a157..7618871ebe 100644 --- a/physicsnemo/experimental/datapipes/healda/dataset.py +++ b/physicsnemo/experimental/datapipes/healda/dataset.py @@ -52,7 +52,6 @@ from typing import Union import numpy as np -import pandas as pd import torch from physicsnemo.experimental.datapipes.healda.indexing import get_flat_indexer @@ -60,6 +59,9 @@ from physicsnemo.experimental.datapipes.healda.protocols import ObsLoader, Transform from physicsnemo.experimental.datapipes.healda.time_utils import as_cftime from physicsnemo.experimental.datapipes.healda.types import VariableConfig +from physicsnemo.core.version_check import OptionalImport + +pd = OptionalImport("pandas") # HEALPix level-6 pixel count: 12 * 4^6 NPIX_HPX6 = 12 * 4**6 diff --git a/physicsnemo/experimental/datapipes/healda/loaders/era5.py b/physicsnemo/experimental/datapipes/healda/loaders/era5.py index 8986ce8abe..e3f58442a1 100644 --- a/physicsnemo/experimental/datapipes/healda/loaders/era5.py +++ b/physicsnemo/experimental/datapipes/healda/loaders/era5.py @@ -25,7 +25,6 @@ from typing import Optional import numpy as np -import pandas as pd from physicsnemo.core.version_check import OptionalImport @@ -35,6 +34,8 @@ from physicsnemo.experimental.datapipes.healda.loaders.zarr_loader import NO_LEVEL, ZarrLoader from physicsnemo.experimental.datapipes.healda.types import BatchInfo, TimeUnit, VariableConfig +pd = OptionalImport("pandas") + __all__ = ["ERA5Loader", "get_batch_info"] SST_LAND_FILL_VALUE = 290 @@ -174,7 +175,7 @@ def _encode_channel(channel) -> str: return name -def _load_raw_stats(config: VariableConfig) -> pd.DataFrame: +def _load_raw_stats(config: VariableConfig) -> "pd.DataFrame": if config.name == "ufs": file_name = "ufs_v0_stats.csv" elif config.name == "era5": diff --git a/physicsnemo/experimental/datapipes/healda/loaders/ufs_obs.py b/physicsnemo/experimental/datapipes/healda/loaders/ufs_obs.py index 337b217728..11689eca03 100644 --- a/physicsnemo/experimental/datapipes/healda/loaders/ufs_obs.py +++ b/physicsnemo/experimental/datapipes/healda/loaders/ufs_obs.py @@ -38,7 +38,6 @@ import fsspec import numpy as np -import pandas as pd from physicsnemo.core.version_check import OptionalImport pa = OptionalImport("pyarrow") @@ -53,6 +52,8 @@ from physicsnemo.experimental.datapipes.healda.configs.sensors import SENSOR_CONFIGS from physicsnemo.experimental.datapipes.healda.transforms.obs_filtering import filter_observations +pd = OptionalImport("pandas") + LOCAL_CHANNEL_ID = pa.field("local_channel_id", pa.uint16()) @@ -163,7 +164,7 @@ def channel_table(self) -> pa.Table: array = pa.array(local_channel_ids).cast(LOCAL_CHANNEL_ID.type) return table.append_column(LOCAL_CHANNEL_ID, array) - def _get_interval_times(self, dt: datetime) -> pd.DatetimeIndex: + def _get_interval_times(self, dt: datetime) -> "pd.DatetimeIndex": start, end = self.obs_context_hours start += self.data_spacing return pd.date_range( @@ -172,7 +173,7 @@ def _get_interval_times(self, dt: datetime) -> pd.DatetimeIndex: freq=f"{self.data_spacing}h", ) - def _get_parquet_files_to_read(self, interval_times: pd.DatetimeIndex): + def _get_parquet_files_to_read(self, interval_times: "pd.DatetimeIndex"): required_dates = {t.strftime("%Y%m%d") for t in interval_times} for sensor in self.sensors: for date in required_dates: @@ -247,7 +248,7 @@ def _add_channel_metadata(self, table): GLOBAL_CHANNEL_ID.name, ) - async def sel_time(self, times: pd.DatetimeIndex) -> dict: + async def sel_time(self, times: "pd.DatetimeIndex") -> dict: """Load observation data for specified times. Args: diff --git a/physicsnemo/experimental/datapipes/healda/loaders/zarr_loader.py b/physicsnemo/experimental/datapipes/healda/loaders/zarr_loader.py index 220d19c5e5..3e803b088c 100644 --- a/physicsnemo/experimental/datapipes/healda/loaders/zarr_loader.py +++ b/physicsnemo/experimental/datapipes/healda/loaders/zarr_loader.py @@ -27,10 +27,11 @@ import cftime import numpy as np -import pandas as pd from physicsnemo.core.version_check import OptionalImport +pd = OptionalImport("pandas") + xr = OptionalImport("xarray") zarr = OptionalImport("zarr") _zarr_sync = OptionalImport("zarr.core.sync") diff --git a/physicsnemo/experimental/datapipes/healda/protocols.py b/physicsnemo/experimental/datapipes/healda/protocols.py index c12eb428d0..83ed444f3f 100644 --- a/physicsnemo/experimental/datapipes/healda/protocols.py +++ b/physicsnemo/experimental/datapipes/healda/protocols.py @@ -27,8 +27,10 @@ from typing import Any, Protocol, runtime_checkable import cftime -import pandas as pd import torch +from physicsnemo.core.version_check import OptionalImport + +pd = OptionalImport("pandas") @runtime_checkable @@ -51,7 +53,7 @@ async def sel_time(self, times): return {"obs": tables} """ - async def sel_time(self, times: pd.DatetimeIndex) -> dict[str, list[Any]]: + async def sel_time(self, times: "pd.DatetimeIndex") -> dict[str, list[Any]]: """Load observation data for the given timestamps. Args: diff --git a/physicsnemo/experimental/datapipes/healda/time_utils.py b/physicsnemo/experimental/datapipes/healda/time_utils.py index 9a1bc1b96a..043fa627f5 100644 --- a/physicsnemo/experimental/datapipes/healda/time_utils.py +++ b/physicsnemo/experimental/datapipes/healda/time_utils.py @@ -19,7 +19,9 @@ import cftime import numpy as np -import pandas as pd +from physicsnemo.core.version_check import OptionalImport + +pd = OptionalImport("pandas") def as_pydatetime(time) -> datetime.datetime: diff --git a/physicsnemo/models/dlwp_healpix/HEALPixRecUNet.py b/physicsnemo/models/dlwp_healpix/HEALPixRecUNet.py index 334ed1e8dd..3472756ae1 100644 --- a/physicsnemo/models/dlwp_healpix/HEALPixRecUNet.py +++ b/physicsnemo/models/dlwp_healpix/HEALPixRecUNet.py @@ -27,13 +27,13 @@ from dataclasses import dataclass from typing import Any, Dict, Sequence -import pandas as pd import torch from hydra.utils import instantiate from omegaconf import DictConfig from physicsnemo.core.meta import ModelMetaData from physicsnemo.core.module import Module +from physicsnemo.core.version_check import OptionalImport from physicsnemo.nn.module.hpx import HEALPixFoldFaces, HEALPixUnfoldFaces from .layers import ( @@ -43,6 +43,8 @@ _remap_obj, ) +pd = OptionalImport("pandas") + logger = logging.getLogger(__name__) diff --git a/physicsnemo/utils/insolation.py b/physicsnemo/utils/insolation.py index d57cd816a1..a0dbfec6cf 100644 --- a/physicsnemo/utils/insolation.py +++ b/physicsnemo/utils/insolation.py @@ -14,8 +14,53 @@ # See the License for the specific language governing permissions and # limitations under the License. +import datetime +import re + import numpy as np -import pandas as pd + +_LEADING_YEAR = re.compile(r"\s*(-?\d{4,})") + + +def _calendar_year(d) -> int: + """Calendar year of one date-like value, as ``pandas.Timestamp(d).year`` reports it. + + For timezone-aware datetimes and ISO strings that carry a UTC offset, this is + the year of the *local* wall-clock time, not of the UTC instant numpy stores. + """ + if isinstance( + d, datetime.date + ): # datetime.datetime, datetime.date, pandas.Timestamp + return d.year + if isinstance(d, str): + match = _LEADING_YEAR.match(d) + if match: + return int(match.group(1)) + return int(np.datetime64(d).astype("datetime64[Y]").astype(np.int64)) + 1970 + + +def _days_since_year_start(dates) -> np.ndarray: + """Fractional days from January 1st 00:00 of each date's calendar year. + + Reproduces ``(np.array(dates, dtype="datetime64") - Timestamp(year, 1, 1)) / 1 day`` + from the earlier pandas-based implementation exactly: instants are numpy's + conversion of ``dates`` (UTC for timezone-aware input), the year is the local + calendar year, and year starts are held at microsecond resolution, which is the + unit numpy inferred from the ``Timestamp`` objects, so unit promotion in the + subtraction is unchanged for every input dtype. + """ + instants = np.array(dates, dtype="datetime64") + if np.isnat(instants).any(): + raise ValueError("insolation: 'dates' contains NaT or None entries") + if isinstance(dates, np.ndarray) and np.issubdtype(dates.dtype, np.datetime64): + # Naive datetime64 arrays: the calendar year is unambiguous, stay vectorized. + years = instants.astype("datetime64[Y]") + else: + flat = np.asarray(dates, dtype=object).ravel() + years = np.array([_calendar_year(d) for d in flat], dtype=np.int64) - 1970 + years = years.astype("datetime64[Y]").reshape(instants.shape) + start_years = years.astype("datetime64[us]") + return (instants - start_years) / np.timedelta64(1, "D") def insolation( @@ -69,12 +114,7 @@ def insolation( beta = np.sqrt(1 - ecc**2.0) # Get the day of year as a float. - start_years = np.array( - [pd.Timestamp(pd.Timestamp(d).year, 1, 1) for d in dates], dtype="datetime64" - ) - days_arr = (np.array(dates, dtype="datetime64") - start_years) / np.timedelta64( - 1, "D" - ) + days_arr = _days_since_year_start(dates) for d in range(n_dim): days_arr = np.expand_dims(days_arr, -1) # For daily max values, set the day to 0.5 and the longitude everywhere to 0 (this is approx noon) diff --git a/pyproject.toml b/pyproject.toml index e2e7491350..b1b2ebe82e 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -21,14 +21,11 @@ classifiers = [ dynamic = ["version", "optional_dependencies"] dependencies = [ - "onnx>=1.14.0", "warp-lang>=1.14.0", - "pandas>=2.2.0", "nvtx>=0.2.10", "treelib>=1.2.5", "numpy>=1.22.4", "torch>=2.10.0", - "torchvision>=0.25.0a0", "tqdm>=4.60.0", "requests>=2.32.2", # urllib3 is a transitive dependency (via requests/botocore/etc.); pin to @@ -304,10 +301,12 @@ nn-extras = [ ] model-extras = [ "nvidia-physicsnemo[nn-extras]", + "pandas>=2.2.0", "pyvista>=0.46.4", "vtk", ] datapipes-extras = [ + "pandas>=2.2.0", "tfrecord", "dask", "netCDF4", diff --git a/test/datapipes/healda/test_time_utils.py b/test/datapipes/healda/test_time_utils.py index b7fd5066dd..c09e5b1d00 100644 --- a/test/datapipes/healda/test_time_utils.py +++ b/test/datapipes/healda/test_time_utils.py @@ -18,9 +18,9 @@ import datetime import numpy as np -import pandas as pd import pytest +pd = pytest.importorskip("pandas") cftime = pytest.importorskip("cftime") from physicsnemo.experimental.datapipes.healda.time_utils import ( # noqa: E402 diff --git a/test/models/dlwp_healpix/test_healpix_recunet_model.py b/test/models/dlwp_healpix/test_healpix_recunet_model.py index 42f72318e8..d63a4ab7b8 100644 --- a/test/models/dlwp_healpix/test_healpix_recunet_model.py +++ b/test/models/dlwp_healpix/test_healpix_recunet_model.py @@ -21,8 +21,13 @@ from physicsnemo.models.dlwp_healpix import HEALPixRecUNet from test import common +from test.conftest import requires_module from test.models.graphcast.utils import fix_random_seeds +# HEALPixRecUNet parses ``delta_time`` / ``reset_cycle`` with ``pandas.Timedelta``, +# an optional dependency (``model-extras``). +pytestmark = requires_module("pandas") + @pytest.fixture def conv_next_block_dict(in_channels=3, out_channels=1): diff --git a/test/utils/test_insolation.py b/test/utils/test_insolation.py new file mode 100644 index 0000000000..0db327f358 --- /dev/null +++ b/test/utils/test_insolation.py @@ -0,0 +1,322 @@ +# SPDX-FileCopyrightText: Copyright (c) 2023 - 2026 NVIDIA CORPORATION & AFFILIATES. +# SPDX-FileCopyrightText: All rights reserved. +# SPDX-License-Identifier: Apache-2.0 +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Regression tests pinning the day-of-year semantics of ``insolation``. + +``_days_since_year_start`` replaced a pandas expression, +``(np.array(dates, "datetime64") - Timestamp(Timestamp(d).year, 1, 1)) / 1 day``, +and must reproduce it bit-for-bit. Two oracles enforce that: + +* a pure-Python reference: the instant in UTC (numpy's conversion) minus + January 1st of the *local* calendar year (pandas' ``Timestamp.year``); +* when pandas is installed, the original expression itself. + +Both run over a sweep of dates (every month, leap days, year boundaries, +1600-3000), timezones (whole, half and 45-minute offsets, both extremes, +named zones), every ``datetime64`` unit, and every ISO offset spelling. +""" + +import datetime as dt +import itertools +from zoneinfo import ZoneInfo + +import numpy as np +import pytest + +from physicsnemo.utils.insolation import _days_since_year_start, insolation + +# ---------------------------------------------------------------- fixtures -- + +UTC = dt.timezone.utc +OFFSETS = { + "UTC": UTC, + "-12:00": dt.timezone(dt.timedelta(hours=-12)), + "-05:00": dt.timezone(dt.timedelta(hours=-5)), + "-03:30": dt.timezone(dt.timedelta(hours=-3, minutes=-30)), + "+01:00": dt.timezone(dt.timedelta(hours=1)), + "+05:30": dt.timezone(dt.timedelta(hours=5, minutes=30)), + "+05:45": dt.timezone(dt.timedelta(hours=5, minutes=45)), + "+14:00": dt.timezone(dt.timedelta(hours=14)), + "US/Eastern": ZoneInfo("America/New_York"), + "Europe/Berlin": ZoneInfo("Europe/Berlin"), + "Asia/Kolkata": ZoneInfo("Asia/Kolkata"), + "Pacific/Kiritimati": ZoneInfo("Pacific/Kiritimati"), + "Pacific/Apia": ZoneInfo("Pacific/Apia"), +} + +# Naive wall-clock times spanning the year: one per month, leap days, and the +# instants on either side of every year boundary where a UTC offset can move +# the calendar year. +NAIVE_DATES = ( + [dt.datetime(2020, m, 15, 6 * (m % 4), 17, 3) for m in range(1, 13)] + + [ + dt.datetime(2020, 2, 29, 23, 59, 59, 999999), # leap day, last microsecond + dt.datetime(2024, 2, 29, 0, 0, 0, 1), + dt.datetime(1900, 3, 1), # 1900 is not a leap year + dt.datetime(2000, 2, 29, 12), # 2000 is + dt.datetime(1600, 3, 1, 12), # outside pandas' ns range + dt.datetime(1950, 6, 15, 12), # pre-1970 (negative epoch) + dt.datetime(1969, 12, 31, 23, 59, 59), + dt.datetime(1970, 1, 1, 0, 0, 0), + dt.datetime(2262, 4, 11, 12), # edge of the ns range + dt.datetime(3000, 7, 4), + ] + + [ + dt.datetime(y, 12, 31, h, mi) + for y in (1999, 2019, 2020, 2023) + for h, mi in ((9, 0), (12, 0), (20, 30), (23, 59)) + ] + + [ + dt.datetime(y, 1, 1, h, mi) + for y in (2000, 2020, 2021, 2024) + for h, mi in ((0, 0), (0, 1), (3, 30), (14, 59)) + ] +) + + +def reference_days(d): + """Pure-Python oracle for one date: UTC instant minus local-year start.""" + if d.tzinfo is not None: + instant = d.astimezone(UTC).replace(tzinfo=None) # numpy stores UTC + else: + instant = d + start = dt.datetime(d.year, 1, 1) # pandas: Timestamp(d).year is local + # Exact integer microseconds divided once, which is what numpy does with + # datetime64[us] arithmetic; timedelta.total_seconds() would round differently. + delta = instant - start + micros = (delta.days * 86_400 + delta.seconds) * 1_000_000 + delta.microseconds + return micros / 86_400_000_000 + + +# ------------------------------------------------------------- naive dates -- + + +@pytest.mark.parametrize("date", NAIVE_DATES, ids=str) +def test_naive_datetime_matches_reference(date): + """Every naive date agrees with the reference to the last bit.""" + np.testing.assert_array_equal( + _days_since_year_start([date]), [reference_days(date)] + ) + + +def test_naive_input_types_are_interchangeable(): + """datetime objects, datetime64 arrays, datetime64 scalars and ISO strings agree.""" + dates = [d for d in NAIVE_DATES if 1678 < d.year < 2262] # ns-representable + expected = _days_since_year_start(dates) + as_ns = np.array(dates, dtype="datetime64[ns]") + as_scalars = [np.datetime64(d) for d in dates] + as_str = [d.isoformat() for d in dates] + for variant in (as_ns, as_scalars, as_str): + np.testing.assert_array_equal(_days_since_year_start(variant), expected) + + +@pytest.mark.parametrize("unit", ["ns", "us", "ms", "s", "m", "h", "D", "M", "Y"]) +def test_every_datetime64_unit(unit): + """Each numpy unit, including the non-linear month and year units, is accepted.""" + dates = np.array( + ["2020-02-29T13:30:15.123456789", "2021-12-31T23:59:59.999999999"], + dtype="datetime64[ns]", + ).astype(f"datetime64[{unit}]") + # The truth is the same values re-expressed as datetime objects at that unit's + # resolution: convert back through microseconds and let the reference do the rest. + truth = [reference_days(d) for d in dates.astype("datetime64[us]").astype(object)] + got = _days_since_year_start(dates) + if unit == "ns": + # microsecond truth cannot see nanoseconds; assert against numpy directly + ns = ( + dates - dates.astype("datetime64[Y]").astype("datetime64[us]") + ) / np.timedelta64(1, "D") + np.testing.assert_array_equal(got, ns) + else: + np.testing.assert_array_equal(got, truth) + + +# ---------------------------------------------------------- timezone-aware -- + + +@pytest.mark.parametrize( + ("date", "zone"), + list(itertools.product(NAIVE_DATES, OFFSETS)), + ids=lambda v: v if isinstance(v, str) else str(v), +) +def test_timezone_aware_uses_local_year_and_utc_instant(date, zone): + """For every date x timezone, the year is local and the instant is UTC.""" + aware = date.replace(tzinfo=OFFSETS[zone]) + if aware.utcoffset().total_seconds() % 60: + # Historical local-mean-time offsets carry seconds; numpy drops them when + # converting tz-aware datetimes, and the original pandas code shared that + # numpy conversion, so the pure-Python oracle cannot be exact here. The + # pandas oracle below still covers these inputs bit-for-bit. + pytest.skip("sub-minute UTC offset") + with np.testing.suppress_warnings() as sup: + sup.filter(DeprecationWarning) # numpy deprecates tz-aware datetime conversion + got = _days_since_year_start([aware]) + np.testing.assert_array_equal(got, [reference_days(aware)]) + + +def test_year_boundary_crossing_examples(): + """Hand-checked cases where the local and UTC calendar years differ.""" + est = OFFSETS["-05:00"] + # Dec 31 21:00 EST is Jan 1 02:00 UTC: local year 2020 -> day 366 + 2h (leap year) + np.testing.assert_array_equal( + _days_since_year_start([dt.datetime(2020, 12, 31, 21, 0, tzinfo=est)]), + [366 + 2 / 24], + ) + # The same instant written in UTC is day 0 + 2h of 2021 + np.testing.assert_array_equal( + _days_since_year_start([dt.datetime(2021, 1, 1, 2, 0, tzinfo=UTC)]), [2 / 24] + ) + # Jan 1 03:00 IST is still Dec 31 21:30 UTC: local year 2021 -> negative day + ist = OFFSETS["+05:30"] + np.testing.assert_array_equal( + _days_since_year_start([dt.datetime(2021, 1, 1, 3, 0, tzinfo=ist)]), + [-(2.5 / 24)], + ) + # Mid-year, the offset only shifts the fraction of the day + np.testing.assert_array_equal( + _days_since_year_start([dt.datetime(2020, 6, 1, 12, tzinfo=est)]), + [reference_days(dt.datetime(2020, 6, 1, 17))], + ) + + +@pytest.mark.parametrize( + "spelling", + ["-05:00", "-0500", "-05", "+05:30", "+0530", "+14:00", "Z", "+00:00"], +) +def test_iso_strings_with_offsets(spelling): + """Every ISO offset spelling takes the year from the wall-clock digits.""" + wall = "2020-12-31T21:00:00" + aware = dt.datetime.fromisoformat( + wall + + ( + "+00:00" + if spelling == "Z" + else spelling + if ":" in spelling or len(spelling) == 3 + else spelling[:3] + ":" + spelling[3:] + ) + ) + with np.testing.suppress_warnings() as sup: + sup.filter(DeprecationWarning) + got = _days_since_year_start([wall + spelling]) + np.testing.assert_array_equal(got, [reference_days(aware)]) + + +def test_mixed_naive_and_aware_lists(): + """Lists may mix naive datetimes, aware datetimes, dates and strings.""" + items = [ + dt.datetime(2020, 12, 31, 21, 0, tzinfo=OFFSETS["-05:00"]), + dt.datetime(2020, 6, 1), + dt.date(2021, 3, 1), + "2021-12-31T23:30Z", + ] + expected = [ + reference_days(items[0]), + reference_days(items[1]), + reference_days(dt.datetime(2021, 3, 1)), + reference_days(dt.datetime(2021, 12, 31, 23, 30, tzinfo=UTC)), + ] + with np.testing.suppress_warnings() as sup: + sup.filter(DeprecationWarning) + got = _days_since_year_start(items) + np.testing.assert_array_equal(got, expected) + + +# ------------------------------------------------------------ pandas oracle -- + + +@pytest.fixture +def pandas_oracle(): + """The exact pre-refactor expression, for environments that have pandas.""" + pd = pytest.importorskip("pandas") + + def _oracle(dates): + start_years = np.array( + [pd.Timestamp(pd.Timestamp(d).year, 1, 1) for d in dates], + dtype="datetime64", + ) + return (np.array(dates, dtype="datetime64") - start_years) / np.timedelta64( + 1, "D" + ) + + return pd, _oracle + + +def test_matches_pandas_expression_exactly(pandas_oracle): + """Bit-identical to the original pandas code over the full date x zone sweep.""" + pd, oracle = pandas_oracle + dates = [d for d in NAIVE_DATES if 1678 < d.year < 2262] + inputs = list(dates) + for zone in OFFSETS.values(): + inputs += [d.replace(tzinfo=zone) for d in dates if d.year >= 1900] + inputs += [pd.Timestamp(d) for d in dates] + inputs += [pd.Timestamp(d, tz="US/Eastern") for d in dates if d.year >= 1900] + with np.testing.suppress_warnings() as sup: + sup.filter(DeprecationWarning) + np.testing.assert_array_equal(_days_since_year_start(inputs), oracle(inputs)) + + +def test_pandas_containers(pandas_oracle): + """DatetimeIndex (naive and tz-aware) and Series behave like the oracle.""" + pd, oracle = pandas_oracle + idx = pd.date_range("2020-12-30 18:00", periods=12, freq="3h") + tz_idx = pd.date_range("2020-12-30 18:00", periods=12, freq="3h", tz="US/Eastern") + with np.testing.suppress_warnings() as sup: + sup.filter(DeprecationWarning) + for container in (idx, tz_idx, pd.Series(idx), idx.to_numpy()): + np.testing.assert_array_equal( + _days_since_year_start(container), oracle(container) + ) + + +# ------------------------------------------------------------- end-to-end -- + + +def test_nat_is_rejected(): + """NaT entries raise instead of propagating silently.""" + with pytest.raises(ValueError, match="NaT"): + _days_since_year_start(np.array(["NaT", "2020-06-01"], dtype="datetime64[D]")) + with pytest.raises(ValueError, match="NaT"): + _days_since_year_start([None, dt.datetime(2020, 6, 1)]) + + +def test_insolation_shapes_daily_and_clipping(): + """End-to-end call on 1-D and 2-D grids, including daily max and clipping.""" + dates = np.array(["2020-06-21T12:00", "2020-12-21T00:00"], dtype="datetime64[s]") + lat = np.linspace(-90, 90, 5) + lon = np.linspace(0, 360, 5, endpoint=False) + out = insolation(dates, lat, lon) + assert out.shape == (2, 5) and np.all(out >= 0) + unclipped = insolation(dates, lat, lon, clip_zero=False) + assert np.any(unclipped < 0) + out2d = insolation(dates, lat, lon, enforce_2d=True, daily=True) + assert out2d.shape == (2, 5, 5) + # daily max is longitude-independent + np.testing.assert_array_equal(out2d, np.repeat(out2d[..., :1], 5, axis=-1)) + + +def test_insolation_timezone_aware_end_to_end(): + """The full model sees identical fields for the same instant in any zone.""" + lon, lat = np.meshgrid( + np.linspace(0, 360, 9, endpoint=False), np.linspace(-60, 60, 7) + ) + est = dt.datetime(2020, 6, 30, 21, 0, tzinfo=OFFSETS["-05:00"]) + utc = dt.datetime(2020, 7, 1, 2, 0, tzinfo=UTC) + with np.testing.suppress_warnings() as sup: + sup.filter(DeprecationWarning) + a, b = insolation([est], lat, lon), insolation([utc], lat, lon) + assert a.shape == (1, 7, 9) + np.testing.assert_array_equal(a, b) diff --git a/uv.lock b/uv.lock index c0481bbcf8..7ea2ce219b 100644 --- a/uv.lock +++ b/uv.lock @@ -5882,9 +5882,7 @@ dependencies = [ { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "(platform_machine == 'ARM64' and sys_platform == 'win32') or (platform_machine != 'ARM64' and extra != 'extra-18-nvidia-physicsnemo-cu12') or (platform_machine != 'ARM64' and extra == 'extra-18-nvidia-physicsnemo-cu12' and extra == 'extra-18-nvidia-physicsnemo-cu13') or (platform_machine != 'ARM64' and extra == 'extra-18-nvidia-physicsnemo-cu12' and extra == 'extra-18-nvidia-physicsnemo-natten-cu12' and extra == 'extra-18-nvidia-physicsnemo-natten-cu13') or (platform_machine != 'ARM64' and extra == 'extra-18-nvidia-physicsnemo-cu12' and extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu12' and extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu13') or (sys_platform != 'win32' and extra == 'extra-18-nvidia-physicsnemo-cu13') or (sys_platform != 'win32' and extra != 'extra-18-nvidia-physicsnemo-cu12') or (sys_platform != 'win32' and extra == 'extra-18-nvidia-physicsnemo-natten-cu12' and extra == 'extra-18-nvidia-physicsnemo-natten-cu13') or (sys_platform != 'win32' and extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu12' and extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu13')" }, { name = "nvtx" }, { name = "omegaconf" }, - { name = "onnx" }, { name = "packaging" }, - { name = "pandas" }, { name = "psutil" }, { name = "requests" }, { name = "s3fs" }, @@ -5894,9 +5892,6 @@ dependencies = [ { name = "torch", version = "2.11.0+cu128", source = { registry = "https://download.pytorch.org/whl/cu128" }, marker = "extra == 'extra-18-nvidia-physicsnemo-cu12' or (extra == 'extra-18-nvidia-physicsnemo-natten-cu12' and extra == 'extra-18-nvidia-physicsnemo-natten-cu13') or (extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu12' and extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu13')" }, { name = "torch", version = "2.12.0", source = { registry = "https://pypi.org/simple" }, marker = "(extra == 'extra-18-nvidia-physicsnemo-cu12' and extra == 'extra-18-nvidia-physicsnemo-cu13') or (extra == 'extra-18-nvidia-physicsnemo-natten-cu12' and extra == 'extra-18-nvidia-physicsnemo-natten-cu13') or (extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu12' and extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu13') or (extra != 'extra-18-nvidia-physicsnemo-cu12' and extra != 'extra-18-nvidia-physicsnemo-cu13')" }, { name = "torch", version = "2.12.0+cu130", source = { registry = "https://download.pytorch.org/whl/cu130" }, marker = "extra == 'extra-18-nvidia-physicsnemo-cu13' or (extra == 'extra-18-nvidia-physicsnemo-natten-cu12' and extra == 'extra-18-nvidia-physicsnemo-natten-cu13') or (extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu12' and extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu13')" }, - { name = "torchvision", version = "0.26.0+cu128", source = { registry = "https://download.pytorch.org/whl/cu128" }, marker = "extra == 'extra-18-nvidia-physicsnemo-cu12' or (extra == 'extra-18-nvidia-physicsnemo-natten-cu12' and extra == 'extra-18-nvidia-physicsnemo-natten-cu13') or (extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu12' and extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu13')" }, - { name = "torchvision", version = "0.27.0", source = { registry = "https://pypi.org/simple" }, marker = "(extra == 'extra-18-nvidia-physicsnemo-cu12' and extra == 'extra-18-nvidia-physicsnemo-cu13') or (extra == 'extra-18-nvidia-physicsnemo-natten-cu12' and extra == 'extra-18-nvidia-physicsnemo-natten-cu13') or (extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu12' and extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu13') or (extra != 'extra-18-nvidia-physicsnemo-cu12' and extra != 'extra-18-nvidia-physicsnemo-cu13')" }, - { name = "torchvision", version = "0.27.0+cu130", source = { registry = "https://download.pytorch.org/whl/cu130" }, marker = "extra == 'extra-18-nvidia-physicsnemo-cu13' or (extra == 'extra-18-nvidia-physicsnemo-natten-cu12' and extra == 'extra-18-nvidia-physicsnemo-natten-cu13') or (extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu12' and extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu13')" }, { name = "tqdm" }, { name = "treelib" }, { name = "urllib3" }, @@ -5926,6 +5921,7 @@ cu13 = [ datapipes-extras = [ { name = "dask" }, { name = "netcdf4" }, + { name = "pandas" }, { name = "tfrecord" }, { name = "xarray" }, { name = "zarr", version = "3.1.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.12' or (extra == 'extra-18-nvidia-physicsnemo-cu12' and extra == 'extra-18-nvidia-physicsnemo-cu13') or (extra == 'extra-18-nvidia-physicsnemo-natten-cu12' and extra == 'extra-18-nvidia-physicsnemo-natten-cu13') or (extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu12' and extra == 'extra-18-nvidia-physicsnemo-transformer-engine-cu13')" }, @@ -5934,6 +5930,7 @@ datapipes-extras = [ gnns = [ { name = "line-profiler" }, { name = "mlflow" }, + { name = "pandas" }, { name = "pyvista" }, { name = "pyyaml" }, { name = "scipy" }, @@ -5952,6 +5949,7 @@ mesh-extras = [ model-extras = [ { name = "line-profiler" }, { name = "mlflow" }, + { name = "pandas" }, { name = "pyvista" }, { name = "scipy" }, { name = "vtk" }, @@ -6037,9 +6035,10 @@ requires-dist = [ { name = "nvidia-dali-cuda130", marker = "extra == 'cu13'", index = "https://pypi.nvidia.com/" }, { name = "nvtx", specifier = ">=0.2.10" }, { name = "omegaconf", specifier = ">=2.3.0" }, - { name = "onnx", specifier = ">=1.14.0" }, { name = "packaging", specifier = ">=24.2" }, - { name = "pandas", specifier = ">=2.2.0" }, + { name = "pandas", marker = "extra == 'datapipes-extras'", specifier = ">=2.2.0" }, + { name = "pandas", marker = "extra == 'gnns'", specifier = ">=2.2.0" }, + { name = "pandas", marker = "extra == 'model-extras'", specifier = ">=2.2.0" }, { name = "psutil", specifier = ">=6.0.0" }, { name = "pylibraft-cu12", marker = "extra == 'cu12'", specifier = ">=26.2.0", index = "https://pypi.nvidia.com/" }, { name = "pylibraft-cu13", marker = "extra == 'cu13'", specifier = ">=26.2.0", index = "https://pypi.nvidia.com/" }, @@ -6064,7 +6063,6 @@ requires-dist = [ { name = "torch-geometric", marker = "extra == 'gnns'" }, { name = "torch-scatter", marker = "extra == 'gnns'" }, { name = "torch-sparse", marker = "extra == 'gnns'" }, - { name = "torchvision", specifier = ">=0.25.0a0" }, { name = "torchvision", marker = "extra == 'cu12'", specifier = ">=0.25.0a0", index = "https://download.pytorch.org/whl/cu128", conflict = { package = "nvidia-physicsnemo", extra = "cu12" } }, { name = "torchvision", marker = "extra == 'cu13'", specifier = ">=0.25.0a0", index = "https://download.pytorch.org/whl/cu130", conflict = { package = "nvidia-physicsnemo", extra = "cu13" } }, { name = "tqdm", specifier = ">=4.60.0" },