Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -114,6 +114,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

### Dependencies

- Drops `onnx`, `torchvision` and `pandas` from the required dependencies.
`pandas` is now optional via `datapipes-extras` or `model-extras`.

## [2.2.1] - 2026-XX-YY

### Fixed
Expand Down
5 changes: 5 additions & 0 deletions physicsnemo/core/version_check.py
Original file line number Diff line number Diff line change
Expand Up @@ -214,6 +214,11 @@ def _format_install_hint(
"h5py",
group="datapipes-extras",
),
"pandas": _format_install_hint(
"pandas",
group="datapipes-extras",
direct_hint="pip install pandas (also included in [model-extras])",
),
"netCDF4": _format_install_hint(
"netCDF4",
group="model-extras",
Expand Down
3 changes: 2 additions & 1 deletion physicsnemo/datapipes/gnn/drivaernet_dataset.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,6 @@
from pathlib import Path
from typing import Iterable

import pandas as pd
import torch
from torch import Tensor
from torch.utils.data import Dataset
Expand All @@ -28,6 +27,8 @@
from physicsnemo.datapipes.meta import DatapipeMetaData
from physicsnemo.nn.module.gnn_layers.utils import PyGData

pd = OptionalImport("pandas")

# Lazy imports for optional dependencies
pyg = OptionalImport("torch_geometric")
pv = OptionalImport("pyvista")
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,6 @@
from typing import Optional, Sequence, Union

import numpy as np
import pandas as pd
import torch
from omegaconf import DictConfig, OmegaConf

Expand All @@ -34,6 +33,8 @@
from . import couplers
from .timeseries_dataset import TimeSeriesDataset

pd = OptionalImport("pandas")

xr = OptionalImport("xarray")

logger = logging.getLogger(__name__)
Expand Down
15 changes: 8 additions & 7 deletions physicsnemo/datapipes/healpix/couplers.py
Original file line number Diff line number Diff line change
Expand Up @@ -20,11 +20,12 @@
from typing import Sequence

import numpy as np
import pandas as pd
import torch as th

from physicsnemo.core.version_check import OptionalImport

pd = OptionalImport("pandas")

xr = OptionalImport("xarray")

logger = logging.getLogger(__name__)
Expand All @@ -46,7 +47,7 @@ def __init__(
presteps: int = 0,
input_time_dim: int = 2,
output_time_dim: int = 2,
input_times: Sequence = [pd.Timedelta("24h"), pd.Timedelta("48h")],
input_times: Sequence = ["24h", "48h"],
prepared_coupled_data=True,
):
"""
Expand All @@ -68,8 +69,8 @@ def __init__(
output_time_dim: int, optional
number of output times for each model step, default 2
input_times: Sequence, optional
sequence of pandas Timedelta objects that indicate which times are to be coupled,
default [pd.Timedelta("24h"), pd.Timedelta("48h")]
sequence of pandas Timedelta objects (or strings accepted by ``pd.Timedelta``)
that indicate which times are to be coupled, default ["24h", "48h"]
prepared_coupled_data: boolean, optional
If True assumes data in dataset has been prepared approiately for training:
averages have already been calculated so that each time step denotes
Expand Down Expand Up @@ -289,7 +290,7 @@ def __init__(
input_time_dim: int = 2,
output_time_dim: int = 2,
averaging_window: str = "24h",
input_times: Sequence = [pd.Timedelta("24h"), pd.Timedelta("48h")],
input_times: Sequence = ["24h", "48h"],
prepared_coupled_data=True,
):
"""
Expand All @@ -313,8 +314,8 @@ def __init__(
averaging_window: str, optional
period over which coupled data is averaged before sent back to model, default "24h"
input_times: Sequence, optional
sequence of pandas Timedelta objects that indicate which times are to be coupled,
default [pd.Timedelta("24h"), pd.Timedelta("48h")]
sequence of pandas Timedelta objects (or strings accepted by ``pd.Timedelta``)
that indicate which times are to be coupled, default ["24h", "48h"]
prepared_coupled_data: boolean, optional
If True assumes data in dataset has been prepared approiately for training:
averages have already been calculated so that each time step denotes
Expand Down
3 changes: 2 additions & 1 deletion physicsnemo/datapipes/healpix/timeseries_dataset.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,6 @@
from typing import Optional, Sequence, Union

import numpy as np
import pandas as pd
import torch
from omegaconf import DictConfig, OmegaConf

Expand All @@ -32,6 +31,8 @@
from physicsnemo.datapipes.meta import DatapipeMetaData
from physicsnemo.utils.insolation import insolation

pd = OptionalImport("pandas")

xr = OptionalImport("xarray")

logger = logging.getLogger(__name__)
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -24,7 +24,9 @@
from dataclasses import dataclass, field

import numpy as np
import pandas as pd
from physicsnemo.core.version_check import OptionalImport

pd = OptionalImport("pandas")

# Recipe-side directory containing per-sensor `*_normalizations.csv` files
# (and the ERA5 stats CSV consumed by `loaders.era5`). When unset, sensor
Expand Down
4 changes: 3 additions & 1 deletion physicsnemo/experimental/datapipes/healda/dataset.py
Original file line number Diff line number Diff line change
Expand Up @@ -52,14 +52,16 @@
from typing import Union

import numpy as np
import pandas as pd
import torch

from physicsnemo.experimental.datapipes.healda.indexing import get_flat_indexer
from physicsnemo.experimental.datapipes.healda.loaders.era5 import get_batch_info
from physicsnemo.experimental.datapipes.healda.protocols import ObsLoader, Transform
from physicsnemo.experimental.datapipes.healda.time_utils import as_cftime
from physicsnemo.experimental.datapipes.healda.types import VariableConfig
from physicsnemo.core.version_check import OptionalImport

pd = OptionalImport("pandas")

# HEALPix level-6 pixel count: 12 * 4^6
NPIX_HPX6 = 12 * 4**6
Expand Down
5 changes: 3 additions & 2 deletions physicsnemo/experimental/datapipes/healda/loaders/era5.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,6 @@
from typing import Optional

import numpy as np
import pandas as pd

from physicsnemo.core.version_check import OptionalImport

Expand All @@ -35,6 +34,8 @@
from physicsnemo.experimental.datapipes.healda.loaders.zarr_loader import NO_LEVEL, ZarrLoader
from physicsnemo.experimental.datapipes.healda.types import BatchInfo, TimeUnit, VariableConfig

pd = OptionalImport("pandas")

__all__ = ["ERA5Loader", "get_batch_info"]

SST_LAND_FILL_VALUE = 290
Expand Down Expand Up @@ -174,7 +175,7 @@ def _encode_channel(channel) -> str:
return name


def _load_raw_stats(config: VariableConfig) -> pd.DataFrame:
def _load_raw_stats(config: VariableConfig) -> "pd.DataFrame":
if config.name == "ufs":
file_name = "ufs_v0_stats.csv"
elif config.name == "era5":
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,6 @@

import fsspec
import numpy as np
import pandas as pd
from physicsnemo.core.version_check import OptionalImport

pa = OptionalImport("pyarrow")
Expand All @@ -53,6 +52,8 @@
from physicsnemo.experimental.datapipes.healda.configs.sensors import SENSOR_CONFIGS
from physicsnemo.experimental.datapipes.healda.transforms.obs_filtering import filter_observations

pd = OptionalImport("pandas")

LOCAL_CHANNEL_ID = pa.field("local_channel_id", pa.uint16())


Expand Down Expand Up @@ -163,7 +164,7 @@ def channel_table(self) -> pa.Table:
array = pa.array(local_channel_ids).cast(LOCAL_CHANNEL_ID.type)
return table.append_column(LOCAL_CHANNEL_ID, array)

def _get_interval_times(self, dt: datetime) -> pd.DatetimeIndex:
def _get_interval_times(self, dt: datetime) -> "pd.DatetimeIndex":
start, end = self.obs_context_hours
start += self.data_spacing
return pd.date_range(
Expand All @@ -172,7 +173,7 @@ def _get_interval_times(self, dt: datetime) -> pd.DatetimeIndex:
freq=f"{self.data_spacing}h",
)

def _get_parquet_files_to_read(self, interval_times: pd.DatetimeIndex):
def _get_parquet_files_to_read(self, interval_times: "pd.DatetimeIndex"):
required_dates = {t.strftime("%Y%m%d") for t in interval_times}
for sensor in self.sensors:
for date in required_dates:
Expand Down Expand Up @@ -247,7 +248,7 @@ def _add_channel_metadata(self, table):
GLOBAL_CHANNEL_ID.name,
)

async def sel_time(self, times: pd.DatetimeIndex) -> dict:
async def sel_time(self, times: "pd.DatetimeIndex") -> dict:
"""Load observation data for specified times.

Args:
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -27,10 +27,11 @@

import cftime
import numpy as np
import pandas as pd

from physicsnemo.core.version_check import OptionalImport

pd = OptionalImport("pandas")

xr = OptionalImport("xarray")
zarr = OptionalImport("zarr")
_zarr_sync = OptionalImport("zarr.core.sync")
Expand Down
6 changes: 4 additions & 2 deletions physicsnemo/experimental/datapipes/healda/protocols.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,8 +27,10 @@
from typing import Any, Protocol, runtime_checkable

import cftime
import pandas as pd
import torch
from physicsnemo.core.version_check import OptionalImport

pd = OptionalImport("pandas")


@runtime_checkable
Expand All @@ -51,7 +53,7 @@ async def sel_time(self, times):
return {"obs": tables}
"""

async def sel_time(self, times: pd.DatetimeIndex) -> dict[str, list[Any]]:
async def sel_time(self, times: "pd.DatetimeIndex") -> dict[str, list[Any]]:
"""Load observation data for the given timestamps.

Args:
Expand Down
4 changes: 3 additions & 1 deletion physicsnemo/experimental/datapipes/healda/time_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,9 @@

import cftime
import numpy as np
import pandas as pd
from physicsnemo.core.version_check import OptionalImport

pd = OptionalImport("pandas")


def as_pydatetime(time) -> datetime.datetime:
Expand Down
4 changes: 3 additions & 1 deletion physicsnemo/models/dlwp_healpix/HEALPixRecUNet.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,13 +27,13 @@
from dataclasses import dataclass
from typing import Any, Dict, Sequence

import pandas as pd
import torch
from hydra.utils import instantiate
from omegaconf import DictConfig

from physicsnemo.core.meta import ModelMetaData
from physicsnemo.core.module import Module
from physicsnemo.core.version_check import OptionalImport
from physicsnemo.nn.module.hpx import HEALPixFoldFaces, HEALPixUnfoldFaces

from .layers import (
Expand All @@ -43,6 +43,8 @@
_remap_obj,
)

pd = OptionalImport("pandas")

logger = logging.getLogger(__name__)


Expand Down
54 changes: 47 additions & 7 deletions physicsnemo/utils/insolation.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,8 +14,53 @@
# See the License for the specific language governing permissions and
# limitations under the License.

import datetime
import re

import numpy as np
import pandas as pd

_LEADING_YEAR = re.compile(r"\s*(-?\d{4,})")


def _calendar_year(d) -> int:
"""Calendar year of one date-like value, as ``pandas.Timestamp(d).year`` reports it.

For timezone-aware datetimes and ISO strings that carry a UTC offset, this is
the year of the *local* wall-clock time, not of the UTC instant numpy stores.
"""
if isinstance(
d, datetime.date
): # datetime.datetime, datetime.date, pandas.Timestamp
return d.year
if isinstance(d, str):
match = _LEADING_YEAR.match(d)
if match:
return int(match.group(1))
return int(np.datetime64(d).astype("datetime64[Y]").astype(np.int64)) + 1970


def _days_since_year_start(dates) -> np.ndarray:
"""Fractional days from January 1st 00:00 of each date's calendar year.

Reproduces ``(np.array(dates, dtype="datetime64") - Timestamp(year, 1, 1)) / 1 day``
from the earlier pandas-based implementation exactly: instants are numpy's
conversion of ``dates`` (UTC for timezone-aware input), the year is the local
calendar year, and year starts are held at microsecond resolution, which is the
unit numpy inferred from the ``Timestamp`` objects, so unit promotion in the
subtraction is unchanged for every input dtype.
"""
instants = np.array(dates, dtype="datetime64")
if np.isnat(instants).any():
raise ValueError("insolation: 'dates' contains NaT or None entries")
if isinstance(dates, np.ndarray) and np.issubdtype(dates.dtype, np.datetime64):
# Naive datetime64 arrays: the calendar year is unambiguous, stay vectorized.
years = instants.astype("datetime64[Y]")
else:
flat = np.asarray(dates, dtype=object).ravel()
years = np.array([_calendar_year(d) for d in flat], dtype=np.int64) - 1970
years = years.astype("datetime64[Y]").reshape(instants.shape)
start_years = years.astype("datetime64[us]")
return (instants - start_years) / np.timedelta64(1, "D")


def insolation(
Expand Down Expand Up @@ -69,12 +114,7 @@ def insolation(
beta = np.sqrt(1 - ecc**2.0)

# Get the day of year as a float.
start_years = np.array(
[pd.Timestamp(pd.Timestamp(d).year, 1, 1) for d in dates], dtype="datetime64"
)
days_arr = (np.array(dates, dtype="datetime64") - start_years) / np.timedelta64(
1, "D"
)
days_arr = _days_since_year_start(dates)
for d in range(n_dim):
days_arr = np.expand_dims(days_arr, -1)
# For daily max values, set the day to 0.5 and the longitude everywhere to 0 (this is approx noon)
Expand Down
5 changes: 2 additions & 3 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -21,14 +21,11 @@ classifiers = [
dynamic = ["version", "optional_dependencies"]

dependencies = [
"onnx>=1.14.0",
"warp-lang>=1.14.0",
"pandas>=2.2.0",
"nvtx>=0.2.10",
"treelib>=1.2.5",
"numpy>=1.22.4",
"torch>=2.10.0",
"torchvision>=0.25.0a0",
Comment thread
coreyjadams marked this conversation as resolved.
"tqdm>=4.60.0",
"requests>=2.32.2",
# urllib3 is a transitive dependency (via requests/botocore/etc.); pin to
Expand Down Expand Up @@ -304,10 +301,12 @@ nn-extras = [
]
model-extras = [
"nvidia-physicsnemo[nn-extras]",
"pandas>=2.2.0",
"pyvista>=0.46.4",
"vtk",
]
datapipes-extras = [
"pandas>=2.2.0",
Comment thread
coreyjadams marked this conversation as resolved.
"tfrecord",
"dask",
"netCDF4",
Expand Down
Loading
Loading