diff --git a/README.md b/README.md index d5d2827a..8745472f 100644 --- a/README.md +++ b/README.md @@ -46,8 +46,8 @@ Extra packages: |---------------|-----------------------------------------|----------| | SQLAlchemy | `pip install PyAthena[SQLAlchemy]` | >=2.0.0 | | AioSQLAlchemy | `pip install PyAthena[AioSQLAlchemy]` | >=2.0.0 | -| Pandas | `pip install PyAthena[Pandas]` | >=1.3.0 | -| Arrow | `pip install PyAthena[Arrow]` | >=10.0.0 | +| Pandas | `pip install PyAthena[Pandas]` | >=3.0.0 | +| Arrow | `pip install PyAthena[Arrow]` | >=22.0.0 | | Polars | `pip install PyAthena[Polars]` | >=1.39.0 | ## Usage diff --git a/docs/arrow.md b/docs/arrow.md index d70d82df..a99f1d24 100644 --- a/docs/arrow.md +++ b/docs/arrow.md @@ -293,10 +293,6 @@ cursor = connect( The timeout parameters accept float values in seconds and apply to all S3 operations performed by the cursor, including HeadObject and GetObject operations when retrieving query results. -```{note} -These timeout parameters require PyArrow >= 10.0.0, which added support for configuring S3FileSystem timeouts. -``` - (async-arrow-cursor)= ## AsyncArrowCursor diff --git a/docs/introduction.md b/docs/introduction.md index 03942490..580d139b 100644 --- a/docs/introduction.md +++ b/docs/introduction.md @@ -33,8 +33,8 @@ Extra packages: |---------------|-----------------------------------------|----------| | SQLAlchemy | `pip install PyAthena[SQLAlchemy]` | >=2.0.0 | | AioSQLAlchemy | `pip install PyAthena[AioSQLAlchemy]` | >=2.0.0 | -| Pandas | `pip install PyAthena[Pandas]` | >=1.3.0 | -| Arrow | `pip install PyAthena[Arrow]` | >=10.0.0 | +| Pandas | `pip install PyAthena[Pandas]` | >=3.0.0 | +| Arrow | `pip install PyAthena[Arrow]` | >=22.0.0 | | Polars | `pip install PyAthena[Polars]` | >=1.39.0 | (features)= diff --git a/docs/null_handling.md b/docs/null_handling.md index 5b60d372..7c745abb 100644 --- a/docs/null_handling.md +++ b/docs/null_handling.md @@ -55,7 +55,7 @@ based on actual testing with Athena: | `Cursor` (default) | Athena API | `''` | `None` | ✅ Yes | | `DictCursor` | Athena API | `''` | `None` | ✅ Yes | | `PandasCursor` | CSV file | `NaN` | `NaN` | ❌ No | -| `PandasCursor` + unload | Parquet file | `''` | `NaN` (pandas 3) or `None` (pandas 2) | ✅ Yes | +| `PandasCursor` + unload | Parquet file | `''` | `NaN` | ✅ Yes | | `ArrowCursor` | CSV file | `''` | `''` | ❌ No | | `ArrowCursor` + unload | Parquet file | `''` | `null` | ✅ Yes | | `PolarsCursor` | CSV file | `''` | `null` | ✅ Yes | @@ -177,28 +177,26 @@ df = cursor.execute(""" print(df) # id value description # 0 1 empty_string <- Empty string preserved -# 1 2 NaN null_value <- NULL is NaN (None with pandas 2) +# 1 2 NaN null_value <- NULL is NaN # 2 3 hello normal_string print(df['value'].isna().tolist()) # [False, True, False] <- Only NULL is missing, empty string is not ``` -String columns follow the installed pandas version. -pandas 3 infers its `str` dtype and represents NULL as `NaN`; pandas 2 uses `object` columns and `None`. +String columns use the pandas `str` dtype, which represents NULL as `NaN`. `fetchone()`, `fetchmany()`, and `fetchall()` return the same values as the DataFrame. -Use `isna()` or `pandas.isna()` to detect NULL regardless of the pandas version. +Use `isna()` or `pandas.isna()` to detect NULL. -To get `None` for NULL strings with pandas 3, convert the columns after reading: +To get `None` for NULL strings, convert the columns after reading: ```python df = cursor.execute("SELECT ...").as_pandas() df = df.astype({"value": object}).where(df.notna(), None) ``` -Alternatively, turn off the pandas 3 string dtype for the whole process before executing queries. +Alternatively, turn off the pandas `str` dtype for the whole process before executing queries. String columns then use `object` with `None` for NULL, including rows returned by `fetchone()`, `fetchmany()`, and `fetchall()`. -The `future.infer_string` option exists in pandas 2.1 and later. ```python import pandas as pd diff --git a/pyproject.toml b/pyproject.toml index 0ce6035a..c58f8866 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -56,12 +56,10 @@ aiosqlalchemy = [ "sqlalchemy[asyncio]>=2.0.0", ] pandas = [ - "pandas>=1.3.0; python_version<'3.13'", - "pandas>=2.3.0; python_version>='3.13'", + "pandas>=3.0.0", ] arrow = [ - "pyarrow>=10.0.0; python_version<'3.14'", - "pyarrow>=22.0.0; python_version>='3.14'", + "pyarrow>=22.0.0", ] polars = [ "polars>=1.39.0", @@ -70,12 +68,8 @@ polars = [ [dependency-groups] dev = [ "sqlalchemy[asyncio]>=2.0.0", - "pandas>=1.3.0; python_version<'3.13'", - "pandas>=2.3.0; python_version>='3.13'", - "numpy>=1.26.0; python_version<'3.13'", - "numpy>=2.3.0; python_version>='3.14'", - "pyarrow>=10.0.0; python_version<'3.14'", - "pyarrow>=22.0.0; python_version>='3.14'", + "pandas>=3.0.0", + "pyarrow>=22.0.0", "polars>=1.39.0", "Jinja2>=3.1.0", "mypy>=0.900", diff --git a/tests/pyathena/pandas/test_async_cursor.py b/tests/pyathena/pandas/test_async_cursor.py index a0e9261d..01313627 100644 --- a/tests/pyathena/pandas/test_async_cursor.py +++ b/tests/pyathena/pandas/test_async_cursor.py @@ -17,11 +17,6 @@ from tests import ENV from tests.pyathena.conftest import connect -# pandas 3 infers its "str" dtype for strings, which represents NULL as NaN; pandas 2 uses -# object columns with None. -STRING_TYPE = pd.Series(["a"]).dtype.type -STRING_NULL = pd.Series(["a", None]).iloc[1] - class TestAsyncPandasCursor: def test_binary_null_vs_empty(self, async_pandas_cursor): @@ -592,7 +587,7 @@ def test_empty_and_null_string(self, async_pandas_cursor, parquet_engine): # NULL and empty characters are correctly converted when the UNLOAD option is enabled. np.testing.assert_equal( result_set.fetchall(), - [("", "a"), ("N/A", "a"), ("NULL", "a"), (STRING_NULL, "a")], + [("", "a"), ("N/A", "a"), ("NULL", "a"), (np.nan, "a")], ) else: np.testing.assert_equal( @@ -605,7 +600,7 @@ def test_empty_and_null_string(self, async_pandas_cursor, parquet_engine): # NULL and empty characters are correctly converted when the UNLOAD option is enabled. np.testing.assert_equal( result_set.fetchall(), - [("", "a"), ("N/A", "a"), ("NULL", "a"), (STRING_NULL, "a")], + [("", "a"), ("N/A", "a"), ("NULL", "a"), (np.nan, "a")], ) else: assert result_set.fetchall() == [ diff --git a/tests/pyathena/pandas/test_cursor.py b/tests/pyathena/pandas/test_cursor.py index 0e19ff18..943bc935 100644 --- a/tests/pyathena/pandas/test_cursor.py +++ b/tests/pyathena/pandas/test_cursor.py @@ -20,11 +20,6 @@ from tests import ENV from tests.pyathena.conftest import connect -# pandas 3 infers its "str" dtype for strings, which represents NULL as NaN; pandas 2 uses -# object columns with None. -STRING_TYPE = pd.Series(["a"]).dtype.type -STRING_NULL = pd.Series(["a", None]).iloc[1] - class TestPandasCursor: @pytest.mark.parametrize( @@ -646,17 +641,17 @@ def test_complex_as_pandas(self, pandas_cursor, chunksize): np.int64, np.float64, np.float64, - STRING_TYPE, - STRING_TYPE, + str, + str, np.datetime64, np.object_, np.datetime64, np.object_, - STRING_TYPE, + str, np.object_, - STRING_TYPE, + str, np.object_, - STRING_TYPE, + str, np.object_, ) rows = [ @@ -768,8 +763,8 @@ def test_complex_unload_as_pandas_pyarrow(self, pandas_cursor, parquet_engine): np.int64, np.float32, np.float64, - STRING_TYPE, - STRING_TYPE, + str, + str, np.datetime64, np.object_, np.object_, @@ -1200,7 +1195,7 @@ def test_null_vs_empty_string(self, pandas_cursor, parquet_engine): # NULL and empty characters are correctly converted when the UNLOAD option is enabled. np.testing.assert_equal( pandas_cursor.fetchall(), - [("", "a"), ("N/A", "a"), ("NULL", "a"), (STRING_NULL, "a")], + [("", "a"), ("N/A", "a"), ("NULL", "a"), (np.nan, "a")], ) else: np.testing.assert_equal( @@ -1212,7 +1207,7 @@ def test_null_vs_empty_string(self, pandas_cursor, parquet_engine): # NULL and empty characters are correctly converted when the UNLOAD option is enabled. np.testing.assert_equal( pandas_cursor.fetchall(), - [("", "a"), ("N/A", "a"), ("NULL", "a"), (STRING_NULL, "a")], + [("", "a"), ("N/A", "a"), ("NULL", "a"), (np.nan, "a")], ) else: assert pandas_cursor.fetchall() == [ diff --git a/uv.lock b/uv.lock index 79adb7af..aaf1f338 100644 --- a/uv.lock +++ b/uv.lock @@ -9,10 +9,10 @@ resolution-markers = [ "python_full_version == '3.13.*' and sys_platform == 'emscripten'", "python_full_version == '3.13.*' and sys_platform != 'emscripten' and sys_platform != 'win32'", "python_full_version == '3.12.*' and sys_platform == 'win32'", - "python_full_version < '3.12' and sys_platform == 'win32'", "python_full_version == '3.12.*' and sys_platform == 'emscripten'", - "python_full_version < '3.12' and sys_platform == 'emscripten'", "python_full_version == '3.12.*' and sys_platform != 'emscripten' and sys_platform != 'win32'", + "python_full_version < '3.12' and sys_platform == 'win32'", + "python_full_version < '3.12' and sys_platform == 'emscripten'", "python_full_version < '3.12' and sys_platform != 'emscripten' and sys_platform != 'win32'", ] @@ -957,8 +957,6 @@ dev = [ { name = "jinja2" }, { name = "mypy" }, { name = "myst-parser" }, - { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.12'" }, - { name = "numpy", version = "2.5.3", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version == '3.12.*' or python_full_version >= '3.14'" }, { name = "pandas" }, { name = "polars" }, { name = "pyarrow" }, @@ -981,11 +979,9 @@ requires-dist = [ { name = "boto3", specifier = ">=1.41.2" }, { name = "botocore", specifier = ">=1.41.2" }, { name = "fsspec" }, - { name = "pandas", marker = "python_full_version >= '3.13' and extra == 'pandas'", specifier = ">=2.3.0" }, - { name = "pandas", marker = "python_full_version < '3.13' and extra == 'pandas'", specifier = ">=1.3.0" }, + { name = "pandas", marker = "extra == 'pandas'", specifier = ">=3.0.0" }, { name = "polars", marker = "extra == 'polars'", specifier = ">=1.39.0" }, - { name = "pyarrow", marker = "python_full_version >= '3.14' and extra == 'arrow'", specifier = ">=22.0.0" }, - { name = "pyarrow", marker = "python_full_version < '3.14' and extra == 'arrow'", specifier = ">=10.0.0" }, + { name = "pyarrow", marker = "extra == 'arrow'", specifier = ">=22.0.0" }, { name = "python-dateutil" }, { name = "sqlalchemy", marker = "extra == 'sqlalchemy'", specifier = ">=2.0.0" }, { name = "sqlalchemy", extras = ["asyncio"], marker = "extra == 'aiosqlalchemy'", specifier = ">=2.0.0" }, @@ -1000,13 +996,9 @@ dev = [ { name = "jinja2", specifier = ">=3.1.0" }, { name = "mypy", specifier = ">=0.900" }, { name = "myst-parser" }, - { name = "numpy", marker = "python_full_version < '3.13'", specifier = ">=1.26.0" }, - { name = "numpy", marker = "python_full_version >= '3.14'", specifier = ">=2.3.0" }, - { name = "pandas", marker = "python_full_version < '3.13'", specifier = ">=1.3.0" }, - { name = "pandas", marker = "python_full_version >= '3.13'", specifier = ">=2.3.0" }, + { name = "pandas", specifier = ">=3.0.0" }, { name = "polars", specifier = ">=1.39.0" }, - { name = "pyarrow", marker = "python_full_version < '3.14'", specifier = ">=10.0.0" }, - { name = "pyarrow", marker = "python_full_version >= '3.14'", specifier = ">=22.0.0" }, + { name = "pyarrow", specifier = ">=22.0.0" }, { name = "pytest", specifier = ">=3.5" }, { name = "pytest-asyncio" }, { name = "pytest-cov" },