From 03f5b5dee66bfb09c1405a8ce3e79cb5ee786388 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Tue, 29 Sep 2026 01:38:50 +0900 Subject: [PATCH 1/2] Require pandas 3.0 and pyarrow 22.0 The pandas and arrow extras declared floors that CI never installed (pandas>=1.3.0, pyarrow>=10.0.0), so they promised compatibility that was not verified. With Python 3.10 dropped, the extras can require the versions the project tests: - pandas>=3.0.0, the first release of the only major version CI tests. - pyarrow>=22.0.0, the first release with wheels for every supported Python version, including 3.14. The dev group follows the extras and no longer lists numpy, whose markers only restated pandas' own numpy requirement. The pandas cursor tests and the NULL handling guide drop their pandas 2 branches. Closes #852 Co-Authored-By: Claude Opus 5.5 --- README.md | 4 ++-- docs/introduction.md | 4 ++-- docs/null_handling.md | 14 ++++++------- pyproject.toml | 14 ++++--------- tests/pyathena/pandas/test_async_cursor.py | 9 ++------- tests/pyathena/pandas/test_cursor.py | 23 +++++++++------------- uv.lock | 20 ++++++------------- 7 files changed, 31 insertions(+), 57 deletions(-) diff --git a/README.md b/README.md index d5d2827a..8745472f 100644 --- a/README.md +++ b/README.md @@ -46,8 +46,8 @@ Extra packages: |---------------|-----------------------------------------|----------| | SQLAlchemy | `pip install PyAthena[SQLAlchemy]` | >=2.0.0 | | AioSQLAlchemy | `pip install PyAthena[AioSQLAlchemy]` | >=2.0.0 | -| Pandas | `pip install PyAthena[Pandas]` | >=1.3.0 | -| Arrow | `pip install PyAthena[Arrow]` | >=10.0.0 | +| Pandas | `pip install PyAthena[Pandas]` | >=3.0.0 | +| Arrow | `pip install PyAthena[Arrow]` | >=22.0.0 | | Polars | `pip install PyAthena[Polars]` | >=1.39.0 | ## Usage diff --git a/docs/introduction.md b/docs/introduction.md index 03942490..580d139b 100644 --- a/docs/introduction.md +++ b/docs/introduction.md @@ -33,8 +33,8 @@ Extra packages: |---------------|-----------------------------------------|----------| | SQLAlchemy | `pip install PyAthena[SQLAlchemy]` | >=2.0.0 | | AioSQLAlchemy | `pip install PyAthena[AioSQLAlchemy]` | >=2.0.0 | -| Pandas | `pip install PyAthena[Pandas]` | >=1.3.0 | -| Arrow | `pip install PyAthena[Arrow]` | >=10.0.0 | +| Pandas | `pip install PyAthena[Pandas]` | >=3.0.0 | +| Arrow | `pip install PyAthena[Arrow]` | >=22.0.0 | | Polars | `pip install PyAthena[Polars]` | >=1.39.0 | (features)= diff --git a/docs/null_handling.md b/docs/null_handling.md index 5b60d372..7c745abb 100644 --- a/docs/null_handling.md +++ b/docs/null_handling.md @@ -55,7 +55,7 @@ based on actual testing with Athena: | `Cursor` (default) | Athena API | `''` | `None` | ✅ Yes | | `DictCursor` | Athena API | `''` | `None` | ✅ Yes | | `PandasCursor` | CSV file | `NaN` | `NaN` | ❌ No | -| `PandasCursor` + unload | Parquet file | `''` | `NaN` (pandas 3) or `None` (pandas 2) | ✅ Yes | +| `PandasCursor` + unload | Parquet file | `''` | `NaN` | ✅ Yes | | `ArrowCursor` | CSV file | `''` | `''` | ❌ No | | `ArrowCursor` + unload | Parquet file | `''` | `null` | ✅ Yes | | `PolarsCursor` | CSV file | `''` | `null` | ✅ Yes | @@ -177,28 +177,26 @@ df = cursor.execute(""" print(df) # id value description # 0 1 empty_string <- Empty string preserved -# 1 2 NaN null_value <- NULL is NaN (None with pandas 2) +# 1 2 NaN null_value <- NULL is NaN # 2 3 hello normal_string print(df['value'].isna().tolist()) # [False, True, False] <- Only NULL is missing, empty string is not ``` -String columns follow the installed pandas version. -pandas 3 infers its `str` dtype and represents NULL as `NaN`; pandas 2 uses `object` columns and `None`. +String columns use the pandas `str` dtype, which represents NULL as `NaN`. `fetchone()`, `fetchmany()`, and `fetchall()` return the same values as the DataFrame. -Use `isna()` or `pandas.isna()` to detect NULL regardless of the pandas version. +Use `isna()` or `pandas.isna()` to detect NULL. -To get `None` for NULL strings with pandas 3, convert the columns after reading: +To get `None` for NULL strings, convert the columns after reading: ```python df = cursor.execute("SELECT ...").as_pandas() df = df.astype({"value": object}).where(df.notna(), None) ``` -Alternatively, turn off the pandas 3 string dtype for the whole process before executing queries. +Alternatively, turn off the pandas `str` dtype for the whole process before executing queries. String columns then use `object` with `None` for NULL, including rows returned by `fetchone()`, `fetchmany()`, and `fetchall()`. -The `future.infer_string` option exists in pandas 2.1 and later. ```python import pandas as pd diff --git a/pyproject.toml b/pyproject.toml index 0ce6035a..c58f8866 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -56,12 +56,10 @@ aiosqlalchemy = [ "sqlalchemy[asyncio]>=2.0.0", ] pandas = [ - "pandas>=1.3.0; python_version<'3.13'", - "pandas>=2.3.0; python_version>='3.13'", + "pandas>=3.0.0", ] arrow = [ - "pyarrow>=10.0.0; python_version<'3.14'", - "pyarrow>=22.0.0; python_version>='3.14'", + "pyarrow>=22.0.0", ] polars = [ "polars>=1.39.0", @@ -70,12 +68,8 @@ polars = [ [dependency-groups] dev = [ "sqlalchemy[asyncio]>=2.0.0", - "pandas>=1.3.0; python_version<'3.13'", - "pandas>=2.3.0; python_version>='3.13'", - "numpy>=1.26.0; python_version<'3.13'", - "numpy>=2.3.0; python_version>='3.14'", - "pyarrow>=10.0.0; python_version<'3.14'", - "pyarrow>=22.0.0; python_version>='3.14'", + "pandas>=3.0.0", + "pyarrow>=22.0.0", "polars>=1.39.0", "Jinja2>=3.1.0", "mypy>=0.900", diff --git a/tests/pyathena/pandas/test_async_cursor.py b/tests/pyathena/pandas/test_async_cursor.py index a0e9261d..01313627 100644 --- a/tests/pyathena/pandas/test_async_cursor.py +++ b/tests/pyathena/pandas/test_async_cursor.py @@ -17,11 +17,6 @@ from tests import ENV from tests.pyathena.conftest import connect -# pandas 3 infers its "str" dtype for strings, which represents NULL as NaN; pandas 2 uses -# object columns with None. -STRING_TYPE = pd.Series(["a"]).dtype.type -STRING_NULL = pd.Series(["a", None]).iloc[1] - class TestAsyncPandasCursor: def test_binary_null_vs_empty(self, async_pandas_cursor): @@ -592,7 +587,7 @@ def test_empty_and_null_string(self, async_pandas_cursor, parquet_engine): # NULL and empty characters are correctly converted when the UNLOAD option is enabled. np.testing.assert_equal( result_set.fetchall(), - [("", "a"), ("N/A", "a"), ("NULL", "a"), (STRING_NULL, "a")], + [("", "a"), ("N/A", "a"), ("NULL", "a"), (np.nan, "a")], ) else: np.testing.assert_equal( @@ -605,7 +600,7 @@ def test_empty_and_null_string(self, async_pandas_cursor, parquet_engine): # NULL and empty characters are correctly converted when the UNLOAD option is enabled. np.testing.assert_equal( result_set.fetchall(), - [("", "a"), ("N/A", "a"), ("NULL", "a"), (STRING_NULL, "a")], + [("", "a"), ("N/A", "a"), ("NULL", "a"), (np.nan, "a")], ) else: assert result_set.fetchall() == [ diff --git a/tests/pyathena/pandas/test_cursor.py b/tests/pyathena/pandas/test_cursor.py index 0e19ff18..943bc935 100644 --- a/tests/pyathena/pandas/test_cursor.py +++ b/tests/pyathena/pandas/test_cursor.py @@ -20,11 +20,6 @@ from tests import ENV from tests.pyathena.conftest import connect -# pandas 3 infers its "str" dtype for strings, which represents NULL as NaN; pandas 2 uses -# object columns with None. -STRING_TYPE = pd.Series(["a"]).dtype.type -STRING_NULL = pd.Series(["a", None]).iloc[1] - class TestPandasCursor: @pytest.mark.parametrize( @@ -646,17 +641,17 @@ def test_complex_as_pandas(self, pandas_cursor, chunksize): np.int64, np.float64, np.float64, - STRING_TYPE, - STRING_TYPE, + str, + str, np.datetime64, np.object_, np.datetime64, np.object_, - STRING_TYPE, + str, np.object_, - STRING_TYPE, + str, np.object_, - STRING_TYPE, + str, np.object_, ) rows = [ @@ -768,8 +763,8 @@ def test_complex_unload_as_pandas_pyarrow(self, pandas_cursor, parquet_engine): np.int64, np.float32, np.float64, - STRING_TYPE, - STRING_TYPE, + str, + str, np.datetime64, np.object_, np.object_, @@ -1200,7 +1195,7 @@ def test_null_vs_empty_string(self, pandas_cursor, parquet_engine): # NULL and empty characters are correctly converted when the UNLOAD option is enabled. np.testing.assert_equal( pandas_cursor.fetchall(), - [("", "a"), ("N/A", "a"), ("NULL", "a"), (STRING_NULL, "a")], + [("", "a"), ("N/A", "a"), ("NULL", "a"), (np.nan, "a")], ) else: np.testing.assert_equal( @@ -1212,7 +1207,7 @@ def test_null_vs_empty_string(self, pandas_cursor, parquet_engine): # NULL and empty characters are correctly converted when the UNLOAD option is enabled. np.testing.assert_equal( pandas_cursor.fetchall(), - [("", "a"), ("N/A", "a"), ("NULL", "a"), (STRING_NULL, "a")], + [("", "a"), ("N/A", "a"), ("NULL", "a"), (np.nan, "a")], ) else: assert pandas_cursor.fetchall() == [ diff --git a/uv.lock b/uv.lock index 79adb7af..aaf1f338 100644 --- a/uv.lock +++ b/uv.lock @@ -9,10 +9,10 @@ resolution-markers = [ "python_full_version == '3.13.*' and sys_platform == 'emscripten'", "python_full_version == '3.13.*' and sys_platform != 'emscripten' and sys_platform != 'win32'", "python_full_version == '3.12.*' and sys_platform == 'win32'", - "python_full_version < '3.12' and sys_platform == 'win32'", "python_full_version == '3.12.*' and sys_platform == 'emscripten'", - "python_full_version < '3.12' and sys_platform == 'emscripten'", "python_full_version == '3.12.*' and sys_platform != 'emscripten' and sys_platform != 'win32'", + "python_full_version < '3.12' and sys_platform == 'win32'", + "python_full_version < '3.12' and sys_platform == 'emscripten'", "python_full_version < '3.12' and sys_platform != 'emscripten' and sys_platform != 'win32'", ] @@ -957,8 +957,6 @@ dev = [ { name = "jinja2" }, { name = "mypy" }, { name = "myst-parser" }, - { name = "numpy", version = "2.4.6", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.12'" }, - { name = "numpy", version = "2.5.3", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version == '3.12.*' or python_full_version >= '3.14'" }, { name = "pandas" }, { name = "polars" }, { name = "pyarrow" }, @@ -981,11 +979,9 @@ requires-dist = [ { name = "boto3", specifier = ">=1.41.2" }, { name = "botocore", specifier = ">=1.41.2" }, { name = "fsspec" }, - { name = "pandas", marker = "python_full_version >= '3.13' and extra == 'pandas'", specifier = ">=2.3.0" }, - { name = "pandas", marker = "python_full_version < '3.13' and extra == 'pandas'", specifier = ">=1.3.0" }, + { name = "pandas", marker = "extra == 'pandas'", specifier = ">=3.0.0" }, { name = "polars", marker = "extra == 'polars'", specifier = ">=1.39.0" }, - { name = "pyarrow", marker = "python_full_version >= '3.14' and extra == 'arrow'", specifier = ">=22.0.0" }, - { name = "pyarrow", marker = "python_full_version < '3.14' and extra == 'arrow'", specifier = ">=10.0.0" }, + { name = "pyarrow", marker = "extra == 'arrow'", specifier = ">=22.0.0" }, { name = "python-dateutil" }, { name = "sqlalchemy", marker = "extra == 'sqlalchemy'", specifier = ">=2.0.0" }, { name = "sqlalchemy", extras = ["asyncio"], marker = "extra == 'aiosqlalchemy'", specifier = ">=2.0.0" }, @@ -1000,13 +996,9 @@ dev = [ { name = "jinja2", specifier = ">=3.1.0" }, { name = "mypy", specifier = ">=0.900" }, { name = "myst-parser" }, - { name = "numpy", marker = "python_full_version < '3.13'", specifier = ">=1.26.0" }, - { name = "numpy", marker = "python_full_version >= '3.14'", specifier = ">=2.3.0" }, - { name = "pandas", marker = "python_full_version < '3.13'", specifier = ">=1.3.0" }, - { name = "pandas", marker = "python_full_version >= '3.13'", specifier = ">=2.3.0" }, + { name = "pandas", specifier = ">=3.0.0" }, { name = "polars", specifier = ">=1.39.0" }, - { name = "pyarrow", marker = "python_full_version < '3.14'", specifier = ">=10.0.0" }, - { name = "pyarrow", marker = "python_full_version >= '3.14'", specifier = ">=22.0.0" }, + { name = "pyarrow", specifier = ">=22.0.0" }, { name = "pytest", specifier = ">=3.5" }, { name = "pytest-asyncio" }, { name = "pytest-cov" }, From 842f0f05b1ac8dc456f94b35d390e89a85defc70 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Tue, 29 Sep 2026 01:39:57 +0900 Subject: [PATCH 2/2] Drop the obsolete PyArrow 10 note from the Arrow timeout docs Every supported pyarrow version accepts the S3FileSystem timeouts. Co-Authored-By: Claude Opus 5.5 --- docs/arrow.md | 4 ---- 1 file changed, 4 deletions(-) diff --git a/docs/arrow.md b/docs/arrow.md index d70d82df..a99f1d24 100644 --- a/docs/arrow.md +++ b/docs/arrow.md @@ -293,10 +293,6 @@ cursor = connect( The timeout parameters accept float values in seconds and apply to all S3 operations performed by the cursor, including HeadObject and GetObject operations when retrieving query results. -```{note} -These timeout parameters require PyArrow >= 10.0.0, which added support for configuring S3FileSystem timeouts. -``` - (async-arrow-cursor)= ## AsyncArrowCursor