From a3fcbb3e01032bb7c50566333b9d33ac42b75c0c Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 12:35:21 +0900 Subject: [PATCH 1/6] Read string columns as text for the pandas PyArrow engine pandas' PyArrow engine lets pyarrow infer a type for each column and only then applies dtype=str, so a VARCHAR column whose values look like numbers came back changed ("007" as "7.0", "1e3" as "1000.0", the text "nan" as NaN), and without future.infer_string, astype(str) turned NULL into the string "nan". _read_csv_with_pyarrow() reproduced both. Read the columns with a string dtype as pyarrow strings, and give the columns with a NumPy str dtype object values that keep missing values missing, so these columns have the C engine's values. pandas' own PyArrow engine, used for options PyAthena does not read itself, keeps pandas' behavior (pandas-dev/pandas#57666), which the docs now state. Closes #1062 Co-Authored-By: Claude Opus 5.5 --- docs/pandas.md | 3 ++- pyathena/pandas/result_set.py | 32 +++++++++++++++++++++++- tests/pyathena/pandas/test_cursor.py | 26 +++++++++++++++++++ tests/pyathena/pandas/test_result_set.py | 24 +++++++++++++++++- 4 files changed, 82 insertions(+), 3 deletions(-) diff --git a/docs/pandas.md b/docs/pandas.md index bdb0f749..9610a0db 100644 --- a/docs/pandas.md +++ b/docs/pandas.md @@ -556,8 +556,9 @@ Common performance options: With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, `json`, `time with time zone`, and `timestamp with time zone` with the default converter), and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. Otherwise, it falls back to the C engine. -With the PyArrow engine, when `keep_default_na` and `na_values` are the defaults and the only pandas.read_csv() options passed to `execute()` are `dtype` as a mapping and `parse_dates` as a list, PyAthena reads the file with `pyarrow.csv`, allowing newlines in quoted values, and converts it to a DataFrame as `pandas.read_csv(engine="pyarrow")` does. +With the PyArrow engine, when `keep_default_na` and `na_values` are the defaults and the only pandas.read_csv() options passed to `execute()` are `dtype` as a mapping and `parse_dates` as a list, PyAthena reads the file with `pyarrow.csv`, allowing newlines in quoted values, and converts it to a DataFrame as `pandas.read_csv(engine="pyarrow")` does, except that columns with a string dtype keep their text and their missing values, as with the C engine. Otherwise, pandas reads the file, and its PyArrow engine raises an error or returns wrong values when a quoted value containing a newline crosses one of pyarrow's read blocks. +It also changes strings that look like numbers in a column with a string dtype, such as `"007"` to `"7.0"`, and without `future.infer_string`, turns NULL into the string `"nan"`. Apart from PandasCursor's own options such as `engine` and `chunksize`, an option passed here replaces the value PyAthena sets for the same pandas.read_csv() argument. For example, `dtype` replaces the whole column type mapping, and `parse_dates` replaces the list of date, time, and timestamp columns. diff --git a/pyathena/pandas/result_set.py b/pyathena/pandas/result_set.py index 3b87348a..665f8210 100644 --- a/pyathena/pandas/result_set.py +++ b/pyathena/pandas/result_set.py @@ -56,6 +56,10 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] ``parse_dates`` columns are parsed with ``pandas.to_datetime()``, and without ``future.infer_string``, strings become objects. + Unlike pandas' PyArrow engine, columns with a string dtype are read as + text, and a NumPy str dtype keeps missing values missing, so these columns + have the values of pandas' C engine. + Args: source: The result file. read_csv_kwargs: The pandas.read_csv() options built by @@ -71,6 +75,19 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] header = read_csv_kwargs["header"] names = read_csv_kwargs["names"] null_values = list(read_csv_kwargs["na_values"]) + string_columns = { + column + for column, value in read_csv_kwargs["dtype"].items() + if isinstance(column_dtype := pd.api.types.pandas_dtype(value), pd.StringDtype) + or column_dtype.kind == "U" + } + if header is None: + # pyarrow names the columns of a header-less file f0, f1, ... + column_types = { + f"f{index}": pa.string() for index, name in enumerate(names) if name in string_columns + } + else: + column_types = {column: pa.string() for column in string_columns} table = pyarrow_csv.read_csv( source, read_options=pyarrow_csv.ReadOptions(autogenerate_column_names=header is None), @@ -80,7 +97,11 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] newlines_in_values=True, ), convert_options=pyarrow_csv.ConvertOptions( - null_values=null_values, strings_can_be_null="" in null_values + # Columns with a string dtype keep their text instead of the type pyarrow + # infers, so "007" stays "007" as with pandas' C engine. + column_types=column_types, + null_values=null_values, + strings_can_be_null="" in null_values, ), ) schema = table.schema @@ -109,7 +130,16 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] for column, value in dtype.items() if column in df.columns } + # astype() with a NumPy str dtype, which str means without future.infer_string, + # turns missing values into "nan"; these columns become objects that keep them + # missing, as with pandas' C engine. + numpy_string_columns = {column for column, value in dtype.items() if value.kind == "U"} + dtype = {column: value for column, value in dtype.items() if column not in numpy_string_columns} df = df.astype(dtype) + for index, column in enumerate(df.columns): + if column in numpy_string_columns: + values = df.iloc[:, index] + df.isetitem(index, values.astype(object).where(values.notna(), float("nan"))) if not pd.get_option("future.infer_string"): # Without the string dtype, pandas returns strings, and the string # categories of categorical columns, as objects. diff --git a/tests/pyathena/pandas/test_cursor.py b/tests/pyathena/pandas/test_cursor.py index 7ca3052a..b10d03d0 100644 --- a/tests/pyathena/pandas/test_cursor.py +++ b/tests/pyathena/pandas/test_cursor.py @@ -303,6 +303,32 @@ def test_csv_storage_options(self, pandas_cursor, query, expected, binary, with_ # pandas opens and closes the file itself. assert pandas_cursor.result_set._csv_stream is None + @pytest.mark.parametrize("infer_string", [True, False]) + def test_pyarrow_engine_string_values(self, pandas_cursor, infer_string): + # The PyArrow engine keeps the text of strings that look like numbers, and + # a NULL string missing, as the C engine does. + query = """ + SELECT * FROM (VALUES + (1, '1', 'abcdefghijklmnopqrstuvwxyz'), + (2, CAST(NULL AS VARCHAR), 'abcdefghijklmnopqrstuvwxyz'), + (3, 'nan', 'abcdefghijklmnopqrstuvwxyz'), + (4, '007', 'abcdefghijklmnopqrstuvwxyz'), + (5, '1e3', 'abcdefghijklmnopqrstuvwxyz') + ) AS t(id, v, padding) ORDER BY id + """ + with pd.option_context("future.infer_string", infer_string): + with patch( + "pyathena.pandas.result_set._read_csv_with_pyarrow", wraps=_read_csv_with_pyarrow + ) as read_csv_with_pyarrow: + pandas_cursor.execute(query, engine="pyarrow") + actual = pandas_cursor.as_pandas()["v"] + read_csv_with_pyarrow.assert_called_once() + pandas_cursor.execute(query, engine="c") + expected = pandas_cursor.as_pandas()["v"] + pd.testing.assert_series_equal(actual, expected, check_exact=True) + assert actual.iloc[[0, 2, 3, 4]].tolist() == ["1", "nan", "007", "1e3"] + assert pd.isna(actual.iloc[1]) + def test_pyarrow_engine_multiline_values_across_blocks(self, pandas_cursor): # The 2.4 MB result spans several 1 MiB pyarrow read blocks, and its # values contain a newline and quotes. diff --git a/tests/pyathena/pandas/test_result_set.py b/tests/pyathena/pandas/test_result_set.py index ef6022a5..42391018 100644 --- a/tests/pyathena/pandas/test_result_set.py +++ b/tests/pyathena/pandas/test_result_set.py @@ -224,6 +224,10 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): '"x","x","d"\n"1","2","2024-01-01"\n,,\n', _pyarrow_read_csv_kwargs({"x": "integer", "d": "date"}), ), + ( + '"v","n"\n"1","1"\n,\n"nan","3"\n"007","4"\n"1e3","5"\n', + _pyarrow_read_csv_kwargs({"v": "varchar", "n": "integer"}), + ), ( '"v"\n"plain"\n"2024-01-01"\n\n', _pyarrow_read_csv_kwargs({"v": "varchar"}, parse_dates=["v"]), @@ -244,6 +248,7 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): "dtype_of_date_column", "dtype_none", "duplicate_names", + "numeric_looking_strings", "unparsed_dates", "tab_separated_extra_fields", "tab_separated", @@ -251,12 +256,29 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): ) def test_read_csv_with_pyarrow_matches_pandas(data, read_csv_kwargs, infer_string): # Without values that cross a read block, the result is the one of - # pandas.read_csv(engine="pyarrow"). + # pandas.read_csv(engine="pyarrow"), except that the columns with a string + # dtype have the values of pandas' C engine. Where the C engine parses a + # parse_dates column despite its string dtype, the PyArrow engine's + # applying the dtype again is kept. with pd.option_context("future.infer_string", infer_string): expected = pd.read_csv( io.BytesIO(data.encode()), **{**read_csv_kwargs, "dtype": dict(read_csv_kwargs["dtype"])}, ) + c_engine = pd.read_csv( + io.BytesIO(data.encode()), + **{**read_csv_kwargs, "engine": "c", "dtype": dict(read_csv_kwargs["dtype"])}, + ) + string_columns = { + column + for column, value in read_csv_kwargs["dtype"].items() + if isinstance(dtype := pd.api.types.pandas_dtype(value), pd.StringDtype) + or dtype.kind == "U" + } + for index, column in enumerate(expected.columns): + if column in string_columns and c_engine[column].dtype.kind != "M": + # The C engine makes the extra fields of a header-less file the index. + expected.isetitem(index, c_engine[column].array) actual = _read_csv_with_pyarrow( io.BytesIO(data.encode()), {**read_csv_kwargs, "dtype": dict(read_csv_kwargs["dtype"])}, From 51129926ca9a62f03120fe8c3cc95cb949b6ab9e Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 13:06:11 +0900 Subject: [PATCH 2/6] Pass only column names to pyarrow column_types A dtype mapping can have keys that are not column names, such as positions, which pandas ignores; pyarrow's column_types raised TypeError for them. Co-Authored-By: Claude Opus 5.5 --- pyathena/pandas/result_set.py | 3 ++- tests/pyathena/pandas/test_result_set.py | 8 ++++++++ 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/pyathena/pandas/result_set.py b/pyathena/pandas/result_set.py index 665f8210..b7748a1c 100644 --- a/pyathena/pandas/result_set.py +++ b/pyathena/pandas/result_set.py @@ -87,7 +87,8 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] f"f{index}": pa.string() for index, name in enumerate(names) if name in string_columns } else: - column_types = {column: pa.string() for column in string_columns} + # pandas ignores dtype keys that are not column names, such as positions. + column_types = {column: pa.string() for column in string_columns if isinstance(column, str)} table = pyarrow_csv.read_csv( source, read_options=pyarrow_csv.ReadOptions(autogenerate_column_names=header is None), diff --git a/tests/pyathena/pandas/test_result_set.py b/tests/pyathena/pandas/test_result_set.py index 42391018..23a9065f 100644 --- a/tests/pyathena/pandas/test_result_set.py +++ b/tests/pyathena/pandas/test_result_set.py @@ -228,6 +228,13 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): '"v","n"\n"1","1"\n,\n"nan","3"\n"007","4"\n"1e3","5"\n', _pyarrow_read_csv_kwargs({"v": "varchar", "n": "integer"}), ), + ( + '"v","n"\n"007","1"\n', + _pyarrow_read_csv_kwargs( + {"v": "varchar", "n": "integer"}, + dtype={**_pyarrow_read_csv_kwargs({"v": "varchar"})["dtype"], 0: str}, + ), + ), ( '"v"\n"plain"\n"2024-01-01"\n\n', _pyarrow_read_csv_kwargs({"v": "varchar"}, parse_dates=["v"]), @@ -249,6 +256,7 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): "dtype_none", "duplicate_names", "numeric_looking_strings", + "dtype_position_key", "unparsed_dates", "tab_separated_extra_fields", "tab_separated", From 487d88ef71399f48639da61af83eca8a9b806822 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 13:06:31 +0900 Subject: [PATCH 3/6] Name the conditions of the PyArrow engine's string changes in the docs Co-Authored-By: Claude Opus 5.5 --- docs/pandas.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/pandas.md b/docs/pandas.md index 9610a0db..ada2c71a 100644 --- a/docs/pandas.md +++ b/docs/pandas.md @@ -558,7 +558,7 @@ With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow Otherwise, it falls back to the C engine. With the PyArrow engine, when `keep_default_na` and `na_values` are the defaults and the only pandas.read_csv() options passed to `execute()` are `dtype` as a mapping and `parse_dates` as a list, PyAthena reads the file with `pyarrow.csv`, allowing newlines in quoted values, and converts it to a DataFrame as `pandas.read_csv(engine="pyarrow")` does, except that columns with a string dtype keep their text and their missing values, as with the C engine. Otherwise, pandas reads the file, and its PyArrow engine raises an error or returns wrong values when a quoted value containing a newline crosses one of pyarrow's read blocks. -It also changes strings that look like numbers in a column with a string dtype, such as `"007"` to `"7.0"`, and without `future.infer_string`, turns NULL into the string `"nan"`. +It also changes the values of a column with a string dtype whose values all look like numbers, such as `"007"` to `"7.0"`, and without `future.infer_string`, turns NULL in a column with a string dtype into the string `"nan"`. Apart from PandasCursor's own options such as `engine` and `chunksize`, an option passed here replaces the value PyAthena sets for the same pandas.read_csv() argument. For example, `dtype` replaces the whole column type mapping, and `parse_dates` replaces the list of date, time, and timestamp columns. From 7bad8fe55571c5f22ea610c666e1db2b6a351f59 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 13:29:32 +0900 Subject: [PATCH 4/6] Keep requested string dtypes and header-less field positions in the PyArrow reader The independent review found that the string-column change: - applied the NumPy str handling to Arrow string dtypes, whose kind is also "U", returning objects instead of the requested dtype; - typed the first fields of a header-less file while pandas gives the names to the last ones, changing other columns' types when a row has extra fields; - left parse_dates columns with a NumPy str dtype as datetimes, while the dtype used to apply again after parsing; - failed on invalid dtype entries for columns not in the result, which pandas ignores. Apply the dtype mapping through _astype_keeping_missing_strings() before and after parsing dates, limit its NumPy str handling to NumPy dtypes, count the fields of a header-less file's first line to type the named ones, and skip dtype values that do not resolve when choosing string columns. Co-Authored-By: Claude Opus 5.5 --- docs/pandas.md | 4 +- pyathena/pandas/result_set.py | 78 +++++++++++++++++------- tests/pyathena/pandas/test_result_set.py | 60 ++++++++++++++++-- 3 files changed, 115 insertions(+), 27 deletions(-) diff --git a/docs/pandas.md b/docs/pandas.md index ada2c71a..334568ab 100644 --- a/docs/pandas.md +++ b/docs/pandas.md @@ -556,9 +556,9 @@ Common performance options: With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, `json`, `time with time zone`, and `timestamp with time zone` with the default converter), and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. Otherwise, it falls back to the C engine. -With the PyArrow engine, when `keep_default_na` and `na_values` are the defaults and the only pandas.read_csv() options passed to `execute()` are `dtype` as a mapping and `parse_dates` as a list, PyAthena reads the file with `pyarrow.csv`, allowing newlines in quoted values, and converts it to a DataFrame as `pandas.read_csv(engine="pyarrow")` does, except that columns with a string dtype keep their text and their missing values, as with the C engine. +With the PyArrow engine, when `keep_default_na` and `na_values` are the defaults and the only pandas.read_csv() options passed to `execute()` are `dtype` as a mapping and `parse_dates` as a list, PyAthena reads the file with `pyarrow.csv`, allowing newlines in quoted values, and converts it to a DataFrame as `pandas.read_csv(engine="pyarrow")` does, except that columns with a string dtype are read as text, and without `future.infer_string`, a column with dtype `str` keeps NULL as a missing value. Otherwise, pandas reads the file, and its PyArrow engine raises an error or returns wrong values when a quoted value containing a newline crosses one of pyarrow's read blocks. -It also changes the values of a column with a string dtype whose values all look like numbers, such as `"007"` to `"7.0"`, and without `future.infer_string`, turns NULL in a column with a string dtype into the string `"nan"`. +It also changes the values of a column with a string dtype whose values all look like numbers, such as `"007"` to `"7.0"`, and without `future.infer_string`, turns NULL in a column with dtype `str` into the string `"nan"`. Apart from PandasCursor's own options such as `engine` and `chunksize`, an option passed here replaces the value PyAthena sets for the same pandas.read_csv() argument. For example, `dtype` replaces the whole column type mapping, and `parse_dates` replaces the list of date, time, and timestamp columns. diff --git a/pyathena/pandas/result_set.py b/pyathena/pandas/result_set.py index b7748a1c..2e004d8c 100644 --- a/pyathena/pandas/result_set.py +++ b/pyathena/pandas/result_set.py @@ -43,7 +43,40 @@ def _no_trunc_date(df: DataFrame) -> DataFrame: return df -def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any]) -> DataFrame: +def _astype_keeping_missing_strings(df: DataFrame, dtype: dict[Any, Any]) -> DataFrame: + """Apply a dtype mapping, keeping missing values missing in NumPy str columns. + + ``astype()`` with a NumPy str dtype, which ``str`` means without + ``future.infer_string``, turns missing values into strings such as ``"nan"``. + Columns mapped to a NumPy str dtype become object columns of strings with + missing values kept, as pandas' C engine returns them. + + Args: + df: The DataFrame to convert. + dtype: The normalized dtypes, keyed by column name. + + Returns: + The converted DataFrame. + """ + import pandas as pd + + numpy_string_columns = { + column + for column, value in dtype.items() + if not isinstance(value, pd.api.extensions.ExtensionDtype) and value.kind == "U" + } + df = df.astype( + {column: value for column, value in dtype.items() if column not in numpy_string_columns} + ) + for index, column in enumerate(df.columns): + if column in numpy_string_columns: + values = df.iloc[:, index] + strings = values.astype(str).astype(object) + df.isetitem(index, strings.where(values.notna(), float("nan"))) + return df + + +def _read_csv_with_pyarrow(source: IOBase, read_csv_kwargs: dict[str, Any]) -> DataFrame: """Read a CSV result with pyarrow as ``pandas.read_csv(engine="pyarrow")`` does. pandas' PyArrow engine does not set ``newlines_in_values``, so pyarrow @@ -75,16 +108,24 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] header = read_csv_kwargs["header"] names = read_csv_kwargs["names"] null_values = list(read_csv_kwargs["na_values"]) - string_columns = { - column - for column, value in read_csv_kwargs["dtype"].items() - if isinstance(column_dtype := pd.api.types.pandas_dtype(value), pd.StringDtype) - or column_dtype.kind == "U" - } + string_columns = set() + for column, value in read_csv_kwargs["dtype"].items(): + try: + column_dtype = pd.api.types.pandas_dtype(value) + except TypeError: + # pandas validates only the entries of columns in the result. + continue + if isinstance(column_dtype, pd.StringDtype) or column_dtype.kind == "U": + string_columns.add(column) if header is None: - # pyarrow names the columns of a header-less file f0, f1, ... + # pyarrow names the fields of a header-less file f0, f1, ..., and pandas + # gives the names to the last fields, so count the fields of the first line. + offset = len(source.readline().split(read_csv_kwargs["sep"].encode())) - len(names) + source.seek(0) column_types = { - f"f{index}": pa.string() for index, name in enumerate(names) if name in string_columns + f"f{offset + index}": pa.string() + for index, name in enumerate(names) + if name in string_columns } else: # pandas ignores dtype keys that are not column names, such as positions. @@ -131,16 +172,7 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] for column, value in dtype.items() if column in df.columns } - # astype() with a NumPy str dtype, which str means without future.infer_string, - # turns missing values into "nan"; these columns become objects that keep them - # missing, as with pandas' C engine. - numpy_string_columns = {column for column, value in dtype.items() if value.kind == "U"} - dtype = {column: value for column, value in dtype.items() if column not in numpy_string_columns} - df = df.astype(dtype) - for index, column in enumerate(df.columns): - if column in numpy_string_columns: - values = df.iloc[:, index] - df.isetitem(index, values.astype(object).where(values.notna(), float("nan"))) + df = _astype_keeping_missing_strings(df, dtype) if not pd.get_option("future.infer_string"): # Without the string dtype, pandas returns strings, and the string # categories of categorical columns, as objects. @@ -168,7 +200,7 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] # pandas keeps the column as strings if it cannot parse it. df[column] = values.to_numpy(dtype=object, na_value=float("nan")) # pandas applies the dtype mapping again after parsing dates. - df = df.astype(dtype) + df = _astype_keeping_missing_strings(df, dtype) return df @@ -827,7 +859,11 @@ def _read_csv(self) -> TextFileReader | DataFrame: source = self._csv_stream = stack.enter_context( self._fs.open(self.output_location, mode="rb") ) - if csv_engine == "pyarrow" and self._reads_csv_with_pyarrow(): + if ( + csv_engine == "pyarrow" + and isinstance(source, IOBase) + and self._reads_csv_with_pyarrow() + ): result = _read_csv_with_pyarrow(source, read_csv_kwargs) else: result = pd.read_csv(source, **read_csv_kwargs) diff --git a/tests/pyathena/pandas/test_result_set.py b/tests/pyathena/pandas/test_result_set.py index 23a9065f..e3a16557 100644 --- a/tests/pyathena/pandas/test_result_set.py +++ b/tests/pyathena/pandas/test_result_set.py @@ -10,6 +10,7 @@ from unittest.mock import MagicMock, PropertyMock, patch import pandas as pd +import pyarrow as pa import pytest from pandas.testing import assert_frame_equal @@ -153,6 +154,15 @@ def test_read_parquet_filesystem(self, execute_kwargs, path, filesystem_kwargs): } +def _is_string_dtype(value): + """Return whether a dtype mapping value is a string dtype, ignoring invalid values.""" + try: + dtype = pd.api.types.pandas_dtype(value) + except TypeError: + return False + return isinstance(dtype, pd.StringDtype) or dtype.kind == "U" + + def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): """Build the pandas.read_csv() options with AthenaPandasResultSet._get_csv_read_options(). @@ -228,6 +238,21 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): '"v","n"\n"1","1"\n,\n"nan","3"\n"007","4"\n"1e3","5"\n', _pyarrow_read_csv_kwargs({"v": "varchar", "n": "integer"}), ), + ( + '"v","w","x"\n"007","a","1"\n,,"2"\n', + _pyarrow_read_csv_kwargs( + {"x": "integer"}, + dtype={ + "v": pd.ArrowDtype(pa.string()), + "w": pd.ArrowDtype(pa.large_string()), + "x": pd.Int64Dtype(), + }, + ), + ), + ( + "001\t2\t003\n004\t5\t\n", + _pyarrow_read_csv_kwargs({"v": "varchar"}, True), + ), ( '"v","n"\n"007","1"\n', _pyarrow_read_csv_kwargs( @@ -256,6 +281,8 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): "dtype_none", "duplicate_names", "numeric_looking_strings", + "arrow_string_dtypes", + "tab_separated_numeric_fields", "dtype_position_key", "unparsed_dates", "tab_separated_extra_fields", @@ -278,10 +305,7 @@ def test_read_csv_with_pyarrow_matches_pandas(data, read_csv_kwargs, infer_strin **{**read_csv_kwargs, "engine": "c", "dtype": dict(read_csv_kwargs["dtype"])}, ) string_columns = { - column - for column, value in read_csv_kwargs["dtype"].items() - if isinstance(dtype := pd.api.types.pandas_dtype(value), pd.StringDtype) - or dtype.kind == "U" + column for column, value in read_csv_kwargs["dtype"].items() if _is_string_dtype(value) } for index, column in enumerate(expected.columns): if column in string_columns and c_engine[column].dtype.kind != "M": @@ -294,6 +318,34 @@ def test_read_csv_with_pyarrow_matches_pandas(data, read_csv_kwargs, infer_strin assert_frame_equal(actual, expected, check_exact=True) +@pytest.mark.parametrize("infer_string", [True, False]) +def test_read_csv_with_pyarrow_string_dtype_after_dates(infer_string): + # A string dtype applies again after parse_dates, as with pandas' PyArrow + # engine, and keeps NULL missing. + with pd.option_context("future.infer_string", infer_string): + df = _read_csv_with_pyarrow( + io.BytesIO(b'"v"\n"2024-01-01"\n\n'), + _pyarrow_read_csv_kwargs({"v": "varchar"}, parse_dates=["v"]), + ) + assert df["v"].tolist()[0] == "2024-01-01" + assert pd.isna(df["v"].tolist()[1]) + + +def test_read_csv_with_pyarrow_ignores_unused_dtype_entries(): + # As with pandas' PyArrow engine, a dtype entry for a column that is not in + # the result is not validated. + read_csv_kwargs = _pyarrow_read_csv_kwargs( + {"v": "varchar"}, dtype={"v": str, "unused": "not-a-dtype"} + ) + data = b'"v"\n"007"\n' + expected = pd.read_csv(io.BytesIO(data), **{**read_csv_kwargs, "dtype": {"v": str}}) + actual = _read_csv_with_pyarrow( + io.BytesIO(data), {**read_csv_kwargs, "dtype": dict(read_csv_kwargs["dtype"])} + ) + assert actual.columns.tolist() == expected.columns.tolist() == ["v"] + assert actual["v"].tolist() == ["007"] + + def test_read_csv_with_pyarrow_multiline_values_across_blocks(): # The 2.4 MB file spans several 1 MiB pyarrow read blocks, and its values # contain newlines and quotes. From d09bccd4e9d655791a041ad9cbcd9e13fb108573 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 13:42:01 +0900 Subject: [PATCH 5/6] Count header-less fields with CSV quoting and ignore any unresolved dtype The follow-up review found that splitting the first line of a header-less file on the delimiter miscounted a quoted delimiter, which typed the wrong field as text, and that dtype values pandas_dtype() rejects with NotImplementedError still failed for absent columns. Read the first record with the csv module, which follows the same quoting as the reader, and skip dtype values that raise TypeError, ValueError, or NotImplementedError when choosing string columns. Co-Authored-By: Claude Opus 5.5 --- pyathena/pandas/result_set.py | 9 ++++++--- tests/pyathena/pandas/test_result_set.py | 13 ++++++++++++- 2 files changed, 18 insertions(+), 4 deletions(-) diff --git a/pyathena/pandas/result_set.py b/pyathena/pandas/result_set.py index 2e004d8c..d2053658 100644 --- a/pyathena/pandas/result_set.py +++ b/pyathena/pandas/result_set.py @@ -112,16 +112,19 @@ def _read_csv_with_pyarrow(source: IOBase, read_csv_kwargs: dict[str, Any]) -> D for column, value in read_csv_kwargs["dtype"].items(): try: column_dtype = pd.api.types.pandas_dtype(value) - except TypeError: + except (TypeError, ValueError, NotImplementedError): # pandas validates only the entries of columns in the result. continue if isinstance(column_dtype, pd.StringDtype) or column_dtype.kind == "U": string_columns.add(column) if header is None: # pyarrow names the fields of a header-less file f0, f1, ..., and pandas - # gives the names to the last fields, so count the fields of the first line. - offset = len(source.readline().split(read_csv_kwargs["sep"].encode())) - len(names) + # gives the names to the last fields, so count the fields of the first + # record, which has one field even if it is empty. + lines = (line.decode("utf-8") for line in iter(source.readline, b"")) + first_record = next(csv.reader(lines, delimiter=read_csv_kwargs["sep"]), []) source.seek(0) + offset = max(len(first_record), 1) - len(names) column_types = { f"f{offset + index}": pa.string() for index, name in enumerate(names) diff --git a/tests/pyathena/pandas/test_result_set.py b/tests/pyathena/pandas/test_result_set.py index e3a16557..e78aa172 100644 --- a/tests/pyathena/pandas/test_result_set.py +++ b/tests/pyathena/pandas/test_result_set.py @@ -253,6 +253,14 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): "001\t2\t003\n004\t5\t\n", _pyarrow_read_csv_kwargs({"v": "varchar"}, True), ), + ( + '007\t"a\tb"\n010\tc\n', + _pyarrow_read_csv_kwargs({"v": "varchar", "x": "varchar"}, True), + ), + ( + '007\t"a\nb"\n010\tc\n', + _pyarrow_read_csv_kwargs({"v": "varchar", "x": "varchar"}, True), + ), ( '"v","n"\n"007","1"\n', _pyarrow_read_csv_kwargs( @@ -283,6 +291,8 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): "numeric_looking_strings", "arrow_string_dtypes", "tab_separated_numeric_fields", + "tab_separated_quoted_tab", + "tab_separated_quoted_newline", "dtype_position_key", "unparsed_dates", "tab_separated_extra_fields", @@ -335,7 +345,8 @@ def test_read_csv_with_pyarrow_ignores_unused_dtype_entries(): # As with pandas' PyArrow engine, a dtype entry for a column that is not in # the result is not validated. read_csv_kwargs = _pyarrow_read_csv_kwargs( - {"v": "varchar"}, dtype={"v": str, "unused": "not-a-dtype"} + {"v": "varchar"}, + dtype={"v": str, "unused": "not-a-dtype", "unsupported": "decimal128(10, 2)[pyarrow]"}, ) data = b'"v"\n"007"\n' expected = pd.read_csv(io.BytesIO(data), **{**read_csv_kwargs, "dtype": {"v": str}}) From 6bd7711b3b5b51e3035a7755db53dbc072969291 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 13:59:54 +0900 Subject: [PATCH 6/6] Keep the inferred types of header-less fields in the PyArrow reader Typing the named fields of a header-less file needs their positions before reading, and each way of counting the first record's fields (delimiter split, the csv module) disagreed with pyarrow on some input: quoted delimiters, long fields, bare CR line ends, a byte order mark. Header-less files are the tab-separated results of DDL statements, so leave their fields with the types pyarrow infers, as before, and keep only the missing-value handling of NumPy str columns for them. Also convert NumPy str columns once, after the dates, instead of through a separate helper. Co-Authored-By: Claude Opus 5.5 --- docs/pandas.md | 2 +- pyathena/pandas/result_set.py | 94 ++++++++---------------- tests/pyathena/pandas/test_result_set.py | 26 +++---- 3 files changed, 44 insertions(+), 78 deletions(-) diff --git a/docs/pandas.md b/docs/pandas.md index 334568ab..1ceb6d7e 100644 --- a/docs/pandas.md +++ b/docs/pandas.md @@ -556,7 +556,7 @@ Common performance options: With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, `json`, `time with time zone`, and `timestamp with time zone` with the default converter), and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. Otherwise, it falls back to the C engine. -With the PyArrow engine, when `keep_default_na` and `na_values` are the defaults and the only pandas.read_csv() options passed to `execute()` are `dtype` as a mapping and `parse_dates` as a list, PyAthena reads the file with `pyarrow.csv`, allowing newlines in quoted values, and converts it to a DataFrame as `pandas.read_csv(engine="pyarrow")` does, except that columns with a string dtype are read as text, and without `future.infer_string`, a column with dtype `str` keeps NULL as a missing value. +With the PyArrow engine, when `keep_default_na` and `na_values` are the defaults and the only pandas.read_csv() options passed to `execute()` are `dtype` as a mapping and `parse_dates` as a list, PyAthena reads the file with `pyarrow.csv`, allowing newlines in quoted values, and converts it to a DataFrame as `pandas.read_csv(engine="pyarrow")` does, except that columns with a string dtype are read as text, other than in the tab-separated results of DDL statements, and without `future.infer_string`, a column with dtype `str` keeps NULL as a missing value. Otherwise, pandas reads the file, and its PyArrow engine raises an error or returns wrong values when a quoted value containing a newline crosses one of pyarrow's read blocks. It also changes the values of a column with a string dtype whose values all look like numbers, such as `"007"` to `"7.0"`, and without `future.infer_string`, turns NULL in a column with dtype `str` into the string `"nan"`. diff --git a/pyathena/pandas/result_set.py b/pyathena/pandas/result_set.py index d2053658..c03db6d6 100644 --- a/pyathena/pandas/result_set.py +++ b/pyathena/pandas/result_set.py @@ -43,40 +43,7 @@ def _no_trunc_date(df: DataFrame) -> DataFrame: return df -def _astype_keeping_missing_strings(df: DataFrame, dtype: dict[Any, Any]) -> DataFrame: - """Apply a dtype mapping, keeping missing values missing in NumPy str columns. - - ``astype()`` with a NumPy str dtype, which ``str`` means without - ``future.infer_string``, turns missing values into strings such as ``"nan"``. - Columns mapped to a NumPy str dtype become object columns of strings with - missing values kept, as pandas' C engine returns them. - - Args: - df: The DataFrame to convert. - dtype: The normalized dtypes, keyed by column name. - - Returns: - The converted DataFrame. - """ - import pandas as pd - - numpy_string_columns = { - column - for column, value in dtype.items() - if not isinstance(value, pd.api.extensions.ExtensionDtype) and value.kind == "U" - } - df = df.astype( - {column: value for column, value in dtype.items() if column not in numpy_string_columns} - ) - for index, column in enumerate(df.columns): - if column in numpy_string_columns: - values = df.iloc[:, index] - strings = values.astype(str).astype(object) - df.isetitem(index, strings.where(values.notna(), float("nan"))) - return df - - -def _read_csv_with_pyarrow(source: IOBase, read_csv_kwargs: dict[str, Any]) -> DataFrame: +def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any]) -> DataFrame: """Read a CSV result with pyarrow as ``pandas.read_csv(engine="pyarrow")`` does. pandas' PyArrow engine does not set ``newlines_in_values``, so pyarrow @@ -89,9 +56,9 @@ def _read_csv_with_pyarrow(source: IOBase, read_csv_kwargs: dict[str, Any]) -> D ``parse_dates`` columns are parsed with ``pandas.to_datetime()``, and without ``future.infer_string``, strings become objects. - Unlike pandas' PyArrow engine, columns with a string dtype are read as - text, and a NumPy str dtype keeps missing values missing, so these columns - have the values of pandas' C engine. + Unlike pandas' PyArrow engine, columns with a string dtype are read as text, + except in a header-less file, and a NumPy str dtype keeps missing values + missing, as with pandas' C engine. Args: source: The result file. @@ -117,22 +84,16 @@ def _read_csv_with_pyarrow(source: IOBase, read_csv_kwargs: dict[str, Any]) -> D continue if isinstance(column_dtype, pd.StringDtype) or column_dtype.kind == "U": string_columns.add(column) - if header is None: - # pyarrow names the fields of a header-less file f0, f1, ..., and pandas - # gives the names to the last fields, so count the fields of the first - # record, which has one field even if it is empty. - lines = (line.decode("utf-8") for line in iter(source.readline, b"")) - first_record = next(csv.reader(lines, delimiter=read_csv_kwargs["sep"]), []) - source.seek(0) - offset = max(len(first_record), 1) - len(names) - column_types = { - f"f{offset + index}": pa.string() - for index, name in enumerate(names) - if name in string_columns - } - else: - # pandas ignores dtype keys that are not column names, such as positions. - column_types = {column: pa.string() for column in string_columns if isinstance(column, str)} + # Columns with a string dtype keep their text instead of the type pyarrow infers, + # so "007" stays "007" as with pandas' C engine. The fields of a header-less file + # get their names only after reading, as pandas gives the names to the last + # fields, so they keep the inferred types. pandas ignores dtype keys that are not + # column names, such as positions. + column_types = ( + {} + if header is None + else {column: pa.string() for column in string_columns if isinstance(column, str)} + ) table = pyarrow_csv.read_csv( source, read_options=pyarrow_csv.ReadOptions(autogenerate_column_names=header is None), @@ -142,8 +103,6 @@ def _read_csv_with_pyarrow(source: IOBase, read_csv_kwargs: dict[str, Any]) -> D newlines_in_values=True, ), convert_options=pyarrow_csv.ConvertOptions( - # Columns with a string dtype keep their text instead of the type pyarrow - # infers, so "007" stays "007" as with pandas' C engine. column_types=column_types, null_values=null_values, strings_can_be_null="" in null_values, @@ -175,7 +134,16 @@ def _read_csv_with_pyarrow(source: IOBase, read_csv_kwargs: dict[str, Any]) -> D for column, value in dtype.items() if column in df.columns } - df = _astype_keeping_missing_strings(df, dtype) + # astype() with a NumPy str dtype, which str means without future.infer_string, + # turns missing values into strings such as "nan", so these columns are + # converted after the dates instead. + numpy_string_columns = { + column + for column, value in dtype.items() + if not isinstance(value, pd.api.extensions.ExtensionDtype) and value.kind == "U" + } + dtype = {column: value for column, value in dtype.items() if column not in numpy_string_columns} + df = df.astype(dtype) if not pd.get_option("future.infer_string"): # Without the string dtype, pandas returns strings, and the string # categories of categorical columns, as objects. @@ -203,7 +171,13 @@ def _read_csv_with_pyarrow(source: IOBase, read_csv_kwargs: dict[str, Any]) -> D # pandas keeps the column as strings if it cannot parse it. df[column] = values.to_numpy(dtype=object, na_value=float("nan")) # pandas applies the dtype mapping again after parsing dates. - df = _astype_keeping_missing_strings(df, dtype) + df = df.astype(dtype) + for index, column in enumerate(df.columns): + if column in numpy_string_columns: + # Object strings with missing values kept, as pandas' C engine returns. + values = df.iloc[:, index] + strings = values.astype(str).astype(object) + df.isetitem(index, strings.where(values.notna(), float("nan"))) return df @@ -862,11 +836,7 @@ def _read_csv(self) -> TextFileReader | DataFrame: source = self._csv_stream = stack.enter_context( self._fs.open(self.output_location, mode="rb") ) - if ( - csv_engine == "pyarrow" - and isinstance(source, IOBase) - and self._reads_csv_with_pyarrow() - ): + if csv_engine == "pyarrow" and self._reads_csv_with_pyarrow(): result = _read_csv_with_pyarrow(source, read_csv_kwargs) else: result = pd.read_csv(source, **read_csv_kwargs) diff --git a/tests/pyathena/pandas/test_result_set.py b/tests/pyathena/pandas/test_result_set.py index e78aa172..0ed38d81 100644 --- a/tests/pyathena/pandas/test_result_set.py +++ b/tests/pyathena/pandas/test_result_set.py @@ -253,14 +253,6 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): "001\t2\t003\n004\t5\t\n", _pyarrow_read_csv_kwargs({"v": "varchar"}, True), ), - ( - '007\t"a\tb"\n010\tc\n', - _pyarrow_read_csv_kwargs({"v": "varchar", "x": "varchar"}, True), - ), - ( - '007\t"a\nb"\n010\tc\n', - _pyarrow_read_csv_kwargs({"v": "varchar", "x": "varchar"}, True), - ), ( '"v","n"\n"007","1"\n', _pyarrow_read_csv_kwargs( @@ -291,8 +283,6 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): "numeric_looking_strings", "arrow_string_dtypes", "tab_separated_numeric_fields", - "tab_separated_quoted_tab", - "tab_separated_quoted_newline", "dtype_position_key", "unparsed_dates", "tab_separated_extra_fields", @@ -302,9 +292,9 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): def test_read_csv_with_pyarrow_matches_pandas(data, read_csv_kwargs, infer_string): # Without values that cross a read block, the result is the one of # pandas.read_csv(engine="pyarrow"), except that the columns with a string - # dtype have the values of pandas' C engine. Where the C engine parses a - # parse_dates column despite its string dtype, the PyArrow engine's - # applying the dtype again is kept. + # dtype have the values of pandas' C engine, and in a header-less file, its + # missing values. Where the C engine parses a parse_dates column despite its + # string dtype, the PyArrow engine's applying the dtype again is kept. with pd.option_context("future.infer_string", infer_string): expected = pd.read_csv( io.BytesIO(data.encode()), @@ -318,8 +308,14 @@ def test_read_csv_with_pyarrow_matches_pandas(data, read_csv_kwargs, infer_strin column for column, value in read_csv_kwargs["dtype"].items() if _is_string_dtype(value) } for index, column in enumerate(expected.columns): - if column in string_columns and c_engine[column].dtype.kind != "M": - # The C engine makes the extra fields of a header-less file the index. + if column not in string_columns or c_engine[column].dtype.kind == "M": + continue + if read_csv_kwargs["header"] is None: + # Header-less fields keep the inferred types, and only their missing + # values follow the C engine, which makes extra fields the index. + missing = c_engine[column].isna().to_numpy() + expected.isetitem(index, expected.iloc[:, index].mask(missing, float("nan"))) + else: expected.isetitem(index, c_engine[column].array) actual = _read_csv_with_pyarrow( io.BytesIO(data.encode()),