From 5339aff7e2993ebbc72bd0385891a3588bdb1b1d Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 09:57:09 +0900 Subject: [PATCH 1/5] Read CSV results for the pandas PyArrow engine with pyarrow.csv pandas.read_csv(engine="pyarrow") builds its pyarrow ParseOptions without newlines_in_values and does not let callers set it, so a quoted value containing a newline that crosses one of pyarrow's 1 MiB read blocks made PandasCursor raise "CSV parser got out of sync with chunker" or return wrong values. Read the file with pyarrow.csv and newlines_in_values=True, and convert the table as pandas' PyArrow engine does for the options PandasCursor sets. The PyArrow engine is now used only when keep_default_na is False and the read_csv() options given to execute() are limited to dtype (a mapping), parse_dates, and storage_options; other options use the C engine. Document that PolarsCursor with use_pyarrow=True has the same limitation, which comes from pl.read_csv(). Closes #1048 Co-Authored-By: Claude Opus 5.5 --- docs/pandas.md | 3 +- docs/polars.md | 4 + pyathena/pandas/result_set.py | 123 +++++++++++++++++++++-- tests/pyathena/pandas/test_cursor.py | 63 +++++++++++- tests/pyathena/pandas/test_result_set.py | 119 ++++++++++++++++++++++ 5 files changed, 302 insertions(+), 10 deletions(-) diff --git a/docs/pandas.md b/docs/pandas.md index 20387867d..26b354caa 100644 --- a/docs/pandas.md +++ b/docs/pandas.md @@ -554,8 +554,9 @@ Common performance options: - `dtype`: Explicit column data types - `parse_dates`: Columns to parse as dates -With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, `json`, `time with time zone`, and `timestamp with time zone` with the default converter), and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. +With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, `keep_default_na` is `False`, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, `json`, `time with time zone`, and `timestamp with time zone` with the default converter), the only pandas.read_csv() options passed to `execute()` are `dtype` as a mapping, `parse_dates`, and `storage_options`, and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. Otherwise, it falls back to the C engine. +The PyArrow engine reads the file with `pyarrow.csv`, allowing newlines in quoted values, and converts it to a DataFrame as `pandas.read_csv(engine="pyarrow")` does. Apart from PandasCursor's own options such as `engine` and `chunksize`, an option passed here replaces the value PyAthena sets for the same pandas.read_csv() argument. For example, `dtype` replaces the whole column type mapping, and `parse_dates` replaces the list of date, time, and timestamp columns. diff --git a/docs/polars.md b/docs/polars.md index 9fcc692aa..37bb77b50 100644 --- a/docs/polars.md +++ b/docs/polars.md @@ -15,6 +15,10 @@ does not require PyArrow as a dependency. CSV results read without the chunksize PyAthena's own S3FileSystem (fsspec compatible), and other reads use Polars' native S3 access, so s3fs is also not required. +Options passed to `execute()`, other than PolarsCursor's own options such as `chunksize`, are passed to the Polars read function. +With `use_pyarrow=True` and `schema_overrides=None`, `pl.read_csv()` reads the CSV result with pyarrow, which raises an error or returns wrong values when a quoted value containing a newline crosses one of pyarrow's read blocks. +The default Polars CSV reader reads such values. + You can use the PolarsCursor by specifying the `cursor_class` with the connect method or connection object. diff --git a/pyathena/pandas/result_set.py b/pyathena/pandas/result_set.py index 54a864e07..d26ae9ab8 100644 --- a/pyathena/pandas/result_set.py +++ b/pyathena/pandas/result_set.py @@ -43,6 +43,102 @@ def _no_trunc_date(df: DataFrame) -> DataFrame: return df +def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any]) -> DataFrame: + """Read a CSV result with pyarrow as ``pandas.read_csv(engine="pyarrow")`` does. + + pandas' PyArrow engine does not set ``newlines_in_values``, so pyarrow + raises or returns wrong values when a quoted value containing a newline + crosses one of its read blocks. This reads the file with that option and + converts the table as pandas does for the options that + ``AthenaPandasResultSet._get_csv_engine()`` lets through: NULL-typed + columns become float64, integer columns without a ``dtype`` entry keep + NumPy integer types, the ``dtype`` mapping is applied, and the ``parse_dates`` columns + are parsed with ``pandas.to_datetime()``. + + Args: + source: The result file. + read_csv_kwargs: The pandas.read_csv() options built by + ``AthenaPandasResultSet._get_csv_read_options()``. + + Returns: + The result as a DataFrame. + """ + import pandas as pd + import pyarrow as pa + from pyarrow import csv as pyarrow_csv + + header = read_csv_kwargs["header"] + names = read_csv_kwargs["names"] + dtype = read_csv_kwargs["dtype"] + null_values = list(read_csv_kwargs["na_values"] or []) + table = pyarrow_csv.read_csv( + source, + read_options=pyarrow_csv.ReadOptions(autogenerate_column_names=header is None), + parse_options=pyarrow_csv.ParseOptions( + delimiter=read_csv_kwargs["sep"], + ignore_empty_lines=read_csv_kwargs["skip_blank_lines"], + newlines_in_values=True, + ), + convert_options=pyarrow_csv.ConvertOptions( + null_values=null_values, strings_can_be_null="" in null_values + ), + ) + schema = table.schema + for index, type_ in enumerate(schema.types): + if pa.types.is_null(type_): + schema = schema.set(index, schema.field(index).with_type(pa.float64())) + integer_dtypes = { + pa.int8(): pd.Int8Dtype(), + pa.int16(): pd.Int16Dtype(), + pa.int32(): pd.Int32Dtype(), + pa.int64(): pd.Int64Dtype(), + } + # Integers become nullable dtypes first so that a dtype entry converts them + # without going through float64. + df = table.cast(schema).to_pandas(types_mapper=integer_dtypes.get) + if names is not None: + df.columns = names + df = df.astype( + { + # Integer columns without a dtype entry get NumPy integer types. + **{ + column: df[column].dtype.numpy_dtype + for column in df.columns + if df[column].dtype in integer_dtypes.values() + }, + **{column: dtype[column] for column in dtype if column in df.columns}, + } + ) + if not pd.get_option("future.infer_string"): + # Without the string dtype, pandas returns strings, and the string + # categories of categorical columns, as objects. + for column in df.columns: + column_dtype = df[column].dtype + if column_dtype == "str": + df[column] = df[column].astype(object).fillna(None) + elif isinstance(column_dtype, pd.CategoricalDtype) and ( + column_dtype.categories.dtype == "str" + ): + df[column] = df[column].astype( + pd.CategoricalDtype( + column_dtype.categories.astype(object), ordered=column_dtype.ordered + ) + ) + parse_dates = read_csv_kwargs["parse_dates"] + for column in parse_dates if isinstance(parse_dates, list) else []: + if isinstance(column, int) and column not in df.columns: + column = df.columns[column] + if df[column].dtype.kind in "Mm": + continue + values = df[column].astype("string") + try: + df[column] = pd.to_datetime(values, utc=False) + except (ValueError, TypeError): + # pandas keeps the column as strings if it cannot parse it. + df[column] = values.to_numpy(dtype=object, na_value=float("nan")) + return df + + class PandasDataFrameIterator(abc.Iterator): # type: ignore[type-arg] """Iterator for chunked DataFrame results from Athena queries. @@ -259,6 +355,10 @@ class AthenaPandasResultSet(AthenaResultSet): "time", "timestamp", ] + # The pandas.read_csv() options given to execute() that the PyArrow engine reads. + _PYARROW_READ_CSV_OPTIONS: ClassVar[frozenset[str]] = frozenset( + {"dtype", "parse_dates", "storage_options"} + ) def __init__( self, @@ -402,7 +502,11 @@ def _get_csv_engine( is_compatible = ( effective_chunksize is None and self._quoting == 1 + and not self._keep_default_na and not self.converters + # _read_csv_with_pyarrow() handles only these pandas.read_csv() options. + and self._kwargs.keys() <= self._PYARROW_READ_CSV_OPTIONS + and isinstance(self._kwargs.get("dtype", {}), dict) and (file_size_bytes is None or file_size_bytes >= self.PYARROW_MIN_FILE_SIZE_BYTES) ) if is_compatible: @@ -590,7 +694,17 @@ def _read_csv(self) -> TextFileReader | DataFrame: source = self._csv_stream = stack.enter_context( self._fs.open(self.output_location, mode="rb") ) - result = pd.read_csv(source, **read_csv_kwargs) + elif csv_engine == "pyarrow": + # Given storage_options, even None, open the file through fsspec + # as pandas does. + storage_options = read_csv_kwargs.pop("storage_options") or {} + source = self._csv_stream = stack.enter_context( + filesystem_open(self.output_location, mode="rb", **storage_options) + ) + if csv_engine == "pyarrow": + result = _read_csv_with_pyarrow(source, read_csv_kwargs) + else: + result = pd.read_csv(source, **read_csv_kwargs) if not isinstance(result, pd.DataFrame): # The chunk iterator takes ownership of the stream. stack.pop_all() @@ -635,13 +749,6 @@ def _get_csv_read_options(self, csv_engine: str, chunksize: int | None) -> dict[ "chunksize": chunksize, "engine": csv_engine, } - - # Engine-specific compatibility adjustments - if csv_engine == "pyarrow": - # PyArrow doesn't support these pandas-specific options - read_csv_kwargs.pop("quoting", None) - read_csv_kwargs.pop("converters", None) - read_csv_kwargs.update(self._kwargs) return read_csv_kwargs diff --git a/tests/pyathena/pandas/test_cursor.py b/tests/pyathena/pandas/test_cursor.py index d4d78c9c9..0a5c0aa3a 100644 --- a/tests/pyathena/pandas/test_cursor.py +++ b/tests/pyathena/pandas/test_cursor.py @@ -19,7 +19,11 @@ from pyathena.model import AthenaQueryExecution from pyathena.pandas.converter import DefaultPandasTypeConverter from pyathena.pandas.cursor import PandasCursor -from pyathena.pandas.result_set import AthenaPandasResultSet, PandasDataFrameIterator +from pyathena.pandas.result_set import ( + AthenaPandasResultSet, + PandasDataFrameIterator, + _read_csv_with_pyarrow, +) from pyathena.util import RetryConfig from tests import ENV from tests.pyathena.conftest import connect @@ -293,6 +297,28 @@ def test_csv_storage_options(self, pandas_cursor, query, expected, binary, with_ # pandas opens and closes the file itself. assert pandas_cursor.result_set._csv_stream is None + @pytest.mark.parametrize( + "read_options", [{}, {"storage_options": None}], ids=["filesystem", "storage_options"] + ) + def test_pyarrow_engine_multiline_values_across_blocks(self, pandas_cursor, read_options): + # The 2.4 MB result spans several 1 MiB pyarrow read blocks, and its + # values contain a newline and quotes. + with patch( + "pyathena.pandas.result_set._read_csv_with_pyarrow", wraps=_read_csv_with_pyarrow + ) as read_csv_with_pyarrow: + pandas_cursor.execute( + """ + SELECT array_join(repeat('x', 600), '') || chr(10) + || '"' || array_join(repeat('y', 598), '') || '"' AS v + FROM UNNEST(sequence(1, 2000)) AS t(i) + """, + engine="pyarrow", + **read_options, + ) + df = pandas_cursor.as_pandas() + read_csv_with_pyarrow.assert_called_once() + assert df["v"].tolist() == ["x" * 600 + '\n"' + "y" * 598 + '"'] * 2000 + @pytest.mark.parametrize( ("pandas_cursor", "parquet_engine", "chunksize"), [ @@ -961,6 +987,8 @@ def test_get_csv_engine_explicit_specification(self): result_set = AthenaPandasResultSet.__new__(AthenaPandasResultSet) result_set._chunksize = None # Default values result_set._quoting = 1 + result_set._keep_default_na = False + result_set._kwargs = {} # Test C engine specification result_set._engine = "c" @@ -1036,6 +1064,39 @@ def test_get_csv_engine_explicit_specification(self): engine = result_set._get_csv_engine() assert engine == "c" + @pytest.mark.parametrize( + ("keep_default_na", "kwargs", "expected"), + [ + ( + False, + {"dtype": {"a": "str"}, "parse_dates": ["b"], "storage_options": None}, + "pyarrow", + ), + (True, {}, "c"), + (False, {"usecols": ["a"]}, "c"), + (False, {"dtype_backend": "pyarrow"}, "c"), + (False, {"dtype": "str"}, "c"), + ], + ids=["supported_options", "keep_default_na", "usecols", "dtype_backend", "single_dtype"], + ) + def test_get_csv_engine_pyarrow_read_options(self, keep_default_na, kwargs, expected): + # The PyArrow engine reads only the pandas.read_csv() options that + # _read_csv_with_pyarrow() handles; other options use the C engine. + with patch("pyathena.pandas.result_set.AthenaResultSet.__init__"): + result_set = AthenaPandasResultSet.__new__(AthenaPandasResultSet) + result_set._engine = "pyarrow" + result_set._chunksize = None + result_set._quoting = 1 + result_set._keep_default_na = keep_default_na + result_set._kwargs = kwargs + with ( + patch.object(result_set, "_get_available_engine", return_value="pyarrow"), + patch.object( + type(result_set), "converters", new_callable=PropertyMock, return_value={} + ), + ): + assert result_set._get_csv_engine() == expected + @pytest.mark.parametrize( ("pandas_cursor", "parquet_engine", "chunksize"), [ diff --git a/tests/pyathena/pandas/test_result_set.py b/tests/pyathena/pandas/test_result_set.py index 84ee98922..c311b2718 100644 --- a/tests/pyathena/pandas/test_result_set.py +++ b/tests/pyathena/pandas/test_result_set.py @@ -5,16 +5,20 @@ # # SPDX-License-Identifier: MIT +import csv import io from unittest.mock import MagicMock, patch import pandas as pd import pytest +from pandas.testing import assert_frame_equal +from pyathena.pandas.converter import DefaultPandasTypeConverter from pyathena.pandas.result_set import ( AthenaPandasResultSet, PandasDataFrameIterator, _no_trunc_date, + _read_csv_with_pyarrow, ) @@ -113,3 +117,118 @@ def test_read_parquet_filesystem(self, execute_kwargs, path, filesystem_kwargs): **execute_kwargs, **filesystem_kwargs, } + + +# A CSV result of Athena with each type the PyArrow engine reads, then a row of NULLs. +_TYPES_CSV = ( + '"ti","si","i","bi","r","d","c","v","ml","arr","m","rw","dt","ts","ts6","tm","iv","nul","u",' + '"empty","na"\n' + '"1","2","3","4","1.5","2.25","ab ","plain","multi\nline ""q"", x","[1, 2]","{k=1}",' + '"{a=1, b=x}","2024-02-29","2024-02-29 23:59:58.123","2024-02-29 23:59:58.123456",' + '"12:34:56.789","2 00:00:00.000",,"589f6631-9c50-4f58-a121-e2608a04fc64","","NA"\n' + ",,,,,,,,,,,,,,,,,,,,\n" +) +_TYPES = { + "ti": "tinyint", + "si": "smallint", + "i": "integer", + "bi": "bigint", + "r": "float", + "d": "double", + "c": "char", + "v": "varchar", + "ml": "varchar", + "arr": "array", + "m": "map", + "rw": "row", + "dt": "date", + "ts": "timestamp", + "ts6": "timestamp", + "tm": "time", + "iv": "interval day to second", + "nul": "unknown", + "u": "uuid", + "empty": "varchar", + "na": "varchar", +} + + +def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): + """Build the pandas.read_csv() options that PandasCursor passes to the PyArrow engine.""" + converter = DefaultPandasTypeConverter() + return { + "sep": "\t" if tab_separated else ",", + "header": None if tab_separated else 0, + "names": list(types) if tab_separated else None, + "dtype": { + name: dtype + for name, type_ in types.items() + if (dtype := converter.get_dtype(type_, 0, 0)) is not None + }, + "parse_dates": [ + name for name, type_ in types.items() if type_ in ("date", "time", "timestamp") + ], + "skip_blank_lines": False, + "keep_default_na": False, + "na_values": ("",), + **kwargs, + } + + +@pytest.mark.filterwarnings("ignore:Could not infer format") +@pytest.mark.parametrize("infer_string", [True, False]) +@pytest.mark.parametrize( + ("data", "read_csv_kwargs"), + [ + (_TYPES_CSV, _pyarrow_read_csv_kwargs(_TYPES)), + (_TYPES_CSV, _pyarrow_read_csv_kwargs(_TYPES, na_values=("", "NA"))), + ( + _TYPES_CSV, + _pyarrow_read_csv_kwargs( + _TYPES, + dtype={ + **_pyarrow_read_csv_kwargs(_TYPES)["dtype"], + "ti": "float32", + "v": "category", + "missing": "int64", + }, + ), + ), + (_TYPES_CSV, _pyarrow_read_csv_kwargs(_TYPES, parse_dates=[12, "ts"])), + ( + "x\t1\t2024-01-01\n\t\t\ny y\t3\t2024-01-02\n", + _pyarrow_read_csv_kwargs({"a": "varchar", "b": "bigint", "c": "date"}, True), + ), + ], + ids=["types", "na_values", "dtype", "parse_dates", "tab_separated"], +) +def test_read_csv_with_pyarrow_matches_pandas(data, read_csv_kwargs, infer_string): + # Without values that cross a read block, the result is the one of + # pandas.read_csv(engine="pyarrow"). + with pd.option_context("future.infer_string", infer_string): + expected = pd.read_csv( + io.BytesIO(data.encode()), + engine="pyarrow", + **{**read_csv_kwargs, "dtype": dict(read_csv_kwargs["dtype"])}, + ) + actual = _read_csv_with_pyarrow( + io.BytesIO(data.encode()), + {**read_csv_kwargs, "dtype": dict(read_csv_kwargs["dtype"])}, + ) + assert_frame_equal(actual, expected, check_exact=True) + + +def test_read_csv_with_pyarrow_multiline_values_across_blocks(): + # The 2.4 MB file spans several 1 MiB pyarrow read blocks, and its values + # contain newlines and quotes. + rows = [(i, f'{"x" * 600}\n"{"y" * 598}"' if i % 2 else "z") for i in range(4000)] + buffer = io.StringIO() + writer = csv.writer(buffer, quoting=csv.QUOTE_ALL, lineterminator="\n") + writer.writerow(["id", "v"]) + writer.writerows(rows) + df = _read_csv_with_pyarrow( + io.BytesIO(buffer.getvalue().encode()), + _pyarrow_read_csv_kwargs({"id": "integer", "v": "varchar"}), + ) + assert df["id"].tolist() == [i for i, _ in rows] + assert df["v"].tolist() == [v for _, v in rows] From e272719123fce03c291f22f015ba00ff04b8eb0d Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 09:58:45 +0900 Subject: [PATCH 2/5] Name the conditions under which pl.read_csv() reads with pyarrow Co-Authored-By: Claude Opus 5.5 --- docs/polars.md | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/docs/polars.md b/docs/polars.md index 37bb77b50..f347e4214 100644 --- a/docs/polars.md +++ b/docs/polars.md @@ -16,7 +16,8 @@ PyAthena's own S3FileSystem (fsspec compatible), and other reads use Polars' nat so s3fs is also not required. Options passed to `execute()`, other than PolarsCursor's own options such as `chunksize`, are passed to the Polars read function. -With `use_pyarrow=True` and `schema_overrides=None`, `pl.read_csv()` reads the CSV result with pyarrow, which raises an error or returns wrong values when a quoted value containing a newline crosses one of pyarrow's read blocks. +`pl.read_csv()` reads the CSV result with pyarrow when `use_pyarrow=True` is given with `schema_overrides=None`, which replaces PyAthena's column types, and Polars' other conditions for it hold. +That read raises an error or returns wrong values when a quoted value containing a newline crosses one of pyarrow's read blocks. The default Polars CSV reader reads such values. You can use the PolarsCursor by specifying the `cursor_class` From 84e2fca36c98d8a6306d1ef3ecfd0a9a591e51c8 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 10:15:20 +0900 Subject: [PATCH 3/5] Let pandas read the PyArrow engine's CSV results it handles differently The independent review found that _read_csv_with_pyarrow() diverged from pandas.read_csv(engine="pyarrow"), and that sending other cases to the C engine broke callers: - pandas applies the dtype mapping again after parsing dates, so a dtype entry for a parse_dates column wins; apply it again. - Duplicate column names made df[column] a DataFrame; look up a column's dtype only when it has no dtype entry, as pandas does, and convert strings to objects by position. - Numeric na_values, keep_default_na, storage_options (with pandas' anonymous retry), and PyArrow-only options such as a callable on_bad_lines are not reproduced, and the C engine rejects some of them. Keep the engine selection unchanged, and read the file with _read_csv_with_pyarrow() only when keep_default_na and na_values are the defaults and execute() passes no read_csv() options other than a dtype mapping and parse_dates; pandas reads it otherwise, as before. Co-Authored-By: Claude Opus 5.5 --- docs/pandas.md | 5 +- pyathena/pandas/result_set.py | 96 +++++++++++++----------- tests/pyathena/pandas/test_cursor.py | 56 ++++++-------- tests/pyathena/pandas/test_result_set.py | 20 ++++- 4 files changed, 99 insertions(+), 78 deletions(-) diff --git a/docs/pandas.md b/docs/pandas.md index 26b354caa..5b2706bbc 100644 --- a/docs/pandas.md +++ b/docs/pandas.md @@ -554,9 +554,10 @@ Common performance options: - `dtype`: Explicit column data types - `parse_dates`: Columns to parse as dates -With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, `keep_default_na` is `False`, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, `json`, `time with time zone`, and `timestamp with time zone` with the default converter), the only pandas.read_csv() options passed to `execute()` are `dtype` as a mapping, `parse_dates`, and `storage_options`, and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. +With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, `json`, `time with time zone`, and `timestamp with time zone` with the default converter), and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. Otherwise, it falls back to the C engine. -The PyArrow engine reads the file with `pyarrow.csv`, allowing newlines in quoted values, and converts it to a DataFrame as `pandas.read_csv(engine="pyarrow")` does. +With the PyArrow engine, when `keep_default_na` and `na_values` are the defaults and the only pandas.read_csv() options passed to `execute()` are `dtype` as a mapping and `parse_dates`, PyAthena reads the file with `pyarrow.csv`, allowing newlines in quoted values, and converts it to a DataFrame as `pandas.read_csv(engine="pyarrow")` does. +Otherwise, pandas reads the file, and its PyArrow engine raises an error or returns wrong values when a quoted value containing a newline crosses one of pyarrow's read blocks. Apart from PandasCursor's own options such as `engine` and `chunksize`, an option passed here replaces the value PyAthena sets for the same pandas.read_csv() argument. For example, `dtype` replaces the whole column type mapping, and `parse_dates` replaces the list of date, time, and timestamp columns. diff --git a/pyathena/pandas/result_set.py b/pyathena/pandas/result_set.py index d26ae9ab8..d6a66a947 100644 --- a/pyathena/pandas/result_set.py +++ b/pyathena/pandas/result_set.py @@ -50,10 +50,11 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] raises or returns wrong values when a quoted value containing a newline crosses one of its read blocks. This reads the file with that option and converts the table as pandas does for the options that - ``AthenaPandasResultSet._get_csv_engine()`` lets through: NULL-typed - columns become float64, integer columns without a ``dtype`` entry keep - NumPy integer types, the ``dtype`` mapping is applied, and the ``parse_dates`` columns - are parsed with ``pandas.to_datetime()``. + ``AthenaPandasResultSet._reads_csv_with_pyarrow()`` accepts: NULL-typed + columns become float64, integer columns without a ``dtype`` entry get + NumPy integer types, the ``dtype`` mapping is applied before and after the + ``parse_dates`` columns are parsed with ``pandas.to_datetime()``, and + without ``future.infer_string``, strings become objects. Args: source: The result file. @@ -69,8 +70,7 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] header = read_csv_kwargs["header"] names = read_csv_kwargs["names"] - dtype = read_csv_kwargs["dtype"] - null_values = list(read_csv_kwargs["na_values"] or []) + null_values = list(read_csv_kwargs["na_values"]) table = pyarrow_csv.read_csv( source, read_options=pyarrow_csv.ReadOptions(autogenerate_column_names=header is None), @@ -98,31 +98,27 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] df = table.cast(schema).to_pandas(types_mapper=integer_dtypes.get) if names is not None: df.columns = names - df = df.astype( - { - # Integer columns without a dtype entry get NumPy integer types. - **{ - column: df[column].dtype.numpy_dtype - for column in df.columns - if df[column].dtype in integer_dtypes.values() - }, - **{column: dtype[column] for column in dtype if column in df.columns}, - } - ) + dtype = dict(read_csv_kwargs["dtype"]) + for column in df.columns: + # Integer columns without a dtype entry get NumPy integer types. + if column not in dtype and df[column].dtype in integer_dtypes.values(): + dtype[column] = df[column].dtype.numpy_dtype + dtype = {column: value for column, value in dtype.items() if column in df.columns} + df = df.astype(dtype) if not pd.get_option("future.infer_string"): # Without the string dtype, pandas returns strings, and the string # categories of categorical columns, as objects. - for column in df.columns: - column_dtype = df[column].dtype - if column_dtype == "str": - df[column] = df[column].astype(object).fillna(None) - elif isinstance(column_dtype, pd.CategoricalDtype) and ( - column_dtype.categories.dtype == "str" + for index in range(len(df.columns)): + values = df.iloc[:, index] + if values.dtype == "str": + df.isetitem(index, values.astype(object).fillna(None)) + elif isinstance(values.dtype, pd.CategoricalDtype) and ( + values.dtype.categories.dtype == "str" ): - df[column] = df[column].astype( - pd.CategoricalDtype( - column_dtype.categories.astype(object), ordered=column_dtype.ordered - ) + categories = values.dtype.categories.astype(object) + df.isetitem( + index, + values.astype(pd.CategoricalDtype(categories, ordered=values.dtype.ordered)), ) parse_dates = read_csv_kwargs["parse_dates"] for column in parse_dates if isinstance(parse_dates, list) else []: @@ -136,6 +132,8 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] except (ValueError, TypeError): # pandas keeps the column as strings if it cannot parse it. df[column] = values.to_numpy(dtype=object, na_value=float("nan")) + # pandas applies the dtype mapping again after parsing dates. + df = df.astype(dtype) return df @@ -355,10 +353,8 @@ class AthenaPandasResultSet(AthenaResultSet): "time", "timestamp", ] - # The pandas.read_csv() options given to execute() that the PyArrow engine reads. - _PYARROW_READ_CSV_OPTIONS: ClassVar[frozenset[str]] = frozenset( - {"dtype", "parse_dates", "storage_options"} - ) + # The pandas.read_csv() options given to execute() that _read_csv_with_pyarrow() reads. + _PYARROW_READ_CSV_OPTIONS: ClassVar[frozenset[str]] = frozenset({"dtype", "parse_dates"}) def __init__( self, @@ -502,11 +498,7 @@ def _get_csv_engine( is_compatible = ( effective_chunksize is None and self._quoting == 1 - and not self._keep_default_na and not self.converters - # _read_csv_with_pyarrow() handles only these pandas.read_csv() options. - and self._kwargs.keys() <= self._PYARROW_READ_CSV_OPTIONS - and isinstance(self._kwargs.get("dtype", {}), dict) and (file_size_bytes is None or file_size_bytes >= self.PYARROW_MIN_FILE_SIZE_BYTES) ) if is_compatible: @@ -694,14 +686,7 @@ def _read_csv(self) -> TextFileReader | DataFrame: source = self._csv_stream = stack.enter_context( self._fs.open(self.output_location, mode="rb") ) - elif csv_engine == "pyarrow": - # Given storage_options, even None, open the file through fsspec - # as pandas does. - storage_options = read_csv_kwargs.pop("storage_options") or {} - source = self._csv_stream = stack.enter_context( - filesystem_open(self.output_location, mode="rb", **storage_options) - ) - if csv_engine == "pyarrow": + if csv_engine == "pyarrow" and self._reads_csv_with_pyarrow(): result = _read_csv_with_pyarrow(source, read_csv_kwargs) else: result = pd.read_csv(source, **read_csv_kwargs) @@ -724,6 +709,24 @@ def _read_csv(self) -> TextFileReader | DataFrame: _logger.exception(f"Failed to read {self.output_location}.") raise OperationalError(*e.args) from e + def _reads_csv_with_pyarrow(self) -> bool: + """Whether ``_read_csv_with_pyarrow()`` reads the CSV result for the PyArrow engine. + + It reproduces ``pandas.read_csv(engine="pyarrow")`` for PyAthena's default NA + values and for the ``dtype`` mapping and ``parse_dates`` options given to + ``execute()``. With other options, pandas reads the file. + + Returns: + True if ``_read_csv_with_pyarrow()`` reads the result. + """ + return ( + not self._keep_default_na + and isinstance(self._na_values, (list, tuple)) + and list(self._na_values) == [""] + and self._kwargs.keys() <= self._PYARROW_READ_CSV_OPTIONS + and isinstance(self._kwargs.get("dtype", {}), dict) + ) + def _get_csv_read_options(self, csv_engine: str, chunksize: int | None) -> dict[str, Any]: """Build pandas options for Athena CSV or tab-separated results.""" if self.output_location and self.output_location.endswith(".txt"): @@ -749,6 +752,13 @@ def _get_csv_read_options(self, csv_engine: str, chunksize: int | None) -> dict[ "chunksize": chunksize, "engine": csv_engine, } + + # Engine-specific compatibility adjustments + if csv_engine == "pyarrow": + # PyArrow doesn't support these pandas-specific options + read_csv_kwargs.pop("quoting", None) + read_csv_kwargs.pop("converters", None) + read_csv_kwargs.update(self._kwargs) return read_csv_kwargs diff --git a/tests/pyathena/pandas/test_cursor.py b/tests/pyathena/pandas/test_cursor.py index 0a5c0aa3a..a30526b5e 100644 --- a/tests/pyathena/pandas/test_cursor.py +++ b/tests/pyathena/pandas/test_cursor.py @@ -297,10 +297,7 @@ def test_csv_storage_options(self, pandas_cursor, query, expected, binary, with_ # pandas opens and closes the file itself. assert pandas_cursor.result_set._csv_stream is None - @pytest.mark.parametrize( - "read_options", [{}, {"storage_options": None}], ids=["filesystem", "storage_options"] - ) - def test_pyarrow_engine_multiline_values_across_blocks(self, pandas_cursor, read_options): + def test_pyarrow_engine_multiline_values_across_blocks(self, pandas_cursor): # The 2.4 MB result spans several 1 MiB pyarrow read blocks, and its # values contain a newline and quotes. with patch( @@ -313,7 +310,6 @@ def test_pyarrow_engine_multiline_values_across_blocks(self, pandas_cursor, read FROM UNNEST(sequence(1, 2000)) AS t(i) """, engine="pyarrow", - **read_options, ) df = pandas_cursor.as_pandas() read_csv_with_pyarrow.assert_called_once() @@ -987,8 +983,6 @@ def test_get_csv_engine_explicit_specification(self): result_set = AthenaPandasResultSet.__new__(AthenaPandasResultSet) result_set._chunksize = None # Default values result_set._quoting = 1 - result_set._keep_default_na = False - result_set._kwargs = {} # Test C engine specification result_set._engine = "c" @@ -1065,37 +1059,37 @@ def test_get_csv_engine_explicit_specification(self): assert engine == "c" @pytest.mark.parametrize( - ("keep_default_na", "kwargs", "expected"), + ("keep_default_na", "na_values", "kwargs", "expected"), [ - ( - False, - {"dtype": {"a": "str"}, "parse_dates": ["b"], "storage_options": None}, - "pyarrow", - ), - (True, {}, "c"), - (False, {"usecols": ["a"]}, "c"), - (False, {"dtype_backend": "pyarrow"}, "c"), - (False, {"dtype": "str"}, "c"), + (False, ("",), {}, True), + (False, ("",), {"dtype": {"a": "str"}, "parse_dates": ["b"]}, True), + (True, ("",), {}, False), + (False, ("", "NA"), {}, False), + (False, np.array([""]), {}, False), + (False, ("",), {"storage_options": None}, False), + (False, ("",), {"on_bad_lines": "skip"}, False), + (False, ("",), {"dtype": "str"}, False), + ], + ids=[ + "default", + "dtype_parse_dates", + "keep_default_na", + "na_values", + "na_values_array", + "storage_options", + "on_bad_lines", + "single_dtype", ], - ids=["supported_options", "keep_default_na", "usecols", "dtype_backend", "single_dtype"], ) - def test_get_csv_engine_pyarrow_read_options(self, keep_default_na, kwargs, expected): - # The PyArrow engine reads only the pandas.read_csv() options that - # _read_csv_with_pyarrow() handles; other options use the C engine. + def test_reads_csv_with_pyarrow(self, keep_default_na, na_values, kwargs, expected): + # PyAthena reads the CSV result for the PyArrow engine only with the options + # that _read_csv_with_pyarrow() reproduces; pandas reads it otherwise. with patch("pyathena.pandas.result_set.AthenaResultSet.__init__"): result_set = AthenaPandasResultSet.__new__(AthenaPandasResultSet) - result_set._engine = "pyarrow" - result_set._chunksize = None - result_set._quoting = 1 result_set._keep_default_na = keep_default_na + result_set._na_values = na_values result_set._kwargs = kwargs - with ( - patch.object(result_set, "_get_available_engine", return_value="pyarrow"), - patch.object( - type(result_set), "converters", new_callable=PropertyMock, return_value={} - ), - ): - assert result_set._get_csv_engine() == expected + assert result_set._reads_csv_with_pyarrow() is expected @pytest.mark.parametrize( ("pandas_cursor", "parquet_engine", "chunksize"), diff --git a/tests/pyathena/pandas/test_result_set.py b/tests/pyathena/pandas/test_result_set.py index c311b2718..9872c73c8 100644 --- a/tests/pyathena/pandas/test_result_set.py +++ b/tests/pyathena/pandas/test_result_set.py @@ -181,7 +181,6 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): ("data", "read_csv_kwargs"), [ (_TYPES_CSV, _pyarrow_read_csv_kwargs(_TYPES)), - (_TYPES_CSV, _pyarrow_read_csv_kwargs(_TYPES, na_values=("", "NA"))), ( _TYPES_CSV, _pyarrow_read_csv_kwargs( @@ -195,12 +194,29 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): ), ), (_TYPES_CSV, _pyarrow_read_csv_kwargs(_TYPES, parse_dates=[12, "ts"])), + ( + _TYPES_CSV, + _pyarrow_read_csv_kwargs( + _TYPES, dtype={**_pyarrow_read_csv_kwargs(_TYPES)["dtype"], "dt": "string"} + ), + ), + ( + '"x","x","d"\n"1","2","2024-01-01"\n,,\n', + _pyarrow_read_csv_kwargs({"x": "integer", "d": "date"}), + ), ( "x\t1\t2024-01-01\n\t\t\ny y\t3\t2024-01-02\n", _pyarrow_read_csv_kwargs({"a": "varchar", "b": "bigint", "c": "date"}, True), ), ], - ids=["types", "na_values", "dtype", "parse_dates", "tab_separated"], + ids=[ + "types", + "dtype", + "parse_dates", + "dtype_of_date_column", + "duplicate_names", + "tab_separated", + ], ) def test_read_csv_with_pyarrow_matches_pandas(data, read_csv_kwargs, infer_string): # Without values that cross a read block, the result is the one of From fbceb11b894f3b208554bc53ba86af3e87545119 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 10:50:17 +0900 Subject: [PATCH 4/5] Normalize dtype values and take only a parse_dates list in the PyArrow reader pandas normalizes dtype mapping values with pandas_dtype(), so None means float64, and rejects a parse_dates value other than a boolean or a list. Normalize the values the same way, and let pandas read the file when parse_dates given to execute() is not a list. Co-Authored-By: Claude Opus 5.5 --- docs/pandas.md | 2 +- pyathena/pandas/result_set.py | 12 ++++++++---- tests/pyathena/pandas/test_cursor.py | 4 ++++ tests/pyathena/pandas/test_result_set.py | 5 +++++ 4 files changed, 18 insertions(+), 5 deletions(-) diff --git a/docs/pandas.md b/docs/pandas.md index 5b2706bbc..bdb0f7498 100644 --- a/docs/pandas.md +++ b/docs/pandas.md @@ -556,7 +556,7 @@ Common performance options: With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, `json`, `time with time zone`, and `timestamp with time zone` with the default converter), and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. Otherwise, it falls back to the C engine. -With the PyArrow engine, when `keep_default_na` and `na_values` are the defaults and the only pandas.read_csv() options passed to `execute()` are `dtype` as a mapping and `parse_dates`, PyAthena reads the file with `pyarrow.csv`, allowing newlines in quoted values, and converts it to a DataFrame as `pandas.read_csv(engine="pyarrow")` does. +With the PyArrow engine, when `keep_default_na` and `na_values` are the defaults and the only pandas.read_csv() options passed to `execute()` are `dtype` as a mapping and `parse_dates` as a list, PyAthena reads the file with `pyarrow.csv`, allowing newlines in quoted values, and converts it to a DataFrame as `pandas.read_csv(engine="pyarrow")` does. Otherwise, pandas reads the file, and its PyArrow engine raises an error or returns wrong values when a quoted value containing a newline crosses one of pyarrow's read blocks. Apart from PandasCursor's own options such as `engine` and `chunksize`, an option passed here replaces the value PyAthena sets for the same pandas.read_csv() argument. diff --git a/pyathena/pandas/result_set.py b/pyathena/pandas/result_set.py index d6a66a947..f52c58602 100644 --- a/pyathena/pandas/result_set.py +++ b/pyathena/pandas/result_set.py @@ -103,7 +103,11 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] # Integer columns without a dtype entry get NumPy integer types. if column not in dtype and df[column].dtype in integer_dtypes.values(): dtype[column] = df[column].dtype.numpy_dtype - dtype = {column: value for column, value in dtype.items() if column in df.columns} + dtype = { + column: pd.api.types.pandas_dtype(value) + for column, value in dtype.items() + if column in df.columns + } df = df.astype(dtype) if not pd.get_option("future.infer_string"): # Without the string dtype, pandas returns strings, and the string @@ -120,8 +124,7 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] index, values.astype(pd.CategoricalDtype(categories, ordered=values.dtype.ordered)), ) - parse_dates = read_csv_kwargs["parse_dates"] - for column in parse_dates if isinstance(parse_dates, list) else []: + for column in read_csv_kwargs["parse_dates"]: if isinstance(column, int) and column not in df.columns: column = df.columns[column] if df[column].dtype.kind in "Mm": @@ -713,7 +716,7 @@ def _reads_csv_with_pyarrow(self) -> bool: """Whether ``_read_csv_with_pyarrow()`` reads the CSV result for the PyArrow engine. It reproduces ``pandas.read_csv(engine="pyarrow")`` for PyAthena's default NA - values and for the ``dtype`` mapping and ``parse_dates`` options given to + values and for ``dtype`` as a mapping and ``parse_dates`` as a list given to ``execute()``. With other options, pandas reads the file. Returns: @@ -725,6 +728,7 @@ def _reads_csv_with_pyarrow(self) -> bool: and list(self._na_values) == [""] and self._kwargs.keys() <= self._PYARROW_READ_CSV_OPTIONS and isinstance(self._kwargs.get("dtype", {}), dict) + and isinstance(self._kwargs.get("parse_dates", []), list) ) def _get_csv_read_options(self, csv_engine: str, chunksize: int | None) -> dict[str, Any]: diff --git a/tests/pyathena/pandas/test_cursor.py b/tests/pyathena/pandas/test_cursor.py index a30526b5e..0d78019dc 100644 --- a/tests/pyathena/pandas/test_cursor.py +++ b/tests/pyathena/pandas/test_cursor.py @@ -1069,6 +1069,8 @@ def test_get_csv_engine_explicit_specification(self): (False, ("",), {"storage_options": None}, False), (False, ("",), {"on_bad_lines": "skip"}, False), (False, ("",), {"dtype": "str"}, False), + (False, ("",), {"parse_dates": "d"}, False), + (False, ("",), {"parse_dates": ("d",)}, False), ], ids=[ "default", @@ -1079,6 +1081,8 @@ def test_get_csv_engine_explicit_specification(self): "storage_options", "on_bad_lines", "single_dtype", + "parse_dates_string", + "parse_dates_tuple", ], ) def test_reads_csv_with_pyarrow(self, keep_default_na, na_values, kwargs, expected): diff --git a/tests/pyathena/pandas/test_result_set.py b/tests/pyathena/pandas/test_result_set.py index 9872c73c8..e0b01cecf 100644 --- a/tests/pyathena/pandas/test_result_set.py +++ b/tests/pyathena/pandas/test_result_set.py @@ -200,6 +200,10 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): _TYPES, dtype={**_pyarrow_read_csv_kwargs(_TYPES)["dtype"], "dt": "string"} ), ), + ( + '"x","d"\n"1","2024-01-01"\n,\n', + _pyarrow_read_csv_kwargs({"x": "integer", "d": "date"}, dtype={"x": None}), + ), ( '"x","x","d"\n"1","2","2024-01-01"\n,,\n', _pyarrow_read_csv_kwargs({"x": "integer", "d": "date"}), @@ -214,6 +218,7 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): "dtype", "parse_dates", "dtype_of_date_column", + "dtype_none", "duplicate_names", "tab_separated", ], From 2302049431ff2ee9d2b49ba271ae634cc5252a10 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 11:58:50 +0900 Subject: [PATCH 5/5] Name extra fields of header-less results as pandas does pandas' PyArrow engine names the columns of a header-less file beyond the given names by their positions, while _read_csv_with_pyarrow() assigned the names directly and raised a length mismatch. Build the parity tests' options with the real _get_csv_read_options(), and cover the extra fields and a parse_dates column that does not parse. Co-Authored-By: Claude Opus 5.5 --- pyathena/pandas/result_set.py | 5 +- tests/pyathena/pandas/test_result_set.py | 67 ++++++++++++++++-------- 2 files changed, 49 insertions(+), 23 deletions(-) diff --git a/pyathena/pandas/result_set.py b/pyathena/pandas/result_set.py index f58fe3976..3b87348a0 100644 --- a/pyathena/pandas/result_set.py +++ b/pyathena/pandas/result_set.py @@ -96,8 +96,9 @@ def _read_csv_with_pyarrow(source: str | IOBase, read_csv_kwargs: dict[str, Any] # Integers become nullable dtypes first so that a dtype entry converts them # without going through float64. df = table.cast(schema).to_pandas(types_mapper=integer_dtypes.get) - if names is not None: - df.columns = names + if header is None: + # pandas names the columns beyond the given names by their positions. + df.columns = [str(index) for index in range(len(df.columns) - len(names))] + names dtype = dict(read_csv_kwargs["dtype"]) for column in df.columns: # Integer columns without a dtype entry get NumPy integer types. diff --git a/tests/pyathena/pandas/test_result_set.py b/tests/pyathena/pandas/test_result_set.py index e0b01cecf..ef6022a57 100644 --- a/tests/pyathena/pandas/test_result_set.py +++ b/tests/pyathena/pandas/test_result_set.py @@ -7,7 +7,7 @@ import csv import io -from unittest.mock import MagicMock, patch +from unittest.mock import MagicMock, PropertyMock, patch import pandas as pd import pytest @@ -154,25 +154,41 @@ def test_read_parquet_filesystem(self, execute_kwargs, path, filesystem_kwargs): def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): - """Build the pandas.read_csv() options that PandasCursor passes to the PyArrow engine.""" - converter = DefaultPandasTypeConverter() - return { - "sep": "\t" if tab_separated else ",", - "header": None if tab_separated else 0, - "names": list(types) if tab_separated else None, - "dtype": { - name: dtype - for name, type_ in types.items() - if (dtype := converter.get_dtype(type_, 0, 0)) is not None - }, - "parse_dates": [ - name for name, type_ in types.items() if type_ in ("date", "time", "timestamp") - ], - "skip_blank_lines": False, - "keep_default_na": False, - "na_values": ("",), - **kwargs, - } + """Build the pandas.read_csv() options with AthenaPandasResultSet._get_csv_read_options(). + + Args: + types: The Athena types of the result columns, keyed by column name. + tab_separated: Whether the result is a tab-separated ``.txt`` file. + **kwargs: The pandas.read_csv() options given to ``execute()``. + + Returns: + The options for the PyArrow engine. + """ + with patch("pyathena.pandas.result_set.AthenaResultSet.__init__", return_value=None): + result_set = AthenaPandasResultSet.__new__(AthenaPandasResultSet) + result_set._converter = DefaultPandasTypeConverter() + result_set._keep_default_na = False + result_set._na_values = ("",) + result_set._quoting = 1 + result_set._kwargs = kwargs + description = [(name, type_, None, None, 0, 0, "UNKNOWN") for name, type_ in types.items()] + location = f"s3://bucket/result.{'txt' if tab_separated else 'csv'}" + with ( + patch.object( + AthenaPandasResultSet, + "description", + new_callable=PropertyMock, + return_value=description, + ), + patch.object( + AthenaPandasResultSet, + "output_location", + new_callable=PropertyMock, + return_value=location, + ), + ): + assert result_set._reads_csv_with_pyarrow() + return result_set._get_csv_read_options("pyarrow", None) @pytest.mark.filterwarnings("ignore:Could not infer format") @@ -208,6 +224,14 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): '"x","x","d"\n"1","2","2024-01-01"\n,,\n', _pyarrow_read_csv_kwargs({"x": "integer", "d": "date"}), ), + ( + '"v"\n"plain"\n"2024-01-01"\n\n', + _pyarrow_read_csv_kwargs({"v": "varchar"}, parse_dates=["v"]), + ), + ( + "id \tint \t \nname \tstring \t \n", + _pyarrow_read_csv_kwargs({"col_name": "varchar"}, True), + ), ( "x\t1\t2024-01-01\n\t\t\ny y\t3\t2024-01-02\n", _pyarrow_read_csv_kwargs({"a": "varchar", "b": "bigint", "c": "date"}, True), @@ -220,6 +244,8 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): "dtype_of_date_column", "dtype_none", "duplicate_names", + "unparsed_dates", + "tab_separated_extra_fields", "tab_separated", ], ) @@ -229,7 +255,6 @@ def test_read_csv_with_pyarrow_matches_pandas(data, read_csv_kwargs, infer_strin with pd.option_context("future.infer_string", infer_string): expected = pd.read_csv( io.BytesIO(data.encode()), - engine="pyarrow", **{**read_csv_kwargs, "dtype": dict(read_csv_kwargs["dtype"])}, ) actual = _read_csv_with_pyarrow(