From 4596cb242cea8a23214f6587c58797ac196fe06a Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 16:54:37 +0900 Subject: [PATCH] Avoid Pandas JSON dtype warnings by defaulting to whole-chunk parsing --- docs/pandas.md | 10 +- pyathena/pandas/result_set.py | 6 + tests/pyathena/pandas/test_result_set.py | 135 +++++++++++++++++++++++ 3 files changed, 150 insertions(+), 1 deletion(-) diff --git a/docs/pandas.md b/docs/pandas.md index eb0d46ed..f4816ff4 100644 --- a/docs/pandas.md +++ b/docs/pandas.md @@ -552,10 +552,18 @@ cursor.execute("SELECT * FROM typed_table", Common performance options: - `engine`: CSV parsing engine ('auto', 'c', 'python', 'pyarrow'); 'auto' uses the C engine -- `low_memory`: Parse the file in internal chunks to reduce memory use (C engine only; pandas default `True`) +- `low_memory`: Parse the file in internal chunks to reduce memory use (C engine only) - `dtype`: Explicit column data types - `parse_dates`: Columns to parse as dates +When the C engine's converter mapping includes PyAthena's JSON converter, `low_memory` defaults to `False`. +This avoids pandas' `DtypeWarning` when JSON numbers or booleans and NULLs occur in different internal parser blocks. +The setting applies to the entire CSV read, including other columns whose dtypes pandas infers. +Parsing each result or chunk at once can use more memory; `chunksize` still limits the rows read per chunk. +An explicit `low_memory=True` or `low_memory=False` passed to `execute()` takes precedence. +With `low_memory=True`, the warning can return. +Without PyAthena's JSON converter, the C engine keeps pandas' default `low_memory=True`. + With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when the result is not a tab-separated `.txt` file, pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, `json`, `time with time zone`, and `timestamp with time zone` with the default converter), and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. Otherwise, it falls back to the C engine. With `engine="pyarrow"`, tab-separated `.txt` results from DDL statements such as `SHOW TABLES`, `SHOW COLUMNS`, and `DESCRIBE` use the C engine to preserve leading zeros, exponent notation, and padding in string values. diff --git a/pyathena/pandas/result_set.py b/pyathena/pandas/result_set.py index 85a0e22f..669683d0 100644 --- a/pyathena/pandas/result_set.py +++ b/pyathena/pandas/result_set.py @@ -899,6 +899,12 @@ def _read_csv(self) -> TextFileReader | DataFrame: # After _configure_binary_csv_read(), which checks for the header row. self._read_csv_header_as_labels(read_csv_kwargs, csv_engine) self._csv_converters = read_csv_kwargs.get("converters") or {} + if csv_engine == "c" and any( + isinstance(converter, _JSONConverter) + for converter in self._csv_converters.values() + ): + # A NULL placeholder can give internal parser blocks different dtypes. + read_csv_kwargs.setdefault("low_memory", False) if data is not None: # The rows are in memory, with nothing to open with storage options. read_csv_kwargs.pop("storage_options", None) diff --git a/tests/pyathena/pandas/test_result_set.py b/tests/pyathena/pandas/test_result_set.py index 8d7ceead..0a8cb680 100644 --- a/tests/pyathena/pandas/test_result_set.py +++ b/tests/pyathena/pandas/test_result_set.py @@ -300,6 +300,141 @@ def _pyarrow_read_csv_kwargs(types, tab_separated=False, **kwargs): return result_set._get_csv_read_options("pyarrow", None) +def _read_csv_result(data, types, engine="c", chunksize=None, **kwargs): + """Read an in-memory CSV through the result set and the real pandas parser.""" + result_set = AthenaPandasResultSet.__new__(AthenaPandasResultSet) + result_set._converter = DefaultPandasTypeConverter() + result_set._keep_default_na = False + result_set._na_values = ("",) + result_set._quoting = 1 + result_set._engine = engine + result_set._chunksize = chunksize + result_set._auto_optimize_chunksize = False + result_set._kwargs = kwargs + result_set._time_columns = [] + result_set._csv_stream = None + result_set._fs = MagicMock() + result_set._fs.open.return_value = io.BytesIO(data.encode()) + description = [(name, type_, None, None, 0, 0, "UNKNOWN") for name, type_ in types.items()] + with ( + patch.object( + AthenaPandasResultSet, + "description", + new_callable=PropertyMock, + return_value=description, + ), + patch.object( + AthenaPandasResultSet, + "output_location", + new_callable=PropertyMock, + return_value="s3://bucket/result.csv", + ), + patch.object( + AthenaPandasResultSet, + "substatement_type", + new_callable=PropertyMock, + return_value=None, + ), + patch.object(AthenaPandasResultSet, "_get_content_length", return_value=len(data)), + patch("pandas.read_csv", wraps=pd.read_csv) as read_csv, + patch( + "pyathena.pandas.result_set._read_csv_with_pyarrow", + wraps=_read_csv_with_pyarrow, + ) as read_pyarrow, + ): + result = result_set._read_csv() + options = read_pyarrow.call_args.args[1] if read_pyarrow.called else read_csv.call_args.kwargs + if isinstance(result, pd.DataFrame): + return result_set._finish_csv_frame(result), options + return PandasDataFrameIterator( + result, result_set._finish_csv_frame, result_set._csv_stream + ), options + + +@pytest.mark.filterwarnings("error::pandas.errors.DtypeWarning") +@pytest.mark.parametrize( + ("json_value", "expected"), [("9007199254740993", 9007199254740993), ("true", True)] +) +@pytest.mark.parametrize( + ("engine", "kwargs", "access"), + [ + pytest.param("c", {}, "whole", id="c"), + pytest.param("auto", {}, "whole", id="auto"), + pytest.param("pyarrow", {}, "whole", id="pyarrow-fallback"), + pytest.param("c", {"chunksize": 400005}, "iteration", id="iteration"), + pytest.param("c", {"chunksize": 400010}, "get_chunk", id="get-chunk"), + pytest.param("c", {"index_col": "j"}, "whole", id="index"), + pytest.param("c", {"usecols": ["j"]}, "whole", id="usecols"), + ], +) +def test_read_csv_json_null_without_dtype_warning(json_value, expected, engine, kwargs, access): + """JSON NULLs in a later parser block keep exact values without a warning.""" + data = "n,j\n" + f"1,{json_value}\n" * 400000 + "2,\n" * 10 + result, options = _read_csv_result(data, {"n": "integer", "j": "json"}, engine=engine, **kwargs) + if access != "whole": + with result: + first = result.get_chunk(400005) if access == "get_chunk" else next(result) + assert len(first) == 400005 + df = pd.concat([first, *result]) + else: + df = result + values = df.index if kwargs.get("index_col") == "j" else df["j"] + assert options["engine"] == "c" + assert options["low_memory"] is False + assert len(df) == 400010 + assert values.dtype == object + assert type(values[0]) is type(expected) + assert (values[:400000] == expected).all() + assert values[-10:].tolist() == [None] * 10 + + +def test_read_csv_json_explicit_low_memory_true(): + """An explicit low_memory=True keeps pandas' internal block parsing and warning.""" + data = "n,j\n" + "1,9007199254740993\n" * 400000 + "2,\n" * 10 + with pytest.warns(pd.errors.DtypeWarning, match="have mixed types"): + df, options = _read_csv_result(data, {"n": "integer", "j": "json"}, low_memory=True) + assert options["low_memory"] is True + assert df["j"].iloc[0] == 9007199254740993 + assert df["j"].iloc[-10:].tolist() == [None] * 10 + + +@pytest.mark.filterwarnings("error::pandas.errors.DtypeWarning") +@pytest.mark.parametrize("low_memory", [None, True, False]) +def test_read_csv_json_without_null(low_memory): + """JSON without NULL keeps its inferred dtype with default or explicit low_memory.""" + kwargs = {} if low_memory is None else {"low_memory": low_memory} + df, options = _read_csv_result("n,j\n1,2\n3,4\n", {"n": "integer", "j": "json"}, **kwargs) + assert options["low_memory"] is (False if low_memory is None else low_memory) + assert df["j"].dtype == "int64" + assert df["j"].tolist() == [2, 4] + + +@pytest.mark.parametrize( + ("types", "engine", "kwargs"), + [ + pytest.param({"n": "integer", "j": "integer"}, "c", {}, id="no-json"), + pytest.param({"n": "integer", "j": "json"}, "c", {"usecols": ["n"]}, id="json-excluded"), + pytest.param( + {"n": "integer", "j": "json"}, "c", {"converters": {}}, id="custom-converters" + ), + pytest.param( + {"n": "integer", "j": "json"}, + "c", + {"converters": {"j": int}}, + id="custom-json-converter", + ), + pytest.param({"n": "integer", "j": "json"}, "python", {}, id="python"), + pytest.param({"n": "integer", "j": "integer"}, "pyarrow", {}, id="pyarrow"), + ], +) +def test_read_csv_without_json_c_converter_keeps_low_memory_default(types, engine, kwargs): + """Other columns, converters, and engines do not get a low_memory option.""" + df, options = _read_csv_result("n,j\n" + "1,2\n" * 30, types, engine=engine, **kwargs) + assert options["engine"] == engine + assert "low_memory" not in options + assert len(df) == 30 + + @pytest.mark.filterwarnings("ignore:Could not infer format") @pytest.mark.parametrize("infer_string", [True, False]) @pytest.mark.parametrize(