diff --git a/pyathena/arrow/result_set.py b/pyathena/arrow/result_set.py index ea11a419f..799bbfb97 100644 --- a/pyathena/arrow/result_set.py +++ b/pyathena/arrow/result_set.py @@ -308,7 +308,8 @@ def _read_csv(self) -> Table: parse_opts = csv.ParseOptions( delimiter=",", quote_char='"', - ignore_empty_lines=not binary_columns, + # Athena writes a single-column row with a NULL value as an empty line. + ignore_empty_lines=False, double_quote=True, escape_char=False, ) diff --git a/tests/pyathena/arrow/test_cursor.py b/tests/pyathena/arrow/test_cursor.py index af1abca69..8d8f66bd6 100644 --- a/tests/pyathena/arrow/test_cursor.py +++ b/tests/pyathena/arrow/test_cursor.py @@ -46,6 +46,22 @@ def test_binary_single_null(self, arrow_cursor): arrow_cursor.execute("SELECT CAST(NULL AS VARBINARY) AS value") assert arrow_cursor.fetchall() == [(None,)] + @pytest.mark.parametrize( + ("query", "expected"), + [ + ( + "SELECT x FROM (VALUES 1, NULL, 2) AS t(x) ORDER BY x NULLS FIRST", + [(None,), (1,), (2,)], + ), + # Arrow reads a NULL string from a CSV result as an empty string. + ("SELECT CAST(NULL AS VARCHAR) AS v", [("",)]), + ], + ) + def test_single_column_null(self, arrow_cursor, query, expected): + arrow_cursor.execute(query) + assert arrow_cursor.as_arrow().num_rows == len(expected) + assert arrow_cursor.fetchall() == expected + @pytest.mark.parametrize( "arrow_cursor", [{"cursor_kwargs": {"unload": False}}, {"cursor_kwargs": {"unload": True}}],