diff --git a/pyathena/arrow/result_set.py b/pyathena/arrow/result_set.py index 64f65040..cb12f808 100644 --- a/pyathena/arrow/result_set.py +++ b/pyathena/arrow/result_set.py @@ -327,6 +327,9 @@ def _read_csv(self) -> Table: ignore_empty_lines=False, double_quote=True, escape_char=False, + # A quoted value can contain a newline, so the reader must not split + # blocks inside quotes. + newlines_in_values=True, ) else: return pa.Table.from_pydict({}) diff --git a/tests/pyathena/arrow/test_cursor.py b/tests/pyathena/arrow/test_cursor.py index 055d47b8..f9a66712 100644 --- a/tests/pyathena/arrow/test_cursor.py +++ b/tests/pyathena/arrow/test_cursor.py @@ -47,6 +47,18 @@ def test_binary_null_vs_empty(self, arrow_cursor): ] assert [row[3] for row in rows] == ["", "", "NULL"] + def test_multiline_values_across_blocks(self, arrow_cursor): + # The 50 two-line values of 301 bytes span several 1024-byte blocks. + arrow_cursor.execute( + """ + SELECT array_join(repeat('x', 150), '') || chr(10) || array_join(repeat('y', 150), '') + AS v + FROM UNNEST(sequence(1, 50)) AS t(i) + """, + block_size=1024, + ) + assert arrow_cursor.fetchall() == [("x" * 150 + "\n" + "y" * 150,)] * 50 + def test_binary_single_null(self, arrow_cursor): arrow_cursor.execute("SELECT CAST(NULL AS VARBINARY) AS value") assert arrow_cursor.fetchall() == [(None,)]