From f88ba4d8f066b85273feba39426c8baf6976deed Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sat, 3 Oct 2026 23:13:50 +0900 Subject: [PATCH 1/9] Convert time with time zone values to aware times The cursors returned TIME WITH TIME ZONE values as text, except PandasCursor reading a result file, which parsed them as dates and dropped the offset (and raised AttributeError when the offsets differed). TIME values with no fraction or more than six fraction digits also raised ValueError. - _parse_time() parses TIME values of any precision, truncating digits beyond microseconds as _parse_datetime() does, and _to_time_with_tz() returns a time with a fixed-offset tzinfo. The default, Arrow, Polars, and pandas converters map time with time zone to it. - PandasCursor no longer parses time with time zone columns as dates. - Arrow and Polars time types have no time zone, so with the GetQueryResults fallback their tables keep these values as text, as with a result file, and the fetch methods convert them. Co-Authored-By: Claude Opus 5.5 --- docs/pandas.md | 2 +- docs/s3fs.md | 1 + pyathena/arrow/converter.py | 3 ++ pyathena/arrow/result_set.py | 27 ++++++++---- pyathena/converter.py | 62 +++++++++++++++++++++++++--- pyathena/pandas/converter.py | 2 + pyathena/pandas/result_set.py | 5 +-- pyathena/polars/converter.py | 3 ++ pyathena/polars/result_set.py | 13 +++--- pyathena/result_set.py | 10 ++--- tests/pyathena/arrow/test_cursor.py | 6 +++ tests/pyathena/pandas/test_cursor.py | 5 ++- tests/pyathena/polars/test_cursor.py | 7 +++- tests/pyathena/s3fs/test_cursor.py | 6 +-- tests/pyathena/test_converter.py | 40 +++++++++++++++++- tests/pyathena/test_cursor.py | 6 ++- tests/pyathena/util.py | 22 ++++++++++ 17 files changed, 184 insertions(+), 36 deletions(-) diff --git a/docs/pandas.md b/docs/pandas.md index 3edc0dda0..580af3bc5 100644 --- a/docs/pandas.md +++ b/docs/pandas.md @@ -554,7 +554,7 @@ Common performance options: - `dtype`: Explicit column data types - `parse_dates`: Columns to parse as dates -With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, and `json` with the default converter), and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. +With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, `json`, and `time with time zone` with the default converter), and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. Otherwise, it falls back to the C engine. Apart from PandasCursor's own options such as `engine` and `chunksize`, an option passed here replaces the value PyAthena sets for the same pandas.read_csv() argument. diff --git a/docs/s3fs.md b/docs/s3fs.md index eb1f53709..31e44d73b 100644 --- a/docs/s3fs.md +++ b/docs/s3fs.md @@ -125,6 +125,7 @@ The following type mappings are used: | timestamp | datetime.datetime | | timestamp with time zone | datetime.datetime (timezone-aware) | | time | datetime.time | +| time with time zone | datetime.time (timezone-aware) | | varbinary | bytes | | array, map, row (struct) | Parsed into Python list/dict (see {ref}`usage-type-hints` for the types of nested values); values too complex to parse are returned as the original string | | json | Parsed JSON value (dict, list, or scalar) | diff --git a/pyathena/arrow/converter.py b/pyathena/arrow/converter.py index 88943efba..f65cbb85a 100644 --- a/pyathena/arrow/converter.py +++ b/pyathena/arrow/converter.py @@ -15,6 +15,7 @@ _to_decimal, _to_default, _to_time, + _to_time_with_tz, ) from pyathena.util import override @@ -24,6 +25,7 @@ _DEFAULT_ARROW_CONVERTERS: dict[str, Callable[[str | None], Any | None]] = { "date": _to_date, "time": _to_time, + "time with time zone": _to_time_with_tz, "decimal": _to_decimal, "varbinary": _to_binary, "json": _csv_to_json, @@ -85,6 +87,7 @@ def _dtypes(self) -> dict[str, type[Any]]: "timestamp": pa.timestamp("ms"), "date": pa.timestamp("ms"), "time": pa.string(), + "time with time zone": pa.string(), "varbinary": pa.string(), "array": pa.string(), "map": pa.string(), diff --git a/pyathena/arrow/result_set.py b/pyathena/arrow/result_set.py index 1acd654b7..94a8597c4 100644 --- a/pyathena/arrow/result_set.py +++ b/pyathena/arrow/result_set.py @@ -12,7 +12,7 @@ from pyathena import OperationalError from pyathena.arrow.util import to_column_info -from pyathena.converter import Converter, _json_text_converter, _to_default +from pyathena.converter import Converter, _text_value_converter, _to_default from pyathena.error import ProgrammingError from pyathena.model import AthenaQueryExecution from pyathena.result_set import AthenaResultSet @@ -147,7 +147,8 @@ def __init__( self._table = pa.Table.from_pydict({}) # The fetch methods convert the values read from a result file. GetQueryResults - # values are already converted, except json values, which stay text. + # values are already converted, except json and time with time zone values, + # which stay text. self._convert_rows = bool(self.output_location) self._batches = iter(self._table.to_batches(arraysize)) @@ -255,7 +256,9 @@ def _fetch(self) -> None: else: dict_rows = rows.to_pydict() converters = ( - self.converters if self._convert_rows else self._json_converters(self.converters) + self.converters + if self._convert_rows + else self._text_value_converters(self.converters) ) if converters: column_names = dict_rows.keys() @@ -267,7 +270,16 @@ def _fetch(self) -> None: for row in zip(*dict_rows.values(), strict=False) ] else: - processed_rows = list(zip(*dict_rows.values(), strict=False)) + converters = { + d[0]: self._converter.get(d[1]) + for d in self.description or [] + if d[1] == "time with time zone" + } + column_converters = [converters.get(k) for k in dict_rows] + processed_rows = [ + tuple(c(v) if c else v for c, v in zip(column_converters, row, strict=False)) + for row in zip(*dict_rows.values(), strict=False) + ] self._rows.extend(processed_rows) @override @@ -387,12 +399,13 @@ def _as_arrow_from_api(self, converter: Converter | None = None) -> Table: Args: converter: Type converter for result values. Defaults to - ``DefaultTypeConverter`` with json values kept as text, as in - the CSV result file. + ``DefaultTypeConverter`` with json and time with time zone values kept as + text, as in the CSV result file. Arrow has no type for JSON values or for + times with a time zone. """ import pyarrow as pa - rows = self._fetch_all_rows(converter or _json_text_converter()) + rows = self._fetch_all_rows(converter or _text_value_converter()) if not rows: return pa.Table.from_pydict({}) description = self.description if self.description else [] diff --git a/pyathena/converter.py b/pyathena/converter.py index eff987421..14dbfaf96 100644 --- a/pyathena/converter.py +++ b/pyathena/converter.py @@ -9,7 +9,7 @@ from abc import ABCMeta, abstractmethod from collections.abc import Callable from copy import deepcopy -from datetime import date, datetime, time +from datetime import date, datetime, time, timedelta, timezone from decimal import Decimal from typing import Any, ClassVar @@ -82,10 +82,55 @@ def _to_datetime_with_tz(varchar_value: str | None) -> datetime | None: return _parse_datetime(datetime_).replace(tzinfo=gettz(tz)) +def _parse_time(value: str) -> time: + """Parse an Athena TIME value of any precision. + + Args: + value: The value as ``HH:MM:SS`` followed by an optional fraction of up to + 12 digits. + + Returns: + The time. Digits beyond microseconds are truncated. + """ + seconds, _, fraction = value.partition(".") + parsed = datetime.strptime(seconds, "%H:%M:%S").time() + if fraction: + parsed = parsed.replace(microsecond=int(fraction[:6].ljust(6, "0"))) + return parsed + + def _to_time(varchar_value: str | None) -> time | None: + """Convert an Athena TIME value to a time. + + Args: + varchar_value: The value as text, or None. + + Returns: + The time, or None. + """ if varchar_value is None: return None - return datetime.strptime(varchar_value, "%H:%M:%S.%f").time() + return _parse_time(varchar_value) + + +def _to_time_with_tz(varchar_value: str | None) -> time | None: + """Convert an Athena TIME WITH TIME ZONE value to an aware time. + + Args: + varchar_value: The value as text with a trailing ``+HH:MM`` or ``-HH:MM`` + offset, or None. An empty string, which pandas passes for NULL, is None. + + Returns: + The time with a fixed-offset ``tzinfo``, or None. + """ + if not varchar_value: + return None + index = max(varchar_value.rfind("+"), varchar_value.rfind("-")) + hours, _, minutes = varchar_value[index + 1 :].partition(":") + offset = timedelta(hours=int(hours), minutes=int(minutes)) + if varchar_value[index] == "-": + offset = -offset + return _parse_time(varchar_value[:index]).replace(tzinfo=timezone(offset)) def _to_float(varchar_value: str | None) -> float | None: @@ -494,6 +539,7 @@ def _to_default(varchar_value: str | None) -> str | None: "timestamp with time zone": _to_datetime_with_tz, "date": _to_date, "time": _to_time, + "time with time zone": _to_time_with_tz, "varbinary": _to_binary, "array": _to_array, "map": _to_map, @@ -744,12 +790,18 @@ def _parse_type_hint(self, type_hint: str) -> TypeNode: return self._parsed_hints[normalized] -def _json_text_converter() -> DefaultTypeConverter: - """Return a ``DefaultTypeConverter`` that keeps json values as text. +# The types whose values the Arrow and Polars GetQueryResults fallbacks keep as text, +# as in a CSV result file, and convert when the rows are fetched. +_TEXT_VALUE_TYPES: tuple[str, ...] = ("json", "time with time zone") + + +def _text_value_converter() -> DefaultTypeConverter: + """Return a ``DefaultTypeConverter`` that keeps ``_TEXT_VALUE_TYPES`` values as text. Returns: The converter. """ converter = DefaultTypeConverter() - converter.set("json", _to_default) + for type_ in _TEXT_VALUE_TYPES: + converter.set(type_, _to_default) return converter diff --git a/pyathena/pandas/converter.py b/pyathena/pandas/converter.py index a3b207e90..ee964ebb1 100644 --- a/pyathena/pandas/converter.py +++ b/pyathena/pandas/converter.py @@ -14,6 +14,7 @@ _to_boolean, _to_decimal, _to_default, + _to_time_with_tz, ) from pyathena.util import override @@ -25,6 +26,7 @@ "decimal": _to_decimal, "varbinary": _to_binary, "json": _csv_to_json, + "time with time zone": _to_time_with_tz, } diff --git a/pyathena/pandas/result_set.py b/pyathena/pandas/result_set.py index 638642723..212bd5baf 100644 --- a/pyathena/pandas/result_set.py +++ b/pyathena/pandas/result_set.py @@ -257,7 +257,6 @@ class AthenaPandasResultSet(AthenaResultSet): _PARSE_DATES: ClassVar[list[str]] = [ "date", "time", - "time with time zone", "timestamp", "timestamp with time zone", ] @@ -343,9 +342,7 @@ def __init__( # Cache time column names for efficient _trunc_date processing description = self.description if self.description else [] - self._time_columns: list[str] = [ - d[0] for d in description if d[1] in ("time", "time with time zone") - ] + self._time_columns: list[str] = [d[0] for d in description if d[1] == "time"] import pandas as pd diff --git a/pyathena/polars/converter.py b/pyathena/polars/converter.py index d0b2caf6f..62300a3a1 100644 --- a/pyathena/polars/converter.py +++ b/pyathena/polars/converter.py @@ -21,6 +21,7 @@ _to_default, _to_json, _to_time, + _to_time_with_tz, ) from pyathena.util import override @@ -30,6 +31,7 @@ _DEFAULT_POLARS_CONVERTERS: dict[str, Callable[[str | None], Any | None]] = { "date": _to_date, "time": _to_time, + "time with time zone": _to_time_with_tz, "varbinary": _to_binary, "json": _to_json, } @@ -89,6 +91,7 @@ def _dtypes(self) -> dict[str, Any]: "timestamp": pl.Datetime, "date": pl.Date, "time": pl.String, + "time with time zone": pl.String, "varbinary": pl.String, "array": pl.String, "map": pl.String, diff --git a/pyathena/polars/result_set.py b/pyathena/polars/result_set.py index 4641614ab..7f9bea749 100644 --- a/pyathena/polars/result_set.py +++ b/pyathena/polars/result_set.py @@ -20,7 +20,7 @@ ) from pyathena import OperationalError -from pyathena.converter import Converter, _json_text_converter +from pyathena.converter import Converter, _text_value_converter from pyathena.error import ProgrammingError from pyathena.model import AthenaQueryExecution from pyathena.polars.util import to_column_info @@ -277,8 +277,9 @@ def __init__( self._df_iter = self._create_dataframe_iterator() elif self.state == AthenaQueryExecution.STATE_SUCCEEDED: self._df = self._as_polars_from_api() - # GetQueryResults values are already converted, except json values kept as text. - self._df_converters = self._json_converters(self.converters) + # GetQueryResults values are already converted, except json and time with + # time zone values kept as text. + self._df_converters = self._text_value_converters(self.converters) else: self._df = pl.DataFrame() if self._df is not None: @@ -582,12 +583,12 @@ def _as_polars_from_api(self, converter: Converter | None = None) -> pl.DataFram Args: converter: Type converter for result values. Defaults to - ``DefaultTypeConverter`` with json values kept as text, as in - the CSV result file. + ``DefaultTypeConverter`` with json and time with time zone values kept as + text, as in the CSV result file. A Polars ``Time`` has no time zone. """ import polars as pl - rows = self._fetch_all_rows(converter or _json_text_converter()) + rows = self._fetch_all_rows(converter or _text_value_converter()) if not rows: return pl.DataFrame() description = self.description if self.description else [] diff --git a/pyathena/result_set.py b/pyathena/result_set.py index 65842761e..9ecc91876 100644 --- a/pyathena/result_set.py +++ b/pyathena/result_set.py @@ -12,7 +12,7 @@ ) from pyathena.common import BaseCursor, CursorIterator -from pyathena.converter import Converter, DefaultTypeConverter +from pyathena.converter import _TEXT_VALUE_TYPES, Converter, DefaultTypeConverter from pyathena.error import DataError, OperationalError, ProgrammingError from pyathena.model import AthenaQueryExecution from pyathena.util import RetryConfig, override, parse_output_location, retry_api_call @@ -681,19 +681,19 @@ def _is_first_row_column_labels(self, rows: list[dict[str, Any]]) -> bool: return False return True - def _json_converters( + def _text_value_converters( self, converters: dict[str, Callable[[str | None], Any | None]] ) -> dict[str, Callable[[str | None], Any | None]]: - """Select the converters of the json columns. + """Select the converters of the columns that the fallbacks keep as text. Args: converters: The converters keyed by column name. Returns: - The converters of the columns whose Athena type is json. + The converters of the columns whose Athena type is in ``_TEXT_VALUE_TYPES``. """ description = self.description if self.description else [] - return {d[0]: converters[d[0]] for d in description if d[1] == "json"} + return {d[0]: converters[d[0]] for d in description if d[1] in _TEXT_VALUE_TYPES} def _fetch_all_rows( self, diff --git a/tests/pyathena/arrow/test_cursor.py b/tests/pyathena/arrow/test_cursor.py index fd3f04b42..4dc0523b9 100644 --- a/tests/pyathena/arrow/test_cursor.py +++ b/tests/pyathena/arrow/test_cursor.py @@ -26,6 +26,7 @@ from pyathena.util import RetryConfig from tests import ENV from tests.pyathena.conftest import connect +from tests.pyathena.util import TIME_VALUES_QUERY, TIME_VALUES_ROW class TestArrowCursor: @@ -1009,6 +1010,11 @@ def test_fetch_all_rows(self, arrow_cursor): ] assert arrow_cursor.as_arrow().schema.field("col_json").type == pa.string() + arrow_cursor.execute(TIME_VALUES_QUERY) + assert arrow_cursor.fetchall() == [TIME_VALUES_ROW] + # An Arrow time type has no time zone, so the table keeps the text. + assert arrow_cursor.as_arrow().column("col_time_tz").to_pylist() == ["12:34:56.789+09:00"] + @pytest.mark.parametrize( "execute_kwargs", [{}, {"connect_timeout": 3.0, "request_timeout": 4.0}] ) diff --git a/tests/pyathena/pandas/test_cursor.py b/tests/pyathena/pandas/test_cursor.py index 7c646ded6..7fe790a1f 100644 --- a/tests/pyathena/pandas/test_cursor.py +++ b/tests/pyathena/pandas/test_cursor.py @@ -23,7 +23,7 @@ from pyathena.util import RetryConfig from tests import ENV from tests.pyathena.conftest import connect -from tests.pyathena.util import cached_file_systems +from tests.pyathena.util import TIME_VALUES_QUERY, TIME_VALUES_ROW, cached_file_systems class TestPandasCursor: @@ -1617,6 +1617,9 @@ def test_fetch_all_rows(self, pandas_cursor): (2, datetime(2017, 1, 1, 12, 34, 56).time(), b"\x00\x01", [1, "x"], "s", None), ] + pandas_cursor.execute(TIME_VALUES_QUERY) + assert pandas_cursor.fetchall() == [TIME_VALUES_ROW] + @pytest.mark.parametrize( "execute_kwargs", [ diff --git a/tests/pyathena/polars/test_cursor.py b/tests/pyathena/polars/test_cursor.py index 7988a29c5..f6d1fee93 100644 --- a/tests/pyathena/polars/test_cursor.py +++ b/tests/pyathena/polars/test_cursor.py @@ -24,7 +24,7 @@ from pyathena.util import RetryConfig from tests import ENV from tests.pyathena.conftest import connect -from tests.pyathena.util import cached_file_systems +from tests.pyathena.util import TIME_VALUES_QUERY, TIME_VALUES_ROW, cached_file_systems class TestPolarsCursor: @@ -779,6 +779,11 @@ def test_fetch_all_rows(self, polars_cursor): ] assert polars_cursor.as_polars()["col_json"].dtype == pl.String + polars_cursor.execute(TIME_VALUES_QUERY) + assert polars_cursor.fetchall() == [TIME_VALUES_ROW] + # A Polars Time has no time zone, so the DataFrame keeps the text. + assert polars_cursor.as_polars()["col_time_tz"].to_list() == ["12:34:56.789+09:00"] + @pytest.mark.parametrize( "execute_kwargs", [{}, {"block_size": 2048, "cache_type": "none", "max_workers": 3, "chunksize": 20}], diff --git a/tests/pyathena/s3fs/test_cursor.py b/tests/pyathena/s3fs/test_cursor.py index 4ebaa4130..f7b8abefd 100644 --- a/tests/pyathena/s3fs/test_cursor.py +++ b/tests/pyathena/s3fs/test_cursor.py @@ -19,7 +19,7 @@ from pyathena.util import RetryConfig from tests import ENV from tests.pyathena.conftest import connect -from tests.pyathena.util import cached_file_systems +from tests.pyathena.util import TIME_VALUES_QUERY, TIME_VALUES_ROW, cached_file_systems def _s3fs_converter_with_json_text(): @@ -537,8 +537,8 @@ def test_quoted_string_with_comma(self, csv_reader_class): indirect=["s3fs_cursor"], ) def test_fetch_all_rows(self, s3fs_cursor): - s3fs_cursor.execute("SELECT 1 AS col") - assert s3fs_cursor.fetchall() == [(1,)] + s3fs_cursor.execute(TIME_VALUES_QUERY) + assert s3fs_cursor.fetchall() == [TIME_VALUES_ROW] @pytest.mark.parametrize( "s3fs_cursor", diff --git a/tests/pyathena/test_converter.py b/tests/pyathena/test_converter.py index 5d7ad4012..78e4493c4 100644 --- a/tests/pyathena/test_converter.py +++ b/tests/pyathena/test_converter.py @@ -1,4 +1,4 @@ -from datetime import datetime +from datetime import datetime, time, timedelta, timezone import pytest from dateutil.tz import gettz @@ -11,6 +11,8 @@ _to_datetime_with_tz, _to_map, _to_struct, + _to_time, + _to_time_with_tz, ) @@ -590,3 +592,39 @@ def test_normalize_hive_syntax_mixed(self): type_hint="array", ) assert result == [{"a": 1, "b": "hello"}] + + +@pytest.mark.parametrize( + ("input_value", "expected"), + [ + (None, None), + ("12:34:56", time(12, 34, 56)), + ("12:34:56.1", time(12, 34, 56, 100000)), + ("12:34:56.123", time(12, 34, 56, 123000)), + ("12:34:56.123456", time(12, 34, 56, 123456)), + ("12:34:56.123456789012", time(12, 34, 56, 123456)), + ], +) +def test_to_time_any_precision(input_value, expected): + assert _to_time(input_value) == expected + + +@pytest.mark.parametrize( + ("input_value", "expected"), + [ + (None, None), + ("", None), + ("12:34:56+09:00", time(12, 34, 56, tzinfo=timezone(timedelta(hours=9)))), + ( + "12:34:56.789-05:30", + time(12, 34, 56, 789000, tzinfo=timezone(-timedelta(hours=5, minutes=30))), + ), + ("00:00:00.123456789012+00:00", time(0, 0, 0, 123456, tzinfo=timezone(timedelta(0)))), + ("23:59:59.9-14:00", time(23, 59, 59, 900000, tzinfo=timezone(-timedelta(hours=14)))), + ], +) +def test_to_time_with_tz(input_value, expected): + result = _to_time_with_tz(input_value) + assert result == expected + if expected is not None: + assert result.utcoffset() == expected.utcoffset() diff --git a/tests/pyathena/test_cursor.py b/tests/pyathena/test_cursor.py index f4a6b5528..c66cdce6b 100644 --- a/tests/pyathena/test_cursor.py +++ b/tests/pyathena/test_cursor.py @@ -42,6 +42,8 @@ from tests.pyathena.tables import TABLES, VIEWS from tests.pyathena.util import ( EVENT_TIMEOUT, + TIME_VALUES_QUERY, + TIME_VALUES_ROW, interrupt_start_waits, succeeded_query_execution, throttle_metadata_api, @@ -1688,8 +1690,8 @@ def test_null_vs_empty_string(self, cursor): indirect=["cursor"], ) def test_fetch_all_rows(self, cursor): - cursor.execute("SELECT 1 AS col") - assert cursor.fetchall() == [(1,)] + cursor.execute(TIME_VALUES_QUERY) + assert cursor.fetchall() == [TIME_VALUES_ROW] @staticmethod def _metadata_view(metadata): diff --git a/tests/pyathena/util.py b/tests/pyathena/util.py index b2e0cf400..467fb0d98 100644 --- a/tests/pyathena/util.py +++ b/tests/pyathena/util.py @@ -7,6 +7,7 @@ import time from concurrent.futures import wait +from datetime import datetime, timedelta, timezone from pathlib import Path from unittest.mock import patch @@ -107,6 +108,27 @@ def unreachable_glue(connection): ) +# TIME values of several precisions, with and without a time zone, and the row that +# every cursor should fetch for them. +TIME_VALUES_QUERY = """ +SELECT + 1 AS col + ,CAST('12:34:56' AS TIME(0)) AS col_time_0 + ,CAST('12:34:56.123456789' AS TIME(9)) AS col_time_9 + ,CAST('12:34:56.789 +09:00' AS TIME WITH TIME ZONE) AS col_time_tz + ,CAST('12:34:56 -05:30' AS TIME(0) WITH TIME ZONE) AS col_time_tz_0 + ,CAST(NULL AS TIME WITH TIME ZONE) AS col_time_tz_null +""" +TIME_VALUES_ROW = ( + 1, + datetime(2000, 1, 1, 12, 34, 56).time(), + datetime(2000, 1, 1, 12, 34, 56, 123456).time(), + datetime(2000, 1, 1, 12, 34, 56, 789000, tzinfo=timezone(timedelta(hours=9))).timetz(), + datetime(2000, 1, 1, 12, 34, 56, tzinfo=timezone(-timedelta(hours=5, minutes=30))).timetz(), + None, +) + + # A Spark job whose executor tasks sleep, so that StopCalculationExecution can cancel it. CANCELABLE_SPARK_JOB = """ import time From b1b9c4c4cf8075b37c6b53cbcfe4d6e9a66c352b Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sat, 3 Oct 2026 23:37:20 +0900 Subject: [PATCH 2/9] Treat an empty TIME value as NULL The Arrow CSV reader returns an empty string for a NULL in a column that it reads as text, so ArrowCursor raised ValueError for a NULL TIME. An empty string is now None, as for DECIMAL and TIME WITH TIME ZONE. Co-Authored-By: Claude Opus 5.5 --- pyathena/converter.py | 8 +++++--- tests/pyathena/arrow/test_cursor.py | 16 +++++++++++++--- tests/pyathena/test_converter.py | 1 + 3 files changed, 19 insertions(+), 6 deletions(-) diff --git a/pyathena/converter.py b/pyathena/converter.py index 14dbfaf96..63395cd91 100644 --- a/pyathena/converter.py +++ b/pyathena/converter.py @@ -103,12 +103,13 @@ def _to_time(varchar_value: str | None) -> time | None: """Convert an Athena TIME value to a time. Args: - varchar_value: The value as text, or None. + varchar_value: The value as text, or None. An empty string, which the Arrow + CSV reader returns for NULL, is None. Returns: The time, or None. """ - if varchar_value is None: + if not varchar_value: return None return _parse_time(varchar_value) @@ -118,7 +119,8 @@ def _to_time_with_tz(varchar_value: str | None) -> time | None: Args: varchar_value: The value as text with a trailing ``+HH:MM`` or ``-HH:MM`` - offset, or None. An empty string, which pandas passes for NULL, is None. + offset, or None. An empty string, which pandas and the Arrow CSV reader + return for NULL, is None. Returns: The time with a fixed-offset ``tzinfo``, or None. diff --git a/tests/pyathena/arrow/test_cursor.py b/tests/pyathena/arrow/test_cursor.py index 4dc0523b9..df323d470 100644 --- a/tests/pyathena/arrow/test_cursor.py +++ b/tests/pyathena/arrow/test_cursor.py @@ -998,15 +998,25 @@ def test_fetch_all_rows(self, arrow_cursor): ,json_parse('{"a": 1}') AS col_json ,CAST('{"a": 1}' AS JSON) AS col_json_string ,CAST(NULL AS JSON) AS col_json_null + ,CAST(NULL AS TIME) AS col_time_null UNION ALL SELECT - 2, CAST('12:34:56' AS TIME), X'0102', json_parse('[1, "x"]'), json_parse('"s"'), NULL + 2, CAST('12:34:56' AS TIME), X'0102', json_parse('[1, "x"]'), json_parse('"s"'), NULL, + NULL ORDER BY col """ ) assert arrow_cursor.fetchall() == [ - (1, datetime(2017, 1, 1, 12, 34, 56).time(), b"\x01\x02", {"a": 1}, '{"a": 1}', None), - (2, datetime(2017, 1, 1, 12, 34, 56).time(), b"\x01\x02", [1, "x"], "s", None), + ( + 1, + datetime(2017, 1, 1, 12, 34, 56).time(), + b"\x01\x02", + {"a": 1}, + '{"a": 1}', + None, + None, + ), + (2, datetime(2017, 1, 1, 12, 34, 56).time(), b"\x01\x02", [1, "x"], "s", None, None), ] assert arrow_cursor.as_arrow().schema.field("col_json").type == pa.string() diff --git a/tests/pyathena/test_converter.py b/tests/pyathena/test_converter.py index 78e4493c4..ffeb74cab 100644 --- a/tests/pyathena/test_converter.py +++ b/tests/pyathena/test_converter.py @@ -598,6 +598,7 @@ def test_normalize_hive_syntax_mixed(self): ("input_value", "expected"), [ (None, None), + ("", None), ("12:34:56", time(12, 34, 56)), ("12:34:56.1", time(12, 34, 56, 100000)), ("12:34:56.123", time(12, 34, 56, 123000)), From 824e903639f32b3fc5d1ea772cbe451f372049b8 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sat, 3 Oct 2026 23:53:01 +0900 Subject: [PATCH 3/9] Treat an empty JSON value as NULL pandas passes an empty string to its read_csv converters for a NULL, and the Arrow CSV reader returns one, so a NULL JSON value made PandasCursor fail in execute() and ArrowCursor fail in the fetch methods. Athena never returns empty JSON text, so an empty string is now None. Co-Authored-By: Claude Opus 5.5 --- pyathena/converter.py | 12 +++++++++++- tests/pyathena/arrow/test_cursor.py | 6 +++--- tests/pyathena/pandas/test_cursor.py | 6 +++--- tests/pyathena/polars/test_cursor.py | 6 +++--- tests/pyathena/s3fs/test_cursor.py | 6 +++--- tests/pyathena/test_converter.py | 16 ++++++++++++++++ tests/pyathena/test_cursor.py | 8 ++++---- tests/pyathena/util.py | 10 ++++++---- 8 files changed, 49 insertions(+), 21 deletions(-) diff --git a/pyathena/converter.py b/pyathena/converter.py index 63395cd91..acb9a2752 100644 --- a/pyathena/converter.py +++ b/pyathena/converter.py @@ -166,7 +166,17 @@ def _to_binary(varchar_value: str | None) -> bytes | None: def _to_json(varchar_value: str | None) -> Any | None: - if varchar_value is None: + """Convert an Athena JSON value to a Python value. + + Args: + varchar_value: The JSON text, or None. An empty string, which pandas and the + Arrow CSV reader return for NULL, is None; Athena never returns empty JSON + text. + + Returns: + The decoded value, or None. + """ + if not varchar_value: return None return json.loads(varchar_value) diff --git a/tests/pyathena/arrow/test_cursor.py b/tests/pyathena/arrow/test_cursor.py index df323d470..b74b9a458 100644 --- a/tests/pyathena/arrow/test_cursor.py +++ b/tests/pyathena/arrow/test_cursor.py @@ -26,7 +26,7 @@ from pyathena.util import RetryConfig from tests import ENV from tests.pyathena.conftest import connect -from tests.pyathena.util import TIME_VALUES_QUERY, TIME_VALUES_ROW +from tests.pyathena.util import CONVERTED_VALUES_QUERY, CONVERTED_VALUES_ROW class TestArrowCursor: @@ -1020,8 +1020,8 @@ def test_fetch_all_rows(self, arrow_cursor): ] assert arrow_cursor.as_arrow().schema.field("col_json").type == pa.string() - arrow_cursor.execute(TIME_VALUES_QUERY) - assert arrow_cursor.fetchall() == [TIME_VALUES_ROW] + arrow_cursor.execute(CONVERTED_VALUES_QUERY) + assert arrow_cursor.fetchall() == [CONVERTED_VALUES_ROW] # An Arrow time type has no time zone, so the table keeps the text. assert arrow_cursor.as_arrow().column("col_time_tz").to_pylist() == ["12:34:56.789+09:00"] diff --git a/tests/pyathena/pandas/test_cursor.py b/tests/pyathena/pandas/test_cursor.py index 7fe790a1f..a69165361 100644 --- a/tests/pyathena/pandas/test_cursor.py +++ b/tests/pyathena/pandas/test_cursor.py @@ -23,7 +23,7 @@ from pyathena.util import RetryConfig from tests import ENV from tests.pyathena.conftest import connect -from tests.pyathena.util import TIME_VALUES_QUERY, TIME_VALUES_ROW, cached_file_systems +from tests.pyathena.util import CONVERTED_VALUES_QUERY, CONVERTED_VALUES_ROW, cached_file_systems class TestPandasCursor: @@ -1617,8 +1617,8 @@ def test_fetch_all_rows(self, pandas_cursor): (2, datetime(2017, 1, 1, 12, 34, 56).time(), b"\x00\x01", [1, "x"], "s", None), ] - pandas_cursor.execute(TIME_VALUES_QUERY) - assert pandas_cursor.fetchall() == [TIME_VALUES_ROW] + pandas_cursor.execute(CONVERTED_VALUES_QUERY) + assert pandas_cursor.fetchall() == [CONVERTED_VALUES_ROW] @pytest.mark.parametrize( "execute_kwargs", diff --git a/tests/pyathena/polars/test_cursor.py b/tests/pyathena/polars/test_cursor.py index f6d1fee93..f719a612d 100644 --- a/tests/pyathena/polars/test_cursor.py +++ b/tests/pyathena/polars/test_cursor.py @@ -24,7 +24,7 @@ from pyathena.util import RetryConfig from tests import ENV from tests.pyathena.conftest import connect -from tests.pyathena.util import TIME_VALUES_QUERY, TIME_VALUES_ROW, cached_file_systems +from tests.pyathena.util import CONVERTED_VALUES_QUERY, CONVERTED_VALUES_ROW, cached_file_systems class TestPolarsCursor: @@ -779,8 +779,8 @@ def test_fetch_all_rows(self, polars_cursor): ] assert polars_cursor.as_polars()["col_json"].dtype == pl.String - polars_cursor.execute(TIME_VALUES_QUERY) - assert polars_cursor.fetchall() == [TIME_VALUES_ROW] + polars_cursor.execute(CONVERTED_VALUES_QUERY) + assert polars_cursor.fetchall() == [CONVERTED_VALUES_ROW] # A Polars Time has no time zone, so the DataFrame keeps the text. assert polars_cursor.as_polars()["col_time_tz"].to_list() == ["12:34:56.789+09:00"] diff --git a/tests/pyathena/s3fs/test_cursor.py b/tests/pyathena/s3fs/test_cursor.py index f7b8abefd..4b643d5b0 100644 --- a/tests/pyathena/s3fs/test_cursor.py +++ b/tests/pyathena/s3fs/test_cursor.py @@ -19,7 +19,7 @@ from pyathena.util import RetryConfig from tests import ENV from tests.pyathena.conftest import connect -from tests.pyathena.util import TIME_VALUES_QUERY, TIME_VALUES_ROW, cached_file_systems +from tests.pyathena.util import CONVERTED_VALUES_QUERY, CONVERTED_VALUES_ROW, cached_file_systems def _s3fs_converter_with_json_text(): @@ -537,8 +537,8 @@ def test_quoted_string_with_comma(self, csv_reader_class): indirect=["s3fs_cursor"], ) def test_fetch_all_rows(self, s3fs_cursor): - s3fs_cursor.execute(TIME_VALUES_QUERY) - assert s3fs_cursor.fetchall() == [TIME_VALUES_ROW] + s3fs_cursor.execute(CONVERTED_VALUES_QUERY) + assert s3fs_cursor.fetchall() == [CONVERTED_VALUES_ROW] @pytest.mark.parametrize( "s3fs_cursor", diff --git a/tests/pyathena/test_converter.py b/tests/pyathena/test_converter.py index ffeb74cab..776872faa 100644 --- a/tests/pyathena/test_converter.py +++ b/tests/pyathena/test_converter.py @@ -9,6 +9,7 @@ _to_array, _to_datetime, _to_datetime_with_tz, + _to_json, _to_map, _to_struct, _to_time, @@ -629,3 +630,18 @@ def test_to_time_with_tz(input_value, expected): assert result == expected if expected is not None: assert result.utcoffset() == expected.utcoffset() + + +@pytest.mark.parametrize( + ("input_value", "expected"), + [ + (None, None), + ("", None), + ('""', ""), + ('{"a": 1}', {"a": 1}), + ("[1, 2]", [1, 2]), + ("null", None), + ], +) +def test_to_json(input_value, expected): + assert _to_json(input_value) == expected diff --git a/tests/pyathena/test_cursor.py b/tests/pyathena/test_cursor.py index c66cdce6b..6b012f7f7 100644 --- a/tests/pyathena/test_cursor.py +++ b/tests/pyathena/test_cursor.py @@ -41,9 +41,9 @@ from tests.pyathena.conftest import connect from tests.pyathena.tables import TABLES, VIEWS from tests.pyathena.util import ( + CONVERTED_VALUES_QUERY, + CONVERTED_VALUES_ROW, EVENT_TIMEOUT, - TIME_VALUES_QUERY, - TIME_VALUES_ROW, interrupt_start_waits, succeeded_query_execution, throttle_metadata_api, @@ -1690,8 +1690,8 @@ def test_null_vs_empty_string(self, cursor): indirect=["cursor"], ) def test_fetch_all_rows(self, cursor): - cursor.execute(TIME_VALUES_QUERY) - assert cursor.fetchall() == [TIME_VALUES_ROW] + cursor.execute(CONVERTED_VALUES_QUERY) + assert cursor.fetchall() == [CONVERTED_VALUES_ROW] @staticmethod def _metadata_view(metadata): diff --git a/tests/pyathena/util.py b/tests/pyathena/util.py index 467fb0d98..7189d525e 100644 --- a/tests/pyathena/util.py +++ b/tests/pyathena/util.py @@ -108,9 +108,9 @@ def unreachable_glue(connection): ) -# TIME values of several precisions, with and without a time zone, and the row that -# every cursor should fetch for them. -TIME_VALUES_QUERY = """ +# TIME values of several precisions, with and without a time zone, a NULL JSON value, +# and the row that every cursor should fetch for them. +CONVERTED_VALUES_QUERY = """ SELECT 1 AS col ,CAST('12:34:56' AS TIME(0)) AS col_time_0 @@ -118,14 +118,16 @@ def unreachable_glue(connection): ,CAST('12:34:56.789 +09:00' AS TIME WITH TIME ZONE) AS col_time_tz ,CAST('12:34:56 -05:30' AS TIME(0) WITH TIME ZONE) AS col_time_tz_0 ,CAST(NULL AS TIME WITH TIME ZONE) AS col_time_tz_null + ,CAST(NULL AS JSON) AS col_json_null """ -TIME_VALUES_ROW = ( +CONVERTED_VALUES_ROW = ( 1, datetime(2000, 1, 1, 12, 34, 56).time(), datetime(2000, 1, 1, 12, 34, 56, 123456).time(), datetime(2000, 1, 1, 12, 34, 56, 789000, tzinfo=timezone(timedelta(hours=9))).timetz(), datetime(2000, 1, 1, 12, 34, 56, tzinfo=timezone(-timedelta(hours=5, minutes=30))).timetz(), None, + None, ) From f26016df089b47432dec1bbb85426469bec7f181 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 00:00:27 +0900 Subject: [PATCH 4/9] Keep empty JSON strings inside typed complex values Treating an empty string as NULL in _to_json() also turned an empty JSON string inside a typed complex value (for example array(json) '[""]') into None, because the complex-type parser passes the decoded element to the converter and relied on the decode error to fall back to the native format. _to_json() keeps its previous behavior, and the pandas and Arrow converters, which read CSV result files, use the new _to_json_from_file() that treats an empty string as NULL. Co-Authored-By: Claude Opus 5.5 --- pyathena/converter.py | 6 ++---- tests/pyathena/test_converter.py | 23 ++++++++++++++++++++++- 2 files changed, 24 insertions(+), 5 deletions(-) diff --git a/pyathena/converter.py b/pyathena/converter.py index acb9a2752..540ea70ec 100644 --- a/pyathena/converter.py +++ b/pyathena/converter.py @@ -169,14 +169,12 @@ def _to_json(varchar_value: str | None) -> Any | None: """Convert an Athena JSON value to a Python value. Args: - varchar_value: The JSON text, or None. An empty string, which pandas and the - Arrow CSV reader return for NULL, is None; Athena never returns empty JSON - text. + varchar_value: The JSON text, or None. Returns: The decoded value, or None. """ - if not varchar_value: + if varchar_value is None: return None return json.loads(varchar_value) diff --git a/tests/pyathena/test_converter.py b/tests/pyathena/test_converter.py index 776872faa..135065792 100644 --- a/tests/pyathena/test_converter.py +++ b/tests/pyathena/test_converter.py @@ -1,3 +1,4 @@ +import json from datetime import datetime, time, timedelta, timezone import pytest @@ -636,7 +637,6 @@ def test_to_time_with_tz(input_value, expected): ("input_value", "expected"), [ (None, None), - ("", None), ('""', ""), ('{"a": 1}', {"a": 1}), ("[1, 2]", [1, 2]), @@ -645,3 +645,24 @@ def test_to_time_with_tz(input_value, expected): ) def test_to_json(input_value, expected): assert _to_json(input_value) == expected + assert _csv_to_json(input_value) == expected + + +def test_to_json_empty_string(): + """Only the result-file converter treats an empty string as NULL.""" + assert _csv_to_json("") is None + with pytest.raises(json.JSONDecodeError): + _to_json("") + + +@pytest.mark.parametrize( + ("type_hint", "value", "expected"), + [ + ("array(json)", '[""]', [""]), + ("map(varchar,json)", '{"k": ""}', {"k": ""}), + ], +) +def test_typed_json_empty_string_element(type_hint, value, expected): + """An empty JSON string inside a typed complex value stays an empty string.""" + type_ = type_hint.split("(", 1)[0] + assert DefaultTypeConverter().convert(type_, value, type_hint=type_hint) == expected From 82492b231ea42ce99be9574b10f5501b22fc50aa Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 00:37:06 +0900 Subject: [PATCH 5/9] Decode JSON elements of typed complex values from their JSON text The typed complex-value parser turned a decoded JSON element back into text with str(), so a JSON string element lost its quotes. It relied on the JSON decode error to fall back to the native parser, decoded string scalars that looked like JSON a second time, and raised for row(json) fields. JSON-typed elements are now encoded with json.dumps(), which lets _to_json() treat an empty string as NULL like the other converters, so the separate _to_json_from_file() is gone. Co-Authored-By: Claude Opus 5.5 --- pyathena/arrow/converter.py | 4 +-- pyathena/converter.py | 19 ++---------- pyathena/pandas/converter.py | 4 +-- pyathena/parser.py | 22 +++++++++----- tests/pyathena/test_converter.py | 52 ++++++++++++++------------------ 5 files changed, 44 insertions(+), 57 deletions(-) diff --git a/pyathena/arrow/converter.py b/pyathena/arrow/converter.py index f65cbb85a..0adc49c59 100644 --- a/pyathena/arrow/converter.py +++ b/pyathena/arrow/converter.py @@ -9,11 +9,11 @@ from pyathena.converter import ( Converter, - _csv_to_json, _to_binary, _to_date, _to_decimal, _to_default, + _to_json, _to_time, _to_time_with_tz, ) @@ -28,7 +28,7 @@ "time with time zone": _to_time_with_tz, "decimal": _to_decimal, "varbinary": _to_binary, - "json": _csv_to_json, + "json": _to_json, } diff --git a/pyathena/converter.py b/pyathena/converter.py index 540ea70ec..e00688a9d 100644 --- a/pyathena/converter.py +++ b/pyathena/converter.py @@ -169,22 +169,9 @@ def _to_json(varchar_value: str | None) -> Any | None: """Convert an Athena JSON value to a Python value. Args: - varchar_value: The JSON text, or None. - - Returns: - The decoded value, or None. - """ - if varchar_value is None: - return None - return json.loads(varchar_value) - - -def _csv_to_json(varchar_value: str | None) -> Any | None: - """Convert an Athena JSON value read from a CSV result file. - - Args: - varchar_value: The value as JSON text, or None. An empty string, which - CSV results use for SQL NULL, is also treated as NULL. + varchar_value: The JSON text, or None. An empty string, which pandas and the + Arrow CSV reader return for NULL, is None; Athena never returns empty JSON + text. Returns: The decoded value, or None for SQL NULL. diff --git a/pyathena/pandas/converter.py b/pyathena/pandas/converter.py index ee964ebb1..690ab9e0c 100644 --- a/pyathena/pandas/converter.py +++ b/pyathena/pandas/converter.py @@ -9,11 +9,11 @@ from pyathena.converter import ( Converter, - _csv_to_json, _to_binary, _to_boolean, _to_decimal, _to_default, + _to_json, _to_time_with_tz, ) from pyathena.util import override @@ -25,7 +25,7 @@ "boolean": _to_boolean, "decimal": _to_decimal, "varbinary": _to_binary, - "json": _csv_to_json, + "json": _to_json, "time with time zone": _to_time_with_tz, } diff --git a/pyathena/parser.py b/pyathena/parser.py index 1694cebc3..4535169bd 100644 --- a/pyathena/parser.py +++ b/pyathena/parser.py @@ -268,19 +268,21 @@ def convert(self, value: str, type_node: TypeNode) -> Any: return converter_fn(value) @staticmethod - def _to_json_str(value: Any) -> str: + def _to_json_str(value: Any, type_node: TypeNode) -> str: """Convert a JSON-parsed value back to a string for further conversion. - Uses json.dumps for dict/list to produce valid JSON, and str() for - scalar types to produce converter-compatible strings. + Uses json.dumps for dict/list values and for every value of a JSON type, + so that the JSON converter decodes the original JSON text, and str() for + the other scalar types to produce converter-compatible strings. Args: value: A value from json.loads output. + type_node: The type of the value. Returns: String representation suitable for type conversion. """ - if isinstance(value, (dict, list)): + if isinstance(value, (dict, list)) or type_node.type_name == "json": return json.dumps(value) return str(value) @@ -324,7 +326,7 @@ def _convert_typed_array(self, value: str, type_node: TypeNode) -> list[Any] | N return [ None if elem is None - else self.convert(self._to_json_str(elem), element_type) + else self.convert(self._to_json_str(elem, element_type), element_type) for elem in parsed ] except json.JSONDecodeError: @@ -379,8 +381,12 @@ def _convert_typed_map(self, value: str, type_node: TypeNode) -> dict[str, Any] parsed = json.loads(value) if isinstance(parsed, dict): return { - str(self.convert(self._to_json_str(k), key_type) if k is not None else k): ( - self.convert(self._to_json_str(v), value_type) + str( + self.convert(self._to_json_str(k, key_type), key_type) + if k is not None + else k + ): ( + self.convert(self._to_json_str(v, value_type), value_type) if v is not None else None ) @@ -447,7 +453,7 @@ def _convert_typed_struct(self, value: str, type_node: TypeNode) -> dict[str, An for i, (k, v) in enumerate(parsed.items()): ft = self._get_field_type(k, type_node, i) result[k] = ( - self.convert(self._to_json_str(v), ft) if v is not None else None + self.convert(self._to_json_str(v, ft), ft) if v is not None else None ) return result except json.JSONDecodeError: diff --git a/tests/pyathena/test_converter.py b/tests/pyathena/test_converter.py index 135065792..da5312e88 100644 --- a/tests/pyathena/test_converter.py +++ b/tests/pyathena/test_converter.py @@ -1,4 +1,3 @@ -import json from datetime import datetime, time, timedelta, timezone import pytest @@ -6,7 +5,6 @@ from pyathena.converter import ( DefaultTypeConverter, - _csv_to_json, _to_array, _to_datetime, _to_datetime_with_tz, @@ -330,22 +328,6 @@ def test_to_array_invalid_formats(input_value): assert _to_array(input_value) is None -@pytest.mark.parametrize( - ("input_value", "expected"), - [ - (None, None), - ("", None), - ('{"a":1}', {"a": 1}), - ("[1,2]", [1, 2]), - ('""', ""), - ('"[1, 2]"', "[1, 2]"), - ("null", None), - ], -) -def test_csv_to_json(input_value, expected): - assert _csv_to_json(input_value) == expected - - @pytest.mark.parametrize( ("value", "type_hint", "expected"), [ @@ -637,7 +619,9 @@ def test_to_time_with_tz(input_value, expected): ("input_value", "expected"), [ (None, None), + ("", None), ('""', ""), + ('"[1, 2]"', "[1, 2]"), ('{"a": 1}', {"a": 1}), ("[1, 2]", [1, 2]), ("null", None), @@ -645,24 +629,34 @@ def test_to_time_with_tz(input_value, expected): ) def test_to_json(input_value, expected): assert _to_json(input_value) == expected - assert _csv_to_json(input_value) == expected - - -def test_to_json_empty_string(): - """Only the result-file converter treats an empty string as NULL.""" - assert _csv_to_json("") is None - with pytest.raises(json.JSONDecodeError): - _to_json("") @pytest.mark.parametrize( ("type_hint", "value", "expected"), [ ("array(json)", '[""]', [""]), - ("map(varchar,json)", '{"k": ""}', {"k": ""}), + ( + "array(json)", + '[{"a":1}, "x", 1, true, null, ""]', + [{"a": 1}, "x", 1, True, None, ""], + ), + # JSON string scalars whose text looks like JSON stay strings. + ( + "array(json)", + '["{\\"a\\": 1}", "123", "true", "null"]', + ['{"a": 1}', "123", "true", "null"], + ), + ( + "map(varchar,json)", + '{"k": "", "n": 1, "b": true, "z": null}', + {"k": "", "n": 1, "b": True, "z": None}, + ), + ("row(a json, b json)", '{"a": "x", "b": {"c": 1}}', {"a": "x", "b": {"c": 1}}), + ("row(a json, b json)", '{"a": "", "b": true}', {"a": "", "b": True}), + ("array(varchar)", '["a", "123"]', ["a", "123"]), ], ) -def test_typed_json_empty_string_element(type_hint, value, expected): - """An empty JSON string inside a typed complex value stays an empty string.""" +def test_typed_json_elements(type_hint, value, expected): + """JSON elements of typed complex values decode their original JSON text.""" type_ = type_hint.split("(", 1)[0] assert DefaultTypeConverter().convert(type_, value, type_hint=type_hint) == expected From 61f0be78c0ee6968ae69b378141e587db902d4f6 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 00:53:00 +0900 Subject: [PATCH 6/9] Keep the time zone of time values nested in typed complex values DefaultTypeConverter built its typed complex-value converter from the module-level mappings, so set() did not reach nested values. The Arrow and Polars GetQueryResults fallbacks keep TIME WITH TIME ZONE values as text with set(), but nested ones were still converted and lost their offsets in the Arrow and Polars tables. The typed converter now uses the converter's own mappings. The type signature parser also dropped " with time zone" after a precision, so time(3) with time zone and timestamp(3) with time zone hints used the TIME and TIMESTAMP converters. It now keeps that suffix and still ignores other trailing text. Co-Authored-By: Claude Opus 5.5 --- pyathena/converter.py | 3 ++- pyathena/parser.py | 6 +++++- tests/pyathena/test_converter.py | 26 ++++++++++++++++++++++++++ tests/pyathena/test_parser.py | 12 ++++++++++++ 4 files changed, 45 insertions(+), 2 deletions(-) diff --git a/pyathena/converter.py b/pyathena/converter.py index e00688a9d..7e8786613 100644 --- a/pyathena/converter.py +++ b/pyathena/converter.py @@ -710,8 +710,9 @@ def __init__(self) -> None: """Initialize the converter with the default conversion functions.""" super().__init__(mappings=deepcopy(_DEFAULT_CONVERTERS), default=_to_default) self._parser = TypeSignatureParser() + # Nested values use the same mappings, so set() and remove() apply to them too. self._typed_converter = TypedValueConverter( - converters=_DEFAULT_CONVERTERS, + converters=self.mappings, default_converter=_to_default, struct_parser=_to_struct, ) diff --git a/pyathena/parser.py b/pyathena/parser.py index 4535169bd..120e29980 100644 --- a/pyathena/parser.py +++ b/pyathena/parser.py @@ -141,7 +141,11 @@ def parse(self, type_str: str) -> TypeNode: return TypeNode(type_name=type_name, children=[key_type, value_type]) return TypeNode(type_name=type_name) - # Types with parameters like decimal(10, 2), varchar(255) + # Types with parameters like decimal(10, 2), varchar(255), or + # time(3) with time zone, whose suffix is part of the type name. + # Other trailing text is ignored. + if " ".join(type_str[close_idx + 1 :].lower().split()) == "with time zone": + type_name = f"{type_name} with time zone" return TypeNode(type_name=type_name) def _split_type_args(self, s: str) -> list[str]: diff --git a/tests/pyathena/test_converter.py b/tests/pyathena/test_converter.py index da5312e88..71160b13d 100644 --- a/tests/pyathena/test_converter.py +++ b/tests/pyathena/test_converter.py @@ -8,6 +8,7 @@ _to_array, _to_datetime, _to_datetime_with_tz, + _to_default, _to_json, _to_map, _to_struct, @@ -660,3 +661,28 @@ def test_typed_json_elements(type_hint, value, expected): """JSON elements of typed complex values decode their original JSON text.""" type_ = type_hint.split("(", 1)[0] assert DefaultTypeConverter().convert(type_, value, type_hint=type_hint) == expected + + +def test_typed_time_with_tz_elements(): + """Parameterized time zone types in type hints keep their time zone.""" + converter = DefaultTypeConverter() + jst = timezone(timedelta(hours=9)) + assert converter.convert( + "array", "[12:34:56.789+09:00, null]", type_hint="array(time(3) with time zone)" + ) == [time(12, 34, 56, 789000, tzinfo=jst), None] + assert converter.convert( + "map", "{a=12:34:56+09:00}", type_hint="map(varchar, time(0) with time zone)" + ) == {"a": time(12, 34, 56, tzinfo=jst)} + + +def test_set_applies_to_nested_values(): + """A conversion function set on a converter also converts nested values.""" + converter = DefaultTypeConverter() + converter.set("time with time zone", _to_default) + assert converter.convert( + "array", "[12:34:56.789+09:00]", type_hint="array(time with time zone)" + ) == ["12:34:56.789+09:00"] + # Other converters keep the default mappings. + assert DefaultTypeConverter().convert( + "array", "[12:34:56.789+09:00]", type_hint="array(time with time zone)" + ) == [time(12, 34, 56, 789000, tzinfo=timezone(timedelta(hours=9)))] diff --git a/tests/pyathena/test_parser.py b/tests/pyathena/test_parser.py index 00b03ccaf..dd3b827f1 100644 --- a/tests/pyathena/test_parser.py +++ b/tests/pyathena/test_parser.py @@ -23,6 +23,18 @@ def test_simple_type(self): assert node.children == [] assert node.field_names is None + @pytest.mark.parametrize( + ("type_str", "expected"), + [ + ("time(3) with time zone", "time with time zone"), + ("TIMESTAMP(6) WITH TIME ZONE", "timestamp with time zone"), + ("decimal(10, 2)", "decimal"), + ("varchar(255)", "varchar"), + ], + ) + def test_parameterized_type_keeps_suffix(self, type_str, expected): + assert TypeSignatureParser().parse(type_str).type_name == expected + def test_simple_type_case_insensitive(self): parser = TypeSignatureParser() node = parser.parse("VARCHAR") From 02ebd2764082eb3f38180e775b3bc576b90e5c7e Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 01:04:49 +0900 Subject: [PATCH 7/9] Limit nested time zone text to the Arrow and Polars fallback converter Making the typed converter use the converter's own mappings also kept nested JSON values as text in the Arrow and Polars GetQueryResults fallbacks, which decoded them before. DefaultTypeConverter's typed converter uses the default mappings again, and only the fallback converter, _text_value_converter(), gets a typed converter that keeps nested TIME WITH TIME ZONE values as text. Co-Authored-By: Claude Opus 5.5 --- pyathena/converter.py | 12 ++++++++++-- tests/pyathena/test_converter.py | 12 +++++++----- 2 files changed, 17 insertions(+), 7 deletions(-) diff --git a/pyathena/converter.py b/pyathena/converter.py index 7e8786613..c519b3a45 100644 --- a/pyathena/converter.py +++ b/pyathena/converter.py @@ -710,9 +710,8 @@ def __init__(self) -> None: """Initialize the converter with the default conversion functions.""" super().__init__(mappings=deepcopy(_DEFAULT_CONVERTERS), default=_to_default) self._parser = TypeSignatureParser() - # Nested values use the same mappings, so set() and remove() apply to them too. self._typed_converter = TypedValueConverter( - converters=self.mappings, + converters=_DEFAULT_CONVERTERS, default_converter=_to_default, struct_parser=_to_struct, ) @@ -796,10 +795,19 @@ def _parse_type_hint(self, type_hint: str) -> TypeNode: def _text_value_converter() -> DefaultTypeConverter: """Return a ``DefaultTypeConverter`` that keeps ``_TEXT_VALUE_TYPES`` values as text. + Values nested in typed complex values keep only TIME WITH TIME ZONE values as text, + because Arrow and Polars time types would drop the offset; nested JSON values are + decoded as before. + Returns: The converter. """ converter = DefaultTypeConverter() for type_ in _TEXT_VALUE_TYPES: converter.set(type_, _to_default) + converter._typed_converter = TypedValueConverter( + converters={**_DEFAULT_CONVERTERS, "time with time zone": _to_default}, + default_converter=_to_default, + struct_parser=_to_struct, + ) return converter diff --git a/tests/pyathena/test_converter.py b/tests/pyathena/test_converter.py index 71160b13d..90eb32a67 100644 --- a/tests/pyathena/test_converter.py +++ b/tests/pyathena/test_converter.py @@ -5,10 +5,10 @@ from pyathena.converter import ( DefaultTypeConverter, + _text_value_converter, _to_array, _to_datetime, _to_datetime_with_tz, - _to_default, _to_json, _to_map, _to_struct, @@ -675,13 +675,15 @@ def test_typed_time_with_tz_elements(): ) == {"a": time(12, 34, 56, tzinfo=jst)} -def test_set_applies_to_nested_values(): - """A conversion function set on a converter also converts nested values.""" - converter = DefaultTypeConverter() - converter.set("time with time zone", _to_default) +def test_text_value_converter(): + """The fallback converter keeps text values and nested time zones as text.""" + converter = _text_value_converter() + assert converter.convert("json", '{"a": 1}') == '{"a": 1}' + assert converter.convert("time with time zone", "12:34:56+09:00") == "12:34:56+09:00" assert converter.convert( "array", "[12:34:56.789+09:00]", type_hint="array(time with time zone)" ) == ["12:34:56.789+09:00"] + assert converter.convert("array", '[{"a": 1}]', type_hint="array(json)") == [{"a": 1}] # Other converters keep the default mappings. assert DefaultTypeConverter().convert( "array", "[12:34:56.789+09:00]", type_hint="array(time with time zone)" From d1dee80194f6afa0522368572177988aa522074c Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 01:17:05 +0900 Subject: [PATCH 8/9] Convert Arrow fetch values once The rebase onto #1010 left this branch's earlier conversion code after #1010's, so _fetch() converted every value twice and kept the second result. It now has #1010's single pass with the renamed helper. Co-Authored-By: Claude Opus 5.5 --- pyathena/arrow/result_set.py | 11 +---------- tests/pyathena/arrow/test_cursor.py | 14 ++++++++++++++ 2 files changed, 15 insertions(+), 10 deletions(-) diff --git a/pyathena/arrow/result_set.py b/pyathena/arrow/result_set.py index 94a8597c4..64f650405 100644 --- a/pyathena/arrow/result_set.py +++ b/pyathena/arrow/result_set.py @@ -270,16 +270,7 @@ def _fetch(self) -> None: for row in zip(*dict_rows.values(), strict=False) ] else: - converters = { - d[0]: self._converter.get(d[1]) - for d in self.description or [] - if d[1] == "time with time zone" - } - column_converters = [converters.get(k) for k in dict_rows] - processed_rows = [ - tuple(c(v) if c else v for c, v in zip(column_converters, row, strict=False)) - for row in zip(*dict_rows.values(), strict=False) - ] + processed_rows = list(zip(*dict_rows.values(), strict=False)) self._rows.extend(processed_rows) @override diff --git a/tests/pyathena/arrow/test_cursor.py b/tests/pyathena/arrow/test_cursor.py index b74b9a458..68f62e6a1 100644 --- a/tests/pyathena/arrow/test_cursor.py +++ b/tests/pyathena/arrow/test_cursor.py @@ -19,6 +19,7 @@ import pyarrow as pa import pytest +from pyathena.arrow.converter import DefaultArrowTypeConverter from pyathena.arrow.cursor import ArrowCursor from pyathena.arrow.result_set import AthenaArrowResultSet from pyathena.error import DatabaseError, ProgrammingError @@ -973,6 +974,19 @@ def test_null_vs_empty_string(self, arrow_cursor): assert values[3] == "N/A" assert values[4] == "NULL" + def test_fetch_converts_each_value_once(self): + """The fetch methods call the converter once per value, after a set() too.""" + calls = [] + converter = DefaultArrowTypeConverter() + with ( + contextlib.closing(connect()) as conn, + conn.cursor(ArrowCursor, converter=converter) as cursor, + ): + cursor.execute("SELECT * FROM (VALUES 'a', 'b') AS t(v) ORDER BY v") + converter.set("varchar", lambda value: calls.append(value) or value) + assert cursor.fetchall() == [("a",), ("b",)] + assert calls == ["a", "b"] + @pytest.mark.parametrize( "arrow_cursor", [ From 79fb784848701c674bb537c96624fba9f54614c6 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sun, 4 Oct 2026 01:43:16 +0900 Subject: [PATCH 9/9] Keep the UTC offset of timestamp with time zone values _to_datetime_with_tz() passed the trailing zone to dateutil's gettz(), which returns None for a numeric UTC offset such as +05:30, so Cursor, S3FSCursor, and every GetQueryResults fallback returned a naive datetime. A numeric offset now becomes a fixed-offset time zone; zone names still use gettz(). The other cursors now return aware datetimes for these values as well: the Arrow, Polars, and pandas converters map timestamp with time zone to _to_datetime_with_tz(), PandasCursor no longer parses it as a date (a column with different zones stayed strings), and the Arrow and Polars fallback tables keep it as text, as their result-file tables do, which also avoids a Polars error and a lost Arrow time zone for mixed zones. Reported by @aminghadersohi in #854. Co-Authored-By: Claude Opus 5.5 --- docs/pandas.md | 2 +- pyathena/arrow/converter.py | 3 ++ pyathena/converter.py | 50 ++++++++++++++++++++-------- pyathena/pandas/converter.py | 2 ++ pyathena/pandas/result_set.py | 1 - pyathena/polars/converter.py | 3 ++ tests/pyathena/arrow/test_cursor.py | 3 ++ tests/pyathena/polars/test_cursor.py | 3 ++ tests/pyathena/test_converter.py | 38 +++++++++++++++++++++ tests/pyathena/util.py | 14 ++++++-- 10 files changed, 102 insertions(+), 17 deletions(-) diff --git a/docs/pandas.md b/docs/pandas.md index 580af3bc5..20387867d 100644 --- a/docs/pandas.md +++ b/docs/pandas.md @@ -554,7 +554,7 @@ Common performance options: - `dtype`: Explicit column data types - `parse_dates`: Columns to parse as dates -With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, `json`, and `time with time zone` with the default converter), and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. +With `engine="pyarrow"`, PandasCursor uses the PyArrow engine only when pyarrow is installed, no chunksize is set (explicitly or by `auto_optimize_chunksize`), `quoting` is the default, the result has no columns that need a converter (`boolean`, `decimal`, `varbinary`, `json`, `time with time zone`, and `timestamp with time zone` with the default converter), and the result file is at least `AthenaPandasResultSet.PYARROW_MIN_FILE_SIZE_BYTES` bytes. Otherwise, it falls back to the C engine. Apart from PandasCursor's own options such as `engine` and `chunksize`, an option passed here replaces the value PyAthena sets for the same pandas.read_csv() argument. diff --git a/pyathena/arrow/converter.py b/pyathena/arrow/converter.py index 0adc49c59..7da7bcef3 100644 --- a/pyathena/arrow/converter.py +++ b/pyathena/arrow/converter.py @@ -11,6 +11,7 @@ Converter, _to_binary, _to_date, + _to_datetime_with_tz, _to_decimal, _to_default, _to_json, @@ -26,6 +27,7 @@ "date": _to_date, "time": _to_time, "time with time zone": _to_time_with_tz, + "timestamp with time zone": _to_datetime_with_tz, "decimal": _to_decimal, "varbinary": _to_binary, "json": _to_json, @@ -88,6 +90,7 @@ def _dtypes(self) -> dict[str, type[Any]]: "date": pa.timestamp("ms"), "time": pa.string(), "time with time zone": pa.string(), + "timestamp with time zone": pa.string(), "varbinary": pa.string(), "array": pa.string(), "map": pa.string(), diff --git a/pyathena/converter.py b/pyathena/converter.py index c519b3a45..e573a3657 100644 --- a/pyathena/converter.py +++ b/pyathena/converter.py @@ -67,19 +67,41 @@ def _to_datetime(varchar_value: str | None) -> datetime | None: return _parse_datetime(varchar_value) +_UTC_OFFSET_PATTERN: re.Pattern[str] = re.compile(r"([+-])(\d{2}):(\d{2})") + + +def _parse_utc_offset(value: str) -> timezone | None: + """Parse a ``+HH:MM`` or ``-HH:MM`` UTC offset. + + Args: + value: The text to parse. + + Returns: + The fixed-offset time zone, or None if the text is not an offset. + """ + match = _UTC_OFFSET_PATTERN.fullmatch(value) + if not match: + return None + sign, hours, minutes = match.groups() + offset = timedelta(hours=int(hours), minutes=int(minutes)) + return timezone(-offset if sign == "-" else offset) + + def _to_datetime_with_tz(varchar_value: str | None) -> datetime | None: """Convert an Athena TIMESTAMP WITH TIME ZONE value to an aware datetime. Args: - varchar_value: The value as text with a trailing zone name, or None. + varchar_value: The value as text with a trailing zone name or ``+HH:MM`` / + ``-HH:MM`` UTC offset, or None. An empty string, which pandas and the + Arrow CSV reader return for NULL, is None. Returns: The aware datetime, or None. """ - if varchar_value is None: + if not varchar_value: return None datetime_, _, tz = varchar_value.rpartition(" ") - return _parse_datetime(datetime_).replace(tzinfo=gettz(tz)) + return _parse_datetime(datetime_).replace(tzinfo=_parse_utc_offset(tz) or gettz(tz)) def _parse_time(value: str) -> time: @@ -128,11 +150,9 @@ def _to_time_with_tz(varchar_value: str | None) -> time | None: if not varchar_value: return None index = max(varchar_value.rfind("+"), varchar_value.rfind("-")) - hours, _, minutes = varchar_value[index + 1 :].partition(":") - offset = timedelta(hours=int(hours), minutes=int(minutes)) - if varchar_value[index] == "-": - offset = -offset - return _parse_time(varchar_value[:index]).replace(tzinfo=timezone(offset)) + return _parse_time(varchar_value[:index]).replace( + tzinfo=_parse_utc_offset(varchar_value[index:]) + ) def _to_float(varchar_value: str | None) -> float | None: @@ -789,15 +809,15 @@ def _parse_type_hint(self, type_hint: str) -> TypeNode: # The types whose values the Arrow and Polars GetQueryResults fallbacks keep as text, # as in a CSV result file, and convert when the rows are fetched. -_TEXT_VALUE_TYPES: tuple[str, ...] = ("json", "time with time zone") +_TEXT_VALUE_TYPES: tuple[str, ...] = ("json", "time with time zone", "timestamp with time zone") def _text_value_converter() -> DefaultTypeConverter: """Return a ``DefaultTypeConverter`` that keeps ``_TEXT_VALUE_TYPES`` values as text. - Values nested in typed complex values keep only TIME WITH TIME ZONE values as text, - because Arrow and Polars time types would drop the offset; nested JSON values are - decoded as before. + Values nested in typed complex values keep only the time zone types as text, + because Arrow and Polars time and timestamp types hold one time zone per column; + nested JSON values are decoded as before. Returns: The converter. @@ -806,7 +826,11 @@ def _text_value_converter() -> DefaultTypeConverter: for type_ in _TEXT_VALUE_TYPES: converter.set(type_, _to_default) converter._typed_converter = TypedValueConverter( - converters={**_DEFAULT_CONVERTERS, "time with time zone": _to_default}, + converters={ + **_DEFAULT_CONVERTERS, + "time with time zone": _to_default, + "timestamp with time zone": _to_default, + }, default_converter=_to_default, struct_parser=_to_struct, ) diff --git a/pyathena/pandas/converter.py b/pyathena/pandas/converter.py index 690ab9e0c..1db9b2d7f 100644 --- a/pyathena/pandas/converter.py +++ b/pyathena/pandas/converter.py @@ -11,6 +11,7 @@ Converter, _to_binary, _to_boolean, + _to_datetime_with_tz, _to_decimal, _to_default, _to_json, @@ -27,6 +28,7 @@ "varbinary": _to_binary, "json": _to_json, "time with time zone": _to_time_with_tz, + "timestamp with time zone": _to_datetime_with_tz, } diff --git a/pyathena/pandas/result_set.py b/pyathena/pandas/result_set.py index 212bd5baf..2a78aeef2 100644 --- a/pyathena/pandas/result_set.py +++ b/pyathena/pandas/result_set.py @@ -258,7 +258,6 @@ class AthenaPandasResultSet(AthenaResultSet): "date", "time", "timestamp", - "timestamp with time zone", ] def __init__( diff --git a/pyathena/polars/converter.py b/pyathena/polars/converter.py index 62300a3a1..4d46c702f 100644 --- a/pyathena/polars/converter.py +++ b/pyathena/polars/converter.py @@ -18,6 +18,7 @@ Converter, _to_binary, _to_date, + _to_datetime_with_tz, _to_default, _to_json, _to_time, @@ -32,6 +33,7 @@ "date": _to_date, "time": _to_time, "time with time zone": _to_time_with_tz, + "timestamp with time zone": _to_datetime_with_tz, "varbinary": _to_binary, "json": _to_json, } @@ -92,6 +94,7 @@ def _dtypes(self) -> dict[str, Any]: "date": pl.Date, "time": pl.String, "time with time zone": pl.String, + "timestamp with time zone": pl.String, "varbinary": pl.String, "array": pl.String, "map": pl.String, diff --git a/tests/pyathena/arrow/test_cursor.py b/tests/pyathena/arrow/test_cursor.py index 68f62e6a1..055d47b83 100644 --- a/tests/pyathena/arrow/test_cursor.py +++ b/tests/pyathena/arrow/test_cursor.py @@ -1038,6 +1038,9 @@ def test_fetch_all_rows(self, arrow_cursor): assert arrow_cursor.fetchall() == [CONVERTED_VALUES_ROW] # An Arrow time type has no time zone, so the table keeps the text. assert arrow_cursor.as_arrow().column("col_time_tz").to_pylist() == ["12:34:56.789+09:00"] + assert arrow_cursor.as_arrow().column("col_timestamp_tz").to_pylist() == [ + "2024-02-29 23:59:58.123 +05:30" + ] @pytest.mark.parametrize( "execute_kwargs", [{}, {"connect_timeout": 3.0, "request_timeout": 4.0}] diff --git a/tests/pyathena/polars/test_cursor.py b/tests/pyathena/polars/test_cursor.py index f719a612d..a943d2683 100644 --- a/tests/pyathena/polars/test_cursor.py +++ b/tests/pyathena/polars/test_cursor.py @@ -783,6 +783,9 @@ def test_fetch_all_rows(self, polars_cursor): assert polars_cursor.fetchall() == [CONVERTED_VALUES_ROW] # A Polars Time has no time zone, so the DataFrame keeps the text. assert polars_cursor.as_polars()["col_time_tz"].to_list() == ["12:34:56.789+09:00"] + assert polars_cursor.as_polars()["col_timestamp_tz"].to_list() == [ + "2024-02-29 23:59:58.123 +05:30" + ] @pytest.mark.parametrize( "execute_kwargs", diff --git a/tests/pyathena/test_converter.py b/tests/pyathena/test_converter.py index 90eb32a67..080925191 100644 --- a/tests/pyathena/test_converter.py +++ b/tests/pyathena/test_converter.py @@ -688,3 +688,41 @@ def test_text_value_converter(): assert DefaultTypeConverter().convert( "array", "[12:34:56.789+09:00]", type_hint="array(time with time zone)" ) == [time(12, 34, 56, 789000, tzinfo=timezone(timedelta(hours=9)))] + + +@pytest.mark.parametrize( + ("input_value", "expected"), + [ + (None, None), + ("", None), + ( + "2024-02-29 23:59:58.123 +05:30", + datetime( + 2024, 2, 29, 23, 59, 58, 123000, tzinfo=timezone(timedelta(hours=5, minutes=30)) + ), + ), + ( + "2024-02-29 23:59:58.123456 -08:00", + datetime(2024, 2, 29, 23, 59, 58, 123456, tzinfo=timezone(-timedelta(hours=8))), + ), + ( + "2024-02-29 23:59:58 +00:00", + datetime(2024, 2, 29, 23, 59, 58, tzinfo=timezone(timedelta(0))), + ), + ( + "2024-02-29 23:59:58.123 UTC", + datetime(2024, 2, 29, 23, 59, 58, 123000, tzinfo=gettz("UTC")), + ), + ( + "2024-02-29 23:59:58.123 America/New_York", + datetime(2024, 2, 29, 23, 59, 58, 123000, tzinfo=gettz("America/New_York")), + ), + ], +) +def test_to_datetime_with_tz_offsets_and_zone_names(input_value, expected): + """Numeric UTC offsets give fixed-offset time zones; zone names keep their zone.""" + result = _to_datetime_with_tz(input_value) + assert result == expected + if expected is not None: + assert result.utcoffset() == expected.utcoffset() + assert result.tzinfo is not None diff --git a/tests/pyathena/util.py b/tests/pyathena/util.py index 7189d525e..b8479d363 100644 --- a/tests/pyathena/util.py +++ b/tests/pyathena/util.py @@ -13,6 +13,7 @@ from botocore.config import Config from botocore.exceptions import ClientError +from dateutil.tz import gettz from jinja2 import Environment, FileSystemLoader from sqlalchemy import types @@ -108,8 +109,9 @@ def unreachable_glue(connection): ) -# TIME values of several precisions, with and without a time zone, a NULL JSON value, -# and the row that every cursor should fetch for them. +# TIME values of several precisions, with and without a time zone, TIMESTAMP WITH TIME +# ZONE values with UTC offsets and a zone name, a NULL JSON value, and the row that +# every cursor should fetch for them. CONVERTED_VALUES_QUERY = """ SELECT 1 AS col @@ -119,6 +121,10 @@ def unreachable_glue(connection): ,CAST('12:34:56 -05:30' AS TIME(0) WITH TIME ZONE) AS col_time_tz_0 ,CAST(NULL AS TIME WITH TIME ZONE) AS col_time_tz_null ,CAST(NULL AS JSON) AS col_json_null + ,TIMESTAMP '2024-02-29 23:59:58.123 +05:30' AS col_timestamp_tz + ,TIMESTAMP '2024-02-29 23:59:58.123 -08:00' AS col_timestamp_tz_negative + ,TIMESTAMP '2024-02-29 23:59:58.123 America/New_York' AS col_timestamp_tz_name + ,CAST(NULL AS TIMESTAMP WITH TIME ZONE) AS col_timestamp_tz_null """ CONVERTED_VALUES_ROW = ( 1, @@ -128,6 +134,10 @@ def unreachable_glue(connection): datetime(2000, 1, 1, 12, 34, 56, tzinfo=timezone(-timedelta(hours=5, minutes=30))).timetz(), None, None, + datetime(2024, 2, 29, 23, 59, 58, 123000, tzinfo=timezone(timedelta(hours=5, minutes=30))), + datetime(2024, 2, 29, 23, 59, 58, 123000, tzinfo=timezone(-timedelta(hours=8))), + datetime(2024, 2, 29, 23, 59, 58, 123000, tzinfo=gettz("America/New_York")), + None, )