From 4217bbe65b6203f42522c63646bb2b5b06c02a7b Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sat, 3 Oct 2026 18:00:48 +0900 Subject: [PATCH 1/5] Reflect the field and value types of top-level STRUCT and MAP columns Column reflection parsed MAP and STRUCT/ROW type arguments only inside an ARRAY. A top-level map<...> column was reflected as String and a top-level struct<...> column as an AthenaStruct without fields, so CREATE TABLE compiled from a reflected table rendered MAP columns as STRING and failed for STRUCT columns. Top-level MAP and STRUCT/ROW columns now reflect as AthenaMap and AthenaStruct with their parsed types, from both the metadata API (Hive) and information_schema (Trino) spellings. A top-level type that cannot be parsed warns and reflects as NullType, as an ARRAY already does. Closes #886 Co-Authored-By: Claude Opus 5.5 --- docs/sqlalchemy.md | 6 +++ pyathena/sqlalchemy/base.py | 26 +++++++++++ tests/pyathena/sqlalchemy/test_base.py | 65 ++++++++++++++++++++++---- 3 files changed, 89 insertions(+), 8 deletions(-) diff --git a/docs/sqlalchemy.md b/docs/sqlalchemy.md index f0535fa54..9977baefa 100644 --- a/docs/sqlalchemy.md +++ b/docs/sqlalchemy.md @@ -846,6 +846,9 @@ CREATE TABLE users ( `CREATE TABLE` renders `AthenaStruct` columns with Hive `STRUCT` syntax at every nesting depth, and `CAST` renders them as `ROW(name type, ...)`. See [DDL and CAST types](#ddl-and-cast-types). +Reflected STRUCT and ROW columns use `AthenaStruct` with their field names and types. +Selecting such a column returns the value from the cursor, as described under Data format support below; SQLAlchemy does not convert it to the reflected field types. + #### Querying STRUCT data PyAthena automatically converts STRUCT data between different formats: @@ -991,6 +994,9 @@ CREATE TABLE products ( `CREATE TABLE` renders integer MAP keys and values as `INT`. `CAST` still spells those integers as `INTEGER`. +Reflected MAP columns use `AthenaMap` with their key and value types. +Selecting such a column returns the value from the cursor, as described under Data format support below; SQLAlchemy does not convert it to the reflected key and value types. + #### Querying MAP data PyAthena automatically converts MAP data between different formats: diff --git a/pyathena/sqlalchemy/base.py b/pyathena/sqlalchemy/base.py index d33331e6b..917f6a921 100644 --- a/pyathena/sqlalchemy/base.py +++ b/pyathena/sqlalchemy/base.py @@ -695,6 +695,26 @@ def get_columns(self, connection: Connection, table_name: str, schema: str | Non return self._get_columns(connection, table_name, schema=schema, **kw) def _get_column_type(self, type_: str, _nested: bool = False): + """Map an Athena column type string to a SQLAlchemy type. + + Accepts both the Hive (``struct``, ``map``) and the + Trino (``row(a integer)``, ``map(integer, integer)``) spellings, and + parses the element, key, value, and field types of ARRAY, MAP, and + STRUCT/ROW types. + + Args: + type_: The column type reported by Athena. + _nested: Whether ``type_`` is nested in another type. A nested MAP + or STRUCT/ROW that cannot be parsed raises, so that the + enclosing type is reported as unrecognized. + + Returns: + The SQLAlchemy type, or ``NullType`` with a warning for a type that + is not recognized. + + Raises: + ValueError: If a nested MAP or STRUCT/ROW cannot be parsed. + """ type_ = type_.strip() match = self._pattern_column_type.match(type_) if match: @@ -710,6 +730,12 @@ def _get_column_type(self, type_: str, _nested: bool = False): except (TypeError, ValueError): util.warn(f"Did not recognize type '{type_}'") return types.NullType() + if not _nested and name in ("map", "row", "struct") and length: + try: + return self._get_column_type(type_, _nested=True) + except (TypeError, ValueError): + util.warn(f"Did not recognize type '{type_}'") + return types.NullType() if _nested and name == "map" and length: key, value = _split_type_arguments(length) return AthenaMap( diff --git a/tests/pyathena/sqlalchemy/test_base.py b/tests/pyathena/sqlalchemy/test_base.py index 8af87da47..661c0c2cc 100644 --- a/tests/pyathena/sqlalchemy/test_base.py +++ b/tests/pyathena/sqlalchemy/test_base.py @@ -183,6 +183,7 @@ def open_cursor(cursor_class, converter=None): assert [column["name"] for column in columns] == ["id", "payload", "label", "dt"] assert isinstance(columns[0]["type"], types.INTEGER) assert isinstance(columns[1]["type"], AthenaStruct) + assert list(columns[1]["type"].fields) == ["a", "b"] assert type(columns[2]["type"]) is types.String assert type(columns[3]["type"]) is types.String assert [column["comment"] for column in columns] == ["identifier", None, None, None] @@ -196,6 +197,47 @@ def open_cursor(cursor_class, converter=None): assert "WHERE table_schema = 'my_schema' AND table_name = 'o''neil'" in operation assert kwargs == {"result_reuse_enable": False} + @pytest.mark.parametrize( + ("map_type", "struct_type"), + [ + # Metadata API (Hive) spellings. + ("map", "struct>>"), + # information_schema (Trino) spellings. + ("map(integer, varchar)", 'row(a integer, "b c" array(row(x integer)))'), + ], + ) + def test_top_level_map_and_struct_reflect_their_types(self, map_type, struct_type): + dialect = AthenaDialect() + map_ = dialect._get_column_type(map_type) + struct = dialect._get_column_type(struct_type) + + assert isinstance(map_, AthenaMap) + assert isinstance(map_.key_type, types.INTEGER) + assert isinstance(map_.value_type, (types.String, types.VARCHAR)) + assert isinstance(struct, AthenaStruct) + assert list(struct.fields) == ["a", "b c"] + assert isinstance(struct.fields["a"], types.INTEGER) + assert isinstance(struct.fields["b c"], AthenaArray) + assert list(struct.fields["b c"].item_type.fields) == ["x"] + + table = Table( + "t", + MetaData(), + Column("m", map_), + Column("s", struct), + awsathena_location="s3://bucket/path/", + ) + ddl = str(CreateTable(table).compile(dialect=dialect)) + assert "\tm MAP,\n" in ddl + assert "\ts STRUCT>>\n" in ddl + + @pytest.mark.parametrize( + "type_", ["map", "struct", "row(a)", "struct>", "map>"] + ) + def test_unrecognized_top_level_map_or_struct_reflects_null_type(self, type_): + with pytest.warns(sqlalchemy.exc.SAWarning, match="Did not recognize type"): + assert isinstance(AthenaDialect()._get_column_type(type_), types.NullType) + def test_empty_metadata_comment_is_no_comment(self): # Glue can carry an empty comment, so the metadata path must agree with # the information_schema path rather than reflecting it as a comment. @@ -1499,10 +1541,15 @@ def test_reflect_select(self, engine): assert isinstance(one_row_complex.c.col_binary.type, types.BINARY) assert isinstance(one_row_complex.c.col_array.type, AthenaArray) assert isinstance(one_row_complex.c.col_array.type.item_type, types.INTEGER) - assert isinstance(one_row_complex.c.col_map.type, types.String) - # With struct support, col_struct should now be recognized as AthenaStruct - + assert isinstance(one_row_complex.c.col_map.type, AthenaMap) + assert isinstance(one_row_complex.c.col_map.type.key_type, types.INTEGER) + assert isinstance(one_row_complex.c.col_map.type.value_type, types.INTEGER) assert isinstance(one_row_complex.c.col_struct.type, AthenaStruct) + assert list(one_row_complex.c.col_struct.type.fields) == ["a", "b"] + assert isinstance(one_row_complex.c.col_struct.type.fields["a"], types.INTEGER) + ddl = str(CreateTable(one_row_complex).compile(dialect=engine.dialect)) + assert "\tcol_map MAP,\n" in ddl + assert "\tcol_struct STRUCT,\n" in ddl assert isinstance( one_row_complex.c.col_decimal.type, types.DECIMAL, @@ -1552,11 +1599,13 @@ def test_get_column_type(self, engine): assert isinstance(dialect._get_column_type("date"), types.DATE) assert isinstance(dialect._get_column_type("binary"), types.BINARY) assert isinstance(dialect._get_column_type("array"), AthenaArray) - assert isinstance(dialect._get_column_type("map"), types.String) - # With struct support, struct types should be recognized as AthenaStruct - - assert isinstance(dialect._get_column_type("struct"), AthenaStruct) - assert isinstance(dialect._get_column_type("row"), AthenaStruct) + assert isinstance(dialect._get_column_type("map"), AthenaMap) + struct = dialect._get_column_type("struct") + assert isinstance(struct, AthenaStruct) + assert list(struct.fields) == ["a", "b"] + row = dialect._get_column_type("row") + assert isinstance(row, AthenaStruct) + assert list(row.fields) == ["name", "age"] decimal_with_args = dialect._get_column_type("decimal(10,1)") assert isinstance(decimal_with_args, types.DECIMAL) assert decimal_with_args.precision == 10 From 543d681f4c89376840f00ea349fe4798818b0917 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sat, 3 Oct 2026 18:10:50 +0900 Subject: [PATCH 2/5] Describe how unrecognized nested column types are reflected An unrecognized element, key, value, or field type becomes NullType in place, while arguments that cannot be parsed make the whole type NullType. The docstring only described the second case. Co-Authored-By: Claude Opus 5.5 --- pyathena/sqlalchemy/base.py | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/pyathena/sqlalchemy/base.py b/pyathena/sqlalchemy/base.py index 917f6a921..7cdca804c 100644 --- a/pyathena/sqlalchemy/base.py +++ b/pyathena/sqlalchemy/base.py @@ -709,11 +709,14 @@ def _get_column_type(self, type_: str, _nested: bool = False): enclosing type is reported as unrecognized. Returns: - The SQLAlchemy type, or ``NullType`` with a warning for a type that - is not recognized. + The SQLAlchemy type. A type name that is not recognized becomes + ``NullType`` with a warning, in place when it is an element, key, + value, or field type. An ARRAY, MAP, or STRUCT/ROW whose arguments + cannot be parsed becomes ``NullType`` as a whole, with a warning. Raises: - ValueError: If a nested MAP or STRUCT/ROW cannot be parsed. + ValueError: If the arguments of a nested MAP or STRUCT/ROW cannot be + parsed. """ type_ = type_.strip() match = self._pattern_column_type.match(type_) From 035a3b9eac5a3f36bdac0b3269a343580e84e818 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sat, 3 Oct 2026 18:15:48 +0900 Subject: [PATCH 3/5] List the scalar argument errors that column type parsing raises Co-Authored-By: Claude Opus 5.5 --- pyathena/sqlalchemy/base.py | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/pyathena/sqlalchemy/base.py b/pyathena/sqlalchemy/base.py index 7cdca804c..459e3b83a 100644 --- a/pyathena/sqlalchemy/base.py +++ b/pyathena/sqlalchemy/base.py @@ -715,8 +715,14 @@ def _get_column_type(self, type_: str, _nested: bool = False): cannot be parsed becomes ``NullType`` as a whole, with a warning. Raises: - ValueError: If the arguments of a nested MAP or STRUCT/ROW cannot be - parsed. + ValueError: If the length, precision, or scale of a CHAR, VARCHAR, + or DECIMAL type is not an integer, or if the arguments of a + nested MAP or STRUCT/ROW cannot be parsed. Inside a top-level + ARRAY, MAP, or STRUCT/ROW, these errors make that type + ``NullType`` instead. + TypeError: If a DECIMAL type has more arguments than SQLAlchemy's + ``DECIMAL`` accepts, with the same exception for a top-level + ARRAY, MAP, or STRUCT/ROW. """ type_ = type_.strip() match = self._pattern_column_type.match(type_) From e06653de567a23b53d87e48b45a34ea58ac275ed Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sat, 3 Oct 2026 18:19:40 +0900 Subject: [PATCH 4/5] State where column type parse errors are handled A parse error is handled by the innermost enclosing ARRAY, or else by a top-level MAP or STRUCT/ROW, and raised otherwise. Co-Authored-By: Claude Opus 5.5 --- pyathena/sqlalchemy/base.py | 23 +++++++++++------------ 1 file changed, 11 insertions(+), 12 deletions(-) diff --git a/pyathena/sqlalchemy/base.py b/pyathena/sqlalchemy/base.py index 459e3b83a..d620fce6e 100644 --- a/pyathena/sqlalchemy/base.py +++ b/pyathena/sqlalchemy/base.py @@ -709,20 +709,19 @@ def _get_column_type(self, type_: str, _nested: bool = False): enclosing type is reported as unrecognized. Returns: - The SQLAlchemy type. A type name that is not recognized becomes - ``NullType`` with a warning, in place when it is an element, key, - value, or field type. An ARRAY, MAP, or STRUCT/ROW whose arguments - cannot be parsed becomes ``NullType`` as a whole, with a warning. + The SQLAlchemy type. A type name that is not recognized, such as + ``foo`` in ``struct``, becomes ``NullType`` in place with a + warning. A type that cannot be parsed, such as ``map`` or + ``varchar(x)``, makes its innermost enclosing ARRAY ``NullType`` + with a warning; without an enclosing ARRAY, a top-level MAP or + STRUCT/ROW becomes ``NullType`` instead. Raises: - ValueError: If the length, precision, or scale of a CHAR, VARCHAR, - or DECIMAL type is not an integer, or if the arguments of a - nested MAP or STRUCT/ROW cannot be parsed. Inside a top-level - ARRAY, MAP, or STRUCT/ROW, these errors make that type - ``NullType`` instead. - TypeError: If a DECIMAL type has more arguments than SQLAlchemy's - ``DECIMAL`` accepts, with the same exception for a top-level - ARRAY, MAP, or STRUCT/ROW. + ValueError: If a type cannot be parsed and neither an enclosing + ARRAY nor a top-level MAP or STRUCT/ROW handles it, for example + a top-level ``varchar(x)`` or a nested ``map``. + TypeError: In the same case, for a DECIMAL type with more + arguments than SQLAlchemy's ``DECIMAL`` accepts. """ type_ = type_.strip() match = self._pattern_column_type.match(type_) From db6be39f06803b1fd0e851f20598eefd2512ed61 Mon Sep 17 00:00:00 2001 From: laughingman7743 Date: Sat, 3 Oct 2026 18:44:08 +0900 Subject: [PATCH 5/5] Raise CompileError for NullType in Athena DDL AthenaTypeCompiler rendered NullType as NULL, which Athena DDL rejects. The override existed for CAST, which now uses AthenaDMLTypeCompiler, and for an untyped foreign key column in a compliance test that the dialect's requirements now skip. Reflection can now produce NullType for an unrecognized MAP key or value or STRUCT field type, so CREATE TABLE from such a table raises SQLAlchemy's CompileError instead of emitting invalid DDL, as it already does for JSON and an empty STRUCT. Co-Authored-By: Claude Opus 5.5 --- docs/sqlalchemy.md | 3 +++ pyathena/sqlalchemy/compiler.py | 8 ++------ tests/pyathena/sqlalchemy/test_base.py | 15 +++++++++++++++ tests/pyathena/sqlalchemy/test_compiler.py | 19 +++++++++++++++++++ 4 files changed, 39 insertions(+), 6 deletions(-) diff --git a/docs/sqlalchemy.md b/docs/sqlalchemy.md index 9977baefa..6e285f19c 100644 --- a/docs/sqlalchemy.md +++ b/docs/sqlalchemy.md @@ -755,6 +755,7 @@ Athena parses DDL statements such as `CREATE TABLE` with Hive type syntax, and q Complex types apply the same syntax to their nested types. An `AthenaStruct` without fields raises `CompileError` in both. +`NullType`, including an unrecognized type that reflection reports as `NullType`, raises `CompileError` in DDL at any depth. ## Floating-point types @@ -847,6 +848,7 @@ CREATE TABLE users ( See [DDL and CAST types](#ddl-and-cast-types). Reflected STRUCT and ROW columns use `AthenaStruct` with their field names and types. +A field type that the dialect does not recognize is reflected as `NullType` with a warning. Selecting such a column returns the value from the cursor, as described under Data format support below; SQLAlchemy does not convert it to the reflected field types. #### Querying STRUCT data @@ -995,6 +997,7 @@ CREATE TABLE products ( `CAST` still spells those integers as `INTEGER`. Reflected MAP columns use `AthenaMap` with their key and value types. +A key or value type that the dialect does not recognize is reflected as `NullType` with a warning. Selecting such a column returns the value from the cursor, as described under Data format support below; SQLAlchemy does not convert it to the reflected key and value types. #### Querying MAP data diff --git a/pyathena/sqlalchemy/compiler.py b/pyathena/sqlalchemy/compiler.py index 8c2f81183..189855074 100644 --- a/pyathena/sqlalchemy/compiler.py +++ b/pyathena/sqlalchemy/compiler.py @@ -90,8 +90,8 @@ class AthenaTypeCompiler(GenericTypeCompiler): without a length render as STRING; with a length, CHAR, NCHAR, VARCHAR, and NVARCHAR render as CHAR(n) or VARCHAR(n). Complex types render as ``STRUCT``, ``MAP``, and - ``ARRAY``. TIME, JSON, and a STRUCT without fields have no Athena - DDL type and raise ``CompileError``. + ``ARRAY``. TIME, JSON, a STRUCT without fields, and ``NullType`` + have no Athena DDL type and raise ``CompileError``. See Also: AWS Athena Data Types: @@ -238,10 +238,6 @@ def visit_unicode(self, type_, **kw): def visit_unicode_text(self, type_, **kw): return "STRING" - @override - def visit_null(self, type_, **kw): - return "NULL" - def visit_tinyint(self, type_, **kw): """Render a tinyint type through ``visit_TINYINT``. diff --git a/tests/pyathena/sqlalchemy/test_base.py b/tests/pyathena/sqlalchemy/test_base.py index 661c0c2cc..9e0536bc7 100644 --- a/tests/pyathena/sqlalchemy/test_base.py +++ b/tests/pyathena/sqlalchemy/test_base.py @@ -231,6 +231,21 @@ def test_top_level_map_and_struct_reflect_their_types(self, map_type, struct_typ assert "\tm MAP,\n" in ddl assert "\ts STRUCT>>\n" in ddl + def test_unrecognized_field_type_reflects_null_type_and_blocks_ddl(self): + dialect = AthenaDialect() + with pytest.warns(sqlalchemy.exc.SAWarning, match="Did not recognize type"): + struct = dialect._get_column_type("struct") + assert isinstance(struct.fields["a"], types.INTEGER) + assert isinstance(struct.fields["b"], types.NullType) + table = Table( + "t", MetaData(), Column("payload", struct), awsathena_location="s3://bucket/path/" + ) + with pytest.raises( + sqlalchemy.exc.CompileError, + match=r"column 'payload'.*Can't generate DDL for NullType", + ): + CreateTable(table).compile(dialect=dialect) + @pytest.mark.parametrize( "type_", ["map", "struct", "row(a)", "struct>", "map>"] ) diff --git a/tests/pyathena/sqlalchemy/test_compiler.py b/tests/pyathena/sqlalchemy/test_compiler.py index 0c3548a76..b5a841ba4 100644 --- a/tests/pyathena/sqlalchemy/test_compiler.py +++ b/tests/pyathena/sqlalchemy/test_compiler.py @@ -218,6 +218,25 @@ def test_ddl_and_cast_types(self, type_, ddl, cast_type): assert dialect.type_compiler_instance.process(type_) == ddl assert str(cast(column("x"), type_).compile(dialect=dialect)) == f"CAST(x AS {cast_type})" + @pytest.mark.parametrize( + "type_", + [ + types.NullType(), + AthenaArray(types.NullType()), + AthenaMap(Integer, types.NullType()), + AthenaStruct(("a", Integer), ("b", types.NullType())), + ], + ) + def test_null_type_is_rejected_in_ddl(self, type_): + dialect = AthenaDialect() + with pytest.raises(exc.CompileError, match="Can't generate DDL for NullType"): + dialect.type_compiler_instance.process(type_) + + def test_null_type_cast_is_unchanged(self): + assert str(cast(column("x"), types.NullType()).compile(dialect=AthenaDialect())) == ( + "CAST(x AS NULL)" + ) + @pytest.mark.parametrize( "type_", [types.JSON(), AthenaArray(types.JSON), AthenaMap(String, types.JSON)] )