From ff9f661c2344b6f51532ffebec200331494525b1 Mon Sep 17 00:00:00 2001 From: mokashang Date: Sat, 6 Jun 2026 09:27:20 -0700 Subject: [PATCH 1/2] Preserve dialect-specific ARRAY types instead of adapting to generic ARRAY When reflecting a PostgreSQL ``text[]`` (or any dialect-specific ARRAY) column, sqlacodegen previously walked the type's MRO in ``get_adapted_type`` and substituted ``sqlalchemy.dialects.postgresql.ARRAY`` with the generic ``sqlalchemy.ARRAY``. The generic ARRAY does not implement operators like ``.contains()``, ``.any()`` or ``.all()`` -- calling them on a generated model raises:: NotImplementedError: ARRAY.contains() not implemented for the base ARRAY type; please use the dialect-specific ARRAY type The workaround was to pass ``--options keep_dialect_types``, but that also forces every other type back to its dialect-specific form (``INTEGER`` instead of ``Integer``, etc.), which is more than the user wants. This change special-cases ARRAY in ``get_adapted_type``: when the column type is a subclass of the generic ARRAY (i.e. a dialect-specific ARRAY), the original class is kept and only the item type is adapted. Plain generic ``sqlalchemy.ARRAY`` inputs are unaffected, and no other type's adaptation behavior changes. The existing ``test_arrays`` is updated to expect the dialect-specific import, and a new ``test_array_preserves_dialect_for_runtime_operators`` test locks in the regression scenario described in the issue. Fixes #441 --- CHANGES.rst | 5 ++++ src/sqlacodegen/generators.py | 10 ++++++++ tests/test_generator_tables.py | 44 +++++++++++++++++++++++++++++++++- 3 files changed, 58 insertions(+), 1 deletion(-) diff --git a/CHANGES.rst b/CHANGES.rst index 75be5d1b..eaa1ff60 100644 --- a/CHANGES.rst +++ b/CHANGES.rst @@ -5,6 +5,11 @@ Version history - Added autoincrement to primary key columns to prevent missing field errors. (`#473 `_; PR by @jtmonroe) +- Preserve dialect-specific ``ARRAY`` types (e.g. ``postgresql.ARRAY``) instead + of adapting them to the generic ``sqlalchemy.ARRAY``. The generic type does + not implement operators like ``.contains()``, so adapting silently broke + PostgreSQL array queries on generated models. + (`#441 `_) **4.0.3** diff --git a/src/sqlacodegen/generators.py b/src/sqlacodegen/generators.py index 0dfba366..a89e690f 100644 --- a/src/sqlacodegen/generators.py +++ b/src/sqlacodegen/generators.py @@ -1036,6 +1036,16 @@ def fix_enum_column(col_name: str, enum_type: Enum) -> None: column.server_default = None def get_adapted_type(self, coltype: Any) -> Any: + # The generic ``sqlalchemy.ARRAY`` does not implement operators such as + # ``.contains()`` that dialect-specific ARRAY types (notably + # ``sqlalchemy.dialects.postgresql.ARRAY``) provide. Adapting a + # dialect-specific ARRAY to the generic one would silently break user + # code that relies on those operators at runtime (see GH-441), so we + # keep the original ARRAY class and only adapt its item type. + if isinstance(coltype, ARRAY) and type(coltype) is not ARRAY: + coltype.item_type = self.get_adapted_type(coltype.item_type) + return coltype + compiled_type = coltype.compile(self.bind.engine.dialect) for supercls in coltype.__class__.__mro__: if not supercls.__name__.startswith("_") and hasattr( diff --git a/tests/test_generator_tables.py b/tests/test_generator_tables.py index 8633e3b5..81a39ba6 100644 --- a/tests/test_generator_tables.py +++ b/tests/test_generator_tables.py @@ -127,10 +127,15 @@ def test_arrays(generator: CodeGenerator) -> None: Column("int_array", postgresql.ARRAY(INTEGER)), ) + # The dialect-specific ``postgresql.ARRAY`` is preserved because the generic + # ``sqlalchemy.ARRAY`` does not implement operators such as ``.contains()`` + # (see GH-441). The item types are still adapted to their generic + # equivalents (``DOUBLE_PRECISION`` -> ``Double``, ``INTEGER`` -> ``Integer``). validate_code( generator.generate(), """\ - from sqlalchemy import ARRAY, Column, Double, Integer, MetaData, Table + from sqlalchemy import Column, Double, Integer, MetaData, Table + from sqlalchemy.dialects.postgresql import ARRAY metadata = MetaData() @@ -144,6 +149,43 @@ def test_arrays(generator: CodeGenerator) -> None: ) +@pytest.mark.parametrize("engine", ["postgresql"], indirect=["engine"]) +def test_array_preserves_dialect_for_runtime_operators( + generator: CodeGenerator, +) -> None: + """Regression test for GH-441. + + A ``text[]`` column should generate ``sqlalchemy.dialects.postgresql.ARRAY`` + (not the generic ``sqlalchemy.ARRAY``) so that PostgreSQL array operators + like ``.contains()`` work on the generated model. The generic ARRAY raises + ``NotImplementedError: ARRAY.contains() not implemented for the base ARRAY + type; please use the dialect-specific ARRAY type``. + """ + Table( + "simple_items", + generator.metadata, + Column("id", postgresql.TEXT, primary_key=True), + Column("tags", postgresql.ARRAY(postgresql.TEXT)), + ) + + validate_code( + generator.generate(), + """\ + from sqlalchemy import Column, MetaData, Table, Text + from sqlalchemy.dialects.postgresql import ARRAY + + metadata = MetaData() + + + t_simple_items = Table( + 'simple_items', metadata, + Column('id', Text, primary_key=True), + Column('tags', ARRAY(Text())) + ) + """, + ) + + @pytest.mark.parametrize("engine", ["postgresql"], indirect=["engine"]) def test_jsonb(generator: CodeGenerator) -> None: Table( From 32c2eed015e9f6d10cfc08a2742aa394966b11b5 Mon Sep 17 00:00:00 2001 From: mokashang Date: Sun, 7 Jun 2026 09:07:11 -0700 Subject: [PATCH 2/2] Address review: trim comments in #480 --- src/sqlacodegen/generators.py | 8 ++------ tests/test_generator_tables.py | 13 +------------ 2 files changed, 3 insertions(+), 18 deletions(-) diff --git a/src/sqlacodegen/generators.py b/src/sqlacodegen/generators.py index a89e690f..a10a21e6 100644 --- a/src/sqlacodegen/generators.py +++ b/src/sqlacodegen/generators.py @@ -1036,12 +1036,8 @@ def fix_enum_column(col_name: str, enum_type: Enum) -> None: column.server_default = None def get_adapted_type(self, coltype: Any) -> Any: - # The generic ``sqlalchemy.ARRAY`` does not implement operators such as - # ``.contains()`` that dialect-specific ARRAY types (notably - # ``sqlalchemy.dialects.postgresql.ARRAY``) provide. Adapting a - # dialect-specific ARRAY to the generic one would silently break user - # code that relies on those operators at runtime (see GH-441), so we - # keep the original ARRAY class and only adapt its item type. + # Keep dialect-specific ARRAY subclasses; the generic sqlalchemy.ARRAY + # is missing operators like .contains() (GH-441). if isinstance(coltype, ARRAY) and type(coltype) is not ARRAY: coltype.item_type = self.get_adapted_type(coltype.item_type) return coltype diff --git a/tests/test_generator_tables.py b/tests/test_generator_tables.py index 81a39ba6..e467dc2b 100644 --- a/tests/test_generator_tables.py +++ b/tests/test_generator_tables.py @@ -127,10 +127,6 @@ def test_arrays(generator: CodeGenerator) -> None: Column("int_array", postgresql.ARRAY(INTEGER)), ) - # The dialect-specific ``postgresql.ARRAY`` is preserved because the generic - # ``sqlalchemy.ARRAY`` does not implement operators such as ``.contains()`` - # (see GH-441). The item types are still adapted to their generic - # equivalents (``DOUBLE_PRECISION`` -> ``Double``, ``INTEGER`` -> ``Integer``). validate_code( generator.generate(), """\ @@ -153,14 +149,7 @@ def test_arrays(generator: CodeGenerator) -> None: def test_array_preserves_dialect_for_runtime_operators( generator: CodeGenerator, ) -> None: - """Regression test for GH-441. - - A ``text[]`` column should generate ``sqlalchemy.dialects.postgresql.ARRAY`` - (not the generic ``sqlalchemy.ARRAY``) so that PostgreSQL array operators - like ``.contains()`` work on the generated model. The generic ARRAY raises - ``NotImplementedError: ARRAY.contains() not implemented for the base ARRAY - type; please use the dialect-specific ARRAY type``. - """ + """Regression test for GH-441.""" Table( "simple_items", generator.metadata,