diff --git a/README.md b/README.md index dc8f20d..4839a54 100644 --- a/README.md +++ b/README.md @@ -50,9 +50,13 @@ pgdb compare --dialect mssql --url "Server=host,1433;Database=db;UID=user;PWD=pa Requires the `mssql` extra: `pip install pgdevkit[mssql]` (pulls in [mssql-python](https://github.com/microsoft/mssql-python), which bundles its own driver — no system ODBC driver install needed). MSSQL has no composite -type, native enum, or first-class JSONB column type, so those areas of a -`database/` tree don't have a direct equivalent on this backend — see -`docs/database-layout.md`. +type or native enum equivalent, so those areas of a `database/` tree don't +have a direct equivalent on this backend — see `docs/database-layout.md`. +Current Azure SQL/SQL Server (2025+) does have a native `json` column type, +which parses/introspects/diffs like any other column type; see +"`pgdevkit.db` — helpers for application code" below for how JSON values are +handled on the CRUD side (write-side serialization only, no auto-parsing on +read — `mssql-python` doesn't distinguish `json` columns from `nvarchar`). ## `pgdb testdb` @@ -144,8 +148,17 @@ Install with the `db` extra: `pip install pgdevkit[db]`. table, built on `psycopg` for safe identifier/value handling. The `mssql` extra provides an `mssql_*`-prefixed mirror of the same functions in `pgdevkit.db.mssql_crud`, built on `mssql-python` (`MERGE`-based upsert, - `OUTPUT` instead of `RETURNING`) — MSSQL has no composite/enum/JSONB - equivalent, so `complex_helper` is always `None` on that path. + `OUTPUT` instead of `RETURNING`) — MSSQL has no composite/enum equivalent, + so `complex_helper` is always `None` on that path. It does have a native + `json` column type on current versions (and the older + `NVARCHAR(MAX)`-plus-`OPENJSON()` convention works on any version), but + `mssql-python` has no auto-serialization for dict/list parameter values + (binding one raises `TypeError`) and no way to distinguish a `json` + column from `nvarchar` on fetch — so every `mssql_*` write function + serializes dict/list values to JSON text automatically + (`db.mssql_sql.json_encode_value`), while reads always come back as plain + `str`; deserialize with `json.loads()` yourself if you need the parsed + value back. - **`SqlLoader`** — loads and caches `.sql` files from `{root}//.sql`, for keeping hand-written queries out of Python source. diff --git a/pgdevkit/db/mssql_crud.py b/pgdevkit/db/mssql_crud.py index fc37d47..2da5bbb 100644 --- a/pgdevkit/db/mssql_crud.py +++ b/pgdevkit/db/mssql_crud.py @@ -3,7 +3,7 @@ import asyncio from typing import Any, Callable, Mapping, Optional, Sequence, Type, TypeVar -from .mssql_sql import ident, qualified +from .mssql_sql import ident, json_encode_values, qualified from .model import TableModel T = TypeVar("T", bound=TableModel) @@ -166,6 +166,7 @@ async def mssql_insert( complex_helper: Any | None = None, ) -> dict[str, Any]: """Insert one row and return the full row (`OUTPUT INSERTED.*`).""" + data = json_encode_values(data) sql, params = _build_insert(table_name, data) row = await _execute_returning(con, sql, params) assert row is not None @@ -182,6 +183,7 @@ async def mssql_insert_many( """Batch insert -- no OUTPUT, one round-trip via executemany.""" if not data: return + data = [json_encode_values(row) for row in data] fields = list(data[0]) sql = _build_insert_many(table_name, fields) await _execute_many(con, sql, [[row[k] for k in fields] for row in data]) @@ -194,6 +196,7 @@ async def mssql_update_dict( primary_keys: Sequence[str], ) -> dict | None: """Update a row identified by primary_keys. Returns the updated row.""" + data = json_encode_values(data) sql, params = _build_update(table_name, data, primary_keys) return await _execute_returning(con, sql, params) @@ -212,6 +215,7 @@ async def mssql_upsert_dict( complex_helper: Any | None = None, ) -> dict: """MERGE-based upsert, returns the row as a dict.""" + data = json_encode_values(data) sql, params = _build_upsert_merge(table_name, data, primary_keys) row = await _execute_returning(con, sql, params) assert row is not None @@ -241,6 +245,7 @@ async def mssql_upsert_many_dict( and want a missing row to be a silent no-op rather than create one.""" if not data: return + data = [json_encode_values(row) for row in data] fields = list(data[0]) if must_exist: sql = _build_update_many(table_name, fields, primary_keys) diff --git a/pgdevkit/db/mssql_sql.py b/pgdevkit/db/mssql_sql.py index e3c8a2e..e60ce55 100644 --- a/pgdevkit/db/mssql_sql.py +++ b/pgdevkit/db/mssql_sql.py @@ -1,5 +1,8 @@ from __future__ import annotations +import json +from typing import Any + def ident(name: str) -> str: """Bracket-quote a single identifier, doubling any embedded `]` @@ -12,3 +15,34 @@ def ident(name: str) -> str: def qualified(schema: str, table: str) -> str: return f"{ident(schema)}.{ident(table)}" + + +def json_encode_value(value: Any) -> Any: + """Serialize a value destined for a `json`-typed (or legacy + `nvarchar(max)`-storing-JSON) column to text. + + mssql-python has no auto-serialization for dict/list parameter values -- + binding one directly raises `TypeError: Unsupported parameter type` + (confirmed against mssql-python 1.12.0: its `_map_sql_type` has explicit + branches for every scalar Python type but none for dict/list, and SQL + Server's native `json` type -- a genuine first-class type in current + Azure SQL/SQL Server, unlike the old NVARCHAR(MAX)-plus-OPENJSON() + convention -- has no dedicated ODBC type code in this driver either, so + it's fetched back as plain `str`, indistinguishable from any other text + column). Unlike Postgres, there's no type-registration ambiguity to + resolve here (composite type vs jsonb vs plain array all need different + handling there): MSSQL has no composite types, so a Python dict/list + passed to any MSSQL CRUD call can only sensibly mean "serialize me as + JSON text" -- no per-column-type lookup needed on the write side. + + There is deliberately no read-side counterpart: the driver can't tell + us which columns are `json`-typed (it reports the same opaque `str` for + those as for a plain `nvarchar`), so auto-parsing fetched values back + into dict/list would need its own catalog lookup -- a ComplexHelper-like + mechanism this backend intentionally doesn't have. Callers that know a + column is JSON deserialize it themselves with `json.loads()`.""" + return json.dumps(value) if isinstance(value, (dict, list)) else value + + +def json_encode_values(data: dict) -> dict: + return {k: json_encode_value(v) for k, v in data.items()} diff --git a/pgdevkit/testdb/mssql/api.py b/pgdevkit/testdb/mssql/api.py index 2284233..40af093 100644 --- a/pgdevkit/testdb/mssql/api.py +++ b/pgdevkit/testdb/mssql/api.py @@ -7,7 +7,7 @@ import mssql_python -from ...db.mssql_sql import ident +from ...db.mssql_sql import ident, json_encode_value from ...dialect import MSSQL from .. import query from ..config import ProjectConfig @@ -84,21 +84,18 @@ def _run() -> None: (count,) = cur.fetchone() if count == len(rows): return - # Unlike the Postgres path (ComplexHelper-driven composite/enum/JSONB - # conversion), MSSQL fixtures are limited to flat scalar columns for - # this first cut -- there is no MSSQL equivalent to convert into, so - # dict/list values are serialized as JSON text (matching how a - # NVARCHAR(MAX)-typed "JSON column" is conventionally stored on this - # engine) rather than silently dropped or erroring. + # Unlike the Postgres path (ComplexHelper-driven composite/enum + # conversion), MSSQL has no composite/enum equivalent to convert + # into -- dict/list values are serialized as JSON text instead + # (see db/mssql_sql.json_encode_value), matching a `json`-typed or + # legacy NVARCHAR(MAX)-storing-JSON column, rather than silently + # dropped or erroring. col_names = list(rows[0]) cur.execute(f"DELETE FROM {qualified}") cols = ", ".join(ident(c) for c in col_names) placeholders = ", ".join("?" for _ in col_names) insert_sql = f"INSERT INTO {qualified} ({cols}) VALUES ({placeholders})" - param_rows = [ - [json.dumps(row[c]) if isinstance(row[c], (dict, list)) else row[c] for c in col_names] - for row in rows - ] + param_rows = [[json_encode_value(row[c]) for c in col_names] for row in rows] cur.executemany(insert_sql, param_rows) await asyncio.to_thread(_run) diff --git a/tests/db/test_mssql_crud_live.py b/tests/db/test_mssql_crud_live.py index 22adea4..42e17bf 100644 --- a/tests/db/test_mssql_crud_live.py +++ b/tests/db/test_mssql_crud_live.py @@ -1,6 +1,7 @@ from __future__ import annotations import asyncio +import json from pathlib import Path import mssql_python @@ -71,3 +72,35 @@ async def _run() -> None: conn.close() finally: clean_testdb(project) + + +@requires_mssql +def test_json_column_round_trip(tmp_path: Path): + # app.widget.tags is NVARCHAR(MAX) (see fixtures/database_mssql) -- + # this is the main way the write-side JSON fix actually gets validated + # against a real driver+server: without db/mssql_sql.json_encode_value, + # mssql-python raises TypeError outright on a raw dict/list parameter + # (confirmed via source inspection, not just assumed), so this insert + # wouldn't merely round-trip wrong, it would crash. + project = _make_project(tmp_path, "mssqljsonlive", "main", engine="mssql") + try: + ensure_testdb(project) + conn = mssql_python.connect(status(project)["dsn"], autocommit=True) + + async def _run() -> None: + inserted = await mssql_insert( + conn, ("app", "widget"), {"id": 5, "name": "tagged", "tags": ["red", "blue"]} + ) + assert json.loads(inserted["tags"]) == ["red", "blue"] + + upserted = await mssql_upsert_dict( + conn, ("app", "widget"), {"id": 5, "name": "tagged", "tags": {"a": 1}}, ["id"] + ) + assert json.loads(upserted["tags"]) == {"a": 1} + + try: + asyncio.run(_run()) + finally: + conn.close() + finally: + clean_testdb(project) diff --git a/tests/db/test_mssql_crud_sql.py b/tests/db/test_mssql_crud_sql.py index 3066c53..2fee4b9 100644 --- a/tests/db/test_mssql_crud_sql.py +++ b/tests/db/test_mssql_crud_sql.py @@ -1,5 +1,7 @@ from __future__ import annotations +import asyncio + from pgdevkit.db.mssql_crud import ( _build_delete, _build_insert, @@ -9,8 +11,11 @@ _build_update, _build_update_many, _build_upsert_merge, + mssql_insert, + mssql_update_dict, + mssql_upsert_dict, ) -from pgdevkit.db.mssql_sql import ident, qualified +from pgdevkit.db.mssql_sql import ident, json_encode_value, json_encode_values, qualified def test_ident_brackets_and_doubles_embedded_bracket(): @@ -83,3 +88,76 @@ def test_build_delete_uses_output_deleted_before_where(): sql, params = _build_delete(("dbo", "widget"), {"id": 1}) assert sql == "DELETE FROM [dbo].[widget] OUTPUT DELETED.* WHERE [id] = ?" assert params == [1] + + +# JSON support: mssql-python has no auto-serialization for dict/list +# parameter values (binding one raises TypeError -- confirmed against the +# installed driver), and no distinct type code for SQL Server's native +# `json` type either (it's fetched as plain str, indistinguishable from +# nvarchar). See db/mssql_sql.json_encode_value's docstring for the full +# story; these tests cover the write-side serialization it does. + + +def test_json_encode_value_serializes_dict_and_list(): + assert json_encode_value({"a": 1}) == '{"a": 1}' + assert json_encode_value([1, 2]) == "[1, 2]" + + +def test_json_encode_value_leaves_scalars_untouched(): + assert json_encode_value("hello") == "hello" + assert json_encode_value(5) == 5 + assert json_encode_value(None) is None + + +def test_json_encode_values_only_touches_dict_and_list_entries(): + result = json_encode_values({"name": "sprocket", "tags": ["a", "b"], "count": 3}) + assert result == {"name": "sprocket", "tags": '["a", "b"]', "count": 3} + + +class _FakeCursor: + """A minimal double for mssql-python's Cursor -- just enough to capture + what mssql_insert/mssql_update_dict/mssql_upsert_dict actually bind, so + the JSON-encoding wiring can be verified without a real driver/server.""" + + def __init__(self, row): + self.executed: tuple[str, list] | None = None + self._row = row + self.description = [("id",), ("payload",)] + + def execute(self, sql, params): + self.executed = (sql, params) + + def fetchone(self): + return self._row + + def close(self): + pass + + +class _FakeConnection: + def __init__(self, row): + self.cursor_obj = _FakeCursor(row) + + def cursor(self): + return self.cursor_obj + + +def test_mssql_insert_serializes_dict_value_before_binding(): + con = _FakeConnection(row=(1, '{"a": 1}')) + asyncio.run(mssql_insert(con, ("dbo", "widget"), {"id": 1, "payload": {"a": 1}})) + _, params = con.cursor_obj.executed + assert params == [1, '{"a": 1}'] + + +def test_mssql_update_dict_serializes_dict_value_before_binding(): + con = _FakeConnection(row=(1, '{"a": 1}')) + asyncio.run(mssql_update_dict(con, ("dbo", "widget"), {"id": 1, "payload": {"a": 1}}, ["id"])) + _, params = con.cursor_obj.executed + assert params == ['{"a": 1}', 1] + + +def test_mssql_upsert_dict_serializes_list_value_before_binding(): + con = _FakeConnection(row=(1, "[1, 2]")) + asyncio.run(mssql_upsert_dict(con, ("dbo", "widget"), {"id": 1, "payload": [1, 2]}, ["id"])) + _, params = con.cursor_obj.executed + assert params == [1, "[1, 2]"] diff --git a/tests/test_parser_mssql.py b/tests/test_parser_mssql.py index dd1d8e9..6f842a9 100644 --- a/tests/test_parser_mssql.py +++ b/tests/test_parser_mssql.py @@ -82,6 +82,21 @@ def test_parse_function_details_tsql_extracts_args_and_body_directly(): assert "return @a + @b" in body +def test_parses_native_json_column_type(tmp_path: Path): + # SQL Server's native `json` type (current Azure SQL/SQL Server 2025+ -- + # distinct from the older NVARCHAR(MAX)-plus-OPENJSON() convention). + # sqlglot's tsql dialect already recognizes it as exp.DataType.Type.JSON; + # this just confirms parser.py round-trips it to "json" like any other + # type name, unaffected by the fact that MSSQL has no native enum/ + # composite type (Dialect.supports_enums/supports_composites are about + # CREATE TYPE, not built-in column types like this one). + _write(tmp_path, "widget.sql", "CREATE TABLE dbo.widget (id INT PRIMARY KEY, payload JSON NULL);") + schema = parse_directory(tmp_path, dialect="mssql") + cols = {c.name: c for c in schema.tables["dbo.widget"].columns} + assert cols["payload"].data_type == "json" + assert cols["payload"].is_nullable + + def test_mssql_enum_style_type_is_ignored_not_erroring(tmp_path: Path): # A Postgres-style `CREATE TYPE ... AS ENUM` has no T-SQL equivalent -- # parsing it under dialect="mssql" must not raise, and (since sqlglot's diff --git a/tests/testdb/fixtures/database_mssql/app/tables/widget.sql b/tests/testdb/fixtures/database_mssql/app/tables/widget.sql index 1857bf5..12b3008 100644 --- a/tests/testdb/fixtures/database_mssql/app/tables/widget.sql +++ b/tests/testdb/fixtures/database_mssql/app/tables/widget.sql @@ -2,6 +2,12 @@ IF NOT EXISTS (SELECT 1 FROM sys.tables t JOIN sys.schemas s ON s.schema_id = t. BEGIN CREATE TABLE app.widget ( id INT PRIMARY KEY, - name NVARCHAR(100) NOT NULL + name NVARCHAR(100) NOT NULL, + -- NVARCHAR(MAX)-storing-JSON, not the native `json` type (SQL + -- Server 2025+) -- the CI container image (2022) predates it, and + -- this convention is what the write-side JSON-encoding fix in + -- db/mssql_sql.json_encode_value covers either way (mssql-python + -- has no auto dict/list serialization for either column style). + tags NVARCHAR(MAX) NULL ); END