Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
23 changes: 18 additions & 5 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -50,9 +50,13 @@ pgdb compare --dialect mssql --url "Server=host,1433;Database=db;UID=user;PWD=pa
Requires the `mssql` extra: `pip install pgdevkit[mssql]` (pulls in
[mssql-python](https://github.com/microsoft/mssql-python), which bundles its
own driver — no system ODBC driver install needed). MSSQL has no composite
type, native enum, or first-class JSONB column type, so those areas of a
`database/` tree don't have a direct equivalent on this backend — see
`docs/database-layout.md`.
type or native enum equivalent, so those areas of a `database/` tree don't
have a direct equivalent on this backend — see `docs/database-layout.md`.
Current Azure SQL/SQL Server (2025+) does have a native `json` column type,
which parses/introspects/diffs like any other column type; see
"`pgdevkit.db` — helpers for application code" below for how JSON values are
handled on the CRUD side (write-side serialization only, no auto-parsing on
read — `mssql-python` doesn't distinguish `json` columns from `nvarchar`).

## `pgdb testdb`

Expand Down Expand Up @@ -144,8 +148,17 @@ Install with the `db` extra: `pip install pgdevkit[db]`.
table, built on `psycopg` for safe identifier/value handling. The `mssql`
extra provides an `mssql_*`-prefixed mirror of the same functions in
`pgdevkit.db.mssql_crud`, built on `mssql-python` (`MERGE`-based upsert,
`OUTPUT` instead of `RETURNING`) — MSSQL has no composite/enum/JSONB
equivalent, so `complex_helper` is always `None` on that path.
`OUTPUT` instead of `RETURNING`) — MSSQL has no composite/enum equivalent,
so `complex_helper` is always `None` on that path. It does have a native
`json` column type on current versions (and the older
`NVARCHAR(MAX)`-plus-`OPENJSON()` convention works on any version), but
`mssql-python` has no auto-serialization for dict/list parameter values
(binding one raises `TypeError`) and no way to distinguish a `json`
column from `nvarchar` on fetch — so every `mssql_*` write function
serializes dict/list values to JSON text automatically
(`db.mssql_sql.json_encode_value`), while reads always come back as plain
`str`; deserialize with `json.loads()` yourself if you need the parsed
value back.
- **`SqlLoader`** — loads and caches `.sql` files from
`{root}/<topic>/<name>.sql`, for keeping hand-written queries out of
Python source.
Expand Down
7 changes: 6 additions & 1 deletion pgdevkit/db/mssql_crud.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@
import asyncio
from typing import Any, Callable, Mapping, Optional, Sequence, Type, TypeVar

from .mssql_sql import ident, qualified
from .mssql_sql import ident, json_encode_values, qualified
from .model import TableModel

T = TypeVar("T", bound=TableModel)
Expand Down Expand Up @@ -166,6 +166,7 @@ async def mssql_insert(
complex_helper: Any | None = None,
) -> dict[str, Any]:
"""Insert one row and return the full row (`OUTPUT INSERTED.*`)."""
data = json_encode_values(data)
sql, params = _build_insert(table_name, data)
row = await _execute_returning(con, sql, params)
assert row is not None
Expand All @@ -182,6 +183,7 @@ async def mssql_insert_many(
"""Batch insert -- no OUTPUT, one round-trip via executemany."""
if not data:
return
data = [json_encode_values(row) for row in data]
fields = list(data[0])
sql = _build_insert_many(table_name, fields)
await _execute_many(con, sql, [[row[k] for k in fields] for row in data])
Expand All @@ -194,6 +196,7 @@ async def mssql_update_dict(
primary_keys: Sequence[str],
) -> dict | None:
"""Update a row identified by primary_keys. Returns the updated row."""
data = json_encode_values(data)
sql, params = _build_update(table_name, data, primary_keys)
return await _execute_returning(con, sql, params)

Expand All @@ -212,6 +215,7 @@ async def mssql_upsert_dict(
complex_helper: Any | None = None,
) -> dict:
"""MERGE-based upsert, returns the row as a dict."""
data = json_encode_values(data)
sql, params = _build_upsert_merge(table_name, data, primary_keys)
row = await _execute_returning(con, sql, params)
assert row is not None
Expand Down Expand Up @@ -241,6 +245,7 @@ async def mssql_upsert_many_dict(
and want a missing row to be a silent no-op rather than create one."""
if not data:
return
data = [json_encode_values(row) for row in data]
fields = list(data[0])
if must_exist:
sql = _build_update_many(table_name, fields, primary_keys)
Expand Down
34 changes: 34 additions & 0 deletions pgdevkit/db/mssql_sql.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,8 @@
from __future__ import annotations

import json
from typing import Any


def ident(name: str) -> str:
"""Bracket-quote a single identifier, doubling any embedded `]`
Expand All @@ -12,3 +15,34 @@ def ident(name: str) -> str:

def qualified(schema: str, table: str) -> str:
return f"{ident(schema)}.{ident(table)}"


def json_encode_value(value: Any) -> Any:
"""Serialize a value destined for a `json`-typed (or legacy
`nvarchar(max)`-storing-JSON) column to text.

mssql-python has no auto-serialization for dict/list parameter values --
binding one directly raises `TypeError: Unsupported parameter type`
(confirmed against mssql-python 1.12.0: its `_map_sql_type` has explicit
branches for every scalar Python type but none for dict/list, and SQL
Server's native `json` type -- a genuine first-class type in current
Azure SQL/SQL Server, unlike the old NVARCHAR(MAX)-plus-OPENJSON()
convention -- has no dedicated ODBC type code in this driver either, so
it's fetched back as plain `str`, indistinguishable from any other text
column). Unlike Postgres, there's no type-registration ambiguity to
resolve here (composite type vs jsonb vs plain array all need different
handling there): MSSQL has no composite types, so a Python dict/list
passed to any MSSQL CRUD call can only sensibly mean "serialize me as
JSON text" -- no per-column-type lookup needed on the write side.

There is deliberately no read-side counterpart: the driver can't tell
us which columns are `json`-typed (it reports the same opaque `str` for
those as for a plain `nvarchar`), so auto-parsing fetched values back
into dict/list would need its own catalog lookup -- a ComplexHelper-like
mechanism this backend intentionally doesn't have. Callers that know a
column is JSON deserialize it themselves with `json.loads()`."""
return json.dumps(value) if isinstance(value, (dict, list)) else value


def json_encode_values(data: dict) -> dict:
return {k: json_encode_value(v) for k, v in data.items()}
19 changes: 8 additions & 11 deletions pgdevkit/testdb/mssql/api.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@

import mssql_python

from ...db.mssql_sql import ident
from ...db.mssql_sql import ident, json_encode_value
from ...dialect import MSSQL
from .. import query
from ..config import ProjectConfig
Expand Down Expand Up @@ -84,21 +84,18 @@ def _run() -> None:
(count,) = cur.fetchone()
if count == len(rows):
return
# Unlike the Postgres path (ComplexHelper-driven composite/enum/JSONB
# conversion), MSSQL fixtures are limited to flat scalar columns for
# this first cut -- there is no MSSQL equivalent to convert into, so
# dict/list values are serialized as JSON text (matching how a
# NVARCHAR(MAX)-typed "JSON column" is conventionally stored on this
# engine) rather than silently dropped or erroring.
# Unlike the Postgres path (ComplexHelper-driven composite/enum
# conversion), MSSQL has no composite/enum equivalent to convert
# into -- dict/list values are serialized as JSON text instead
# (see db/mssql_sql.json_encode_value), matching a `json`-typed or
# legacy NVARCHAR(MAX)-storing-JSON column, rather than silently
# dropped or erroring.
col_names = list(rows[0])
cur.execute(f"DELETE FROM {qualified}")
cols = ", ".join(ident(c) for c in col_names)
placeholders = ", ".join("?" for _ in col_names)
insert_sql = f"INSERT INTO {qualified} ({cols}) VALUES ({placeholders})"
param_rows = [
[json.dumps(row[c]) if isinstance(row[c], (dict, list)) else row[c] for c in col_names]
for row in rows
]
param_rows = [[json_encode_value(row[c]) for c in col_names] for row in rows]
cur.executemany(insert_sql, param_rows)

await asyncio.to_thread(_run)
Expand Down
33 changes: 33 additions & 0 deletions tests/db/test_mssql_crud_live.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,7 @@
from __future__ import annotations

import asyncio
import json
from pathlib import Path

import mssql_python
Expand Down Expand Up @@ -71,3 +72,35 @@ async def _run() -> None:
conn.close()
finally:
clean_testdb(project)


@requires_mssql
def test_json_column_round_trip(tmp_path: Path):
# app.widget.tags is NVARCHAR(MAX) (see fixtures/database_mssql) --
# this is the main way the write-side JSON fix actually gets validated
# against a real driver+server: without db/mssql_sql.json_encode_value,
# mssql-python raises TypeError outright on a raw dict/list parameter
# (confirmed via source inspection, not just assumed), so this insert
# wouldn't merely round-trip wrong, it would crash.
project = _make_project(tmp_path, "mssqljsonlive", "main", engine="mssql")
try:
ensure_testdb(project)
conn = mssql_python.connect(status(project)["dsn"], autocommit=True)

async def _run() -> None:
inserted = await mssql_insert(
conn, ("app", "widget"), {"id": 5, "name": "tagged", "tags": ["red", "blue"]}
)
assert json.loads(inserted["tags"]) == ["red", "blue"]

upserted = await mssql_upsert_dict(
conn, ("app", "widget"), {"id": 5, "name": "tagged", "tags": {"a": 1}}, ["id"]
)
assert json.loads(upserted["tags"]) == {"a": 1}

try:
asyncio.run(_run())
finally:
conn.close()
finally:
clean_testdb(project)
80 changes: 79 additions & 1 deletion tests/db/test_mssql_crud_sql.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,7 @@
from __future__ import annotations

import asyncio

from pgdevkit.db.mssql_crud import (
_build_delete,
_build_insert,
Expand All @@ -9,8 +11,11 @@
_build_update,
_build_update_many,
_build_upsert_merge,
mssql_insert,
mssql_update_dict,
mssql_upsert_dict,
)
from pgdevkit.db.mssql_sql import ident, qualified
from pgdevkit.db.mssql_sql import ident, json_encode_value, json_encode_values, qualified


def test_ident_brackets_and_doubles_embedded_bracket():
Expand Down Expand Up @@ -83,3 +88,76 @@ def test_build_delete_uses_output_deleted_before_where():
sql, params = _build_delete(("dbo", "widget"), {"id": 1})
assert sql == "DELETE FROM [dbo].[widget] OUTPUT DELETED.* WHERE [id] = ?"
assert params == [1]


# JSON support: mssql-python has no auto-serialization for dict/list
# parameter values (binding one raises TypeError -- confirmed against the
# installed driver), and no distinct type code for SQL Server's native
# `json` type either (it's fetched as plain str, indistinguishable from
# nvarchar). See db/mssql_sql.json_encode_value's docstring for the full
# story; these tests cover the write-side serialization it does.


def test_json_encode_value_serializes_dict_and_list():
assert json_encode_value({"a": 1}) == '{"a": 1}'
assert json_encode_value([1, 2]) == "[1, 2]"


def test_json_encode_value_leaves_scalars_untouched():
assert json_encode_value("hello") == "hello"
assert json_encode_value(5) == 5
assert json_encode_value(None) is None


def test_json_encode_values_only_touches_dict_and_list_entries():
result = json_encode_values({"name": "sprocket", "tags": ["a", "b"], "count": 3})
assert result == {"name": "sprocket", "tags": '["a", "b"]', "count": 3}


class _FakeCursor:
"""A minimal double for mssql-python's Cursor -- just enough to capture
what mssql_insert/mssql_update_dict/mssql_upsert_dict actually bind, so
the JSON-encoding wiring can be verified without a real driver/server."""

def __init__(self, row):
self.executed: tuple[str, list] | None = None
self._row = row
self.description = [("id",), ("payload",)]

def execute(self, sql, params):
self.executed = (sql, params)

def fetchone(self):
return self._row

def close(self):
pass


class _FakeConnection:
def __init__(self, row):
self.cursor_obj = _FakeCursor(row)

def cursor(self):
return self.cursor_obj


def test_mssql_insert_serializes_dict_value_before_binding():
con = _FakeConnection(row=(1, '{"a": 1}'))
asyncio.run(mssql_insert(con, ("dbo", "widget"), {"id": 1, "payload": {"a": 1}}))
_, params = con.cursor_obj.executed
assert params == [1, '{"a": 1}']


def test_mssql_update_dict_serializes_dict_value_before_binding():
con = _FakeConnection(row=(1, '{"a": 1}'))
asyncio.run(mssql_update_dict(con, ("dbo", "widget"), {"id": 1, "payload": {"a": 1}}, ["id"]))
_, params = con.cursor_obj.executed
assert params == ['{"a": 1}', 1]


def test_mssql_upsert_dict_serializes_list_value_before_binding():
con = _FakeConnection(row=(1, "[1, 2]"))
asyncio.run(mssql_upsert_dict(con, ("dbo", "widget"), {"id": 1, "payload": [1, 2]}, ["id"]))
_, params = con.cursor_obj.executed
assert params == [1, "[1, 2]"]
15 changes: 15 additions & 0 deletions tests/test_parser_mssql.py
Original file line number Diff line number Diff line change
Expand Up @@ -82,6 +82,21 @@ def test_parse_function_details_tsql_extracts_args_and_body_directly():
assert "return @a + @b" in body


def test_parses_native_json_column_type(tmp_path: Path):
# SQL Server's native `json` type (current Azure SQL/SQL Server 2025+ --
# distinct from the older NVARCHAR(MAX)-plus-OPENJSON() convention).
# sqlglot's tsql dialect already recognizes it as exp.DataType.Type.JSON;
# this just confirms parser.py round-trips it to "json" like any other
# type name, unaffected by the fact that MSSQL has no native enum/
# composite type (Dialect.supports_enums/supports_composites are about
# CREATE TYPE, not built-in column types like this one).
_write(tmp_path, "widget.sql", "CREATE TABLE dbo.widget (id INT PRIMARY KEY, payload JSON NULL);")
schema = parse_directory(tmp_path, dialect="mssql")
cols = {c.name: c for c in schema.tables["dbo.widget"].columns}
assert cols["payload"].data_type == "json"
assert cols["payload"].is_nullable


def test_mssql_enum_style_type_is_ignored_not_erroring(tmp_path: Path):
# A Postgres-style `CREATE TYPE ... AS ENUM` has no T-SQL equivalent --
# parsing it under dialect="mssql" must not raise, and (since sqlglot's
Expand Down
8 changes: 7 additions & 1 deletion tests/testdb/fixtures/database_mssql/app/tables/widget.sql
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,12 @@ IF NOT EXISTS (SELECT 1 FROM sys.tables t JOIN sys.schemas s ON s.schema_id = t.
BEGIN
CREATE TABLE app.widget (
id INT PRIMARY KEY,
name NVARCHAR(100) NOT NULL
name NVARCHAR(100) NOT NULL,
-- NVARCHAR(MAX)-storing-JSON, not the native `json` type (SQL
-- Server 2025+) -- the CI container image (2022) predates it, and
-- this convention is what the write-side JSON-encoding fix in
-- db/mssql_sql.json_encode_value covers either way (mssql-python
-- has no auto dict/list serialization for either column style).
tags NVARCHAR(MAX) NULL
);
END
Loading