From 4d896146343e36d149217d54291eb9b5e58b3019 Mon Sep 17 00:00:00 2001 From: Adrian Ehrsam Date: Mon, 5 Oct 2026 09:15:09 +0200 Subject: [PATCH 1/2] Test update-stats against a live MSSQL container instead of a fake driver, bump to 0.11.0 (#43) Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_01SwHJR2uryShYQnsySXZT6G --- pyproject.toml | 2 +- tests/test_stats_cli.py | 57 -------------------------------- tests/test_stats_mssql_live.py | 59 ++++++++++++++++++++++++++++++++++ uv.lock | 2 +- 4 files changed, 61 insertions(+), 59 deletions(-) create mode 100644 tests/test_stats_mssql_live.py diff --git a/pyproject.toml b/pyproject.toml index 54cb50c..47e56b4 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -11,7 +11,7 @@ packages = ["pgdevkit"] [project] name = "pgdevkit" -version = "0.10.0" +version = "0.11.0" description = "A helper for developing with Postgres" readme = "README.md" requires-python = ">=3.14" diff --git a/tests/test_stats_cli.py b/tests/test_stats_cli.py index 2afe0d5..f1a34ca 100644 --- a/tests/test_stats_cli.py +++ b/tests/test_stats_cli.py @@ -74,60 +74,3 @@ def test_update_stats_then_get_stats(tmp_path: Path): with psycopg.connect(admin, autocommit=True) as con: con.execute(f'DROP DATABASE IF EXISTS "{TEST_DB}"') - -class _FakeMssqlCursor: - def __init__(self, conn): - self.conn, self.description, self.rows = conn, None, [] - - def execute(self, sql, params=()): - self.conn.executed.append(sql) - if "sys.dm_db_partition_stats" in sql: - self.description = [(n,) for n in ("schema", "name", "estimated_rows", "table_bytes", "index_bytes", "total_bytes")] - self.rows = [("dbo", "zeta", 10, 8192, 16384, 24576), ("sys", "junk", 1, 0, 0, 0)] - elif "FROM sys.columns" in sql: - self.description = [(n,) for n in ("name", "base_type", "max_length", "precision", "scale")] - self.rows = [("id", "int", 4, 10, 0), ("note", "nvarchar", 100, 0, 0)] - elif "COUNT_BIG(*)" in sql: - names = ("n", "nn0", "nd0", "w0", "nn1", "nd1", "w1") - self.description, self.rows = [(n,) for n in names], [(10, 10, 10, 4.0, 0, 0, None)] - - def fetchall(self): - return self.rows - - def close(self): - pass - - -class _FakeMssqlConn: - def __init__(self): - self.executed: list[str] = [] - - def cursor(self): - return _FakeMssqlCursor(self) - - def execute(self, sql): - self.executed.append(sql) - - def close(self): - pass - - -def test_update_stats_mssql(tmp_path: Path, monkeypatch): - import mssql_python - - conn = _FakeMssqlConn() - monkeypatch.setattr(mssql_python, "connect", lambda *a, **k: conn) - - r = runner.invoke(app, ["update-stats", str(tmp_path), "--url", "x", "--dialect", "mssql", "--exact", "--analyze"]) - assert r.exit_code == 0, r.output - tables = json.loads((tmp_path / "_stats" / "_tables.json").read_text()) - assert list(tables) == ["dbo.zeta"] # system schema skipped - assert tables["dbo.zeta"]["row_count"] == 10 and tables["dbo.zeta"]["row_count_exact"] is True - assert tables["dbo.zeta"]["total_bytes"] == 24576 - cols = json.loads((tmp_path / "_stats" / "dbo.zeta.json").read_text()) - assert cols["note"]["data_type"] == "nvarchar(50)" and cols["note"]["null_fraction"] == 1 - assert cols["id"]["n_distinct"] == 10 and cols["id"]["avg_width"] == 4 - assert "UPDATE STATISTICS [dbo].[zeta]" in conn.executed - - assert runner.invoke(app, ["update-stats", str(tmp_path), "--url", "x", "--dialect", "mssql", "--table", "dbo.nope"]).exit_code == 2 - assert runner.invoke(app, ["update-stats", str(tmp_path), "--url", "x", "--dialect", "oracle"]).exit_code == 2 diff --git a/tests/test_stats_mssql_live.py b/tests/test_stats_mssql_live.py new file mode 100644 index 0000000..7da71fa --- /dev/null +++ b/tests/test_stats_mssql_live.py @@ -0,0 +1,59 @@ +from __future__ import annotations + +import json +from pathlib import Path + +import mssql_python +import pytest +from typer.testing import CliRunner + +from pgdevkit.cli import app +from pgdevkit.testdb.api import clean_testdb, ensure_testdb, status +from tests.testdb.conftest import _make_project, requires_mssql + +# Selected only by the dedicated `mssql-test` CI job; see test_compare_mssql_live.py. +pytestmark = pytest.mark.mssql + +runner = CliRunner() + + +def _exec(dsn: str, sql: str) -> None: + conn = mssql_python.connect(dsn, autocommit=True) + try: + conn.cursor().execute(sql) + finally: + conn.close() + + +@requires_mssql +def test_update_stats_mssql_live(tmp_path: Path): + project = _make_project(tmp_path, "mssqlstats", "main", engine="mssql") + out = tmp_path / "out" + out.mkdir() + try: + ensure_testdb(project) + dsn = status(project)["dsn"] + _exec(dsn, "INSERT INTO app.widget (id, name, tags) VALUES (2, N'gear', NULL), (3, N'cog', NULL)") + + r = runner.invoke(app, ["update-stats", str(out), "--url", dsn, "--dialect", "mssql", "--exact", "--analyze"]) + assert r.exit_code == 0, r.output + tables = json.loads((out / "_stats" / "_tables.json").read_text()) + assert "app.widget" in tables and not any(k.startswith("sys.") for k in tables) + assert tables["app.widget"]["row_count"] == 3 + assert tables["app.widget"]["row_count_exact"] is True + assert tables["app.widget"]["total_bytes"] > 0 + cols = json.loads((out / "_stats" / "app.widget.json").read_text()) + assert cols["name"]["data_type"] == "nvarchar(100)" + assert cols["id"]["n_distinct"] == 3 and cols["id"]["null_fraction"] == 0 + assert cols["tags"]["null_fraction"] == 1 + + # Without --exact: engine-maintained row count, no column measurements. + r = runner.invoke(app, ["update-stats", str(out), "--url", dsn, "--dialect", "mssql", "--table", "app.widget"]) + assert r.exit_code == 0, r.output + assert json.loads((out / "_stats" / "_tables.json").read_text())["app.widget"]["row_count"] == 3 + assert json.loads((out / "_stats" / "app.widget.json").read_text())["id"]["n_distinct"] is None + + bad = ["update-stats", str(out), "--url", dsn, "--dialect", "mssql", "--table", "app.nope"] + assert runner.invoke(app, bad).exit_code == 2 + finally: + clean_testdb(project) diff --git a/uv.lock b/uv.lock index f07cef7..17a3840 100644 --- a/uv.lock +++ b/uv.lock @@ -313,7 +313,7 @@ wheels = [ [[package]] name = "pgdevkit" -version = "0.10.0" +version = "0.11.0" source = { editable = "." } dependencies = [ { name = "docker" }, From 463174ffa8ad8c0beed705b299dbd9aeec7d74b5 Mon Sep 17 00:00:00 2001 From: Adrian Ehrsam Date: Mon, 5 Oct 2026 09:16:38 +0200 Subject: [PATCH 2/2] Skip json/vector columns in MSSQL exact stats (#43) Co-Authored-By: Claude Sonnet 5.5 Claude-Session: https://claude.ai/code/session_01SwHJR2uryShYQnsySXZT6G --- pgdevkit/stats.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pgdevkit/stats.py b/pgdevkit/stats.py index 7e2e756..c183870 100644 --- a/pgdevkit/stats.py +++ b/pgdevkit/stats.py @@ -80,7 +80,7 @@ def _write_json(path: Path, data: Any) -> None: """ # Types that can't be COUNT(DISTINCT)ed / measured with DATALENGTH. -_MSSQL_UNMEASURABLE = {"text", "ntext", "image", "xml", "geography", "geometry", "hierarchyid", "sql_variant"} +_MSSQL_UNMEASURABLE = {"text", "ntext", "image", "xml", "geography", "geometry", "hierarchyid", "sql_variant", "json", "vector"} def _mssql_q(conn: Any, sql: str, params: tuple = ()) -> list[dict[str, Any]]: