diff --git a/ci/get_package_shards.py b/ci/get_package_shards.py index d0dd4e7d4ea2..558abd4a99be 100644 --- a/ci/get_package_shards.py +++ b/ci/get_package_shards.py @@ -49,14 +49,7 @@ } # Packages temporarily excluded from CI test execution. -# NOTE: 'sqlalchemy-bigquery' is temporarily excluded to allow testing in this PR -# to complete due to an upstream packaging issue in sqlalchemy (duplicate normalized -# extra name 'mssql-pymssql' under strict uv PEP 621 parsing in sqlalchemy==2.1.0rc2, -# pulled via global UV_PRERELEASE=allow). Awaiting team feedback on a long-term -# solution (e.g. package migration out of the monorepo or adjusting workflow settings). -EXCLUDED_PACKAGES = { - "sqlalchemy-bigquery", -} +EXCLUDED_PACKAGES = set() def get_package_directories(): diff --git a/packages/sqlalchemy-bigquery/docs/conf.py b/packages/sqlalchemy-bigquery/docs/conf.py index 2a3478cc1616..4da3f9a32770 100644 --- a/packages/sqlalchemy-bigquery/docs/conf.py +++ b/packages/sqlalchemy-bigquery/docs/conf.py @@ -359,7 +359,6 @@ # Example configuration for intersphinx: refer to the Python standard library. intersphinx_mapping = { - "python": ("https://python.readthedocs.org/en/latest/", None), "google-auth": ("https://googleapis.dev/python/google-auth/latest/", None), "google.api_core": ( "https://googleapis.dev/python/google-api-core/latest/", @@ -370,6 +369,22 @@ "protobuf": ("https://googleapis.dev/python/protobuf/latest/", None), } +# Check reachability of the Python standard library inventory before attaching it. +# Because Sphinx is run with `-W` (warnings as errors) in CI, an external network +# failure or upstream outage on docs.python.org would otherwise treat the missing +# inventory as a fatal error and fail the build. +try: + import urllib.error + import urllib.request + + with urllib.request.urlopen("https://docs.python.org/3/objects.inv", timeout=2): + intersphinx_mapping["python"] = ( + "https://python.readthedocs.org/en/latest/", + None, + ) +except (urllib.error.URLError, TimeoutError, OSError): + pass + # Napoleon settings napoleon_google_docstring = True diff --git a/packages/sqlalchemy-bigquery/noxfile.py b/packages/sqlalchemy-bigquery/noxfile.py index f9ebdeb8b331..c99032072a4a 100644 --- a/packages/sqlalchemy-bigquery/noxfile.py +++ b/packages/sqlalchemy-bigquery/noxfile.py @@ -170,6 +170,11 @@ def wrapper(*args, **kwargs): nox.options.stop_on_first_error = True nox.options.error_on_missing_interpreters = True +# NOTE: venv_backend="virtualenv" is used to bypass an upstream packaging issue +# in sqlalchemy (duplicate normalized extra name 'mssql-pymssql' under strict uv +# PEP 621 parsing in sqlalchemy==2.1.0rc2, pulled via global UV_PRERELEASE=allow). +VENV_BACKEND = "virtualenv" + @nox.session(python=DEFAULT_PYTHON_VERSION) @_calculate_duration @@ -289,7 +294,7 @@ def install_unittest_dependencies(session, *constraints): session.install("-e", ".", *constraints) -@nox.session(python=ALL_PYTHON) +@nox.session(python=ALL_PYTHON, venv_backend=VENV_BACKEND) @nox.parametrize( "protobuf_implementation", ["python", "upb"], @@ -407,6 +412,7 @@ def _run_system_test_logic(session, test_type): "mock", "pytest", "pytest-rerunfailures", + "pytest-xdist", "google-cloud-testutils", "-c", constraints_path, @@ -425,12 +431,17 @@ def _run_system_test_logic(session, test_type): # Execution logic if test_type == "compliance": + xdist_args = ["-n=4", "--dist=loadscope"] + if any(arg.startswith(("-n", "--numprocesses")) for arg in session.posargs): + xdist_args = [] + session.run( "py.test", "-vv", + *xdist_args, f"--junitxml=compliance_{session.python}_sponge_log.xml", - "--reruns=3", - "--reruns-delay=60", + "--reruns=2", + "--reruns-delay=30", "--only-rerun=Exceeded rate limits", "--only-rerun=Already Exists", "--only-rerun=Not found", @@ -450,7 +461,7 @@ def _run_system_test_logic(session, test_type): ) -@nox.session(python="3.12") +@nox.session(python="3.12", venv_backend=VENV_BACKEND) @nox.parametrize("test_type", ["system", "system_noextras", "compliance"]) @_calculate_duration def system(session, test_type): @@ -458,21 +469,21 @@ def system(session, test_type): _run_system_test_logic(session, test_type) -@nox.session(python=SYSTEM_TEST_PYTHON_VERSIONS) +@nox.session(python=SYSTEM_TEST_PYTHON_VERSIONS, venv_backend=VENV_BACKEND) @_calculate_duration def system_noextras(session): """Run the system test suite without extras.""" _run_system_test_logic(session, "system_noextras") -@nox.session(python=SYSTEM_TEST_PYTHON_VERSIONS[-1]) +@nox.session(python=DEFAULT_PYTHON_VERSION, venv_backend=VENV_BACKEND) @_calculate_duration def compliance(session): """Run the SQLAlchemy dialect-compliance system tests""" _run_system_test_logic(session, "compliance") -@nox.session(python=DEFAULT_PYTHON_VERSION) +@nox.session(python=DEFAULT_PYTHON_VERSION, venv_backend=VENV_BACKEND) @_calculate_duration def cover(session): """Run the final coverage report. @@ -486,7 +497,7 @@ def cover(session): session.run("coverage", "erase") -@nox.session(python="3.10") +@nox.session(python="3.10", venv_backend=VENV_BACKEND) @_calculate_duration def docs(session): """Build the docs for this library.""" @@ -524,7 +535,7 @@ def docs(session): ) -@nox.session(python="3.10") +@nox.session(python="3.10", venv_backend=VENV_BACKEND) @_calculate_duration def docfx(session): """Build the docfx yaml files for this library.""" @@ -573,7 +584,7 @@ def docfx(session): ) -@nox.session(python=DEFAULT_PYTHON_VERSION) +@nox.session(python=DEFAULT_PYTHON_VERSION, venv_backend=VENV_BACKEND) @nox.parametrize( "protobuf_implementation", ["python", "upb"], @@ -697,7 +708,7 @@ def mypy(session): session.skip("mypy tests are not yet supported") -@nox.session(python=DEFAULT_PYTHON_VERSION) +@nox.session(python=DEFAULT_PYTHON_VERSION, venv_backend=VENV_BACKEND) @nox.parametrize( "protobuf_implementation", ["python", "upb"], diff --git a/packages/sqlalchemy-bigquery/sqlalchemy_bigquery/base.py b/packages/sqlalchemy-bigquery/sqlalchemy_bigquery/base.py index e2922890cea0..db11393d97ae 100644 --- a/packages/sqlalchemy-bigquery/sqlalchemy_bigquery/base.py +++ b/packages/sqlalchemy-bigquery/sqlalchemy_bigquery/base.py @@ -600,7 +600,7 @@ def visit_BOOLEAN(self, type_, **kw): def visit_FLOAT(self, type_, **kw): return "FLOAT64" - visit_REAL = visit_FLOAT + visit_REAL = visit_DOUBLE = visit_DOUBLE_PRECISION = visit_FLOAT def visit_STRING(self, type_, **kw): if (type_.length is not None) and isinstance( diff --git a/packages/sqlalchemy-bigquery/sqlalchemy_bigquery/provision.py b/packages/sqlalchemy-bigquery/sqlalchemy_bigquery/provision.py new file mode 100644 index 000000000000..4ed4b1801985 --- /dev/null +++ b/packages/sqlalchemy-bigquery/sqlalchemy_bigquery/provision.py @@ -0,0 +1,95 @@ +# Copyright 2026 Google LLC +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import contextlib +import datetime +import os +import uuid + +import google.cloud.bigquery +from sqlalchemy.engine import make_url +from sqlalchemy.testing.provision import ( + create_db, + drop_db, + follower_url_from_main, + generate_driver_url, +) + +try: + import test_utils.prefixer # pragma: NO COVER + + prefixer = test_utils.prefixer.Prefixer( # pragma: NO COVER + "python-bigquery-sqlalchemy", "tests/compliance" + ) +except ImportError: + prefixer = None + + +def _dataset_id_from_ident(ident: str) -> str: + """Derive a deterministic BigQuery dataset ID for an xdist follower ident.""" + run_prefix = os.environ.get("COMPLIANCE_RUN_PREFIX") + if not run_prefix: + if prefixer: + run_prefix = prefixer.create_prefix() + else: + now = datetime.datetime.now(datetime.timezone.utc).strftime("%Y%m%d%H%M%S") + run_prefix = f"python_bigquery_sqlalchemy_tests_compliance_{now}_{uuid.uuid4().hex[:6]}" + os.environ["COMPLIANCE_RUN_PREFIX"] = run_prefix + return f"{run_prefix}_{ident}" + + +@generate_driver_url.for_db("bigquery") +def _bigquery_generate_driver_url(url, driver, query_str): + url = make_url(url) + if driver and driver != "bigquery": + new_url = url.set(drivername=f"bigquery+{driver}") + else: + new_url = url.set(drivername="bigquery") + if query_str: + new_url = new_url.update_query_string(query_str) + return new_url + + +@follower_url_from_main.for_db("bigquery") +def _bigquery_follower_url_from_main(url, ident): + url = make_url(url) + dataset_id = _dataset_id_from_ident(ident) + return url.set(database=dataset_id) + + +def ensure_dataset(dataset_id: str) -> None: + """Ensure a BigQuery dataset exists with a 1-hour expiration safety net.""" + with contextlib.closing(google.cloud.bigquery.Client()) as client: + dataset_ref = google.cloud.bigquery.DatasetReference(client.project, dataset_id) + dataset = google.cloud.bigquery.Dataset(dataset_ref) + dataset.default_table_expiration_ms = 3600 * 1000 + client.create_dataset(dataset, exists_ok=True) + + +def drop_dataset(dataset_id: str) -> None: + """Drop a BigQuery dataset and its contents if it exists.""" + with contextlib.closing(google.cloud.bigquery.Client()) as client: + client.delete_dataset(dataset_id, delete_contents=True, not_found_ok=True) + + +@create_db.for_db("bigquery") +def _bigquery_create_db(cfg, eng, ident): + dataset_id = _dataset_id_from_ident(ident) + ensure_dataset(dataset_id) + + +@drop_db.for_db("bigquery") +def _bigquery_drop_db(cfg, eng, ident): + dataset_id = _dataset_id_from_ident(ident) + drop_dataset(dataset_id) diff --git a/packages/sqlalchemy-bigquery/tests/sqlalchemy_dialect_compliance/conftest.py b/packages/sqlalchemy-bigquery/tests/sqlalchemy_dialect_compliance/conftest.py index c5d69ced03dd..f19672222f91 100644 --- a/packages/sqlalchemy-bigquery/tests/sqlalchemy_dialect_compliance/conftest.py +++ b/packages/sqlalchemy-bigquery/tests/sqlalchemy_dialect_compliance/conftest.py @@ -18,21 +18,21 @@ # CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. import contextlib +import os import re import traceback import google.cloud.bigquery.dbapi.connection import test_utils.prefixer from sqlalchemy.testing import config -from sqlalchemy.testing.plugin.pytestplugin import * # noqa -from sqlalchemy.testing.plugin.pytestplugin import ( - pytest_sessionfinish as _pytest_sessionfinish, -) -from sqlalchemy.testing.plugin.pytestplugin import ( - pytest_sessionstart as _pytest_sessionstart, -) +from sqlalchemy.testing.plugin import pytestplugin + +# flake8: F401, F403 are ignored because SQLAlchemy requires importing its testing +# plugin fixtures and hooks directly into the pytest conftest namespace. +from sqlalchemy.testing.plugin.pytestplugin import * # noqa: F401, F403 import sqlalchemy_bigquery.base +import sqlalchemy_bigquery.provision sqlalchemy_bigquery.BigQueryDialect.preexecute_autoincrement_sequences = True @@ -54,6 +54,7 @@ def visit_delete(self, delete_stmt, *args, **kw): + """Compile DELETE statements, appending WHERE true during test teardown.""" text = super(sqlalchemy_bigquery.base.BigQueryCompiler, self).visit_delete( delete_stmt, *args, **kw ) @@ -69,19 +70,62 @@ def visit_delete(self, delete_stmt, *args, **kw): sqlalchemy_bigquery.base.BigQueryCompiler.visit_delete = visit_delete +def _resolve_dataset_id(cfg) -> str: + """Resolve the dataset ID for the current runner process (worker or master).""" + run_prefix = os.environ["COMPLIANCE_RUN_PREFIX"] + if hasattr(cfg, "workerinput"): + ident = cfg.workerinput.get("follower_ident") + if ident: + return sqlalchemy_bigquery.provision._dataset_id_from_ident(ident) + return f"{run_prefix}_worker" + return f"{run_prefix}_master" + + +def pytest_configure(config): + """Configure pytest session, establishing compliance run prefix and target dburi.""" + if hasattr(config, "workerinput"): + prefix = config.workerinput.get("compliance_run_prefix") + if prefix: + os.environ["COMPLIANCE_RUN_PREFIX"] = prefix + else: + if "COMPLIANCE_RUN_PREFIX" not in os.environ: + os.environ["COMPLIANCE_RUN_PREFIX"] = prefixer.create_prefix() + + dataset_id = _resolve_dataset_id(config) + config.option.dburi = [f"bigquery:///{dataset_id}"] + pytestplugin.pytest_configure(config) + + +def pytest_configure_node(node): + """Propagate the compliance run prefix from the controller to worker nodes.""" + node.workerinput["compliance_run_prefix"] = os.environ.get("COMPLIANCE_RUN_PREFIX") + + def pytest_sessionstart(session): - dataset_id = prefixer.create_prefix() + """Ensure the target dataset exists before running compliance tests in the session.""" + dataset_id = _resolve_dataset_id(session.config) session.config.option.dburi = [f"bigquery:///{dataset_id}"] - with contextlib.closing(google.cloud.bigquery.Client()) as client: - client.create_dataset(dataset_id) - _pytest_sessionstart(session) + sqlalchemy_bigquery.provision.ensure_dataset(dataset_id) + pytestplugin.pytest_sessionstart(session) def pytest_sessionfinish(session): - dataset_id = config.db.dialect.dataset_id - _pytest_sessionfinish(session) + """Tear down datasets and clean up leaked test datasets after session completion.""" + if hasattr(session.config, "workerinput"): + # Worker dataset teardown is handled by provision.drop_follower_db + # on the controller node. + pytestplugin.pytest_sessionfinish(session) + return + + pytestplugin.pytest_sessionfinish(session) + run_prefix = os.environ.get("COMPLIANCE_RUN_PREFIX") + db = getattr(session.config, "db", None) or getattr(config, "db", None) + if db is not None and hasattr(db.dialect, "dataset_id"): + sqlalchemy_bigquery.provision.drop_dataset(db.dialect.dataset_id) + elif run_prefix: + sqlalchemy_bigquery.provision.drop_dataset(f"{run_prefix}_master") + with contextlib.closing(google.cloud.bigquery.Client()) as client: - client.delete_dataset(dataset_id, delete_contents=True, not_found_ok=True) for dataset in client.list_datasets(): if prefixer.should_cleanup(dataset.dataset_id): client.delete_dataset(dataset, delete_contents=True, not_found_ok=True) diff --git a/packages/sqlalchemy-bigquery/tests/sqlalchemy_dialect_compliance/test_dialect_compliance.py b/packages/sqlalchemy-bigquery/tests/sqlalchemy_dialect_compliance/test_dialect_compliance.py index dde343c62429..b449de2d9aec 100644 --- a/packages/sqlalchemy-bigquery/tests/sqlalchemy_dialect_compliance/test_dialect_compliance.py +++ b/packages/sqlalchemy-bigquery/tests/sqlalchemy_dialect_compliance/test_dialect_compliance.py @@ -644,6 +644,11 @@ def test_no_results_for_non_returning_insert(cls): PostCompileParamsTest ) # BQ adds backticks to bind parameters, causing failure of tests TODO: fix this? del QuotedNameArgumentTest # Quotes aren't allowed in BigQuery table names. -del ( - WindowFunctionTest.test_window_rows_between -) # test expects BQ to return sorted results +# Test expects BQ to return sorted results, which it does not do. (renamed to _w_caching in SQLAlchemy 2.1+) +for _test_name in ("test_window_rows_between", "test_window_rows_between_w_caching"): + if hasattr(WindowFunctionTest, _test_name): + delattr(WindowFunctionTest, _test_name) + +# BigQuery requires dataset qualification for views, which TableViaSelectTest does not provide. +if "TableViaSelectTest" in globals(): + del TableViaSelectTest diff --git a/packages/sqlalchemy-bigquery/tests/unit/test_compiler.py b/packages/sqlalchemy-bigquery/tests/unit/test_compiler.py index 77f55dd7bf80..4fae78e7b1e6 100644 --- a/packages/sqlalchemy-bigquery/tests/unit/test_compiler.py +++ b/packages/sqlalchemy-bigquery/tests/unit/test_compiler.py @@ -450,3 +450,40 @@ def compile_custom_intersect(element, compiler, **kwargs): found_sql = q.compile(faux_conn).string assert found_sql == expected_sql + + +def test_type_compiler_double_methods(): + from sqlalchemy_bigquery.base import BigQueryTypeCompiler + + assert BigQueryTypeCompiler.visit_DOUBLE == BigQueryTypeCompiler.visit_FLOAT + assert ( + BigQueryTypeCompiler.visit_DOUBLE_PRECISION == BigQueryTypeCompiler.visit_FLOAT + ) + + +_double_types = [ + getattr(sqlalchemy, name) + for name in ("DOUBLE", "DOUBLE_PRECISION", "Double") + if hasattr(sqlalchemy, name) +] + + +@pytest.mark.skipif( + not _double_types, + reason="DOUBLE types not present in this SQLAlchemy version", +) +@pytest.mark.parametrize("type_cls", _double_types or [None]) +def test_double_types_compile_to_float64(type_cls): + import sqlalchemy_bigquery + + dialect = sqlalchemy_bigquery.BigQueryDialect() + assert dialect.type_compiler.process(type_cls()) == "FLOAT64" + + +def test_float_literal_pyformat_bindparam(): + import sqlalchemy_bigquery + + dialect = sqlalchemy_bigquery.BigQueryDialect(paramstyle="pyformat") + stmt = sqlalchemy.select(sqlalchemy.literal(15.7563)) + compiled = stmt.compile(dialect=dialect) + assert "%(param_1:FLOAT64)s" in str(compiled) diff --git a/packages/sqlalchemy-bigquery/tests/unit/test_provision.py b/packages/sqlalchemy-bigquery/tests/unit/test_provision.py new file mode 100644 index 000000000000..6004f7e541f2 --- /dev/null +++ b/packages/sqlalchemy-bigquery/tests/unit/test_provision.py @@ -0,0 +1,128 @@ +# Copyright 2026 Google LLC +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import os +from unittest import mock + +import pytest +from sqlalchemy.engine import make_url + +from sqlalchemy_bigquery import provision + + +def test_dataset_id_from_ident_with_existing_env(monkeypatch): + monkeypatch.setenv("COMPLIANCE_RUN_PREFIX", "custom_prefix") + dataset_id = provision._dataset_id_from_ident("gw0") + assert dataset_id == "custom_prefix_gw0" + + +def test_dataset_id_from_ident_without_env(monkeypatch): + monkeypatch.delenv("COMPLIANCE_RUN_PREFIX", raising=False) + monkeypatch.setattr(provision, "prefixer", None) + dataset_id = provision._dataset_id_from_ident("gw1") + cached_prefix = os.environ.get("COMPLIANCE_RUN_PREFIX") + assert cached_prefix is not None + assert dataset_id == f"{cached_prefix}_gw1" + + +def test_dataset_id_from_ident_with_prefixer(monkeypatch): + monkeypatch.delenv("COMPLIANCE_RUN_PREFIX", raising=False) + mock_prefixer = mock.Mock() + mock_prefixer.create_prefix.return_value = "mock_pfx" + monkeypatch.setattr(provision, "prefixer", mock_prefixer) + dataset_id = provision._dataset_id_from_ident("gw5") + assert dataset_id == "mock_pfx_gw5" + assert os.environ.get("COMPLIANCE_RUN_PREFIX") == "mock_pfx" + + +@pytest.mark.parametrize( + "driver, query_str, expected_drivername, expected_query", + [ + (None, None, "bigquery", {}), + ("bigquery", None, "bigquery", {}), + ("bigquery", "param=1", "bigquery", {"param": "1"}), + ("custom", None, "bigquery+custom", {}), + ("custom", "param=2", "bigquery+custom", {"param": "2"}), + ], +) +def test_generate_driver_url(driver, query_str, expected_drivername, expected_query): + base_url = "bigquery:///test_master" + result = provision._bigquery_generate_driver_url(base_url, driver, query_str) + assert result.drivername == expected_drivername + if expected_query: + for k, v in expected_query.items(): + assert result.query.get(k) == v + + +def test_follower_url_from_main(monkeypatch): + monkeypatch.setenv("COMPLIANCE_RUN_PREFIX", "run_123") + base_url = make_url("bigquery:///master_dataset") + follower_url = provision._bigquery_follower_url_from_main(base_url, "gw2") + assert follower_url.database == "run_123_gw2" + + +def test_create_db(monkeypatch): + monkeypatch.setenv("COMPLIANCE_RUN_PREFIX", "run_456") + mock_client = mock.MagicMock() + mock_client.project = "test-project" + mock_cfg = mock.Mock() + mock_cfg.db.url = make_url("bigquery:///test_dataset") + + with mock.patch("google.cloud.bigquery.Client", return_value=mock_client): + provision._bigquery_create_db(mock_cfg, mock.Mock(), "gw3") + + mock_client.create_dataset.assert_called_once() + created_dataset = mock_client.create_dataset.call_args[0][0] + assert created_dataset.dataset_id == "run_456_gw3" + assert created_dataset.default_table_expiration_ms == 3600 * 1000 + assert mock_client.create_dataset.call_args[1].get("exists_ok") is True + + +def test_drop_db(monkeypatch): + monkeypatch.setenv("COMPLIANCE_RUN_PREFIX", "run_789") + mock_client = mock.MagicMock() + mock_cfg = mock.Mock() + mock_cfg.db.url = make_url("bigquery:///test_dataset") + + with mock.patch("google.cloud.bigquery.Client", return_value=mock_client): + provision._bigquery_drop_db(mock_cfg, mock.Mock(), "gw4") + + mock_client.delete_dataset.assert_called_once_with( + "run_789_gw4", delete_contents=True, not_found_ok=True + ) + + +def test_ensure_dataset(): + mock_client = mock.MagicMock() + mock_client.project = "test-project" + + with mock.patch("google.cloud.bigquery.Client", return_value=mock_client): + provision.ensure_dataset("custom_dataset_123") + + mock_client.create_dataset.assert_called_once() + created = mock_client.create_dataset.call_args[0][0] + assert created.dataset_id == "custom_dataset_123" + assert created.default_table_expiration_ms == 3600 * 1000 + assert mock_client.create_dataset.call_args[1].get("exists_ok") is True + + +def test_drop_dataset(): + mock_client = mock.MagicMock() + + with mock.patch("google.cloud.bigquery.Client", return_value=mock_client): + provision.drop_dataset("custom_dataset_456") + + mock_client.delete_dataset.assert_called_once_with( + "custom_dataset_456", delete_contents=True, not_found_ok=True + )