Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 1 addition & 8 deletions ci/get_package_shards.py
Original file line number Diff line number Diff line change
Expand Up @@ -49,14 +49,7 @@
}

# Packages temporarily excluded from CI test execution.
# NOTE: 'sqlalchemy-bigquery' is temporarily excluded to allow testing in this PR
# to complete due to an upstream packaging issue in sqlalchemy (duplicate normalized
# extra name 'mssql-pymssql' under strict uv PEP 621 parsing in sqlalchemy==2.1.0rc2,
# pulled via global UV_PRERELEASE=allow). Awaiting team feedback on a long-term
# solution (e.g. package migration out of the monorepo or adjusting workflow settings).
EXCLUDED_PACKAGES = {
"sqlalchemy-bigquery",
}
EXCLUDED_PACKAGES = set()


def get_package_directories():
Expand Down
17 changes: 16 additions & 1 deletion packages/sqlalchemy-bigquery/docs/conf.py
Original file line number Diff line number Diff line change
Expand Up @@ -359,7 +359,6 @@

# Example configuration for intersphinx: refer to the Python standard library.
intersphinx_mapping = {
"python": ("https://python.readthedocs.org/en/latest/", None),
"google-auth": ("https://googleapis.dev/python/google-auth/latest/", None),
"google.api_core": (
"https://googleapis.dev/python/google-api-core/latest/",
Expand All @@ -370,6 +369,22 @@
"protobuf": ("https://googleapis.dev/python/protobuf/latest/", None),
}

# Check reachability of the Python standard library inventory before attaching it.

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

FYI: This check is in direct response to an issue that came up today: docs.python.org was not available on line for a period of time and was failing tests.

# Because Sphinx is run with `-W` (warnings as errors) in CI, an external network
# failure or upstream outage on docs.python.org would otherwise treat the missing
# inventory as a fatal error and fail the build.
try:
import urllib.error
import urllib.request

with urllib.request.urlopen("https://docs.python.org/3/objects.inv", timeout=2):
intersphinx_mapping["python"] = (
"https://python.readthedocs.org/en/latest/",
None,
)
except (urllib.error.URLError, TimeoutError, OSError):
pass


# Napoleon settings
napoleon_google_docstring = True
Expand Down
33 changes: 22 additions & 11 deletions packages/sqlalchemy-bigquery/noxfile.py
Original file line number Diff line number Diff line change
Expand Up @@ -170,6 +170,11 @@ def wrapper(*args, **kwargs):
nox.options.stop_on_first_error = True
nox.options.error_on_missing_interpreters = True

# NOTE: venv_backend="virtualenv" is used to bypass an upstream packaging issue
# in sqlalchemy (duplicate normalized extra name 'mssql-pymssql' under strict uv
# PEP 621 parsing in sqlalchemy==2.1.0rc2, pulled via global UV_PRERELEASE=allow).
VENV_BACKEND = "virtualenv"


@nox.session(python=DEFAULT_PYTHON_VERSION)
@_calculate_duration
Expand Down Expand Up @@ -289,7 +294,7 @@ def install_unittest_dependencies(session, *constraints):
session.install("-e", ".", *constraints)


@nox.session(python=ALL_PYTHON)
@nox.session(python=ALL_PYTHON, venv_backend=VENV_BACKEND)
@nox.parametrize(
"protobuf_implementation",
["python", "upb"],
Expand Down Expand Up @@ -407,6 +412,7 @@ def _run_system_test_logic(session, test_type):
"mock",
"pytest",
"pytest-rerunfailures",
"pytest-xdist",
"google-cloud-testutils",
"-c",
constraints_path,
Expand All @@ -425,12 +431,17 @@ def _run_system_test_logic(session, test_type):

# Execution logic
if test_type == "compliance":
xdist_args = ["-n=4", "--dist=loadscope"]
if any(arg.startswith(("-n", "--numprocesses")) for arg in session.posargs):
xdist_args = []

session.run(
"py.test",
"-vv",
*xdist_args,
f"--junitxml=compliance_{session.python}_sponge_log.xml",
"--reruns=3",
"--reruns-delay=60",
"--reruns=2",
"--reruns-delay=30",
"--only-rerun=Exceeded rate limits",
"--only-rerun=Already Exists",
"--only-rerun=Not found",
Expand All @@ -450,29 +461,29 @@ def _run_system_test_logic(session, test_type):
)


@nox.session(python="3.12")
@nox.session(python="3.12", venv_backend=VENV_BACKEND)
@nox.parametrize("test_type", ["system", "system_noextras", "compliance"])
@_calculate_duration
def system(session, test_type):
"""Run the system test suite."""
_run_system_test_logic(session, test_type)


@nox.session(python=SYSTEM_TEST_PYTHON_VERSIONS)
@nox.session(python=SYSTEM_TEST_PYTHON_VERSIONS, venv_backend=VENV_BACKEND)
@_calculate_duration
def system_noextras(session):
"""Run the system test suite without extras."""
_run_system_test_logic(session, "system_noextras")


@nox.session(python=SYSTEM_TEST_PYTHON_VERSIONS[-1])
@nox.session(python=DEFAULT_PYTHON_VERSION, venv_backend=VENV_BACKEND)
@_calculate_duration
def compliance(session):
"""Run the SQLAlchemy dialect-compliance system tests"""
_run_system_test_logic(session, "compliance")


@nox.session(python=DEFAULT_PYTHON_VERSION)
@nox.session(python=DEFAULT_PYTHON_VERSION, venv_backend=VENV_BACKEND)
@_calculate_duration
def cover(session):
"""Run the final coverage report.
Expand All @@ -486,7 +497,7 @@ def cover(session):
session.run("coverage", "erase")


@nox.session(python="3.10")
@nox.session(python="3.10", venv_backend=VENV_BACKEND)
@_calculate_duration
def docs(session):
"""Build the docs for this library."""
Expand Down Expand Up @@ -524,7 +535,7 @@ def docs(session):
)


@nox.session(python="3.10")
@nox.session(python="3.10", venv_backend=VENV_BACKEND)
@_calculate_duration
def docfx(session):
"""Build the docfx yaml files for this library."""
Expand Down Expand Up @@ -573,7 +584,7 @@ def docfx(session):
)


@nox.session(python=DEFAULT_PYTHON_VERSION)
@nox.session(python=DEFAULT_PYTHON_VERSION, venv_backend=VENV_BACKEND)
@nox.parametrize(
"protobuf_implementation",
["python", "upb"],
Expand Down Expand Up @@ -697,7 +708,7 @@ def mypy(session):
session.skip("mypy tests are not yet supported")


@nox.session(python=DEFAULT_PYTHON_VERSION)
@nox.session(python=DEFAULT_PYTHON_VERSION, venv_backend=VENV_BACKEND)
@nox.parametrize(
"protobuf_implementation",
["python", "upb"],
Expand Down
2 changes: 1 addition & 1 deletion packages/sqlalchemy-bigquery/sqlalchemy_bigquery/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -600,7 +600,7 @@ def visit_BOOLEAN(self, type_, **kw):
def visit_FLOAT(self, type_, **kw):
return "FLOAT64"

visit_REAL = visit_FLOAT
visit_REAL = visit_DOUBLE = visit_DOUBLE_PRECISION = visit_FLOAT

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

This change is due to an upstream issue in sqlalchemy and how they handle DOUBLE and DOUBLE_PRECISION, which BigQuery does not recognize.

def visit_STRING(self, type_, **kw):
if (type_.length is not None) and isinstance(
Expand Down
95 changes: 95 additions & 0 deletions packages/sqlalchemy-bigquery/sqlalchemy_bigquery/provision.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,95 @@
# Copyright 2026 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

import contextlib
import datetime
import os
import uuid

import google.cloud.bigquery
from sqlalchemy.engine import make_url
from sqlalchemy.testing.provision import (
create_db,

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

sqlalchemy provides decorators to help manage portions of db and url creation.
Most of the functions below are wrapped by the sqlalchemy decorators to align with their expectations and structured internally to handle nuances that are particular to our needs OR the BigQuery API.

drop_db,
follower_url_from_main,
generate_driver_url,
)

try:
import test_utils.prefixer # pragma: NO COVER

prefixer = test_utils.prefixer.Prefixer( # pragma: NO COVER
"python-bigquery-sqlalchemy", "tests/compliance"
)
except ImportError:
prefixer = None


def _dataset_id_from_ident(ident: str) -> str:
"""Derive a deterministic BigQuery dataset ID for an xdist follower ident."""
run_prefix = os.environ.get("COMPLIANCE_RUN_PREFIX")
if not run_prefix:
if prefixer:
run_prefix = prefixer.create_prefix()
else:
now = datetime.datetime.now(datetime.timezone.utc).strftime("%Y%m%d%H%M%S")
run_prefix = f"python_bigquery_sqlalchemy_tests_compliance_{now}_{uuid.uuid4().hex[:6]}"
os.environ["COMPLIANCE_RUN_PREFIX"] = run_prefix
return f"{run_prefix}_{ident}"
Comment thread
chalmerlowe marked this conversation as resolved.


@generate_driver_url.for_db("bigquery")
def _bigquery_generate_driver_url(url, driver, query_str):
url = make_url(url)
if driver and driver != "bigquery":
new_url = url.set(drivername=f"bigquery+{driver}")
else:
new_url = url.set(drivername="bigquery")
if query_str:
new_url = new_url.update_query_string(query_str)
return new_url


@follower_url_from_main.for_db("bigquery")
def _bigquery_follower_url_from_main(url, ident):
url = make_url(url)
dataset_id = _dataset_id_from_ident(ident)
return url.set(database=dataset_id)


def ensure_dataset(dataset_id: str) -> None:
"""Ensure a BigQuery dataset exists with a 1-hour expiration safety net."""
with contextlib.closing(google.cloud.bigquery.Client()) as client:
dataset_ref = google.cloud.bigquery.DatasetReference(client.project, dataset_id)
dataset = google.cloud.bigquery.Dataset(dataset_ref)
dataset.default_table_expiration_ms = 3600 * 1000
client.create_dataset(dataset, exists_ok=True)


def drop_dataset(dataset_id: str) -> None:
"""Drop a BigQuery dataset and its contents if it exists."""
with contextlib.closing(google.cloud.bigquery.Client()) as client:
client.delete_dataset(dataset_id, delete_contents=True, not_found_ok=True)


@create_db.for_db("bigquery")
def _bigquery_create_db(cfg, eng, ident):
dataset_id = _dataset_id_from_ident(ident)
ensure_dataset(dataset_id)


@drop_db.for_db("bigquery")
def _bigquery_drop_db(cfg, eng, ident):
dataset_id = _dataset_id_from_ident(ident)
drop_dataset(dataset_id)
Original file line number Diff line number Diff line change
Expand Up @@ -18,21 +18,21 @@
# CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.

import contextlib
import os
import re
import traceback

import google.cloud.bigquery.dbapi.connection
import test_utils.prefixer
from sqlalchemy.testing import config
from sqlalchemy.testing.plugin.pytestplugin import * # noqa
from sqlalchemy.testing.plugin.pytestplugin import (
pytest_sessionfinish as _pytest_sessionfinish,
)
from sqlalchemy.testing.plugin.pytestplugin import (
pytest_sessionstart as _pytest_sessionstart,
)
from sqlalchemy.testing.plugin import pytestplugin

# flake8: F401, F403 are ignored because SQLAlchemy requires importing its testing
# plugin fixtures and hooks directly into the pytest conftest namespace.
from sqlalchemy.testing.plugin.pytestplugin import * # noqa: F401, F403

import sqlalchemy_bigquery.base
import sqlalchemy_bigquery.provision

sqlalchemy_bigquery.BigQueryDialect.preexecute_autoincrement_sequences = True

Expand All @@ -54,6 +54,7 @@


def visit_delete(self, delete_stmt, *args, **kw):
"""Compile DELETE statements, appending WHERE true during test teardown."""
text = super(sqlalchemy_bigquery.base.BigQueryCompiler, self).visit_delete(
delete_stmt, *args, **kw
)
Expand All @@ -69,19 +70,62 @@ def visit_delete(self, delete_stmt, *args, **kw):
sqlalchemy_bigquery.base.BigQueryCompiler.visit_delete = visit_delete


def _resolve_dataset_id(cfg) -> str:
"""Resolve the dataset ID for the current runner process (worker or master)."""
run_prefix = os.environ["COMPLIANCE_RUN_PREFIX"]
if hasattr(cfg, "workerinput"):
ident = cfg.workerinput.get("follower_ident")
if ident:
return sqlalchemy_bigquery.provision._dataset_id_from_ident(ident)
return f"{run_prefix}_worker"
return f"{run_prefix}_master"


def pytest_configure(config):
"""Configure pytest session, establishing compliance run prefix and target dburi."""
if hasattr(config, "workerinput"):
prefix = config.workerinput.get("compliance_run_prefix")
if prefix:
os.environ["COMPLIANCE_RUN_PREFIX"] = prefix
else:
if "COMPLIANCE_RUN_PREFIX" not in os.environ:
os.environ["COMPLIANCE_RUN_PREFIX"] = prefixer.create_prefix()

dataset_id = _resolve_dataset_id(config)
config.option.dburi = [f"bigquery:///{dataset_id}"]
pytestplugin.pytest_configure(config)


def pytest_configure_node(node):
"""Propagate the compliance run prefix from the controller to worker nodes."""
node.workerinput["compliance_run_prefix"] = os.environ.get("COMPLIANCE_RUN_PREFIX")


def pytest_sessionstart(session):
dataset_id = prefixer.create_prefix()
"""Ensure the target dataset exists before running compliance tests in the session."""
dataset_id = _resolve_dataset_id(session.config)
session.config.option.dburi = [f"bigquery:///{dataset_id}"]
with contextlib.closing(google.cloud.bigquery.Client()) as client:
client.create_dataset(dataset_id)
_pytest_sessionstart(session)
sqlalchemy_bigquery.provision.ensure_dataset(dataset_id)
pytestplugin.pytest_sessionstart(session)


def pytest_sessionfinish(session):
dataset_id = config.db.dialect.dataset_id
_pytest_sessionfinish(session)
"""Tear down datasets and clean up leaked test datasets after session completion."""
if hasattr(session.config, "workerinput"):
# Worker dataset teardown is handled by provision.drop_follower_db
# on the controller node.
pytestplugin.pytest_sessionfinish(session)
return

pytestplugin.pytest_sessionfinish(session)
run_prefix = os.environ.get("COMPLIANCE_RUN_PREFIX")
db = getattr(session.config, "db", None) or getattr(config, "db", None)
if db is not None and hasattr(db.dialect, "dataset_id"):
sqlalchemy_bigquery.provision.drop_dataset(db.dialect.dataset_id)
elif run_prefix:
sqlalchemy_bigquery.provision.drop_dataset(f"{run_prefix}_master")

with contextlib.closing(google.cloud.bigquery.Client()) as client:
client.delete_dataset(dataset_id, delete_contents=True, not_found_ok=True)
for dataset in client.list_datasets():
if prefixer.should_cleanup(dataset.dataset_id):
client.delete_dataset(dataset, delete_contents=True, not_found_ok=True)
Original file line number Diff line number Diff line change
Expand Up @@ -644,6 +644,11 @@ def test_no_results_for_non_returning_insert(cls):
PostCompileParamsTest
) # BQ adds backticks to bind parameters, causing failure of tests TODO: fix this?
del QuotedNameArgumentTest # Quotes aren't allowed in BigQuery table names.
del (
WindowFunctionTest.test_window_rows_between
) # test expects BQ to return sorted results
# Test expects BQ to return sorted results, which it does not do. (renamed to _w_caching in SQLAlchemy 2.1+)
for _test_name in ("test_window_rows_between", "test_window_rows_between_w_caching"):
if hasattr(WindowFunctionTest, _test_name):
delattr(WindowFunctionTest, _test_name)

# BigQuery requires dataset qualification for views, which TableViaSelectTest does not provide.
if "TableViaSelectTest" in globals():
del TableViaSelectTest
Loading
Loading