Repository navigation
perf(sqlalchemy-bigquery): parallelize compliance tests with worker-partitioned datasets and pytest-xdist #18534
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
b6ee424
50ba897
a4e8eab
f0c1f3b
9b3a5fa
82e4027
0ab9844
796b5ef
afcad64
f2c9fea
8ac2d66
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -600,7 +600,7 @@ def visit_BOOLEAN(self, type_, **kw): | |
| def visit_FLOAT(self, type_, **kw): | ||
| return "FLOAT64" | ||
|
|
||
| visit_REAL = visit_FLOAT | ||
| visit_REAL = visit_DOUBLE = visit_DOUBLE_PRECISION = visit_FLOAT | ||
|
|
||
|
Contributor
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. This change is due to an upstream issue in sqlalchemy and how they handle DOUBLE and DOUBLE_PRECISION, which BigQuery does not recognize. |
||
| def visit_STRING(self, type_, **kw): | ||
| if (type_.length is not None) and isinstance( | ||
|
|
||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,95 @@ | ||
| # Copyright 2026 Google LLC | ||
| # | ||
| # Licensed under the Apache License, Version 2.0 (the "License"); | ||
| # you may not use this file except in compliance with the License. | ||
| # You may obtain a copy of the License at | ||
| # | ||
| # http://www.apache.org/licenses/LICENSE-2.0 | ||
| # | ||
| # Unless required by applicable law or agreed to in writing, software | ||
| # distributed under the License is distributed on an "AS IS" BASIS, | ||
| # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. | ||
| # See the License for the specific language governing permissions and | ||
| # limitations under the License. | ||
|
|
||
| import contextlib | ||
| import datetime | ||
| import os | ||
| import uuid | ||
|
|
||
| import google.cloud.bigquery | ||
| from sqlalchemy.engine import make_url | ||
| from sqlalchemy.testing.provision import ( | ||
| create_db, | ||
|
Contributor
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. sqlalchemy provides decorators to help manage portions of db and url creation. |
||
| drop_db, | ||
| follower_url_from_main, | ||
| generate_driver_url, | ||
| ) | ||
|
|
||
| try: | ||
| import test_utils.prefixer # pragma: NO COVER | ||
|
|
||
| prefixer = test_utils.prefixer.Prefixer( # pragma: NO COVER | ||
| "python-bigquery-sqlalchemy", "tests/compliance" | ||
| ) | ||
| except ImportError: | ||
| prefixer = None | ||
|
|
||
|
|
||
| def _dataset_id_from_ident(ident: str) -> str: | ||
| """Derive a deterministic BigQuery dataset ID for an xdist follower ident.""" | ||
| run_prefix = os.environ.get("COMPLIANCE_RUN_PREFIX") | ||
| if not run_prefix: | ||
| if prefixer: | ||
| run_prefix = prefixer.create_prefix() | ||
| else: | ||
| now = datetime.datetime.now(datetime.timezone.utc).strftime("%Y%m%d%H%M%S") | ||
| run_prefix = f"python_bigquery_sqlalchemy_tests_compliance_{now}_{uuid.uuid4().hex[:6]}" | ||
| os.environ["COMPLIANCE_RUN_PREFIX"] = run_prefix | ||
| return f"{run_prefix}_{ident}" | ||
|
chalmerlowe marked this conversation as resolved.
|
||
|
|
||
|
|
||
| @generate_driver_url.for_db("bigquery") | ||
| def _bigquery_generate_driver_url(url, driver, query_str): | ||
| url = make_url(url) | ||
| if driver and driver != "bigquery": | ||
| new_url = url.set(drivername=f"bigquery+{driver}") | ||
| else: | ||
| new_url = url.set(drivername="bigquery") | ||
| if query_str: | ||
| new_url = new_url.update_query_string(query_str) | ||
| return new_url | ||
|
|
||
|
|
||
| @follower_url_from_main.for_db("bigquery") | ||
| def _bigquery_follower_url_from_main(url, ident): | ||
| url = make_url(url) | ||
| dataset_id = _dataset_id_from_ident(ident) | ||
| return url.set(database=dataset_id) | ||
|
|
||
|
|
||
| def ensure_dataset(dataset_id: str) -> None: | ||
| """Ensure a BigQuery dataset exists with a 1-hour expiration safety net.""" | ||
| with contextlib.closing(google.cloud.bigquery.Client()) as client: | ||
| dataset_ref = google.cloud.bigquery.DatasetReference(client.project, dataset_id) | ||
| dataset = google.cloud.bigquery.Dataset(dataset_ref) | ||
| dataset.default_table_expiration_ms = 3600 * 1000 | ||
| client.create_dataset(dataset, exists_ok=True) | ||
|
|
||
|
|
||
| def drop_dataset(dataset_id: str) -> None: | ||
| """Drop a BigQuery dataset and its contents if it exists.""" | ||
| with contextlib.closing(google.cloud.bigquery.Client()) as client: | ||
| client.delete_dataset(dataset_id, delete_contents=True, not_found_ok=True) | ||
|
|
||
|
|
||
| @create_db.for_db("bigquery") | ||
| def _bigquery_create_db(cfg, eng, ident): | ||
| dataset_id = _dataset_id_from_ident(ident) | ||
| ensure_dataset(dataset_id) | ||
|
|
||
|
|
||
| @drop_db.for_db("bigquery") | ||
| def _bigquery_drop_db(cfg, eng, ident): | ||
| dataset_id = _dataset_id_from_ident(ident) | ||
| drop_dataset(dataset_id) | ||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
FYI: This check is in direct response to an issue that came up today: docs.python.org was not available on line for a period of time and was failing tests.