Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 17 additions & 0 deletions automated_ingestion/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,23 @@
the first source; ``shared/`` holds what later sources reuse.

See ``docs/automated-ingestion-pipeline-plan.md``.

Importing this package puts the repository root on ``sys.path``. That is
unusual and deliberate. The Dagster+ image copies the repository to
``/opt/dagster/app`` but never installs it -- the generated requirements omit
the project, and the build template only runs ``pip install .`` when a
``setup.py`` exists -- so ``db`` and ``domain`` are importable only while that
directory happens to be on the path. It is, when Dagster loads the code
location; it is not guaranteed in the separate process that executes a step,
which is where the loader's imports actually run. Locally the editable install
hides the difference entirely, so the failure appears only once deployed.
"""

import sys as _sys
from pathlib import Path as _Path

_REPOSITORY_ROOT = _Path(__file__).resolve().parent.parent
if str(_REPOSITORY_ROOT) not in _sys.path:
_sys.path.insert(0, str(_REPOSITORY_ROOT))

# ============= EOF =============================================
37 changes: 37 additions & 0 deletions automated_ingestion/iac/main.tf
Original file line number Diff line number Diff line change
Expand Up @@ -101,3 +101,40 @@ resource "google_storage_bucket_iam_member" "ingestion_object_admin" {
role = "roles/storage.objectAdmin"
member = "serviceAccount:${google_service_account.ingestion.email}"
}

# Database access for the ingestion service account.
#
# Only created when `cloud_sql_instance` is set, so the storage half of this
# configuration can be applied before the database half is decided.
#
# These grants are what make IAM database authentication work. Without them the
# Postgres role in automated_ingestion/sql/ingestion_role.sql exists but cannot
# be reached: the connector fails while acquiring a token, which surfaces as an
# authentication error and reads like a missing GRANT.
resource "google_project_iam_member" "ingestion_cloudsql_client" {
count = var.cloud_sql_instance == null ? 0 : 1

project = var.project_id
role = "roles/cloudsql.client"
member = "serviceAccount:${google_service_account.ingestion.email}"
}

resource "google_project_iam_member" "ingestion_cloudsql_instance_user" {
count = var.cloud_sql_instance == null ? 0 : 1

project = var.project_id
role = "roles/cloudsql.instanceUser"
member = "serviceAccount:${google_service_account.ingestion.email}"
}

# Registers the service account as a database user. The Postgres role itself,
# and its grants, come from ingestion_role.sql -- this only makes the login
# possible.
resource "google_sql_user" "ingestion" {
count = var.cloud_sql_instance == null ? 0 : 1

name = trimsuffix(google_service_account.ingestion.email, ".gserviceaccount.com")
instance = var.cloud_sql_instance
project = var.project_id
type = "CLOUD_IAM_SERVICE_ACCOUNT"
}
6 changes: 6 additions & 0 deletions automated_ingestion/iac/variables.tf
Original file line number Diff line number Diff line change
Expand Up @@ -14,3 +14,9 @@ variable "bucket_location" {
description = "Bucket location. US-CENTRAL1 keeps the raw zone in the same region as Cloud SQL, so replay reads do not cross regions."
default = "US-CENTRAL1"
}

variable "cloud_sql_instance" {
type = string
description = "Cloud SQL instance name for the IAM database user. Leave null to skip the database grants entirely -- useful before the instance is known, or when using password authentication instead."
default = null
}
88 changes: 88 additions & 0 deletions automated_ingestion/scripts/set_code_location_env.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,88 @@
#!/usr/bin/env bash
# Set every environment variable the ocotillo-automated-ingestion code location
# needs, in one pass.
#
# Requires a Dagster+ *user* token -- an agent token authenticates but is not
# authorized for these mutations, and dg reports that as an unhelpful KeyError:
# dg plus config set --api-token 'user:...'
#
# Secrets are never passed as arguments. `--from-local-env` reads them from this
# shell, so nothing sensitive reaches the command line, your shell history, or
# the Dagster+ audit log's argument capture. Export them first:
#
# read -rs "DIVERHUB_USERNAME?Diver-HUB username: "; echo
# read -rs "DIVERHUB_PASSWORD?Diver-HUB password: "; echo
# export DIVERHUB_USERNAME DIVERHUB_PASSWORD
#
# Variables are scoped to this code location, not the deployment. If an earlier
# run set them with --global, delete those deployment-level entries in the
# Dagster+ UI afterwards -- otherwise both exist and which one wins is not
# obvious from either place.
#
# Usage:
# ./automated_ingestion/scripts/set_code_location_env.sh storage
# ./automated_ingestion/scripts/set_code_location_env.sh vendor
# ./automated_ingestion/scripts/set_code_location_env.sh database
#
# The phases are separate on purpose. `database` should wait until
# automated_ingestion/sql/ingestion_role.sql has been run: setting CLOUD_SQL_*
# against a role that does not exist yet makes database_connectivity fail in a
# way that looks like the serverless-to-Cloud-SQL problem it is meant to test.
set -euo pipefail

DG="uv run --with dagster-dg-cli dg"
PHASE="${1:-}"

# No --global: that sets the variable at deployment level, where every other
# code location in this deployment can read it. This deployment is shared, so
# the vendor and database credentials stay scoped to this location.
set_var() { echo " $1"; $DG plus create env "$@" -y >/dev/null; }

case "$PHASE" in
storage)
echo "Raw-zone buckets (different value per scope):"
set_var INGESTION_GCS_BUCKET ocotillo-ingestion-production --scope full
set_var INGESTION_GCS_BUCKET ocotillo-ingestion-staging --scope branch
;;
vendor)
: "${DIVERHUB_USERNAME:?export it first, see the header}"
: "${DIVERHUB_PASSWORD:?export it first, see the header}"
echo "Diver-HUB credentials (values read from this shell, not echoed):"
set_var DIVERHUB_USERNAME --from-local-env
set_var DIVERHUB_PASSWORD --from-local-env
;;
database)
: "${CLOUD_SQL_INSTANCE_NAME:?export it first}"
: "${CLOUD_SQL_DATABASE:?export it first}"
echo "Cloud SQL connection:"
set_var DB_DRIVER cloudsql
set_var CLOUD_SQL_IP_TYPE public
set_var CLOUD_SQL_INSTANCE_NAME --from-local-env
set_var CLOUD_SQL_DATABASE --from-local-env

# CLOUD_SQL_USER means different things in the two auth modes, and db/engine.py
# passes it straight to the connector either way. Under IAM auth it must be the
# service account with the .gserviceaccount.com suffix stripped; a plain
# Postgres role name there fails as an authentication error that reads like a
# missing grant. Deriving it here keeps the two settings from contradicting
# each other.
if [ -n "${CLOUD_SQL_PASSWORD:-}" ]; then
echo " (password auth)"
set_var CLOUD_SQL_IAM_AUTH 0
set_var CLOUD_SQL_USER ocotillo_ingestion
set_var CLOUD_SQL_PASSWORD --from-local-env
else
IAM_SA="${INGESTION_SERVICE_ACCOUNT:-ocotillo-ingestion@waterdatainitiative-271000.iam.gserviceaccount.com}"
IAM_USER="${IAM_SA%.gserviceaccount.com}"
echo " (IAM auth as ${IAM_USER})"
set_var CLOUD_SQL_IAM_AUTH 1
set_var CLOUD_SQL_USER "$IAM_USER"
fi
;;
*)
echo "usage: $0 {storage|vendor|database}" >&2
exit 64
;;
esac

echo "Done. Verify in Dagster+ under Deployment -> Environment variables."
27 changes: 19 additions & 8 deletions automated_ingestion/sql/ingestion_role.sql
Original file line number Diff line number Diff line change
Expand Up @@ -11,16 +11,27 @@
-- and NMW_* tables, so a bug in an adapter cannot corrupt data no ingestion
-- path should ever reach.

-- Set the password out of band; do not commit it. It belongs in Secret
-- Manager alongside internal-ogc-api-keys.
-- CREATE ROLE ocotillo_ingestion LOGIN PASSWORD '...';
-- IAM authentication is the configured path, and the reason is that it removes
-- the credential rather than rotating it: Cloud SQL mints a short-lived token
-- from the service account, so there is no password to store in Dagster+, in
-- Secret Manager, or here.
--
-- The role name is the service account with the .gserviceaccount.com suffix
-- stripped. That exact string is also what CLOUD_SQL_USER must be set to --
-- db/engine.py passes it straight to the connector, and a plain role name there
-- fails as an authentication error that reads like a missing grant.
--
-- CREATE ROLE "ocotillo-ingestion@waterdatainitiative-271000.iam" WITH LOGIN;
-- GRANT cloudsqliamuser TO "ocotillo-ingestion@waterdatainitiative-271000.iam";
--
-- Or, preferred, use IAM database authentication and create the role for the
-- service account instead, so there is no password to rotate:
-- CREATE ROLE "ocotillo-ingestion@PROJECT.iam" WITH LOGIN;
-- GRANT cloudsqliamuser TO "ocotillo-ingestion@PROJECT.iam";
-- Password authentication, if IAM is ever unavailable. Set the password out of
-- band and store it in Secret Manager; never commit it, and set
-- CLOUD_SQL_IAM_AUTH=0 so the two settings agree.
--
-- CREATE ROLE ocotillo_ingestion LOGIN PASSWORD '...';

\set role_name ocotillo_ingestion
-- Set to match whichever role was created above.
\set role_name "ocotillo-ingestion@waterdatainitiative-271000.iam"

GRANT CONNECT ON DATABASE :"db_name" TO :"role_name";
GRANT USAGE ON SCHEMA public TO :"role_name";
Expand Down
36 changes: 36 additions & 0 deletions automated_ingestion/tests/test_connectivity.py
Original file line number Diff line number Diff line change
Expand Up @@ -57,4 +57,40 @@ def test_loading_definitions_does_not_import_db_engine():
assert result.stdout.strip() == "False", result.stdout


def test_importing_the_package_makes_the_repository_importable():
# The deployed image never installs this project, so `db` and `domain`
# resolve only if the repository root is on sys.path. Locally an editable
# install provides that and hides the difference, which is why this failed
# only once deployed -- the code location loaded fine and the step that
# imported db died.
import sys

import automated_ingestion

assert str(automated_ingestion._REPOSITORY_ROOT) in sys.path


def test_db_imports_from_an_unrelated_working_directory():
# Reproduces the deployed condition: a process whose cwd is not the
# repository. The lazy imports in the resource and the connectivity asset
# run at step execution, not at load, so this is the path that broke.
import subprocess
import sys

result = subprocess.run(
[
sys.executable,
"-c",
"import automated_ingestion; "
"from db.transducer import TransducerObservation; "
"print(TransducerObservation.__tablename__)",
],
capture_output=True,
text=True,
cwd="/",
)
assert result.returncode == 0, result.stderr
assert "transducer_observation" in result.stdout


# ============= EOF =============================================
10 changes: 10 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -131,6 +131,16 @@ packages = [
[tool.dagster]
module_name = "automated_ingestion.defs.definitions"

# Required by the `dg` CLI, which refuses to run outside a directory it
# recognises as a project. `dagster_cloud.yaml` names the entry point
# explicitly, so this does not affect what the deployed code location loads --
# it only lets `dg plus` manage environment variables from this checkout.
[tool.dg]
directory_type = "project"

[tool.dg.project]
root_module = "automated_ingestion"

# Bare `--cov` measures every imported module, which pulls the whole virtualenv
# into the report. Scope it to first-party code instead. Keep this a single
# source root -- listing each package separately makes coverage treat every
Expand Down
Loading