Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 24 additions & 0 deletions opentofu/gcp/gke/configure/data.tf
Original file line number Diff line number Diff line change
Expand Up @@ -39,3 +39,27 @@ data "google_secret_manager_secret_version" "openbao_root_token" {
secret = var.openbao_root_token_secret_name
project = var.project_id
}

# This directory's ZITADEL project id when the cluster HOSTS it (GCP parity GP-4).
# A fresh directory gets a new id every build; zitadel-oidc-clients.sh publishes
# it here in stage 3. Read at plan time so this stack's later apply agrees with
# the value the sync patched into the ConfigMap. Listed before it is read: absent
# is "first build", and var.zitadel_project_id stands in until stage 3.
#
# The gate checks that the SECRET exists, not that it has a VERSION -- the
# provider has no version-listing data source. store_write creates the secret
# then adds a version, so a failed add leaves a version-less secret and this plan
# then fails on "latest" not found. Recovery: re-run zitadel-oidc-clients.sh
# sync --apply, or delete the empty secret
# (gcloud secrets delete zitadel-project-id --project <id>), then re-apply.
data "google_secret_manager_secrets" "zitadel_project" {
count = var.deploy_identity_provider ? 1 : 0
project = var.project_id
filter = "name:zitadel-project-id"
}

data "google_secret_manager_secret_version" "zitadel_project" {
count = local.zitadel_project_secret_present ? 1 : 0
secret = "zitadel-project-id" # pragma: allowlist secret
project = var.project_id
}
3 changes: 2 additions & 1 deletion opentofu/gcp/gke/configure/kubernetes.tf
Original file line number Diff line number Diff line change
Expand Up @@ -135,7 +135,8 @@ resource "kubectl_manifest" "flux_cluster_vars" {
# scope has to name it. Substituting it means a rename is caught by
# check-substitution.py instead of surfacing as a bare `invalid_grant`
# with oauth2-proxy, the exchange proxy and Headlamp all reporting healthy.
zitadel_project_id = var.zitadel_project_id
# Live-reconciled when this cluster hosts ZITADEL: see local.zitadel_project_id.
zitadel_project_id = local.zitadel_project_id
}
})
server_side_apply = true
Expand Down
8 changes: 8 additions & 0 deletions opentofu/gcp/gke/configure/locals.tf
Original file line number Diff line number Diff line change
Expand Up @@ -31,4 +31,12 @@ locals {
# hosts the IdP" become unanswerable from configuration in the first place.
# See ADR-0024 and the two-gate note on var.deploy_identity_provider.
identity_provider_url = var.deploy_identity_provider ? "https://auth.${var.public_domain_name}" : var.identity_provider_url

# The list filter is a substring match; contains() makes it exact.
zitadel_project_secret_present = var.deploy_identity_provider && contains(
[for s in try(data.google_secret_manager_secrets.zitadel_project[0].secrets, []) : s.secret_id],
"zitadel-project-id"
)
# Not secret: nonsensitive() keeps the whole ConfigMap out of (sensitive) diffs.
zitadel_project_id = local.zitadel_project_secret_present ? nonsensitive(jsondecode(data.google_secret_manager_secret_version.zitadel_project[0].secret_data)["project_id"]) : var.zitadel_project_id
}
86 changes: 71 additions & 15 deletions opentofu/gcp/gke/init/workflows.tm.hcl
Original file line number Diff line number Diff line change
Expand Up @@ -6,10 +6,15 @@
# Stage 1 (this stack): GKE Standard cluster, static spot node pool, Workload
# Identity, Crossplane WIF bootstrap
# Stage 2 (configure): Gateway API CRDs -> Cilium -> Flux Operator -> Flux Instance
# Stage 3: External Secrets grants, then the OIDC clients; when
# gcp-0 hosts the IdP, first the Google IdP and groups
# Action, and the clients mirrored into OpenBao
# Stage 5: a hosting gcp-0's OpenBao OIDC check, which halts the
# deploy on drift
#
# There is no stage 3. The EKS equivalent recycles bootstrap nodes whose ENIs
# predate Cilium, which is specific to ENI prefix delegation and has no GCP
# counterpart -- ipam.mode=kubernetes takes pod CIDRs from the node object.
# Stage 5 keeps aws-0's job name, so one contract suite guards both. EKS's node
# recycle has no GCP counterpart: it exists for ENI prefix delegation, while
# ipam.mode=kubernetes takes pod CIDRs from the node object.
#
# The control-plane endpoint is PRIVATE, so stage 2 must run from a machine on the
# tailnet.
Expand Down Expand Up @@ -260,6 +265,25 @@ script "deploy" {
exit 0
fi

# The workforce provider's audience is the ZITADEL PROJECT id, which does
# not exist until the sync below creates the project. Passing the pool
# lets the script reconcile it; without this, per-user RBAC on this
# cluster fails as a bare `invalid_grant` with everything looking healthy.
# Empty (no such stack / no such key) simply skips that reconciliation.
WORKFORCE_POOL="$(awk -F'=' '/^[[:space:]]*workforce_pool_id/{gsub(/[[:space:]"]/,"",$2); print $2}' "$${ROOT}/opentofu/gcp/workforce-identity/variables.tfvars" 2>/dev/null || true)"
# One flag list per sync, expanded by the real call AND the recovery
# printed when ZITADEL is late: a hand re-run missing the OpenBao flags
# leaves every ExternalSecret on the dead directory's clients.
IDP_SYNC_ARGS=(--cluster "$${NAME}" --cloud gcp --project "$${PROJECT}")
CLIENT_SYNC_ARGS=(
--cluster "$${NAME}" --cloud gcp --project "$${PROJECT}"
--workforce-pool "$${WORKFORCE_POOL}"
--openbao-url "https://bao.$${PRIVATE_DOMAIN}:8200"
--openbao-root-token-secret openbao-priv-gcp-root-token
--openbao-ca-file "$${ROOT}/opentofu/gcp/gke/configure/.tls/ca.pem"
--mirror-openbao
)

# A BUDGET FOR A COLD BUILD, NOT A REBUILD.
#
# ZITADEL is last in a long chain on a fresh cluster --
Expand Down Expand Up @@ -290,23 +314,23 @@ script "deploy" {

if [ "$(kubectl get deploy zitadel -n security -o jsonpath='{.status.readyReplicas}' 2>/dev/null || echo 0)" -lt 1 ] 2>/dev/null; then
echo "[warn] ZITADEL not ready in $(( ZITADEL_WAIT_SECONDS / 60 ))m; skipping OIDC client registration."
echo " Re-run by hand once it is up:"
echo " IDP_URL=https://auth.$${PUBLIC_DOMAIN} PRIVATE_DOMAIN=$${PRIVATE_DOMAIN} \\"
echo " scripts/provision/zitadel-oidc-clients.sh sync --cluster $${NAME} --cloud gcp --project $${PROJECT} --apply"
echo " Re-run by hand once it is up, kubectl pointed at $${NAME}:"
echo " IDP_URL=https://auth.$${PUBLIC_DOMAIN} scripts/provision/zitadel-idp.sh sync$(printf ' %q' "$${IDP_SYNC_ARGS[@]}") --apply"
echo " IDP_URL=https://auth.$${PUBLIC_DOMAIN} PRIVATE_DOMAIN=$${PRIVATE_DOMAIN} scripts/provision/zitadel-oidc-clients.sh sync$(printf ' %q' "$${CLIENT_SYNC_ARGS[@]}") --apply"
exit 0
fi

# The workforce provider's audience is the ZITADEL PROJECT id, which does
# not exist until the sync below creates the project. Passing the pool
# lets the script reconcile it; without this, per-user RBAC on this
# cluster fails as a bare `invalid_grant` with everything looking healthy.
# Empty (no such stack / no such key) simply skips that reconciliation.
WORKFORCE_POOL="$(awk -F'=' '/^[[:space:]]*workforce_pool_id/{gsub(/[[:space:]"]/,"",$2); print $2}' "$${ROOT}/opentofu/gcp/workforce-identity/variables.tfvars" 2>/dev/null || true)"
# A fresh directory (GCP parity GP-3) has neither the Google IdP nor the
# groups Action; both must exist before the clients, or no token carries
# a groups claim.
echo "== registering the Google IdP and the groups Action"
IDP_URL="https://auth.$${PUBLIC_DOMAIN}" \
bash "$${ROOT}/scripts/provision/zitadel-idp.sh" sync "$${IDP_SYNC_ARGS[@]}" --apply || \
echo "[warn] IdP registration failed; re-run it by hand"

echo "== registering the OIDC clients"
IDP_URL="https://auth.$${PUBLIC_DOMAIN}" PRIVATE_DOMAIN="$${PRIVATE_DOMAIN}" \
bash "$${ROOT}/scripts/provision/zitadel-oidc-clients.sh" sync \
--cluster "$${NAME}" --cloud gcp --project "$${PROJECT}" \
--workforce-pool "$${WORKFORCE_POOL}" --apply || \
bash "$${ROOT}/scripts/provision/zitadel-oidc-clients.sh" sync "$${CLIENT_SYNC_ARGS[@]}" --apply || \
echo "[warn] OIDC registration failed; re-run it by hand"

echo "== granting access to the secrets it just created"
Expand All @@ -315,6 +339,38 @@ script "deploy" {
],
]
}

job {
name = "stage5-verify-openbao-oidc"
description = "Fail the deploy when a hosting gcp-0's OpenBao OIDC client disagrees with the store or ZITADEL no longer knows it"
commands = [
["bash", "-c", <<-BASH
${global.cloud_gate}
set -euo pipefail
ROOT="${terramate.root.path.fs.absolute}"

# Stage 3's warn-and-continue lets a failed rotation through; this is
# what stops it. The check is the last statement, so exit 1 (drift) and
# exit 2 (cannot tell) both halt, as on aws-0.
DEPLOY_IDP="$${TF_VAR_deploy_identity_provider:-${global.deploy_identity_provider_gcp}}"
if [ "$${DEPLOY_IDP}" != "true" ]; then
echo "== skipping: this cluster consumes ${global.primary_cloud}'s directory, whose own deploy verifies OpenBao's OIDC client"
exit 0
fi

PROJECT="$(${global.provisioner} output -raw project_id)"
OPENBAO_URL="https://bao.$(${global.provisioner} output -raw private_domain_name):8200"
echo "== verifying OpenBao's OIDC client"
bash "$${ROOT}/scripts/provision/openbao-oidc-check.sh" \
--url "$${OPENBAO_URL}" \
--root-token-secret-name openbao-priv-gcp-root-token \
--ca-file "$${ROOT}/opentofu/gcp/gke/configure/.tls/ca.pem" \
--cloud gcp --project "$${PROJECT}" \
--redirect-uri "$${OPENBAO_URL}/ui/vault/auth/oidc/oidc/callback"
BASH
],
]
}
}

script "deploy-stage1" {
Expand Down
20 changes: 20 additions & 0 deletions scripts/ci/tests/test-bao-map.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,20 @@
#!/usr/bin/env bash
#
# The managed-store -> OpenBao map lives in one file (GCP parity GP-5): migrate
# copies through it, and the OIDC sync mirrors through it. Two copies would drift
# and put a client secret where no ExternalSecret reads it.
set -uo pipefail
REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../../.." && pwd)"
fails=0
check() { if [ "$2" = "$3" ]; then echo " ok $1"; else echo " FAIL $1: want '$2' got '$3'"; fails=1; fi; }
# shellcheck source=scripts/lib/bao-map.sh
. "$REPO_ROOT/scripts/lib/bao-map.sh" \
|| { echo "FAIL cannot source scripts/lib/bao-map.sh"; exit 1; }
check "grafana-envvars" "platform/victoria-metrics/grafana-envvars" "$(bao_target_for observability-victoria-metrics-k8s-stack-grafana-envvars)"
check "harbor-oidc" "platform/harbor/oidc" "$(bao_target_for harbor-oidc)"
check "app-wizard llm" "apps/app-wizard/llm" "$(bao_target_for apps-app-wizard-llm)"
bao_target_for openbao-oidc >/dev/null; check "openbao-oidc is unmapped" "1" "$?"
grep -q '^bao_target_for()' "$REPO_ROOT/scripts/provision/secret-store.sh" && { echo " FAIL secret-store.sh still defines its own copy"; fails=1; }
grep -q 'lib/bao-map.sh' "$REPO_ROOT/scripts/provision/secret-store.sh" || { echo " FAIL secret-store.sh does not source the map"; fails=1; }
[ "$fails" -eq 0 ] && echo "all checks passed"
exit "$fails"
79 changes: 79 additions & 0 deletions scripts/ci/tests/test-cloud-secret-store.sh
Original file line number Diff line number Diff line change
Expand Up @@ -221,4 +221,83 @@ rm -rf "$FAILTMP2"

rm -rf "$FAILSTUB" "$FAILTMP"

# --- store_write: a failed payload write never reaches the cloud ------------
#
# Called under `||`, as every mirror and PAT caller does, store_write runs with
# errexit off. A `cat > "$payload"` cut short (a full disk) then fell through to
# `versions add` / `put-secret-value`, which stored the truncated secret and
# returned 0. The failing `cat` is a function so it can land half the payload
# first, the way a full disk does.
short_write() { # cloud label -> "<rc> <cloud calls>"
: > "$STUB_LOG"
local rc=0
PATH="$STUB:$PATH" bash -c '
set -o errexit -o nounset -o pipefail
# shellcheck source=scripts/lib/cloud-secret-store.sh
. "'"$HERE"'/../../lib/cloud-secret-store.sh"
cat() { command cat >/dev/null; printf "%s" "{\"pat\":\"trunc"; return 1; }
CLOUD="$1" REGION=eu-west-3 GCP_PROJECT=proj
store_write existing-secret <<< "{\"pat\":\"fake-full-token\"}" || exit $?
' _ "$1" || rc=$?
echo "$rc $(grep -cE 'versions add|put-secret-value|create-secret' "$STUB_LOG")"
}
check "aws short payload write: non-zero, no put-secret-value" "1 0" "$(short_write aws)"
check "gcp short payload write: non-zero, no versions add" "1 0" "$(short_write gcp)"

# mktemp failing (TMPDIR gone) is the same fall-through with an empty path.
no_tmp() {
: > "$STUB_LOG"
local rc=0
PATH="$STUB:$PATH" TMPDIR=/nonexistent/cloud-secret-store-test bash -c '
. "'"$HERE"'/../../lib/cloud-secret-store.sh"
CLOUD="$1" REGION=eu-west-3 GCP_PROJECT=proj
store_write existing-secret <<< "{\"pat\":\"fake-full-token\"}" || exit $?
' _ "$1" 2>/dev/null || rc=$?
echo "$rc $(grep -cE 'versions add|put-secret-value|create-secret' "$STUB_LOG")"
}
check "aws mktemp failure: non-zero, no cloud write" "1 0" "$(no_tmp aws)"
check "gcp mktemp failure: non-zero, no cloud write" "1 0" "$(no_tmp gcp)"

# A failed `secrets create` stops before `versions add`, which would only fail
# again on a secret that does not exist.
CREATEFAIL="$(mktemp -d)"
cat > "$CREATEFAIL/gcloud" <<'EOF'
#!/usr/bin/env bash
printf 'CALL:' >> "$STUB_LOG"; printf ' %q' "$@" >> "$STUB_LOG"; printf '\n' >> "$STUB_LOG"
for a in "$@"; do case "$a" in describe|create) exit 1 ;; esac; done
exit 0
EOF
chmod +x "$CREATEFAIL/gcloud"
: > "$STUB_LOG"
rc=0
PATH="$CREATEFAIL:$PATH" bash -c '
. "'"$HERE"'/../../lib/cloud-secret-store.sh"
CLOUD=gcp REGION="" GCP_PROJECT=proj
store_write notfound-secret <<< "{\"pat\":\"fake-full-token\"}" || exit $?
' || rc=$?
check "gcp create failure: non-zero" "1" "$rc"
check_log_absent "gcp create failure: no versions add" 'versions add' "$STUB_LOG"

# A describe that failed transiently sends an existing secret down the create
# path; ALREADY_EXISTS then proves it exists, so the version is still added.
cat > "$CREATEFAIL/gcloud" <<'EOF'
#!/usr/bin/env bash
printf 'CALL:' >> "$STUB_LOG"; printf ' %q' "$@" >> "$STUB_LOG"; printf '\n' >> "$STUB_LOG"
for a in "$@"; do case "$a" in
describe) exit 1 ;;
create) echo "ERROR: (gcloud.secrets.create) ALREADY_EXISTS: Secret [projects/proj/secrets/raced-secret] already exists." >&2; exit 1 ;;
esac; done
exit 0
EOF
: > "$STUB_LOG"
rc=0
PATH="$CREATEFAIL:$PATH" bash -c '
. "'"$HERE"'/../../lib/cloud-secret-store.sh"
CLOUD=gcp REGION="" GCP_PROJECT=proj
store_write raced-secret <<< "{\"pat\":\"fake-full-token\"}" || exit $?
' 2>/dev/null || rc=$?
check "gcp create ALREADY_EXISTS: succeeds" "0" "$rc"
check_log "gcp create ALREADY_EXISTS: versions add still runs" 'versions add' "$STUB_LOG"
rm -rf "$CREATEFAIL"

exit $fail
57 changes: 57 additions & 0 deletions scripts/ci/tests/test-gcp-gke-init-workflow.py
Original file line number Diff line number Diff line change
Expand Up @@ -60,6 +60,63 @@ def before(body, first, second, what):
if not any("state rm -lock-timeout=" in l for l in state_lines):
fails.append("the destroy's custom-role state rm waits for the state lock")

s3 = job_body(TEXT, "stage3-secrets-and-oidc")
before(s3, 'zitadel-idp.sh" sync', "== registering the OIDC clients", "a hosting stage 3 configures the IdP and Action before its clients")
# One flag list per sync, expanded by both the real call and the printed
# recovery, so the recovery cannot drift from what the deploy runs.
client_args = re.search(r"CLIENT_SYNC_ARGS=\((.*?)\n\s*\)", s3, re.S)
client_args = client_args.group(1) if client_args else ""
for flag in ("--openbao-url", "--openbao-root-token-secret openbao-priv-gcp-root-token", "--openbao-ca-file", "--mirror-openbao",
"--workforce-pool", "--cluster", "--project"):
if flag not in client_args:
fails.append(f"CLIENT_SYNC_ARGS carries {flag}")
if "--cluster" not in (re.search(r"IDP_SYNC_ARGS=\((.*?)\)", s3, re.S) or [None, ""])[1]:
fails.append("IDP_SYNC_ARGS carries --cluster")


def call_line(text, script):
"""The logical line (continuations joined) that runs `script" sync`."""
joined = re.sub(r"\\\n\s*", " ", text)
return next((l for l in joined.splitlines() if f'{script}" sync' in l and not l.lstrip().startswith("echo")), "")


real = s3[s3.find("== registering the Google IdP"):] if "== registering the Google IdP" in s3 else ""
if "IDP_SYNC_ARGS[@]" not in call_line(real, "zitadel-idp.sh"):
fails.append("the real IdP sync expands IDP_SYNC_ARGS")
if "CLIENT_SYNC_ARGS[@]" not in call_line(real, "zitadel-oidc-clients.sh"):
fails.append("the real clients sync expands CLIENT_SYNC_ARGS")
m = re.search(r"ZITADEL not ready.*?\n\s*exit 0", s3, re.S)
recovery = m.group(0) if m else ""
before(recovery, "zitadel-idp.sh sync", "zitadel-oidc-clients.sh sync", "the not-ready recovery prints the IdP sync, then the clients sync")
for line_has, arr in (("zitadel-idp.sh sync", "IDP_SYNC_ARGS[@]"), ("zitadel-oidc-clients.sh sync", "CLIENT_SYNC_ARGS[@]")):
line = next((l for l in recovery.splitlines() if line_has in l), "")
if arr not in line or "--apply" not in line:
fails.append(f"the not-ready recovery's {line_has} expands {arr} and passes --apply")
before(s3, "CLIENT_SYNC_ARGS=(", "ZITADEL not ready", "the flag lists exist before the recovery prints them")

# Stage 5 (item 3): the same check aws-0's deploy halts on, for a hosting gcp-0.
deploy = re.search(r'\nscript "deploy" \{(.*?)(?=\nscript "|\Z)', TEXT, re.S)
deploy = deploy.group(1) if deploy else ""
s5 = job_body(deploy, "stage5-verify-openbao-oidc")
if not s5:
fails.append("script deploy has a stage5-verify-openbao-oidc job")
before(deploy, '"stage3-secrets-and-oidc"', '"stage5-verify-openbao-oidc"', "stage 5 runs after stage 3")
if "set -euo pipefail" not in s5:
fails.append("stage 5 runs under errexit")
body5 = s5.split("<<-BASH", 1)[-1].split("\n BASH", 1)[0]
logical5 = [l.strip() for l in re.sub(r"\\\n\s*", " ", body5).splitlines() if l.strip()]
last5 = logical5[-1] if logical5 else ""
if not last5.startswith('bash "$${ROOT}/scripts/provision/openbao-oidc-check.sh"') or re.search(r"\|\||&&|;|\|", last5):
fails.append("stage 5's check is its last statement, so exit 1 and exit 2 both halt the deploy")
for arg in ("--cloud gcp", "--project", "--root-token-secret-name openbao-priv-gcp-root-token",
"/opentofu/gcp/gke/configure/.tls/ca.pem", "/ui/vault/auth/oidc/oidc/callback"):
if arg not in last5:
fails.append(f"stage 5's check passes {arg}")
before(s5, 'DEPLOY_IDP}" != "true" ]', "openbao-oidc-check.sh", "stage 5 skips a consuming cluster before the check")
skip = s5[s5.find('DEPLOY_IDP}" != "true" ]'):s5.find("openbao-oidc-check.sh")]
if "exit 0" not in skip or "echo" not in skip:
fails.append("stage 5's consumer skip says why and exits 0")

for f in fails:
print("FAIL", f)
if fails:
Expand Down
Loading
Loading