diff --git a/EXTENDING.md b/EXTENDING.md index d839930..2dbd38b 100644 --- a/EXTENDING.md +++ b/EXTENDING.md @@ -35,6 +35,7 @@ with an upstream update. | Which layers to create | `enable_runtime`, `enable_portal`, `enable_llm_edge`, `enable_team_auth`, `enable_team_demo`, `enable_mcp_hub_demo` | [terraform/variables.tf](terraform/variables.tf) — the quick start uses them to stage the deploy | | Bedrock direct mode | `CLAUDE_CODE_USE_BEDROCK=1` | runtime env; no key involved | | Your LLM gateway | `enable_llm_edge = true` + the backend's `base_url` in Governance → Model backends | key in Secrets Manager, read only by `llm-edge`; kernels get a per-session grant, never the key | +| An AgentCore Gateway instead | `PLATFORM_AGENTCORE_GATEWAY_CALLER_ROLE_ARN` + the backend's `base_url` (the gateway's `/inference` URL) | no key on the platform side at all and no `llm-edge` to run; kernels get session-tagged STS credentials and SigV4-sign. Requires denying the kernel roles direct `InvokeGateway` on that gateway — see [docs/permissions.md](docs/permissions.md) | | Backend runtime settings (table, buckets, ARNs, Cognito, CORS, admin tiers) | `PLATFORM_*` env vars | [backend/app/config.py](backend/app/config.py) — every field there is `PLATFORM_` (`env_prefix`); `backend/.env.example` for local runs | Terraform state is yours: `terraform/backend.tf.example` shows the remote-state diff --git a/README.md b/README.md index 1903dd0..5b20fb9 100644 --- a/README.md +++ b/README.md @@ -151,9 +151,10 @@ cp terraform.tfvars.example terraform.tfvars # then edit terraform init terraform apply -var enable_runtime=false -var enable_portal=false -# 2. Store your LLM gateway key (skip if using Bedrock direct). +# 2. Store your LLM gateway key (skip for Bedrock direct or AgentCore Gateway). # Readable only by the llm-edge service; it never enters a kernel container. -# Gateway mode also needs enable_llm_edge=true — see docs/deployment.md §2. +# Gateway mode also needs enable_llm_edge=true — see docs/deployment.md §2, +# which also covers the agentcore_gateway backend (no key on this side). aws secretsmanager put-secret-value \ --secret-id agent-platform/llm-gateway-key \ --secret-string '{"api_key":"sk-..."}' diff --git a/backend/app/api/sessions.py b/backend/app/api/sessions.py index 3b98695..1ac7256 100644 --- a/backend/app/api/sessions.py +++ b/backend/app/api/sessions.py @@ -166,11 +166,12 @@ def connect(session_id: str, user: str = Depends(get_current_user)): spec = model_config_service.resolve( item.get("model_backend", ""), item.get("model", "") ) - if spec and spec.get("backend") == "gateway": - # The gateway key stays in llm-edge. The kernel gets an - # endpoint plus a session-scoped token, and the routing fields - # are stripped from what the container sees: the edge re-reads - # them from the grant, so a container has nothing to forge. + if spec and spec.get("backend") in ("gateway", "agentcore_gateway"): + # No upstream credential reaches the container in either mode: + # litellm keeps the key in llm-edge, AgentCore keeps it in the + # gateway's token vault. The kernel gets an endpoint plus a + # session-scoped credential, and the routing fields are + # stripped from what the container sees. creds = llm_credentials_service.mint( item["runtime_session_id"], user, spec ) @@ -178,8 +179,10 @@ def connect(session_id: str, user: str = Depends(get_current_user)): raise HTTPException( status_code=503, detail=( - "gateway model routing is unavailable: the llm-edge " - "service is not deployed (set enable_llm_edge)" + "gateway model routing is unavailable: deploy " + "llm-edge (enable_llm_edge) for the litellm " + "backend, or set the AgentCore gateway caller role " + "for the agentcore_gateway backend" ), ) config["llm_credentials"] = creds diff --git a/backend/app/config.py b/backend/app/config.py index 1e3a137..10e20dd 100644 --- a/backend/app/config.py +++ b/backend/app/config.py @@ -39,6 +39,18 @@ class Settings(BaseSettings): # root in. llm_edge_url: str = "" + # Role the backend assumes once per session to mint the kernel's AgentCore + # Gateway credentials, tagged with that session's runtime id. Same shape as + # workspace_access_role_arn: the kernel role itself has no InvokeGateway + # grant, so a container cannot reach the gateway on its own identity. + # + # The session tag is what makes revocation per-session: ending a session + # adds a Deny conditioned on aws:PrincipalTag/session_id, which stops that + # session's credentials without touching any other live session's. + # Empty = the agentcore_gateway model backend is unavailable and the + # backend refuses it rather than falling back to a shared credential. + agentcore_gateway_caller_role_arn: str = "" + # EventBridge Scheduler wiring (outputs of the PortalStack). When all of # group/lambda/role are set, the scheduler runs in "eventbridge" mode: # each platform schedule is mirrored to an EventBridge Scheduler schedule diff --git a/backend/app/models/schemas.py b/backend/app/models/schemas.py index 16c21ac..5737c80 100644 --- a/backend/app/models/schemas.py +++ b/backend/app/models/schemas.py @@ -9,7 +9,7 @@ class SessionCreateRequest(BaseModel): mcp_server_ids: list[str] = Field(default_factory=list, max_length=10) skill_ids: list[str] = Field(default_factory=list, max_length=10) # "" = platform default backend; otherwise a backend from the model config - model_backend: str = Field(default="", pattern="^(|bedrock|litellm)$") + model_backend: str = Field(default="", pattern="^(|bedrock|litellm|agentcore_gateway)$") model: str = Field(default="", max_length=200) # AgentCore Runtime platform version; "" = deployment default platform_version: str = Field(default="", pattern="^(|V1|V2)$") @@ -127,7 +127,7 @@ class AgentPublishRequest(BaseModel): skill_names: list[str] = Field(default_factory=list, max_length=10) memory_id: str = "" # "" = platform default backend; otherwise a backend from the model config - model_backend: str = Field(default="", pattern="^(|bedrock|litellm)$") + model_backend: str = Field(default="", pattern="^(|bedrock|litellm|agentcore_gateway)$") model: str = Field(default="", max_length=200) # AgentCore Runtime platform version the agent runs on; "" = default. # Pipelines, schedules and channels calling the agent inherit it. @@ -216,12 +216,12 @@ class ModelBackendPatch(BaseModel): class ModelConfigUpdate(BaseModel): - default_backend: str | None = Field(default=None, pattern="^(bedrock|litellm)$") + default_backend: str | None = Field(default=None, pattern="^(bedrock|litellm|agentcore_gateway)$") backends: dict[str, ModelBackendPatch] | None = None class ModelTestRequest(BaseModel): - backend: str = Field(pattern="^(bedrock|litellm)$") + backend: str = Field(pattern="^(bedrock|litellm|agentcore_gateway)$") model: str = Field(default="", max_length=200) diff --git a/backend/app/services/kernel_service.py b/backend/app/services/kernel_service.py index 1c8d9f6..d95c902 100644 --- a/backend/app/services/kernel_service.py +++ b/backend/app/services/kernel_service.py @@ -162,12 +162,17 @@ def invoke_sdk_kernel( payload["memory"] = memory if model: # per-invocation model routing (see model_config_service.resolve) - if model.get("backend") == "gateway": - # The gateway key stays in llm-edge. Mint a grant scoped to this - # invocation's session and strip the routing fields, so the - # kernel receives an endpoint and a token instead of a key it - # could fetch itself. An async run has no refresh channel and - # may execute for hours, so its grant is given matching life. + if model.get("backend") in ("gateway", "agentcore_gateway"): + # No upstream credential reaches the kernel either way: the + # litellm path keeps the key in llm-edge, the AgentCore path + # keeps it in the gateway's token vault. Mint a grant scoped to + # this invocation's session and strip the routing fields, so + # the kernel receives an endpoint plus a session credential + # rather than something it could spend elsewhere. An async run + # has no refresh channel and may execute for hours, so its + # grant is given matching life — for the AgentCore path that + # requires the caller role's MaxSessionDuration to cover it, + # otherwise AssumeRole refuses and this call fails closed. creds = llm_credentials_service.mint( sid, user, @@ -180,9 +185,10 @@ def invoke_sdk_kernel( "result": "", "raw": { "error": ( - "gateway model routing is unavailable: the " - "llm-edge service is not deployed " - "(set enable_llm_edge)" + "gateway model routing is unavailable: deploy " + "llm-edge (enable_llm_edge) for the litellm " + "backend, or set the AgentCore gateway caller " + "role for the agentcore_gateway backend" ) }, } diff --git a/backend/app/services/llm_credentials_service.py b/backend/app/services/llm_credentials_service.py index 8bceaf4..f6f2579 100644 --- a/backend/app/services/llm_credentials_service.py +++ b/backend/app/services/llm_credentials_service.py @@ -27,9 +27,12 @@ import hashlib import hmac +import json import logging +import re import secrets import time +import urllib.parse import boto3 @@ -46,6 +49,51 @@ # session silently stops being found. LLM_TOKEN_PK = "LLMTOKEN" +# ---------------------------------------------------------------- AgentCore + +# STS refuses anything shorter, so this is the floor on how stale a revoked +# AgentCore session's credentials can be if the Deny below fails to apply. +AGENTCORE_TTL_S = 900 + +# Sessions whose AgentCore credentials have been revoked but whose STS +# expiry has not yet passed. The role's inline Deny is rebuilt from this +# partition, and entries are dropped once they expire — that pruning is what +# keeps the policy inside the 10 KB per-policy limit. +LLM_REVOKE_PK = "LLMREVOKE" + +# Inline policy on the caller role carrying the per-session Deny statements. +REVOKE_POLICY_NAME = "AgentCoreSessionRevocations" + +_GATEWAY_HOST_RE = re.compile( + r"^(?P[A-Za-z0-9_-]+)\.gateway\.bedrock-agentcore\.(?P[a-z0-9-]+)\.amazonaws\.com$" +) + + +def gateway_arn_from_base_url(base_url: str, account_id: str) -> str: + """Derive the gateway ARN from the endpoint the model config carries. + + The session policy below has to name the gateway as a resource, and asking + an operator to keep an ARN and a URL in sync is a good way to get a + mismatch. The URL already contains both the gateway id and the region, so + the ARN is derivable; an unparseable host yields "" and the caller falls + back to no session policy (the role's own policy still applies). + """ + host = urllib.parse.urlsplit(str(base_url or "")).hostname or "" + m = _GATEWAY_HOST_RE.match(host) + if not m or not account_id: + return "" + return ( + f"arn:aws:bedrock-agentcore:{m.group('region')}:{account_id}" + f":gateway/{m.group('id')}" + ) + + +def _region_from_gateway_arn(gateway_arn: str) -> str: + """SigV4 has to be signed for the gateway's own region, which is not + necessarily the region this backend runs in.""" + parts = str(gateway_arn or "").split(":") + return parts[3] if len(parts) > 4 else "" + def _sha256(value: str) -> str: return hashlib.sha256(value.encode("utf-8")).hexdigest() @@ -72,11 +120,24 @@ class LlmCredentialsService: def __init__(self) -> None: dynamodb = boto3.resource("dynamodb", region_name=settings.aws_region) self.table = dynamodb.Table(settings.dynamo_table) + self.sts = boto3.client("sts", region_name=settings.aws_region) + self.iam = boto3.client("iam") + self._account_id = "" @property def enabled(self) -> bool: return bool(settings.llm_edge_url) + @property + def agentcore_enabled(self) -> bool: + return bool(settings.agentcore_gateway_caller_role_arn) + + @property + def account_id(self) -> str: + if not self._account_id: + self._account_id = self.sts.get_caller_identity()["Account"] + return self._account_id + def mint( self, runtime_session_id: str, @@ -99,6 +160,10 @@ def mint( never reaches the agent's own subprocess environment either way — the kernel keeps it and hands the CLI a container-local token instead. """ + if str(spec.get("backend") or "") == "agentcore_gateway": + return self.mint_agentcore( + runtime_session_id, user, spec, team=team, ttl_s=ttl_s + ) if not self.enabled: return None base_url = str(spec.get("base_url") or "") @@ -157,6 +222,256 @@ def mint( "expires_at": expires_at, } + # ------------------------------------------------------------ AgentCore + + def mint_agentcore( + self, + runtime_session_id: str, + user: str, + spec: dict, + team: str = "", + ttl_s: int = AGENTCORE_TTL_S, + models: list[str] | None = None, + ) -> dict | None: + """Mint SigV4 credentials for one session's AgentCore Gateway access. + + Same shape of promise as :meth:`mint`, reached differently. There is no + bearer token to leak here and no platform-wide secret anywhere on the + path: the gateway holds the upstream provider credential in its token + vault, and the container authenticates as *this session* using STS + credentials tagged with its runtime session id. + + Two things are bound into the credential itself, so neither depends on + the container behaving: + + - a **session policy** narrowing the grant to `InvokeGateway` on the one + gateway this backend points at, so a leaked credential cannot reach + any other gateway in the account; + - a **session tag** (``session_id``), which is what + :meth:`revoke_agentcore` conditions its Deny on. That is how one + session is revoked without disturbing any other live session. + + What is *not* enforced here is the per-session model allowlist: IAM + conditions cannot see a request body, so which model a session may ask + for is not expressible as an IAM condition. The kernel does a + fail-fast local check (an optimisation, not a boundary — the session's + user is root in that microVM) and real enforcement needs either a + gateway request interceptor or one gateway per model. The allowlist is + recorded on the token item either way so the decision stays auditable. + """ + if not self.agentcore_enabled: + logger.error( + "refusing to mint AgentCore gateway credentials for %s: " + "PLATFORM_AGENTCORE_GATEWAY_CALLER_ROLE_ARN is not set", + runtime_session_id, + ) + return None + base_url = str(spec.get("base_url") or "").rstrip("/") + # ``models`` is passed only by :meth:`rotate`, which re-mints from the + # allowlist already recorded on the token item rather than re-deriving + # it from a spec it deliberately does not re-resolve. + models = list(models) if models else permitted_models(spec) + if not base_url or not models: + logger.error( + "refusing to mint AgentCore gateway credentials for %s: " + "incomplete spec", + runtime_session_id, + ) + return None + + gateway_arn = gateway_arn_from_base_url(base_url, self.account_id) + kwargs: dict = { + "RoleArn": settings.agentcore_gateway_caller_role_arn, + # The runtime session id is already caller-bound upstream (see + # session_binding), so it is safe to use verbatim as the session + # name — and doing so makes CloudTrail read as the session. + "RoleSessionName": runtime_session_id[:64], + "DurationSeconds": max(AGENTCORE_TTL_S, int(ttl_s)), + "Tags": [{"Key": "session_id", "Value": runtime_session_id}], + # Nothing downstream should be able to widen the grant by + # re-assuming with a different tag. + "TransitiveTagKeys": ["session_id"], + } + if gateway_arn: + kwargs["Policy"] = json.dumps( + { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Action": "bedrock-agentcore:InvokeGateway", + "Resource": gateway_arn, + } + ], + } + ) + else: + logger.warning( + "could not derive a gateway ARN from %r — minting without a " + "session policy", + base_url, + ) + try: + creds = self.sts.assume_role(**kwargs)["Credentials"] + except Exception as e: # noqa: BLE001 - any STS failure means refuse + logger.error( + "AssumeRole for AgentCore gateway session %s failed: %s", + runtime_session_id, + e, + ) + return None + + expires_at = int(creds["Expiration"].timestamp()) + try: + self.table.put_item( + Item={ + "PK": LLM_TOKEN_PK, + "SK": f"RSID#{runtime_session_id}", + "mode": "agentcore_gateway", + # No token digest: the credential is an STS session, not a + # bearer value this table could be used to replay. + "expires_at": expires_at, + "runtime_session_id": runtime_session_id, + "user": user, + "team": team, + "upstream_base_url": base_url, + "gateway_arn": gateway_arn, + "allowed_models": models, + }, + # A grant belongs to the session's owner: create, or re-mint the + # owner's own — never overwrite a live grant held by a different + # principal, since doing so would stop the victim's kernel from + # matching and is a targeted denial of service. Session ids are + # caller-bound upstream, so a collision should be unreachable; + # this makes it impossible even if that breaks. + ConditionExpression="attribute_not_exists(PK) OR #u = :user", + ExpressionAttributeNames={"#u": "user"}, + ExpressionAttributeValues={":user": user}, + ) + except self.table.meta.client.exceptions.ConditionalCheckFailedException: + logger.error( + "refusing to mint AgentCore gateway credentials for %s: " + "session grant is held by a different principal", + runtime_session_id, + ) + return None + + return { + "mode": "agentcore_gateway", + "endpoint": base_url, + "session_id": runtime_session_id, + "region": _region_from_gateway_arn(gateway_arn) or settings.aws_region, + # The gateway's data plane signs as bedrock-agentcore, not as the + # control-plane service the SDK would guess from the hostname. + "service": "bedrock-agentcore", + "access_key_id": creds["AccessKeyId"], + "secret_access_key": creds["SecretAccessKey"], + "session_token": creds["SessionToken"], + "expires_at": expires_at, + # Carried for the kernel's fail-fast check only. The kernel must + # not treat this as authorization — see the docstring. + "allowed_models": models, + } + + def revoke_agentcore(self, runtime_session_id: str) -> None: + """Stop one session's gateway credentials from working, now. + + STS sessions cannot be withdrawn, so revocation is expressed as a Deny + on the caller role conditioned on the session's tag. The Deny set is + rebuilt from the LLMREVOKE partition on every call and entries whose + STS expiry has passed are dropped, which is what keeps the inline + policy inside its 10 KB limit: an entry only has to outlive the + credential it revokes. + """ + if not self.agentcore_enabled or not runtime_session_id: + return + item = self.table.get_item( + Key={"PK": LLM_TOKEN_PK, "SK": f"RSID#{runtime_session_id}"} + ).get("Item") + expires_at = int((item or {}).get("expires_at") or 0) + if not expires_at: + # Nothing live to revoke; still fall through so the rebuild prunes. + expires_at = int(time.time()) + AGENTCORE_TTL_S + try: + self.table.put_item( + Item={ + "PK": LLM_REVOKE_PK, + "SK": f"RSID#{runtime_session_id}", + "expires_at": expires_at, + } + ) + except Exception: # noqa: BLE001 + logger.warning("could not record revocation for %s", runtime_session_id) + self._rebuild_agentcore_denies() + + def _rebuild_agentcore_denies(self) -> None: + now = int(time.time()) + live: list[str] = [] + stale: list[str] = [] + try: + resp = self.table.query( + KeyConditionExpression="PK = :pk AND begins_with(SK, :p)", + ExpressionAttributeValues={":pk": LLM_REVOKE_PK, ":p": "RSID#"}, + ) + except Exception as e: # noqa: BLE001 + logger.error("could not read revocation list: %s", e) + return + for it in resp.get("Items", []): + sid = str(it.get("SK", ""))[len("RSID#") :] + if not sid: + continue + # A small grace period past expiry: clock skew between here and + # STS must not un-revoke a session a second early. + if int(it.get("expires_at") or 0) + 60 > now: + live.append(sid) + else: + stale.append(sid) + + role_name = settings.agentcore_gateway_caller_role_arn.rsplit("/", 1)[-1] + try: + if live: + self.iam.put_role_policy( + RoleName=role_name, + PolicyName=REVOKE_POLICY_NAME, + PolicyDocument=json.dumps( + { + "Version": "2012-10-17", + "Statement": [ + { + "Sid": "RevokedSessions", + "Effect": "Deny", + "Action": "bedrock-agentcore:*", + "Resource": "*", + "Condition": { + "StringEquals": { + "aws:PrincipalTag/session_id": sorted(live) + } + }, + } + ], + } + ), + ) + else: + # Nothing outstanding — drop the policy rather than leave an + # empty statement behind. + try: + self.iam.delete_role_policy( + RoleName=role_name, PolicyName=REVOKE_POLICY_NAME + ) + except self.iam.exceptions.NoSuchEntityException: + pass + except Exception as e: # noqa: BLE001 + logger.error("could not update AgentCore revocation policy: %s", e) + return + for sid in stale: + try: + self.table.delete_item( + Key={"PK": LLM_REVOKE_PK, "SK": f"RSID#{sid}"} + ) + except Exception: # noqa: BLE001 + pass + def rotate(self, runtime_session_id: str) -> dict | None: """Issue a fresh token for an existing session grant. @@ -164,7 +479,26 @@ def rotate(self, runtime_session_id: str) -> dict | None: on the session's next warmup, same as every other model-routing change on this platform. Refresh is only about keeping a live session alive. """ - if not self.enabled or not runtime_session_id: + if not runtime_session_id: + return None + existing = self.table.get_item( + Key={"PK": LLM_TOKEN_PK, "SK": f"RSID#{runtime_session_id}"} + ).get("Item") + if existing and existing.get("mode") == "agentcore_gateway": + # STS credentials cannot be extended, so refresh is a re-mint from + # the routing already recorded on the item — which keeps the same + # "config edits apply at next warmup" rule as the edge path. + return self.mint_agentcore( + runtime_session_id, + str(existing.get("user") or ""), + { + "backend": "agentcore_gateway", + "base_url": str(existing.get("upstream_base_url") or ""), + }, + team=str(existing.get("team") or ""), + models=[str(m) for m in (existing.get("allowed_models") or [])], + ) + if not self.enabled: return None token = secrets.token_urlsafe(32) expires_at = int(time.time()) + TOKEN_TTL_S @@ -191,11 +525,22 @@ def rotate(self, runtime_session_id: str) -> dict | None: } def revoke(self, runtime_session_id: str) -> None: - """Drop a session's grant. Called when a session ends so a token + """Drop a session's grant. Called when a session ends so a credential scraped out of a container's memory stops working immediately rather - than at the end of its hour.""" + than at the end of its lifetime. + + Both modes are handled. The edge mode is a single delete — the edge + re-reads this item on every call, so the grant dies with the row. The + AgentCore mode cannot delete an STS session, so it records a Deny + keyed on the session's tag *before* dropping the row (the Deny needs + the row's expiry to know when it may be pruned).""" if not runtime_session_id: return + item = self.table.get_item( + Key={"PK": LLM_TOKEN_PK, "SK": f"RSID#{runtime_session_id}"} + ).get("Item") + if item and item.get("mode") == "agentcore_gateway": + self.revoke_agentcore(runtime_session_id) try: self.table.delete_item( Key={"PK": LLM_TOKEN_PK, "SK": f"RSID#{runtime_session_id}"} diff --git a/backend/app/services/model_config_service.py b/backend/app/services/model_config_service.py index 0fb11ee..84dc10a 100644 --- a/backend/app/services/model_config_service.py +++ b/backend/app/services/model_config_service.py @@ -25,7 +25,7 @@ PK = "GOV" SK = "MODELCONFIG" -BACKEND_NAMES = ("bedrock", "litellm") +BACKEND_NAMES = ("bedrock", "litellm", "agentcore_gateway") DEFAULT_CONFIG: dict = { "default_backend": "bedrock", @@ -44,6 +44,18 @@ "default_model": "", "small_fast_model": "", }, + # AgentCore Gateway inference target. Unlike litellm there is no + # secret_name: the upstream provider credential lives in the gateway's + # token vault, so nothing on the platform side ever holds it. base_url + # is the gateway's /inference base; the kernel signs SigV4 against it + # with per-session credentials rather than presenting a bearer token. + "agentcore_gateway": { + "enabled": False, + "base_url": "", + "models": [], + "default_model": "", + "small_fast_model": "", + }, }, } @@ -124,22 +136,35 @@ def resolve(self, backend: str = "", model: str = "") -> dict | None: spec["small_fast_model"] = b["small_fast_model"] return spec - # litellm → the kernel's generic Anthropic-compatible gateway mode + # Both remaining backends mean "point the kernel at an HTTP endpoint + # rather than Bedrock", and they share these two failure modes. if not b["base_url"]: - raise ValueError("model backend 'litellm' has no base_url configured") + raise ValueError(f"model backend {name!r} has no base_url configured") if not chosen: # without an explicit name the container's baked-in Bedrock model # ID would leak into gateway requests — refuse instead raise ValueError( - "model backend 'litellm' needs a model (set the backend's " + f"model backend {name!r} needs a model (set the backend's " "default_model or the agent's model)" ) - spec = { - "backend": "gateway", - "base_url": b["base_url"], - "secret_name": b["secret_name"] or "agent-platform/llm-gateway-key", - "model": chosen, - } + if name == "agentcore_gateway": + # SigV4 mode. There is no platform-side secret to name: the + # upstream provider credential lives in the gateway's token vault, + # and the kernel authenticates as its own session instead of + # presenting a shared bearer token. + spec = { + "backend": "agentcore_gateway", + "base_url": b["base_url"], + "model": chosen, + } + else: + # litellm → the kernel's generic Anthropic-compatible gateway mode + spec = { + "backend": "gateway", + "base_url": b["base_url"], + "secret_name": b["secret_name"] or "agent-platform/llm-gateway-key", + "model": chosen, + } if b["small_fast_model"]: spec["small_fast_model"] = b["small_fast_model"] # Claude Code's /model picker offers opus/sonnet/haiku aliases that diff --git a/docs/architecture.md b/docs/architecture.md index 1742b86..3f7d1bf 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -255,17 +255,18 @@ Claude Code calls it from a terminal or the SDK kernel calls it per-invoke. ### 2. Model access — LLM gateway first -Both kernels support two model backends: +Both kernels support three model backends: | Mode | How | When to use | |---|---|---| | **Bedrock direct** | `CLAUDE_CODE_USE_BEDROCK=1` + cross-region inference profile (`global.` model ID prefix). Container IAM role; no key of any kind | Simplest path, and there is no model credential to leak | -| **LLM gateway** | Always per session, never a container default. The gateway key lives only in the `llm-edge` service; a kernel gets a session-scoped grant and reaches the gateway through it (`enable_llm_edge`) | Centralized model governance: allow-lists, budgets, cost attribution per team | +| **LLM gateway** (`litellm`) | Always per session, never a container default. The gateway key lives only in the `llm-edge` service; a kernel gets a session-scoped grant and reaches the gateway through it (`enable_llm_edge`) | Centralized model governance: allow-lists, budgets, cost attribution per team | +| **AgentCore Gateway** (`agentcore_gateway`) | Also always per session. The gateway sits in front of the same LiteLLM deployment Option A talks to, and the LiteLLM key lives in AgentCore Identity's token vault — so there is no key on the platform side at all and no broker service to run. A kernel gets STS credentials tagged with its session id and SigV4-signs each request (`agentcore_gateway_caller_role_arn`); LiteLLM keeps its multi-provider routing exactly as it does for every other client | The same governance goal with a managed hop instead of `llm-edge`, and per-session revocation expressed as an IAM Deny | -Gateway mode has no container-level configuration on purpose. A session's user -is root in its microVM and the headless kernel runs agent tools in a +Neither gateway mode has container-level configuration, on purpose. A session's +user is root in its microVM and the headless kernel runs agent tools in a subprocess, so any credential placed in a kernel container is a credential its -users have. Reaching the gateway therefore requires a grant the backend mints +users have. Reaching either gateway therefore requires a grant the backend mints per session; see security-explainer.zh.md §9. The environment sets the **container default**; the model control plane @@ -534,12 +535,15 @@ a new runtime version automatically). (SDK kernel only, and nothing the platform ships needs it, since search authenticates with the kernel's own role). The LLM gateway key is deliberately **not** among them: it is readable only by the `llm-edge` task - role, which no session can enter. + role, which no session can enter. With the `agentcore_gateway` backend there + is no such key on the platform side at all — the upstream credential lives in + the gateway's token vault. - No secrets are baked into images; keys are read from Secrets Manager at container/Lambda start. - The browser and end users hold **no AWS credentials** — all AWS access is server-side under the platform's seven IAM roles (an eighth, - `agent-platform-llm-edge`, in gateway mode), enumerated in + `agent-platform-llm-edge`, in litellm gateway mode; a ninth, the AgentCore + Gateway caller role, in `agentcore_gateway` mode), enumerated in [permissions.md §1](permissions.md#1-principals-at-a-glance). - If your LLM gateway is HTTP-only, traffic from NAT → gateway crosses the network unencrypted; put a TLS listener or PrivateLink in front for diff --git a/docs/deployment.md b/docs/deployment.md index 7684d9f..34913da 100644 --- a/docs/deployment.md +++ b/docs/deployment.md @@ -101,6 +101,119 @@ With `enable_llm_edge = false`, selecting the litellm backend makes the platform refuse the session with a 503 instead of falling back to handing a container the key. +### Option A2 — AgentCore Gateway in front of LiteLLM + +Same promise as Option A with one fewer service to run: the LiteLLM API key +lives in AgentCore Identity's token vault, so there is no bearer secret on +the platform side and no `llm-edge` to deploy. A kernel receives STS +credentials tagged with its session id and SigV4-signs each request; the +gateway substitutes in the LiteLLM key on the way out. LiteLLM keeps its +multi-provider routing, cost accounting and model catalog exactly as it does +for every other client — this option changes only how our platform reaches +LiteLLM, not what LiteLLM does behind it. + +1. **Mint a LiteLLM virtual key** scoped to the models this backend will + serve. The Prod-LiteLLM master key stays where it is; the key stored in + the vault below is the virtual one, so a revocation there does not touch + any other LiteLLM client. + + ```bash + curl -sS -X POST "$LITELLM_URL/key/generate" \ + -H "Authorization: Bearer $LITELLM_MASTER" \ + -H "Content-Type: application/json" \ + -d '{"models":["claude-haiku-4-5","claude-sonnet-5-5"], + "key_alias":"agentcore-gateway"}' | jq -r .key + ``` + +2. **Store the key in AgentCore Identity.** In the AWS console: Amazon + Bedrock AgentCore → **Identity** → **API key credential providers** → + *Create*. Paste the virtual key. AWS returns a + `credentialProviderArn` of the form + `arn:aws:bedrock-agentcore:::token-vault/default/apikeycredentialprovider/` + — note it for step 4. + +3. **Create the gateway.** `protocolType=MCP`, `authorizerType=AWS_IAM`. Its + execution role needs only the token-vault reads that fetch the API key + (`bedrock-agentcore:GetWorkloadAccessToken` and `GetResourceApiKey`, plus + `secretsmanager:GetSecretValue` on the provider's secret); it does *not* + need any `bedrock*` grant, because the outbound call is HTTPS to LiteLLM + with a header-injected API key rather than a Bedrock service call. + +4. **Create the inference *provider* target** pointing at LiteLLM, with the + credential provider substituting the caller's `Authorization` header on + the way out (`credentialPrefix: "Bearer"`, no trailing space). Declare + each path the clients use (`/v1/messages`, `/v1/chat/completions`, + `/v1/responses`; the gateway accepts no others) and the models each + path may route to. The gateway does no protocol translation: it matches + path and `model`, swaps in the key and relays body and SSE stream + verbatim, so LiteLLM decides which model works on which path. + + - **`metadataConfiguration.allowedRequestHeaders`** must list at least + `content-type`, `anthropic-version`, `anthropic-beta` and `accept`. + Left unset the gateway relays whatever the caller sent — including the + caller's own `x-amz-security-token`. Once set, `content-type` has to + be listed explicitly or the target stops relaying it and LiteLLM + answers 400. + - **Enable `drop_params: true` on the LiteLLM model entries** this + target routes to. Claude Code sends first-party-only fields + (`context_management` is the current one) that LiteLLM's Bedrock + adapter rejects otherwise. A REQUEST interceptor at the gateway could + strip them instead, but it adds a synchronous Lambda call per model + request and caps request bodies at Lambda's 6 MB invoke payload + (about 4.5 MB of JSON once base64-encoded), which a long session can + reach. + - **If LiteLLM is not publicly reachable**, add a top-level + `privateEndpoint.managedVpcResource` (sibling of + `targetConfiguration`): the gateway places VPC Lattice resource-gateway + ENIs in your subnets and connects to `routingDomain` (for example an + internal ALB's DNS name) while keeping the target endpoint's hostname + as TLS SNI, so the ALB needs a publicly trusted certificate for that + name. Lattice closes TCP connections idle for 350 seconds, so on this + path a non-streaming call whose response takes longer than that never + returns; clients must stream. + + Measured through a private endpoint (Tokyo, 2026-10): `claude-opus-5-5`, + `claude-sonnet-5-5`, `gpt-6-astra` and `kimi-k3` each answered on all + three paths, streaming and not; 20 concurrent calls and a 634-second SSE + stream went through; a non-streaming call that took 368 seconds never + returned. `GET /inference/v1/models` lists every declared model across the + gateway's targets as `/`, and a client may pin a target + that way in the `model` field. + + The exact `targetConfiguration` and `credentialProviderConfigurations` + payloads are in `scripts/e2e_agentcore_gateway.py`; that script provisions + all of the above against a live account. + +5. **Create the role the backend assumes per session**, trusted by the + backend's own role, granting `bedrock-agentcore:InvokeGateway` on the + gateway ARN and allowing `sts:TagSession`. Its `MaxSessionDuration` must + cover the async grant lifetime (9 h) if headless async runs use this + backend. + + ```hcl + # no dedicated module yet — supply the role ARN you created + ``` + + ```bash + PLATFORM_AGENTCORE_GATEWAY_CALLER_ROLE_ARN=arn:aws:iam:::role/ + ``` + +6. **Deny the kernel roles direct access to this gateway.** The kernel roles + already hold `bedrock-agentcore:InvokeGateway` on `gateway/*` so they can + reach MCP tool gateways, and that wildcard would also cover the inference + gateway — letting a root user in the microVM call it on the kernel role's own + identity, with no session tag to revoke. Add an explicit Deny for the + inference gateway's ARN to each kernel role; see permissions.md §3. Without + this the per-session credential is decorative. + +7. **Point the backend at the gateway in the portal**: *Governance → Model + backends → agentcore_gateway*, setting `base_url` to the `/inference` URL + and the model catalog to whichever LiteLLM aliases the virtual key + authorises. There is no `secret_name` for this backend. + +With the caller role ARN unset, selecting this backend makes the platform refuse +the session rather than fall back to a shared credential. + ### Option B — Bedrock direct Set: diff --git a/docs/permissions.md b/docs/permissions.md index 11b9685..47840dd 100644 --- a/docs/permissions.md +++ b/docs/permissions.md @@ -59,7 +59,8 @@ create their own roles in `terraform/modules/team_auth` and | 5 | **`agent-platform-backend-task`** | `portal` module | The EKS cluster's OIDC provider via **IRSA** — only the `backend` and `entry` service accounts in the `portal` namespace | The control-plane API on EKS: session routing, invoking runtimes, memory/scheduler/eval management, minting workspace credentials. | | 6 | **`agent-platform-schedule-runner`** | `portal` module (CDK: `PortalStack/ScheduleRunner`) | `lambda.amazonaws.com` | Fires scheduled invocations at each occurrence. Packages the same service layer as the backend. | | 7 | **`agent-platform-scheduler`** | `portal` module (CDK: `PortalStack/SchedulerRole`) | `scheduler.amazonaws.com` (conditioned on `aws:SourceAccount`) | The role EventBridge Scheduler assumes to invoke the runner Lambda and send to the DLQ. Holds no data-plane permissions. | -| 8 | **`agent-platform-llm-edge`** — *gateway mode only* | `llm_edge` module | The cluster's OIDC provider via **IRSA** — only the `edge` service account in the `llm-edge` namespace | The gateway broker. Holds exactly two grants: `secretsmanager:GetSecretValue` on the gateway key (the **only** principal that has it) and `dynamodb:GetItem` on the platform table for per-session token lookup — no `Query`, no `Scan`, nothing else. | +| 8 | **`agent-platform-llm-edge`** — *litellm gateway mode only* | `llm_edge` module | The cluster's OIDC provider via **IRSA** — only the `edge` service account in the `llm-edge` namespace | The gateway broker. Holds exactly two grants: `secretsmanager:GetSecretValue` on the gateway key (the **only** principal that has it) and `dynamodb:GetItem` on the platform table for per-session token lookup — no `Query`, no `Scan`, nothing else. | +| 9 | **AgentCore Gateway caller role** — *`agentcore_gateway` mode only* | supplied by the operator (`agentcore_gateway_caller_role_arn`); no module yet | The **backend role**, which must also be allowed `sts:TagSession` | The identity a *session* borrows. Holds one grant: `bedrock-agentcore:InvokeGateway` on the gateway ARN. The backend assumes it once per session with a session tag (`session_id`) and a session policy narrowing it to a single gateway, so the credentials a container ends up holding are strictly weaker than the role. Session revocation is an inline Deny on this role conditioned on `aws:PrincipalTag/session_id`, which is why the backend also needs `iam:PutRolePolicy`/`DeleteRolePolicy` on it. | The single most important property for a security review: **the browser and end users never hold AWS credentials.** All AWS access is server-side under @@ -90,7 +91,23 @@ tools role (#3) carries **only** the first three rows: | `Skills` / `SkillsList` | `s3:GetObject`; `s3:ListBucket` conditioned on `s3:prefix` | `skills/*` in the workspace bucket only | Mount skill packages before the agent starts. Read-only. | | *(no LLM gateway secret grant)* | — | — | Deliberately absent. A kernel role is reachable from inside the session it serves (root shell in the Dev Workbench microVM; agent tools in the headless kernel's CLI subprocess), so a kernel that can read the gateway key is a kernel whose users have it. Only `agent-platform-llm-edge` holds that read; kernels reach the gateway through it with a per-session grant. | | `InvokeMcpRuntimes` | `bedrock-agentcore:InvokeAgentRuntime` | `runtime/mcp_tools_kernel-*` and its `runtime-endpoint/*` only | Kernels reach the AgentCore-hosted MCP server through `mcp-proxy-for-aws`, which SigV4-signs with this role. Scoped to the MCP runtime name — **not** all runtimes. | -| `InvokeGateways` | `bedrock-agentcore:InvokeGateway` | `gateway/*` in this account, in both the platform's region and `us-east-1` | Kernels reach registry entries of kind `agentcore-gateway` (SigV4 through `mcp-proxy-for-aws`) — the feed pipelines' managed Web Search connector is one. Gateway IDs are generated at deploy time, so this is scoped by account+region rather than to a single gateway; `us-east-1` is listed explicitly because the Web Search connector is offered only there. | +| `InvokeGateways` | `bedrock-agentcore:InvokeGateway` | `gateway/*` in this account, in both the platform's region and `us-east-1` | Kernels reach registry entries of kind `agentcore-gateway` (SigV4 through `mcp-proxy-for-aws`) — the feed pipelines' managed Web Search connector is one. Gateway IDs are generated at deploy time, so this is scoped by account+region rather than to a single gateway; `us-east-1` is listed explicitly because the Web Search connector is offered only there. **⚠ Using the `agentcore_gateway` model backend requires narrowing this** — see the note below. | + +> **Prerequisite for the `agentcore_gateway` model backend.** The `InvokeGateways` +> statement above is a wildcard over `gateway/*`, which would also cover the +> *inference* gateway. That defeats the point of minting per-session credentials: +> a root user in the microVM can read the kernel role from the metadata endpoint +> and call the inference gateway on the role's own identity, with no session tag +> to revoke and no session policy to narrow it. Before enabling that backend, add +> an explicit **Deny** on the kernel roles for the inference gateway's ARN — Deny +> wins over the wildcard Allow, so tool gateways keep working while the inference +> gateway becomes reachable only with a backend-minted session credential: +> +> ```json +> { "Sid": "NoDirectInferenceGateway", "Effect": "Deny", +> "Action": "bedrock-agentcore:InvokeGateway", +> "Resource": "arn:aws:bedrock-agentcore:::gateway/" } +> ``` | `BuiltinTools` | `bedrock-agentcore:StartCodeInterpreterSession`, `InvokeCodeInterpreter`, `StopCodeInterpreterSession`, `GetCodeInterpreterSession`, `StartBrowserSession`, `StopBrowserSession`, `GetBrowserSession`, `UpdateBrowserStream`, `ConnectBrowserAutomationStream`, `ConnectBrowserLiveViewStream` | `code-interpreter/aws.codeinterpreter.v1`, `browser/aws.browser.v1` (AWS-managed), plus `{account}:code-interpreter/*` and `{account}:browser/*` (custom variants) | Code Interpreter and Browser built-in tools run in AWS-managed sandboxes under this role — no separate tool runtime to deploy. | | `BedrockInvoke` | `bedrock:InvokeModel`, `InvokeModelWithResponseStream` | `*` | The model control plane (Governance → Model backends) can route any agent to Bedrock per invocation. Cross-region inference profiles span regions, so this cannot be region-pinned. See [§4](#4-wildcard-resource-statements). | diff --git a/docs/security-explainer.zh.md b/docs/security-explainer.zh.md index 1f830d5..424a87a 100644 --- a/docs/security-explainer.zh.md +++ b/docs/security-explainer.zh.md @@ -442,19 +442,28 @@ CLI 子进程,agent 的工具就在那个子进程里执行。 (内核角色对 `workspaces/*` 零权限,由后端按 session 铸凭证,见第 4 节)。 模型凭证现在同样遵循。 -### 9.2 两个后端 +### 9.2 三个后端 -模型调用有两个后端,由治理页的模型控制面按 agent 路由: +模型调用有三个后端,由治理页的模型控制面按 agent 路由: - **Bedrock 直连**:以容器的 IAM 角色调用 `bedrock:InvokeModel*`,不存在长期 密钥,模型 ID 用 `global.` 前缀的跨区推理配置。 -- **LLM 网关(如 LiteLLM)**:容器**不持有网关密钥**。密钥只存在于 +- **LLM 网关(`litellm`,如 LiteLLM)**:容器**不持有网关密钥**。密钥只存在于 `llm-edge` 这个内部服务的任务角色里(`agent-platform-llm-edge`,全平台唯一 被授予该密钥读权的主体;运行时角色已不再有此授权)。内核拿到的是一份按 session 铸出的短期凭证。 +- **AgentCore Gateway(`agentcore_gateway`)**:容器同样不持有上游密钥,而且 + **平台侧也不持有** —— 上游凭证在网关自己的 token vault 里,因此不需要 + `llm-edge` 这个中间服务。内核拿到的是一份**打了 session 标签的 STS 凭证**, + 每个请求由内核 shim 做 SigV4 签名。 + +两种网关模式的共同点是:容器里没有可在别处兑现的东西。差别只在「谁替容器持有 +上游凭证」和「按会话吊销怎么表达」。 ### 9.3 网关模式的调用链 +**`litellm` 模式** + ``` Claude Code / Agent SDK ANTHROPIC_BASE_URL = http://127.0.0.1:8787 ← 容器内 loopback @@ -482,24 +491,87 @@ grant 里的上游地址、密钥名、模型白名单,全部由后端在铸 LiteLLM 的 master key),session 也无法借它调用网关管理接口给自己签发长期 key。仍建议存放按模型限定范围的虚拟 key,而不是 master key。 +**`agentcore_gateway` 模式** + +``` +Claude Code / Agent SDK + ANTHROPIC_BASE_URL = http://127.0.0.1:8787 ← 同上,仍是 loopback + ANTHROPIC_AUTH_TOKEN = <本次 invocation 的随机 token> + ↓ +loopback shim(内核进程内) + STS 凭证(AK/SK/SessionToken)只存在于该进程内存 + 按请求做 SigV4 签名;Claude Code 不会签名,这正是签名必须在这里而不是 + 在 CLI 子进程里的原因 + ↓ +AgentCore Gateway(IAM authorizer,公网端点但只认 SigV4) + 三层策略同时求值: + · caller 角色的身份策略 + · 铸凭证时附加的 session policy —— 只允许这一个 gateway + · 会话结束时写入的标签条件 Deny —— aws:PrincipalTag/session_id + ↓ +(可选)REQUEST interceptor:容器外唯一能做「容器内不可信之事」的位置 + ↓ +上游模型提供方(凭证在网关的 token vault,平台与容器都取不到) +``` + +这一模式下「路由不可篡改」由 IAM 与网关配置共同保证:gateway 地址写在凭证块 +里,模型到 target 的映射是网关侧配置。 + +**部署前置条件(否则 per-session 凭证形同虚设)**:内核角色为了访问 MCP 工具 +网关,已持有 `bedrock-agentcore:InvokeGateway` on `gateway/*`。这个通配同样覆盖 +推理网关,于是 microVM 内的 root 用户可以从 metadata 端点读出内核角色凭证、以 +**角色自身身份**直接调用推理网关 —— 既没有 session 标签可吊销,也没有 session +policy 可收窄。启用本后端前必须给内核角色加一条针对推理网关 ARN 的显式 +**Deny**(Deny 优先于通配 Allow,工具网关不受影响),见 `docs/permissions.md` +第 3 节。 + +**另一条要写明**:IAM 条件看不到请求体,所以**按会话的模型白名单无法表达为 +IAM 条件**。内核 shim 里的那道检查是 +fail-fast 优化而**不是**边界(会话用户在该 microVM 内是 root,可以绕过 shim +直连网关)。要真正强制,需要网关的 request interceptor,或一个模型一个 +gateway 并用 session policy 枚举资源。白名单无论如何都会记在 grant 上,所以 +决策本身是可审计的。 + ### 9.4 可验收的三条 1. 容器的环境变量、文件、进程命令行、Claude Code 会话里,**不存在任何真实 密钥**。原先放平台级网关 key 的位置现在是字面量 `unused`。 2. 把容器内一切可获取的材料(环境变量、文件、内存中的 session 凭证、IAM 角色 - 凭证)导出到外部机器,**均不可用**:`llm-edge` 的监听器在 VPC 内网,公网无 - 路由;session 凭证在一小时内过期,且只对该 session 被允许的模型有效。 -3. 在 session 内滥用只能消耗该 session 自己的额度,且归因到具体用户。 + 凭证)导出到外部机器,其可用性是**受限且有时限**的: + - `litellm` 模式:`llm-edge` 的监听器在 VPC 内网、公网无路由,凭证出了 VPC + 即废;grant 在一小时内过期。 + - `agentcore_gateway` 模式:网关端点在公网,但只接受 SigV4,且凭证被 + session policy 限死在**一个** gateway 上、并绑定 session 标签;STS 会话 + ≥15 分钟即到期,会话结束时另有标签条件 Deny 立即生效。 + 两种模式下,**上游模型提供方的真实凭证都不在导出物之内**。 +3. 每次调用可归因到具体 session 与用户。 ### 9.5 残余边界 用户可以在自己的 session 里直接调用那个 loopback 端口来使用模型。这是设计 -预期:这本就是他有权使用的额度,有上限、可归因、随 session 失效。安全团队 -评估时应当把它理解为"用户正常使用自己的配额",而不是越权。 +预期:这本就是他有权使用的额度,可归因、随 session 失效。安全团队评估时应当 +把它理解为"用户正常使用自己的配额",而不是越权。 + +**但"额度"这个词要说准**:平台侧的配额计的是调用次数,不是 token。按用户/团队 +的 **token 额度强制由上游网关按 key 施加**,平台不做 per-user 强制 —— 同一个 +后端下所有 session 共享同一把上游凭证,因此单个 session 有可能占满该凭证的 +TPM。若需要按用户限额,应在上游网关侧按会话签发受限凭证(例如 LiteLLM 的 +per-key `max_budget`),而不是在平台侧另造一套记账。 + +另一条残余边界:**凭证跨 session 复用挡不住。** 两种网关模式都以 bearer 或 +SigV4 凭证认证,而没有任何机制能证明"发起方确实是该 session"——`llm-edge` 的 +session id 与 token 都来自请求头,AgentCore 的 STS 凭证也可被读出后在别处使用 +(受 session policy 与标签 Deny 约束,但不受"哪个 microVM"约束)。要真正绑定 +需要 microVM 侧的可信证明,AgentCore Runtime 目前不提供。危害有界:同一用户的 +两个 session 之间总额不变,实质风险是**跨模型档位提权**(用被路由到更贵模型的 +会话凭证)与归因失真。 每次调用实际走了哪个后端、哪个模型,记入调用台账(`backend:model` 字段), -路由变更前后可审计;`llm-edge` 另有结构化日志记录每次调用的 session、用户、 -模型、状态与 token 用量。 +路由变更前后可审计;`litellm` 模式下 `llm-edge` 另有结构化日志记录每次调用的 +session、用户、模型、状态与 token 用量。`agentcore_gateway` 模式下**网关不产出 +token 用量指标或日志**(已实测:开启 vended logs 后逐条日志只含路由决策与 OTel +关联字段,无 token、无会话维度),因此 token 级归因要么由上游网关提供,要么由 +一个 RESPONSE interceptor 自行累加。 --- @@ -509,7 +581,8 @@ key。仍建议存放按模型限定范围的虚拟 key,而不是 master key |---|---|---| | Session 文件、对话历史 | Workspace S3 桶 | 公共访问全阻断、强制 SSL、静态加密、仅两个角色可访问 | | 平台记录(session/agent/schedule/台账/审计) | DynamoDB | 静态加密(默认 AWS 拥有密钥,可换 CMK) | -| LLM 网关 key | Secrets Manager | 加密存储;仅 `llm-edge` 任务角色可读,内核角色无此授权;从不写入镜像,也从不进入任何 session 容器 | +| LLM 网关 key(`litellm` 模式) | Secrets Manager | 加密存储;仅 `llm-edge` 任务角色可读,内核角色无此授权;从不写入镜像,也从不进入任何 session 容器 | +| 上游模型凭证(`agentcore_gateway` 模式) | AgentCore Gateway 的 token vault | 平台侧不存储、不可读;由网关在出站时注入。平台只持有一个可 `AssumeRole` 的 caller 角色 ARN | | 可选的第三方 MCP key、管理员口令 | Secrets Manager | 加密存储;按密钥名精确授权;容器/Lambda 启动时读取,从不写入镜像 | | 内核与平台日志 | CloudWatch Logs | 前缀级授权;平台自有日志组 7 天保留 | @@ -551,10 +624,22 @@ AgentCore 服务侧的数据(session 元数据、Memory 记录等)默认用 4. **headless 异步任务的网关凭证寿命较长**。异步运行最长到平台的 8 小时上限, 且没有续期通道,因此其 grant 的有效期与之匹配(9 小时)而不是 1 小时。影响 有限:headless 内核没有终端,且该 grant 从不进入 agent 子进程的环境(内核 - 只给子进程一个容器内本地令牌,见第 9.3 节)。 -5. **加密密钥默认为 AWS 托管**:合规要求 CMK 时,S3/DynamoDB/AgentCore 资源 + 只给子进程一个容器内本地令牌,见第 9.3 节)。`agentcore_gateway` 模式下这个 + 时长要求 caller 角色的 `MaxSessionDuration` 覆盖它,否则 `AssumeRole` 会拒绝 + 而调用 fail-closed。 +5. **`agentcore_gateway` 模式下,网关 interceptor 抛出的异常会原样回传给调用方** + —— 包含异常消息与**完整堆栈(含源码行)**,且与 `exceptionLevel` 的设置无关 + (已实测)。调用方就是租户的 microVM。因此 interceptor 必须自己捕获全部异常 + 并返回通用错误,绝不可让异常逸出;尤其是按 AWS 参考实现从 Secrets Manager + 取凭证的 interceptor,一旦抛错就可能把密钥名或内部 ARN 送到租户面前。 + 相关的正面结论:interceptor 失败时网关是 **fail-closed** 的(请求被拒,不会 + 放行),这一点也已实测。 +6. **`agentcore_gateway` 模式的单次请求有 15 分钟硬上限**(服务配额 + `Request timeout`,不可调整)。单次模型调用受此约束;流式响应实测 46 秒、 + 3980 帧无断流,距上限有充足余量。 +7. **加密密钥默认为 AWS 托管**:合规要求 CMK 时,S3/DynamoDB/AgentCore 资源 均可切换,代价是给相关角色增加精确限定到该密钥 ARN 的 KMS 权限。 -6. **所有平台角色都可加权限边界**,保证未来任何代码改动都不能越过边界扩权。 +8. **所有平台角色都可加权限边界**,保证未来任何代码改动都不能越过边界扩权。 另见第 9.5 节:用户可在自己 session 内直接调用 loopback 端口消耗自己的配额, 这是设计预期而非越权。 diff --git a/docs/user-guide.md b/docs/user-guide.md index e7896eb..c641333 100644 --- a/docs/user-guide.md +++ b/docs/user-guide.md @@ -463,15 +463,18 @@ The **Model backends** card on the Governance page is the routing control plane for every model call — headless invocations and Dev Workbench sessions alike: -- **Two backends** — Amazon Bedrock (direct, via the kernel container's IAM - role; use `global.` cross-region inference profile IDs) and an - Anthropic-compatible **LLM gateway** (e.g. LiteLLM; the API key lives in - Secrets Manager, only its *name* is stored here, and only the `llm-edge` - service can read it — a session container never receives it). Each has an - enable switch and a model catalog that feeds the dropdowns elsewhere. - Gateway mode requires `llm-edge` to be deployed (`enable_llm_edge`); with it - missing, the platform refuses to route a session rather than falling back to - handing out the key. +- **Three backends** — Amazon Bedrock (direct, via the kernel container's IAM + role; use `global.` cross-region inference profile IDs); an + Anthropic-compatible **LLM gateway** (`litellm`, e.g. LiteLLM; the API key + lives in Secrets Manager, only its *name* is stored here, and only the + `llm-edge` service can read it — a session container never receives it); and + **AgentCore Gateway** (`agentcore_gateway`, where the upstream credential + lives in the gateway's own token vault, so there is no key name to store and + no broker service to deploy — a session gets STS credentials tagged with its + own id). Each has an enable switch and a model catalog that feeds the + dropdowns elsewhere. Either gateway mode refuses to route a session when its + prerequisite is missing (`enable_llm_edge`, or the caller role ARN) rather + than falling back to handing out a shared credential. - **Platform default** — which backend an agent uses when it doesn't pick one. - **Per-agent choice** — the Publish page's edit dialog has a *Model diff --git a/runtimes/agent-sdk-kernel/src/llm_shim.py b/runtimes/agent-sdk-kernel/src/llm_shim.py index 8cd4d59..d8db6ef 100644 --- a/runtimes/agent-sdk-kernel/src/llm_shim.py +++ b/runtimes/agent-sdk-kernel/src/llm_shim.py @@ -16,6 +16,14 @@ Per-invocation registration is also what makes concurrent runs safe: two agents routed to different backends each get their own local token, so neither can borrow the other's grant. + +Two upstream shapes are supported, decided by the grant the platform delivers: + +- ``llm-edge``: a bearer token plus the session id, which the edge re-reads its + own grant from. +- ``agentcore_gateway``: STS credentials tagged with the session id, which this + shim SigV4-signs each request with. Claude Code cannot sign, which is exactly + why the signing has to happen here rather than in the CLI subprocess. """ from __future__ import annotations @@ -62,12 +70,53 @@ "content-length", } +# On the SigV4 path a client-supplied x-amz-* header would either be folded +# into the signature or contradict it, so the client never gets to set one. +_SIGV4_RESERVED_PREFIX = "x-amz-" + + +def _sign_sigv4(grant: dict, method: str, url: str, body: bytes) -> dict: + """Return the SigV4 headers for one request against the gateway. + + Only a minimal, deterministic header set is signed (content-type, plus the + host and date botocore adds). Everything the caller sent is attached + *after* signing: headers outside SignedHeaders do not participate in the + signature, so passing Claude Code's ``anthropic-*`` and ``x-stainless-*`` + headers through cannot invalidate it. + """ + from botocore.auth import SigV4Auth + from botocore.awsrequest import AWSRequest + from botocore.credentials import Credentials + + creds = Credentials( + grant["access_key_id"], + grant["secret_access_key"], + grant.get("session_token") or None, + ) + req = AWSRequest( + method=method, + url=url, + data=body, + headers={"content-type": "application/json"}, + ) + SigV4Auth( + creds, + str(grant.get("service") or "bedrock-agentcore"), + str(grant.get("region") or ""), + ).add_auth(req) + return dict(req.headers) + def register(grant: dict) -> str: """Register a platform grant for one invocation; returns the local token to put in the CLI subprocess's ANTHROPIC_AUTH_TOKEN.""" - if not grant or not grant.get("endpoint") or not grant.get("token"): - raise ValueError("gateway grant is missing endpoint/token") + if not grant or not grant.get("endpoint"): + raise ValueError("gateway grant is missing endpoint") + if not grant.get("token") and not grant.get("access_key_id"): + raise ValueError( + "gateway grant carries neither a bearer token (llm-edge) nor STS " + "credentials (agentcore_gateway)" + ) local = secrets.token_urlsafe(24) with _lock: _grants[local] = grant @@ -122,13 +171,49 @@ def _proxy(self) -> None: length = int(self.headers.get("Content-Length") or 0) body = self.rfile.read(length) if length else b"" - headers = { + sigv4 = bool(grant.get("access_key_id")) + passthrough = { k: v for k, v in self.headers.items() if k.lower() not in _DROP_REQUEST_HEADERS + and not (sigv4 and k.lower().startswith(_SIGV4_RESERVED_PREFIX)) } - headers["Authorization"] = f"Bearer {grant['token']}" - headers["x-platform-session-id"] = str(grant.get("session_id") or "") + + if sigv4: + # Fail fast on a model this session was not routed to. This is an + # optimisation, not an authorization boundary: the session's user + # is root in this microVM and can bypass the shim. Real enforcement + # has to sit outside the container — see the platform's + # llm_credentials_service.mint_agentcore docstring. + allowed = [str(m) for m in (grant.get("allowed_models") or [])] + if allowed and body: + try: + requested = str(json.loads(body).get("model") or "") + except Exception: # noqa: BLE001 - non-JSON bodies just pass + requested = "" + if requested and requested not in allowed: + self._fail( + 403, f"model {requested!r} is not permitted for this session" + ) + return + headers = _sign_sigv4( + grant, + self.command, + str(grant["endpoint"]).rstrip("/") + self.path, + body, + ) + # Anything that took part in the signature must keep the signed + # value: the caller's own content-type would otherwise silently + # invalidate it. + signed = {k.lower() for k in headers} + headers.update( + {k: v for k, v in passthrough.items() if k.lower() not in signed} + ) + else: + headers = passthrough + headers["Authorization"] = f"Bearer {grant['token']}" + headers["x-platform-session-id"] = str(grant.get("session_id") or "") + headers["Accept-Encoding"] = "identity" headers["Content-Length"] = str(len(body)) diff --git a/runtimes/agent-sdk-kernel/src/main.py b/runtimes/agent-sdk-kernel/src/main.py index 7b78aea..1f35f17 100644 --- a/runtimes/agent-sdk-kernel/src/main.py +++ b/runtimes/agent-sdk-kernel/src/main.py @@ -317,14 +317,18 @@ def build_model_env( if small: env["ANTHROPIC_SMALL_FAST_MODEL"] = small return env, model, "" - if backend == "gateway": + if backend in ("gateway", "agentcore_gateway"): # No key is fetched and the upstream address is not even in the spec — # the backend strips base_url/secret_name before sending it here. The # CLI talks to the loopback shim with a token that means nothing outside # this container and nothing after this invocation. + # + # Which credential the shim then presents upstream is the grant's + # business: a bearer token for llm-edge, or SigV4 over per-session STS + # credentials for an AgentCore Gateway. The CLI sees neither. if not grant: raise ValueError( - "model.backend=gateway requires llm_credentials in the payload" + f"model.backend={backend} requires llm_credentials in the payload" ) local_token = llm_shim.register(grant) env = { diff --git a/runtimes/claude-code-kernel/contract-server/main.js b/runtimes/claude-code-kernel/contract-server/main.js index 1deb675..b0d9b30 100644 --- a/runtimes/claude-code-kernel/contract-server/main.js +++ b/runtimes/claude-code-kernel/contract-server/main.js @@ -14,6 +14,7 @@ */ const http = require("http"); const https = require("https"); +const crypto = require("crypto"); const fs = require("fs"); const url = require("url"); const { spawn, execFile } = require("child_process"); @@ -124,13 +125,78 @@ setInterval(refreshWorkspaceCredentials, 5 * 60_000); // models this session may use on every call. // --------------------------------------------------------------------------- const LLM_SHIM_PORT = 8787; +// Only the SigV4 path buffers a request body (it needs the payload hash), so +// cap it rather than let a malformed or hostile body exhaust the kernel. +const SHIM_MAX_BODY_BYTES = 32 * 1024 * 1024; // {endpoint, token, expires_at} — set from the warmup payload and rotated by // the same refresh call that renews workspace credentials. let llmGrant = null; +// --------------------------------------------------------------------------- +// SigV4, by hand. +// +// The agentcore_gateway grant is a set of STS credentials rather than a bearer +// token, so each request has to be signed. Claude Code cannot sign — which is +// the whole reason the signing belongs here and not in the CLI subprocess. +// +// Written against node:crypto rather than pulling in an AWS SDK: this contract +// server has exactly one dependency today, and a signing routine is 40 lines. +// --------------------------------------------------------------------------- +const sha256Hex = (b) => crypto.createHash("sha256").update(b).digest("hex"); +const hmac = (key, msg) => crypto.createHmac("sha256", key).update(msg).digest(); + +function sigv4Headers(grant, method, target, body) { + const now = new Date(); + const amzDate = now.toISOString().replace(/[:-]|\.\d{3}/g, ""); + const dateStamp = amzDate.slice(0, 8); + const region = String(grant.region || ""); + const service = String(grant.service || "bedrock-agentcore"); + + // Only a minimal, deterministic set is signed. Everything the caller sent is + // attached after signing, where it cannot affect the signature. + const signed = { + host: target.host, + "content-type": "application/json", + "x-amz-date": amzDate, + }; + if (grant.session_token) signed["x-amz-security-token"] = grant.session_token; + + const names = Object.keys(signed).sort(); + const canonicalHeaders = names.map((n) => `${n}:${String(signed[n]).trim()}\n`).join(""); + const signedHeaders = names.join(";"); + const canonicalRequest = [ + method, + target.pathname, + target.searchParams ? target.searchParams.toString() : "", + canonicalHeaders, + signedHeaders, + sha256Hex(body), + ].join("\n"); + + const scope = `${dateStamp}/${region}/${service}/aws4_request`; + const stringToSign = [ + "AWS4-HMAC-SHA256", + amzDate, + scope, + sha256Hex(Buffer.from(canonicalRequest, "utf8")), + ].join("\n"); + + const kDate = hmac(`AWS4${grant.secret_access_key}`, dateStamp); + const kSigning = hmac(hmac(hmac(kDate, region), service), "aws4_request"); + const signature = crypto.createHmac("sha256", kSigning).update(stringToSign).digest("hex"); + + return { + ...signed, + authorization: + `AWS4-HMAC-SHA256 Credential=${grant.access_key_id}/${scope}, ` + + `SignedHeaders=${signedHeaders}, Signature=${signature}`, + }; +} + function setLlmGrant(grant) { - if (!grant || !grant.endpoint || !grant.token) return; + if (!grant || !grant.endpoint) return; + if (!grant.token && !grant.access_key_id) return; llmGrant = grant; console.log( `[model] gateway grant active via ${grant.endpoint}` + @@ -180,53 +246,108 @@ function startLlmShim() { return; } - const headers = {}; + const sigv4 = Boolean(llmGrant.access_key_id); + const passthrough = {}; for (const [k, v] of Object.entries(req.headers)) { - if (!SHIM_DROP_HEADERS.has(k.toLowerCase())) headers[k] = v; + const lk = k.toLowerCase(); + if (SHIM_DROP_HEADERS.has(lk)) continue; + // On the SigV4 path a caller-supplied x-amz-* header would either be + // folded into the signature or contradict it. + if (sigv4 && lk.startsWith("x-amz-")) continue; + passthrough[k] = v; } - headers["authorization"] = `Bearer ${llmGrant.token}`; - // The grant names the session it was issued for; fall back to the ID - // AgentCore injected in case an older backend omits it. - headers["x-platform-session-id"] = llmGrant.session_id || runtimeSessionId || ""; - headers["accept-encoding"] = "identity"; - - const client = target.protocol === "https:" ? https : http; - const upstream = client.request( - target, - { method: req.method, headers, timeout: 15 * 60_000 }, - (up) => { - const out = {}; - for (const [k, v] of Object.entries(up.headers)) { - if (!["connection", "keep-alive", "transfer-encoding"].includes(k.toLowerCase())) { - out[k] = v; + + // body === null means "stream it through unread", which is what the edge + // path does: the shim has no reason to inspect it and the edge authorizes + // the requested model. SigV4 needs the payload hash, so that path buffers. + const forward = (body) => { + let headers; + if (sigv4) { + headers = sigv4Headers(llmGrant, req.method, target, body); + // Anything that took part in the signature keeps its signed value. + const signed = new Set(Object.keys(headers)); + for (const [k, v] of Object.entries(passthrough)) { + if (!signed.has(k.toLowerCase())) headers[k] = v; + } + } else { + headers = { ...passthrough }; + headers["authorization"] = `Bearer ${llmGrant.token}`; + // The grant names the session it was issued for; fall back to the ID + // AgentCore injected in case an older backend omits it. + headers["x-platform-session-id"] = llmGrant.session_id || runtimeSessionId || ""; + } + headers["accept-encoding"] = "identity"; + if (body !== null) headers["content-length"] = String(body.length); + + const client = target.protocol === "https:" ? https : http; + const upstream = client.request( + target, + { method: req.method, headers, timeout: 15 * 60_000 }, + (up) => { + const out = {}; + for (const [k, v] of Object.entries(up.headers)) { + if (!["connection", "keep-alive", "transfer-encoding"].includes(k.toLowerCase())) { + out[k] = v; + } } + res.writeHead(up.statusCode || 502, out); + // Straight pipe: nothing here may buffer, or token-by-token output + // would arrive as one blob at the end of the response. + up.pipe(res); + }, + ); + + upstream.on("timeout", () => upstream.destroy(new Error("edge timeout"))); + upstream.on("error", (e) => { + console.log(`[model] upstream request failed: ${e.message}`); + if (!res.headersSent) { + res.writeHead(502, { "Content-Type": "application/json" }); + res.end( + JSON.stringify({ + type: "error", + error: { type: "upstream_error", message: e.message }, + }), + ); + } else { + res.destroy(); } - res.writeHead(up.statusCode || 502, out); - // Straight pipe: nothing here may buffer, or token-by-token output - // would arrive as one blob at the end of the response. - up.pipe(res); - }, - ); + }); + + if (body === null) req.pipe(upstream); + else upstream.end(body); + }; + + if (!sigv4) { + forward(null); + return; + } - upstream.on("timeout", () => upstream.destroy(new Error("edge timeout"))); - upstream.on("error", (e) => { - console.log(`[model] edge request failed: ${e.message}`); - if (!res.headersSent) { - res.writeHead(502, { "Content-Type": "application/json" }); + const chunks = []; + let total = 0; + let aborted = false; + req.on("data", (c) => { + if (aborted) return; + total += c.length; + if (total > SHIM_MAX_BODY_BYTES) { + aborted = true; + res.writeHead(413, { "Content-Type": "application/json" }); res.end( JSON.stringify({ type: "error", - error: { type: "upstream_error", message: e.message }, + error: { type: "request_too_large", message: "request body too large" }, }), ); - } else { - res.destroy(); + req.destroy(); + return; } + chunks.push(c); + }); + req.on("end", () => { + if (!aborted) forward(Buffer.concat(chunks)); + }); + req.on("error", () => { + aborted = true; }); - - // The request body is streamed through unread — the shim has no reason to - // inspect it, and the edge is what authorizes the requested model. - req.pipe(upstream); }); server.requestTimeout = 0; @@ -282,7 +403,7 @@ function applyModelSpec(spec) { if (spec.small_fast_model) lines.push(`export ANTHROPIC_SMALL_FAST_MODEL=${shq(spec.small_fast_model)}`); // keep the container's baked-in (Bedrock) alias steering - } else if (backend === "gateway") { + } else if (backend === "gateway" || backend === "agentcore_gateway") { // No credential is written here, and the upstream gateway's address is not // even known to this container: the spec arrives with base_url and // secret_name stripped. Claude Code is pointed at the loopback shim, which diff --git a/scripts/e2e_agentcore_gateway.py b/scripts/e2e_agentcore_gateway.py new file mode 100755 index 0000000..224e648 --- /dev/null +++ b/scripts/e2e_agentcore_gateway.py @@ -0,0 +1,886 @@ +#!/usr/bin/env python3 +"""End-to-end test for the agentcore_gateway model backend, against real AWS. + +Unlike the moto-based suites this one provisions a live AgentCore Gateway and +drives the *real* Claude Code CLI through the *real* kernel shim, so the whole +authentication chain is exercised as deployed: + + Claude Code subprocess per-invocation loopback token + -> llm_shim (kernel) SigV4 over per-session STS credentials + -> AgentCore Gateway IAM authorizer + session policy + -> LiteLLM /v1/messages API key injected from the token vault + -> the upstream model LiteLLM routes to + +Nothing in the chain is stubbed. The only simulation is the microVM itself: +the shim runs in this process instead of inside AgentCore Runtime, which is +exactly where it runs in production (in the kernel process), so the code path +is identical. + +Asserts: + + [chain] Claude Code completes a turn through the shim, and the shim + observed the request — i.e. the CLI really used the gateway and + not some ambient credential. + [protocol] which paths Claude Code actually sends (reported, and asserted to + be a subset of what AgentCore Gateway accepts inbound). + [scoping] the session policy narrows the credential to one gateway: the + same credential is refused on a second gateway. + [failfast] the shim refuses a model this session was not routed to (an + optimisation, not a boundary — see mint_agentcore's docstring). + [revoke] llm_credentials_service.revoke() stops this session's + credentials while a second live session keeps working. + +Usage: python3 scripts/e2e_agentcore_gateway.py [--region us-east-1] [--keep] + +Configuration is by environment: LITELLM_ENDPOINT and +LITELLM_CREDENTIAL_PROVIDER_ARN are required; LITELLM_ALLOWED_MODELS, +LITELLM_CHAIN_MODEL and LITELLM_PATHS have defaults. For a LiteLLM that is not reachable from +the internet, set LITELLM_ROUTING_DOMAIN (an internal ALB's DNS name), +LITELLM_VPC_ID, LITELLM_SUBNET_IDS and LITELLM_LATTICE_SG_IDS: the target then +gets a managed VPC Lattice private endpoint and the gateway never leaves AWS's +network on the way to LiteLLM. + +Requires: credentials able to create IAM roles, an AgentCore Gateway and a +DynamoDB table, plus the `claude` CLI on PATH. Everything it creates is torn +down unless --keep is passed. +""" +import argparse +import json +import os +import secrets +import shutil +import subprocess +import sys +import time +import urllib.error +import urllib.request + +import boto3 +from botocore.exceptions import ClientError + +REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +FAIL: list[str] = [] +SUFFIX = secrets.token_hex(4) + +# The gateway target routes to this LiteLLM deployment. Overridable so a fork +# can point at a staging instance, but the default is the one Prod-LiteLLM +# already runs on this account. +# The gateway target routes to this LiteLLM deployment, and this API-key +# credential provider holds the LiteLLM virtual key. Both are required. The +# provider is created out of band (AWS Console: AgentCore -> Identity -> API +# key credential providers) so a live key never lands in this script or its +# logs; the e2e's job is to prove the provider ARN and the target wiring work +# end to end. +LITELLM_ENDPOINT = os.environ.get("LITELLM_ENDPOINT", "") +LITELLM_CREDENTIAL_PROVIDER_ARN = os.environ.get("LITELLM_CREDENTIAL_PROVIDER_ARN", "") + +# The list of models the virtual key permits. The gateway target refuses any +# model outside this set at its own layer, and LiteLLM refuses it at the key +# layer — either check is enough on its own, both layers are here on purpose +# so a misconfiguration surfaces immediately rather than at the far end. +LITELLM_ALLOWED_MODELS = [ + m.strip() + for m in os.environ.get( + "LITELLM_ALLOWED_MODELS", + "claude-opus-5-5,claude-sonnet-5-5,gpt-6-astra,kimi-k3", + ).split(",") + if m.strip() +] + +# Private path to LiteLLM. When LITELLM_ROUTING_DOMAIN is set the target gets a +# managed VPC Lattice private endpoint: AgentCore places resource-gateway ENIs +# in these subnets and connects to the routing domain (an internal ALB), while +# TLS SNI / Host stay on LITELLM_ENDPOINT's name so the ALB's public cert +# matches. Unset, the target dials LITELLM_ENDPOINT over the internet. +LITELLM_ROUTING_DOMAIN = os.environ.get("LITELLM_ROUTING_DOMAIN", "") +LITELLM_VPC_ID = os.environ.get("LITELLM_VPC_ID", "") +LITELLM_SUBNET_IDS = [ + s.strip() for s in os.environ.get("LITELLM_SUBNET_IDS", "").split(",") if s.strip() +] +LITELLM_LATTICE_SG_IDS = [ + s.strip() for s in os.environ.get("LITELLM_LATTICE_SG_IDS", "").split(",") if s.strip() +] + +# Inbound paths the target accepts. The gateway does no protocol translation: +# it matches path + model, swaps in the API key and relays the body (and the +# SSE stream) verbatim, so whether a given model works on a given path is +# LiteLLM's call. Every allowed model is declared on every path here and the +# [protocol-matrix] probe reports what LiteLLM actually serves. +LITELLM_PATHS = [ + p.strip() + for p in os.environ.get( + "LITELLM_PATHS", "/v1/messages,/v1/chat/completions,/v1/responses" + ).split(",") + if p.strip() +] + +# The one used for the [chain] / multi-turn / revoke / failfast steps, which +# drive the real Claude Code CLI and therefore need an Anthropic-native model. +LITELLM_CHAIN_MODEL = os.environ.get("LITELLM_CHAIN_MODEL", "claude-sonnet-5-5") + +# Anything in the allowed set but not the chain model gets exercised by the +# lightweight [multimodel] smoke: one /v1/messages "hi" per model, no CLI, no +# tools. That is where any Anthropic-to-OpenAI protocol translation LiteLLM +# does for gpt-* / kimi-* is stressed. +LITELLM_SMOKE_MODELS = [ + m for m in LITELLM_ALLOWED_MODELS if m != LITELLM_CHAIN_MODEL +] + +# No REQUEST interceptor. Claude Code sends first-party-only fields +# (context_management today) that LiteLLM's Bedrock adapter rejects; the +# LiteLLM model entries carry `drop_params: true` so LiteLLM strips them. An +# interceptor Lambda could do the same at the gateway, but every model call +# then pays a synchronous Lambda hop and the request body is capped by +# Lambda's 6 MB invoke payload (about 4.5 MB of JSON after base64), which a +# long Claude Code session can reach. + + +def check(label: str, ok: bool, detail: str = "") -> None: + print(f" {'ok ' if ok else 'FAIL'} {label}" + (f" ({detail})" if detail else "")) + if not ok: + FAIL.append(label) + + +def step(msg: str) -> None: + print(f"\n=== {msg}") + + +# --------------------------------------------------------------- provisioning + + +class Fixture: + """Live AWS resources for one run.""" + + def __init__(self, region: str) -> None: + self.region = region + self.iam = boto3.client("iam") + self.sts = boto3.client("sts") + self.ddb = boto3.client("dynamodb", region_name=region) + self.acc = boto3.client("bedrock-agentcore-control", region_name=region) + self.account = self.sts.get_caller_identity()["Account"] + self.gw_role = f"e2e-acgw-exec-{SUFFIX}" + self.caller_role = f"e2e-acgw-caller-{SUFFIX}" + self.table = f"e2e-acgw-{SUFFIX}" + self.gateways: list[str] = [] + + # -- helpers ---------------------------------------------------------- + + def _caller_principal(self) -> str: + """The role this script runs as, so the caller role can trust it. + + In production the trusting principal is the backend's IRSA role; here + it is whatever identity is running the test. SSO reserved roles + (``arn:aws:iam:::role/AWSReservedSSO_*``) cannot be used as an + AssumeRole *Principal* — IAM refuses the trust policy with + MalformedPolicyDocument — so for those we widen to the account root. + That is safe here because the role lives for the length of one run. + """ + arn = self.sts.get_caller_identity()["Arn"] + if ":assumed-role/" in arn: + name = arn.split(":assumed-role/")[1].split("/")[0] + if name.startswith("AWSReservedSSO_"): + return f"arn:aws:iam::{self.account}:root" + return f"arn:aws:iam::{self.account}:role/{name}" + return arn + + def _gw_role_policy(self) -> dict: + """Policy for the gateway execution role. + + The one capability needed is retrieving the outbound API key (the + LiteLLM virtual key) from AgentCore Identity's token vault during a + request; without the GetResourceApiKey / secretsmanager + grants the gateway returns 400 "Failed to fetch outbound api key. + Failed to get workload identity token" and every call — chain, + scoping, revoke — surfaces the same error. + """ + vault = f"arn:aws:bedrock-agentcore:{self.region}:{self.account}" + provider_name = LITELLM_CREDENTIAL_PROVIDER_ARN.rsplit("/", 1)[-1] + return { + "Version": "2012-10-17", + "Statement": [ + { + # Two actions, one resource set. GetWorkloadAccessToken + # is the first hop the gateway makes: it exchanges its + # execution role for a workload-scoped token, which it + # then presents to GetResourceApiKey to unwrap the + # LiteLLM key. Without the first action the docs' policy + # is not sufficient — this is the exact failure mode we + # observed as "Failed to get workload identity token" + # despite GetResourceApiKey being granted. + "Sid": "AgentCoreApiKeyTokenVaultDefault", + "Effect": "Allow", + "Action": [ + "bedrock-agentcore:GetWorkloadAccessToken", + "bedrock-agentcore:GetResourceApiKey", + ], + "Resource": [ + f"{vault}:token-vault/default", + f"{vault}:workload-identity-directory/default", + f"{vault}:workload-identity-directory/default/" + "workload-identity/*", + ], + }, + { + "Sid": "AgentCoreApiKeyTokenVaultPerKey", + "Effect": "Allow", + "Action": "bedrock-agentcore:GetResourceApiKey", + "Resource": LITELLM_CREDENTIAL_PROVIDER_ARN, + }, + { + "Sid": "AgentCoreApiKeySecret", + "Effect": "Allow", + "Action": "secretsmanager:GetSecretValue", + "Resource": ( + f"arn:aws:secretsmanager:{self.region}:" + f"{self.account}:secret:" + f"bedrock-agentcore-identity!default/apikey/" + f"{provider_name}-*" + ), + }, + ], + } + + def _role(self, name: str, trust: dict, policy: dict, max_session: int = 3600) -> str: + try: + arn = self.iam.create_role( + RoleName=name, + AssumeRolePolicyDocument=json.dumps(trust), + MaxSessionDuration=max_session, + Description="temporary: e2e_agentcore_gateway", + )["Role"]["Arn"] + except ClientError as e: + if e.response["Error"]["Code"] != "EntityAlreadyExists": + raise + arn = self.iam.get_role(RoleName=name)["Role"]["Arn"] + self.iam.put_role_policy( + RoleName=name, PolicyName="e2e", PolicyDocument=json.dumps(policy) + ) + return arn + + # -- create ----------------------------------------------------------- + + def create(self) -> None: + step("provisioning live AWS resources") + self.ddb.create_table( + TableName=self.table, + KeySchema=[ + {"AttributeName": "PK", "KeyType": "HASH"}, + {"AttributeName": "SK", "KeyType": "RANGE"}, + ], + AttributeDefinitions=[ + {"AttributeName": "PK", "AttributeType": "S"}, + {"AttributeName": "SK", "AttributeType": "S"}, + ], + BillingMode="PAY_PER_REQUEST", + ) + self.ddb.get_waiter("table_exists").wait(TableName=self.table) + print(f" table {self.table}") + + gw_trust = { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Principal": {"Service": "bedrock-agentcore.amazonaws.com"}, + "Action": "sts:AssumeRole", + "Condition": {"StringEquals": {"aws:SourceAccount": self.account}}, + } + ], + } + # The gateway execution role must reach AgentCore Identity's token + # vault so the LiteLLM API key can be fetched and injected into + # outbound requests. + gw_policy = self._gw_role_policy() + gw_arn_role = self._role(self.gw_role, gw_trust, gw_policy) + print(f" role {self.gw_role} (gateway execution)") + + # Two gateways: the session credential is scoped to the first, and the + # second exists only to prove the scoping actually bites. + for label in ("primary", "decoy"): + g = self.acc.create_gateway( + name=f"e2e-{label}-{SUFFIX}", + roleArn=gw_arn_role, + protocolType="MCP", + authorizerType="AWS_IAM", + exceptionLevel="DEBUG", + description="temporary: e2e_agentcore_gateway", + ) + self.gateways.append(g["gatewayId"]) + print(f" gateway {g['gatewayId']} ({label})") + for gid in self.gateways: + for _ in range(40): + if self.acc.get_gateway(gatewayIdentifier=gid)["status"] != "CREATING": + break + time.sleep(4) + # IAM propagation before the gateway's first outbound. + time.sleep(15) + for gid in self.gateways: + t = self.acc.create_gateway_target( + gatewayIdentifier=gid, + name="litellm", + # An inference *provider* target rather than the built-in + # bedrock-mantle connector: the gateway forwards Anthropic + # Messages requests to our LiteLLM deployment, and LiteLLM + # keeps its multi-provider routing (Bedrock Anthropic today, + # OpenAI / Gemini tomorrow) exactly as it does for every other + # client. What this backend gives up in return is the + # llm-edge pod — one fewer service on the platform side, no + # provider key living anywhere in our EKS. + targetConfiguration={ + "inference": { + "provider": { + "endpoint": LITELLM_ENDPOINT, + "operations": [ + { + "path": path, + # The models named here are what a caller + # is *allowed* to route to; the shim + # additionally fail-fasts against the + # per-session allowlist minted by the + # backend. + "models": [ + {"model": m} + for m in LITELLM_ALLOWED_MODELS + ], + } + for path in LITELLM_PATHS + ], + } + } + }, + # An API_KEY provider held in AgentCore Identity's token vault + # substitutes the caller's Authorization header on the way + # out, so a container never sees the LiteLLM key. The + # provider is created out of band (Console) and referenced by + # ARN; see LITELLM_CREDENTIAL_PROVIDER_ARN at the top. + credentialProviderConfigurations=[ + { + "credentialProviderType": "API_KEY", + "credentialProvider": { + "apiKeyCredentialProvider": { + "providerArn": LITELLM_CREDENTIAL_PROVIDER_ARN, + "credentialLocation": "HEADER", + "credentialParameterName": "Authorization", + # No trailing space: the gateway inserts the + # separator itself. "Bearer " goes out as + # "Bearer " and LiteLLM rejects it. + "credentialPrefix": "Bearer", + } + }, + } + ], + # Curating headers here is the AgentCore-side counterpart of + # llm-edge's FORWARD_HEADERS allowlist, and it is not optional: + # left unset the gateway relays whatever the caller sent, + # including the caller's own x-amz-security-token. LiteLLM + # accepts anthropic-beta (so Claude Code's prompt-caching + # advertisement passes through), unlike the bedrock-mantle + # connector, which is the one payoff of not routing through + # that connector. content-type has to be listed explicitly: + # once allowedRequestHeaders is set, a connector-style target + # stops relaying it. + metadataConfiguration={ + "allowedRequestHeaders": [ + "content-type", + "anthropic-version", + "anthropic-beta", + "accept", + ] + }, + **self._private_endpoint(), + ) + for _ in range(90): + st = self.acc.get_gateway_target( + gatewayIdentifier=gid, targetId=t["targetId"] + ) + if st["status"] != "CREATING": + break + time.sleep(4) + print(f" target {gid} -> {st['status']} {st.get('statusReasons', '')}") + for r in st.get("privateEndpointManagedResources") or []: + print(f" private endpoint {r.get('domain')} via {r.get('resourceGatewayArn')}") + if st["status"] != "READY": + raise SystemExit("inference target did not become READY") + + self.caller_arn = self._role( + self.caller_role, + { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Principal": {"AWS": self._caller_principal()}, + "Action": ["sts:AssumeRole", "sts:TagSession"], + } + ], + }, + { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Action": "bedrock-agentcore:InvokeGateway", + "Resource": [ + f"arn:aws:bedrock-agentcore:{self.region}:{self.account}:gateway/{g}" + for g in self.gateways + ], + } + ], + }, + ) + print(f" role {self.caller_role} (per-session caller)") + # The role policy deliberately allows BOTH gateways; the narrowing to + # one is the session policy the service attaches at mint time, so the + # scoping assertion tests the session policy and not this role. + time.sleep(12) + + def _private_endpoint(self) -> dict: + """Top-level privateEndpoint kwarg for create_gateway_target, if any. + + It sits beside targetConfiguration, not inside inference.provider. + The first target creation in an account also creates the + AWSServiceRoleForBedrockAgentCoreGatewayNetwork service-linked role. + """ + if not LITELLM_ROUTING_DOMAIN: + return {} + mvr = { + "vpcIdentifier": LITELLM_VPC_ID, + "subnetIds": LITELLM_SUBNET_IDS, + "endpointIpAddressType": "IPV4", + "routingDomain": LITELLM_ROUTING_DOMAIN, + } + if LITELLM_LATTICE_SG_IDS: + mvr["securityGroupIds"] = LITELLM_LATTICE_SG_IDS + return {"privateEndpoint": {"managedVpcResource": mvr}} + + def base_url(self, which: int = 0) -> str: + gid = self.gateways[which] + return ( + f"https://{gid}.gateway.bedrock-agentcore.{self.region}" + f".amazonaws.com/inference" + ) + + # -- destroy ---------------------------------------------------------- + + def destroy(self) -> None: + step("tearing down") + for gid in self.gateways: + try: + for t in self.acc.list_gateway_targets(gatewayIdentifier=gid).get( + "items", [] + ): + self.acc.delete_gateway_target( + gatewayIdentifier=gid, targetId=t["targetId"] + ) + # A target with a private endpoint takes tens of seconds to + # release its Lattice association; the gateway refuses + # deletion until the target is gone. + for _ in range(40): + if not self.acc.list_gateway_targets( + gatewayIdentifier=gid + ).get("items"): + break + time.sleep(5) + self.acc.delete_gateway(gatewayIdentifier=gid) + print(f" deleted gateway {gid}") + except Exception as e: # noqa: BLE001 + print(f" gateway {gid}: {str(e)[:100]}") + for name in (self.caller_role, self.gw_role): + for pol in self.iam.list_role_policies(RoleName=name).get( + "PolicyNames", [] + ): + try: + self.iam.delete_role_policy(RoleName=name, PolicyName=pol) + except Exception: # noqa: BLE001 + pass + try: + self.iam.delete_role(RoleName=name) + print(f" deleted role {name}") + except Exception as e: # noqa: BLE001 + print(f" role {name}: {str(e)[:100]}") + try: + self.ddb.delete_table(TableName=self.table) + print(f" deleted table {self.table}") + except Exception as e: # noqa: BLE001 + print(f" table: {str(e)[:100]}") + + +# ------------------------------------------------------------------ the test + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--region", default=os.environ.get("AWS_REGION", "us-east-1")) + ap.add_argument("--model", default=LITELLM_CHAIN_MODEL) + ap.add_argument("--keep", action="store_true", help="leave AWS resources behind") + args = ap.parse_args() + + if not (LITELLM_ENDPOINT and LITELLM_CREDENTIAL_PROVIDER_ARN): + print("set LITELLM_ENDPOINT (https://...) and LITELLM_CREDENTIAL_PROVIDER_ARN") + return 2 + + if not any( + os.access(os.path.join(p, "claude"), os.X_OK) + for p in os.environ.get("PATH", "").split(os.pathsep) + if p + ): + print("the `claude` CLI is not on PATH — cannot run the chain test") + return 2 + + fx = Fixture(args.region) + fx.create() + + # ---- env must be set before app.config instantiates Settings() ---- + os.environ.update( + PLATFORM_AWS_REGION=args.region, + PLATFORM_DYNAMO_TABLE=fx.table, + PLATFORM_AGENTCORE_GATEWAY_CALLER_ROLE_ARN=fx.caller_arn, + ) + sys.path.insert(0, os.path.join(REPO, "backend")) + sys.path.insert(0, os.path.join(REPO, "runtimes", "agent-sdk-kernel", "src")) + from app.services.llm_credentials_service import LlmCredentialsService # noqa: E402 + import llm_shim # noqa: E402 + + svc = LlmCredentialsService() + + # Observability only, so a chain assertion cannot be satisfied by Claude + # Code quietly using some other credential. Both wrappers delegate to the + # real implementation; nothing about the path under test changes. + seen: list[str] = [] + _orig_proxy = llm_shim._Handler._proxy + + def _counting_proxy(self): # noqa: ANN001, ANN202 + seen.append(f"{self.command} {self.path}") + return _orig_proxy(self) + + # _sign_sigv4 is the one place the full request body is in hand, which is + # what makes the shape of a multi-turn tool loop visible. + sent: list[dict] = [] + _orig_sign = llm_shim._sign_sigv4 + + def _recording_sign(grant, method, url, body): # noqa: ANN001, ANN202 + try: + b = json.loads(body) if body else {} + except Exception: # noqa: BLE001 + b = {} + msgs = [m for m in (b.get("messages") or []) if isinstance(m, dict)] + + def _blocks(m: dict) -> list: + c = m.get("content") + return c if isinstance(c, list) else [] + + sent.append( + { + "path": url.split("amazonaws.com", 1)[-1] or url, + "bytes": len(body or b""), + "model": b.get("model"), + "msgs": len(msgs), + "tools": len(b.get("tools") or []), + "tool_result": sum( + 1 + for m in msgs + for c in _blocks(m) + if isinstance(c, dict) and c.get("type") == "tool_result" + ), + "stream": bool(b.get("stream")), + } + ) + return _orig_sign(grant, method, url, body) + + llm_shim._Handler._proxy = _counting_proxy + llm_shim._sign_sigv4 = _recording_sign + llm_shim.start() + + spec = { + "backend": "agentcore_gateway", + "base_url": fx.base_url(0), + "model": args.model, + "alias_models": {"haiku": args.model}, + } + + try: + step("[chain] mint per-session credentials and run real Claude Code") + sid_a = f"ses-{secrets.token_hex(20)}" + grant = svc.mint_agentcore(sid_a, "alice", spec, team="team-alpha") + check("mint_agentcore returned a grant", bool(grant)) + if not grant: + return 1 + print(f" mode={grant['mode']} region={grant['region']} " + f"service={grant['service']} expires_at={grant['expires_at']}") + print(f" credential is STS: access_key_id={grant['access_key_id'][:8]}…, " + f"session_token={'yes' if grant.get('session_token') else 'no'}") + check( + "grant carries no bearer token and no upstream key", + "token" not in grant and "secret_name" not in grant, + ) + + local_token = llm_shim.register(grant) + check("shim issued a per-invocation local token", bool(local_token)) + check("local token differs from the STS secret", local_token != grant["secret_access_key"]) + + env = dict(os.environ) + # Exactly what the kernel writes for a gateway-mode session. + env.pop("CLAUDE_CODE_USE_BEDROCK", None) + env.pop("CLAUDE_CODE_PROVIDER", None) + env.pop("ANTHROPIC_DEFAULT_OPUS_MODEL", None) + env.update( + ANTHROPIC_BASE_URL=llm_shim.BASE_URL, + ANTHROPIC_AUTH_TOKEN=local_token, + ANTHROPIC_MODEL=args.model, + ANTHROPIC_SMALL_FAST_MODEL=args.model, + CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC="1", + CLAUDE_CONFIG_DIR=os.path.join("/tmp", f"e2e-acgw-cfg-{SUFFIX}"), + ) + before = len(seen) + proc = subprocess.run( + ["claude", "-p", "Reply with the result of 6*7 and nothing else."], + env=env, + capture_output=True, + text=True, + timeout=240, + stdin=subprocess.DEVNULL, + ) + out = (proc.stdout or "") + (proc.stderr or "") + print(f" claude output: {out.strip()[:160]!r}") + check("claude produced no API error", "API Error" not in out, out.strip()[:90]) + check("claude answered", "42" in out) + check( + "the shim forwarded Claude Code's request(s)", + len(seen) > before, + f"{len(seen) - before} request(s)", + ) + print(f" paths Claude Code sent through the chain: {seen[before:]}") + check( + "every path is one AgentCore Gateway accepts inbound", + all( + p.split(" ", 1)[1].split("?")[0] + in ("/v1/messages", "/v1/chat/completions", "/v1/responses", "/v1/models") + for p in seen[before:] + ), + ",".join(sorted({p.split(" ", 1)[1] for p in seen[before:]})), + ) + + step("[task] a real multi-turn task with tool use, not a single turn") + work = os.path.join("/tmp", f"e2e-acgw-work-{SUFFIX}") + os.makedirs(work, exist_ok=True) + before_n, before_sent = len(seen), len(sent) + proc = subprocess.run( + [ + "claude", + "-p", + "Create fizz.py that prints the FizzBuzz sequence for 1 to 15, " + "one value per line. Run it with python3. Then reply with only " + "the 15th line of its output.", + "--allowedTools", + "Write,Read,Bash", + ], + env=env, + cwd=work, + capture_output=True, + text=True, + timeout=600, + stdin=subprocess.DEVNULL, + ) + tout = (proc.stdout or "") + (proc.stderr or "") + print(f" claude output: {tout.strip()[:200]!r}") + rounds = sent[before_sent:] + print(f" {len(seen) - before_n} request(s) through the shim:") + print(f" {'#':>2} {'path':26} {'bytes':>7} {'msgs':>5} {'tools':>5}" + f" {'tool_result':>11} {'stream':>6}") + for i, r in enumerate(rounds, 1): + print( + f" {i:>2} {r['path'][:26]:26} {r['bytes']:>7} {r['msgs']:>5}" + f" {r['tools']:>5} {r['tool_result']:>11} {str(r['stream']):>6}" + ) + check("task produced no API error", "API Error" not in tout, tout.strip()[:90]) + check("task answered correctly", "FizzBuzz" in tout) + check( + "the task took several turns through the chain", + len(rounds) > 1, + f"{len(rounds)} model calls", + ) + check( + "tool results were carried back to the model", + any(r["tool_result"] for r in rounds), + f"max {max((r['tool_result'] for r in rounds), default=0)} per request", + ) + check( + "the agent actually wrote the file", + os.path.exists(os.path.join(work, "fizz.py")), + ) + check( + "context grew across turns (the loop really accumulated)", + len(rounds) > 1 and rounds[-1]["bytes"] > rounds[0]["bytes"], + f"{rounds[0]['bytes']} -> {rounds[-1]['bytes']} bytes" + if len(rounds) > 1 + else "", + ) + paths = sorted({r["path"] for r in rounds}) + print(f" distinct paths used by the task: {paths}") + check( + "no path outside AgentCore Gateway's inbound set", + all( + p.split("?")[0].removeprefix("/inference") + in ("/v1/messages", "/v1/chat/completions", "/v1/responses", "/v1/models") + for p in paths + ), + ",".join(paths), + ) + + if LITELLM_SMOKE_MODELS: + step( + "[multimodel] one /v1/messages 'hi' per non-chain model on the " + "virtual key — sanity-checks LiteLLM's protocol handling for " + "each upstream, no CLI, no tools" + ) + # A fresh grant whose allowed_models covers every model the vk + # permits, so the shim's fail-fast doesn't shadow LiteLLM's + # protocol result. mint_agentcore accepts an explicit models list. + sid_mm = f"ses-{secrets.token_hex(20)}" + grant_mm = svc.mint_agentcore( + sid_mm, "alice", spec, team="team-alpha", + models=LITELLM_ALLOWED_MODELS, + ) + token_mm = llm_shim.register(grant_mm) if grant_mm else "" + # A 200 confirms end-to-end for that model; a non-200 is a data + # point about how Anthropic Messages format survives LiteLLM's + # translation to the upstream — for gpt-* / kimi-* against + # /v1/messages this is not guaranteed to be lossless. Reported + # inline; the [chain] step is what a green run is judged on. + for m in LITELLM_SMOKE_MODELS: + if not token_mm: + print(f" {m}: could not mint a smoke grant") + continue + code, body = shim_call( + token_mm, + { + "model": m, + "max_tokens": 8, + "messages": [{"role": "user", "content": "hi"}], + }, + ) + verdict = "ok" if code == 200 else "fail" + print(f" {verdict:>4} {m:20} HTTP {code} {body[:90]}") + svc.revoke_agentcore(sid_mm) + + step("[failfast] shim refuses a model this session was not routed to") + code, body = shim_call(local_token, {"model": "claude-opus-4-7", "max_tokens": 8, + "messages": [{"role": "user", "content": "hi"}]}) + check("non-permitted model refused locally", code == 403, f"HTTP {code}") + + step("[scoping] the session policy narrows the credential to one gateway") + code_ok, _ = raw_gateway_call(grant, fx.base_url(0), args.model) + code_no, body_no = raw_gateway_call(grant, fx.base_url(1), args.model) + check("credential works on its own gateway", code_ok == 200, f"HTTP {code_ok}") + check( + "same credential refused on a second gateway", + code_no == 403, + f"HTTP {code_no} {body_no[:80]}", + ) + + step("[revoke] revoking one session leaves another untouched") + sid_b = f"ses-{secrets.token_hex(20)}" + grant_b = svc.mint_agentcore(sid_b, "bob", spec, team="team-beta") + check("second session minted", bool(grant_b)) + svc.revoke(sid_a) + print(" revoke(sid_a) issued; polling for effect") + t0 = time.time() + revoked_at = None + while time.time() - t0 < 120: + code, _ = raw_gateway_call(grant, fx.base_url(0), args.model) + if code != 200: + revoked_at = time.time() - t0 + break + time.sleep(3) + check( + "session A's credential stopped working", + revoked_at is not None, + f"after {revoked_at:.1f}s" if revoked_at else "still valid after 120s", + ) + code_b, _ = raw_gateway_call(grant_b, fx.base_url(0), args.model) + check("session B unaffected", code_b == 200, f"HTTP {code_b}") + svc.revoke(sid_b) + finally: + for d in ( + os.path.join("/tmp", f"e2e-acgw-work-{SUFFIX}"), + os.path.join("/tmp", f"e2e-acgw-cfg-{SUFFIX}"), + ): + shutil.rmtree(d, ignore_errors=True) + if not args.keep: + fx.destroy() + else: + print(f"\n--keep: left {fx.gateways}, roles {fx.gw_role}/{fx.caller_role}, " + f"table {fx.table}") + + print() + if FAIL: + print(f"FAILED ({len(FAIL)}): " + "; ".join(FAIL)) + return 1 + print("all assertions passed") + return 0 + + +def shim_call(local_token: str, body: dict) -> tuple[int, str]: + """Call the loopback shim the way the CLI subprocess does.""" + req = urllib.request.Request( + llm_shim_base() + "/v1/messages", + data=json.dumps(body).encode(), + headers={ + "content-type": "application/json", + "authorization": f"Bearer {local_token}", + "anthropic-version": "2023-06-01", + }, + method="POST", + ) + try: + with urllib.request.urlopen(req, timeout=120) as r: + return r.status, r.read().decode()[:200] + except urllib.error.HTTPError as e: + return e.code, e.read().decode()[:200] + + +def llm_shim_base() -> str: + import llm_shim + + return llm_shim.BASE_URL + + +def raw_gateway_call(grant: dict, base_url: str, model: str) -> tuple[int, str]: + """Sign and call the gateway directly, bypassing the shim. + + This is what a session's user could do with the credential they can read + out of the kernel process, which is why the meaningful boundaries (session + policy, revocation) are asserted through this path rather than the shim's. + """ + from botocore.auth import SigV4Auth + from botocore.awsrequest import AWSRequest + from botocore.credentials import Credentials + + data = json.dumps( + {"model": model, "max_tokens": 8, "messages": [{"role": "user", "content": "hi"}]} + ) + req = AWSRequest( + method="POST", + url=base_url.rstrip("/") + "/v1/messages", + data=data, + headers={"content-type": "application/json", "anthropic-version": "2023-06-01"}, + ) + SigV4Auth( + Credentials( + grant["access_key_id"], grant["secret_access_key"], grant.get("session_token") + ), + grant.get("service") or "bedrock-agentcore", + grant["region"], + ).add_auth(req) + r = urllib.request.Request( + req.url, data=data.encode(), headers=dict(req.headers), method="POST" + ) + try: + with urllib.request.urlopen(r, timeout=120) as resp: + return resp.status, resp.read().decode()[:200] + except urllib.error.HTTPError as e: + return e.code, e.read().decode()[:200] + + +if __name__ == "__main__": + sys.exit(main())