From 5d5ef522e28afaf53a851cdf93853f402221e21c Mon Sep 17 00:00:00 2001 From: zzkamzn Date: Fri, 18 Sep 2026 02:53:50 +0000 Subject: [PATCH 1/3] Route model traffic through AgentCore Gateway with per-session credentials MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds an agentcore_gateway model backend alongside bedrock and litellm. It reaches the same goal as llm-edge — no upstream model credential ever lands in a container whose user is root — with one fewer service to run: the provider credential lives in the gateway's token vault, so there is nothing on the platform side to hold either. A session's credential is an STS session tagged with its runtime session id, minted by the backend and delivered in the existing llm_credentials block: - a session policy narrows it to InvokeGateway on the one gateway the backend points at, so a credential read out of kernel memory cannot reach any other gateway in the account; - the session tag is what revoke() conditions a Deny on, which is how one session is stopped without disturbing another. The Deny set is rebuilt from a LLMREVOKE partition and pruned once entries outlive the credential they revoke, keeping the inline policy inside its 10 KB limit; - STS cannot extend a session, so rotate() re-mints from the routing already recorded on the token item, preserving the "config edits apply at next warmup" rule the edge path already has. Claude Code cannot sign SigV4, which is exactly why the signing belongs in the kernel shim rather than the CLI subprocess: both shims now sign per request (botocore in the SDK kernel, ~40 lines of node:crypto in the contract server, which keeps its single dependency). Only a minimal header set is signed and the caller's headers are attached afterwards, so Claude Code's anthropic-* and x-stainless-* headers cannot invalidate a signature; its own x-amz-* headers are dropped rather than folded in. The SigV4 path buffers the request body because it needs the payload hash, and is capped; the edge path still streams it through unread. What this backend deliberately does not claim: IAM conditions cannot see a request body, so the per-session model allowlist is not expressible as an IAM condition. The shim's check is a fail-fast optimisation and says so — the session's user is root in that microVM and can bypass it. Real enforcement needs a gateway request interceptor or one gateway per model; the allowlist is recorded on the token item either way so the decision stays auditable. scripts/e2e_agentcore_gateway.py provisions a live gateway and drives the real Claude Code CLI through the real shim, asserting the chain end to end: a multi-turn tool-using task completes over three model calls, the credential is refused on a second gateway, and revoking one session leaves another working. Terraform for the caller role is not included yet. Co-Authored-By: Claude Opus 5 (1M context) --- backend/app/api/sessions.py | 17 +- backend/app/config.py | 12 + backend/app/models/schemas.py | 8 +- backend/app/services/kernel_service.py | 24 +- .../app/services/llm_credentials_service.py | 351 +++++++- backend/app/services/model_config_service.py | 45 +- runtimes/agent-sdk-kernel/src/llm_shim.py | 95 ++- runtimes/agent-sdk-kernel/src/main.py | 8 +- .../contract-server/main.js | 195 ++++- scripts/e2e_agentcore_gateway.py | 777 ++++++++++++++++++ 10 files changed, 1455 insertions(+), 77 deletions(-) create mode 100755 scripts/e2e_agentcore_gateway.py diff --git a/backend/app/api/sessions.py b/backend/app/api/sessions.py index 9846efb..ad80bdc 100644 --- a/backend/app/api/sessions.py +++ b/backend/app/api/sessions.py @@ -156,11 +156,12 @@ def connect(session_id: str, user: str = Depends(get_current_user)): spec = model_config_service.resolve( item.get("model_backend", ""), item.get("model", "") ) - if spec and spec.get("backend") == "gateway": - # The gateway key stays in llm-edge. The kernel gets an - # endpoint plus a session-scoped token, and the routing fields - # are stripped from what the container sees: the edge re-reads - # them from the grant, so a container has nothing to forge. + if spec and spec.get("backend") in ("gateway", "agentcore_gateway"): + # No upstream credential reaches the container in either mode: + # litellm keeps the key in llm-edge, AgentCore keeps it in the + # gateway's token vault. The kernel gets an endpoint plus a + # session-scoped credential, and the routing fields are + # stripped from what the container sees. creds = llm_credentials_service.mint( item["runtime_session_id"], user, spec ) @@ -168,8 +169,10 @@ def connect(session_id: str, user: str = Depends(get_current_user)): raise HTTPException( status_code=503, detail=( - "gateway model routing is unavailable: the llm-edge " - "service is not deployed (set enable_llm_edge)" + "gateway model routing is unavailable: deploy " + "llm-edge (enable_llm_edge) for the litellm " + "backend, or set the AgentCore gateway caller role " + "for the agentcore_gateway backend" ), ) config["llm_credentials"] = creds diff --git a/backend/app/config.py b/backend/app/config.py index 41ed8d6..99a3a7b 100644 --- a/backend/app/config.py +++ b/backend/app/config.py @@ -31,6 +31,18 @@ class Settings(BaseSettings): # root in. llm_edge_url: str = "" + # Role the backend assumes once per session to mint the kernel's AgentCore + # Gateway credentials, tagged with that session's runtime id. Same shape as + # workspace_access_role_arn: the kernel role itself has no InvokeGateway + # grant, so a container cannot reach the gateway on its own identity. + # + # The session tag is what makes revocation per-session: ending a session + # adds a Deny conditioned on aws:PrincipalTag/session_id, which stops that + # session's credentials without touching any other live session's. + # Empty = the agentcore_gateway model backend is unavailable and the + # backend refuses it rather than falling back to a shared credential. + agentcore_gateway_caller_role_arn: str = "" + # EventBridge Scheduler wiring (outputs of the PortalStack). When all of # group/lambda/role are set, the scheduler runs in "eventbridge" mode: # each platform schedule is mirrored to an EventBridge Scheduler schedule diff --git a/backend/app/models/schemas.py b/backend/app/models/schemas.py index 104b0e1..d0eccd1 100644 --- a/backend/app/models/schemas.py +++ b/backend/app/models/schemas.py @@ -9,7 +9,7 @@ class SessionCreateRequest(BaseModel): mcp_server_ids: list[str] = Field(default_factory=list, max_length=10) skill_ids: list[str] = Field(default_factory=list, max_length=10) # "" = platform default backend; otherwise a backend from the model config - model_backend: str = Field(default="", pattern="^(|bedrock|litellm)$") + model_backend: str = Field(default="", pattern="^(|bedrock|litellm|agentcore_gateway)$") model: str = Field(default="", max_length=200) @@ -118,7 +118,7 @@ class AgentPublishRequest(BaseModel): skill_names: list[str] = Field(default_factory=list, max_length=10) memory_id: str = "" # "" = platform default backend; otherwise a backend from the model config - model_backend: str = Field(default="", pattern="^(|bedrock|litellm)$") + model_backend: str = Field(default="", pattern="^(|bedrock|litellm|agentcore_gateway)$") model: str = Field(default="", max_length=200) @@ -204,12 +204,12 @@ class ModelBackendPatch(BaseModel): class ModelConfigUpdate(BaseModel): - default_backend: str | None = Field(default=None, pattern="^(bedrock|litellm)$") + default_backend: str | None = Field(default=None, pattern="^(bedrock|litellm|agentcore_gateway)$") backends: dict[str, ModelBackendPatch] | None = None class ModelTestRequest(BaseModel): - backend: str = Field(pattern="^(bedrock|litellm)$") + backend: str = Field(pattern="^(bedrock|litellm|agentcore_gateway)$") model: str = Field(default="", max_length=200) diff --git a/backend/app/services/kernel_service.py b/backend/app/services/kernel_service.py index b101f37..5e9763b 100644 --- a/backend/app/services/kernel_service.py +++ b/backend/app/services/kernel_service.py @@ -133,12 +133,17 @@ def invoke_sdk_kernel( payload["memory"] = memory if model: # per-invocation model routing (see model_config_service.resolve) - if model.get("backend") == "gateway": - # The gateway key stays in llm-edge. Mint a grant scoped to this - # invocation's session and strip the routing fields, so the - # kernel receives an endpoint and a token instead of a key it - # could fetch itself. An async run has no refresh channel and - # may execute for hours, so its grant is given matching life. + if model.get("backend") in ("gateway", "agentcore_gateway"): + # No upstream credential reaches the kernel either way: the + # litellm path keeps the key in llm-edge, the AgentCore path + # keeps it in the gateway's token vault. Mint a grant scoped to + # this invocation's session and strip the routing fields, so + # the kernel receives an endpoint plus a session credential + # rather than something it could spend elsewhere. An async run + # has no refresh channel and may execute for hours, so its + # grant is given matching life — for the AgentCore path that + # requires the caller role's MaxSessionDuration to cover it, + # otherwise AssumeRole refuses and this call fails closed. creds = llm_credentials_service.mint( sid, user, @@ -151,9 +156,10 @@ def invoke_sdk_kernel( "result": "", "raw": { "error": ( - "gateway model routing is unavailable: the " - "llm-edge service is not deployed " - "(set enable_llm_edge)" + "gateway model routing is unavailable: deploy " + "llm-edge (enable_llm_edge) for the litellm " + "backend, or set the AgentCore gateway caller " + "role for the agentcore_gateway backend" ) }, } diff --git a/backend/app/services/llm_credentials_service.py b/backend/app/services/llm_credentials_service.py index 77d5b79..2c512fb 100644 --- a/backend/app/services/llm_credentials_service.py +++ b/backend/app/services/llm_credentials_service.py @@ -27,9 +27,12 @@ import hashlib import hmac +import json import logging +import re import secrets import time +import urllib.parse import boto3 @@ -46,6 +49,51 @@ # session silently stops being found. LLM_TOKEN_PK = "LLMTOKEN" +# ---------------------------------------------------------------- AgentCore + +# STS refuses anything shorter, so this is the floor on how stale a revoked +# AgentCore session's credentials can be if the Deny below fails to apply. +AGENTCORE_TTL_S = 900 + +# Sessions whose AgentCore credentials have been revoked but whose STS +# expiry has not yet passed. The role's inline Deny is rebuilt from this +# partition, and entries are dropped once they expire — that pruning is what +# keeps the policy inside the 10 KB per-policy limit. +LLM_REVOKE_PK = "LLMREVOKE" + +# Inline policy on the caller role carrying the per-session Deny statements. +REVOKE_POLICY_NAME = "AgentCoreSessionRevocations" + +_GATEWAY_HOST_RE = re.compile( + r"^(?P[A-Za-z0-9_-]+)\.gateway\.bedrock-agentcore\.(?P[a-z0-9-]+)\.amazonaws\.com$" +) + + +def gateway_arn_from_base_url(base_url: str, account_id: str) -> str: + """Derive the gateway ARN from the endpoint the model config carries. + + The session policy below has to name the gateway as a resource, and asking + an operator to keep an ARN and a URL in sync is a good way to get a + mismatch. The URL already contains both the gateway id and the region, so + the ARN is derivable; an unparseable host yields "" and the caller falls + back to no session policy (the role's own policy still applies). + """ + host = urllib.parse.urlsplit(str(base_url or "")).hostname or "" + m = _GATEWAY_HOST_RE.match(host) + if not m or not account_id: + return "" + return ( + f"arn:aws:bedrock-agentcore:{m.group('region')}:{account_id}" + f":gateway/{m.group('id')}" + ) + + +def _region_from_gateway_arn(gateway_arn: str) -> str: + """SigV4 has to be signed for the gateway's own region, which is not + necessarily the region this backend runs in.""" + parts = str(gateway_arn or "").split(":") + return parts[3] if len(parts) > 4 else "" + def _sha256(value: str) -> str: return hashlib.sha256(value.encode("utf-8")).hexdigest() @@ -72,11 +120,24 @@ class LlmCredentialsService: def __init__(self) -> None: dynamodb = boto3.resource("dynamodb", region_name=settings.aws_region) self.table = dynamodb.Table(settings.dynamo_table) + self.sts = boto3.client("sts", region_name=settings.aws_region) + self.iam = boto3.client("iam") + self._account_id = "" @property def enabled(self) -> bool: return bool(settings.llm_edge_url) + @property + def agentcore_enabled(self) -> bool: + return bool(settings.agentcore_gateway_caller_role_arn) + + @property + def account_id(self) -> str: + if not self._account_id: + self._account_id = self.sts.get_caller_identity()["Account"] + return self._account_id + def mint( self, runtime_session_id: str, @@ -99,6 +160,10 @@ def mint( never reaches the agent's own subprocess environment either way — the kernel keeps it and hands the CLI a container-local token instead. """ + if str(spec.get("backend") or "") == "agentcore_gateway": + return self.mint_agentcore( + runtime_session_id, user, spec, team=team, ttl_s=ttl_s + ) if not self.enabled: return None base_url = str(spec.get("base_url") or "") @@ -140,6 +205,256 @@ def mint( "expires_at": expires_at, } + # ------------------------------------------------------------ AgentCore + + def mint_agentcore( + self, + runtime_session_id: str, + user: str, + spec: dict, + team: str = "", + ttl_s: int = AGENTCORE_TTL_S, + models: list[str] | None = None, + ) -> dict | None: + """Mint SigV4 credentials for one session's AgentCore Gateway access. + + Same shape of promise as :meth:`mint`, reached differently. There is no + bearer token to leak here and no platform-wide secret anywhere on the + path: the gateway holds the upstream provider credential in its token + vault, and the container authenticates as *this session* using STS + credentials tagged with its runtime session id. + + Two things are bound into the credential itself, so neither depends on + the container behaving: + + - a **session policy** narrowing the grant to `InvokeGateway` on the one + gateway this backend points at, so a leaked credential cannot reach + any other gateway in the account; + - a **session tag** (``session_id``), which is what + :meth:`revoke_agentcore` conditions its Deny on. That is how one + session is revoked without disturbing any other live session. + + What is *not* enforced here is the per-session model allowlist: IAM + conditions cannot see a request body, so which model a session may ask + for is not expressible as an IAM condition. The kernel does a + fail-fast local check (an optimisation, not a boundary — the session's + user is root in that microVM) and real enforcement needs either a + gateway request interceptor or one gateway per model. The allowlist is + recorded on the token item either way so the decision stays auditable. + """ + if not self.agentcore_enabled: + logger.error( + "refusing to mint AgentCore gateway credentials for %s: " + "PLATFORM_AGENTCORE_GATEWAY_CALLER_ROLE_ARN is not set", + runtime_session_id, + ) + return None + base_url = str(spec.get("base_url") or "").rstrip("/") + # ``models`` is passed only by :meth:`rotate`, which re-mints from the + # allowlist already recorded on the token item rather than re-deriving + # it from a spec it deliberately does not re-resolve. + models = list(models) if models else permitted_models(spec) + if not base_url or not models: + logger.error( + "refusing to mint AgentCore gateway credentials for %s: " + "incomplete spec", + runtime_session_id, + ) + return None + + gateway_arn = gateway_arn_from_base_url(base_url, self.account_id) + kwargs: dict = { + "RoleArn": settings.agentcore_gateway_caller_role_arn, + # The runtime session id is already caller-bound upstream (see + # session_binding), so it is safe to use verbatim as the session + # name — and doing so makes CloudTrail read as the session. + "RoleSessionName": runtime_session_id[:64], + "DurationSeconds": max(AGENTCORE_TTL_S, int(ttl_s)), + "Tags": [{"Key": "session_id", "Value": runtime_session_id}], + # Nothing downstream should be able to widen the grant by + # re-assuming with a different tag. + "TransitiveTagKeys": ["session_id"], + } + if gateway_arn: + kwargs["Policy"] = json.dumps( + { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Action": "bedrock-agentcore:InvokeGateway", + "Resource": gateway_arn, + } + ], + } + ) + else: + logger.warning( + "could not derive a gateway ARN from %r — minting without a " + "session policy", + base_url, + ) + try: + creds = self.sts.assume_role(**kwargs)["Credentials"] + except Exception as e: # noqa: BLE001 - any STS failure means refuse + logger.error( + "AssumeRole for AgentCore gateway session %s failed: %s", + runtime_session_id, + e, + ) + return None + + expires_at = int(creds["Expiration"].timestamp()) + try: + self.table.put_item( + Item={ + "PK": LLM_TOKEN_PK, + "SK": f"RSID#{runtime_session_id}", + "mode": "agentcore_gateway", + # No token digest: the credential is an STS session, not a + # bearer value this table could be used to replay. + "expires_at": expires_at, + "runtime_session_id": runtime_session_id, + "user": user, + "team": team, + "upstream_base_url": base_url, + "gateway_arn": gateway_arn, + "allowed_models": models, + }, + # A grant belongs to the session's owner: create, or re-mint the + # owner's own — never overwrite a live grant held by a different + # principal, since doing so would stop the victim's kernel from + # matching and is a targeted denial of service. Session ids are + # caller-bound upstream, so a collision should be unreachable; + # this makes it impossible even if that breaks. + ConditionExpression="attribute_not_exists(PK) OR #u = :user", + ExpressionAttributeNames={"#u": "user"}, + ExpressionAttributeValues={":user": user}, + ) + except self.table.meta.client.exceptions.ConditionalCheckFailedException: + logger.error( + "refusing to mint AgentCore gateway credentials for %s: " + "session grant is held by a different principal", + runtime_session_id, + ) + return None + + return { + "mode": "agentcore_gateway", + "endpoint": base_url, + "session_id": runtime_session_id, + "region": _region_from_gateway_arn(gateway_arn) or settings.aws_region, + # The gateway's data plane signs as bedrock-agentcore, not as the + # control-plane service the SDK would guess from the hostname. + "service": "bedrock-agentcore", + "access_key_id": creds["AccessKeyId"], + "secret_access_key": creds["SecretAccessKey"], + "session_token": creds["SessionToken"], + "expires_at": expires_at, + # Carried for the kernel's fail-fast check only. The kernel must + # not treat this as authorization — see the docstring. + "allowed_models": models, + } + + def revoke_agentcore(self, runtime_session_id: str) -> None: + """Stop one session's gateway credentials from working, now. + + STS sessions cannot be withdrawn, so revocation is expressed as a Deny + on the caller role conditioned on the session's tag. The Deny set is + rebuilt from the LLMREVOKE partition on every call and entries whose + STS expiry has passed are dropped, which is what keeps the inline + policy inside its 10 KB limit: an entry only has to outlive the + credential it revokes. + """ + if not self.agentcore_enabled or not runtime_session_id: + return + item = self.table.get_item( + Key={"PK": LLM_TOKEN_PK, "SK": f"RSID#{runtime_session_id}"} + ).get("Item") + expires_at = int((item or {}).get("expires_at") or 0) + if not expires_at: + # Nothing live to revoke; still fall through so the rebuild prunes. + expires_at = int(time.time()) + AGENTCORE_TTL_S + try: + self.table.put_item( + Item={ + "PK": LLM_REVOKE_PK, + "SK": f"RSID#{runtime_session_id}", + "expires_at": expires_at, + } + ) + except Exception: # noqa: BLE001 + logger.warning("could not record revocation for %s", runtime_session_id) + self._rebuild_agentcore_denies() + + def _rebuild_agentcore_denies(self) -> None: + now = int(time.time()) + live: list[str] = [] + stale: list[str] = [] + try: + resp = self.table.query( + KeyConditionExpression="PK = :pk AND begins_with(SK, :p)", + ExpressionAttributeValues={":pk": LLM_REVOKE_PK, ":p": "RSID#"}, + ) + except Exception as e: # noqa: BLE001 + logger.error("could not read revocation list: %s", e) + return + for it in resp.get("Items", []): + sid = str(it.get("SK", ""))[len("RSID#") :] + if not sid: + continue + # A small grace period past expiry: clock skew between here and + # STS must not un-revoke a session a second early. + if int(it.get("expires_at") or 0) + 60 > now: + live.append(sid) + else: + stale.append(sid) + + role_name = settings.agentcore_gateway_caller_role_arn.rsplit("/", 1)[-1] + try: + if live: + self.iam.put_role_policy( + RoleName=role_name, + PolicyName=REVOKE_POLICY_NAME, + PolicyDocument=json.dumps( + { + "Version": "2012-10-17", + "Statement": [ + { + "Sid": "RevokedSessions", + "Effect": "Deny", + "Action": "bedrock-agentcore:*", + "Resource": "*", + "Condition": { + "StringEquals": { + "aws:PrincipalTag/session_id": sorted(live) + } + }, + } + ], + } + ), + ) + else: + # Nothing outstanding — drop the policy rather than leave an + # empty statement behind. + try: + self.iam.delete_role_policy( + RoleName=role_name, PolicyName=REVOKE_POLICY_NAME + ) + except self.iam.exceptions.NoSuchEntityException: + pass + except Exception as e: # noqa: BLE001 + logger.error("could not update AgentCore revocation policy: %s", e) + return + for sid in stale: + try: + self.table.delete_item( + Key={"PK": LLM_REVOKE_PK, "SK": f"RSID#{sid}"} + ) + except Exception: # noqa: BLE001 + pass + def rotate(self, runtime_session_id: str) -> dict | None: """Issue a fresh token for an existing session grant. @@ -147,7 +462,26 @@ def rotate(self, runtime_session_id: str) -> dict | None: on the session's next warmup, same as every other model-routing change on this platform. Refresh is only about keeping a live session alive. """ - if not self.enabled or not runtime_session_id: + if not runtime_session_id: + return None + existing = self.table.get_item( + Key={"PK": LLM_TOKEN_PK, "SK": f"RSID#{runtime_session_id}"} + ).get("Item") + if existing and existing.get("mode") == "agentcore_gateway": + # STS credentials cannot be extended, so refresh is a re-mint from + # the routing already recorded on the item — which keeps the same + # "config edits apply at next warmup" rule as the edge path. + return self.mint_agentcore( + runtime_session_id, + str(existing.get("user") or ""), + { + "backend": "agentcore_gateway", + "base_url": str(existing.get("upstream_base_url") or ""), + }, + team=str(existing.get("team") or ""), + models=[str(m) for m in (existing.get("allowed_models") or [])], + ) + if not self.enabled: return None token = secrets.token_urlsafe(32) expires_at = int(time.time()) + TOKEN_TTL_S @@ -174,11 +508,22 @@ def rotate(self, runtime_session_id: str) -> dict | None: } def revoke(self, runtime_session_id: str) -> None: - """Drop a session's grant. Called when a session ends so a token + """Drop a session's grant. Called when a session ends so a credential scraped out of a container's memory stops working immediately rather - than at the end of its hour.""" + than at the end of its lifetime. + + Both modes are handled. The edge mode is a single delete — the edge + re-reads this item on every call, so the grant dies with the row. The + AgentCore mode cannot delete an STS session, so it records a Deny + keyed on the session's tag *before* dropping the row (the Deny needs + the row's expiry to know when it may be pruned).""" if not runtime_session_id: return + item = self.table.get_item( + Key={"PK": LLM_TOKEN_PK, "SK": f"RSID#{runtime_session_id}"} + ).get("Item") + if item and item.get("mode") == "agentcore_gateway": + self.revoke_agentcore(runtime_session_id) try: self.table.delete_item( Key={"PK": LLM_TOKEN_PK, "SK": f"RSID#{runtime_session_id}"} diff --git a/backend/app/services/model_config_service.py b/backend/app/services/model_config_service.py index 0fb11ee..84dc10a 100644 --- a/backend/app/services/model_config_service.py +++ b/backend/app/services/model_config_service.py @@ -25,7 +25,7 @@ PK = "GOV" SK = "MODELCONFIG" -BACKEND_NAMES = ("bedrock", "litellm") +BACKEND_NAMES = ("bedrock", "litellm", "agentcore_gateway") DEFAULT_CONFIG: dict = { "default_backend": "bedrock", @@ -44,6 +44,18 @@ "default_model": "", "small_fast_model": "", }, + # AgentCore Gateway inference target. Unlike litellm there is no + # secret_name: the upstream provider credential lives in the gateway's + # token vault, so nothing on the platform side ever holds it. base_url + # is the gateway's /inference base; the kernel signs SigV4 against it + # with per-session credentials rather than presenting a bearer token. + "agentcore_gateway": { + "enabled": False, + "base_url": "", + "models": [], + "default_model": "", + "small_fast_model": "", + }, }, } @@ -124,22 +136,35 @@ def resolve(self, backend: str = "", model: str = "") -> dict | None: spec["small_fast_model"] = b["small_fast_model"] return spec - # litellm → the kernel's generic Anthropic-compatible gateway mode + # Both remaining backends mean "point the kernel at an HTTP endpoint + # rather than Bedrock", and they share these two failure modes. if not b["base_url"]: - raise ValueError("model backend 'litellm' has no base_url configured") + raise ValueError(f"model backend {name!r} has no base_url configured") if not chosen: # without an explicit name the container's baked-in Bedrock model # ID would leak into gateway requests — refuse instead raise ValueError( - "model backend 'litellm' needs a model (set the backend's " + f"model backend {name!r} needs a model (set the backend's " "default_model or the agent's model)" ) - spec = { - "backend": "gateway", - "base_url": b["base_url"], - "secret_name": b["secret_name"] or "agent-platform/llm-gateway-key", - "model": chosen, - } + if name == "agentcore_gateway": + # SigV4 mode. There is no platform-side secret to name: the + # upstream provider credential lives in the gateway's token vault, + # and the kernel authenticates as its own session instead of + # presenting a shared bearer token. + spec = { + "backend": "agentcore_gateway", + "base_url": b["base_url"], + "model": chosen, + } + else: + # litellm → the kernel's generic Anthropic-compatible gateway mode + spec = { + "backend": "gateway", + "base_url": b["base_url"], + "secret_name": b["secret_name"] or "agent-platform/llm-gateway-key", + "model": chosen, + } if b["small_fast_model"]: spec["small_fast_model"] = b["small_fast_model"] # Claude Code's /model picker offers opus/sonnet/haiku aliases that diff --git a/runtimes/agent-sdk-kernel/src/llm_shim.py b/runtimes/agent-sdk-kernel/src/llm_shim.py index 8cd4d59..d8db6ef 100644 --- a/runtimes/agent-sdk-kernel/src/llm_shim.py +++ b/runtimes/agent-sdk-kernel/src/llm_shim.py @@ -16,6 +16,14 @@ Per-invocation registration is also what makes concurrent runs safe: two agents routed to different backends each get their own local token, so neither can borrow the other's grant. + +Two upstream shapes are supported, decided by the grant the platform delivers: + +- ``llm-edge``: a bearer token plus the session id, which the edge re-reads its + own grant from. +- ``agentcore_gateway``: STS credentials tagged with the session id, which this + shim SigV4-signs each request with. Claude Code cannot sign, which is exactly + why the signing has to happen here rather than in the CLI subprocess. """ from __future__ import annotations @@ -62,12 +70,53 @@ "content-length", } +# On the SigV4 path a client-supplied x-amz-* header would either be folded +# into the signature or contradict it, so the client never gets to set one. +_SIGV4_RESERVED_PREFIX = "x-amz-" + + +def _sign_sigv4(grant: dict, method: str, url: str, body: bytes) -> dict: + """Return the SigV4 headers for one request against the gateway. + + Only a minimal, deterministic header set is signed (content-type, plus the + host and date botocore adds). Everything the caller sent is attached + *after* signing: headers outside SignedHeaders do not participate in the + signature, so passing Claude Code's ``anthropic-*`` and ``x-stainless-*`` + headers through cannot invalidate it. + """ + from botocore.auth import SigV4Auth + from botocore.awsrequest import AWSRequest + from botocore.credentials import Credentials + + creds = Credentials( + grant["access_key_id"], + grant["secret_access_key"], + grant.get("session_token") or None, + ) + req = AWSRequest( + method=method, + url=url, + data=body, + headers={"content-type": "application/json"}, + ) + SigV4Auth( + creds, + str(grant.get("service") or "bedrock-agentcore"), + str(grant.get("region") or ""), + ).add_auth(req) + return dict(req.headers) + def register(grant: dict) -> str: """Register a platform grant for one invocation; returns the local token to put in the CLI subprocess's ANTHROPIC_AUTH_TOKEN.""" - if not grant or not grant.get("endpoint") or not grant.get("token"): - raise ValueError("gateway grant is missing endpoint/token") + if not grant or not grant.get("endpoint"): + raise ValueError("gateway grant is missing endpoint") + if not grant.get("token") and not grant.get("access_key_id"): + raise ValueError( + "gateway grant carries neither a bearer token (llm-edge) nor STS " + "credentials (agentcore_gateway)" + ) local = secrets.token_urlsafe(24) with _lock: _grants[local] = grant @@ -122,13 +171,49 @@ def _proxy(self) -> None: length = int(self.headers.get("Content-Length") or 0) body = self.rfile.read(length) if length else b"" - headers = { + sigv4 = bool(grant.get("access_key_id")) + passthrough = { k: v for k, v in self.headers.items() if k.lower() not in _DROP_REQUEST_HEADERS + and not (sigv4 and k.lower().startswith(_SIGV4_RESERVED_PREFIX)) } - headers["Authorization"] = f"Bearer {grant['token']}" - headers["x-platform-session-id"] = str(grant.get("session_id") or "") + + if sigv4: + # Fail fast on a model this session was not routed to. This is an + # optimisation, not an authorization boundary: the session's user + # is root in this microVM and can bypass the shim. Real enforcement + # has to sit outside the container — see the platform's + # llm_credentials_service.mint_agentcore docstring. + allowed = [str(m) for m in (grant.get("allowed_models") or [])] + if allowed and body: + try: + requested = str(json.loads(body).get("model") or "") + except Exception: # noqa: BLE001 - non-JSON bodies just pass + requested = "" + if requested and requested not in allowed: + self._fail( + 403, f"model {requested!r} is not permitted for this session" + ) + return + headers = _sign_sigv4( + grant, + self.command, + str(grant["endpoint"]).rstrip("/") + self.path, + body, + ) + # Anything that took part in the signature must keep the signed + # value: the caller's own content-type would otherwise silently + # invalidate it. + signed = {k.lower() for k in headers} + headers.update( + {k: v for k, v in passthrough.items() if k.lower() not in signed} + ) + else: + headers = passthrough + headers["Authorization"] = f"Bearer {grant['token']}" + headers["x-platform-session-id"] = str(grant.get("session_id") or "") + headers["Accept-Encoding"] = "identity" headers["Content-Length"] = str(len(body)) diff --git a/runtimes/agent-sdk-kernel/src/main.py b/runtimes/agent-sdk-kernel/src/main.py index 443c020..f33a107 100644 --- a/runtimes/agent-sdk-kernel/src/main.py +++ b/runtimes/agent-sdk-kernel/src/main.py @@ -264,14 +264,18 @@ def build_model_env( if small: env["ANTHROPIC_SMALL_FAST_MODEL"] = small return env, model, "" - if backend == "gateway": + if backend in ("gateway", "agentcore_gateway"): # No key is fetched and the upstream address is not even in the spec — # the backend strips base_url/secret_name before sending it here. The # CLI talks to the loopback shim with a token that means nothing outside # this container and nothing after this invocation. + # + # Which credential the shim then presents upstream is the grant's + # business: a bearer token for llm-edge, or SigV4 over per-session STS + # credentials for an AgentCore Gateway. The CLI sees neither. if not grant: raise ValueError( - "model.backend=gateway requires llm_credentials in the payload" + f"model.backend={backend} requires llm_credentials in the payload" ) local_token = llm_shim.register(grant) env = { diff --git a/runtimes/claude-code-kernel/contract-server/main.js b/runtimes/claude-code-kernel/contract-server/main.js index 1deb675..b0d9b30 100644 --- a/runtimes/claude-code-kernel/contract-server/main.js +++ b/runtimes/claude-code-kernel/contract-server/main.js @@ -14,6 +14,7 @@ */ const http = require("http"); const https = require("https"); +const crypto = require("crypto"); const fs = require("fs"); const url = require("url"); const { spawn, execFile } = require("child_process"); @@ -124,13 +125,78 @@ setInterval(refreshWorkspaceCredentials, 5 * 60_000); // models this session may use on every call. // --------------------------------------------------------------------------- const LLM_SHIM_PORT = 8787; +// Only the SigV4 path buffers a request body (it needs the payload hash), so +// cap it rather than let a malformed or hostile body exhaust the kernel. +const SHIM_MAX_BODY_BYTES = 32 * 1024 * 1024; // {endpoint, token, expires_at} — set from the warmup payload and rotated by // the same refresh call that renews workspace credentials. let llmGrant = null; +// --------------------------------------------------------------------------- +// SigV4, by hand. +// +// The agentcore_gateway grant is a set of STS credentials rather than a bearer +// token, so each request has to be signed. Claude Code cannot sign — which is +// the whole reason the signing belongs here and not in the CLI subprocess. +// +// Written against node:crypto rather than pulling in an AWS SDK: this contract +// server has exactly one dependency today, and a signing routine is 40 lines. +// --------------------------------------------------------------------------- +const sha256Hex = (b) => crypto.createHash("sha256").update(b).digest("hex"); +const hmac = (key, msg) => crypto.createHmac("sha256", key).update(msg).digest(); + +function sigv4Headers(grant, method, target, body) { + const now = new Date(); + const amzDate = now.toISOString().replace(/[:-]|\.\d{3}/g, ""); + const dateStamp = amzDate.slice(0, 8); + const region = String(grant.region || ""); + const service = String(grant.service || "bedrock-agentcore"); + + // Only a minimal, deterministic set is signed. Everything the caller sent is + // attached after signing, where it cannot affect the signature. + const signed = { + host: target.host, + "content-type": "application/json", + "x-amz-date": amzDate, + }; + if (grant.session_token) signed["x-amz-security-token"] = grant.session_token; + + const names = Object.keys(signed).sort(); + const canonicalHeaders = names.map((n) => `${n}:${String(signed[n]).trim()}\n`).join(""); + const signedHeaders = names.join(";"); + const canonicalRequest = [ + method, + target.pathname, + target.searchParams ? target.searchParams.toString() : "", + canonicalHeaders, + signedHeaders, + sha256Hex(body), + ].join("\n"); + + const scope = `${dateStamp}/${region}/${service}/aws4_request`; + const stringToSign = [ + "AWS4-HMAC-SHA256", + amzDate, + scope, + sha256Hex(Buffer.from(canonicalRequest, "utf8")), + ].join("\n"); + + const kDate = hmac(`AWS4${grant.secret_access_key}`, dateStamp); + const kSigning = hmac(hmac(hmac(kDate, region), service), "aws4_request"); + const signature = crypto.createHmac("sha256", kSigning).update(stringToSign).digest("hex"); + + return { + ...signed, + authorization: + `AWS4-HMAC-SHA256 Credential=${grant.access_key_id}/${scope}, ` + + `SignedHeaders=${signedHeaders}, Signature=${signature}`, + }; +} + function setLlmGrant(grant) { - if (!grant || !grant.endpoint || !grant.token) return; + if (!grant || !grant.endpoint) return; + if (!grant.token && !grant.access_key_id) return; llmGrant = grant; console.log( `[model] gateway grant active via ${grant.endpoint}` + @@ -180,53 +246,108 @@ function startLlmShim() { return; } - const headers = {}; + const sigv4 = Boolean(llmGrant.access_key_id); + const passthrough = {}; for (const [k, v] of Object.entries(req.headers)) { - if (!SHIM_DROP_HEADERS.has(k.toLowerCase())) headers[k] = v; + const lk = k.toLowerCase(); + if (SHIM_DROP_HEADERS.has(lk)) continue; + // On the SigV4 path a caller-supplied x-amz-* header would either be + // folded into the signature or contradict it. + if (sigv4 && lk.startsWith("x-amz-")) continue; + passthrough[k] = v; } - headers["authorization"] = `Bearer ${llmGrant.token}`; - // The grant names the session it was issued for; fall back to the ID - // AgentCore injected in case an older backend omits it. - headers["x-platform-session-id"] = llmGrant.session_id || runtimeSessionId || ""; - headers["accept-encoding"] = "identity"; - - const client = target.protocol === "https:" ? https : http; - const upstream = client.request( - target, - { method: req.method, headers, timeout: 15 * 60_000 }, - (up) => { - const out = {}; - for (const [k, v] of Object.entries(up.headers)) { - if (!["connection", "keep-alive", "transfer-encoding"].includes(k.toLowerCase())) { - out[k] = v; + + // body === null means "stream it through unread", which is what the edge + // path does: the shim has no reason to inspect it and the edge authorizes + // the requested model. SigV4 needs the payload hash, so that path buffers. + const forward = (body) => { + let headers; + if (sigv4) { + headers = sigv4Headers(llmGrant, req.method, target, body); + // Anything that took part in the signature keeps its signed value. + const signed = new Set(Object.keys(headers)); + for (const [k, v] of Object.entries(passthrough)) { + if (!signed.has(k.toLowerCase())) headers[k] = v; + } + } else { + headers = { ...passthrough }; + headers["authorization"] = `Bearer ${llmGrant.token}`; + // The grant names the session it was issued for; fall back to the ID + // AgentCore injected in case an older backend omits it. + headers["x-platform-session-id"] = llmGrant.session_id || runtimeSessionId || ""; + } + headers["accept-encoding"] = "identity"; + if (body !== null) headers["content-length"] = String(body.length); + + const client = target.protocol === "https:" ? https : http; + const upstream = client.request( + target, + { method: req.method, headers, timeout: 15 * 60_000 }, + (up) => { + const out = {}; + for (const [k, v] of Object.entries(up.headers)) { + if (!["connection", "keep-alive", "transfer-encoding"].includes(k.toLowerCase())) { + out[k] = v; + } } + res.writeHead(up.statusCode || 502, out); + // Straight pipe: nothing here may buffer, or token-by-token output + // would arrive as one blob at the end of the response. + up.pipe(res); + }, + ); + + upstream.on("timeout", () => upstream.destroy(new Error("edge timeout"))); + upstream.on("error", (e) => { + console.log(`[model] upstream request failed: ${e.message}`); + if (!res.headersSent) { + res.writeHead(502, { "Content-Type": "application/json" }); + res.end( + JSON.stringify({ + type: "error", + error: { type: "upstream_error", message: e.message }, + }), + ); + } else { + res.destroy(); } - res.writeHead(up.statusCode || 502, out); - // Straight pipe: nothing here may buffer, or token-by-token output - // would arrive as one blob at the end of the response. - up.pipe(res); - }, - ); + }); + + if (body === null) req.pipe(upstream); + else upstream.end(body); + }; + + if (!sigv4) { + forward(null); + return; + } - upstream.on("timeout", () => upstream.destroy(new Error("edge timeout"))); - upstream.on("error", (e) => { - console.log(`[model] edge request failed: ${e.message}`); - if (!res.headersSent) { - res.writeHead(502, { "Content-Type": "application/json" }); + const chunks = []; + let total = 0; + let aborted = false; + req.on("data", (c) => { + if (aborted) return; + total += c.length; + if (total > SHIM_MAX_BODY_BYTES) { + aborted = true; + res.writeHead(413, { "Content-Type": "application/json" }); res.end( JSON.stringify({ type: "error", - error: { type: "upstream_error", message: e.message }, + error: { type: "request_too_large", message: "request body too large" }, }), ); - } else { - res.destroy(); + req.destroy(); + return; } + chunks.push(c); + }); + req.on("end", () => { + if (!aborted) forward(Buffer.concat(chunks)); + }); + req.on("error", () => { + aborted = true; }); - - // The request body is streamed through unread — the shim has no reason to - // inspect it, and the edge is what authorizes the requested model. - req.pipe(upstream); }); server.requestTimeout = 0; @@ -282,7 +403,7 @@ function applyModelSpec(spec) { if (spec.small_fast_model) lines.push(`export ANTHROPIC_SMALL_FAST_MODEL=${shq(spec.small_fast_model)}`); // keep the container's baked-in (Bedrock) alias steering - } else if (backend === "gateway") { + } else if (backend === "gateway" || backend === "agentcore_gateway") { // No credential is written here, and the upstream gateway's address is not // even known to this container: the spec arrives with base_url and // secret_name stripped. Claude Code is pointed at the loopback shim, which diff --git a/scripts/e2e_agentcore_gateway.py b/scripts/e2e_agentcore_gateway.py new file mode 100755 index 0000000..ad3415a --- /dev/null +++ b/scripts/e2e_agentcore_gateway.py @@ -0,0 +1,777 @@ +#!/usr/bin/env python3 +"""End-to-end test for the agentcore_gateway model backend, against real AWS. + +Unlike the moto-based suites this one provisions a live AgentCore Gateway and +drives the *real* Claude Code CLI through the *real* kernel shim, so the whole +authentication chain is exercised as deployed: + + Claude Code subprocess per-invocation loopback token + -> llm_shim (kernel) SigV4 over per-session STS credentials + -> AgentCore Gateway IAM authorizer + session policy + -> bedrock-mantle gateway's own execution role + -> Anthropic model + +Nothing in the chain is stubbed. The only simulation is the microVM itself: +the shim runs in this process instead of inside AgentCore Runtime, which is +exactly where it runs in production (in the kernel process), so the code path +is identical. + +Asserts: + + [chain] Claude Code completes a turn through the shim, and the shim + observed the request — i.e. the CLI really used the gateway and + not some ambient credential. + [protocol] which paths Claude Code actually sends (reported, and asserted to + be a subset of what AgentCore Gateway accepts inbound). + [scoping] the session policy narrows the credential to one gateway: the + same credential is refused on a second gateway. + [failfast] the shim refuses a model this session was not routed to (an + optimisation, not a boundary — see mint_agentcore's docstring). + [revoke] llm_credentials_service.revoke() stops this session's + credentials while a second live session keeps working. + +Usage: python3 scripts/e2e_agentcore_gateway.py [--region us-east-1] [--keep] + +Requires: credentials able to create IAM roles, an AgentCore Gateway and a +DynamoDB table, plus the `claude` CLI on PATH. Everything it creates is torn +down unless --keep is passed. +""" +import argparse +import io +import json +import os +import secrets +import shutil +import subprocess +import sys +import time +import urllib.error +import urllib.request +import zipfile + +import boto3 +from botocore.exceptions import ClientError + +REPO = os.path.dirname(os.path.dirname(os.path.abspath(__file__))) +FAIL: list[str] = [] +SUFFIX = secrets.token_hex(4) + +# A gateway REQUEST interceptor. V4 needs one anyway — that is where the +# tenant identity used for LiteLLM's per-end-user billing gets injected from a +# verified JWT claim, which the client cannot forge. Body adaptation rides +# along in the same function: Claude Code sends first-party-only fields that +# the bedrock-mantle passthrough refuses outright ("400 context_management: +# Extra inputs are not permitted"), and the gateway forwards bodies verbatim, +# so there is nowhere else outside the container to drop them. A LiteLLM +# upstream configured with drop_params absorbs the same class of mismatch. +INTERCEPTOR_SRC = ''' +import base64, json + +# Fields Claude Code sends that the strict Anthropic-on-Bedrock passthrough +# rejects. Dropping them is lossy (prompt caching, server-side context +# management) and deliberately explicit rather than a blanket filter. +DROP_TOP_LEVEL = ("context_management",) + +def lambda_handler(event, context): + req = event["http"]["gatewayRequest"] + raw = base64.b64decode(req.get("body") or b"") + try: + body = json.loads(raw) if raw else {} + except Exception: + return {"interceptorOutputVersion": "1.0", + "http": {"transformedGatewayRequest": req}} + dropped = [k for k in DROP_TOP_LEVEL if k in body] + for k in dropped: + body.pop(k, None) + if dropped: + print("dropped unsupported fields: " + ",".join(dropped)) + new = json.dumps(body).encode() + # Only the fields being changed go back. Echoing the whole gatewayRequest + # (httpMethod/path included) is rejected with "Received invalid response + # from interceptor", and the body must stay base64 — a JSON object is + # rejected the same way, even though the MCP examples in the docs use one. + hdrs = dict(req.get("headers") or {}) + hdrs["Content-Length"] = str(len(new)) + return { + "interceptorOutputVersion": "1.0", + "http": { + "transformedGatewayRequest": { + "headers": hdrs, + "body": base64.b64encode(new).decode(), + } + }, + } +''' + + +def check(label: str, ok: bool, detail: str = "") -> None: + print(f" {'ok ' if ok else 'FAIL'} {label}" + (f" ({detail})" if detail else "")) + if not ok: + FAIL.append(label) + + +def step(msg: str) -> None: + print(f"\n=== {msg}") + + +# --------------------------------------------------------------- provisioning + + +class Fixture: + """Live AWS resources for one run.""" + + def __init__(self, region: str) -> None: + self.region = region + self.iam = boto3.client("iam") + self.sts = boto3.client("sts") + self.ddb = boto3.client("dynamodb", region_name=region) + self.acc = boto3.client("bedrock-agentcore-control", region_name=region) + self.lam = boto3.client("lambda", region_name=region) + self.logs = boto3.client("logs", region_name=region) + self.account = self.sts.get_caller_identity()["Account"] + self.gw_role = f"e2e-acgw-exec-{SUFFIX}" + self.caller_role = f"e2e-acgw-caller-{SUFFIX}" + self.icept_role = f"e2e-acgw-icept-{SUFFIX}" + self.icept_fn = f"e2e-acgw-icept-{SUFFIX}" + self.table = f"e2e-acgw-{SUFFIX}" + self.gateways: list[str] = [] + self.icept_arn = "" + + # -- helpers ---------------------------------------------------------- + + def _caller_principal(self) -> str: + """The role this script runs as, so the caller role can trust it. + + In production the trusting principal is the backend's IRSA role; here + it is whatever identity is running the test. + """ + arn = self.sts.get_caller_identity()["Arn"] + if ":assumed-role/" in arn: + name = arn.split(":assumed-role/")[1].split("/")[0] + return f"arn:aws:iam::{self.account}:role/{name}" + return arn + + def _role(self, name: str, trust: dict, policy: dict, max_session: int = 3600) -> str: + try: + arn = self.iam.create_role( + RoleName=name, + AssumeRolePolicyDocument=json.dumps(trust), + MaxSessionDuration=max_session, + Description="temporary: e2e_agentcore_gateway", + )["Role"]["Arn"] + except ClientError as e: + if e.response["Error"]["Code"] != "EntityAlreadyExists": + raise + arn = self.iam.get_role(RoleName=name)["Role"]["Arn"] + self.iam.put_role_policy( + RoleName=name, PolicyName="e2e", PolicyDocument=json.dumps(policy) + ) + return arn + + # -- create ----------------------------------------------------------- + + def create(self) -> None: + step("provisioning live AWS resources") + self.ddb.create_table( + TableName=self.table, + KeySchema=[ + {"AttributeName": "PK", "KeyType": "HASH"}, + {"AttributeName": "SK", "KeyType": "RANGE"}, + ], + AttributeDefinitions=[ + {"AttributeName": "PK", "AttributeType": "S"}, + {"AttributeName": "SK", "AttributeType": "S"}, + ], + BillingMode="PAY_PER_REQUEST", + ) + self.ddb.get_waiter("table_exists").wait(TableName=self.table) + print(f" table {self.table}") + + # -- interceptor Lambda (created first: the gateway references it) --- + lrole = self._role( + self.icept_role, + { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Principal": {"Service": "lambda.amazonaws.com"}, + "Action": "sts:AssumeRole", + } + ], + }, + { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Action": [ + "logs:CreateLogGroup", + "logs:CreateLogStream", + "logs:PutLogEvents", + ], + "Resource": "*", + } + ], + }, + ) + time.sleep(12) + buf = io.BytesIO() + with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as z: + z.writestr("lambda_function.py", INTERCEPTOR_SRC) + self.icept_arn = self.lam.create_function( + FunctionName=self.icept_fn, + Runtime="python3.12", + Role=lrole, + Handler="lambda_function.lambda_handler", + Code={"ZipFile": buf.getvalue()}, + Timeout=15, + MemorySize=256, + Description="temporary: e2e_agentcore_gateway request interceptor", + )["FunctionArn"] + for _ in range(20): + if self.lam.get_function(FunctionName=self.icept_fn)["Configuration"][ + "State" + ] == "Active": + break + time.sleep(3) + print(f" lambda {self.icept_fn} (REQUEST interceptor)") + + gw_trust = { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Principal": {"Service": "bedrock-agentcore.amazonaws.com"}, + "Action": "sts:AssumeRole", + "Condition": {"StringEquals": {"aws:SourceAccount": self.account}}, + } + ], + } + gw_arn_role = self._role( + self.gw_role, + gw_trust, + { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Action": ["bedrock:*", "bedrock-mantle:*"], + "Resource": "*", + }, + { + # Scoped to the one interceptor, per the gateway + # interceptor security guidance. + "Effect": "Allow", + "Action": "lambda:InvokeFunction", + "Resource": self.icept_arn, + }, + ], + }, + ) + print(f" role {self.gw_role} (gateway execution)") + + # Two gateways: the session credential is scoped to the first, and the + # second exists only to prove the scoping actually bites. + for label in ("primary", "decoy"): + extra: dict = {} + if label == "primary": + extra["interceptorConfigurations"] = [ + { + "interceptor": {"lambda": {"arn": self.icept_arn}}, + "interceptionPoints": ["REQUEST"], + # Required in the billing configuration so the + # interceptor can read the verified JWT it derives the + # tenant identity from; kept on here so the shape under + # test matches the one V4 actually deploys. + "inputConfiguration": {"passRequestHeaders": True}, + } + ] + g = self.acc.create_gateway( + name=f"e2e-{label}-{SUFFIX}", + roleArn=gw_arn_role, + protocolType="MCP", + authorizerType="AWS_IAM", + exceptionLevel="DEBUG", + description="temporary: e2e_agentcore_gateway", + **extra, + ) + self.gateways.append(g["gatewayId"]) + print(f" gateway {g['gatewayId']} ({label})") + for gid in self.gateways: + for _ in range(40): + if self.acc.get_gateway(gatewayIdentifier=gid)["status"] != "CREATING": + break + time.sleep(4) + # IAM propagation before the connector's model-discovery call. + time.sleep(15) + for gid in self.gateways: + t = self.acc.create_gateway_target( + gatewayIdentifier=gid, + name="bedrock", + targetConfiguration={ + "inference": {"connector": {"source": {"connectorId": "bedrock-mantle"}}} + }, + # Curating headers here is the AgentCore-side counterpart of + # llm-edge's FORWARD_HEADERS allowlist, and it is not optional: + # + # - left unset, the gateway relays whatever the caller sent, + # including the caller's own x-amz-security-token; and + # - Claude Code advertises a prompt-caching beta that this + # upstream rejects outright ("400 Unexpected value(s) + # `prompt-caching-scope-...` for the `anthropic-beta` + # header"), so anthropic-beta has to be dropped for the + # bedrock-mantle connector. A LiteLLM upstream accepts it, + # and would be allowlisted instead. + # + # content-type has to be listed explicitly: once + # allowedRequestHeaders is set, a connector target stops + # relaying it and the upstream answers "400 Expected request + # with `Content-Type: application/json`". (A provider target + # supplies it itself, which is why this is easy to miss.) + metadataConfiguration={ + "allowedRequestHeaders": [ + "content-type", + "anthropic-version", + "accept", + ] + }, + credentialProviderConfigurations=[ + {"credentialProviderType": "GATEWAY_IAM_ROLE"} + ], + ) + for _ in range(50): + st = self.acc.get_gateway_target( + gatewayIdentifier=gid, targetId=t["targetId"] + ) + if st["status"] != "CREATING": + break + time.sleep(4) + print(f" target {gid} -> {st['status']} {st.get('statusReasons', '')}") + if st["status"] != "READY": + raise SystemExit("inference target did not become READY") + + self.caller_arn = self._role( + self.caller_role, + { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Principal": {"AWS": self._caller_principal()}, + "Action": ["sts:AssumeRole", "sts:TagSession"], + } + ], + }, + { + "Version": "2012-10-17", + "Statement": [ + { + "Effect": "Allow", + "Action": "bedrock-agentcore:InvokeGateway", + "Resource": [ + f"arn:aws:bedrock-agentcore:{self.region}:{self.account}:gateway/{g}" + for g in self.gateways + ], + } + ], + }, + ) + print(f" role {self.caller_role} (per-session caller)") + # The role policy deliberately allows BOTH gateways; the narrowing to + # one is the session policy the service attaches at mint time, so the + # scoping assertion tests the session policy and not this role. + time.sleep(12) + + def base_url(self, which: int = 0) -> str: + gid = self.gateways[which] + return ( + f"https://{gid}.gateway.bedrock-agentcore.{self.region}" + f".amazonaws.com/inference" + ) + + # -- destroy ---------------------------------------------------------- + + def destroy(self) -> None: + step("tearing down") + for gid in self.gateways: + try: + for t in self.acc.list_gateway_targets(gatewayIdentifier=gid).get( + "items", [] + ): + self.acc.delete_gateway_target( + gatewayIdentifier=gid, targetId=t["targetId"] + ) + time.sleep(6) + self.acc.delete_gateway(gatewayIdentifier=gid) + print(f" deleted gateway {gid}") + except Exception as e: # noqa: BLE001 + print(f" gateway {gid}: {str(e)[:100]}") + try: + self.lam.delete_function(FunctionName=self.icept_fn) + print(f" deleted lambda {self.icept_fn}") + except Exception as e: # noqa: BLE001 + print(f" lambda: {str(e)[:100]}") + try: + self.logs.delete_log_group(logGroupName=f"/aws/lambda/{self.icept_fn}") + except Exception: # noqa: BLE001 + pass + for name in (self.caller_role, self.gw_role, self.icept_role): + for pol in self.iam.list_role_policies(RoleName=name).get( + "PolicyNames", [] + ): + try: + self.iam.delete_role_policy(RoleName=name, PolicyName=pol) + except Exception: # noqa: BLE001 + pass + try: + self.iam.delete_role(RoleName=name) + print(f" deleted role {name}") + except Exception as e: # noqa: BLE001 + print(f" role {name}: {str(e)[:100]}") + try: + self.ddb.delete_table(TableName=self.table) + print(f" deleted table {self.table}") + except Exception as e: # noqa: BLE001 + print(f" table: {str(e)[:100]}") + + +# ------------------------------------------------------------------ the test + + +def main() -> int: + ap = argparse.ArgumentParser() + ap.add_argument("--region", default=os.environ.get("AWS_REGION", "us-east-1")) + ap.add_argument("--model", default="claude-haiku-4-5") + ap.add_argument("--keep", action="store_true", help="leave AWS resources behind") + args = ap.parse_args() + + if not any( + os.access(os.path.join(p, "claude"), os.X_OK) + for p in os.environ.get("PATH", "").split(os.pathsep) + if p + ): + print("the `claude` CLI is not on PATH — cannot run the chain test") + return 2 + + fx = Fixture(args.region) + fx.create() + + # ---- env must be set before app.config instantiates Settings() ---- + os.environ.update( + PLATFORM_AWS_REGION=args.region, + PLATFORM_DYNAMO_TABLE=fx.table, + PLATFORM_AGENTCORE_GATEWAY_CALLER_ROLE_ARN=fx.caller_arn, + ) + sys.path.insert(0, os.path.join(REPO, "backend")) + sys.path.insert(0, os.path.join(REPO, "runtimes", "agent-sdk-kernel", "src")) + from app.services.llm_credentials_service import LlmCredentialsService # noqa: E402 + import llm_shim # noqa: E402 + + svc = LlmCredentialsService() + + # Observability only, so a chain assertion cannot be satisfied by Claude + # Code quietly using some other credential. Both wrappers delegate to the + # real implementation; nothing about the path under test changes. + seen: list[str] = [] + _orig_proxy = llm_shim._Handler._proxy + + def _counting_proxy(self): # noqa: ANN001, ANN202 + seen.append(f"{self.command} {self.path}") + return _orig_proxy(self) + + # _sign_sigv4 is the one place the full request body is in hand, which is + # what makes the shape of a multi-turn tool loop visible. + sent: list[dict] = [] + _orig_sign = llm_shim._sign_sigv4 + + def _recording_sign(grant, method, url, body): # noqa: ANN001, ANN202 + try: + b = json.loads(body) if body else {} + except Exception: # noqa: BLE001 + b = {} + msgs = [m for m in (b.get("messages") or []) if isinstance(m, dict)] + + def _blocks(m: dict) -> list: + c = m.get("content") + return c if isinstance(c, list) else [] + + sent.append( + { + "path": url.split("amazonaws.com", 1)[-1] or url, + "bytes": len(body or b""), + "model": b.get("model"), + "msgs": len(msgs), + "tools": len(b.get("tools") or []), + "tool_result": sum( + 1 + for m in msgs + for c in _blocks(m) + if isinstance(c, dict) and c.get("type") == "tool_result" + ), + "stream": bool(b.get("stream")), + } + ) + return _orig_sign(grant, method, url, body) + + llm_shim._Handler._proxy = _counting_proxy + llm_shim._sign_sigv4 = _recording_sign + llm_shim.start() + + spec = { + "backend": "agentcore_gateway", + "base_url": fx.base_url(0), + "model": args.model, + "alias_models": {"haiku": args.model}, + } + + try: + step("[chain] mint per-session credentials and run real Claude Code") + sid_a = f"ses-{secrets.token_hex(20)}" + grant = svc.mint_agentcore(sid_a, "alice", spec, team="team-alpha") + check("mint_agentcore returned a grant", bool(grant)) + if not grant: + return 1 + print(f" mode={grant['mode']} region={grant['region']} " + f"service={grant['service']} expires_at={grant['expires_at']}") + print(f" credential is STS: access_key_id={grant['access_key_id'][:8]}…, " + f"session_token={'yes' if grant.get('session_token') else 'no'}") + check( + "grant carries no bearer token and no upstream key", + "token" not in grant and "secret_name" not in grant, + ) + + local_token = llm_shim.register(grant) + check("shim issued a per-invocation local token", bool(local_token)) + check("local token differs from the STS secret", local_token != grant["secret_access_key"]) + + env = dict(os.environ) + # Exactly what the kernel writes for a gateway-mode session. + env.pop("CLAUDE_CODE_USE_BEDROCK", None) + env.pop("CLAUDE_CODE_PROVIDER", None) + env.pop("ANTHROPIC_DEFAULT_OPUS_MODEL", None) + env.update( + ANTHROPIC_BASE_URL=llm_shim.BASE_URL, + ANTHROPIC_AUTH_TOKEN=local_token, + ANTHROPIC_MODEL=args.model, + ANTHROPIC_SMALL_FAST_MODEL=args.model, + CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC="1", + CLAUDE_CONFIG_DIR=os.path.join("/tmp", f"e2e-acgw-cfg-{SUFFIX}"), + ) + before = len(seen) + proc = subprocess.run( + ["claude", "-p", "Reply with the result of 6*7 and nothing else."], + env=env, + capture_output=True, + text=True, + timeout=240, + stdin=subprocess.DEVNULL, + ) + out = (proc.stdout or "") + (proc.stderr or "") + print(f" claude output: {out.strip()[:160]!r}") + check("claude produced no API error", "API Error" not in out, out.strip()[:90]) + check("claude answered", "42" in out) + check( + "the shim forwarded Claude Code's request(s)", + len(seen) > before, + f"{len(seen) - before} request(s)", + ) + print(f" paths Claude Code sent through the chain: {seen[before:]}") + check( + "every path is one AgentCore Gateway accepts inbound", + all( + p.split(" ", 1)[1].split("?")[0] + in ("/v1/messages", "/v1/chat/completions", "/v1/responses", "/v1/models") + for p in seen[before:] + ), + ",".join(sorted({p.split(" ", 1)[1] for p in seen[before:]})), + ) + + step("[task] a real multi-turn task with tool use, not a single turn") + work = os.path.join("/tmp", f"e2e-acgw-work-{SUFFIX}") + os.makedirs(work, exist_ok=True) + before_n, before_sent = len(seen), len(sent) + proc = subprocess.run( + [ + "claude", + "-p", + "Create fizz.py that prints the FizzBuzz sequence for 1 to 15, " + "one value per line. Run it with python3. Then reply with only " + "the 15th line of its output.", + "--allowedTools", + "Write,Read,Bash", + ], + env=env, + cwd=work, + capture_output=True, + text=True, + timeout=600, + stdin=subprocess.DEVNULL, + ) + tout = (proc.stdout or "") + (proc.stderr or "") + print(f" claude output: {tout.strip()[:200]!r}") + rounds = sent[before_sent:] + print(f" {len(seen) - before_n} request(s) through the shim:") + print(f" {'#':>2} {'path':26} {'bytes':>7} {'msgs':>5} {'tools':>5}" + f" {'tool_result':>11} {'stream':>6}") + for i, r in enumerate(rounds, 1): + print( + f" {i:>2} {r['path'][:26]:26} {r['bytes']:>7} {r['msgs']:>5}" + f" {r['tools']:>5} {r['tool_result']:>11} {str(r['stream']):>6}" + ) + check("task produced no API error", "API Error" not in tout, tout.strip()[:90]) + check("task answered correctly", "FizzBuzz" in tout) + check( + "the task took several turns through the chain", + len(rounds) > 1, + f"{len(rounds)} model calls", + ) + check( + "tool results were carried back to the model", + any(r["tool_result"] for r in rounds), + f"max {max((r['tool_result'] for r in rounds), default=0)} per request", + ) + check( + "the agent actually wrote the file", + os.path.exists(os.path.join(work, "fizz.py")), + ) + check( + "context grew across turns (the loop really accumulated)", + len(rounds) > 1 and rounds[-1]["bytes"] > rounds[0]["bytes"], + f"{rounds[0]['bytes']} -> {rounds[-1]['bytes']} bytes" + if len(rounds) > 1 + else "", + ) + paths = sorted({r["path"] for r in rounds}) + print(f" distinct paths used by the task: {paths}") + check( + "no path outside AgentCore Gateway's inbound set", + all( + p.split("?")[0].removeprefix("/inference") + in ("/v1/messages", "/v1/chat/completions", "/v1/responses", "/v1/models") + for p in paths + ), + ",".join(paths), + ) + + step("[failfast] shim refuses a model this session was not routed to") + code, body = shim_call(local_token, {"model": "claude-opus-4-7", "max_tokens": 8, + "messages": [{"role": "user", "content": "hi"}]}) + check("non-permitted model refused locally", code == 403, f"HTTP {code}") + + step("[scoping] the session policy narrows the credential to one gateway") + code_ok, _ = raw_gateway_call(grant, fx.base_url(0), args.model) + code_no, body_no = raw_gateway_call(grant, fx.base_url(1), args.model) + check("credential works on its own gateway", code_ok == 200, f"HTTP {code_ok}") + check( + "same credential refused on a second gateway", + code_no == 403, + f"HTTP {code_no} {body_no[:80]}", + ) + + step("[revoke] revoking one session leaves another untouched") + sid_b = f"ses-{secrets.token_hex(20)}" + grant_b = svc.mint_agentcore(sid_b, "bob", spec, team="team-beta") + check("second session minted", bool(grant_b)) + svc.revoke(sid_a) + print(" revoke(sid_a) issued; polling for effect") + t0 = time.time() + revoked_at = None + while time.time() - t0 < 120: + code, _ = raw_gateway_call(grant, fx.base_url(0), args.model) + if code != 200: + revoked_at = time.time() - t0 + break + time.sleep(3) + check( + "session A's credential stopped working", + revoked_at is not None, + f"after {revoked_at:.1f}s" if revoked_at else "still valid after 120s", + ) + code_b, _ = raw_gateway_call(grant_b, fx.base_url(0), args.model) + check("session B unaffected", code_b == 200, f"HTTP {code_b}") + svc.revoke(sid_b) + finally: + for d in ( + os.path.join("/tmp", f"e2e-acgw-work-{SUFFIX}"), + os.path.join("/tmp", f"e2e-acgw-cfg-{SUFFIX}"), + ): + shutil.rmtree(d, ignore_errors=True) + if not args.keep: + fx.destroy() + else: + print(f"\n--keep: left {fx.gateways}, roles {fx.gw_role}/{fx.caller_role}, " + f"table {fx.table}") + + print() + if FAIL: + print(f"FAILED ({len(FAIL)}): " + "; ".join(FAIL)) + return 1 + print("all assertions passed") + return 0 + + +def shim_call(local_token: str, body: dict) -> tuple[int, str]: + """Call the loopback shim the way the CLI subprocess does.""" + req = urllib.request.Request( + llm_shim_base() + "/v1/messages", + data=json.dumps(body).encode(), + headers={ + "content-type": "application/json", + "authorization": f"Bearer {local_token}", + "anthropic-version": "2023-06-01", + }, + method="POST", + ) + try: + with urllib.request.urlopen(req, timeout=120) as r: + return r.status, r.read().decode()[:200] + except urllib.error.HTTPError as e: + return e.code, e.read().decode()[:200] + + +def llm_shim_base() -> str: + import llm_shim + + return llm_shim.BASE_URL + + +def raw_gateway_call(grant: dict, base_url: str, model: str) -> tuple[int, str]: + """Sign and call the gateway directly, bypassing the shim. + + This is what a session's user could do with the credential they can read + out of the kernel process, which is why the meaningful boundaries (session + policy, revocation) are asserted through this path rather than the shim's. + """ + from botocore.auth import SigV4Auth + from botocore.awsrequest import AWSRequest + from botocore.credentials import Credentials + + data = json.dumps( + {"model": model, "max_tokens": 8, "messages": [{"role": "user", "content": "hi"}]} + ) + req = AWSRequest( + method="POST", + url=base_url.rstrip("/") + "/v1/messages", + data=data, + headers={"content-type": "application/json", "anthropic-version": "2023-06-01"}, + ) + SigV4Auth( + Credentials( + grant["access_key_id"], grant["secret_access_key"], grant.get("session_token") + ), + grant.get("service") or "bedrock-agentcore", + grant["region"], + ).add_auth(req) + r = urllib.request.Request( + req.url, data=data.encode(), headers=dict(req.headers), method="POST" + ) + try: + with urllib.request.urlopen(r, timeout=120) as resp: + return resp.status, resp.read().decode()[:200] + except urllib.error.HTTPError as e: + return e.code, e.read().decode()[:200] + + +if __name__ == "__main__": + sys.exit(main()) From 46b67bc39e3f08101d23414acca8bd77b019023a Mon Sep 17 00:00:00 2001 From: zzkamzn Date: Fri, 18 Sep 2026 03:02:36 +0000 Subject: [PATCH 2/3] Docs: the agentcore_gateway model backend, and what it does not claim MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Folds the third model backend into the existing model-access sections rather than adding a new page: architecture.md's backend table, security-explainer §9 (which gains the second call chain), deployment.md §2 as Option A2, permissions.md's principal table, the user guide's Governance card, and the EXTENDING.md configuration index. Three things the evaluation turned up that operators need in writing: - The kernel roles already hold InvokeGateway on gateway/* so they can reach MCP tool gateways, and that wildcard covers an inference gateway too — a root user in the microVM could call it on the role's own identity, with no session tag to revoke. Enabling this backend requires an explicit Deny for the inference gateway's ARN; without it the per-session credential is decorative. Documented as a prerequisite in deployment.md and permissions.md. - A gateway interceptor's escaping exception is relayed to the caller with its full stack trace, regardless of exceptionLevel — and the caller is the tenant's microVM. Interceptors must catch everything themselves. The reassuring half: interceptor failure is fail-closed. - allowedRequestHeaders is not optional. Left unset the gateway relays the caller's own x-amz-security-token; once set, a connector target stops relaying content-type unless it is listed. Also corrects two claims §9.4 and §9.5 had been overstating, independent of this backend: the platform's quota counts invocations rather than tokens, so there is no per-user token ceiling on either gateway mode; and credential reuse across sessions is not prevented by either, because nothing proves a caller is the session it claims to be. Co-Authored-By: Claude Opus 5 (1M context) --- EXTENDING.md | 1 + README.md | 5 +- docs/architecture.md | 18 +++--- docs/deployment.md | 55 +++++++++++++++++ docs/permissions.md | 21 ++++++- docs/security-explainer.zh.md | 113 +++++++++++++++++++++++++++++----- docs/user-guide.md | 21 ++++--- 7 files changed, 200 insertions(+), 34 deletions(-) diff --git a/EXTENDING.md b/EXTENDING.md index d839930..2dbd38b 100644 --- a/EXTENDING.md +++ b/EXTENDING.md @@ -35,6 +35,7 @@ with an upstream update. | Which layers to create | `enable_runtime`, `enable_portal`, `enable_llm_edge`, `enable_team_auth`, `enable_team_demo`, `enable_mcp_hub_demo` | [terraform/variables.tf](terraform/variables.tf) — the quick start uses them to stage the deploy | | Bedrock direct mode | `CLAUDE_CODE_USE_BEDROCK=1` | runtime env; no key involved | | Your LLM gateway | `enable_llm_edge = true` + the backend's `base_url` in Governance → Model backends | key in Secrets Manager, read only by `llm-edge`; kernels get a per-session grant, never the key | +| An AgentCore Gateway instead | `PLATFORM_AGENTCORE_GATEWAY_CALLER_ROLE_ARN` + the backend's `base_url` (the gateway's `/inference` URL) | no key on the platform side at all and no `llm-edge` to run; kernels get session-tagged STS credentials and SigV4-sign. Requires denying the kernel roles direct `InvokeGateway` on that gateway — see [docs/permissions.md](docs/permissions.md) | | Backend runtime settings (table, buckets, ARNs, Cognito, CORS, admin tiers) | `PLATFORM_*` env vars | [backend/app/config.py](backend/app/config.py) — every field there is `PLATFORM_` (`env_prefix`); `backend/.env.example` for local runs | Terraform state is yours: `terraform/backend.tf.example` shows the remote-state diff --git a/README.md b/README.md index 1903dd0..5b20fb9 100644 --- a/README.md +++ b/README.md @@ -151,9 +151,10 @@ cp terraform.tfvars.example terraform.tfvars # then edit terraform init terraform apply -var enable_runtime=false -var enable_portal=false -# 2. Store your LLM gateway key (skip if using Bedrock direct). +# 2. Store your LLM gateway key (skip for Bedrock direct or AgentCore Gateway). # Readable only by the llm-edge service; it never enters a kernel container. -# Gateway mode also needs enable_llm_edge=true — see docs/deployment.md §2. +# Gateway mode also needs enable_llm_edge=true — see docs/deployment.md §2, +# which also covers the agentcore_gateway backend (no key on this side). aws secretsmanager put-secret-value \ --secret-id agent-platform/llm-gateway-key \ --secret-string '{"api_key":"sk-..."}' diff --git a/docs/architecture.md b/docs/architecture.md index 4b49d0b..508edb7 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -217,17 +217,18 @@ Claude Code calls it from a terminal or the SDK kernel calls it per-invoke. ### 2. Model access — LLM gateway first -Both kernels support two model backends: +Both kernels support three model backends: | Mode | How | When to use | |---|---|---| | **Bedrock direct** | `CLAUDE_CODE_USE_BEDROCK=1` + cross-region inference profile (`global.` model ID prefix). Container IAM role; no key of any kind | Simplest path, and there is no model credential to leak | -| **LLM gateway** | Always per session, never a container default. The gateway key lives only in the `llm-edge` service; a kernel gets a session-scoped grant and reaches the gateway through it (`enable_llm_edge`) | Centralized model governance: allow-lists, budgets, cost attribution per team | +| **LLM gateway** (`litellm`) | Always per session, never a container default. The gateway key lives only in the `llm-edge` service; a kernel gets a session-scoped grant and reaches the gateway through it (`enable_llm_edge`) | Centralized model governance: allow-lists, budgets, cost attribution per team | +| **AgentCore Gateway** (`agentcore_gateway`) | Also always per session. The upstream provider credential lives in the gateway's own token vault, so there is no key on the platform side at all and no broker service to run. A kernel gets STS credentials tagged with its session id and SigV4-signs each request (`agentcore_gateway_caller_role_arn`) | The same governance goal with a managed hop instead of `llm-edge`, and per-session revocation expressed as an IAM Deny | -Gateway mode has no container-level configuration on purpose. A session's user -is root in its microVM and the headless kernel runs agent tools in a +Neither gateway mode has container-level configuration, on purpose. A session's +user is root in its microVM and the headless kernel runs agent tools in a subprocess, so any credential placed in a kernel container is a credential its -users have. Reaching the gateway therefore requires a grant the backend mints +users have. Reaching either gateway therefore requires a grant the backend mints per session; see security-explainer.zh.md §9. The environment sets the **container default**; the model control plane @@ -489,12 +490,15 @@ a new runtime version automatically). (SDK kernel only, and nothing the platform ships needs it, since search authenticates with the kernel's own role). The LLM gateway key is deliberately **not** among them: it is readable only by the `llm-edge` task - role, which no session can enter. + role, which no session can enter. With the `agentcore_gateway` backend there + is no such key on the platform side at all — the upstream credential lives in + the gateway's token vault. - No secrets are baked into images; keys are read from Secrets Manager at container/Lambda start. - The browser and end users hold **no AWS credentials** — all AWS access is server-side under the platform's seven IAM roles (an eighth, - `agent-platform-llm-edge`, in gateway mode), enumerated in + `agent-platform-llm-edge`, in litellm gateway mode; a ninth, the AgentCore + Gateway caller role, in `agentcore_gateway` mode), enumerated in [permissions.md §1](permissions.md#1-principals-at-a-glance). - If your LLM gateway is HTTP-only, traffic from NAT → gateway crosses the network unencrypted; put a TLS listener or PrivateLink in front for diff --git a/docs/deployment.md b/docs/deployment.md index 8d21339..c19fef6 100644 --- a/docs/deployment.md +++ b/docs/deployment.md @@ -98,6 +98,61 @@ With `enable_llm_edge = false`, selecting the litellm backend makes the platform refuse the session with a 503 instead of falling back to handing a container the key. +### Option A2 — AgentCore Gateway + +Same promise as Option A with one fewer service to run: the upstream provider +credential lives in the gateway's own token vault, so there is no key on the +platform side and no `llm-edge` to deploy. A kernel gets STS credentials tagged +with its session id and signs each request. + +1. Create the gateway and its inference target, and note the `/inference` URL. + Two settings are not optional: + + - **`metadataConfiguration.allowedRequestHeaders`** must list at least + `content-type`, `anthropic-version` and `accept`. Left unset the gateway + relays whatever the caller sent — including the caller's own + `x-amz-security-token` — and Claude Code's prompt-caching `anthropic-beta` + value, which a strict upstream refuses. Once it *is* set, a connector + target stops relaying `content-type` unless it is listed. + - a **REQUEST interceptor** if the upstream rejects fields Claude Code sends + (`context_management` is one). The gateway forwards bodies verbatim, so an + interceptor is the only place outside the container to adapt them. It is + also where a per-tenant identity would be injected for upstream billing. + Interceptors must catch every exception themselves — an escaping one is + relayed to the caller with its stack trace. + +2. Create the role the backend assumes per session, trusted by the backend's + own role, granting `bedrock-agentcore:InvokeGateway` on the gateway ARN and + allowing `sts:TagSession`. Its `MaxSessionDuration` must cover the async + grant lifetime (9 h) if headless async runs use this backend. + + ```hcl + # no dedicated module yet — supply the role ARN you created + ``` + + ```bash + PLATFORM_AGENTCORE_GATEWAY_CALLER_ROLE_ARN=arn:aws:iam:::role/ + ``` + +3. **Deny the kernel roles direct access to this gateway.** The kernel roles + already hold `bedrock-agentcore:InvokeGateway` on `gateway/*` so they can + reach MCP tool gateways, and that wildcard would also cover the inference + gateway — letting a root user in the microVM call it on the kernel role's own + identity, with no session tag to revoke. Add an explicit Deny for the + inference gateway's ARN to each kernel role; see permissions.md §3. Without + this the per-session credential is decorative. + +4. Point the backend at the gateway in the portal: **Governance → Model + backends → agentcore_gateway**, setting `base_url` to the `/inference` URL + and the model catalog. There is no `secret_name` for this backend. + +`scripts/e2e_agentcore_gateway.py` provisions all of the above against a live +account and asserts the chain end to end, including per-session revocation; it +is the fastest way to see a working configuration. + +With the caller role ARN unset, selecting this backend makes the platform refuse +the session rather than fall back to a shared credential. + ### Option B — Bedrock direct Set: diff --git a/docs/permissions.md b/docs/permissions.md index e911be6..0bf7f50 100644 --- a/docs/permissions.md +++ b/docs/permissions.md @@ -59,7 +59,8 @@ create their own roles in `terraform/modules/team_auth` and | 5 | **`agent-platform-backend-task`** | `portal` module | The EKS cluster's OIDC provider via **IRSA** — only the `backend` and `entry` service accounts in the `portal` namespace | The control-plane API on EKS: session routing, invoking runtimes, memory/scheduler/eval management, minting workspace credentials. | | 6 | **`agent-platform-schedule-runner`** | `portal` module (CDK: `PortalStack/ScheduleRunner`) | `lambda.amazonaws.com` | Fires scheduled invocations at each occurrence. Packages the same service layer as the backend. | | 7 | **`agent-platform-scheduler`** | `portal` module (CDK: `PortalStack/SchedulerRole`) | `scheduler.amazonaws.com` (conditioned on `aws:SourceAccount`) | The role EventBridge Scheduler assumes to invoke the runner Lambda and send to the DLQ. Holds no data-plane permissions. | -| 8 | **`agent-platform-llm-edge`** — *gateway mode only* | `llm_edge` module | The cluster's OIDC provider via **IRSA** — only the `edge` service account in the `llm-edge` namespace | The gateway broker. Holds exactly two grants: `secretsmanager:GetSecretValue` on the gateway key (the **only** principal that has it) and `dynamodb:GetItem` on the platform table for per-session token lookup — no `Query`, no `Scan`, nothing else. | +| 8 | **`agent-platform-llm-edge`** — *litellm gateway mode only* | `llm_edge` module | The cluster's OIDC provider via **IRSA** — only the `edge` service account in the `llm-edge` namespace | The gateway broker. Holds exactly two grants: `secretsmanager:GetSecretValue` on the gateway key (the **only** principal that has it) and `dynamodb:GetItem` on the platform table for per-session token lookup — no `Query`, no `Scan`, nothing else. | +| 9 | **AgentCore Gateway caller role** — *`agentcore_gateway` mode only* | supplied by the operator (`agentcore_gateway_caller_role_arn`); no module yet | The **backend role**, which must also be allowed `sts:TagSession` | The identity a *session* borrows. Holds one grant: `bedrock-agentcore:InvokeGateway` on the gateway ARN. The backend assumes it once per session with a session tag (`session_id`) and a session policy narrowing it to a single gateway, so the credentials a container ends up holding are strictly weaker than the role. Session revocation is an inline Deny on this role conditioned on `aws:PrincipalTag/session_id`, which is why the backend also needs `iam:PutRolePolicy`/`DeleteRolePolicy` on it. | The single most important property for a security review: **the browser and end users never hold AWS credentials.** All AWS access is server-side under @@ -90,7 +91,23 @@ tools role (#3) carries **only** the first three rows: | `Skills` / `SkillsList` | `s3:GetObject`; `s3:ListBucket` conditioned on `s3:prefix` | `skills/*` in the workspace bucket only | Mount skill packages before the agent starts. Read-only. | | *(no LLM gateway secret grant)* | — | — | Deliberately absent. A kernel role is reachable from inside the session it serves (root shell in the Dev Workbench microVM; agent tools in the headless kernel's CLI subprocess), so a kernel that can read the gateway key is a kernel whose users have it. Only `agent-platform-llm-edge` holds that read; kernels reach the gateway through it with a per-session grant. | | `InvokeMcpRuntimes` | `bedrock-agentcore:InvokeAgentRuntime` | `runtime/mcp_tools_kernel-*` and its `runtime-endpoint/*` only | Kernels reach the AgentCore-hosted MCP server through `mcp-proxy-for-aws`, which SigV4-signs with this role. Scoped to the MCP runtime name — **not** all runtimes. | -| `InvokeGateways` | `bedrock-agentcore:InvokeGateway` | `gateway/*` in this account, in both the platform's region and `us-east-1` | Kernels reach registry entries of kind `agentcore-gateway` (SigV4 through `mcp-proxy-for-aws`) — the feed pipelines' managed Web Search connector is one. Gateway IDs are generated at deploy time, so this is scoped by account+region rather than to a single gateway; `us-east-1` is listed explicitly because the Web Search connector is offered only there. | +| `InvokeGateways` | `bedrock-agentcore:InvokeGateway` | `gateway/*` in this account, in both the platform's region and `us-east-1` | Kernels reach registry entries of kind `agentcore-gateway` (SigV4 through `mcp-proxy-for-aws`) — the feed pipelines' managed Web Search connector is one. Gateway IDs are generated at deploy time, so this is scoped by account+region rather than to a single gateway; `us-east-1` is listed explicitly because the Web Search connector is offered only there. **⚠ Using the `agentcore_gateway` model backend requires narrowing this** — see the note below. | + +> **Prerequisite for the `agentcore_gateway` model backend.** The `InvokeGateways` +> statement above is a wildcard over `gateway/*`, which would also cover the +> *inference* gateway. That defeats the point of minting per-session credentials: +> a root user in the microVM can read the kernel role from the metadata endpoint +> and call the inference gateway on the role's own identity, with no session tag +> to revoke and no session policy to narrow it. Before enabling that backend, add +> an explicit **Deny** on the kernel roles for the inference gateway's ARN — Deny +> wins over the wildcard Allow, so tool gateways keep working while the inference +> gateway becomes reachable only with a backend-minted session credential: +> +> ```json +> { "Sid": "NoDirectInferenceGateway", "Effect": "Deny", +> "Action": "bedrock-agentcore:InvokeGateway", +> "Resource": "arn:aws:bedrock-agentcore:::gateway/" } +> ``` | `BuiltinTools` | `bedrock-agentcore:StartCodeInterpreterSession`, `InvokeCodeInterpreter`, `StopCodeInterpreterSession`, `GetCodeInterpreterSession`, `StartBrowserSession`, `StopBrowserSession`, `GetBrowserSession`, `UpdateBrowserStream`, `ConnectBrowserAutomationStream`, `ConnectBrowserLiveViewStream` | `code-interpreter/aws.codeinterpreter.v1`, `browser/aws.browser.v1` (AWS-managed), plus `{account}:code-interpreter/*` and `{account}:browser/*` (custom variants) | Code Interpreter and Browser built-in tools run in AWS-managed sandboxes under this role — no separate tool runtime to deploy. | | `BedrockInvoke` | `bedrock:InvokeModel`, `InvokeModelWithResponseStream` | `*` | The model control plane (Governance → Model backends) can route any agent to Bedrock per invocation. Cross-region inference profiles span regions, so this cannot be region-pinned. See [§4](#4-wildcard-resource-statements). | diff --git a/docs/security-explainer.zh.md b/docs/security-explainer.zh.md index 19f0a83..aaa8726 100644 --- a/docs/security-explainer.zh.md +++ b/docs/security-explainer.zh.md @@ -442,19 +442,28 @@ CLI 子进程,agent 的工具就在那个子进程里执行。 (内核角色对 `workspaces/*` 零权限,由后端按 session 铸凭证,见第 4 节)。 模型凭证现在同样遵循。 -### 9.2 两个后端 +### 9.2 三个后端 -模型调用有两个后端,由治理页的模型控制面按 agent 路由: +模型调用有三个后端,由治理页的模型控制面按 agent 路由: - **Bedrock 直连**:以容器的 IAM 角色调用 `bedrock:InvokeModel*`,不存在长期 密钥,模型 ID 用 `global.` 前缀的跨区推理配置。 -- **LLM 网关(如 LiteLLM)**:容器**不持有网关密钥**。密钥只存在于 +- **LLM 网关(`litellm`,如 LiteLLM)**:容器**不持有网关密钥**。密钥只存在于 `llm-edge` 这个内部服务的任务角色里(`agent-platform-llm-edge`,全平台唯一 被授予该密钥读权的主体;运行时角色已不再有此授权)。内核拿到的是一份按 session 铸出的短期凭证。 +- **AgentCore Gateway(`agentcore_gateway`)**:容器同样不持有上游密钥,而且 + **平台侧也不持有** —— 上游凭证在网关自己的 token vault 里,因此不需要 + `llm-edge` 这个中间服务。内核拿到的是一份**打了 session 标签的 STS 凭证**, + 每个请求由内核 shim 做 SigV4 签名。 + +两种网关模式的共同点是:容器里没有可在别处兑现的东西。差别只在「谁替容器持有 +上游凭证」和「按会话吊销怎么表达」。 ### 9.3 网关模式的调用链 +**`litellm` 模式** + ``` Claude Code / Agent SDK ANTHROPIC_BASE_URL = http://127.0.0.1:8787 ← 容器内 loopback @@ -476,24 +485,87 @@ grant 里的上游地址、密钥名、模型白名单,全部由后端在铸 `llm-edge` 每次调用都重新读取。**容器上报的任何路由信息都不被采信**,所以租户 无法通过改请求把自己路由到别的模型或别的上游。 +**`agentcore_gateway` 模式** + +``` +Claude Code / Agent SDK + ANTHROPIC_BASE_URL = http://127.0.0.1:8787 ← 同上,仍是 loopback + ANTHROPIC_AUTH_TOKEN = <本次 invocation 的随机 token> + ↓ +loopback shim(内核进程内) + STS 凭证(AK/SK/SessionToken)只存在于该进程内存 + 按请求做 SigV4 签名;Claude Code 不会签名,这正是签名必须在这里而不是 + 在 CLI 子进程里的原因 + ↓ +AgentCore Gateway(IAM authorizer,公网端点但只认 SigV4) + 三层策略同时求值: + · caller 角色的身份策略 + · 铸凭证时附加的 session policy —— 只允许这一个 gateway + · 会话结束时写入的标签条件 Deny —— aws:PrincipalTag/session_id + ↓ +(可选)REQUEST interceptor:容器外唯一能做「容器内不可信之事」的位置 + ↓ +上游模型提供方(凭证在网关的 token vault,平台与容器都取不到) +``` + +这一模式下「路由不可篡改」由 IAM 与网关配置共同保证:gateway 地址写在凭证块 +里,模型到 target 的映射是网关侧配置。 + +**部署前置条件(否则 per-session 凭证形同虚设)**:内核角色为了访问 MCP 工具 +网关,已持有 `bedrock-agentcore:InvokeGateway` on `gateway/*`。这个通配同样覆盖 +推理网关,于是 microVM 内的 root 用户可以从 metadata 端点读出内核角色凭证、以 +**角色自身身份**直接调用推理网关 —— 既没有 session 标签可吊销,也没有 session +policy 可收窄。启用本后端前必须给内核角色加一条针对推理网关 ARN 的显式 +**Deny**(Deny 优先于通配 Allow,工具网关不受影响),见 `docs/permissions.md` +第 3 节。 + +**另一条要写明**:IAM 条件看不到请求体,所以**按会话的模型白名单无法表达为 +IAM 条件**。内核 shim 里的那道检查是 +fail-fast 优化而**不是**边界(会话用户在该 microVM 内是 root,可以绕过 shim +直连网关)。要真正强制,需要网关的 request interceptor,或一个模型一个 +gateway 并用 session policy 枚举资源。白名单无论如何都会记在 grant 上,所以 +决策本身是可审计的。 + ### 9.4 可验收的三条 1. 容器的环境变量、文件、进程命令行、Claude Code 会话里,**不存在任何真实 密钥**。原先放平台级网关 key 的位置现在是字面量 `unused`。 2. 把容器内一切可获取的材料(环境变量、文件、内存中的 session 凭证、IAM 角色 - 凭证)导出到外部机器,**均不可用**:`llm-edge` 的监听器在 VPC 内网,公网无 - 路由;session 凭证在一小时内过期,且只对该 session 被允许的模型有效。 -3. 在 session 内滥用只能消耗该 session 自己的额度,且归因到具体用户。 + 凭证)导出到外部机器,其可用性是**受限且有时限**的: + - `litellm` 模式:`llm-edge` 的监听器在 VPC 内网、公网无路由,凭证出了 VPC + 即废;grant 在一小时内过期。 + - `agentcore_gateway` 模式:网关端点在公网,但只接受 SigV4,且凭证被 + session policy 限死在**一个** gateway 上、并绑定 session 标签;STS 会话 + ≥15 分钟即到期,会话结束时另有标签条件 Deny 立即生效。 + 两种模式下,**上游模型提供方的真实凭证都不在导出物之内**。 +3. 每次调用可归因到具体 session 与用户。 ### 9.5 残余边界 用户可以在自己的 session 里直接调用那个 loopback 端口来使用模型。这是设计 -预期:这本就是他有权使用的额度,有上限、可归因、随 session 失效。安全团队 -评估时应当把它理解为"用户正常使用自己的配额",而不是越权。 +预期:这本就是他有权使用的额度,可归因、随 session 失效。安全团队评估时应当 +把它理解为"用户正常使用自己的配额",而不是越权。 + +**但"额度"这个词要说准**:平台侧的配额计的是调用次数,不是 token。按用户/团队 +的 **token 额度强制由上游网关按 key 施加**,平台不做 per-user 强制 —— 同一个 +后端下所有 session 共享同一把上游凭证,因此单个 session 有可能占满该凭证的 +TPM。若需要按用户限额,应在上游网关侧按会话签发受限凭证(例如 LiteLLM 的 +per-key `max_budget`),而不是在平台侧另造一套记账。 + +另一条残余边界:**凭证跨 session 复用挡不住。** 两种网关模式都以 bearer 或 +SigV4 凭证认证,而没有任何机制能证明"发起方确实是该 session"——`llm-edge` 的 +session id 与 token 都来自请求头,AgentCore 的 STS 凭证也可被读出后在别处使用 +(受 session policy 与标签 Deny 约束,但不受"哪个 microVM"约束)。要真正绑定 +需要 microVM 侧的可信证明,AgentCore Runtime 目前不提供。危害有界:同一用户的 +两个 session 之间总额不变,实质风险是**跨模型档位提权**(用被路由到更贵模型的 +会话凭证)与归因失真。 每次调用实际走了哪个后端、哪个模型,记入调用台账(`backend:model` 字段), -路由变更前后可审计;`llm-edge` 另有结构化日志记录每次调用的 session、用户、 -模型、状态与 token 用量。 +路由变更前后可审计;`litellm` 模式下 `llm-edge` 另有结构化日志记录每次调用的 +session、用户、模型、状态与 token 用量。`agentcore_gateway` 模式下**网关不产出 +token 用量指标或日志**(已实测:开启 vended logs 后逐条日志只含路由决策与 OTel +关联字段,无 token、无会话维度),因此 token 级归因要么由上游网关提供,要么由 +一个 RESPONSE interceptor 自行累加。 --- @@ -503,7 +575,8 @@ grant 里的上游地址、密钥名、模型白名单,全部由后端在铸 |---|---|---| | Session 文件、对话历史 | Workspace S3 桶 | 公共访问全阻断、强制 SSL、静态加密、仅两个角色可访问 | | 平台记录(session/agent/schedule/台账/审计) | DynamoDB | 静态加密(默认 AWS 拥有密钥,可换 CMK) | -| LLM 网关 key | Secrets Manager | 加密存储;仅 `llm-edge` 任务角色可读,内核角色无此授权;从不写入镜像,也从不进入任何 session 容器 | +| LLM 网关 key(`litellm` 模式) | Secrets Manager | 加密存储;仅 `llm-edge` 任务角色可读,内核角色无此授权;从不写入镜像,也从不进入任何 session 容器 | +| 上游模型凭证(`agentcore_gateway` 模式) | AgentCore Gateway 的 token vault | 平台侧不存储、不可读;由网关在出站时注入。平台只持有一个可 `AssumeRole` 的 caller 角色 ARN | | 可选的第三方 MCP key、管理员口令 | Secrets Manager | 加密存储;按密钥名精确授权;容器/Lambda 启动时读取,从不写入镜像 | | 内核与平台日志 | CloudWatch Logs | 前缀级授权;平台自有日志组 7 天保留 | @@ -545,10 +618,22 @@ AgentCore 服务侧的数据(session 元数据、Memory 记录等)默认用 4. **headless 异步任务的网关凭证寿命较长**。异步运行最长到平台的 8 小时上限, 且没有续期通道,因此其 grant 的有效期与之匹配(9 小时)而不是 1 小时。影响 有限:headless 内核没有终端,且该 grant 从不进入 agent 子进程的环境(内核 - 只给子进程一个容器内本地令牌,见第 9.3 节)。 -5. **加密密钥默认为 AWS 托管**:合规要求 CMK 时,S3/DynamoDB/AgentCore 资源 + 只给子进程一个容器内本地令牌,见第 9.3 节)。`agentcore_gateway` 模式下这个 + 时长要求 caller 角色的 `MaxSessionDuration` 覆盖它,否则 `AssumeRole` 会拒绝 + 而调用 fail-closed。 +5. **`agentcore_gateway` 模式下,网关 interceptor 抛出的异常会原样回传给调用方** + —— 包含异常消息与**完整堆栈(含源码行)**,且与 `exceptionLevel` 的设置无关 + (已实测)。调用方就是租户的 microVM。因此 interceptor 必须自己捕获全部异常 + 并返回通用错误,绝不可让异常逸出;尤其是按 AWS 参考实现从 Secrets Manager + 取凭证的 interceptor,一旦抛错就可能把密钥名或内部 ARN 送到租户面前。 + 相关的正面结论:interceptor 失败时网关是 **fail-closed** 的(请求被拒,不会 + 放行),这一点也已实测。 +6. **`agentcore_gateway` 模式的单次请求有 15 分钟硬上限**(服务配额 + `Request timeout`,不可调整)。单次模型调用受此约束;流式响应实测 46 秒、 + 3980 帧无断流,距上限有充足余量。 +7. **加密密钥默认为 AWS 托管**:合规要求 CMK 时,S3/DynamoDB/AgentCore 资源 均可切换,代价是给相关角色增加精确限定到该密钥 ARN 的 KMS 权限。 -6. **所有平台角色都可加权限边界**,保证未来任何代码改动都不能越过边界扩权。 +8. **所有平台角色都可加权限边界**,保证未来任何代码改动都不能越过边界扩权。 另见第 9.5 节:用户可在自己 session 内直接调用 loopback 端口消耗自己的配额, 这是设计预期而非越权。 diff --git a/docs/user-guide.md b/docs/user-guide.md index 119f23f..bbcfdea 100644 --- a/docs/user-guide.md +++ b/docs/user-guide.md @@ -461,15 +461,18 @@ The **Model backends** card on the Governance page is the routing control plane for every model call — headless invocations and Dev Workbench sessions alike: -- **Two backends** — Amazon Bedrock (direct, via the kernel container's IAM - role; use `global.` cross-region inference profile IDs) and an - Anthropic-compatible **LLM gateway** (e.g. LiteLLM; the API key lives in - Secrets Manager, only its *name* is stored here, and only the `llm-edge` - service can read it — a session container never receives it). Each has an - enable switch and a model catalog that feeds the dropdowns elsewhere. - Gateway mode requires `llm-edge` to be deployed (`enable_llm_edge`); with it - missing, the platform refuses to route a session rather than falling back to - handing out the key. +- **Three backends** — Amazon Bedrock (direct, via the kernel container's IAM + role; use `global.` cross-region inference profile IDs); an + Anthropic-compatible **LLM gateway** (`litellm`, e.g. LiteLLM; the API key + lives in Secrets Manager, only its *name* is stored here, and only the + `llm-edge` service can read it — a session container never receives it); and + **AgentCore Gateway** (`agentcore_gateway`, where the upstream credential + lives in the gateway's own token vault, so there is no key name to store and + no broker service to deploy — a session gets STS credentials tagged with its + own id). Each has an enable switch and a model catalog that feeds the + dropdowns elsewhere. Either gateway mode refuses to route a session when its + prerequisite is missing (`enable_llm_edge`, or the caller role ARN) rather + than falling back to handing out a shared credential. - **Platform default** — which backend an agent uses when it doesn't pick one. - **Per-agent choice** — the Publish page's edit dialog has a *Model From 292d41b47322711f140c0c98d9aae10e0d1ca957 Mon Sep 17 00:00:00 2001 From: odinwang <48194321+odinwang@users.noreply.github.com> Date: Thu, 1 Oct 2026 18:09:58 +0800 Subject: [PATCH 3/3] agentcore_gateway: front LiteLLM, private path, no interceptor Repoint the gateway target from the bedrock-mantle connector to an inference *provider* target in front of LiteLLM, which is what this backend is for: the LiteLLM virtual key lives in an AgentCore Identity API-key credential provider and the gateway injects it, so neither the platform nor the container holds it and llm-edge is not needed. - Target: endpoint = LiteLLM, operations /v1/messages, /v1/chat/completions and /v1/responses with the models the virtual key permits. The gateway does no protocol translation (it matches path and model, swaps the key, relays body and SSE verbatim), so which model works on which path is LiteLLM's call; all four test models answered on all three. credentialPrefix "Bearer", no trailing space (the gateway adds it). allowedRequestHeaders keeps anthropic-beta, which LiteLLM accepts. - Gateway execution role: GetWorkloadAccessToken as well as GetResourceApiKey (the documented policy alone fails with "Failed to get workload identity token"), plus the provider's secret. - Private path: a top-level privateEndpoint.managedVpcResource places VPC Lattice resource-gateway ENIs in the LiteLLM VPC and connects to an internal ALB (routingDomain) while keeping the public cert name as SNI. Driven by LITELLM_ROUTING_DOMAIN / LITELLM_VPC_ID / LITELLM_SUBNET_IDS / LITELLM_LATTICE_SG_IDS. - No REQUEST interceptor: drop_params on the LiteLLM model entries strips Claude Code's first-party-only fields (context_management). An interceptor adds a Lambda hop per call and caps bodies at ~4.5 MB. - e2e: [multimodel] smoke over every model on the key; SSO reserved roles trusted via the account root; teardown waits for private-endpoint targets; LITELLM_ENDPOINT / LITELLM_CREDENTIAL_PROVIDER_ARN required. - deployment.md Option A2 rewritten for LiteLLM, including the 350-second Lattice idle timeout that makes long non-streaming calls hang. Verified end to end in ap-northeast-1 through the private endpoint: chain, multi-turn tool task, multi-model, failfast, scoping, revoke (8.6 s). Co-Authored-By: Claude Opus 5.5 (1M context) --- docs/architecture.md | 2 +- docs/deployment.md | 120 +++++--- scripts/e2e_agentcore_gateway.py | 459 +++++++++++++++++++------------ 3 files changed, 374 insertions(+), 207 deletions(-) diff --git a/docs/architecture.md b/docs/architecture.md index d008d48..3f7d1bf 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -261,7 +261,7 @@ Both kernels support three model backends: |---|---|---| | **Bedrock direct** | `CLAUDE_CODE_USE_BEDROCK=1` + cross-region inference profile (`global.` model ID prefix). Container IAM role; no key of any kind | Simplest path, and there is no model credential to leak | | **LLM gateway** (`litellm`) | Always per session, never a container default. The gateway key lives only in the `llm-edge` service; a kernel gets a session-scoped grant and reaches the gateway through it (`enable_llm_edge`) | Centralized model governance: allow-lists, budgets, cost attribution per team | -| **AgentCore Gateway** (`agentcore_gateway`) | Also always per session. The upstream provider credential lives in the gateway's own token vault, so there is no key on the platform side at all and no broker service to run. A kernel gets STS credentials tagged with its session id and SigV4-signs each request (`agentcore_gateway_caller_role_arn`) | The same governance goal with a managed hop instead of `llm-edge`, and per-session revocation expressed as an IAM Deny | +| **AgentCore Gateway** (`agentcore_gateway`) | Also always per session. The gateway sits in front of the same LiteLLM deployment Option A talks to, and the LiteLLM key lives in AgentCore Identity's token vault — so there is no key on the platform side at all and no broker service to run. A kernel gets STS credentials tagged with its session id and SigV4-signs each request (`agentcore_gateway_caller_role_arn`); LiteLLM keeps its multi-provider routing exactly as it does for every other client | The same governance goal with a managed hop instead of `llm-edge`, and per-session revocation expressed as an IAM Deny | Neither gateway mode has container-level configuration, on purpose. A session's user is root in its microVM and the headless kernel runs agent tools in a diff --git a/docs/deployment.md b/docs/deployment.md index a95d2ce..34913da 100644 --- a/docs/deployment.md +++ b/docs/deployment.md @@ -101,33 +101,94 @@ With `enable_llm_edge = false`, selecting the litellm backend makes the platform refuse the session with a 503 instead of falling back to handing a container the key. -### Option A2 — AgentCore Gateway +### Option A2 — AgentCore Gateway in front of LiteLLM + +Same promise as Option A with one fewer service to run: the LiteLLM API key +lives in AgentCore Identity's token vault, so there is no bearer secret on +the platform side and no `llm-edge` to deploy. A kernel receives STS +credentials tagged with its session id and SigV4-signs each request; the +gateway substitutes in the LiteLLM key on the way out. LiteLLM keeps its +multi-provider routing, cost accounting and model catalog exactly as it does +for every other client — this option changes only how our platform reaches +LiteLLM, not what LiteLLM does behind it. + +1. **Mint a LiteLLM virtual key** scoped to the models this backend will + serve. The Prod-LiteLLM master key stays where it is; the key stored in + the vault below is the virtual one, so a revocation there does not touch + any other LiteLLM client. -Same promise as Option A with one fewer service to run: the upstream provider -credential lives in the gateway's own token vault, so there is no key on the -platform side and no `llm-edge` to deploy. A kernel gets STS credentials tagged -with its session id and signs each request. + ```bash + curl -sS -X POST "$LITELLM_URL/key/generate" \ + -H "Authorization: Bearer $LITELLM_MASTER" \ + -H "Content-Type: application/json" \ + -d '{"models":["claude-haiku-4-5","claude-sonnet-5-5"], + "key_alias":"agentcore-gateway"}' | jq -r .key + ``` -1. Create the gateway and its inference target, and note the `/inference` URL. - Two settings are not optional: +2. **Store the key in AgentCore Identity.** In the AWS console: Amazon + Bedrock AgentCore → **Identity** → **API key credential providers** → + *Create*. Paste the virtual key. AWS returns a + `credentialProviderArn` of the form + `arn:aws:bedrock-agentcore:::token-vault/default/apikeycredentialprovider/` + — note it for step 4. + +3. **Create the gateway.** `protocolType=MCP`, `authorizerType=AWS_IAM`. Its + execution role needs only the token-vault reads that fetch the API key + (`bedrock-agentcore:GetWorkloadAccessToken` and `GetResourceApiKey`, plus + `secretsmanager:GetSecretValue` on the provider's secret); it does *not* + need any `bedrock*` grant, because the outbound call is HTTPS to LiteLLM + with a header-injected API key rather than a Bedrock service call. + +4. **Create the inference *provider* target** pointing at LiteLLM, with the + credential provider substituting the caller's `Authorization` header on + the way out (`credentialPrefix: "Bearer"`, no trailing space). Declare + each path the clients use (`/v1/messages`, `/v1/chat/completions`, + `/v1/responses`; the gateway accepts no others) and the models each + path may route to. The gateway does no protocol translation: it matches + path and `model`, swaps in the key and relays body and SSE stream + verbatim, so LiteLLM decides which model works on which path. - **`metadataConfiguration.allowedRequestHeaders`** must list at least - `content-type`, `anthropic-version` and `accept`. Left unset the gateway - relays whatever the caller sent — including the caller's own - `x-amz-security-token` — and Claude Code's prompt-caching `anthropic-beta` - value, which a strict upstream refuses. Once it *is* set, a connector - target stops relaying `content-type` unless it is listed. - - a **REQUEST interceptor** if the upstream rejects fields Claude Code sends - (`context_management` is one). The gateway forwards bodies verbatim, so an - interceptor is the only place outside the container to adapt them. It is - also where a per-tenant identity would be injected for upstream billing. - Interceptors must catch every exception themselves — an escaping one is - relayed to the caller with its stack trace. - -2. Create the role the backend assumes per session, trusted by the backend's - own role, granting `bedrock-agentcore:InvokeGateway` on the gateway ARN and - allowing `sts:TagSession`. Its `MaxSessionDuration` must cover the async - grant lifetime (9 h) if headless async runs use this backend. + `content-type`, `anthropic-version`, `anthropic-beta` and `accept`. + Left unset the gateway relays whatever the caller sent — including the + caller's own `x-amz-security-token`. Once set, `content-type` has to + be listed explicitly or the target stops relaying it and LiteLLM + answers 400. + - **Enable `drop_params: true` on the LiteLLM model entries** this + target routes to. Claude Code sends first-party-only fields + (`context_management` is the current one) that LiteLLM's Bedrock + adapter rejects otherwise. A REQUEST interceptor at the gateway could + strip them instead, but it adds a synchronous Lambda call per model + request and caps request bodies at Lambda's 6 MB invoke payload + (about 4.5 MB of JSON once base64-encoded), which a long session can + reach. + - **If LiteLLM is not publicly reachable**, add a top-level + `privateEndpoint.managedVpcResource` (sibling of + `targetConfiguration`): the gateway places VPC Lattice resource-gateway + ENIs in your subnets and connects to `routingDomain` (for example an + internal ALB's DNS name) while keeping the target endpoint's hostname + as TLS SNI, so the ALB needs a publicly trusted certificate for that + name. Lattice closes TCP connections idle for 350 seconds, so on this + path a non-streaming call whose response takes longer than that never + returns; clients must stream. + + Measured through a private endpoint (Tokyo, 2026-10): `claude-opus-5-5`, + `claude-sonnet-5-5`, `gpt-6-astra` and `kimi-k3` each answered on all + three paths, streaming and not; 20 concurrent calls and a 634-second SSE + stream went through; a non-streaming call that took 368 seconds never + returned. `GET /inference/v1/models` lists every declared model across the + gateway's targets as `/`, and a client may pin a target + that way in the `model` field. + + The exact `targetConfiguration` and `credentialProviderConfigurations` + payloads are in `scripts/e2e_agentcore_gateway.py`; that script provisions + all of the above against a live account. + +5. **Create the role the backend assumes per session**, trusted by the + backend's own role, granting `bedrock-agentcore:InvokeGateway` on the + gateway ARN and allowing `sts:TagSession`. Its `MaxSessionDuration` must + cover the async grant lifetime (9 h) if headless async runs use this + backend. ```hcl # no dedicated module yet — supply the role ARN you created @@ -137,7 +198,7 @@ with its session id and signs each request. PLATFORM_AGENTCORE_GATEWAY_CALLER_ROLE_ARN=arn:aws:iam:::role/ ``` -3. **Deny the kernel roles direct access to this gateway.** The kernel roles +6. **Deny the kernel roles direct access to this gateway.** The kernel roles already hold `bedrock-agentcore:InvokeGateway` on `gateway/*` so they can reach MCP tool gateways, and that wildcard would also cover the inference gateway — letting a root user in the microVM call it on the kernel role's own @@ -145,13 +206,10 @@ with its session id and signs each request. inference gateway's ARN to each kernel role; see permissions.md §3. Without this the per-session credential is decorative. -4. Point the backend at the gateway in the portal: **Governance → Model - backends → agentcore_gateway**, setting `base_url` to the `/inference` URL - and the model catalog. There is no `secret_name` for this backend. - -`scripts/e2e_agentcore_gateway.py` provisions all of the above against a live -account and asserts the chain end to end, including per-session revocation; it -is the fastest way to see a working configuration. +7. **Point the backend at the gateway in the portal**: *Governance → Model + backends → agentcore_gateway*, setting `base_url` to the `/inference` URL + and the model catalog to whichever LiteLLM aliases the virtual key + authorises. There is no `secret_name` for this backend. With the caller role ARN unset, selecting this backend makes the platform refuse the session rather than fall back to a shared credential. diff --git a/scripts/e2e_agentcore_gateway.py b/scripts/e2e_agentcore_gateway.py index ad3415a..224e648 100755 --- a/scripts/e2e_agentcore_gateway.py +++ b/scripts/e2e_agentcore_gateway.py @@ -8,8 +8,8 @@ Claude Code subprocess per-invocation loopback token -> llm_shim (kernel) SigV4 over per-session STS credentials -> AgentCore Gateway IAM authorizer + session policy - -> bedrock-mantle gateway's own execution role - -> Anthropic model + -> LiteLLM /v1/messages API key injected from the token vault + -> the upstream model LiteLLM routes to Nothing in the chain is stubbed. The only simulation is the microVM itself: the shim runs in this process instead of inside AgentCore Runtime, which is @@ -32,12 +32,19 @@ Usage: python3 scripts/e2e_agentcore_gateway.py [--region us-east-1] [--keep] +Configuration is by environment: LITELLM_ENDPOINT and +LITELLM_CREDENTIAL_PROVIDER_ARN are required; LITELLM_ALLOWED_MODELS, +LITELLM_CHAIN_MODEL and LITELLM_PATHS have defaults. For a LiteLLM that is not reachable from +the internet, set LITELLM_ROUTING_DOMAIN (an internal ALB's DNS name), +LITELLM_VPC_ID, LITELLM_SUBNET_IDS and LITELLM_LATTICE_SG_IDS: the target then +gets a managed VPC Lattice private endpoint and the gateway never leaves AWS's +network on the way to LiteLLM. + Requires: credentials able to create IAM roles, an AgentCore Gateway and a DynamoDB table, plus the `claude` CLI on PATH. Everything it creates is torn down unless --keep is passed. """ import argparse -import io import json import os import secrets @@ -47,7 +54,6 @@ import time import urllib.error import urllib.request -import zipfile import boto3 from botocore.exceptions import ClientError @@ -56,52 +62,77 @@ FAIL: list[str] = [] SUFFIX = secrets.token_hex(4) -# A gateway REQUEST interceptor. V4 needs one anyway — that is where the -# tenant identity used for LiteLLM's per-end-user billing gets injected from a -# verified JWT claim, which the client cannot forge. Body adaptation rides -# along in the same function: Claude Code sends first-party-only fields that -# the bedrock-mantle passthrough refuses outright ("400 context_management: -# Extra inputs are not permitted"), and the gateway forwards bodies verbatim, -# so there is nowhere else outside the container to drop them. A LiteLLM -# upstream configured with drop_params absorbs the same class of mismatch. -INTERCEPTOR_SRC = ''' -import base64, json - -# Fields Claude Code sends that the strict Anthropic-on-Bedrock passthrough -# rejects. Dropping them is lossy (prompt caching, server-side context -# management) and deliberately explicit rather than a blanket filter. -DROP_TOP_LEVEL = ("context_management",) - -def lambda_handler(event, context): - req = event["http"]["gatewayRequest"] - raw = base64.b64decode(req.get("body") or b"") - try: - body = json.loads(raw) if raw else {} - except Exception: - return {"interceptorOutputVersion": "1.0", - "http": {"transformedGatewayRequest": req}} - dropped = [k for k in DROP_TOP_LEVEL if k in body] - for k in dropped: - body.pop(k, None) - if dropped: - print("dropped unsupported fields: " + ",".join(dropped)) - new = json.dumps(body).encode() - # Only the fields being changed go back. Echoing the whole gatewayRequest - # (httpMethod/path included) is rejected with "Received invalid response - # from interceptor", and the body must stay base64 — a JSON object is - # rejected the same way, even though the MCP examples in the docs use one. - hdrs = dict(req.get("headers") or {}) - hdrs["Content-Length"] = str(len(new)) - return { - "interceptorOutputVersion": "1.0", - "http": { - "transformedGatewayRequest": { - "headers": hdrs, - "body": base64.b64encode(new).decode(), - } - }, - } -''' +# The gateway target routes to this LiteLLM deployment. Overridable so a fork +# can point at a staging instance, but the default is the one Prod-LiteLLM +# already runs on this account. +# The gateway target routes to this LiteLLM deployment, and this API-key +# credential provider holds the LiteLLM virtual key. Both are required. The +# provider is created out of band (AWS Console: AgentCore -> Identity -> API +# key credential providers) so a live key never lands in this script or its +# logs; the e2e's job is to prove the provider ARN and the target wiring work +# end to end. +LITELLM_ENDPOINT = os.environ.get("LITELLM_ENDPOINT", "") +LITELLM_CREDENTIAL_PROVIDER_ARN = os.environ.get("LITELLM_CREDENTIAL_PROVIDER_ARN", "") + +# The list of models the virtual key permits. The gateway target refuses any +# model outside this set at its own layer, and LiteLLM refuses it at the key +# layer — either check is enough on its own, both layers are here on purpose +# so a misconfiguration surfaces immediately rather than at the far end. +LITELLM_ALLOWED_MODELS = [ + m.strip() + for m in os.environ.get( + "LITELLM_ALLOWED_MODELS", + "claude-opus-5-5,claude-sonnet-5-5,gpt-6-astra,kimi-k3", + ).split(",") + if m.strip() +] + +# Private path to LiteLLM. When LITELLM_ROUTING_DOMAIN is set the target gets a +# managed VPC Lattice private endpoint: AgentCore places resource-gateway ENIs +# in these subnets and connects to the routing domain (an internal ALB), while +# TLS SNI / Host stay on LITELLM_ENDPOINT's name so the ALB's public cert +# matches. Unset, the target dials LITELLM_ENDPOINT over the internet. +LITELLM_ROUTING_DOMAIN = os.environ.get("LITELLM_ROUTING_DOMAIN", "") +LITELLM_VPC_ID = os.environ.get("LITELLM_VPC_ID", "") +LITELLM_SUBNET_IDS = [ + s.strip() for s in os.environ.get("LITELLM_SUBNET_IDS", "").split(",") if s.strip() +] +LITELLM_LATTICE_SG_IDS = [ + s.strip() for s in os.environ.get("LITELLM_LATTICE_SG_IDS", "").split(",") if s.strip() +] + +# Inbound paths the target accepts. The gateway does no protocol translation: +# it matches path + model, swaps in the API key and relays the body (and the +# SSE stream) verbatim, so whether a given model works on a given path is +# LiteLLM's call. Every allowed model is declared on every path here and the +# [protocol-matrix] probe reports what LiteLLM actually serves. +LITELLM_PATHS = [ + p.strip() + for p in os.environ.get( + "LITELLM_PATHS", "/v1/messages,/v1/chat/completions,/v1/responses" + ).split(",") + if p.strip() +] + +# The one used for the [chain] / multi-turn / revoke / failfast steps, which +# drive the real Claude Code CLI and therefore need an Anthropic-native model. +LITELLM_CHAIN_MODEL = os.environ.get("LITELLM_CHAIN_MODEL", "claude-sonnet-5-5") + +# Anything in the allowed set but not the chain model gets exercised by the +# lightweight [multimodel] smoke: one /v1/messages "hi" per model, no CLI, no +# tools. That is where any Anthropic-to-OpenAI protocol translation LiteLLM +# does for gpt-* / kimi-* is stressed. +LITELLM_SMOKE_MODELS = [ + m for m in LITELLM_ALLOWED_MODELS if m != LITELLM_CHAIN_MODEL +] + +# No REQUEST interceptor. Claude Code sends first-party-only fields +# (context_management today) that LiteLLM's Bedrock adapter rejects; the +# LiteLLM model entries carry `drop_params: true` so LiteLLM strips them. An +# interceptor Lambda could do the same at the gateway, but every model call +# then pays a synchronous Lambda hop and the request body is capped by +# Lambda's 6 MB invoke payload (about 4.5 MB of JSON after base64), which a +# long Claude Code session can reach. def check(label: str, ok: bool, detail: str = "") -> None: @@ -126,16 +157,11 @@ def __init__(self, region: str) -> None: self.sts = boto3.client("sts") self.ddb = boto3.client("dynamodb", region_name=region) self.acc = boto3.client("bedrock-agentcore-control", region_name=region) - self.lam = boto3.client("lambda", region_name=region) - self.logs = boto3.client("logs", region_name=region) self.account = self.sts.get_caller_identity()["Account"] self.gw_role = f"e2e-acgw-exec-{SUFFIX}" self.caller_role = f"e2e-acgw-caller-{SUFFIX}" - self.icept_role = f"e2e-acgw-icept-{SUFFIX}" - self.icept_fn = f"e2e-acgw-icept-{SUFFIX}" self.table = f"e2e-acgw-{SUFFIX}" self.gateways: list[str] = [] - self.icept_arn = "" # -- helpers ---------------------------------------------------------- @@ -143,14 +169,77 @@ def _caller_principal(self) -> str: """The role this script runs as, so the caller role can trust it. In production the trusting principal is the backend's IRSA role; here - it is whatever identity is running the test. + it is whatever identity is running the test. SSO reserved roles + (``arn:aws:iam:::role/AWSReservedSSO_*``) cannot be used as an + AssumeRole *Principal* — IAM refuses the trust policy with + MalformedPolicyDocument — so for those we widen to the account root. + That is safe here because the role lives for the length of one run. """ arn = self.sts.get_caller_identity()["Arn"] if ":assumed-role/" in arn: name = arn.split(":assumed-role/")[1].split("/")[0] + if name.startswith("AWSReservedSSO_"): + return f"arn:aws:iam::{self.account}:root" return f"arn:aws:iam::{self.account}:role/{name}" return arn + def _gw_role_policy(self) -> dict: + """Policy for the gateway execution role. + + The one capability needed is retrieving the outbound API key (the + LiteLLM virtual key) from AgentCore Identity's token vault during a + request; without the GetResourceApiKey / secretsmanager + grants the gateway returns 400 "Failed to fetch outbound api key. + Failed to get workload identity token" and every call — chain, + scoping, revoke — surfaces the same error. + """ + vault = f"arn:aws:bedrock-agentcore:{self.region}:{self.account}" + provider_name = LITELLM_CREDENTIAL_PROVIDER_ARN.rsplit("/", 1)[-1] + return { + "Version": "2012-10-17", + "Statement": [ + { + # Two actions, one resource set. GetWorkloadAccessToken + # is the first hop the gateway makes: it exchanges its + # execution role for a workload-scoped token, which it + # then presents to GetResourceApiKey to unwrap the + # LiteLLM key. Without the first action the docs' policy + # is not sufficient — this is the exact failure mode we + # observed as "Failed to get workload identity token" + # despite GetResourceApiKey being granted. + "Sid": "AgentCoreApiKeyTokenVaultDefault", + "Effect": "Allow", + "Action": [ + "bedrock-agentcore:GetWorkloadAccessToken", + "bedrock-agentcore:GetResourceApiKey", + ], + "Resource": [ + f"{vault}:token-vault/default", + f"{vault}:workload-identity-directory/default", + f"{vault}:workload-identity-directory/default/" + "workload-identity/*", + ], + }, + { + "Sid": "AgentCoreApiKeyTokenVaultPerKey", + "Effect": "Allow", + "Action": "bedrock-agentcore:GetResourceApiKey", + "Resource": LITELLM_CREDENTIAL_PROVIDER_ARN, + }, + { + "Sid": "AgentCoreApiKeySecret", + "Effect": "Allow", + "Action": "secretsmanager:GetSecretValue", + "Resource": ( + f"arn:aws:secretsmanager:{self.region}:" + f"{self.account}:secret:" + f"bedrock-agentcore-identity!default/apikey/" + f"{provider_name}-*" + ), + }, + ], + } + def _role(self, name: str, trust: dict, policy: dict, max_session: int = 3600) -> str: try: arn = self.iam.create_role( @@ -187,56 +276,6 @@ def create(self) -> None: self.ddb.get_waiter("table_exists").wait(TableName=self.table) print(f" table {self.table}") - # -- interceptor Lambda (created first: the gateway references it) --- - lrole = self._role( - self.icept_role, - { - "Version": "2012-10-17", - "Statement": [ - { - "Effect": "Allow", - "Principal": {"Service": "lambda.amazonaws.com"}, - "Action": "sts:AssumeRole", - } - ], - }, - { - "Version": "2012-10-17", - "Statement": [ - { - "Effect": "Allow", - "Action": [ - "logs:CreateLogGroup", - "logs:CreateLogStream", - "logs:PutLogEvents", - ], - "Resource": "*", - } - ], - }, - ) - time.sleep(12) - buf = io.BytesIO() - with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as z: - z.writestr("lambda_function.py", INTERCEPTOR_SRC) - self.icept_arn = self.lam.create_function( - FunctionName=self.icept_fn, - Runtime="python3.12", - Role=lrole, - Handler="lambda_function.lambda_handler", - Code={"ZipFile": buf.getvalue()}, - Timeout=15, - MemorySize=256, - Description="temporary: e2e_agentcore_gateway request interceptor", - )["FunctionArn"] - for _ in range(20): - if self.lam.get_function(FunctionName=self.icept_fn)["Configuration"][ - "State" - ] == "Active": - break - time.sleep(3) - print(f" lambda {self.icept_fn} (REQUEST interceptor)") - gw_trust = { "Version": "2012-10-17", "Statement": [ @@ -248,45 +287,16 @@ def create(self) -> None: } ], } - gw_arn_role = self._role( - self.gw_role, - gw_trust, - { - "Version": "2012-10-17", - "Statement": [ - { - "Effect": "Allow", - "Action": ["bedrock:*", "bedrock-mantle:*"], - "Resource": "*", - }, - { - # Scoped to the one interceptor, per the gateway - # interceptor security guidance. - "Effect": "Allow", - "Action": "lambda:InvokeFunction", - "Resource": self.icept_arn, - }, - ], - }, - ) + # The gateway execution role must reach AgentCore Identity's token + # vault so the LiteLLM API key can be fetched and injected into + # outbound requests. + gw_policy = self._gw_role_policy() + gw_arn_role = self._role(self.gw_role, gw_trust, gw_policy) print(f" role {self.gw_role} (gateway execution)") # Two gateways: the session credential is scoped to the first, and the # second exists only to prove the scoping actually bites. for label in ("primary", "decoy"): - extra: dict = {} - if label == "primary": - extra["interceptorConfigurations"] = [ - { - "interceptor": {"lambda": {"arn": self.icept_arn}}, - "interceptionPoints": ["REQUEST"], - # Required in the billing configuration so the - # interceptor can read the verified JWT it derives the - # tenant identity from; kept on here so the shape under - # test matches the one V4 actually deploys. - "inputConfiguration": {"passRequestHeaders": True}, - } - ] g = self.acc.create_gateway( name=f"e2e-{label}-{SUFFIX}", roleArn=gw_arn_role, @@ -294,7 +304,6 @@ def create(self) -> None: authorizerType="AWS_IAM", exceptionLevel="DEBUG", description="temporary: e2e_agentcore_gateway", - **extra, ) self.gateways.append(g["gatewayId"]) print(f" gateway {g['gatewayId']} ({label})") @@ -303,44 +312,84 @@ def create(self) -> None: if self.acc.get_gateway(gatewayIdentifier=gid)["status"] != "CREATING": break time.sleep(4) - # IAM propagation before the connector's model-discovery call. + # IAM propagation before the gateway's first outbound. time.sleep(15) for gid in self.gateways: t = self.acc.create_gateway_target( gatewayIdentifier=gid, - name="bedrock", + name="litellm", + # An inference *provider* target rather than the built-in + # bedrock-mantle connector: the gateway forwards Anthropic + # Messages requests to our LiteLLM deployment, and LiteLLM + # keeps its multi-provider routing (Bedrock Anthropic today, + # OpenAI / Gemini tomorrow) exactly as it does for every other + # client. What this backend gives up in return is the + # llm-edge pod — one fewer service on the platform side, no + # provider key living anywhere in our EKS. targetConfiguration={ - "inference": {"connector": {"source": {"connectorId": "bedrock-mantle"}}} + "inference": { + "provider": { + "endpoint": LITELLM_ENDPOINT, + "operations": [ + { + "path": path, + # The models named here are what a caller + # is *allowed* to route to; the shim + # additionally fail-fasts against the + # per-session allowlist minted by the + # backend. + "models": [ + {"model": m} + for m in LITELLM_ALLOWED_MODELS + ], + } + for path in LITELLM_PATHS + ], + } + } }, + # An API_KEY provider held in AgentCore Identity's token vault + # substitutes the caller's Authorization header on the way + # out, so a container never sees the LiteLLM key. The + # provider is created out of band (Console) and referenced by + # ARN; see LITELLM_CREDENTIAL_PROVIDER_ARN at the top. + credentialProviderConfigurations=[ + { + "credentialProviderType": "API_KEY", + "credentialProvider": { + "apiKeyCredentialProvider": { + "providerArn": LITELLM_CREDENTIAL_PROVIDER_ARN, + "credentialLocation": "HEADER", + "credentialParameterName": "Authorization", + # No trailing space: the gateway inserts the + # separator itself. "Bearer " goes out as + # "Bearer " and LiteLLM rejects it. + "credentialPrefix": "Bearer", + } + }, + } + ], # Curating headers here is the AgentCore-side counterpart of # llm-edge's FORWARD_HEADERS allowlist, and it is not optional: - # - # - left unset, the gateway relays whatever the caller sent, - # including the caller's own x-amz-security-token; and - # - Claude Code advertises a prompt-caching beta that this - # upstream rejects outright ("400 Unexpected value(s) - # `prompt-caching-scope-...` for the `anthropic-beta` - # header"), so anthropic-beta has to be dropped for the - # bedrock-mantle connector. A LiteLLM upstream accepts it, - # and would be allowlisted instead. - # - # content-type has to be listed explicitly: once - # allowedRequestHeaders is set, a connector target stops - # relaying it and the upstream answers "400 Expected request - # with `Content-Type: application/json`". (A provider target - # supplies it itself, which is why this is easy to miss.) + # left unset the gateway relays whatever the caller sent, + # including the caller's own x-amz-security-token. LiteLLM + # accepts anthropic-beta (so Claude Code's prompt-caching + # advertisement passes through), unlike the bedrock-mantle + # connector, which is the one payoff of not routing through + # that connector. content-type has to be listed explicitly: + # once allowedRequestHeaders is set, a connector-style target + # stops relaying it. metadataConfiguration={ "allowedRequestHeaders": [ "content-type", "anthropic-version", + "anthropic-beta", "accept", ] }, - credentialProviderConfigurations=[ - {"credentialProviderType": "GATEWAY_IAM_ROLE"} - ], + **self._private_endpoint(), ) - for _ in range(50): + for _ in range(90): st = self.acc.get_gateway_target( gatewayIdentifier=gid, targetId=t["targetId"] ) @@ -348,6 +397,8 @@ def create(self) -> None: break time.sleep(4) print(f" target {gid} -> {st['status']} {st.get('statusReasons', '')}") + for r in st.get("privateEndpointManagedResources") or []: + print(f" private endpoint {r.get('domain')} via {r.get('resourceGatewayArn')}") if st["status"] != "READY": raise SystemExit("inference target did not become READY") @@ -383,6 +434,25 @@ def create(self) -> None: # scoping assertion tests the session policy and not this role. time.sleep(12) + def _private_endpoint(self) -> dict: + """Top-level privateEndpoint kwarg for create_gateway_target, if any. + + It sits beside targetConfiguration, not inside inference.provider. + The first target creation in an account also creates the + AWSServiceRoleForBedrockAgentCoreGatewayNetwork service-linked role. + """ + if not LITELLM_ROUTING_DOMAIN: + return {} + mvr = { + "vpcIdentifier": LITELLM_VPC_ID, + "subnetIds": LITELLM_SUBNET_IDS, + "endpointIpAddressType": "IPV4", + "routingDomain": LITELLM_ROUTING_DOMAIN, + } + if LITELLM_LATTICE_SG_IDS: + mvr["securityGroupIds"] = LITELLM_LATTICE_SG_IDS + return {"privateEndpoint": {"managedVpcResource": mvr}} + def base_url(self, which: int = 0) -> str: gid = self.gateways[which] return ( @@ -402,21 +472,20 @@ def destroy(self) -> None: self.acc.delete_gateway_target( gatewayIdentifier=gid, targetId=t["targetId"] ) - time.sleep(6) + # A target with a private endpoint takes tens of seconds to + # release its Lattice association; the gateway refuses + # deletion until the target is gone. + for _ in range(40): + if not self.acc.list_gateway_targets( + gatewayIdentifier=gid + ).get("items"): + break + time.sleep(5) self.acc.delete_gateway(gatewayIdentifier=gid) print(f" deleted gateway {gid}") except Exception as e: # noqa: BLE001 print(f" gateway {gid}: {str(e)[:100]}") - try: - self.lam.delete_function(FunctionName=self.icept_fn) - print(f" deleted lambda {self.icept_fn}") - except Exception as e: # noqa: BLE001 - print(f" lambda: {str(e)[:100]}") - try: - self.logs.delete_log_group(logGroupName=f"/aws/lambda/{self.icept_fn}") - except Exception: # noqa: BLE001 - pass - for name in (self.caller_role, self.gw_role, self.icept_role): + for name in (self.caller_role, self.gw_role): for pol in self.iam.list_role_policies(RoleName=name).get( "PolicyNames", [] ): @@ -442,10 +511,14 @@ def destroy(self) -> None: def main() -> int: ap = argparse.ArgumentParser() ap.add_argument("--region", default=os.environ.get("AWS_REGION", "us-east-1")) - ap.add_argument("--model", default="claude-haiku-4-5") + ap.add_argument("--model", default=LITELLM_CHAIN_MODEL) ap.add_argument("--keep", action="store_true", help="leave AWS resources behind") args = ap.parse_args() + if not (LITELLM_ENDPOINT and LITELLM_CREDENTIAL_PROVIDER_ARN): + print("set LITELLM_ENDPOINT (https://...) and LITELLM_CREDENTIAL_PROVIDER_ARN") + return 2 + if not any( os.access(os.path.join(p, "claude"), os.X_OK) for p in os.environ.get("PATH", "").split(os.pathsep) @@ -654,6 +727,42 @@ def _blocks(m: dict) -> list: ",".join(paths), ) + if LITELLM_SMOKE_MODELS: + step( + "[multimodel] one /v1/messages 'hi' per non-chain model on the " + "virtual key — sanity-checks LiteLLM's protocol handling for " + "each upstream, no CLI, no tools" + ) + # A fresh grant whose allowed_models covers every model the vk + # permits, so the shim's fail-fast doesn't shadow LiteLLM's + # protocol result. mint_agentcore accepts an explicit models list. + sid_mm = f"ses-{secrets.token_hex(20)}" + grant_mm = svc.mint_agentcore( + sid_mm, "alice", spec, team="team-alpha", + models=LITELLM_ALLOWED_MODELS, + ) + token_mm = llm_shim.register(grant_mm) if grant_mm else "" + # A 200 confirms end-to-end for that model; a non-200 is a data + # point about how Anthropic Messages format survives LiteLLM's + # translation to the upstream — for gpt-* / kimi-* against + # /v1/messages this is not guaranteed to be lossless. Reported + # inline; the [chain] step is what a green run is judged on. + for m in LITELLM_SMOKE_MODELS: + if not token_mm: + print(f" {m}: could not mint a smoke grant") + continue + code, body = shim_call( + token_mm, + { + "model": m, + "max_tokens": 8, + "messages": [{"role": "user", "content": "hi"}], + }, + ) + verdict = "ok" if code == 200 else "fail" + print(f" {verdict:>4} {m:20} HTTP {code} {body[:90]}") + svc.revoke_agentcore(sid_mm) + step("[failfast] shim refuses a model this session was not routed to") code, body = shim_call(local_token, {"model": "claude-opus-4-7", "max_tokens": 8, "messages": [{"role": "user", "content": "hi"}]})