Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions EXTENDING.md
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,7 @@ with an upstream update.
| Which layers to create | `enable_runtime`, `enable_portal`, `enable_llm_edge`, `enable_team_auth`, `enable_team_demo`, `enable_mcp_hub_demo` | [terraform/variables.tf](terraform/variables.tf) — the quick start uses them to stage the deploy |
| Bedrock direct mode | `CLAUDE_CODE_USE_BEDROCK=1` | runtime env; no key involved |
| Your LLM gateway | `enable_llm_edge = true` + the backend's `base_url` in Governance → Model backends | key in Secrets Manager, read only by `llm-edge`; kernels get a per-session grant, never the key |
| An AgentCore Gateway instead | `PLATFORM_AGENTCORE_GATEWAY_CALLER_ROLE_ARN` + the backend's `base_url` (the gateway's `/inference` URL) | no key on the platform side at all and no `llm-edge` to run; kernels get session-tagged STS credentials and SigV4-sign. Requires denying the kernel roles direct `InvokeGateway` on that gateway — see [docs/permissions.md](docs/permissions.md) |
| Backend runtime settings (table, buckets, ARNs, Cognito, CORS, admin tiers) | `PLATFORM_*` env vars | [backend/app/config.py](backend/app/config.py) — every field there is `PLATFORM_<FIELD>` (`env_prefix`); `backend/.env.example` for local runs |

Terraform state is yours: `terraform/backend.tf.example` shows the remote-state
Expand Down
5 changes: 3 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -151,9 +151,10 @@ cp terraform.tfvars.example terraform.tfvars # then edit
terraform init
terraform apply -var enable_runtime=false -var enable_portal=false

# 2. Store your LLM gateway key (skip if using Bedrock direct).
# 2. Store your LLM gateway key (skip for Bedrock direct or AgentCore Gateway).
# Readable only by the llm-edge service; it never enters a kernel container.
# Gateway mode also needs enable_llm_edge=true — see docs/deployment.md §2.
# Gateway mode also needs enable_llm_edge=true — see docs/deployment.md §2,
# which also covers the agentcore_gateway backend (no key on this side).
aws secretsmanager put-secret-value \
--secret-id agent-platform/llm-gateway-key \
--secret-string '{"api_key":"sk-..."}'
Expand Down
17 changes: 10 additions & 7 deletions backend/app/api/sessions.py
Original file line number Diff line number Diff line change
Expand Up @@ -166,20 +166,23 @@ def connect(session_id: str, user: str = Depends(get_current_user)):
spec = model_config_service.resolve(
item.get("model_backend", ""), item.get("model", "")
)
if spec and spec.get("backend") == "gateway":
# The gateway key stays in llm-edge. The kernel gets an
# endpoint plus a session-scoped token, and the routing fields
# are stripped from what the container sees: the edge re-reads
# them from the grant, so a container has nothing to forge.
if spec and spec.get("backend") in ("gateway", "agentcore_gateway"):
# No upstream credential reaches the container in either mode:
# litellm keeps the key in llm-edge, AgentCore keeps it in the
# gateway's token vault. The kernel gets an endpoint plus a
# session-scoped credential, and the routing fields are
# stripped from what the container sees.
creds = llm_credentials_service.mint(
item["runtime_session_id"], user, spec
)
if not creds:
raise HTTPException(
status_code=503,
detail=(
"gateway model routing is unavailable: the llm-edge "
"service is not deployed (set enable_llm_edge)"
"gateway model routing is unavailable: deploy "
"llm-edge (enable_llm_edge) for the litellm "
"backend, or set the AgentCore gateway caller role "
"for the agentcore_gateway backend"
),
)
config["llm_credentials"] = creds
Expand Down
12 changes: 12 additions & 0 deletions backend/app/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,18 @@ class Settings(BaseSettings):
# root in.
llm_edge_url: str = ""

# Role the backend assumes once per session to mint the kernel's AgentCore
# Gateway credentials, tagged with that session's runtime id. Same shape as
# workspace_access_role_arn: the kernel role itself has no InvokeGateway
# grant, so a container cannot reach the gateway on its own identity.
#
# The session tag is what makes revocation per-session: ending a session
# adds a Deny conditioned on aws:PrincipalTag/session_id, which stops that
# session's credentials without touching any other live session's.
# Empty = the agentcore_gateway model backend is unavailable and the
# backend refuses it rather than falling back to a shared credential.
agentcore_gateway_caller_role_arn: str = ""

# EventBridge Scheduler wiring (outputs of the PortalStack). When all of
# group/lambda/role are set, the scheduler runs in "eventbridge" mode:
# each platform schedule is mirrored to an EventBridge Scheduler schedule
Expand Down
8 changes: 4 additions & 4 deletions backend/app/models/schemas.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,7 +9,7 @@ class SessionCreateRequest(BaseModel):
mcp_server_ids: list[str] = Field(default_factory=list, max_length=10)
skill_ids: list[str] = Field(default_factory=list, max_length=10)
# "" = platform default backend; otherwise a backend from the model config
model_backend: str = Field(default="", pattern="^(|bedrock|litellm)$")
model_backend: str = Field(default="", pattern="^(|bedrock|litellm|agentcore_gateway)$")
model: str = Field(default="", max_length=200)
# AgentCore Runtime platform version; "" = deployment default
platform_version: str = Field(default="", pattern="^(|V1|V2)$")
Expand Down Expand Up @@ -127,7 +127,7 @@ class AgentPublishRequest(BaseModel):
skill_names: list[str] = Field(default_factory=list, max_length=10)
memory_id: str = ""
# "" = platform default backend; otherwise a backend from the model config
model_backend: str = Field(default="", pattern="^(|bedrock|litellm)$")
model_backend: str = Field(default="", pattern="^(|bedrock|litellm|agentcore_gateway)$")
model: str = Field(default="", max_length=200)
# AgentCore Runtime platform version the agent runs on; "" = default.
# Pipelines, schedules and channels calling the agent inherit it.
Expand Down Expand Up @@ -216,12 +216,12 @@ class ModelBackendPatch(BaseModel):


class ModelConfigUpdate(BaseModel):
default_backend: str | None = Field(default=None, pattern="^(bedrock|litellm)$")
default_backend: str | None = Field(default=None, pattern="^(bedrock|litellm|agentcore_gateway)$")
backends: dict[str, ModelBackendPatch] | None = None


class ModelTestRequest(BaseModel):
backend: str = Field(pattern="^(bedrock|litellm)$")
backend: str = Field(pattern="^(bedrock|litellm|agentcore_gateway)$")
model: str = Field(default="", max_length=200)


Expand Down
24 changes: 15 additions & 9 deletions backend/app/services/kernel_service.py
Original file line number Diff line number Diff line change
Expand Up @@ -162,12 +162,17 @@ def invoke_sdk_kernel(
payload["memory"] = memory
if model:
# per-invocation model routing (see model_config_service.resolve)
if model.get("backend") == "gateway":
# The gateway key stays in llm-edge. Mint a grant scoped to this
# invocation's session and strip the routing fields, so the
# kernel receives an endpoint and a token instead of a key it
# could fetch itself. An async run has no refresh channel and
# may execute for hours, so its grant is given matching life.
if model.get("backend") in ("gateway", "agentcore_gateway"):
# No upstream credential reaches the kernel either way: the
# litellm path keeps the key in llm-edge, the AgentCore path
# keeps it in the gateway's token vault. Mint a grant scoped to
# this invocation's session and strip the routing fields, so
# the kernel receives an endpoint plus a session credential
# rather than something it could spend elsewhere. An async run
# has no refresh channel and may execute for hours, so its
# grant is given matching life — for the AgentCore path that
# requires the caller role's MaxSessionDuration to cover it,
# otherwise AssumeRole refuses and this call fails closed.
creds = llm_credentials_service.mint(
sid,
user,
Expand All @@ -180,9 +185,10 @@ def invoke_sdk_kernel(
"result": "",
"raw": {
"error": (
"gateway model routing is unavailable: the "
"llm-edge service is not deployed "
"(set enable_llm_edge)"
"gateway model routing is unavailable: deploy "
"llm-edge (enable_llm_edge) for the litellm "
"backend, or set the AgentCore gateway caller "
"role for the agentcore_gateway backend"
)
},
}
Expand Down
Loading