Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
25 changes: 25 additions & 0 deletions config/litellm.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
# LiteLLM proxy configuration: the one place an approved external provider is
# named. The gateway reaches providers only through this proxy, so provider
# credentials never enter the gateway and adding a provider never changes it.
#
# Every alias under model_list must match a non-local model card in
# config/registry.yaml; a test enforces it. Keys are read from the environment,
# which the cluster secret manager populates. Nothing secret is committed here.
model_list:
- model_name: approved-external-fallback
litellm_params:
model: anthropic/claude-sonnet-5-5
api_key: os.environ/EXTERNAL_PROVIDER_API_KEY

general_settings:
master_key: os.environ/LITELLM_MASTER_KEY

litellm_settings:
# Only public requests may reach this proxy, but prompts are still kept out
# of its logs: the gateway's redacted traces are the record of a request.
turn_off_message_logging: true
# The gateway owns retries and fallback, so a failure is reported once
# rather than retried here and again upstream.
num_retries: 0
request_timeout: 60
drop_params: true
156 changes: 156 additions & 0 deletions deploy/kubernetes/external-provider.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,156 @@
# LiteLLM proxy for the approved external fallback. It is the only workload in
# the namespace allowed to reach the internet, and only the gateway may call it.
apiVersion: v1
kind: ConfigMap
metadata:
name: litellm-config
namespace: llm-routing
data:
# Kept identical to config/litellm.yaml; a test fails if the two drift.
config.yaml: |
model_list:
- model_name: approved-external-fallback
litellm_params:
model: anthropic/claude-sonnet-5-5
api_key: os.environ/EXTERNAL_PROVIDER_API_KEY
general_settings:
master_key: os.environ/LITELLM_MASTER_KEY
litellm_settings:
turn_off_message_logging: true
num_retries: 0
request_timeout: 60
drop_params: true
---
apiVersion: apps/v1
kind: Deployment
metadata:
name: litellm-proxy
namespace: llm-routing
labels:
app.kubernetes.io/name: litellm-proxy
app.kubernetes.io/part-of: local-llm-router
spec:
replicas: 2
selector:
matchLabels:
app.kubernetes.io/name: litellm-proxy
template:
metadata:
labels:
app.kubernetes.io/name: litellm-proxy
app.kubernetes.io/part-of: local-llm-router
spec:
securityContext:
runAsNonRoot: true
runAsUser: 10001
seccompProfile:
type: RuntimeDefault
containers:
- name: litellm
image: ghcr.io/berriai/litellm@sha256:REPLACE_ME
imagePullPolicy: IfNotPresent
args: ["--config", "/etc/litellm/config.yaml", "--port", "4000"]
ports:
- name: http
containerPort: 4000
env:
- name: LITELLM_MASTER_KEY
valueFrom:
secretKeyRef:
name: llm-gateway-credentials
key: external-proxy-key
- name: EXTERNAL_PROVIDER_API_KEY
valueFrom:
secretKeyRef:
name: llm-gateway-credentials
key: external-provider-key
securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities:
drop: [ALL]
resources:
requests:
cpu: 250m
memory: 512Mi
limits:
cpu: "1"
memory: 1Gi
livenessProbe:
httpGet:
path: /health/liveliness
port: http
initialDelaySeconds: 15
periodSeconds: 30
readinessProbe:
httpGet:
path: /health/readiness
port: http
initialDelaySeconds: 15
periodSeconds: 10
volumeMounts:
- name: config
mountPath: /etc/litellm
readOnly: true
- name: tmp
mountPath: /tmp
volumes:
- name: config
configMap:
name: litellm-config
- name: tmp
emptyDir: {}
terminationGracePeriodSeconds: 60
---
apiVersion: v1
kind: Service
metadata:
name: litellm-proxy
namespace: llm-routing
labels:
app.kubernetes.io/name: litellm-proxy
spec:
selector:
app.kubernetes.io/name: litellm-proxy
ports:
- name: http
port: 4000
targetPort: http
---
apiVersion: networking.k8s.io/v1
kind: NetworkPolicy
metadata:
name: litellm-proxy
namespace: llm-routing
spec:
podSelector:
matchLabels:
app.kubernetes.io/name: litellm-proxy
policyTypes: [Ingress, Egress]
ingress:
- from:
- podSelector:
matchLabels:
app.kubernetes.io/name: llm-gateway
ports:
- protocol: TCP
port: 4000
egress:
# Name resolution for the provider endpoint.
- to:
- namespaceSelector:
matchLabels:
kubernetes.io/metadata.name: kube-system
ports:
- protocol: UDP
port: 53
- protocol: TCP
port: 53
# HTTPS to the provider, and nothing inside the cluster or private ranges.
- to:
- ipBlock:
cidr: 0.0.0.0/0
except: [10.0.0.0/8, 172.16.0.0/12, 192.168.0.0/16]
ports:
- protocol: TCP
port: 443
9 changes: 9 additions & 0 deletions deploy/kubernetes/gateway.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -42,6 +42,15 @@ spec:
secretKeyRef:
name: llm-gateway-credentials
key: api-keys
# External fallback stays off until an operator enables it; the
# proxy address and key are in place so enabling it is one setting.
- name: ROUTER_EXTERNAL_BASE_URL
value: http://litellm-proxy.llm-routing.svc.cluster.local:4000
- name: ROUTER_EXTERNAL_API_KEY
valueFrom:
secretKeyRef:
name: llm-gateway-credentials
key: external-proxy-key
securityContext:
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
Expand Down
1 change: 1 addition & 0 deletions deploy/kubernetes/kustomization.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8,4 +8,5 @@ resources:
- vllm-serve.yaml
- state.yaml
- network-policy.yaml
- external-provider.yaml
- observability.yaml
8 changes: 8 additions & 0 deletions deploy/kubernetes/network-policy.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,14 @@ spec:
ports:
- protocol: TCP
port: 6379
# The gateway never reaches a provider itself, only the proxy that does.
- to:
- podSelector:
matchLabels:
app.kubernetes.io/name: litellm-proxy
ports:
- protocol: TCP
port: 4000
---
apiVersion: networking.k8s.io/v1
kind: NetworkPolicy
Expand Down
4 changes: 4 additions & 0 deletions deploy/kubernetes/state.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -86,3 +86,7 @@ spec:
remoteRef:
key: llm-routing/gateway
property: external_provider_key
- secretKey: external-proxy-key
remoteRef:
key: llm-routing/gateway
property: external_proxy_key
96 changes: 91 additions & 5 deletions src/llm_router/app.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,8 @@
BackendOutOfMemoryError,
BackendResult,
BackendUnavailableError,
DispatchingBackend,
ExternalDispatchRefusedError,
InferenceBackend,
MockInferenceBackend,
VLLMBackend,
Expand Down Expand Up @@ -51,6 +53,7 @@
ChatCompletionRequest,
ChatCompletionResponse,
ChatMessage,
PrivacyClass,
RouteDecision,
Usage,
)
Expand Down Expand Up @@ -163,7 +166,32 @@ def create_app(
failure_threshold=runtime_settings.engine_failure_threshold,
cooldown_seconds=runtime_settings.engine_cooldown_seconds,
)
inference_backend: InferenceBackend = ResilientBackend(raw_backend, circuit)
external_client = (
httpx.AsyncClient() if backend is None and runtime_settings.external_base_url else None
)
# An injected or mock backend answers external routes too, which keeps
# tests and local development free of a provider.
raw_external: InferenceBackend = (
VLLMBackend(
base_url=runtime_settings.external_base_url,
client=external_client,
request_timeout_seconds=runtime_settings.backend_timeout_seconds,
api_key=runtime_settings.external_api_key,
health_path="/health/liveliness",
)
if external_client is not None
else raw_backend
)
# Each target has its own circuit: a lost GPU node must not close the
# route to the provider, nor a provider outage the route to the engine.
external_circuit = CircuitBreaker(
failure_threshold=runtime_settings.engine_failure_threshold,
cooldown_seconds=runtime_settings.engine_cooldown_seconds,
)
inference_backend: InferenceBackend = DispatchingBackend(
local=ResilientBackend(raw_backend, circuit),
external=ResilientBackend(raw_external, external_circuit),
)
telemetry = metrics if metrics is not None else Metrics()
engine_telemetry = engine_stats or (
EngineStatsCollector(base_url=runtime_settings.vllm_base_url, client=engine_client)
Expand Down Expand Up @@ -211,6 +239,8 @@ async def lifespan(app: FastAPI) -> AsyncIterator[None]:
await admission.drain(runtime_settings.shutdown_grace_seconds)
if engine_client is not None:
await engine_client.aclose()
if external_client is not None:
await external_client.aclose()

app = FastAPI(
title="Local LLM Inference Router",
Expand Down Expand Up @@ -261,6 +291,16 @@ async def admission_handler(_: Request, error: AdmissionRejectedError) -> JSONRe
content={"error": {"message": str(error), "type": "overloaded"}},
)

@app.exception_handler(ExternalDispatchRefusedError)
async def refused_handler(_: Request, error: ExternalDispatchRefusedError) -> JSONResponse:
# Reaching here means routing chose an external model for data that
# must stay local. The request is refused, and counted so it is seen.
telemetry.record_rejection("external_dispatch_refused")
return JSONResponse(
status_code=500,
content={"error": {"message": str(error), "type": "policy_violation"}},
)

@app.exception_handler(BackendOutOfMemoryError)
async def out_of_memory_handler(_: Request, error: BackendOutOfMemoryError) -> JSONResponse:
# Not a retry-as-is condition: the same request at the same size will
Expand Down Expand Up @@ -323,6 +363,7 @@ async def prometheus_metrics() -> Response:
# gateway stays the single scrape target for the whole serving path and
# no background poller runs when nobody is collecting.
telemetry.record_circuit_state(circuit.state, engine=engine_label)
telemetry.record_circuit_state(external_circuit.state, engine="external")
if engine_telemetry is not None:
stats = await engine_telemetry.sample()
if stats is not None:
Expand Down Expand Up @@ -555,6 +596,38 @@ def _record_canary(
if verdict is not None and verdict.action == "rollback":
telemetry.record_canary_rollback(decision.canary_subject)

def _fallback_route(
payload: ChatCompletionRequest,
failed: RouteDecision,
tenant: TenantRecord | None,
privacy_raised_from: PrivacyClass | None,
) -> RouteDecision | None:
"""Find an approved external route for a request the local engine failed.

Only a local failure falls back, and only to a model the request was
already entitled to: the same privacy, tenant, operator, and opt-in
rules apply as on the first attempt.
"""

if not failed.profile.local:
return None
try:
return router.select(
payload,
task=failed.task,
permitted_models=(
catalog.permitted_models_for(tenant) if catalog is not None else None
),
quality_floor=tenant.quality_floor if tenant is not None else 0.0,
tenant_allows_external=(
tenant.allow_external_fallback is not False if tenant is not None else True
),
privacy_raised_from=privacy_raised_from,
fallback_from=failed.profile.id,
)
except NoEligibleModelError:
return None

async def _consume_quota(subject: str, tenant: TenantRecord | None) -> None:
# A tenant may carry its own limit; absent one the platform default
# applies, which is why None means "defer" rather than "unlimited".
Expand Down Expand Up @@ -827,10 +900,23 @@ async def _complete(
)

try:
result = await inference_backend.generate(payload, decision)
except BackendUnavailableError:
_record_canary(decision, ok=False, started=started)
raise
try:
result = await inference_backend.generate(payload, decision)
except BackendUnavailableError as error:
_record_canary(decision, ok=False, started=started)
fallback = _fallback_route(payload, decision, tenant, privacy_raised_from)
if fallback is None:
raise
# Declared, attributed, and counted: never a silent switch.
telemetry.record_fallback(
from_model=decision.profile.id,
to_model=fallback.profile.id,
cause=type(error).__name__,
)
telemetry.record_route(fallback, privacy=payload.routing.privacy.value)
decision = fallback
span.set_route(decision)
result = await inference_backend.generate(payload, decision)
finally:
telemetry.inflight_requests.dec()
admission.release()
Expand Down
Loading
Loading