diff --git a/.github/workflows/cd.yml b/.github/workflows/cd.yml index 7f2d17b..244de26 100644 --- a/.github/workflows/cd.yml +++ b/.github/workflows/cd.yml @@ -62,6 +62,10 @@ jobs: run: python -m llm_router.serving > /tmp/ray-serve.yaml && diff -u config/ray-serve.yaml /tmp/ray-serve.yaml - name: Verify the Ray topology matches the catalog run: python -m llm_router.topology > /tmp/ray-service.yaml && diff -u deploy/overlays/ray/ray-service.yaml /tmp/ray-service.yaml + - name: Verify the Helm chart matches the manifests + run: python -m llm_router.chart --check && helm lint deploy/helm/llm-routing + - name: Package the Helm chart + run: helm package deploy/helm/llm-routing --destination chart - name: Render the canary and rollback plans # One plan per track (model, adapter, policy), each naming what it # rolls back to and the criteria that trigger it. @@ -87,6 +91,7 @@ jobs: canary-plan.json governance-plan.json config/ray-serve.yaml + chart rendered-base.yaml rendered-ray.yaml deploy diff --git a/README.md b/README.md index e7252e2..0a37414 100644 --- a/README.md +++ b/README.md @@ -211,6 +211,29 @@ kubectl apply -k deploy/kubernetes # one vLLM engine kubectl apply -k deploy/overlays/ray # Ray Serve across GPU pools ``` +The same manifests install as a Helm chart, [`deploy/helm/llm-routing`](deploy/helm/llm-routing): + +```bash +helm upgrade --install llm-routing deploy/helm/llm-routing \ + --namespace llm-routing --create-namespace \ + --set serving.mode=ray +``` + +| Value | Default | Purpose | +|---|---|---| +| `serving.mode` | `vllm` | `vllm` for one engine, `ray` for the Ray Serve topology below. | +| `images.*` | digest placeholders | One digest-pinned reference per workload. | +| `gateway.replicas` | `2` | Starting gateway size. | +| `gateway.autoscaling.minReplicas` / `maxReplicas` | `2` / `20` | KEDA bounds. | + +The chart is generated from the manifests and never edited by hand; tests check that it renders +exactly what kustomize renders, in both modes. + +```bash +python -m llm_router.chart # rebuild after changing a manifest +python -m llm_router.chart --check # exit 1 if the committed chart is stale +``` + The base runs a single vLLM engine, which serves one model. The [`deploy/overlays/ray`](deploy/overlays/ray) overlay replaces it with a KubeRay `RayService` that serves every local model in the catalog behind one OpenAI-compatible endpoint: @@ -258,7 +281,8 @@ python -m pip install -e ".[redis]" CD renders the canary plans (one per track, each with its rollback target) and the governance plan, verifies `config/ray-serve.yaml` and the Ray topology against the catalog, and validates both rendered -topologies with kubeconform. +topologies with kubeconform. It also checks the Helm chart against the manifests, lints it, and +packages it. Applying to a cluster stays disabled until a deployment destination is configured. ## Model registry diff --git a/deploy/helm/llm-routing/Chart.yaml b/deploy/helm/llm-routing/Chart.yaml new file mode 100644 index 0000000..8b4be27 --- /dev/null +++ b/deploy/helm/llm-routing/Chart.yaml @@ -0,0 +1,7 @@ +apiVersion: v2 +name: llm-routing +description: OpenAI-compatible gateway, policy router and local LLM serving plane. +type: application +version: 0.1.0 +appVersion: "0.1.0" +kubeVersion: ">=1.27.0-0" diff --git a/deploy/helm/llm-routing/dashboards/gateway.json b/deploy/helm/llm-routing/dashboards/gateway.json new file mode 100644 index 0000000..8844596 --- /dev/null +++ b/deploy/helm/llm-routing/dashboards/gateway.json @@ -0,0 +1,449 @@ +{ + "uid": "llm-routing-gateway", + "title": "LLM routing: gateway and router", + "tags": [ + "llm-routing" + ], + "schemaVersion": 39, + "version": 1, + "editable": false, + "refresh": "30s", + "time": { + "from": "now-1h", + "to": "now" + }, + "templating": { + "list": [ + { + "name": "datasource", + "type": "datasource", + "query": "prometheus", + "label": "Data source" + } + ] + }, + "panels": [ + { + "id": 1, + "title": "Requests per second by model and outcome", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "fieldConfig": { + "defaults": { + "unit": "reqps" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (model, outcome) (rate(router_requests_total[5m]))", + "legendFormat": "{{model}} {{outcome}}" + } + ] + }, + { + "id": 2, + "title": "End-to-end latency", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "fieldConfig": { + "defaults": { + "unit": "s" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "histogram_quantile(0.50, sum by (le, model) (rate(router_request_latency_seconds_bucket[5m])))", + "legendFormat": "p50 {{model}}" + }, + { + "refId": "B", + "expr": "histogram_quantile(0.95, sum by (le, model) (rate(router_request_latency_seconds_bucket[5m])))", + "legendFormat": "p95 {{model}}" + }, + { + "refId": "C", + "expr": "histogram_quantile(0.99, sum by (le, model) (rate(router_request_latency_seconds_bucket[5m])))", + "legendFormat": "p99 {{model}}" + } + ] + }, + { + "id": 3, + "title": "Time to first token (p95)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "fieldConfig": { + "defaults": { + "unit": "s" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "histogram_quantile(0.95, sum by (le, model) (rate(router_time_to_first_token_seconds_bucket[5m])))", + "legendFormat": "{{model}}" + } + ] + }, + { + "id": 4, + "title": "Time per output token (p95)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "fieldConfig": { + "defaults": { + "unit": "s" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "histogram_quantile(0.95, sum by (le, model) (rate(router_time_per_output_token_seconds_bucket[5m])))", + "legendFormat": "{{model}}" + } + ] + }, + { + "id": 5, + "title": "Tokens per second", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (model, kind) (rate(router_tokens_total[5m]))", + "legendFormat": "{{model}} {{kind}}" + } + ] + }, + { + "id": 6, + "title": "In-flight and queued requests", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 16 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum(router_inflight_requests)", + "legendFormat": "in flight" + }, + { + "refId": "B", + "expr": "sum(router_queued_requests)", + "legendFormat": "queued" + } + ] + }, + { + "id": 7, + "title": "Routes by model and privacy class", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 24 + }, + "fieldConfig": { + "defaults": { + "unit": "reqps" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (model, privacy) (rate(router_routes_total[5m]))", + "legendFormat": "{{model}} {{privacy}}" + } + ] + }, + { + "id": 8, + "title": "Rejections by type", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 24 + }, + "fieldConfig": { + "defaults": { + "unit": "reqps" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (type) (rate(router_rejections_total[5m]))", + "legendFormat": "{{type}}" + } + ] + }, + { + "id": 9, + "title": "Fallbacks", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 32 + }, + "fieldConfig": { + "defaults": { + "unit": "reqps" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (from_model, to_model, cause) (rate(router_fallbacks_total[5m]))", + "legendFormat": "{{from_model}} to {{to_model}} ({{cause}})" + }, + { + "refId": "B", + "expr": "sum by (model) (rate(router_external_fallback_total[5m]))", + "legendFormat": "external {{model}}" + } + ] + }, + { + "id": 10, + "title": "Cache hit ratio", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 32 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (cache) (rate(router_cache_events_total{result=\"hit\"}[5m])) / sum by (cache) (rate(router_cache_events_total[5m]))", + "legendFormat": "{{cache}}" + } + ] + }, + { + "id": 11, + "title": "Predicted versus observed quality", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 40 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (model) (router_predicted_quality)", + "legendFormat": "predicted {{model}}" + }, + { + "refId": "B", + "expr": "avg by (model) (router_observed_quality)", + "legendFormat": "observed {{model}}" + } + ] + }, + { + "id": 12, + "title": "Queue-delay prediction error", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 40 + }, + "fieldConfig": { + "defaults": { + "unit": "ms" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (model) (router_queue_delay_prediction_error_ms)", + "legendFormat": "{{model}}" + } + ] + }, + { + "id": 13, + "title": "Canary traffic and rollbacks", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 48 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (subject, arm, outcome) (rate(router_canary_requests_total[5m]))", + "legendFormat": "{{subject}} {{arm}} {{outcome}}" + }, + { + "refId": "B", + "expr": "sum by (subject) (increase(router_canary_rollbacks_total[1h]))", + "legendFormat": "rollback {{subject}}" + } + ] + }, + { + "id": 14, + "title": "Structured-output validity", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 48 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (model) (rate(router_structured_output_total{result=\"valid\"}[5m])) / sum by (model) (rate(router_structured_output_total[5m]))", + "legendFormat": "{{model}}" + } + ] + } + ] +} diff --git a/deploy/helm/llm-routing/dashboards/serving.json b/deploy/helm/llm-routing/dashboards/serving.json new file mode 100644 index 0000000..f931fee --- /dev/null +++ b/deploy/helm/llm-routing/dashboards/serving.json @@ -0,0 +1,317 @@ +{ + "uid": "llm-routing-serving", + "title": "LLM routing: engines and GPUs", + "tags": [ + "llm-routing" + ], + "schemaVersion": 39, + "version": 1, + "editable": false, + "refresh": "30s", + "time": { + "from": "now-1h", + "to": "now" + }, + "templating": { + "list": [ + { + "name": "datasource", + "type": "datasource", + "query": "prometheus", + "label": "Data source" + } + ] + }, + "panels": [ + { + "id": 1, + "title": "Engine running and waiting requests", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (engine) (router_engine_running_requests)", + "legendFormat": "running {{engine}}" + }, + { + "refId": "B", + "expr": "sum by (engine) (router_engine_waiting_requests)", + "legendFormat": "waiting {{engine}}" + } + ] + }, + { + "id": 2, + "title": "Engine batch size", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (engine) (router_engine_batch_size)", + "legendFormat": "{{engine}}" + } + ] + }, + { + "id": 3, + "title": "KV-cache occupancy", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "max by (engine) (router_engine_kv_cache_occupancy_ratio)", + "legendFormat": "{{engine}}" + } + ] + }, + { + "id": 4, + "title": "Engine preemptions", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (engine) (rate(router_engine_preemptions_total[5m]))", + "legendFormat": "{{engine}}" + } + ] + }, + { + "id": 5, + "title": "Engine circuit state (0 closed, 0.5 half-open, 1 open)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "max by (engine) (router_engine_circuit_open)", + "legendFormat": "{{engine}}" + } + ] + }, + { + "id": 6, + "title": "Model load and cold start (p95)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 16 + }, + "fieldConfig": { + "defaults": { + "unit": "s" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "histogram_quantile(0.95, sum by (le, model) (rate(router_model_load_seconds_bucket[5m])))", + "legendFormat": "{{model}}" + } + ] + }, + { + "id": 7, + "title": "GPU utilization as seen by the gateway", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 24 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (gpu) (router_gpu_utilization_ratio)", + "legendFormat": "gpu {{gpu}}" + } + ] + }, + { + "id": 8, + "title": "GPU memory as seen by the gateway", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 24 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (gpu) (router_gpu_memory_used_bytes) / sum by (gpu) (router_gpu_memory_total_bytes)", + "legendFormat": "gpu {{gpu}}" + } + ] + }, + { + "id": 9, + "title": "GPU utilization by node (DCGM exporter)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 32 + }, + "fieldConfig": { + "defaults": { + "unit": "percent" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (Hostname, gpu) (DCGM_FI_DEV_GPU_UTIL)", + "legendFormat": "{{Hostname}} gpu {{gpu}}" + } + ] + }, + { + "id": 10, + "title": "GPU memory by node (DCGM exporter)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 32 + }, + "fieldConfig": { + "defaults": { + "unit": "decmbytes" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (Hostname, gpu) (DCGM_FI_DEV_FB_USED)", + "legendFormat": "used {{Hostname}} gpu {{gpu}}" + }, + { + "refId": "B", + "expr": "sum by (Hostname, gpu) (DCGM_FI_DEV_FB_FREE)", + "legendFormat": "free {{Hostname}} gpu {{gpu}}" + } + ] + } + ] +} diff --git a/deploy/helm/llm-routing/templates/_helpers.tpl b/deploy/helm/llm-routing/templates/_helpers.tpl new file mode 100644 index 0000000..1f1ef7c --- /dev/null +++ b/deploy/helm/llm-routing/templates/_helpers.tpl @@ -0,0 +1,19 @@ +{{/* The OpenAI-compatible endpoint the gateway sends inference to. */}} +{{- define "llm-routing.engineUrl" -}} +{{- if eq .Values.serving.mode "ray" -}} +http://llm-serve-serve-svc.{{ .Release.Namespace }}.svc.cluster.local:8000 +{{- else -}} +http://vllm-serve.{{ .Release.Namespace }}.svc.cluster.local:8000 +{{- end -}} +{{- end -}} + +{{/* The pods that endpoint resolves to, for the gateway's network policy. */}} +{{- define "llm-routing.engineSelector" -}} +{{- if eq .Values.serving.mode "ray" -}}ray-serve{{- else -}}vllm-serve{{- end -}} +{{- end -}} + +{{- define "llm-routing.validate" -}} +{{- if not (has .Values.serving.mode (list "vllm" "ray")) -}} +{{- fail "serving.mode must be vllm or ray" -}} +{{- end -}} +{{- end -}} diff --git a/deploy/helm/llm-routing/templates/autoscaling.yaml b/deploy/helm/llm-routing/templates/autoscaling.yaml new file mode 100644 index 0000000..a02a1fa --- /dev/null +++ b/deploy/helm/llm-routing/templates/autoscaling.yaml @@ -0,0 +1,25 @@ +# Stateless ingress scales on queue depth, independent of GPU replicas. +apiVersion: keda.sh/v1alpha1 +kind: ScaledObject +metadata: + name: llm-gateway + namespace: {{ .Release.Namespace }} +spec: + scaleTargetRef: + name: llm-gateway + minReplicaCount: {{ .Values.gateway.autoscaling.minReplicas }} + maxReplicaCount: {{ .Values.gateway.autoscaling.maxReplicas }} + cooldownPeriod: 120 + triggers: + - type: prometheus + metadata: + serverAddress: http://prometheus.monitoring.svc.cluster.local:9090 + metricName: router_queued_requests + query: sum(router_queued_requests{namespace="{{ .Release.Namespace }}"}) + threshold: "4" + - type: prometheus + metadata: + serverAddress: http://prometheus.monitoring.svc.cluster.local:9090 + metricName: router_request_latency_p95 + query: histogram_quantile(0.95, sum(rate(router_request_latency_seconds_bucket[5m])) by (le)) + threshold: "2" diff --git a/deploy/helm/llm-routing/templates/dashboards.yaml b/deploy/helm/llm-routing/templates/dashboards.yaml new file mode 100644 index 0000000..7bc08a8 --- /dev/null +++ b/deploy/helm/llm-routing/templates/dashboards.yaml @@ -0,0 +1,13 @@ +{{- include "llm-routing.validate" . -}} +# Grafana's dashboard sidecar loads any ConfigMap carrying this label. +apiVersion: v1 +kind: ConfigMap +metadata: + name: llm-routing-dashboards + namespace: {{ .Release.Namespace }} + labels: + grafana_dashboard: "1" +data: +{{- range $path, $_ := .Files.Glob "dashboards/*.json" }} + {{ base $path }}: {{ $.Files.Get $path | quote }} +{{- end }} diff --git a/deploy/helm/llm-routing/templates/external-provider.yaml b/deploy/helm/llm-routing/templates/external-provider.yaml new file mode 100644 index 0000000..c5c6444 --- /dev/null +++ b/deploy/helm/llm-routing/templates/external-provider.yaml @@ -0,0 +1,156 @@ +# LiteLLM proxy for the approved external fallback. It is the only workload in +# the namespace allowed to reach the internet, and only the gateway may call it. +apiVersion: v1 +kind: ConfigMap +metadata: + name: litellm-config + namespace: {{ .Release.Namespace }} +data: + # Kept identical to config/litellm.yaml; a test fails if the two drift. + config.yaml: | + model_list: + - model_name: approved-external-fallback + litellm_params: + model: anthropic/claude-sonnet-5-5 + api_key: os.environ/EXTERNAL_PROVIDER_API_KEY + general_settings: + master_key: os.environ/LITELLM_MASTER_KEY + litellm_settings: + turn_off_message_logging: true + num_retries: 0 + request_timeout: 60 + drop_params: true +--- +apiVersion: apps/v1 +kind: Deployment +metadata: + name: litellm-proxy + namespace: {{ .Release.Namespace }} + labels: + app.kubernetes.io/name: litellm-proxy + app.kubernetes.io/part-of: local-llm-router +spec: + replicas: 2 + selector: + matchLabels: + app.kubernetes.io/name: litellm-proxy + template: + metadata: + labels: + app.kubernetes.io/name: litellm-proxy + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 10001 + seccompProfile: + type: RuntimeDefault + containers: + - name: litellm + image: {{ .Values.images.externalProxy | quote }} + imagePullPolicy: IfNotPresent + args: ["--config", "/etc/litellm/config.yaml", "--port", "4000"] + ports: + - name: http + containerPort: 4000 + env: + - name: LITELLM_MASTER_KEY + valueFrom: + secretKeyRef: + name: llm-gateway-credentials + key: external-proxy-key + - name: EXTERNAL_PROVIDER_API_KEY + valueFrom: + secretKeyRef: + name: llm-gateway-credentials + key: external-provider-key + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: [ALL] + resources: + requests: + cpu: 250m + memory: 512Mi + limits: + cpu: "1" + memory: 1Gi + livenessProbe: + httpGet: + path: /health/liveliness + port: http + initialDelaySeconds: 15 + periodSeconds: 30 + readinessProbe: + httpGet: + path: /health/readiness + port: http + initialDelaySeconds: 15 + periodSeconds: 10 + volumeMounts: + - name: config + mountPath: /etc/litellm + readOnly: true + - name: tmp + mountPath: /tmp + volumes: + - name: config + configMap: + name: litellm-config + - name: tmp + emptyDir: {} + terminationGracePeriodSeconds: 60 +--- +apiVersion: v1 +kind: Service +metadata: + name: litellm-proxy + namespace: {{ .Release.Namespace }} + labels: + app.kubernetes.io/name: litellm-proxy +spec: + selector: + app.kubernetes.io/name: litellm-proxy + ports: + - name: http + port: 4000 + targetPort: http +--- +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: litellm-proxy + namespace: {{ .Release.Namespace }} +spec: + podSelector: + matchLabels: + app.kubernetes.io/name: litellm-proxy + policyTypes: [Ingress, Egress] + ingress: + - from: + - podSelector: + matchLabels: + app.kubernetes.io/name: llm-gateway + ports: + - protocol: TCP + port: 4000 + egress: + # Name resolution for the provider endpoint. + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: kube-system + ports: + - protocol: UDP + port: 53 + - protocol: TCP + port: 53 + # HTTPS to the provider, and nothing inside the cluster or private ranges. + - to: + - ipBlock: + cidr: 0.0.0.0/0 + except: [10.0.0.0/8, 172.16.0.0/12, 192.168.0.0/16] + ports: + - protocol: TCP + port: 443 diff --git a/deploy/helm/llm-routing/templates/gateway.yaml b/deploy/helm/llm-routing/templates/gateway.yaml new file mode 100644 index 0000000..478d875 --- /dev/null +++ b/deploy/helm/llm-routing/templates/gateway.yaml @@ -0,0 +1,97 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: llm-gateway + namespace: {{ .Release.Namespace }} + labels: + app.kubernetes.io/name: llm-gateway + app.kubernetes.io/part-of: local-llm-router +spec: + replicas: {{ .Values.gateway.replicas }} + selector: + matchLabels: + app.kubernetes.io/name: llm-gateway + template: + metadata: + labels: + app.kubernetes.io/name: llm-gateway + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 10001 + seccompProfile: + type: RuntimeDefault + containers: + - name: gateway + # Replaced at deploy time with the digest recorded in the DeploymentRevision. + image: {{ .Values.images.gateway | quote }} + imagePullPolicy: IfNotPresent + ports: + - name: http + containerPort: 8000 + env: + - name: ROUTER_ENVIRONMENT + value: production + - name: ROUTER_BACKEND + value: vllm + - name: ROUTER_VLLM_BASE_URL + value: {{ include "llm-routing.engineUrl" . }} + - name: ROUTER_API_KEYS + valueFrom: + secretKeyRef: + name: llm-gateway-credentials + key: api-keys + # External fallback stays off until an operator enables it; the + # proxy address and key are in place so enabling it is one setting. + - name: ROUTER_EXTERNAL_BASE_URL + value: http://litellm-proxy.{{ .Release.Namespace }}.svc.cluster.local:4000 + - name: ROUTER_EXTERNAL_API_KEY + valueFrom: + secretKeyRef: + name: llm-gateway-credentials + key: external-proxy-key + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: [ALL] + resources: + requests: + cpu: 250m + memory: 512Mi + limits: + cpu: "2" + memory: 2Gi + livenessProbe: + httpGet: + path: /healthz + port: http + initialDelaySeconds: 5 + periodSeconds: 30 + readinessProbe: + httpGet: + path: /readyz + port: http + initialDelaySeconds: 5 + periodSeconds: 10 + lifecycle: + preStop: + exec: + command: ["sleep", "10"] + terminationGracePeriodSeconds: 60 +--- +apiVersion: v1 +kind: Service +metadata: + name: llm-gateway + namespace: {{ .Release.Namespace }} + labels: + app.kubernetes.io/name: llm-gateway +spec: + selector: + app.kubernetes.io/name: llm-gateway + ports: + - name: http + port: 80 + targetPort: http diff --git a/deploy/helm/llm-routing/templates/mlflow.yaml b/deploy/helm/llm-routing/templates/mlflow.yaml new file mode 100644 index 0000000..c739fef --- /dev/null +++ b/deploy/helm/llm-routing/templates/mlflow.yaml @@ -0,0 +1,181 @@ +# MLflow tracking server and model registry: the governance record of every +# model and adapter revision, its benchmark evidence, and its promotions. +# The gateway never reads it; the catalog shipped in the image decides what is +# served. Only the delivery pipeline writes here, with +# `python -m llm_router.governance sync`. +apiVersion: apps/v1 +kind: Deployment +metadata: + name: mlflow + namespace: {{ .Release.Namespace }} + labels: + app.kubernetes.io/name: mlflow + app.kubernetes.io/part-of: local-llm-router +spec: + replicas: 1 + selector: + matchLabels: + app.kubernetes.io/name: mlflow + template: + metadata: + labels: + app.kubernetes.io/name: mlflow + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 10001 + seccompProfile: + type: RuntimeDefault + containers: + - name: mlflow + image: {{ .Values.images.mlflow | quote }} + imagePullPolicy: IfNotPresent + command: [mlflow, server] + args: + - --host=0.0.0.0 + - --port=5000 + # MLflow rejects requests whose Host header it does not expect. + - --allowed-hosts=mlflow.{{ .Release.Namespace }}.svc.cluster.local,mlflow.{{ .Release.Namespace }}.svc.cluster.local:5000 + # Clients upload and download through the server, so object + # storage credentials stay here and are never handed to a client. + - --serve-artifacts + - --artifacts-destination=s3://llm-routing-artifacts + ports: + - name: http + containerPort: 5000 + env: + # The database address carries its password, so the whole value + # comes from the secret manager. + - name: MLFLOW_BACKEND_STORE_URI + valueFrom: + secretKeyRef: + name: mlflow-credentials + key: backend-store-uri + - name: MLFLOW_S3_ENDPOINT_URL + value: http://object-storage.storage.svc.cluster.local:9000 + - name: AWS_ACCESS_KEY_ID + valueFrom: + secretKeyRef: + name: mlflow-credentials + key: artifact-access-key-id + - name: AWS_SECRET_ACCESS_KEY + valueFrom: + secretKeyRef: + name: mlflow-credentials + key: artifact-secret-access-key + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: [ALL] + resources: + requests: + cpu: 250m + memory: 512Mi + limits: + cpu: "1" + memory: 2Gi + livenessProbe: + httpGet: + path: /health + port: http + initialDelaySeconds: 20 + periodSeconds: 30 + readinessProbe: + httpGet: + path: /health + port: http + initialDelaySeconds: 10 + periodSeconds: 10 + volumeMounts: + - name: tmp + mountPath: /tmp + volumes: + - name: tmp + emptyDir: {} + terminationGracePeriodSeconds: 30 +--- +apiVersion: v1 +kind: Service +metadata: + name: mlflow + namespace: {{ .Release.Namespace }} + labels: + app.kubernetes.io/name: mlflow +spec: + selector: + app.kubernetes.io/name: mlflow + ports: + - name: http + port: 5000 + targetPort: http +--- +# Kept apart from the gateway's credentials: the gateway has no use for the +# governance database or the artifact store, so it is never given either. +apiVersion: external-secrets.io/v1 +kind: ExternalSecret +metadata: + name: mlflow-credentials + namespace: {{ .Release.Namespace }} +spec: + refreshInterval: 1h + secretStoreRef: + name: platform-secret-store + kind: ClusterSecretStore + target: + name: mlflow-credentials + data: + - secretKey: backend-store-uri + remoteRef: + key: llm-routing/mlflow + property: backend_store_uri + - secretKey: artifact-access-key-id + remoteRef: + key: llm-routing/mlflow + property: artifact_access_key_id + - secretKey: artifact-secret-access-key + remoteRef: + key: llm-routing/mlflow + property: artifact_secret_access_key +--- +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: mlflow + namespace: {{ .Release.Namespace }} +spec: + podSelector: + matchLabels: + app.kubernetes.io/name: mlflow + policyTypes: [Ingress, Egress] + ingress: + # The delivery pipeline syncs and verifies the catalog from here. No + # workload in this namespace, the gateway included, may reach MLflow. + - from: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: platform-delivery + ports: + - protocol: TCP + port: 5000 + egress: + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: kube-system + ports: + - protocol: UDP + port: 53 + - protocol: TCP + port: 53 + # The backing database and the artifact object store, and nothing else. + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: storage + ports: + - protocol: TCP + port: 5432 + - protocol: TCP + port: 9000 diff --git a/deploy/helm/llm-routing/templates/network-policy.yaml b/deploy/helm/llm-routing/templates/network-policy.yaml new file mode 100644 index 0000000..9da2060 --- /dev/null +++ b/deploy/helm/llm-routing/templates/network-policy.yaml @@ -0,0 +1,69 @@ +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: llm-gateway + namespace: {{ .Release.Namespace }} +spec: + podSelector: + matchLabels: + app.kubernetes.io/name: llm-gateway + policyTypes: [Ingress, Egress] + ingress: + - from: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: applications + ports: + - protocol: TCP + port: 8000 + - from: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: monitoring + ports: + - protocol: TCP + port: 8000 + egress: + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: {{ include "llm-routing.engineSelector" . }} + ports: + - protocol: TCP + port: 8000 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: redis + ports: + - protocol: TCP + port: 6379 + # The gateway never reaches a provider itself, only the proxy that does. + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: litellm-proxy + ports: + - protocol: TCP + port: 4000 +--- +{{- if eq .Values.serving.mode "vllm" }} +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: vllm-serve + namespace: {{ .Release.Namespace }} +spec: + podSelector: + matchLabels: + app.kubernetes.io/name: vllm-serve + policyTypes: [Ingress] + ingress: + - from: + - podSelector: + matchLabels: + app.kubernetes.io/name: llm-gateway + ports: + - protocol: TCP + port: 8000 +{{- end }} diff --git a/deploy/helm/llm-routing/templates/observability.yaml b/deploy/helm/llm-routing/templates/observability.yaml new file mode 100644 index 0000000..3274b75 --- /dev/null +++ b/deploy/helm/llm-routing/templates/observability.yaml @@ -0,0 +1,87 @@ +# Metrics are unauthenticated by design and reachable only from monitoring. +apiVersion: monitoring.coreos.com/v1 +kind: ServiceMonitor +metadata: + name: llm-gateway + namespace: {{ .Release.Namespace }} + labels: + release: prometheus +spec: + selector: + matchLabels: + app.kubernetes.io/name: llm-gateway + endpoints: + - port: http + path: /metrics + interval: 15s +--- +# Alerts on the conditions the platform promises to make visible: overload, +# an open engine circuit, a canary that rolled back, and a GPU out of memory. +apiVersion: monitoring.coreos.com/v1 +kind: PrometheusRule +metadata: + name: llm-routing + namespace: {{ .Release.Namespace }} + labels: + release: prometheus +spec: + groups: + - name: llm-routing.gateway + rules: + - alert: GatewayLatencyHigh + expr: histogram_quantile(0.95, sum by (le, model) (rate(router_request_latency_seconds_bucket[5m]))) > 5 + for: 10m + labels: + severity: warning + annotations: + summary: p95 latency for {{ "{{" }} $labels.model }} has been above 5s for 10 minutes. + - alert: GatewayRejectingRequests + expr: sum by (type) (rate(router_rejections_total{type=~"overloaded|engine_unavailable"}[5m])) > 1 + for: 5m + labels: + severity: warning + annotations: + summary: The gateway is shedding load ({{ "{{" }} $labels.type }}). + - alert: GatewayQueueNotDraining + expr: sum(router_queued_requests) > 16 + for: 10m + labels: + severity: warning + annotations: + summary: Requests have been queueing for admission for 10 minutes. + - alert: CanaryRolledBack + expr: increase(router_canary_rollbacks_total[15m]) > 0 + labels: + severity: warning + annotations: + summary: The canary for {{ "{{" }} $labels.subject }} failed its criteria and was rolled back. + - alert: FallbackRateHigh + expr: sum(rate(router_fallbacks_total[10m])) / sum(rate(router_requests_total[10m])) > 0.05 + for: 10m + labels: + severity: warning + annotations: + summary: More than 5% of requests are being answered by a fallback model. + - name: llm-routing.serving + rules: + - alert: EngineCircuitOpen + expr: max by (engine) (router_engine_circuit_open) == 1 + for: 2m + labels: + severity: critical + annotations: + summary: The circuit for engine {{ "{{" }} $labels.engine }} is open; requests to it are failing fast. + - alert: KVCacheNearlyFull + expr: max by (engine) (router_engine_kv_cache_occupancy_ratio) > 0.95 + for: 10m + labels: + severity: warning + annotations: + summary: KV-cache occupancy on {{ "{{" }} $labels.engine }} has been above 95% for 10 minutes. + - alert: GPUMemoryNearlyFull + expr: DCGM_FI_DEV_FB_USED / (DCGM_FI_DEV_FB_USED + DCGM_FI_DEV_FB_FREE) > 0.97 + for: 10m + labels: + severity: warning + annotations: + summary: GPU {{ "{{" }} $labels.gpu }} on {{ "{{" }} $labels.Hostname }} is nearly out of memory. diff --git a/deploy/helm/llm-routing/templates/ray-monitoring.yaml b/deploy/helm/llm-routing/templates/ray-monitoring.yaml new file mode 100644 index 0000000..bd71ba5 --- /dev/null +++ b/deploy/helm/llm-routing/templates/ray-monitoring.yaml @@ -0,0 +1,20 @@ +{{- if eq .Values.serving.mode "ray" }} +# Ray exports its own metrics and, with log_engine_metrics, each vLLM engine's +# metrics on every head and worker pod. Under this overlay engine telemetry is +# read from here by Prometheus; the gateway cannot scrape a whole cluster from +# one address, so its own router_engine_* gauges stay empty. +apiVersion: monitoring.coreos.com/v1 +kind: PodMonitor +metadata: + name: ray-serve + namespace: {{ .Release.Namespace }} + labels: + release: prometheus +spec: + selector: + matchLabels: + app.kubernetes.io/name: ray-serve + podMetricsEndpoints: + - port: metrics + interval: 15s +{{- end }} diff --git a/deploy/helm/llm-routing/templates/ray-network-policy.yaml b/deploy/helm/llm-routing/templates/ray-network-policy.yaml new file mode 100644 index 0000000..8329457 --- /dev/null +++ b/deploy/helm/llm-routing/templates/ray-network-policy.yaml @@ -0,0 +1,67 @@ +{{- if eq .Values.serving.mode "ray" }} +# Ray head and workers. Inference traffic comes only from the gateway; the +# cluster talks to itself freely; the operator drives it through the dashboard. +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: ray-serve + namespace: {{ .Release.Namespace }} +spec: + podSelector: + matchLabels: + app.kubernetes.io/name: ray-serve + policyTypes: [Ingress, Egress] + ingress: + - from: + - podSelector: + matchLabels: + app.kubernetes.io/name: llm-gateway + ports: + - protocol: TCP + port: 8000 + # Head and workers exchange control, object and worker traffic on ports + # Ray assigns at start-up, so the cluster is open to itself. + - from: + - podSelector: + matchLabels: + app.kubernetes.io/name: ray-serve + # KubeRay submits the Serve configuration and reads health here. + - from: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: kuberay-system + ports: + - protocol: TCP + port: 8265 + - protocol: TCP + port: 52365 + - from: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: monitoring + ports: + - protocol: TCP + port: 8080 + egress: + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: ray-serve + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: kube-system + ports: + - protocol: UDP + port: 53 + - protocol: TCP + port: 53 + # Model and adapter artifacts, from the store governance records them in. + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: storage + ports: + - protocol: TCP + port: 9000 +{{- end }} diff --git a/deploy/helm/llm-routing/templates/ray-ray-service.yaml b/deploy/helm/llm-routing/templates/ray-ray-service.yaml new file mode 100644 index 0000000..0d329d2 --- /dev/null +++ b/deploy/helm/llm-routing/templates/ray-ray-service.yaml @@ -0,0 +1,388 @@ +{{- if eq .Values.serving.mode "ray" }} +# Generated from config/registry.yaml by `python -m llm_router.topology`. +# Do not edit: change the catalog and regenerate. A test and CD both fail on drift. +apiVersion: ray.io/v1 +kind: RayService +metadata: + name: llm-serve + namespace: {{ .Release.Namespace }} + labels: + app.kubernetes.io/part-of: local-llm-router + annotations: + llm-routing/policy-version: v1 +spec: + serveConfigV2: | + applications: + - name: llm_app + route_prefix: / + import_path: ray.serve.llm:build_openai_app + args: + llm_configs: + - model_loading_config: + model_id: small-specialist + model_source: s3://llm-routing-artifacts/models/small-specialist/mock-small-sha256-dev + accelerator_type: L4 + deployment_config: + autoscaling_config: + min_replicas: 1 + max_replicas: 4 + target_ongoing_requests: 8 + upscale_delay_s: 10 + downscale_delay_s: 300 + max_ongoing_requests: 16 + ray_actor_options: + num_gpus: 1 + resources: + gpu_pool_nvidia-l4: 0.001 + engine_kwargs: + max_model_len: 8192 + tensor_parallel_size: 1 + enable_prefix_caching: true + enable_chunked_prefill: true + quantization: awq + enable_lora: true + max_loras: 3 + max_lora_rank: 32 + log_engine_metrics: true + lora_config: + dynamic_lora_loading_path: s3://llm-routing-artifacts/adapters + max_num_adapters_per_replica: 3 + - model_loading_config: + model_id: general-local + model_source: s3://llm-routing-artifacts/models/general-local/mock-general-sha256-dev + accelerator_type: A10G + deployment_config: + autoscaling_config: + min_replicas: 1 + max_replicas: 4 + target_ongoing_requests: 8 + upscale_delay_s: 10 + downscale_delay_s: 300 + max_ongoing_requests: 16 + ray_actor_options: + num_gpus: 1 + resources: + gpu_pool_nvidia-a10g: 0.001 + engine_kwargs: + max_model_len: 32768 + tensor_parallel_size: 1 + enable_prefix_caching: true + enable_chunked_prefill: true + log_engine_metrics: true + - model_loading_config: + model_id: high-capability + model_source: s3://llm-routing-artifacts/models/high-capability/mock-high-sha256-dev + accelerator_type: A100 + deployment_config: + autoscaling_config: + min_replicas: 0 + max_replicas: 2 + target_ongoing_requests: 8 + upscale_delay_s: 10 + downscale_delay_s: 60 + max_ongoing_requests: 16 + ray_actor_options: + num_gpus: 2 + resources: + gpu_pool_nvidia-a100: 0.001 + engine_kwargs: + max_model_len: 65536 + tensor_parallel_size: 2 + enable_prefix_caching: true + enable_chunked_prefill: true + log_engine_metrics: true + - model_loading_config: + model_id: general-local--general-gptq + model_source: s3://llm-routing-artifacts/models/general-local/mock-general-sha256-dev + accelerator_type: A10G + deployment_config: + autoscaling_config: + min_replicas: 0 + max_replicas: 1 + ray_actor_options: + num_gpus: 1 + resources: + gpu_pool_nvidia-a10g: 0.001 + engine_kwargs: + max_model_len: 32768 + tensor_parallel_size: 1 + enable_prefix_caching: true + enable_chunked_prefill: true + quantization: gptq + log_engine_metrics: true + - model_loading_config: + model_id: high-capability--high-capability-speculative + model_source: s3://llm-routing-artifacts/models/high-capability/mock-high-sha256-dev + accelerator_type: A100 + deployment_config: + autoscaling_config: + min_replicas: 0 + max_replicas: 1 + ray_actor_options: + num_gpus: 2 + resources: + gpu_pool_nvidia-a100: 0.001 + engine_kwargs: + max_model_len: 65536 + tensor_parallel_size: 2 + enable_prefix_caching: true + enable_chunked_prefill: true + speculative_config: + model: s3://llm-routing-artifacts/models/small-specialist/mock-small-sha256-dev + num_speculative_tokens: 5 + log_engine_metrics: true + rayClusterConfig: + rayVersion: 2.51.0 + enableInTreeAutoscaling: true + headGroupSpec: + rayStartParams: + dashboard-host: 0.0.0.0 + metrics-export-port: '8080' + num-cpus: '0' + template: + metadata: + labels: + app.kubernetes.io/name: ray-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 1000 + seccompProfile: + type: RuntimeDefault + containers: + - name: ray-head + image: {{ .Values.images.ray | quote }} + imagePullPolicy: IfNotPresent + ports: + - name: metrics + containerPort: 8080 + - name: gcs + containerPort: 6379 + - name: dashboard + containerPort: 8265 + - name: serve + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: + - ALL + resources: + requests: + cpu: '2' + memory: 8Gi + limits: + cpu: '2' + memory: 8Gi + volumeMounts: + - name: tmp + mountPath: /tmp + - name: cache + mountPath: /home/ray/.cache + - name: shm + mountPath: /dev/shm + volumes: + - name: tmp + emptyDir: {} + - name: cache + emptyDir: {} + - name: shm + emptyDir: + medium: Memory + terminationGracePeriodSeconds: 60 + workerGroupSpecs: + - groupName: a100-pool + replicas: 0 + minReplicas: 0 + maxReplicas: 3 + rayStartParams: + metrics-export-port: '8080' + resources: '"{\"gpu_pool_nvidia-a100\": 1}"' + template: + metadata: + labels: + app.kubernetes.io/name: ray-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 1000 + seccompProfile: + type: RuntimeDefault + containers: + - name: ray-worker + image: {{ .Values.images.ray | quote }} + imagePullPolicy: IfNotPresent + ports: + - name: metrics + containerPort: 8080 + - name: serve + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: + - ALL + resources: + requests: + cpu: '8' + memory: 160Gi + nvidia.com/gpu: '2' + limits: + cpu: '8' + memory: 160Gi + nvidia.com/gpu: '2' + volumeMounts: + - name: tmp + mountPath: /tmp + - name: cache + mountPath: /home/ray/.cache + - name: shm + mountPath: /dev/shm + volumes: + - name: tmp + emptyDir: {} + - name: cache + emptyDir: {} + - name: shm + emptyDir: + medium: Memory + terminationGracePeriodSeconds: 120 + nodeSelector: + nvidia.com/gpu.product: NVIDIA-A100-SXM4-80GB + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule + - groupName: a10g-pool + replicas: 1 + minReplicas: 1 + maxReplicas: 5 + rayStartParams: + metrics-export-port: '8080' + resources: '"{\"gpu_pool_nvidia-a10g\": 1}"' + template: + metadata: + labels: + app.kubernetes.io/name: ray-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 1000 + seccompProfile: + type: RuntimeDefault + containers: + - name: ray-worker + image: {{ .Values.images.ray | quote }} + imagePullPolicy: IfNotPresent + ports: + - name: metrics + containerPort: 8080 + - name: serve + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: + - ALL + resources: + requests: + cpu: '4' + memory: 48Gi + nvidia.com/gpu: '1' + limits: + cpu: '4' + memory: 48Gi + nvidia.com/gpu: '1' + volumeMounts: + - name: tmp + mountPath: /tmp + - name: cache + mountPath: /home/ray/.cache + - name: shm + mountPath: /dev/shm + volumes: + - name: tmp + emptyDir: {} + - name: cache + emptyDir: {} + - name: shm + emptyDir: + medium: Memory + terminationGracePeriodSeconds: 120 + nodeSelector: + nvidia.com/gpu.product: NVIDIA-A10G + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule + - groupName: l4-pool + replicas: 1 + minReplicas: 1 + maxReplicas: 4 + rayStartParams: + metrics-export-port: '8080' + resources: '"{\"gpu_pool_nvidia-l4\": 1}"' + template: + metadata: + labels: + app.kubernetes.io/name: ray-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 1000 + seccompProfile: + type: RuntimeDefault + containers: + - name: ray-worker + image: {{ .Values.images.ray | quote }} + imagePullPolicy: IfNotPresent + ports: + - name: metrics + containerPort: 8080 + - name: serve + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: + - ALL + resources: + requests: + cpu: '4' + memory: 24Gi + nvidia.com/gpu: '1' + limits: + cpu: '4' + memory: 24Gi + nvidia.com/gpu: '1' + volumeMounts: + - name: tmp + mountPath: /tmp + - name: cache + mountPath: /home/ray/.cache + - name: shm + mountPath: /dev/shm + volumes: + - name: tmp + emptyDir: {} + - name: cache + emptyDir: {} + - name: shm + emptyDir: + medium: Memory + terminationGracePeriodSeconds: 120 + nodeSelector: + nvidia.com/gpu.product: NVIDIA-L4 + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule +{{- end }} diff --git a/deploy/helm/llm-routing/templates/state.yaml b/deploy/helm/llm-routing/templates/state.yaml new file mode 100644 index 0000000..8719f99 --- /dev/null +++ b/deploy/helm/llm-routing/templates/state.yaml @@ -0,0 +1,92 @@ +# Cache and quota state. Credentials come from the cluster secret manager; +# no secret material is committed to this repository. +apiVersion: apps/v1 +kind: StatefulSet +metadata: + name: redis + namespace: {{ .Release.Namespace }} + labels: + app.kubernetes.io/name: redis +spec: + serviceName: redis + replicas: 1 + selector: + matchLabels: + app.kubernetes.io/name: redis + template: + metadata: + labels: + app.kubernetes.io/name: redis + spec: + securityContext: + runAsNonRoot: true + runAsUser: 10001 + seccompProfile: + type: RuntimeDefault + containers: + - name: redis + image: {{ .Values.images.redis | quote }} + ports: + - name: redis + containerPort: 6379 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: [ALL] + resources: + requests: + cpu: 100m + memory: 256Mi + limits: + cpu: "1" + memory: 1Gi + livenessProbe: + tcpSocket: + port: redis + initialDelaySeconds: 10 + periodSeconds: 30 + readinessProbe: + tcpSocket: + port: redis + initialDelaySeconds: 5 + periodSeconds: 10 +--- +apiVersion: v1 +kind: Service +metadata: + name: redis + namespace: {{ .Release.Namespace }} +spec: + selector: + app.kubernetes.io/name: redis + ports: + - name: redis + port: 6379 + targetPort: redis +--- +apiVersion: external-secrets.io/v1 +kind: ExternalSecret +metadata: + name: llm-gateway-credentials + namespace: {{ .Release.Namespace }} +spec: + refreshInterval: 1h + secretStoreRef: + name: platform-secret-store + kind: ClusterSecretStore + target: + name: llm-gateway-credentials + data: + - secretKey: api-keys + remoteRef: + key: llm-routing/gateway + property: api_keys + - secretKey: external-provider-key + remoteRef: + key: llm-routing/gateway + property: external_provider_key + - secretKey: external-proxy-key + remoteRef: + key: llm-routing/gateway + property: external_proxy_key diff --git a/deploy/helm/llm-routing/templates/vllm-serve.yaml b/deploy/helm/llm-routing/templates/vllm-serve.yaml new file mode 100644 index 0000000..199a601 --- /dev/null +++ b/deploy/helm/llm-routing/templates/vllm-serve.yaml @@ -0,0 +1,88 @@ +{{- if eq .Values.serving.mode "vllm" }} +# GPU serving replicas are scheduled onto accelerator-specific pools and scale +# separately from the stateless gateway. +apiVersion: apps/v1 +kind: Deployment +metadata: + name: vllm-serve + namespace: {{ .Release.Namespace }} + labels: + app.kubernetes.io/name: vllm-serve + app.kubernetes.io/part-of: local-llm-router +spec: + replicas: 1 + selector: + matchLabels: + app.kubernetes.io/name: vllm-serve + template: + metadata: + labels: + app.kubernetes.io/name: vllm-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 10001 + seccompProfile: + type: RuntimeDefault + nodeSelector: + nvidia.com/gpu.product: NVIDIA-L4 + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule + containers: + - name: engine + image: {{ .Values.images.engine | quote }} + args: + - --served-model-name=small-specialist + - --enable-prefix-caching + - --enable-lora + - --max-model-len=8192 + - --quantization=awq + ports: + - name: http + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: [ALL] + resources: + requests: + cpu: "4" + memory: 32Gi + nvidia.com/gpu: "1" + limits: + cpu: "8" + memory: 48Gi + nvidia.com/gpu: "1" + livenessProbe: + httpGet: + path: /health + port: http + initialDelaySeconds: 120 + periodSeconds: 30 + readinessProbe: + httpGet: + path: /health + port: http + initialDelaySeconds: 120 + periodSeconds: 10 + terminationGracePeriodSeconds: 120 +{{- end }} +--- +{{- if eq .Values.serving.mode "vllm" }} +apiVersion: v1 +kind: Service +metadata: + name: vllm-serve + namespace: {{ .Release.Namespace }} +spec: + selector: + app.kubernetes.io/name: vllm-serve + ports: + - name: http + port: 8000 + targetPort: http +{{- end }} diff --git a/deploy/helm/llm-routing/values.yaml b/deploy/helm/llm-routing/values.yaml new file mode 100644 index 0000000..93419dc --- /dev/null +++ b/deploy/helm/llm-routing/values.yaml @@ -0,0 +1,23 @@ +# Which serving topology to run. +# vllm: one vLLM engine, serving one model. +# ray: a KubeRay RayService serving every local model in the catalog from +# worker pools split by accelerator type. Needs the KubeRay operator. +serving: + mode: vllm + +# Every image is pinned by digest. Replace each placeholder with the digest +# recorded for the release; a tag alone is refused by the contract tests. +images: + gateway: ghcr.io/REPLACE_ME/local-llm-router@sha256:REPLACE_ME + engine: docker.io/vllm/vllm-openai@sha256:REPLACE_ME + ray: docker.io/rayproject/ray-llm@sha256:REPLACE_ME + redis: docker.io/library/redis@sha256:REPLACE_ME + externalProxy: ghcr.io/berriai/litellm@sha256:REPLACE_ME + mlflow: ghcr.io/mlflow/mlflow@sha256:REPLACE_ME + +gateway: + # Starting size; the KEDA ScaledObject takes over within the bounds below. + replicas: 2 + autoscaling: + minReplicas: 2 + maxReplicas: 20 diff --git a/src/llm_router/chart.py b/src/llm_router/chart.py new file mode 100644 index 0000000..5ec6e1a --- /dev/null +++ b/src/llm_router/chart.py @@ -0,0 +1,254 @@ +"""Helm chart built from the Kubernetes manifests (section 17). + +The manifests under ``deploy/kubernetes`` and the Ray overlay are the reviewed +source. The chart is those same documents with the values an installation +changes lifted out: image references, the namespace, gateway scale, and which +serving topology to run. Building it rather than writing it means the chart +cannot drift from the manifests the contract tests check. +""" + +import re +import tomllib +from collections.abc import Sequence +from pathlib import Path +from typing import Any + +import yaml + +CHART_NAME = "llm-routing" +BASE = Path("deploy/kubernetes") +RAY_OVERLAY = Path("deploy/overlays/ray") +CHART = Path("deploy/helm") / CHART_NAME +# Applied by the overlay as patches rather than shipped as resources. +OVERLAY_ONLY = {"kustomization.yaml", "gateway-network-policy.patch.yaml"} +ENGINE_URL = "http://vllm-serve.llm-routing.svc.cluster.local:8000" + +IMAGES = { + "gateway": "ghcr.io/REPLACE_ME/local-llm-router@sha256:REPLACE_ME", + "engine": "docker.io/vllm/vllm-openai@sha256:REPLACE_ME", + "ray": "docker.io/rayproject/ray-llm@sha256:REPLACE_ME", + "redis": "docker.io/library/redis@sha256:REPLACE_ME", + "externalProxy": "ghcr.io/berriai/litellm@sha256:REPLACE_ME", + "mlflow": "ghcr.io/mlflow/mlflow@sha256:REPLACE_ME", +} + +HELPERS = """{{/* The OpenAI-compatible endpoint the gateway sends inference to. */}} +{{- define "llm-routing.engineUrl" -}} +{{- if eq .Values.serving.mode "ray" -}} +http://llm-serve-serve-svc.{{ .Release.Namespace }}.svc.cluster.local:8000 +{{- else -}} +http://vllm-serve.{{ .Release.Namespace }}.svc.cluster.local:8000 +{{- end -}} +{{- end -}} + +{{/* The pods that endpoint resolves to, for the gateway's network policy. */}} +{{- define "llm-routing.engineSelector" -}} +{{- if eq .Values.serving.mode "ray" -}}ray-serve{{- else -}}vllm-serve{{- end -}} +{{- end -}} + +{{- define "llm-routing.validate" -}} +{{- if not (has .Values.serving.mode (list "vllm" "ray")) -}} +{{- fail "serving.mode must be vllm or ray" -}} +{{- end -}} +{{- end -}} +""" + +DASHBOARDS = """{{- include "llm-routing.validate" . -}} +# Grafana's dashboard sidecar loads any ConfigMap carrying this label. +apiVersion: v1 +kind: ConfigMap +metadata: + name: llm-routing-dashboards + namespace: {{ .Release.Namespace }} + labels: + grafana_dashboard: "1" +data: +{{- range $path, $_ := .Files.Glob "dashboards/*.json" }} + {{ base $path }}: {{ $.Files.Get $path | quote }} +{{- end }} +""" + + +class ChartError(RuntimeError): + """Raised when a manifest no longer has the shape the chart expects.""" + + +def _documents(path: Path) -> list[str]: + """Split a manifest into its documents, keeping comments where they sit.""" + + text = path.read_text(encoding="utf-8") + return [chunk.strip("\n") for chunk in re.split(r"(?m)^---\s*$", text) if chunk.strip()] + + +def _swap(text: str, old: str, new: str, *, where: str) -> str: + if old not in text: + raise ChartError(f"{where}: expected to find {old!r}") + return text.replace(old, new) + + +def _template(document: str, *, source: str) -> str | None: + """Turn one manifest document into a chart template, or drop it.""" + + parsed: dict[str, Any] = yaml.safe_load(document) + kind, name = parsed["kind"], parsed["metadata"]["name"] + where = f"{source} {kind}/{name}" + if kind == "Namespace": + # Helm installs into a namespace it is given; it does not own one. + return None + + # Alert annotations use Prometheus templating, which Helm must pass through. + text = document.replace("{{", '{{ "{{" }}') + text = text.replace("namespace: llm-routing", "namespace: {{ .Release.Namespace }}") + text = text.replace('namespace="llm-routing"', 'namespace="{{ .Release.Namespace }}"') + for key, image in IMAGES.items(): + text = text.replace(f"image: {image}", f"image: {{{{ .Values.images.{key} | quote }}}}") + + if kind == "Deployment" and name == "llm-gateway": + text = _swap(text, ENGINE_URL, '{{ include "llm-routing.engineUrl" . }}', where=where) + text = _swap(text, "replicas: 2", "replicas: {{ .Values.gateway.replicas }}", where=where) + if kind == "NetworkPolicy" and name == "llm-gateway": + text = _swap( + text, + "app.kubernetes.io/name: vllm-serve", + 'app.kubernetes.io/name: {{ include "llm-routing.engineSelector" . }}', + where=where, + ) + if kind == "ScaledObject": + for field, value in ( + ("minReplicaCount", "minReplicas"), + ("maxReplicaCount", "maxReplicas"), + ): + current = parsed["spec"][field] + text = _swap( + text, + f"{field}: {current}", + f"{field}: {{{{ .Values.gateway.autoscaling.{value} }}}}", + where=where, + ) + text = text.replace(".llm-routing.svc", ".{{ .Release.Namespace }}.svc") + + if name == "vllm-serve": + return f'{{{{- if eq .Values.serving.mode "vllm" }}}}\n{text}\n{{{{- end }}}}' + if source.startswith("ray/"): + return f'{{{{- if eq .Values.serving.mode "ray" }}}}\n{text}\n{{{{- end }}}}' + return text + + +def _values(source_root: Path) -> str: + scaled = next( + yaml.safe_load(document) + for document in _documents(source_root / BASE / "autoscaling.yaml") + if yaml.safe_load(document)["kind"] == "ScaledObject" + ) + lines = [ + "# Which serving topology to run.", + "# vllm: one vLLM engine, serving one model.", + "# ray: a KubeRay RayService serving every local model in the catalog from", + "# worker pools split by accelerator type. Needs the KubeRay operator.", + "serving:", + " mode: vllm", + "", + "# Every image is pinned by digest. Replace each placeholder with the digest", + "# recorded for the release; a tag alone is refused by the contract tests.", + "images:", + *(f" {key}: {image}" for key, image in IMAGES.items()), + "", + "gateway:", + " # Starting size; the KEDA ScaledObject takes over within the bounds below.", + " replicas: 2", + " autoscaling:", + f" minReplicas: {scaled['spec']['minReplicaCount']}", + f" maxReplicas: {scaled['spec']['maxReplicaCount']}", + ] + return "\n".join(lines) + "\n" + + +def _chart_metadata(source_root: Path) -> str: + project = tomllib.loads((source_root / "pyproject.toml").read_text(encoding="utf-8")) + version = project["project"]["version"] + return ( + "apiVersion: v2\n" + f"name: {CHART_NAME}\n" + "description: OpenAI-compatible gateway, policy router and local LLM serving plane.\n" + "type: application\n" + f"version: {version}\n" + f'appVersion: "{version}"\n' + 'kubeVersion: ">=1.27.0-0"\n' + ) + + +def build_chart(source_root: Path = Path(".")) -> dict[str, str]: + """Every file of the chart, keyed by its path inside the chart directory.""" + + files = { + "Chart.yaml": _chart_metadata(source_root), + "values.yaml": _values(source_root), + "templates/_helpers.tpl": HELPERS, + "templates/dashboards.yaml": DASHBOARDS, + } + sources = [ + (path, path.name) + for path in sorted((source_root / BASE).glob("*.yaml")) + if path.name != "kustomization.yaml" + ] + [ + (path, f"ray/{path.name}") + for path in sorted((source_root / RAY_OVERLAY).glob("*.yaml")) + if path.name not in OVERLAY_ONLY + ] + for path, source in sources: + templates = [ + template + for document in _documents(path) + if (template := _template(document, source=source)) is not None + ] + if templates: + target = f"templates/{source.replace('/', '-')}" + files[target] = "\n---\n".join(templates) + "\n" + for path in sorted((source_root / BASE / "dashboards").glob("*.json")): + files[f"dashboards/{path.name}"] = path.read_text(encoding="utf-8") + return files + + +def write_chart(target: Path, source_root: Path = Path(".")) -> None: + for relative, content in build_chart(source_root).items(): + path = target / relative + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(content, encoding="utf-8", newline="\n") + + +def stale_files(target: Path, source_root: Path = Path(".")) -> list[str]: + """Chart files that are missing, changed, or no longer generated.""" + + expected = build_chart(source_root) + on_disk = { + path.relative_to(target).as_posix(): path.read_text(encoding="utf-8") + for path in target.rglob("*") + if path.is_file() + } + return sorted( + name for name in expected.keys() | on_disk.keys() if expected.get(name) != on_disk.get(name) + ) + + +def main(argv: Sequence[str] | None = None) -> int: + """Write the chart, or with --check exit 1 if the committed chart is stale.""" + + import argparse + + parser = argparse.ArgumentParser(description=main.__doc__) + parser.add_argument("--output", default=str(CHART)) + parser.add_argument("--check", action="store_true") + arguments = parser.parse_args(argv) + + target = Path(arguments.output) + if arguments.check: + stale = stale_files(target) + for name in stale: + print(f"stale: {target.as_posix()}/{name}") + return 1 if stale else 0 + write_chart(target) + return 0 + + +if __name__ == "__main__": # pragma: no cover - command-line entry point + raise SystemExit(main()) diff --git a/tests/unit/test_chart.py b/tests/unit/test_chart.py new file mode 100644 index 0000000..2ebf57f --- /dev/null +++ b/tests/unit/test_chart.py @@ -0,0 +1,178 @@ +"""The Helm chart is the reviewed manifests with installation values lifted out.""" + +import os +import shutil +import subprocess +from pathlib import Path +from typing import Any + +import pytest +import yaml + +from llm_router.chart import ( + CHART, + IMAGES, + ChartError, + _template, + build_chart, + main, + stale_files, + write_chart, +) + +TOOLS_PRESENT = shutil.which("helm") is not None and shutil.which("kubectl") is not None +# Skipping is for a workstation without the tools. In CI their absence is a failure. +needs_tools = pytest.mark.skipif( + not TOOLS_PRESENT and not os.environ.get("CI"), reason="helm and kubectl are not installed" +) + +Key = tuple[str, str] + + +def run(*command: str) -> str: + return subprocess.run(command, capture_output=True, text=True, check=True).stdout + + +def resources(rendered: str) -> dict[Key, dict[str, Any]]: + found: dict[Key, dict[str, Any]] = {} + for document in yaml.safe_load_all(rendered): + if not document or document["kind"] == "Namespace": + continue + if document["kind"] == "Deployment": + # Kustomize moves a patched variable to the front; order means nothing. + for container in document["spec"]["template"]["spec"]["containers"]: + container.get("env", []).sort(key=lambda item: item["name"]) + found[document["kind"], document["metadata"]["name"]] = document + return found + + +def helm_template(*arguments: str, namespace: str = "llm-routing") -> dict[Key, dict[str, Any]]: + return resources(run("helm", "template", "release", str(CHART), "-n", namespace, *arguments)) + + +def test_the_committed_chart_is_what_the_manifests_generate() -> None: + assert stale_files(CHART) == [] + assert main(["--check"]) == 0 + + +def test_a_changed_missing_or_stray_chart_file_is_reported( + tmp_path: Path, capsys: pytest.CaptureFixture[str] +) -> None: + assert main(["--output", str(tmp_path)]) == 0 + assert stale_files(tmp_path) == [] + + (tmp_path / "values.yaml").write_text("serving:\n mode: ray\n", encoding="utf-8") + (tmp_path / "templates" / "gateway.yaml").unlink() + (tmp_path / "templates" / "hand-written.yaml").write_text("kind: Secret\n", encoding="utf-8") + + assert stale_files(tmp_path) == [ + "templates/gateway.yaml", + "templates/hand-written.yaml", + "values.yaml", + ] + assert main(["--output", str(tmp_path), "--check"]) == 1 + assert "stale:" in capsys.readouterr().out + + write_chart(tmp_path) + assert stale_files(tmp_path) == ["templates/hand-written.yaml"] + + +def test_the_chart_holds_every_manifest_and_owns_no_namespace() -> None: + chart = build_chart() + + for name in ( + "gateway", + "state", + "mlflow", + "external-provider", + "vllm-serve", + "ray-ray-service", + ): + assert f"templates/{name}.yaml" in chart + assert "templates/namespace.yaml" not in chart + assert "templates/ray-kustomization.yaml" not in chart + assert all("kind: Namespace" not in content for content in chart.values()) + assert set(yaml.safe_load(chart["values.yaml"])["images"]) == set(IMAGES) + assert yaml.safe_load(chart["Chart.yaml"])["name"] == "llm-routing" + + +def test_every_image_and_namespace_reference_is_a_value() -> None: + templates = { + name: content + for name, content in build_chart().items() + if name.startswith("templates/") and name.endswith(".yaml") + } + + for name, content in templates.items(): + assert "REPLACE_ME" not in content, name + assert "namespace: llm-routing" not in content, name + assert ".llm-routing.svc" not in content, name + for image in yaml.safe_load(build_chart()["values.yaml"])["images"].values(): + assert "@sha256:" in image + + +def test_prometheus_templating_in_alerts_survives_helm() -> None: + rules = build_chart()["templates/observability.yaml"] + + assert '{{ "{{" }} $labels.model }}' in rules + assert "{{ $labels" not in rules + + +def test_a_manifest_that_lost_what_the_chart_parameterizes_is_refused() -> None: + gateway = ( + "apiVersion: apps/v1\nkind: Deployment\nmetadata:\n name: llm-gateway\n" + "spec:\n replicas: 2\n" + ) + + with pytest.raises(ChartError, match=r"Deployment/llm-gateway: expected to find 'http://"): + _template(gateway, source="gateway.yaml") + + +@needs_tools +def test_the_chart_lints_cleanly() -> None: + assert "0 chart(s) failed" in run("helm", "lint", str(CHART)) + + +@needs_tools +def test_default_values_render_exactly_the_single_engine_base() -> None: + assert helm_template() == resources(run("kubectl", "kustomize", "deploy/kubernetes")) + + +@needs_tools +def test_ray_mode_renders_exactly_the_ray_overlay() -> None: + rendered = helm_template("--set", "serving.mode=ray") + + assert rendered == resources(run("kubectl", "kustomize", "deploy/overlays/ray")) + assert ("RayService", "llm-serve") in rendered + assert not any(name == "vllm-serve" for _, name in rendered) + + +@needs_tools +def test_values_reach_the_rendered_workloads() -> None: + rendered = helm_template( + "--set", + "images.gateway=registry.example/router@sha256:abc", + "--set", + "gateway.replicas=5", + "--set", + "gateway.autoscaling.maxReplicas=40", + namespace="inference", + ) + gateway = rendered["Deployment", "llm-gateway"] + container = gateway["spec"]["template"]["spec"]["containers"][0] + environment = {item["name"]: item.get("value") for item in container["env"]} + + assert container["image"] == "registry.example/router@sha256:abc" + assert gateway["spec"]["replicas"] == 5 + assert rendered["ScaledObject", "llm-gateway"]["spec"]["maxReplicaCount"] == 40 + assert environment["ROUTER_VLLM_BASE_URL"] == ( + "http://vllm-serve.inference.svc.cluster.local:8000" + ) + assert {item["metadata"]["namespace"] for item in rendered.values()} == {"inference"} + + +@needs_tools +def test_an_unknown_serving_mode_is_refused() -> None: + with pytest.raises(subprocess.CalledProcessError) as raised: + helm_template("--set", "serving.mode=both") + assert "serving.mode must be vllm or ray" in raised.value.stderr