diff --git a/.github/workflows/cd.yml b/.github/workflows/cd.yml index a9d8d8e..7f2d17b 100644 --- a/.github/workflows/cd.yml +++ b/.github/workflows/cd.yml @@ -60,6 +60,8 @@ jobs: - run: python -m pip install -e ".[dev]" - name: Verify the serving configuration matches the catalog run: python -m llm_router.serving > /tmp/ray-serve.yaml && diff -u config/ray-serve.yaml /tmp/ray-serve.yaml + - name: Verify the Ray topology matches the catalog + run: python -m llm_router.topology > /tmp/ray-service.yaml && diff -u deploy/overlays/ray/ray-service.yaml /tmp/ray-service.yaml - name: Render the canary and rollback plans # One plan per track (model, adapter, policy), each naming what it # rolls back to and the criteria that trigger it. @@ -73,7 +75,11 @@ jobs: run: | curl -sSLo kubeconform.tar.gz https://github.com/yannh/kubeconform/releases/download/v0.7.0/kubeconform-linux-amd64.tar.gz tar xf kubeconform.tar.gz kubeconform - ./kubeconform -strict -ignore-missing-schemas -summary deploy/kubernetes + # Both topologies are rendered the way they would be applied: the + # single-engine base, and the Ray overlay that replaces the engine. + kubectl kustomize deploy/kubernetes > rendered-base.yaml + kubectl kustomize deploy/overlays/ray > rendered-ray.yaml + ./kubeconform -strict -ignore-missing-schemas -summary rendered-base.yaml rendered-ray.yaml - uses: actions/upload-artifact@v7 with: name: deployment-plan-${{ env.RELEASE_REF }} @@ -81,5 +87,7 @@ jobs: canary-plan.json governance-plan.json config/ray-serve.yaml - deploy/kubernetes + rendered-base.yaml + rendered-ray.yaml + deploy retention-days: 14 diff --git a/README.md b/README.md index 6d97258..e7252e2 100644 --- a/README.md +++ b/README.md @@ -207,9 +207,46 @@ unprivileged workloads, digest-pinned images, bounded resources, real probes, GP pinning, and `/metrics` reachable only from monitoring. ```bash -kubectl apply -k deploy/kubernetes +kubectl apply -k deploy/kubernetes # one vLLM engine +kubectl apply -k deploy/overlays/ray # Ray Serve across GPU pools ``` +The base runs a single vLLM engine, which serves one model. The +[`deploy/overlays/ray`](deploy/overlays/ray) overlay replaces it with a KubeRay `RayService` +that serves every local model in the catalog behind one OpenAI-compatible endpoint: + +- A head that schedules and never runs a model. +- One worker group per accelerator type, pinned by `nvidia.com/gpu.product`, so GPU pools stay + separate. Each group is sized from the autoscaling bounds of the models placed on it, and a + worker holds as many GPUs as the largest replica on its pool needs. +- Model weights are loaded from the location [governance](#governance-in-mlflow) records for + each revision. + +The `RayService` is generated from the catalog, never hand-edited, and verified by a test and +by CD: + +```bash +python -m llm_router.topology > deploy/overlays/ray/ray-service.yaml +``` + +Under the overlay, engine metrics are scraped from the Ray pods by Prometheus. The gateway +cannot read a whole cluster from one address, so its own `router_engine_*` and `router_gpu_*` +gauges stay empty there and live load does not influence routing. + +GPU support comes from the NVIDIA GPU Operator, installed cluster-wide with +[`deploy/gpu-operator/values.yaml`](deploy/gpu-operator/values.yaml). It provides the +`nvidia.com/gpu` resource, the node label the pools select on, and the DCGM GPU exporter. + +Two Grafana dashboards in [`deploy/kubernetes/dashboards`](deploy/kubernetes/dashboards) ship as +a labelled ConfigMap for Grafana's sidecar: one for the gateway and router, one for engines and +GPUs. A `PrometheusRule` alerts on latency, load shedding, a stuck queue, fallback rate, canary +rollback, an open engine circuit, KV-cache pressure, and GPU memory. A test fails if a dashboard +or alert queries a metric the gateway does not publish. + +None of this has been applied to a cluster. The manifests are schema-validated and +contract-tested; the Ray image, GPU product labels, and node sizes are placeholders to set for +the hardware you have. + Stateless ingress scales separately from GPU replicas. Set `ROUTER_REDIS_URL` so cache and quota state are shared once the gateway runs more than one replica; without it both are in-process and correct for a single replica only. Install the client with the extra: @@ -220,7 +257,8 @@ python -m pip install -e ".[redis]" CD renders the canary plans (one per track, each with its rollback target) and the governance plan, verifies -`config/ray-serve.yaml` against the catalog, and validates the manifests with kubeconform. +`config/ray-serve.yaml` and the Ray topology against the catalog, and validates both rendered +topologies with kubeconform. Applying to a cluster stays disabled until a deployment destination is configured. ## Model registry diff --git a/deploy/gpu-operator/values.yaml b/deploy/gpu-operator/values.yaml new file mode 100644 index 0000000..4bceb39 --- /dev/null +++ b/deploy/gpu-operator/values.yaml @@ -0,0 +1,43 @@ +# Values for the NVIDIA GPU Operator chart (nvidia/gpu-operator). The operator +# is cluster-wide infrastructure, installed once and not part of this +# platform's namespace: +# +# helm upgrade --install gpu-operator nvidia/gpu-operator \ +# --namespace gpu-operator --create-namespace \ +# --values deploy/gpu-operator/values.yaml +# +# It supplies the three things the serving manifests rely on. + +# 1. The nvidia.com/gpu resource that engine and Ray worker pods request. +driver: + enabled: true +toolkit: + enabled: true +devicePlugin: + enabled: true + +# 2. The nvidia.com/gpu.product node label that pins each pool to one +# accelerator type. Node feature discovery finds the hardware and GPU +# feature discovery turns it into the label. +nfd: + enabled: true +gfd: + enabled: true + +# 3. The GPU exporter. DCGM publishes the DCGM_FI_DEV_* series the dashboards +# and alerts read, and the ServiceMonitor hands them to Prometheus. +dcgmExporter: + enabled: true + serviceMonitor: + enabled: true + interval: 15s + additionalLabels: + release: prometheus + +# GPU nodes are tainted so only inference lands on them; the operator's own +# daemons have to tolerate that taint to run there at all. +daemonsets: + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule diff --git a/deploy/kubernetes/dashboards/gateway.json b/deploy/kubernetes/dashboards/gateway.json new file mode 100644 index 0000000..8844596 --- /dev/null +++ b/deploy/kubernetes/dashboards/gateway.json @@ -0,0 +1,449 @@ +{ + "uid": "llm-routing-gateway", + "title": "LLM routing: gateway and router", + "tags": [ + "llm-routing" + ], + "schemaVersion": 39, + "version": 1, + "editable": false, + "refresh": "30s", + "time": { + "from": "now-1h", + "to": "now" + }, + "templating": { + "list": [ + { + "name": "datasource", + "type": "datasource", + "query": "prometheus", + "label": "Data source" + } + ] + }, + "panels": [ + { + "id": 1, + "title": "Requests per second by model and outcome", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "fieldConfig": { + "defaults": { + "unit": "reqps" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (model, outcome) (rate(router_requests_total[5m]))", + "legendFormat": "{{model}} {{outcome}}" + } + ] + }, + { + "id": 2, + "title": "End-to-end latency", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "fieldConfig": { + "defaults": { + "unit": "s" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "histogram_quantile(0.50, sum by (le, model) (rate(router_request_latency_seconds_bucket[5m])))", + "legendFormat": "p50 {{model}}" + }, + { + "refId": "B", + "expr": "histogram_quantile(0.95, sum by (le, model) (rate(router_request_latency_seconds_bucket[5m])))", + "legendFormat": "p95 {{model}}" + }, + { + "refId": "C", + "expr": "histogram_quantile(0.99, sum by (le, model) (rate(router_request_latency_seconds_bucket[5m])))", + "legendFormat": "p99 {{model}}" + } + ] + }, + { + "id": 3, + "title": "Time to first token (p95)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "fieldConfig": { + "defaults": { + "unit": "s" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "histogram_quantile(0.95, sum by (le, model) (rate(router_time_to_first_token_seconds_bucket[5m])))", + "legendFormat": "{{model}}" + } + ] + }, + { + "id": 4, + "title": "Time per output token (p95)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "fieldConfig": { + "defaults": { + "unit": "s" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "histogram_quantile(0.95, sum by (le, model) (rate(router_time_per_output_token_seconds_bucket[5m])))", + "legendFormat": "{{model}}" + } + ] + }, + { + "id": 5, + "title": "Tokens per second", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (model, kind) (rate(router_tokens_total[5m]))", + "legendFormat": "{{model}} {{kind}}" + } + ] + }, + { + "id": 6, + "title": "In-flight and queued requests", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 16 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum(router_inflight_requests)", + "legendFormat": "in flight" + }, + { + "refId": "B", + "expr": "sum(router_queued_requests)", + "legendFormat": "queued" + } + ] + }, + { + "id": 7, + "title": "Routes by model and privacy class", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 24 + }, + "fieldConfig": { + "defaults": { + "unit": "reqps" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (model, privacy) (rate(router_routes_total[5m]))", + "legendFormat": "{{model}} {{privacy}}" + } + ] + }, + { + "id": 8, + "title": "Rejections by type", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 24 + }, + "fieldConfig": { + "defaults": { + "unit": "reqps" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (type) (rate(router_rejections_total[5m]))", + "legendFormat": "{{type}}" + } + ] + }, + { + "id": 9, + "title": "Fallbacks", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 32 + }, + "fieldConfig": { + "defaults": { + "unit": "reqps" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (from_model, to_model, cause) (rate(router_fallbacks_total[5m]))", + "legendFormat": "{{from_model}} to {{to_model}} ({{cause}})" + }, + { + "refId": "B", + "expr": "sum by (model) (rate(router_external_fallback_total[5m]))", + "legendFormat": "external {{model}}" + } + ] + }, + { + "id": 10, + "title": "Cache hit ratio", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 32 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (cache) (rate(router_cache_events_total{result=\"hit\"}[5m])) / sum by (cache) (rate(router_cache_events_total[5m]))", + "legendFormat": "{{cache}}" + } + ] + }, + { + "id": 11, + "title": "Predicted versus observed quality", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 40 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (model) (router_predicted_quality)", + "legendFormat": "predicted {{model}}" + }, + { + "refId": "B", + "expr": "avg by (model) (router_observed_quality)", + "legendFormat": "observed {{model}}" + } + ] + }, + { + "id": 12, + "title": "Queue-delay prediction error", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 40 + }, + "fieldConfig": { + "defaults": { + "unit": "ms" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (model) (router_queue_delay_prediction_error_ms)", + "legendFormat": "{{model}}" + } + ] + }, + { + "id": 13, + "title": "Canary traffic and rollbacks", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 48 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (subject, arm, outcome) (rate(router_canary_requests_total[5m]))", + "legendFormat": "{{subject}} {{arm}} {{outcome}}" + }, + { + "refId": "B", + "expr": "sum by (subject) (increase(router_canary_rollbacks_total[1h]))", + "legendFormat": "rollback {{subject}}" + } + ] + }, + { + "id": 14, + "title": "Structured-output validity", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 48 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (model) (rate(router_structured_output_total{result=\"valid\"}[5m])) / sum by (model) (rate(router_structured_output_total[5m]))", + "legendFormat": "{{model}}" + } + ] + } + ] +} diff --git a/deploy/kubernetes/dashboards/serving.json b/deploy/kubernetes/dashboards/serving.json new file mode 100644 index 0000000..f931fee --- /dev/null +++ b/deploy/kubernetes/dashboards/serving.json @@ -0,0 +1,317 @@ +{ + "uid": "llm-routing-serving", + "title": "LLM routing: engines and GPUs", + "tags": [ + "llm-routing" + ], + "schemaVersion": 39, + "version": 1, + "editable": false, + "refresh": "30s", + "time": { + "from": "now-1h", + "to": "now" + }, + "templating": { + "list": [ + { + "name": "datasource", + "type": "datasource", + "query": "prometheus", + "label": "Data source" + } + ] + }, + "panels": [ + { + "id": 1, + "title": "Engine running and waiting requests", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (engine) (router_engine_running_requests)", + "legendFormat": "running {{engine}}" + }, + { + "refId": "B", + "expr": "sum by (engine) (router_engine_waiting_requests)", + "legendFormat": "waiting {{engine}}" + } + ] + }, + { + "id": 2, + "title": "Engine batch size", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (engine) (router_engine_batch_size)", + "legendFormat": "{{engine}}" + } + ] + }, + { + "id": 3, + "title": "KV-cache occupancy", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "max by (engine) (router_engine_kv_cache_occupancy_ratio)", + "legendFormat": "{{engine}}" + } + ] + }, + { + "id": 4, + "title": "Engine preemptions", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (engine) (rate(router_engine_preemptions_total[5m]))", + "legendFormat": "{{engine}}" + } + ] + }, + { + "id": 5, + "title": "Engine circuit state (0 closed, 0.5 half-open, 1 open)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "max by (engine) (router_engine_circuit_open)", + "legendFormat": "{{engine}}" + } + ] + }, + { + "id": 6, + "title": "Model load and cold start (p95)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 16 + }, + "fieldConfig": { + "defaults": { + "unit": "s" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "histogram_quantile(0.95, sum by (le, model) (rate(router_model_load_seconds_bucket[5m])))", + "legendFormat": "{{model}}" + } + ] + }, + { + "id": 7, + "title": "GPU utilization as seen by the gateway", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 24 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (gpu) (router_gpu_utilization_ratio)", + "legendFormat": "gpu {{gpu}}" + } + ] + }, + { + "id": 8, + "title": "GPU memory as seen by the gateway", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 24 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (gpu) (router_gpu_memory_used_bytes) / sum by (gpu) (router_gpu_memory_total_bytes)", + "legendFormat": "gpu {{gpu}}" + } + ] + }, + { + "id": 9, + "title": "GPU utilization by node (DCGM exporter)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 32 + }, + "fieldConfig": { + "defaults": { + "unit": "percent" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (Hostname, gpu) (DCGM_FI_DEV_GPU_UTIL)", + "legendFormat": "{{Hostname}} gpu {{gpu}}" + } + ] + }, + { + "id": 10, + "title": "GPU memory by node (DCGM exporter)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 32 + }, + "fieldConfig": { + "defaults": { + "unit": "decmbytes" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (Hostname, gpu) (DCGM_FI_DEV_FB_USED)", + "legendFormat": "used {{Hostname}} gpu {{gpu}}" + }, + { + "refId": "B", + "expr": "sum by (Hostname, gpu) (DCGM_FI_DEV_FB_FREE)", + "legendFormat": "free {{Hostname}} gpu {{gpu}}" + } + ] + } + ] +} diff --git a/deploy/kubernetes/kustomization.yaml b/deploy/kubernetes/kustomization.yaml index 6503194..9f8b7fd 100644 --- a/deploy/kubernetes/kustomization.yaml +++ b/deploy/kubernetes/kustomization.yaml @@ -11,3 +11,13 @@ resources: - external-provider.yaml - mlflow.yaml - observability.yaml +# Grafana's dashboard sidecar loads any ConfigMap carrying this label. +configMapGenerator: + - name: llm-routing-dashboards + files: + - dashboards/gateway.json + - dashboards/serving.json + options: + disableNameSuffixHash: true + labels: + grafana_dashboard: "1" diff --git a/deploy/kubernetes/observability.yaml b/deploy/kubernetes/observability.yaml index a2e19f7..1705938 100644 --- a/deploy/kubernetes/observability.yaml +++ b/deploy/kubernetes/observability.yaml @@ -14,3 +14,74 @@ spec: - port: http path: /metrics interval: 15s +--- +# Alerts on the conditions the platform promises to make visible: overload, +# an open engine circuit, a canary that rolled back, and a GPU out of memory. +apiVersion: monitoring.coreos.com/v1 +kind: PrometheusRule +metadata: + name: llm-routing + namespace: llm-routing + labels: + release: prometheus +spec: + groups: + - name: llm-routing.gateway + rules: + - alert: GatewayLatencyHigh + expr: histogram_quantile(0.95, sum by (le, model) (rate(router_request_latency_seconds_bucket[5m]))) > 5 + for: 10m + labels: + severity: warning + annotations: + summary: p95 latency for {{ $labels.model }} has been above 5s for 10 minutes. + - alert: GatewayRejectingRequests + expr: sum by (type) (rate(router_rejections_total{type=~"overloaded|engine_unavailable"}[5m])) > 1 + for: 5m + labels: + severity: warning + annotations: + summary: The gateway is shedding load ({{ $labels.type }}). + - alert: GatewayQueueNotDraining + expr: sum(router_queued_requests) > 16 + for: 10m + labels: + severity: warning + annotations: + summary: Requests have been queueing for admission for 10 minutes. + - alert: CanaryRolledBack + expr: increase(router_canary_rollbacks_total[15m]) > 0 + labels: + severity: warning + annotations: + summary: The canary for {{ $labels.subject }} failed its criteria and was rolled back. + - alert: FallbackRateHigh + expr: sum(rate(router_fallbacks_total[10m])) / sum(rate(router_requests_total[10m])) > 0.05 + for: 10m + labels: + severity: warning + annotations: + summary: More than 5% of requests are being answered by a fallback model. + - name: llm-routing.serving + rules: + - alert: EngineCircuitOpen + expr: max by (engine) (router_engine_circuit_open) == 1 + for: 2m + labels: + severity: critical + annotations: + summary: The circuit for engine {{ $labels.engine }} is open; requests to it are failing fast. + - alert: KVCacheNearlyFull + expr: max by (engine) (router_engine_kv_cache_occupancy_ratio) > 0.95 + for: 10m + labels: + severity: warning + annotations: + summary: KV-cache occupancy on {{ $labels.engine }} has been above 95% for 10 minutes. + - alert: GPUMemoryNearlyFull + expr: DCGM_FI_DEV_FB_USED / (DCGM_FI_DEV_FB_USED + DCGM_FI_DEV_FB_FREE) > 0.97 + for: 10m + labels: + severity: warning + annotations: + summary: GPU {{ $labels.gpu }} on {{ $labels.Hostname }} is nearly out of memory. diff --git a/deploy/overlays/ray/gateway-network-policy.patch.yaml b/deploy/overlays/ray/gateway-network-policy.patch.yaml new file mode 100644 index 0000000..1fba3a2 --- /dev/null +++ b/deploy/overlays/ray/gateway-network-policy.patch.yaml @@ -0,0 +1,30 @@ +# The gateway's egress with the Ray Serve pods in place of the single engine. +# A list of egress rules has no merge key, so the whole list is restated. +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: llm-gateway + namespace: llm-routing +spec: + egress: + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: ray-serve + ports: + - protocol: TCP + port: 8000 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: redis + ports: + - protocol: TCP + port: 6379 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: litellm-proxy + ports: + - protocol: TCP + port: 4000 diff --git a/deploy/overlays/ray/kustomization.yaml b/deploy/overlays/ray/kustomization.yaml new file mode 100644 index 0000000..7848da1 --- /dev/null +++ b/deploy/overlays/ray/kustomization.yaml @@ -0,0 +1,51 @@ +# Ray Serve control plane in place of the single engine Deployment. +# +# The base runs one vLLM engine, which can serve one model. This overlay swaps +# it for a RayService with a head and one worker group per accelerator type, +# so every local model in the catalog is served from its own GPU pool behind +# one OpenAI-compatible endpoint. Requires the KubeRay operator. +apiVersion: kustomize.config.k8s.io/v1beta1 +kind: Kustomization +resources: + - ../../kubernetes + - ray-service.yaml + - network-policy.yaml + - monitoring.yaml +patches: + - patch: | + $patch: delete + apiVersion: apps/v1 + kind: Deployment + metadata: + name: vllm-serve + namespace: llm-routing + - patch: | + $patch: delete + apiVersion: v1 + kind: Service + metadata: + name: vllm-serve + namespace: llm-routing + - patch: | + $patch: delete + apiVersion: networking.k8s.io/v1 + kind: NetworkPolicy + metadata: + name: vllm-serve + namespace: llm-routing + # KubeRay publishes the Serve endpoint as -serve-svc. + - patch: | + apiVersion: apps/v1 + kind: Deployment + metadata: + name: llm-gateway + namespace: llm-routing + spec: + template: + spec: + containers: + - name: gateway + env: + - name: ROUTER_VLLM_BASE_URL + value: http://llm-serve-serve-svc.llm-routing.svc.cluster.local:8000 + - path: gateway-network-policy.patch.yaml diff --git a/deploy/overlays/ray/monitoring.yaml b/deploy/overlays/ray/monitoring.yaml new file mode 100644 index 0000000..b367cd1 --- /dev/null +++ b/deploy/overlays/ray/monitoring.yaml @@ -0,0 +1,18 @@ +# Ray exports its own metrics and, with log_engine_metrics, each vLLM engine's +# metrics on every head and worker pod. Under this overlay engine telemetry is +# read from here by Prometheus; the gateway cannot scrape a whole cluster from +# one address, so its own router_engine_* gauges stay empty. +apiVersion: monitoring.coreos.com/v1 +kind: PodMonitor +metadata: + name: ray-serve + namespace: llm-routing + labels: + release: prometheus +spec: + selector: + matchLabels: + app.kubernetes.io/name: ray-serve + podMetricsEndpoints: + - port: metrics + interval: 15s diff --git a/deploy/overlays/ray/network-policy.yaml b/deploy/overlays/ray/network-policy.yaml new file mode 100644 index 0000000..01f9408 --- /dev/null +++ b/deploy/overlays/ray/network-policy.yaml @@ -0,0 +1,65 @@ +# Ray head and workers. Inference traffic comes only from the gateway; the +# cluster talks to itself freely; the operator drives it through the dashboard. +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: ray-serve + namespace: llm-routing +spec: + podSelector: + matchLabels: + app.kubernetes.io/name: ray-serve + policyTypes: [Ingress, Egress] + ingress: + - from: + - podSelector: + matchLabels: + app.kubernetes.io/name: llm-gateway + ports: + - protocol: TCP + port: 8000 + # Head and workers exchange control, object and worker traffic on ports + # Ray assigns at start-up, so the cluster is open to itself. + - from: + - podSelector: + matchLabels: + app.kubernetes.io/name: ray-serve + # KubeRay submits the Serve configuration and reads health here. + - from: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: kuberay-system + ports: + - protocol: TCP + port: 8265 + - protocol: TCP + port: 52365 + - from: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: monitoring + ports: + - protocol: TCP + port: 8080 + egress: + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: ray-serve + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: kube-system + ports: + - protocol: UDP + port: 53 + - protocol: TCP + port: 53 + # Model and adapter artifacts, from the store governance records them in. + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: storage + ports: + - protocol: TCP + port: 9000 diff --git a/deploy/overlays/ray/ray-service.yaml b/deploy/overlays/ray/ray-service.yaml new file mode 100644 index 0000000..2208612 --- /dev/null +++ b/deploy/overlays/ray/ray-service.yaml @@ -0,0 +1,386 @@ +# Generated from config/registry.yaml by `python -m llm_router.topology`. +# Do not edit: change the catalog and regenerate. A test and CD both fail on drift. +apiVersion: ray.io/v1 +kind: RayService +metadata: + name: llm-serve + namespace: llm-routing + labels: + app.kubernetes.io/part-of: local-llm-router + annotations: + llm-routing/policy-version: v1 +spec: + serveConfigV2: | + applications: + - name: llm_app + route_prefix: / + import_path: ray.serve.llm:build_openai_app + args: + llm_configs: + - model_loading_config: + model_id: small-specialist + model_source: s3://llm-routing-artifacts/models/small-specialist/mock-small-sha256-dev + accelerator_type: L4 + deployment_config: + autoscaling_config: + min_replicas: 1 + max_replicas: 4 + target_ongoing_requests: 8 + upscale_delay_s: 10 + downscale_delay_s: 300 + max_ongoing_requests: 16 + ray_actor_options: + num_gpus: 1 + resources: + gpu_pool_nvidia-l4: 0.001 + engine_kwargs: + max_model_len: 8192 + tensor_parallel_size: 1 + enable_prefix_caching: true + enable_chunked_prefill: true + quantization: awq + enable_lora: true + max_loras: 3 + max_lora_rank: 32 + log_engine_metrics: true + lora_config: + dynamic_lora_loading_path: s3://llm-routing-artifacts/adapters + max_num_adapters_per_replica: 3 + - model_loading_config: + model_id: general-local + model_source: s3://llm-routing-artifacts/models/general-local/mock-general-sha256-dev + accelerator_type: A10G + deployment_config: + autoscaling_config: + min_replicas: 1 + max_replicas: 4 + target_ongoing_requests: 8 + upscale_delay_s: 10 + downscale_delay_s: 300 + max_ongoing_requests: 16 + ray_actor_options: + num_gpus: 1 + resources: + gpu_pool_nvidia-a10g: 0.001 + engine_kwargs: + max_model_len: 32768 + tensor_parallel_size: 1 + enable_prefix_caching: true + enable_chunked_prefill: true + log_engine_metrics: true + - model_loading_config: + model_id: high-capability + model_source: s3://llm-routing-artifacts/models/high-capability/mock-high-sha256-dev + accelerator_type: A100 + deployment_config: + autoscaling_config: + min_replicas: 0 + max_replicas: 2 + target_ongoing_requests: 8 + upscale_delay_s: 10 + downscale_delay_s: 60 + max_ongoing_requests: 16 + ray_actor_options: + num_gpus: 2 + resources: + gpu_pool_nvidia-a100: 0.001 + engine_kwargs: + max_model_len: 65536 + tensor_parallel_size: 2 + enable_prefix_caching: true + enable_chunked_prefill: true + log_engine_metrics: true + - model_loading_config: + model_id: general-local--general-gptq + model_source: s3://llm-routing-artifacts/models/general-local/mock-general-sha256-dev + accelerator_type: A10G + deployment_config: + autoscaling_config: + min_replicas: 0 + max_replicas: 1 + ray_actor_options: + num_gpus: 1 + resources: + gpu_pool_nvidia-a10g: 0.001 + engine_kwargs: + max_model_len: 32768 + tensor_parallel_size: 1 + enable_prefix_caching: true + enable_chunked_prefill: true + quantization: gptq + log_engine_metrics: true + - model_loading_config: + model_id: high-capability--high-capability-speculative + model_source: s3://llm-routing-artifacts/models/high-capability/mock-high-sha256-dev + accelerator_type: A100 + deployment_config: + autoscaling_config: + min_replicas: 0 + max_replicas: 1 + ray_actor_options: + num_gpus: 2 + resources: + gpu_pool_nvidia-a100: 0.001 + engine_kwargs: + max_model_len: 65536 + tensor_parallel_size: 2 + enable_prefix_caching: true + enable_chunked_prefill: true + speculative_config: + model: s3://llm-routing-artifacts/models/small-specialist/mock-small-sha256-dev + num_speculative_tokens: 5 + log_engine_metrics: true + rayClusterConfig: + rayVersion: 2.51.0 + enableInTreeAutoscaling: true + headGroupSpec: + rayStartParams: + dashboard-host: 0.0.0.0 + metrics-export-port: '8080' + num-cpus: '0' + template: + metadata: + labels: + app.kubernetes.io/name: ray-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 1000 + seccompProfile: + type: RuntimeDefault + containers: + - name: ray-head + image: docker.io/rayproject/ray-llm@sha256:REPLACE_ME + imagePullPolicy: IfNotPresent + ports: + - name: metrics + containerPort: 8080 + - name: gcs + containerPort: 6379 + - name: dashboard + containerPort: 8265 + - name: serve + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: + - ALL + resources: + requests: + cpu: '2' + memory: 8Gi + limits: + cpu: '2' + memory: 8Gi + volumeMounts: + - name: tmp + mountPath: /tmp + - name: cache + mountPath: /home/ray/.cache + - name: shm + mountPath: /dev/shm + volumes: + - name: tmp + emptyDir: {} + - name: cache + emptyDir: {} + - name: shm + emptyDir: + medium: Memory + terminationGracePeriodSeconds: 60 + workerGroupSpecs: + - groupName: a100-pool + replicas: 0 + minReplicas: 0 + maxReplicas: 3 + rayStartParams: + metrics-export-port: '8080' + resources: '"{\"gpu_pool_nvidia-a100\": 1}"' + template: + metadata: + labels: + app.kubernetes.io/name: ray-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 1000 + seccompProfile: + type: RuntimeDefault + containers: + - name: ray-worker + image: docker.io/rayproject/ray-llm@sha256:REPLACE_ME + imagePullPolicy: IfNotPresent + ports: + - name: metrics + containerPort: 8080 + - name: serve + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: + - ALL + resources: + requests: + cpu: '8' + memory: 160Gi + nvidia.com/gpu: '2' + limits: + cpu: '8' + memory: 160Gi + nvidia.com/gpu: '2' + volumeMounts: + - name: tmp + mountPath: /tmp + - name: cache + mountPath: /home/ray/.cache + - name: shm + mountPath: /dev/shm + volumes: + - name: tmp + emptyDir: {} + - name: cache + emptyDir: {} + - name: shm + emptyDir: + medium: Memory + terminationGracePeriodSeconds: 120 + nodeSelector: + nvidia.com/gpu.product: NVIDIA-A100-SXM4-80GB + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule + - groupName: a10g-pool + replicas: 1 + minReplicas: 1 + maxReplicas: 5 + rayStartParams: + metrics-export-port: '8080' + resources: '"{\"gpu_pool_nvidia-a10g\": 1}"' + template: + metadata: + labels: + app.kubernetes.io/name: ray-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 1000 + seccompProfile: + type: RuntimeDefault + containers: + - name: ray-worker + image: docker.io/rayproject/ray-llm@sha256:REPLACE_ME + imagePullPolicy: IfNotPresent + ports: + - name: metrics + containerPort: 8080 + - name: serve + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: + - ALL + resources: + requests: + cpu: '4' + memory: 48Gi + nvidia.com/gpu: '1' + limits: + cpu: '4' + memory: 48Gi + nvidia.com/gpu: '1' + volumeMounts: + - name: tmp + mountPath: /tmp + - name: cache + mountPath: /home/ray/.cache + - name: shm + mountPath: /dev/shm + volumes: + - name: tmp + emptyDir: {} + - name: cache + emptyDir: {} + - name: shm + emptyDir: + medium: Memory + terminationGracePeriodSeconds: 120 + nodeSelector: + nvidia.com/gpu.product: NVIDIA-A10G + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule + - groupName: l4-pool + replicas: 1 + minReplicas: 1 + maxReplicas: 4 + rayStartParams: + metrics-export-port: '8080' + resources: '"{\"gpu_pool_nvidia-l4\": 1}"' + template: + metadata: + labels: + app.kubernetes.io/name: ray-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 1000 + seccompProfile: + type: RuntimeDefault + containers: + - name: ray-worker + image: docker.io/rayproject/ray-llm@sha256:REPLACE_ME + imagePullPolicy: IfNotPresent + ports: + - name: metrics + containerPort: 8080 + - name: serve + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: + - ALL + resources: + requests: + cpu: '4' + memory: 24Gi + nvidia.com/gpu: '1' + limits: + cpu: '4' + memory: 24Gi + nvidia.com/gpu: '1' + volumeMounts: + - name: tmp + mountPath: /tmp + - name: cache + mountPath: /home/ray/.cache + - name: shm + mountPath: /dev/shm + volumes: + - name: tmp + emptyDir: {} + - name: cache + emptyDir: {} + - name: shm + emptyDir: + medium: Memory + terminationGracePeriodSeconds: 120 + nodeSelector: + nvidia.com/gpu.product: NVIDIA-L4 + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule diff --git a/src/llm_router/topology.py b/src/llm_router/topology.py new file mode 100644 index 0000000..4f0dd72 --- /dev/null +++ b/src/llm_router/topology.py @@ -0,0 +1,329 @@ +"""Ray head and worker topology derived from the registry (sections 7.3 and 17). + +One KubeRay ``RayService`` holds the whole serving plane: a head that +schedules and runs no model, and one worker group per accelerator type, so +GPU pools stay separate and each is sized from the models placed on it. Like +the serving configuration, the manifest is generated from the catalog and +never edited by hand. +""" + +import json +from collections.abc import Sequence +from typing import Any + +import yaml +from pydantic import BaseModel + +from llm_router.governance import DEFAULT_ARTIFACT_ROOT, governed_versions +from llm_router.registry import Registry +from llm_router.serving import build_serving_config + +NAMESPACE = "llm-routing" +SERVICE_NAME = "llm-serve" +POD_LABEL = "ray-serve" +RAY_VERSION = "2.51.0" +RAY_IMAGE = "docker.io/rayproject/ray-llm@sha256:REPLACE_ME" +SERVE_PORT = 8000 +METRICS_PORT = 8080 +DASHBOARD_PORT = 8265 +# Node labels published by GPU feature discovery. They differ between clouds +# and card variants, so anything not listed falls back to NVIDIA-. +GPU_PRODUCT_LABELS = {"A100": "NVIDIA-A100-SXM4-80GB"} + + +class _Block(str): + """A string rendered as a YAML literal block, so embedded YAML stays readable.""" + + +class _Dumper(yaml.SafeDumper): + pass + + +_Dumper.add_representer( + _Block, + lambda dumper, value: dumper.represent_scalar("tag:yaml.org,2002:str", value, style="|"), +) + + +class WorkerPool(BaseModel): + """The capacity one accelerator type needs for the models placed on it.""" + + accelerator: str + ray_accelerator: str + gpus_per_worker: int + memory_gb: int + min_workers: int + max_workers: int + models: tuple[str, ...] + + @property + def group_name(self) -> str: + return f"{self.ray_accelerator.lower()}-pool" + + @property + def resource(self) -> str: + return f"gpu_pool_{self.accelerator}" + + +def ray_accelerator(accelerator: str) -> str: + """Translate a catalog accelerator into the name Ray knows it by.""" + + return accelerator.removeprefix("nvidia-").upper() + + +def worker_pools(registry: Registry) -> tuple[WorkerPool, ...]: + """One pool per accelerator type, sized from its models' autoscaling bounds. + + A worker holds as many GPUs as the largest replica on its pool needs, so a + tensor-parallel model always fits on one node. Experiments add headroom + to the ceiling but never to the floor: they are never kept warm. + """ + + config = build_serving_config(registry) + pools: dict[str, dict[str, Any]] = {} + for entry in (*config["applications"], *config.get("experiments", ())): + card = registry.model_card(entry.get("base_model_id", entry["model_id"])) + scaling = entry["deployment_config"]["autoscaling_config"] + pool = pools.setdefault( + card.hardware.accelerator, + {"gpus": 0, "memory": 0, "min": 0, "max": 0, "models": []}, + ) + pool["gpus"] = max(pool["gpus"], card.hardware.count) + pool["memory"] = max(pool["memory"], card.hardware.minimum_memory_gb) + pool["min"] += scaling["min_replicas"] + pool["max"] += scaling["max_replicas"] + pool["models"].append(entry["model_id"]) + return tuple( + WorkerPool( + accelerator=accelerator, + ray_accelerator=ray_accelerator(accelerator), + gpus_per_worker=pool["gpus"], + memory_gb=pool["memory"], + min_workers=pool["min"], + max_workers=pool["max"], + models=tuple(pool["models"]), + ) + for accelerator, pool in sorted(pools.items()) + ) + + +def build_serve_application( + registry: Registry, artifact_root: str = DEFAULT_ARTIFACT_ROOT +) -> dict[str, Any]: + """The Ray Serve LLM application: one OpenAI-compatible endpoint for every model. + + Only fields Ray's ``LLMConfig`` accepts are kept. Model weights are read + from the location governance records for each revision, so what Ray loads + is what MLflow says was promoted. + """ + + root = artifact_root.rstrip("/") + sources = {item.name: item.source for item in governed_versions(registry, root)} + config = build_serving_config(registry) + llm_configs: list[dict[str, Any]] = [] + for entry in (*config["applications"], *config.get("experiments", ())): + base = entry.get("base_model_id", entry["model_id"]) + card = registry.model_card(base) + deployment = { + **entry["deployment_config"], + "ray_actor_options": { + "num_gpus": card.hardware.count, + "resources": {f"gpu_pool_{card.hardware.accelerator}": 0.001}, + }, + } + llm_config: dict[str, Any] = { + "model_loading_config": {"model_id": entry["model_id"], "model_source": sources[base]}, + "accelerator_type": ray_accelerator(card.hardware.accelerator), + "deployment_config": deployment, + "engine_kwargs": _engine_kwargs(entry["engine_kwargs"], sources), + "log_engine_metrics": True, + } + if "lora_config" in entry: + llm_config["lora_config"] = { + # Ray resolves /; the deploy step publishes + # each promoted adapter revision there. + "dynamic_lora_loading_path": f"{root}/adapters", + "max_num_adapters_per_replica": len(entry["lora_config"]["adapters"]), + } + llm_configs.append(llm_config) + return { + "applications": [ + { + "name": "llm_app", + "route_prefix": "/", + "import_path": "ray.serve.llm:build_openai_app", + "args": {"llm_configs": llm_configs}, + } + ] + } + + +def _engine_kwargs(engine: dict[str, Any], sources: dict[str, str]) -> dict[str, Any]: + """Point a speculative draft model at its recorded artifact as well.""" + + speculative = engine.get("speculative_config") + if speculative is None: + return engine + draft = str(speculative["model"]).removeprefix("registry://").split("@", 1)[0] + return {**engine, "speculative_config": {**speculative, "model": sources[draft]}} + + +def _pod( + *, + image: str, + cpu: str, + memory: str, + gpus: int = 0, + node_selector: dict[str, str] | None = None, + head: bool = False, +) -> dict[str, Any]: + resources: dict[str, str] = {"cpu": cpu, "memory": memory} + if gpus: + resources["nvidia.com/gpu"] = str(gpus) + ports = [{"name": "metrics", "containerPort": METRICS_PORT}] + if head: + ports += [ + {"name": "gcs", "containerPort": 6379}, + {"name": "dashboard", "containerPort": DASHBOARD_PORT}, + {"name": "serve", "containerPort": SERVE_PORT}, + ] + else: + ports.append({"name": "serve", "containerPort": SERVE_PORT}) + spec: dict[str, Any] = { + "securityContext": { + "runAsNonRoot": True, + "runAsUser": 1000, + "seccompProfile": {"type": "RuntimeDefault"}, + }, + "containers": [ + { + "name": "ray-head" if head else "ray-worker", + "image": image, + "imagePullPolicy": "IfNotPresent", + "ports": ports, + "securityContext": { + "allowPrivilegeEscalation": False, + "readOnlyRootFilesystem": True, + "capabilities": {"drop": ["ALL"]}, + }, + "resources": {"requests": dict(resources), "limits": dict(resources)}, + "volumeMounts": [ + {"name": "tmp", "mountPath": "/tmp"}, + {"name": "cache", "mountPath": "/home/ray/.cache"}, + {"name": "shm", "mountPath": "/dev/shm"}, + ], + } + ], + "volumes": [ + {"name": "tmp", "emptyDir": {}}, + {"name": "cache", "emptyDir": {}}, + {"name": "shm", "emptyDir": {"medium": "Memory"}}, + ], + # Long enough for a replica to drain its in-flight generations. + "terminationGracePeriodSeconds": 60 if head else 120, + } + if node_selector: + spec["nodeSelector"] = node_selector + spec["tolerations"] = [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ] + return { + "metadata": { + "labels": { + "app.kubernetes.io/name": POD_LABEL, + "app.kubernetes.io/part-of": "local-llm-router", + } + }, + "spec": spec, + } + + +def _worker_group(pool: WorkerPool, image: str) -> dict[str, Any]: + product = GPU_PRODUCT_LABELS.get(pool.ray_accelerator, f"NVIDIA-{pool.ray_accelerator}") + return { + "groupName": pool.group_name, + "replicas": pool.min_workers, + "minReplicas": pool.min_workers, + "maxReplicas": pool.max_workers, + "rayStartParams": { + "metrics-export-port": str(METRICS_PORT), + # The custom resource pins a model to its own pool even when two + # pools could both satisfy a bare GPU request. + "resources": json.dumps(json.dumps({pool.resource: 1})), + }, + "template": _pod( + image=image, + cpu=str(4 * pool.gpus_per_worker), + memory=f"{pool.memory_gb}Gi", + gpus=pool.gpus_per_worker, + node_selector={"nvidia.com/gpu.product": product}, + ), + } + + +def build_ray_service( + registry: Registry, + artifact_root: str = DEFAULT_ARTIFACT_ROOT, + image: str = RAY_IMAGE, +) -> dict[str, Any]: + """Render the RayService that serves every local model in the catalog.""" + + serve_config = yaml.safe_dump(build_serve_application(registry, artifact_root), sort_keys=False) + return { + "apiVersion": "ray.io/v1", + "kind": "RayService", + "metadata": { + "name": SERVICE_NAME, + "namespace": NAMESPACE, + "labels": {"app.kubernetes.io/part-of": "local-llm-router"}, + "annotations": {"llm-routing/policy-version": registry.policy.version}, + }, + "spec": { + "serveConfigV2": _Block(serve_config), + "rayClusterConfig": { + "rayVersion": RAY_VERSION, + "enableInTreeAutoscaling": True, + "headGroupSpec": { + "rayStartParams": { + "dashboard-host": "0.0.0.0", + "metrics-export-port": str(METRICS_PORT), + # The head schedules; it never runs a replica. + "num-cpus": "0", + }, + "template": _pod(image=image, cpu="2", memory="8Gi", head=True), + }, + "workerGroupSpecs": [_worker_group(pool, image) for pool in worker_pools(registry)], + }, + }, + } + + +HEADER = ( + "# Generated from config/registry.yaml by `python -m llm_router.topology`.\n" + "# Do not edit: change the catalog and regenerate. A test and CD both fail on drift.\n" +) + + +def render_ray_service(registry: Registry, artifact_root: str = DEFAULT_ARTIFACT_ROOT) -> str: + manifest = build_ray_service(registry, artifact_root) + return HEADER + yaml.dump(manifest, Dumper=_Dumper, sort_keys=False, width=1000) + + +def main(argv: Sequence[str] | None = None) -> int: + """Print the RayService manifest for the committed catalog.""" + + import argparse + import sys + + from llm_router.registry import load_registry + + parser = argparse.ArgumentParser(description=main.__doc__) + parser.add_argument("--catalog", default="config/registry.yaml") + parser.add_argument("--artifact-root", default=DEFAULT_ARTIFACT_ROOT) + arguments = parser.parse_args(argv) + sys.stdout.write(render_ray_service(load_registry(arguments.catalog), arguments.artifact_root)) + return 0 + + +if __name__ == "__main__": # pragma: no cover - command-line entry point + raise SystemExit(main()) diff --git a/tests/unit/test_topology.py b/tests/unit/test_topology.py new file mode 100644 index 0000000..bbd1bd7 --- /dev/null +++ b/tests/unit/test_topology.py @@ -0,0 +1,372 @@ +"""The Ray head and worker topology, GPU support, dashboards and alerts (section 17).""" + +import json +import re +import shutil +import subprocess +from pathlib import Path +from typing import Any + +import pytest +import yaml +from fastapi.testclient import TestClient + +from llm_router.app import create_app +from llm_router.config import Settings +from llm_router.governance import governed_versions +from llm_router.registry import load_registry +from llm_router.topology import ( + build_ray_service, + build_serve_application, + main, + ray_accelerator, + render_ray_service, + worker_pools, +) + +CATALOG = load_registry("config/registry.yaml") +BASE = Path("deploy/kubernetes") +OVERLAY = Path("deploy/overlays/ray") +SERVICE = build_ray_service(CATALOG) +CLUSTER = SERVICE["spec"]["rayClusterConfig"] +# The fields Ray's LLMConfig accepts; anything else is rejected at deploy time. +LLM_CONFIG_FIELDS = { + "model_loading_config", + "engine_kwargs", + "accelerator_type", + "deployment_config", + "lora_config", + "log_engine_metrics", +} + + +def llm_configs() -> dict[str, dict[str, Any]]: + application = build_serve_application(CATALOG)["applications"][0] + return { + item["model_loading_config"]["model_id"]: item + for item in application["args"]["llm_configs"] + } + + +def worker_group(name: str) -> dict[str, Any]: + return next(group for group in CLUSTER["workerGroupSpecs"] if group["groupName"] == name) + + +def pod_templates() -> list[dict[str, Any]]: + return [ + CLUSTER["headGroupSpec"]["template"], + *(group["template"] for group in CLUSTER["workerGroupSpecs"]), + ] + + +def test_accelerators_are_named_the_way_ray_knows_them() -> None: + assert ray_accelerator("nvidia-l4") == "L4" + assert ray_accelerator("nvidia-a10g") == "A10G" + + +def test_each_accelerator_type_gets_its_own_pool_sized_from_its_models() -> None: + pools = {pool.group_name: pool for pool in worker_pools(CATALOG)} + + assert set(pools) == {"l4-pool", "a10g-pool", "a100-pool"} + assert (pools["l4-pool"].min_workers, pools["l4-pool"].max_workers) == (1, 4) + # The experiment adds headroom to the ceiling and nothing to the floor. + assert (pools["a10g-pool"].min_workers, pools["a10g-pool"].max_workers) == (1, 5) + assert pools["a10g-pool"].models == ("general-local", "general-local--general-gptq") + # The high-capability tier scales to zero and needs both GPUs on one node. + assert (pools["a100-pool"].min_workers, pools["a100-pool"].max_workers) == (0, 3) + assert pools["a100-pool"].gpus_per_worker == 2 + + +def test_the_head_schedules_and_never_runs_a_model() -> None: + head = CLUSTER["headGroupSpec"] + container = head["template"]["spec"]["containers"][0] + + assert head["rayStartParams"]["num-cpus"] == "0" + assert "nvidia.com/gpu" not in container["resources"]["limits"] + assert "nodeSelector" not in head["template"]["spec"] + assert CLUSTER["enableInTreeAutoscaling"] is True + + +def test_workers_are_pinned_to_their_accelerator_and_advertise_their_pool() -> None: + group = worker_group("a100-pool") + spec = group["template"]["spec"] + + assert spec["nodeSelector"] == {"nvidia.com/gpu.product": "NVIDIA-A100-SXM4-80GB"} + assert spec["tolerations"][0]["key"] == "nvidia.com/gpu" + assert spec["containers"][0]["resources"]["limits"]["nvidia.com/gpu"] == "2" + assert (group["minReplicas"], group["maxReplicas"]) == (0, 3) + # KubeRay wants the resource map as a quoted JSON string. + assert json.loads(json.loads(group["rayStartParams"]["resources"])) == { + "gpu_pool_nvidia-a100": 1 + } + assert worker_group("l4-pool")["template"]["spec"]["nodeSelector"] == { + "nvidia.com/gpu.product": "NVIDIA-L4" + } + + +def test_every_model_asks_for_the_pool_resource_a_worker_group_provides() -> None: + provided = { + resource + for group in CLUSTER["workerGroupSpecs"] + for resource in json.loads(json.loads(group["rayStartParams"]["resources"])) + } + + for model_id, config in llm_configs().items(): + wanted = set(config["deployment_config"]["ray_actor_options"]["resources"]) + assert wanted <= provided, model_id + + +def test_the_serve_application_uses_only_fields_ray_accepts() -> None: + application = build_serve_application(CATALOG)["applications"][0] + + assert application["import_path"] == "ray.serve.llm:build_openai_app" + for model_id, config in llm_configs().items(): + assert set(config) <= LLM_CONFIG_FIELDS, model_id + assert set(config["model_loading_config"]) == {"model_id", "model_source"} + small = llm_configs()["small-specialist"] + assert small["accelerator_type"] == "L4" + assert small["lora_config"] == { + "dynamic_lora_loading_path": "s3://llm-routing-artifacts/adapters", + "max_num_adapters_per_replica": 3, + } + assert "lora_config" not in llm_configs()["general-local"] + + +def test_ray_loads_the_artifact_governance_recorded() -> None: + recorded = {item.name: item.source for item in governed_versions(CATALOG, "s3://bucket")} + application = build_serve_application(CATALOG, "s3://bucket/")["applications"][0] + configs = { + item["model_loading_config"]["model_id"]: item + for item in application["args"]["llm_configs"] + } + + for model_id in ("small-specialist", "general-local", "high-capability"): + assert configs[model_id]["model_loading_config"]["model_source"] == recorded[model_id] + # A variant serves its base model's weights, and a draft model its own. + speculative = configs["high-capability--high-capability-speculative"] + assert speculative["model_loading_config"]["model_source"] == recorded["high-capability"] + assert ( + speculative["engine_kwargs"]["speculative_config"]["model"] + == (recorded["small-specialist"]) + ) + assert "approved-external-fallback" not in configs + + +@pytest.mark.parametrize( + "template", pod_templates(), ids=lambda item: item["spec"]["containers"][0]["name"] +) +def test_ray_pods_meet_the_same_contract_as_every_other_workload(template: dict[str, Any]) -> None: + spec = template["spec"] + container = spec["containers"][0] + + assert template["metadata"]["labels"]["app.kubernetes.io/name"] == "ray-serve" + assert spec["securityContext"]["runAsNonRoot"] is True + assert container["securityContext"]["allowPrivilegeEscalation"] is False + assert container["securityContext"]["readOnlyRootFilesystem"] is True + assert container["securityContext"]["capabilities"]["drop"] == ["ALL"] + assert "@sha256:" in container["image"] + assert container["resources"]["requests"] == container["resources"]["limits"] + assert any(port["name"] == "metrics" for port in container["ports"]) + + +def test_the_committed_manifest_is_what_the_catalog_generates( + capsys: pytest.CaptureFixture[str], +) -> None: + committed = (OVERLAY / "ray-service.yaml").read_text(encoding="utf-8") + + assert committed == render_ray_service(CATALOG) + assert main([]) == 0 + assert capsys.readouterr().out == committed + embedded = yaml.safe_load(yaml.safe_load(committed)["spec"]["serveConfigV2"]) + assert embedded == build_serve_application(CATALOG) + + +def test_the_overlay_lists_files_that_exist_and_replaces_the_single_engine() -> None: + kustomization = yaml.safe_load((OVERLAY / "kustomization.yaml").read_text(encoding="utf-8")) + + assert kustomization["resources"][0] == "../../kubernetes" + for resource in kustomization["resources"][1:]: + assert (OVERLAY / resource).is_file(), resource + removed = { + (patch["kind"], patch["metadata"]["name"]) + for patch in ( + yaml.safe_load(item["patch"]) for item in kustomization["patches"] if "patch" in item + ) + if patch.get("$patch") == "delete" + } + assert removed == { + ("Deployment", "vllm-serve"), + ("Service", "vllm-serve"), + ("NetworkPolicy", "vllm-serve"), + } + + +def test_inference_reaches_ray_only_from_the_gateway() -> None: + policy = yaml.safe_load((OVERLAY / "network-policy.yaml").read_text(encoding="utf-8"))["spec"] + gateway = yaml.safe_load( + (OVERLAY / "gateway-network-policy.patch.yaml").read_text(encoding="utf-8") + )["spec"] + + serve_sources = [ + entry["from"] + for entry in policy["ingress"] + if any(port["port"] == 8000 for port in entry.get("ports", [])) + ] + assert serve_sources == [ + [{"podSelector": {"matchLabels": {"app.kubernetes.io/name": "llm-gateway"}}}] + ] + targets = { + rule["podSelector"]["matchLabels"]["app.kubernetes.io/name"] + for entry in gateway["egress"] + for rule in entry["to"] + } + assert targets == {"ray-serve", "redis", "litellm-proxy"} + for entry in policy["egress"]: + assert all("ipBlock" not in rule for rule in entry["to"]) + + +@pytest.mark.skipif(shutil.which("kubectl") is None, reason="kubectl is not installed") +def test_the_overlay_renders_with_ray_in_place_of_the_single_engine() -> None: + rendered = subprocess.run( + ["kubectl", "kustomize", str(OVERLAY)], capture_output=True, text=True, check=True + ) + documents = [item for item in yaml.safe_load_all(rendered.stdout) if item] + names = {(item["kind"], item["metadata"]["name"]) for item in documents} + + assert ("RayService", "llm-serve") in names + assert ("PodMonitor", "ray-serve") in names + assert not any(name == "vllm-serve" for _, name in names) + gateway = next( + item + for item in documents + if item["metadata"]["name"] == "llm-gateway" and item["kind"] == "Deployment" + ) + environment = { + item["name"]: item.get("value") + for item in gateway["spec"]["template"]["spec"]["containers"][0]["env"] + } + assert environment["ROUTER_VLLM_BASE_URL"] == ( + "http://llm-serve-serve-svc.llm-routing.svc.cluster.local:8000" + ) + # Everything the base promised about the gateway is still there. + assert environment["ROUTER_BACKEND"] == "vllm" + assert "ROUTER_API_KEYS" in environment + + +def emitted_metrics() -> set[str]: + """Metric families the gateway publishes, without their sample suffixes.""" + + with TestClient(create_app(Settings(api_keys="k"))) as client: + exposition = client.get("/metrics").text + return { + re.sub(r"_total$", "", line.split()[2]) + for line in exposition.splitlines() + if line.startswith("# TYPE router_") + } + + +def referenced_metrics(expression: str) -> set[str]: + names = set(re.findall(r"\b(?:router_|DCGM_)[A-Za-z0-9_]+", expression)) + return {re.sub(r"_(total|bucket|sum|count)$", "", name) for name in names} + + +def dashboards() -> dict[str, dict[str, Any]]: + return { + path.name: json.loads(path.read_text(encoding="utf-8")) + for path in sorted((BASE / "dashboards").glob("*.json")) + } + + +def alert_rules() -> list[dict[str, Any]]: + documents = yaml.safe_load_all((BASE / "observability.yaml").read_text(encoding="utf-8")) + rule = next(item for item in documents if item["kind"] == "PrometheusRule") + return [alert for group in rule["spec"]["groups"] for alert in group["rules"]] + + +def test_dashboards_and_alerts_only_query_metrics_that_exist() -> None: + emitted = emitted_metrics() + expressions = [ + target["expr"] + for board in dashboards().values() + for panel in board["panels"] + for target in panel["targets"] + ] + [alert["expr"] for alert in alert_rules()] + + assert len(expressions) > 30 + for expression in expressions: + used = referenced_metrics(expression) + assert used, expression + for name in used: + if name.startswith("DCGM_"): + assert name in { + "DCGM_FI_DEV_GPU_UTIL", + "DCGM_FI_DEV_FB_USED", + "DCGM_FI_DEV_FB_FREE", + } + else: + assert name in emitted, f"{name} in {expression}" + + +def test_the_spec_inference_and_routing_metrics_are_all_on_a_dashboard() -> None: + charted = { + name + for board in dashboards().values() + for panel in board["panels"] + for target in panel["targets"] + for name in referenced_metrics(target["expr"]) + } + + assert { + "router_time_to_first_token_seconds", + "router_time_per_output_token_seconds", + "router_request_latency_seconds", + "router_requests", + "router_tokens", + "router_engine_batch_size", + "router_queued_requests", + "router_inflight_requests", + "router_gpu_utilization_ratio", + "router_gpu_memory_used_bytes", + "router_engine_kv_cache_occupancy_ratio", + "router_model_load_seconds", + "router_routes", + "router_fallbacks", + "router_queue_delay_prediction_error_ms", + "router_predicted_quality", + "router_observed_quality", + "router_cache_events", + "router_rejections", + } <= charted + + +def test_every_dashboard_is_shipped_to_grafana() -> None: + kustomization = yaml.safe_load((BASE / "kustomization.yaml").read_text(encoding="utf-8")) + generator = kustomization["configMapGenerator"][0] + + assert {Path(item).name for item in generator["files"]} == set(dashboards()) + assert generator["options"]["labels"] == {"grafana_dashboard": "1"} + assert generator["options"]["disableNameSuffixHash"] is True + for board in dashboards().values(): + assert board["uid"].startswith("llm-routing-") and board["panels"] + + +def test_alerts_filter_on_rejection_types_the_gateway_really_reports() -> None: + source = Path("src/llm_router/app.py").read_text(encoding="utf-8") + shedding = next(item for item in alert_rules() if item["alert"] == "GatewayRejectingRequests") + + for rejection in re.search(r'type=~"([^"]+)"', shedding["expr"]).group(1).split("|"): # type: ignore[union-attr] + assert f'record_rejection("{rejection}")' in source + for alert in alert_rules(): + assert alert["labels"]["severity"] in {"warning", "critical"} + assert alert["annotations"]["summary"] + + +def test_the_gpu_operator_supplies_what_the_serving_manifests_rely_on() -> None: + values = yaml.safe_load(Path("deploy/gpu-operator/values.yaml").read_text(encoding="utf-8")) + + # The nvidia.com/gpu resource, the gpu.product label, and the GPU exporter. + assert values["devicePlugin"]["enabled"] and values["driver"]["enabled"] + assert values["nfd"]["enabled"] and values["gfd"]["enabled"] + assert values["dcgmExporter"]["enabled"] + assert values["dcgmExporter"]["serviceMonitor"]["enabled"] + assert values["daemonsets"]["tolerations"][0]["key"] == "nvidia.com/gpu"