From d57ebd9201b7978a208d196c8250dabee317cbf4 Mon Sep 17 00:00:00 2001 From: Yash-Chindam Date: Sun, 4 Oct 2026 11:37:49 +0530 Subject: [PATCH] feat: serve every catalog model from Ray worker pools split by accelerator Generate a KubeRay RayService from the catalog with a head and one worker group per accelerator type, shipped as an overlay that replaces the single engine. Add GPU Operator values with the DCGM exporter, Grafana dashboards, and alert rules checked against the metrics the gateway publishes. Co-Authored-By: Claude Opus 5.5 --- .github/workflows/cd.yml | 12 +- README.md | 42 +- deploy/gpu-operator/values.yaml | 43 ++ deploy/kubernetes/dashboards/gateway.json | 449 ++++++++++++++++++ deploy/kubernetes/dashboards/serving.json | 317 +++++++++++++ deploy/kubernetes/kustomization.yaml | 10 + deploy/kubernetes/observability.yaml | 71 +++ .../ray/gateway-network-policy.patch.yaml | 30 ++ deploy/overlays/ray/kustomization.yaml | 51 ++ deploy/overlays/ray/monitoring.yaml | 18 + deploy/overlays/ray/network-policy.yaml | 65 +++ deploy/overlays/ray/ray-service.yaml | 386 +++++++++++++++ src/llm_router/topology.py | 329 +++++++++++++ tests/unit/test_topology.py | 372 +++++++++++++++ 14 files changed, 2191 insertions(+), 4 deletions(-) create mode 100644 deploy/gpu-operator/values.yaml create mode 100644 deploy/kubernetes/dashboards/gateway.json create mode 100644 deploy/kubernetes/dashboards/serving.json create mode 100644 deploy/overlays/ray/gateway-network-policy.patch.yaml create mode 100644 deploy/overlays/ray/kustomization.yaml create mode 100644 deploy/overlays/ray/monitoring.yaml create mode 100644 deploy/overlays/ray/network-policy.yaml create mode 100644 deploy/overlays/ray/ray-service.yaml create mode 100644 src/llm_router/topology.py create mode 100644 tests/unit/test_topology.py diff --git a/.github/workflows/cd.yml b/.github/workflows/cd.yml index a9d8d8e..7f2d17b 100644 --- a/.github/workflows/cd.yml +++ b/.github/workflows/cd.yml @@ -60,6 +60,8 @@ jobs: - run: python -m pip install -e ".[dev]" - name: Verify the serving configuration matches the catalog run: python -m llm_router.serving > /tmp/ray-serve.yaml && diff -u config/ray-serve.yaml /tmp/ray-serve.yaml + - name: Verify the Ray topology matches the catalog + run: python -m llm_router.topology > /tmp/ray-service.yaml && diff -u deploy/overlays/ray/ray-service.yaml /tmp/ray-service.yaml - name: Render the canary and rollback plans # One plan per track (model, adapter, policy), each naming what it # rolls back to and the criteria that trigger it. @@ -73,7 +75,11 @@ jobs: run: | curl -sSLo kubeconform.tar.gz https://github.com/yannh/kubeconform/releases/download/v0.7.0/kubeconform-linux-amd64.tar.gz tar xf kubeconform.tar.gz kubeconform - ./kubeconform -strict -ignore-missing-schemas -summary deploy/kubernetes + # Both topologies are rendered the way they would be applied: the + # single-engine base, and the Ray overlay that replaces the engine. + kubectl kustomize deploy/kubernetes > rendered-base.yaml + kubectl kustomize deploy/overlays/ray > rendered-ray.yaml + ./kubeconform -strict -ignore-missing-schemas -summary rendered-base.yaml rendered-ray.yaml - uses: actions/upload-artifact@v7 with: name: deployment-plan-${{ env.RELEASE_REF }} @@ -81,5 +87,7 @@ jobs: canary-plan.json governance-plan.json config/ray-serve.yaml - deploy/kubernetes + rendered-base.yaml + rendered-ray.yaml + deploy retention-days: 14 diff --git a/README.md b/README.md index 6d97258..e7252e2 100644 --- a/README.md +++ b/README.md @@ -207,9 +207,46 @@ unprivileged workloads, digest-pinned images, bounded resources, real probes, GP pinning, and `/metrics` reachable only from monitoring. ```bash -kubectl apply -k deploy/kubernetes +kubectl apply -k deploy/kubernetes # one vLLM engine +kubectl apply -k deploy/overlays/ray # Ray Serve across GPU pools ``` +The base runs a single vLLM engine, which serves one model. The +[`deploy/overlays/ray`](deploy/overlays/ray) overlay replaces it with a KubeRay `RayService` +that serves every local model in the catalog behind one OpenAI-compatible endpoint: + +- A head that schedules and never runs a model. +- One worker group per accelerator type, pinned by `nvidia.com/gpu.product`, so GPU pools stay + separate. Each group is sized from the autoscaling bounds of the models placed on it, and a + worker holds as many GPUs as the largest replica on its pool needs. +- Model weights are loaded from the location [governance](#governance-in-mlflow) records for + each revision. + +The `RayService` is generated from the catalog, never hand-edited, and verified by a test and +by CD: + +```bash +python -m llm_router.topology > deploy/overlays/ray/ray-service.yaml +``` + +Under the overlay, engine metrics are scraped from the Ray pods by Prometheus. The gateway +cannot read a whole cluster from one address, so its own `router_engine_*` and `router_gpu_*` +gauges stay empty there and live load does not influence routing. + +GPU support comes from the NVIDIA GPU Operator, installed cluster-wide with +[`deploy/gpu-operator/values.yaml`](deploy/gpu-operator/values.yaml). It provides the +`nvidia.com/gpu` resource, the node label the pools select on, and the DCGM GPU exporter. + +Two Grafana dashboards in [`deploy/kubernetes/dashboards`](deploy/kubernetes/dashboards) ship as +a labelled ConfigMap for Grafana's sidecar: one for the gateway and router, one for engines and +GPUs. A `PrometheusRule` alerts on latency, load shedding, a stuck queue, fallback rate, canary +rollback, an open engine circuit, KV-cache pressure, and GPU memory. A test fails if a dashboard +or alert queries a metric the gateway does not publish. + +None of this has been applied to a cluster. The manifests are schema-validated and +contract-tested; the Ray image, GPU product labels, and node sizes are placeholders to set for +the hardware you have. + Stateless ingress scales separately from GPU replicas. Set `ROUTER_REDIS_URL` so cache and quota state are shared once the gateway runs more than one replica; without it both are in-process and correct for a single replica only. Install the client with the extra: @@ -220,7 +257,8 @@ python -m pip install -e ".[redis]" CD renders the canary plans (one per track, each with its rollback target) and the governance plan, verifies -`config/ray-serve.yaml` against the catalog, and validates the manifests with kubeconform. +`config/ray-serve.yaml` and the Ray topology against the catalog, and validates both rendered +topologies with kubeconform. Applying to a cluster stays disabled until a deployment destination is configured. ## Model registry diff --git a/deploy/gpu-operator/values.yaml b/deploy/gpu-operator/values.yaml new file mode 100644 index 0000000..4bceb39 --- /dev/null +++ b/deploy/gpu-operator/values.yaml @@ -0,0 +1,43 @@ +# Values for the NVIDIA GPU Operator chart (nvidia/gpu-operator). The operator +# is cluster-wide infrastructure, installed once and not part of this +# platform's namespace: +# +# helm upgrade --install gpu-operator nvidia/gpu-operator \ +# --namespace gpu-operator --create-namespace \ +# --values deploy/gpu-operator/values.yaml +# +# It supplies the three things the serving manifests rely on. + +# 1. The nvidia.com/gpu resource that engine and Ray worker pods request. +driver: + enabled: true +toolkit: + enabled: true +devicePlugin: + enabled: true + +# 2. The nvidia.com/gpu.product node label that pins each pool to one +# accelerator type. Node feature discovery finds the hardware and GPU +# feature discovery turns it into the label. +nfd: + enabled: true +gfd: + enabled: true + +# 3. The GPU exporter. DCGM publishes the DCGM_FI_DEV_* series the dashboards +# and alerts read, and the ServiceMonitor hands them to Prometheus. +dcgmExporter: + enabled: true + serviceMonitor: + enabled: true + interval: 15s + additionalLabels: + release: prometheus + +# GPU nodes are tainted so only inference lands on them; the operator's own +# daemons have to tolerate that taint to run there at all. +daemonsets: + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule diff --git a/deploy/kubernetes/dashboards/gateway.json b/deploy/kubernetes/dashboards/gateway.json new file mode 100644 index 0000000..8844596 --- /dev/null +++ b/deploy/kubernetes/dashboards/gateway.json @@ -0,0 +1,449 @@ +{ + "uid": "llm-routing-gateway", + "title": "LLM routing: gateway and router", + "tags": [ + "llm-routing" + ], + "schemaVersion": 39, + "version": 1, + "editable": false, + "refresh": "30s", + "time": { + "from": "now-1h", + "to": "now" + }, + "templating": { + "list": [ + { + "name": "datasource", + "type": "datasource", + "query": "prometheus", + "label": "Data source" + } + ] + }, + "panels": [ + { + "id": 1, + "title": "Requests per second by model and outcome", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "fieldConfig": { + "defaults": { + "unit": "reqps" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (model, outcome) (rate(router_requests_total[5m]))", + "legendFormat": "{{model}} {{outcome}}" + } + ] + }, + { + "id": 2, + "title": "End-to-end latency", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "fieldConfig": { + "defaults": { + "unit": "s" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "histogram_quantile(0.50, sum by (le, model) (rate(router_request_latency_seconds_bucket[5m])))", + "legendFormat": "p50 {{model}}" + }, + { + "refId": "B", + "expr": "histogram_quantile(0.95, sum by (le, model) (rate(router_request_latency_seconds_bucket[5m])))", + "legendFormat": "p95 {{model}}" + }, + { + "refId": "C", + "expr": "histogram_quantile(0.99, sum by (le, model) (rate(router_request_latency_seconds_bucket[5m])))", + "legendFormat": "p99 {{model}}" + } + ] + }, + { + "id": 3, + "title": "Time to first token (p95)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "fieldConfig": { + "defaults": { + "unit": "s" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "histogram_quantile(0.95, sum by (le, model) (rate(router_time_to_first_token_seconds_bucket[5m])))", + "legendFormat": "{{model}}" + } + ] + }, + { + "id": 4, + "title": "Time per output token (p95)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "fieldConfig": { + "defaults": { + "unit": "s" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "histogram_quantile(0.95, sum by (le, model) (rate(router_time_per_output_token_seconds_bucket[5m])))", + "legendFormat": "{{model}}" + } + ] + }, + { + "id": 5, + "title": "Tokens per second", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (model, kind) (rate(router_tokens_total[5m]))", + "legendFormat": "{{model}} {{kind}}" + } + ] + }, + { + "id": 6, + "title": "In-flight and queued requests", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 16 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum(router_inflight_requests)", + "legendFormat": "in flight" + }, + { + "refId": "B", + "expr": "sum(router_queued_requests)", + "legendFormat": "queued" + } + ] + }, + { + "id": 7, + "title": "Routes by model and privacy class", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 24 + }, + "fieldConfig": { + "defaults": { + "unit": "reqps" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (model, privacy) (rate(router_routes_total[5m]))", + "legendFormat": "{{model}} {{privacy}}" + } + ] + }, + { + "id": 8, + "title": "Rejections by type", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 24 + }, + "fieldConfig": { + "defaults": { + "unit": "reqps" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (type) (rate(router_rejections_total[5m]))", + "legendFormat": "{{type}}" + } + ] + }, + { + "id": 9, + "title": "Fallbacks", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 32 + }, + "fieldConfig": { + "defaults": { + "unit": "reqps" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (from_model, to_model, cause) (rate(router_fallbacks_total[5m]))", + "legendFormat": "{{from_model}} to {{to_model}} ({{cause}})" + }, + { + "refId": "B", + "expr": "sum by (model) (rate(router_external_fallback_total[5m]))", + "legendFormat": "external {{model}}" + } + ] + }, + { + "id": 10, + "title": "Cache hit ratio", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 32 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (cache) (rate(router_cache_events_total{result=\"hit\"}[5m])) / sum by (cache) (rate(router_cache_events_total[5m]))", + "legendFormat": "{{cache}}" + } + ] + }, + { + "id": 11, + "title": "Predicted versus observed quality", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 40 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (model) (router_predicted_quality)", + "legendFormat": "predicted {{model}}" + }, + { + "refId": "B", + "expr": "avg by (model) (router_observed_quality)", + "legendFormat": "observed {{model}}" + } + ] + }, + { + "id": 12, + "title": "Queue-delay prediction error", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 40 + }, + "fieldConfig": { + "defaults": { + "unit": "ms" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (model) (router_queue_delay_prediction_error_ms)", + "legendFormat": "{{model}}" + } + ] + }, + { + "id": 13, + "title": "Canary traffic and rollbacks", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 48 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (subject, arm, outcome) (rate(router_canary_requests_total[5m]))", + "legendFormat": "{{subject}} {{arm}} {{outcome}}" + }, + { + "refId": "B", + "expr": "sum by (subject) (increase(router_canary_rollbacks_total[1h]))", + "legendFormat": "rollback {{subject}}" + } + ] + }, + { + "id": 14, + "title": "Structured-output validity", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 48 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (model) (rate(router_structured_output_total{result=\"valid\"}[5m])) / sum by (model) (rate(router_structured_output_total[5m]))", + "legendFormat": "{{model}}" + } + ] + } + ] +} diff --git a/deploy/kubernetes/dashboards/serving.json b/deploy/kubernetes/dashboards/serving.json new file mode 100644 index 0000000..f931fee --- /dev/null +++ b/deploy/kubernetes/dashboards/serving.json @@ -0,0 +1,317 @@ +{ + "uid": "llm-routing-serving", + "title": "LLM routing: engines and GPUs", + "tags": [ + "llm-routing" + ], + "schemaVersion": 39, + "version": 1, + "editable": false, + "refresh": "30s", + "time": { + "from": "now-1h", + "to": "now" + }, + "templating": { + "list": [ + { + "name": "datasource", + "type": "datasource", + "query": "prometheus", + "label": "Data source" + } + ] + }, + "panels": [ + { + "id": 1, + "title": "Engine running and waiting requests", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 0 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (engine) (router_engine_running_requests)", + "legendFormat": "running {{engine}}" + }, + { + "refId": "B", + "expr": "sum by (engine) (router_engine_waiting_requests)", + "legendFormat": "waiting {{engine}}" + } + ] + }, + { + "id": 2, + "title": "Engine batch size", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 0 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (engine) (router_engine_batch_size)", + "legendFormat": "{{engine}}" + } + ] + }, + { + "id": 3, + "title": "KV-cache occupancy", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 8 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "max by (engine) (router_engine_kv_cache_occupancy_ratio)", + "legendFormat": "{{engine}}" + } + ] + }, + { + "id": 4, + "title": "Engine preemptions", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 8 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (engine) (rate(router_engine_preemptions_total[5m]))", + "legendFormat": "{{engine}}" + } + ] + }, + { + "id": 5, + "title": "Engine circuit state (0 closed, 0.5 half-open, 1 open)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 16 + }, + "fieldConfig": { + "defaults": { + "unit": "short" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "max by (engine) (router_engine_circuit_open)", + "legendFormat": "{{engine}}" + } + ] + }, + { + "id": 6, + "title": "Model load and cold start (p95)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 16 + }, + "fieldConfig": { + "defaults": { + "unit": "s" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "histogram_quantile(0.95, sum by (le, model) (rate(router_model_load_seconds_bucket[5m])))", + "legendFormat": "{{model}}" + } + ] + }, + { + "id": 7, + "title": "GPU utilization as seen by the gateway", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 24 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (gpu) (router_gpu_utilization_ratio)", + "legendFormat": "gpu {{gpu}}" + } + ] + }, + { + "id": 8, + "title": "GPU memory as seen by the gateway", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 24 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (gpu) (router_gpu_memory_used_bytes) / sum by (gpu) (router_gpu_memory_total_bytes)", + "legendFormat": "gpu {{gpu}}" + } + ] + }, + { + "id": 9, + "title": "GPU utilization by node (DCGM exporter)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 32 + }, + "fieldConfig": { + "defaults": { + "unit": "percent" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "avg by (Hostname, gpu) (DCGM_FI_DEV_GPU_UTIL)", + "legendFormat": "{{Hostname}} gpu {{gpu}}" + } + ] + }, + { + "id": 10, + "title": "GPU memory by node (DCGM exporter)", + "type": "timeseries", + "datasource": { + "type": "prometheus", + "uid": "${datasource}" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 32 + }, + "fieldConfig": { + "defaults": { + "unit": "decmbytes" + }, + "overrides": [] + }, + "targets": [ + { + "refId": "A", + "expr": "sum by (Hostname, gpu) (DCGM_FI_DEV_FB_USED)", + "legendFormat": "used {{Hostname}} gpu {{gpu}}" + }, + { + "refId": "B", + "expr": "sum by (Hostname, gpu) (DCGM_FI_DEV_FB_FREE)", + "legendFormat": "free {{Hostname}} gpu {{gpu}}" + } + ] + } + ] +} diff --git a/deploy/kubernetes/kustomization.yaml b/deploy/kubernetes/kustomization.yaml index 6503194..9f8b7fd 100644 --- a/deploy/kubernetes/kustomization.yaml +++ b/deploy/kubernetes/kustomization.yaml @@ -11,3 +11,13 @@ resources: - external-provider.yaml - mlflow.yaml - observability.yaml +# Grafana's dashboard sidecar loads any ConfigMap carrying this label. +configMapGenerator: + - name: llm-routing-dashboards + files: + - dashboards/gateway.json + - dashboards/serving.json + options: + disableNameSuffixHash: true + labels: + grafana_dashboard: "1" diff --git a/deploy/kubernetes/observability.yaml b/deploy/kubernetes/observability.yaml index a2e19f7..1705938 100644 --- a/deploy/kubernetes/observability.yaml +++ b/deploy/kubernetes/observability.yaml @@ -14,3 +14,74 @@ spec: - port: http path: /metrics interval: 15s +--- +# Alerts on the conditions the platform promises to make visible: overload, +# an open engine circuit, a canary that rolled back, and a GPU out of memory. +apiVersion: monitoring.coreos.com/v1 +kind: PrometheusRule +metadata: + name: llm-routing + namespace: llm-routing + labels: + release: prometheus +spec: + groups: + - name: llm-routing.gateway + rules: + - alert: GatewayLatencyHigh + expr: histogram_quantile(0.95, sum by (le, model) (rate(router_request_latency_seconds_bucket[5m]))) > 5 + for: 10m + labels: + severity: warning + annotations: + summary: p95 latency for {{ $labels.model }} has been above 5s for 10 minutes. + - alert: GatewayRejectingRequests + expr: sum by (type) (rate(router_rejections_total{type=~"overloaded|engine_unavailable"}[5m])) > 1 + for: 5m + labels: + severity: warning + annotations: + summary: The gateway is shedding load ({{ $labels.type }}). + - alert: GatewayQueueNotDraining + expr: sum(router_queued_requests) > 16 + for: 10m + labels: + severity: warning + annotations: + summary: Requests have been queueing for admission for 10 minutes. + - alert: CanaryRolledBack + expr: increase(router_canary_rollbacks_total[15m]) > 0 + labels: + severity: warning + annotations: + summary: The canary for {{ $labels.subject }} failed its criteria and was rolled back. + - alert: FallbackRateHigh + expr: sum(rate(router_fallbacks_total[10m])) / sum(rate(router_requests_total[10m])) > 0.05 + for: 10m + labels: + severity: warning + annotations: + summary: More than 5% of requests are being answered by a fallback model. + - name: llm-routing.serving + rules: + - alert: EngineCircuitOpen + expr: max by (engine) (router_engine_circuit_open) == 1 + for: 2m + labels: + severity: critical + annotations: + summary: The circuit for engine {{ $labels.engine }} is open; requests to it are failing fast. + - alert: KVCacheNearlyFull + expr: max by (engine) (router_engine_kv_cache_occupancy_ratio) > 0.95 + for: 10m + labels: + severity: warning + annotations: + summary: KV-cache occupancy on {{ $labels.engine }} has been above 95% for 10 minutes. + - alert: GPUMemoryNearlyFull + expr: DCGM_FI_DEV_FB_USED / (DCGM_FI_DEV_FB_USED + DCGM_FI_DEV_FB_FREE) > 0.97 + for: 10m + labels: + severity: warning + annotations: + summary: GPU {{ $labels.gpu }} on {{ $labels.Hostname }} is nearly out of memory. diff --git a/deploy/overlays/ray/gateway-network-policy.patch.yaml b/deploy/overlays/ray/gateway-network-policy.patch.yaml new file mode 100644 index 0000000..1fba3a2 --- /dev/null +++ b/deploy/overlays/ray/gateway-network-policy.patch.yaml @@ -0,0 +1,30 @@ +# The gateway's egress with the Ray Serve pods in place of the single engine. +# A list of egress rules has no merge key, so the whole list is restated. +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: llm-gateway + namespace: llm-routing +spec: + egress: + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: ray-serve + ports: + - protocol: TCP + port: 8000 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: redis + ports: + - protocol: TCP + port: 6379 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: litellm-proxy + ports: + - protocol: TCP + port: 4000 diff --git a/deploy/overlays/ray/kustomization.yaml b/deploy/overlays/ray/kustomization.yaml new file mode 100644 index 0000000..7848da1 --- /dev/null +++ b/deploy/overlays/ray/kustomization.yaml @@ -0,0 +1,51 @@ +# Ray Serve control plane in place of the single engine Deployment. +# +# The base runs one vLLM engine, which can serve one model. This overlay swaps +# it for a RayService with a head and one worker group per accelerator type, +# so every local model in the catalog is served from its own GPU pool behind +# one OpenAI-compatible endpoint. Requires the KubeRay operator. +apiVersion: kustomize.config.k8s.io/v1beta1 +kind: Kustomization +resources: + - ../../kubernetes + - ray-service.yaml + - network-policy.yaml + - monitoring.yaml +patches: + - patch: | + $patch: delete + apiVersion: apps/v1 + kind: Deployment + metadata: + name: vllm-serve + namespace: llm-routing + - patch: | + $patch: delete + apiVersion: v1 + kind: Service + metadata: + name: vllm-serve + namespace: llm-routing + - patch: | + $patch: delete + apiVersion: networking.k8s.io/v1 + kind: NetworkPolicy + metadata: + name: vllm-serve + namespace: llm-routing + # KubeRay publishes the Serve endpoint as -serve-svc. + - patch: | + apiVersion: apps/v1 + kind: Deployment + metadata: + name: llm-gateway + namespace: llm-routing + spec: + template: + spec: + containers: + - name: gateway + env: + - name: ROUTER_VLLM_BASE_URL + value: http://llm-serve-serve-svc.llm-routing.svc.cluster.local:8000 + - path: gateway-network-policy.patch.yaml diff --git a/deploy/overlays/ray/monitoring.yaml b/deploy/overlays/ray/monitoring.yaml new file mode 100644 index 0000000..b367cd1 --- /dev/null +++ b/deploy/overlays/ray/monitoring.yaml @@ -0,0 +1,18 @@ +# Ray exports its own metrics and, with log_engine_metrics, each vLLM engine's +# metrics on every head and worker pod. Under this overlay engine telemetry is +# read from here by Prometheus; the gateway cannot scrape a whole cluster from +# one address, so its own router_engine_* gauges stay empty. +apiVersion: monitoring.coreos.com/v1 +kind: PodMonitor +metadata: + name: ray-serve + namespace: llm-routing + labels: + release: prometheus +spec: + selector: + matchLabels: + app.kubernetes.io/name: ray-serve + podMetricsEndpoints: + - port: metrics + interval: 15s diff --git a/deploy/overlays/ray/network-policy.yaml b/deploy/overlays/ray/network-policy.yaml new file mode 100644 index 0000000..01f9408 --- /dev/null +++ b/deploy/overlays/ray/network-policy.yaml @@ -0,0 +1,65 @@ +# Ray head and workers. Inference traffic comes only from the gateway; the +# cluster talks to itself freely; the operator drives it through the dashboard. +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: ray-serve + namespace: llm-routing +spec: + podSelector: + matchLabels: + app.kubernetes.io/name: ray-serve + policyTypes: [Ingress, Egress] + ingress: + - from: + - podSelector: + matchLabels: + app.kubernetes.io/name: llm-gateway + ports: + - protocol: TCP + port: 8000 + # Head and workers exchange control, object and worker traffic on ports + # Ray assigns at start-up, so the cluster is open to itself. + - from: + - podSelector: + matchLabels: + app.kubernetes.io/name: ray-serve + # KubeRay submits the Serve configuration and reads health here. + - from: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: kuberay-system + ports: + - protocol: TCP + port: 8265 + - protocol: TCP + port: 52365 + - from: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: monitoring + ports: + - protocol: TCP + port: 8080 + egress: + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: ray-serve + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: kube-system + ports: + - protocol: UDP + port: 53 + - protocol: TCP + port: 53 + # Model and adapter artifacts, from the store governance records them in. + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: storage + ports: + - protocol: TCP + port: 9000 diff --git a/deploy/overlays/ray/ray-service.yaml b/deploy/overlays/ray/ray-service.yaml new file mode 100644 index 0000000..2208612 --- /dev/null +++ b/deploy/overlays/ray/ray-service.yaml @@ -0,0 +1,386 @@ +# Generated from config/registry.yaml by `python -m llm_router.topology`. +# Do not edit: change the catalog and regenerate. A test and CD both fail on drift. +apiVersion: ray.io/v1 +kind: RayService +metadata: + name: llm-serve + namespace: llm-routing + labels: + app.kubernetes.io/part-of: local-llm-router + annotations: + llm-routing/policy-version: v1 +spec: + serveConfigV2: | + applications: + - name: llm_app + route_prefix: / + import_path: ray.serve.llm:build_openai_app + args: + llm_configs: + - model_loading_config: + model_id: small-specialist + model_source: s3://llm-routing-artifacts/models/small-specialist/mock-small-sha256-dev + accelerator_type: L4 + deployment_config: + autoscaling_config: + min_replicas: 1 + max_replicas: 4 + target_ongoing_requests: 8 + upscale_delay_s: 10 + downscale_delay_s: 300 + max_ongoing_requests: 16 + ray_actor_options: + num_gpus: 1 + resources: + gpu_pool_nvidia-l4: 0.001 + engine_kwargs: + max_model_len: 8192 + tensor_parallel_size: 1 + enable_prefix_caching: true + enable_chunked_prefill: true + quantization: awq + enable_lora: true + max_loras: 3 + max_lora_rank: 32 + log_engine_metrics: true + lora_config: + dynamic_lora_loading_path: s3://llm-routing-artifacts/adapters + max_num_adapters_per_replica: 3 + - model_loading_config: + model_id: general-local + model_source: s3://llm-routing-artifacts/models/general-local/mock-general-sha256-dev + accelerator_type: A10G + deployment_config: + autoscaling_config: + min_replicas: 1 + max_replicas: 4 + target_ongoing_requests: 8 + upscale_delay_s: 10 + downscale_delay_s: 300 + max_ongoing_requests: 16 + ray_actor_options: + num_gpus: 1 + resources: + gpu_pool_nvidia-a10g: 0.001 + engine_kwargs: + max_model_len: 32768 + tensor_parallel_size: 1 + enable_prefix_caching: true + enable_chunked_prefill: true + log_engine_metrics: true + - model_loading_config: + model_id: high-capability + model_source: s3://llm-routing-artifacts/models/high-capability/mock-high-sha256-dev + accelerator_type: A100 + deployment_config: + autoscaling_config: + min_replicas: 0 + max_replicas: 2 + target_ongoing_requests: 8 + upscale_delay_s: 10 + downscale_delay_s: 60 + max_ongoing_requests: 16 + ray_actor_options: + num_gpus: 2 + resources: + gpu_pool_nvidia-a100: 0.001 + engine_kwargs: + max_model_len: 65536 + tensor_parallel_size: 2 + enable_prefix_caching: true + enable_chunked_prefill: true + log_engine_metrics: true + - model_loading_config: + model_id: general-local--general-gptq + model_source: s3://llm-routing-artifacts/models/general-local/mock-general-sha256-dev + accelerator_type: A10G + deployment_config: + autoscaling_config: + min_replicas: 0 + max_replicas: 1 + ray_actor_options: + num_gpus: 1 + resources: + gpu_pool_nvidia-a10g: 0.001 + engine_kwargs: + max_model_len: 32768 + tensor_parallel_size: 1 + enable_prefix_caching: true + enable_chunked_prefill: true + quantization: gptq + log_engine_metrics: true + - model_loading_config: + model_id: high-capability--high-capability-speculative + model_source: s3://llm-routing-artifacts/models/high-capability/mock-high-sha256-dev + accelerator_type: A100 + deployment_config: + autoscaling_config: + min_replicas: 0 + max_replicas: 1 + ray_actor_options: + num_gpus: 2 + resources: + gpu_pool_nvidia-a100: 0.001 + engine_kwargs: + max_model_len: 65536 + tensor_parallel_size: 2 + enable_prefix_caching: true + enable_chunked_prefill: true + speculative_config: + model: s3://llm-routing-artifacts/models/small-specialist/mock-small-sha256-dev + num_speculative_tokens: 5 + log_engine_metrics: true + rayClusterConfig: + rayVersion: 2.51.0 + enableInTreeAutoscaling: true + headGroupSpec: + rayStartParams: + dashboard-host: 0.0.0.0 + metrics-export-port: '8080' + num-cpus: '0' + template: + metadata: + labels: + app.kubernetes.io/name: ray-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 1000 + seccompProfile: + type: RuntimeDefault + containers: + - name: ray-head + image: docker.io/rayproject/ray-llm@sha256:REPLACE_ME + imagePullPolicy: IfNotPresent + ports: + - name: metrics + containerPort: 8080 + - name: gcs + containerPort: 6379 + - name: dashboard + containerPort: 8265 + - name: serve + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: + - ALL + resources: + requests: + cpu: '2' + memory: 8Gi + limits: + cpu: '2' + memory: 8Gi + volumeMounts: + - name: tmp + mountPath: /tmp + - name: cache + mountPath: /home/ray/.cache + - name: shm + mountPath: /dev/shm + volumes: + - name: tmp + emptyDir: {} + - name: cache + emptyDir: {} + - name: shm + emptyDir: + medium: Memory + terminationGracePeriodSeconds: 60 + workerGroupSpecs: + - groupName: a100-pool + replicas: 0 + minReplicas: 0 + maxReplicas: 3 + rayStartParams: + metrics-export-port: '8080' + resources: '"{\"gpu_pool_nvidia-a100\": 1}"' + template: + metadata: + labels: + app.kubernetes.io/name: ray-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 1000 + seccompProfile: + type: RuntimeDefault + containers: + - name: ray-worker + image: docker.io/rayproject/ray-llm@sha256:REPLACE_ME + imagePullPolicy: IfNotPresent + ports: + - name: metrics + containerPort: 8080 + - name: serve + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: + - ALL + resources: + requests: + cpu: '8' + memory: 160Gi + nvidia.com/gpu: '2' + limits: + cpu: '8' + memory: 160Gi + nvidia.com/gpu: '2' + volumeMounts: + - name: tmp + mountPath: /tmp + - name: cache + mountPath: /home/ray/.cache + - name: shm + mountPath: /dev/shm + volumes: + - name: tmp + emptyDir: {} + - name: cache + emptyDir: {} + - name: shm + emptyDir: + medium: Memory + terminationGracePeriodSeconds: 120 + nodeSelector: + nvidia.com/gpu.product: NVIDIA-A100-SXM4-80GB + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule + - groupName: a10g-pool + replicas: 1 + minReplicas: 1 + maxReplicas: 5 + rayStartParams: + metrics-export-port: '8080' + resources: '"{\"gpu_pool_nvidia-a10g\": 1}"' + template: + metadata: + labels: + app.kubernetes.io/name: ray-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 1000 + seccompProfile: + type: RuntimeDefault + containers: + - name: ray-worker + image: docker.io/rayproject/ray-llm@sha256:REPLACE_ME + imagePullPolicy: IfNotPresent + ports: + - name: metrics + containerPort: 8080 + - name: serve + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: + - ALL + resources: + requests: + cpu: '4' + memory: 48Gi + nvidia.com/gpu: '1' + limits: + cpu: '4' + memory: 48Gi + nvidia.com/gpu: '1' + volumeMounts: + - name: tmp + mountPath: /tmp + - name: cache + mountPath: /home/ray/.cache + - name: shm + mountPath: /dev/shm + volumes: + - name: tmp + emptyDir: {} + - name: cache + emptyDir: {} + - name: shm + emptyDir: + medium: Memory + terminationGracePeriodSeconds: 120 + nodeSelector: + nvidia.com/gpu.product: NVIDIA-A10G + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule + - groupName: l4-pool + replicas: 1 + minReplicas: 1 + maxReplicas: 4 + rayStartParams: + metrics-export-port: '8080' + resources: '"{\"gpu_pool_nvidia-l4\": 1}"' + template: + metadata: + labels: + app.kubernetes.io/name: ray-serve + app.kubernetes.io/part-of: local-llm-router + spec: + securityContext: + runAsNonRoot: true + runAsUser: 1000 + seccompProfile: + type: RuntimeDefault + containers: + - name: ray-worker + image: docker.io/rayproject/ray-llm@sha256:REPLACE_ME + imagePullPolicy: IfNotPresent + ports: + - name: metrics + containerPort: 8080 + - name: serve + containerPort: 8000 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: + - ALL + resources: + requests: + cpu: '4' + memory: 24Gi + nvidia.com/gpu: '1' + limits: + cpu: '4' + memory: 24Gi + nvidia.com/gpu: '1' + volumeMounts: + - name: tmp + mountPath: /tmp + - name: cache + mountPath: /home/ray/.cache + - name: shm + mountPath: /dev/shm + volumes: + - name: tmp + emptyDir: {} + - name: cache + emptyDir: {} + - name: shm + emptyDir: + medium: Memory + terminationGracePeriodSeconds: 120 + nodeSelector: + nvidia.com/gpu.product: NVIDIA-L4 + tolerations: + - key: nvidia.com/gpu + operator: Exists + effect: NoSchedule diff --git a/src/llm_router/topology.py b/src/llm_router/topology.py new file mode 100644 index 0000000..4f0dd72 --- /dev/null +++ b/src/llm_router/topology.py @@ -0,0 +1,329 @@ +"""Ray head and worker topology derived from the registry (sections 7.3 and 17). + +One KubeRay ``RayService`` holds the whole serving plane: a head that +schedules and runs no model, and one worker group per accelerator type, so +GPU pools stay separate and each is sized from the models placed on it. Like +the serving configuration, the manifest is generated from the catalog and +never edited by hand. +""" + +import json +from collections.abc import Sequence +from typing import Any + +import yaml +from pydantic import BaseModel + +from llm_router.governance import DEFAULT_ARTIFACT_ROOT, governed_versions +from llm_router.registry import Registry +from llm_router.serving import build_serving_config + +NAMESPACE = "llm-routing" +SERVICE_NAME = "llm-serve" +POD_LABEL = "ray-serve" +RAY_VERSION = "2.51.0" +RAY_IMAGE = "docker.io/rayproject/ray-llm@sha256:REPLACE_ME" +SERVE_PORT = 8000 +METRICS_PORT = 8080 +DASHBOARD_PORT = 8265 +# Node labels published by GPU feature discovery. They differ between clouds +# and card variants, so anything not listed falls back to NVIDIA-. +GPU_PRODUCT_LABELS = {"A100": "NVIDIA-A100-SXM4-80GB"} + + +class _Block(str): + """A string rendered as a YAML literal block, so embedded YAML stays readable.""" + + +class _Dumper(yaml.SafeDumper): + pass + + +_Dumper.add_representer( + _Block, + lambda dumper, value: dumper.represent_scalar("tag:yaml.org,2002:str", value, style="|"), +) + + +class WorkerPool(BaseModel): + """The capacity one accelerator type needs for the models placed on it.""" + + accelerator: str + ray_accelerator: str + gpus_per_worker: int + memory_gb: int + min_workers: int + max_workers: int + models: tuple[str, ...] + + @property + def group_name(self) -> str: + return f"{self.ray_accelerator.lower()}-pool" + + @property + def resource(self) -> str: + return f"gpu_pool_{self.accelerator}" + + +def ray_accelerator(accelerator: str) -> str: + """Translate a catalog accelerator into the name Ray knows it by.""" + + return accelerator.removeprefix("nvidia-").upper() + + +def worker_pools(registry: Registry) -> tuple[WorkerPool, ...]: + """One pool per accelerator type, sized from its models' autoscaling bounds. + + A worker holds as many GPUs as the largest replica on its pool needs, so a + tensor-parallel model always fits on one node. Experiments add headroom + to the ceiling but never to the floor: they are never kept warm. + """ + + config = build_serving_config(registry) + pools: dict[str, dict[str, Any]] = {} + for entry in (*config["applications"], *config.get("experiments", ())): + card = registry.model_card(entry.get("base_model_id", entry["model_id"])) + scaling = entry["deployment_config"]["autoscaling_config"] + pool = pools.setdefault( + card.hardware.accelerator, + {"gpus": 0, "memory": 0, "min": 0, "max": 0, "models": []}, + ) + pool["gpus"] = max(pool["gpus"], card.hardware.count) + pool["memory"] = max(pool["memory"], card.hardware.minimum_memory_gb) + pool["min"] += scaling["min_replicas"] + pool["max"] += scaling["max_replicas"] + pool["models"].append(entry["model_id"]) + return tuple( + WorkerPool( + accelerator=accelerator, + ray_accelerator=ray_accelerator(accelerator), + gpus_per_worker=pool["gpus"], + memory_gb=pool["memory"], + min_workers=pool["min"], + max_workers=pool["max"], + models=tuple(pool["models"]), + ) + for accelerator, pool in sorted(pools.items()) + ) + + +def build_serve_application( + registry: Registry, artifact_root: str = DEFAULT_ARTIFACT_ROOT +) -> dict[str, Any]: + """The Ray Serve LLM application: one OpenAI-compatible endpoint for every model. + + Only fields Ray's ``LLMConfig`` accepts are kept. Model weights are read + from the location governance records for each revision, so what Ray loads + is what MLflow says was promoted. + """ + + root = artifact_root.rstrip("/") + sources = {item.name: item.source for item in governed_versions(registry, root)} + config = build_serving_config(registry) + llm_configs: list[dict[str, Any]] = [] + for entry in (*config["applications"], *config.get("experiments", ())): + base = entry.get("base_model_id", entry["model_id"]) + card = registry.model_card(base) + deployment = { + **entry["deployment_config"], + "ray_actor_options": { + "num_gpus": card.hardware.count, + "resources": {f"gpu_pool_{card.hardware.accelerator}": 0.001}, + }, + } + llm_config: dict[str, Any] = { + "model_loading_config": {"model_id": entry["model_id"], "model_source": sources[base]}, + "accelerator_type": ray_accelerator(card.hardware.accelerator), + "deployment_config": deployment, + "engine_kwargs": _engine_kwargs(entry["engine_kwargs"], sources), + "log_engine_metrics": True, + } + if "lora_config" in entry: + llm_config["lora_config"] = { + # Ray resolves /; the deploy step publishes + # each promoted adapter revision there. + "dynamic_lora_loading_path": f"{root}/adapters", + "max_num_adapters_per_replica": len(entry["lora_config"]["adapters"]), + } + llm_configs.append(llm_config) + return { + "applications": [ + { + "name": "llm_app", + "route_prefix": "/", + "import_path": "ray.serve.llm:build_openai_app", + "args": {"llm_configs": llm_configs}, + } + ] + } + + +def _engine_kwargs(engine: dict[str, Any], sources: dict[str, str]) -> dict[str, Any]: + """Point a speculative draft model at its recorded artifact as well.""" + + speculative = engine.get("speculative_config") + if speculative is None: + return engine + draft = str(speculative["model"]).removeprefix("registry://").split("@", 1)[0] + return {**engine, "speculative_config": {**speculative, "model": sources[draft]}} + + +def _pod( + *, + image: str, + cpu: str, + memory: str, + gpus: int = 0, + node_selector: dict[str, str] | None = None, + head: bool = False, +) -> dict[str, Any]: + resources: dict[str, str] = {"cpu": cpu, "memory": memory} + if gpus: + resources["nvidia.com/gpu"] = str(gpus) + ports = [{"name": "metrics", "containerPort": METRICS_PORT}] + if head: + ports += [ + {"name": "gcs", "containerPort": 6379}, + {"name": "dashboard", "containerPort": DASHBOARD_PORT}, + {"name": "serve", "containerPort": SERVE_PORT}, + ] + else: + ports.append({"name": "serve", "containerPort": SERVE_PORT}) + spec: dict[str, Any] = { + "securityContext": { + "runAsNonRoot": True, + "runAsUser": 1000, + "seccompProfile": {"type": "RuntimeDefault"}, + }, + "containers": [ + { + "name": "ray-head" if head else "ray-worker", + "image": image, + "imagePullPolicy": "IfNotPresent", + "ports": ports, + "securityContext": { + "allowPrivilegeEscalation": False, + "readOnlyRootFilesystem": True, + "capabilities": {"drop": ["ALL"]}, + }, + "resources": {"requests": dict(resources), "limits": dict(resources)}, + "volumeMounts": [ + {"name": "tmp", "mountPath": "/tmp"}, + {"name": "cache", "mountPath": "/home/ray/.cache"}, + {"name": "shm", "mountPath": "/dev/shm"}, + ], + } + ], + "volumes": [ + {"name": "tmp", "emptyDir": {}}, + {"name": "cache", "emptyDir": {}}, + {"name": "shm", "emptyDir": {"medium": "Memory"}}, + ], + # Long enough for a replica to drain its in-flight generations. + "terminationGracePeriodSeconds": 60 if head else 120, + } + if node_selector: + spec["nodeSelector"] = node_selector + spec["tolerations"] = [ + {"key": "nvidia.com/gpu", "operator": "Exists", "effect": "NoSchedule"} + ] + return { + "metadata": { + "labels": { + "app.kubernetes.io/name": POD_LABEL, + "app.kubernetes.io/part-of": "local-llm-router", + } + }, + "spec": spec, + } + + +def _worker_group(pool: WorkerPool, image: str) -> dict[str, Any]: + product = GPU_PRODUCT_LABELS.get(pool.ray_accelerator, f"NVIDIA-{pool.ray_accelerator}") + return { + "groupName": pool.group_name, + "replicas": pool.min_workers, + "minReplicas": pool.min_workers, + "maxReplicas": pool.max_workers, + "rayStartParams": { + "metrics-export-port": str(METRICS_PORT), + # The custom resource pins a model to its own pool even when two + # pools could both satisfy a bare GPU request. + "resources": json.dumps(json.dumps({pool.resource: 1})), + }, + "template": _pod( + image=image, + cpu=str(4 * pool.gpus_per_worker), + memory=f"{pool.memory_gb}Gi", + gpus=pool.gpus_per_worker, + node_selector={"nvidia.com/gpu.product": product}, + ), + } + + +def build_ray_service( + registry: Registry, + artifact_root: str = DEFAULT_ARTIFACT_ROOT, + image: str = RAY_IMAGE, +) -> dict[str, Any]: + """Render the RayService that serves every local model in the catalog.""" + + serve_config = yaml.safe_dump(build_serve_application(registry, artifact_root), sort_keys=False) + return { + "apiVersion": "ray.io/v1", + "kind": "RayService", + "metadata": { + "name": SERVICE_NAME, + "namespace": NAMESPACE, + "labels": {"app.kubernetes.io/part-of": "local-llm-router"}, + "annotations": {"llm-routing/policy-version": registry.policy.version}, + }, + "spec": { + "serveConfigV2": _Block(serve_config), + "rayClusterConfig": { + "rayVersion": RAY_VERSION, + "enableInTreeAutoscaling": True, + "headGroupSpec": { + "rayStartParams": { + "dashboard-host": "0.0.0.0", + "metrics-export-port": str(METRICS_PORT), + # The head schedules; it never runs a replica. + "num-cpus": "0", + }, + "template": _pod(image=image, cpu="2", memory="8Gi", head=True), + }, + "workerGroupSpecs": [_worker_group(pool, image) for pool in worker_pools(registry)], + }, + }, + } + + +HEADER = ( + "# Generated from config/registry.yaml by `python -m llm_router.topology`.\n" + "# Do not edit: change the catalog and regenerate. A test and CD both fail on drift.\n" +) + + +def render_ray_service(registry: Registry, artifact_root: str = DEFAULT_ARTIFACT_ROOT) -> str: + manifest = build_ray_service(registry, artifact_root) + return HEADER + yaml.dump(manifest, Dumper=_Dumper, sort_keys=False, width=1000) + + +def main(argv: Sequence[str] | None = None) -> int: + """Print the RayService manifest for the committed catalog.""" + + import argparse + import sys + + from llm_router.registry import load_registry + + parser = argparse.ArgumentParser(description=main.__doc__) + parser.add_argument("--catalog", default="config/registry.yaml") + parser.add_argument("--artifact-root", default=DEFAULT_ARTIFACT_ROOT) + arguments = parser.parse_args(argv) + sys.stdout.write(render_ray_service(load_registry(arguments.catalog), arguments.artifact_root)) + return 0 + + +if __name__ == "__main__": # pragma: no cover - command-line entry point + raise SystemExit(main()) diff --git a/tests/unit/test_topology.py b/tests/unit/test_topology.py new file mode 100644 index 0000000..bbd1bd7 --- /dev/null +++ b/tests/unit/test_topology.py @@ -0,0 +1,372 @@ +"""The Ray head and worker topology, GPU support, dashboards and alerts (section 17).""" + +import json +import re +import shutil +import subprocess +from pathlib import Path +from typing import Any + +import pytest +import yaml +from fastapi.testclient import TestClient + +from llm_router.app import create_app +from llm_router.config import Settings +from llm_router.governance import governed_versions +from llm_router.registry import load_registry +from llm_router.topology import ( + build_ray_service, + build_serve_application, + main, + ray_accelerator, + render_ray_service, + worker_pools, +) + +CATALOG = load_registry("config/registry.yaml") +BASE = Path("deploy/kubernetes") +OVERLAY = Path("deploy/overlays/ray") +SERVICE = build_ray_service(CATALOG) +CLUSTER = SERVICE["spec"]["rayClusterConfig"] +# The fields Ray's LLMConfig accepts; anything else is rejected at deploy time. +LLM_CONFIG_FIELDS = { + "model_loading_config", + "engine_kwargs", + "accelerator_type", + "deployment_config", + "lora_config", + "log_engine_metrics", +} + + +def llm_configs() -> dict[str, dict[str, Any]]: + application = build_serve_application(CATALOG)["applications"][0] + return { + item["model_loading_config"]["model_id"]: item + for item in application["args"]["llm_configs"] + } + + +def worker_group(name: str) -> dict[str, Any]: + return next(group for group in CLUSTER["workerGroupSpecs"] if group["groupName"] == name) + + +def pod_templates() -> list[dict[str, Any]]: + return [ + CLUSTER["headGroupSpec"]["template"], + *(group["template"] for group in CLUSTER["workerGroupSpecs"]), + ] + + +def test_accelerators_are_named_the_way_ray_knows_them() -> None: + assert ray_accelerator("nvidia-l4") == "L4" + assert ray_accelerator("nvidia-a10g") == "A10G" + + +def test_each_accelerator_type_gets_its_own_pool_sized_from_its_models() -> None: + pools = {pool.group_name: pool for pool in worker_pools(CATALOG)} + + assert set(pools) == {"l4-pool", "a10g-pool", "a100-pool"} + assert (pools["l4-pool"].min_workers, pools["l4-pool"].max_workers) == (1, 4) + # The experiment adds headroom to the ceiling and nothing to the floor. + assert (pools["a10g-pool"].min_workers, pools["a10g-pool"].max_workers) == (1, 5) + assert pools["a10g-pool"].models == ("general-local", "general-local--general-gptq") + # The high-capability tier scales to zero and needs both GPUs on one node. + assert (pools["a100-pool"].min_workers, pools["a100-pool"].max_workers) == (0, 3) + assert pools["a100-pool"].gpus_per_worker == 2 + + +def test_the_head_schedules_and_never_runs_a_model() -> None: + head = CLUSTER["headGroupSpec"] + container = head["template"]["spec"]["containers"][0] + + assert head["rayStartParams"]["num-cpus"] == "0" + assert "nvidia.com/gpu" not in container["resources"]["limits"] + assert "nodeSelector" not in head["template"]["spec"] + assert CLUSTER["enableInTreeAutoscaling"] is True + + +def test_workers_are_pinned_to_their_accelerator_and_advertise_their_pool() -> None: + group = worker_group("a100-pool") + spec = group["template"]["spec"] + + assert spec["nodeSelector"] == {"nvidia.com/gpu.product": "NVIDIA-A100-SXM4-80GB"} + assert spec["tolerations"][0]["key"] == "nvidia.com/gpu" + assert spec["containers"][0]["resources"]["limits"]["nvidia.com/gpu"] == "2" + assert (group["minReplicas"], group["maxReplicas"]) == (0, 3) + # KubeRay wants the resource map as a quoted JSON string. + assert json.loads(json.loads(group["rayStartParams"]["resources"])) == { + "gpu_pool_nvidia-a100": 1 + } + assert worker_group("l4-pool")["template"]["spec"]["nodeSelector"] == { + "nvidia.com/gpu.product": "NVIDIA-L4" + } + + +def test_every_model_asks_for_the_pool_resource_a_worker_group_provides() -> None: + provided = { + resource + for group in CLUSTER["workerGroupSpecs"] + for resource in json.loads(json.loads(group["rayStartParams"]["resources"])) + } + + for model_id, config in llm_configs().items(): + wanted = set(config["deployment_config"]["ray_actor_options"]["resources"]) + assert wanted <= provided, model_id + + +def test_the_serve_application_uses_only_fields_ray_accepts() -> None: + application = build_serve_application(CATALOG)["applications"][0] + + assert application["import_path"] == "ray.serve.llm:build_openai_app" + for model_id, config in llm_configs().items(): + assert set(config) <= LLM_CONFIG_FIELDS, model_id + assert set(config["model_loading_config"]) == {"model_id", "model_source"} + small = llm_configs()["small-specialist"] + assert small["accelerator_type"] == "L4" + assert small["lora_config"] == { + "dynamic_lora_loading_path": "s3://llm-routing-artifacts/adapters", + "max_num_adapters_per_replica": 3, + } + assert "lora_config" not in llm_configs()["general-local"] + + +def test_ray_loads_the_artifact_governance_recorded() -> None: + recorded = {item.name: item.source for item in governed_versions(CATALOG, "s3://bucket")} + application = build_serve_application(CATALOG, "s3://bucket/")["applications"][0] + configs = { + item["model_loading_config"]["model_id"]: item + for item in application["args"]["llm_configs"] + } + + for model_id in ("small-specialist", "general-local", "high-capability"): + assert configs[model_id]["model_loading_config"]["model_source"] == recorded[model_id] + # A variant serves its base model's weights, and a draft model its own. + speculative = configs["high-capability--high-capability-speculative"] + assert speculative["model_loading_config"]["model_source"] == recorded["high-capability"] + assert ( + speculative["engine_kwargs"]["speculative_config"]["model"] + == (recorded["small-specialist"]) + ) + assert "approved-external-fallback" not in configs + + +@pytest.mark.parametrize( + "template", pod_templates(), ids=lambda item: item["spec"]["containers"][0]["name"] +) +def test_ray_pods_meet_the_same_contract_as_every_other_workload(template: dict[str, Any]) -> None: + spec = template["spec"] + container = spec["containers"][0] + + assert template["metadata"]["labels"]["app.kubernetes.io/name"] == "ray-serve" + assert spec["securityContext"]["runAsNonRoot"] is True + assert container["securityContext"]["allowPrivilegeEscalation"] is False + assert container["securityContext"]["readOnlyRootFilesystem"] is True + assert container["securityContext"]["capabilities"]["drop"] == ["ALL"] + assert "@sha256:" in container["image"] + assert container["resources"]["requests"] == container["resources"]["limits"] + assert any(port["name"] == "metrics" for port in container["ports"]) + + +def test_the_committed_manifest_is_what_the_catalog_generates( + capsys: pytest.CaptureFixture[str], +) -> None: + committed = (OVERLAY / "ray-service.yaml").read_text(encoding="utf-8") + + assert committed == render_ray_service(CATALOG) + assert main([]) == 0 + assert capsys.readouterr().out == committed + embedded = yaml.safe_load(yaml.safe_load(committed)["spec"]["serveConfigV2"]) + assert embedded == build_serve_application(CATALOG) + + +def test_the_overlay_lists_files_that_exist_and_replaces_the_single_engine() -> None: + kustomization = yaml.safe_load((OVERLAY / "kustomization.yaml").read_text(encoding="utf-8")) + + assert kustomization["resources"][0] == "../../kubernetes" + for resource in kustomization["resources"][1:]: + assert (OVERLAY / resource).is_file(), resource + removed = { + (patch["kind"], patch["metadata"]["name"]) + for patch in ( + yaml.safe_load(item["patch"]) for item in kustomization["patches"] if "patch" in item + ) + if patch.get("$patch") == "delete" + } + assert removed == { + ("Deployment", "vllm-serve"), + ("Service", "vllm-serve"), + ("NetworkPolicy", "vllm-serve"), + } + + +def test_inference_reaches_ray_only_from_the_gateway() -> None: + policy = yaml.safe_load((OVERLAY / "network-policy.yaml").read_text(encoding="utf-8"))["spec"] + gateway = yaml.safe_load( + (OVERLAY / "gateway-network-policy.patch.yaml").read_text(encoding="utf-8") + )["spec"] + + serve_sources = [ + entry["from"] + for entry in policy["ingress"] + if any(port["port"] == 8000 for port in entry.get("ports", [])) + ] + assert serve_sources == [ + [{"podSelector": {"matchLabels": {"app.kubernetes.io/name": "llm-gateway"}}}] + ] + targets = { + rule["podSelector"]["matchLabels"]["app.kubernetes.io/name"] + for entry in gateway["egress"] + for rule in entry["to"] + } + assert targets == {"ray-serve", "redis", "litellm-proxy"} + for entry in policy["egress"]: + assert all("ipBlock" not in rule for rule in entry["to"]) + + +@pytest.mark.skipif(shutil.which("kubectl") is None, reason="kubectl is not installed") +def test_the_overlay_renders_with_ray_in_place_of_the_single_engine() -> None: + rendered = subprocess.run( + ["kubectl", "kustomize", str(OVERLAY)], capture_output=True, text=True, check=True + ) + documents = [item for item in yaml.safe_load_all(rendered.stdout) if item] + names = {(item["kind"], item["metadata"]["name"]) for item in documents} + + assert ("RayService", "llm-serve") in names + assert ("PodMonitor", "ray-serve") in names + assert not any(name == "vllm-serve" for _, name in names) + gateway = next( + item + for item in documents + if item["metadata"]["name"] == "llm-gateway" and item["kind"] == "Deployment" + ) + environment = { + item["name"]: item.get("value") + for item in gateway["spec"]["template"]["spec"]["containers"][0]["env"] + } + assert environment["ROUTER_VLLM_BASE_URL"] == ( + "http://llm-serve-serve-svc.llm-routing.svc.cluster.local:8000" + ) + # Everything the base promised about the gateway is still there. + assert environment["ROUTER_BACKEND"] == "vllm" + assert "ROUTER_API_KEYS" in environment + + +def emitted_metrics() -> set[str]: + """Metric families the gateway publishes, without their sample suffixes.""" + + with TestClient(create_app(Settings(api_keys="k"))) as client: + exposition = client.get("/metrics").text + return { + re.sub(r"_total$", "", line.split()[2]) + for line in exposition.splitlines() + if line.startswith("# TYPE router_") + } + + +def referenced_metrics(expression: str) -> set[str]: + names = set(re.findall(r"\b(?:router_|DCGM_)[A-Za-z0-9_]+", expression)) + return {re.sub(r"_(total|bucket|sum|count)$", "", name) for name in names} + + +def dashboards() -> dict[str, dict[str, Any]]: + return { + path.name: json.loads(path.read_text(encoding="utf-8")) + for path in sorted((BASE / "dashboards").glob("*.json")) + } + + +def alert_rules() -> list[dict[str, Any]]: + documents = yaml.safe_load_all((BASE / "observability.yaml").read_text(encoding="utf-8")) + rule = next(item for item in documents if item["kind"] == "PrometheusRule") + return [alert for group in rule["spec"]["groups"] for alert in group["rules"]] + + +def test_dashboards_and_alerts_only_query_metrics_that_exist() -> None: + emitted = emitted_metrics() + expressions = [ + target["expr"] + for board in dashboards().values() + for panel in board["panels"] + for target in panel["targets"] + ] + [alert["expr"] for alert in alert_rules()] + + assert len(expressions) > 30 + for expression in expressions: + used = referenced_metrics(expression) + assert used, expression + for name in used: + if name.startswith("DCGM_"): + assert name in { + "DCGM_FI_DEV_GPU_UTIL", + "DCGM_FI_DEV_FB_USED", + "DCGM_FI_DEV_FB_FREE", + } + else: + assert name in emitted, f"{name} in {expression}" + + +def test_the_spec_inference_and_routing_metrics_are_all_on_a_dashboard() -> None: + charted = { + name + for board in dashboards().values() + for panel in board["panels"] + for target in panel["targets"] + for name in referenced_metrics(target["expr"]) + } + + assert { + "router_time_to_first_token_seconds", + "router_time_per_output_token_seconds", + "router_request_latency_seconds", + "router_requests", + "router_tokens", + "router_engine_batch_size", + "router_queued_requests", + "router_inflight_requests", + "router_gpu_utilization_ratio", + "router_gpu_memory_used_bytes", + "router_engine_kv_cache_occupancy_ratio", + "router_model_load_seconds", + "router_routes", + "router_fallbacks", + "router_queue_delay_prediction_error_ms", + "router_predicted_quality", + "router_observed_quality", + "router_cache_events", + "router_rejections", + } <= charted + + +def test_every_dashboard_is_shipped_to_grafana() -> None: + kustomization = yaml.safe_load((BASE / "kustomization.yaml").read_text(encoding="utf-8")) + generator = kustomization["configMapGenerator"][0] + + assert {Path(item).name for item in generator["files"]} == set(dashboards()) + assert generator["options"]["labels"] == {"grafana_dashboard": "1"} + assert generator["options"]["disableNameSuffixHash"] is True + for board in dashboards().values(): + assert board["uid"].startswith("llm-routing-") and board["panels"] + + +def test_alerts_filter_on_rejection_types_the_gateway_really_reports() -> None: + source = Path("src/llm_router/app.py").read_text(encoding="utf-8") + shedding = next(item for item in alert_rules() if item["alert"] == "GatewayRejectingRequests") + + for rejection in re.search(r'type=~"([^"]+)"', shedding["expr"]).group(1).split("|"): # type: ignore[union-attr] + assert f'record_rejection("{rejection}")' in source + for alert in alert_rules(): + assert alert["labels"]["severity"] in {"warning", "critical"} + assert alert["annotations"]["summary"] + + +def test_the_gpu_operator_supplies_what_the_serving_manifests_rely_on() -> None: + values = yaml.safe_load(Path("deploy/gpu-operator/values.yaml").read_text(encoding="utf-8")) + + # The nvidia.com/gpu resource, the gpu.product label, and the GPU exporter. + assert values["devicePlugin"]["enabled"] and values["driver"]["enabled"] + assert values["nfd"]["enabled"] and values["gfd"]["enabled"] + assert values["dcgmExporter"]["enabled"] + assert values["dcgmExporter"]["serviceMonitor"]["enabled"] + assert values["daemonsets"]["tolerations"][0]["key"] == "nvidia.com/gpu"