Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions .github/workflows/cd.yml
Original file line number Diff line number Diff line change
Expand Up @@ -66,6 +66,8 @@ jobs:
run: python -m llm_router.chart --check && helm lint deploy/helm/llm-routing
- name: Package the Helm chart
run: helm package deploy/helm/llm-routing --destination chart
- name: Verify every catalog adapter has a matching recipe
run: python -m llm_router.adapters check
- name: Render the canary and rollback plans
# One plan per track (model, adapter, policy), each naming what it
# rolls back to and the criteria that trigger it.
Expand Down
26 changes: 26 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -310,6 +310,32 @@ be served. A request can never introduce a model path, revision, or adapter.
Send `routing.domain` to request a domain adapter; the router applies the promoted adapter
with the largest measured quality gain for that base revision and task, or none at all.

### Adapter recipes

Every adapter in the catalog has a recipe in [`config/adapters`](config/adapters): the base
model commit it was trained from, the dataset version, and the LoRA hyperparameters. A recipe
is what makes an adapter reproducible, so a catalog adapter without one fails the check.

```bash
python -m pip install -e ".[training]" # PEFT; add ".[unsloth]" for Unsloth
python -m llm_router.adapters check # recipes agree with the catalog
python -m llm_router.adapters plan claims-extraction-lora # the arguments a run would get
python -m llm_router.adapters train claims-extraction-lora --output ./out/claims
python -m llm_router.adapters register claims-extraction-lora --output ./out/claims
```

- `method: lora` trains over the full-precision base; `method: qlora` loads the base in 4-bit.
`framework` selects PEFT or Unsloth, and both are given the same rank, alpha, target modules,
commit, and seed.
- The base is pinned to a 40-character commit. A branch or tag is rejected, because it can move.
- A rank above 32 is rejected: the serving configuration would not load it.
- `register` runs the [artifact scan](#artifact-scanning) on the finished adapter and prints its
catalog entry. The entry starts in `development` with no measured gain, and its revision is the
artifact's digest. Promotion is a separate, reviewed change once a benchmark exists.

Training has not been run: it needs a GPU. The recipes' `hf_repo` and `hf_revision` are
placeholders, as the catalog's base models are mocks.

### Governance in MLflow

The catalog decides what is served; [MLflow](https://mlflow.org/docs/latest/) keeps the record.
Expand Down
24 changes: 24 additions & 0 deletions config/adapters/claims-extraction-lora-next.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
# Recipe for the staged claims adapter: same base, the next dataset version,
# and full-precision LoRA. hf_repo and hf_revision are placeholders.
id: claims-extraction-lora-next
base_model_id: small-specialist
base_revision: mock-small@sha256:dev
hf_repo: REPLACE_ME/small-specialist
hf_revision: "0000000000000000000000000000000000000000"
domain: claims
intended_tasks: [extraction]
method: lora
framework: peft
rank: 16
alpha: 32
dropout: 0.05
target_modules: [q_proj, k_proj, v_proj, o_proj]
dataset:
path: s3://llm-routing-artifacts/datasets/claims-2026-06.jsonl
version: claims-2026-06
training:
epochs: 3
learning_rate: 0.0002
batch_size: 8
max_seq_length: 2048
seed: 7
25 changes: 25 additions & 0 deletions config/adapters/claims-extraction-lora.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
# Recipe for the claims extraction adapter. The catalog's base model is a mock,
# so hf_repo and hf_revision are placeholders to replace with the real base.
id: claims-extraction-lora
base_model_id: small-specialist
base_revision: mock-small@sha256:dev
hf_repo: REPLACE_ME/small-specialist
hf_revision: "0000000000000000000000000000000000000000"
domain: claims
intended_tasks: [extraction]
# QLoRA: trained over the base loaded in 4-bit, to fit an L4.
method: qlora
framework: unsloth
rank: 16
alpha: 32
dropout: 0.05
target_modules: [q_proj, k_proj, v_proj, o_proj]
dataset:
path: s3://llm-routing-artifacts/datasets/claims-2026-05.jsonl
version: claims-2026-05
training:
epochs: 3
learning_rate: 0.0002
batch_size: 8
max_seq_length: 2048
seed: 7
24 changes: 24 additions & 0 deletions config/adapters/support-classification-lora.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
# Recipe for the support classification adapter. hf_repo and hf_revision are
# placeholders to replace with the real base.
id: support-classification-lora
base_model_id: small-specialist
base_revision: mock-small@sha256:dev
hf_repo: REPLACE_ME/small-specialist
hf_revision: "0000000000000000000000000000000000000000"
domain: support
intended_tasks: [classification]
method: lora
framework: peft
rank: 8
alpha: 16
dropout: 0.05
target_modules: [q_proj, v_proj]
dataset:
path: s3://llm-routing-artifacts/datasets/support-2026-04.jsonl
version: support-2026-04
training:
epochs: 2
learning_rate: 0.0001
batch_size: 16
max_seq_length: 1024
seed: 7
19 changes: 19 additions & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,18 @@ governance = [
"mlflow-skinny>=3.5,<4",
"sqlalchemy>=2,<3",
]
# Adapter training. Needs a GPU; nothing in the verification environment
# installs it. Unsloth is an alternative trainer, installed separately.
training = [
"accelerate>=1.0",
"bitsandbytes>=0.44",
"datasets>=3.0",
"peft>=0.13",
"transformers>=4.46",
]
unsloth = [
"unsloth",
]
tracing = [
"opentelemetry-exporter-otlp-proto-http>=1.30,<2",
"opentelemetry-sdk>=1.30,<2",
Expand Down Expand Up @@ -87,6 +99,13 @@ module = ["redis.*"]
ignore_missing_imports = true
follow_imports = "skip"

[[tool.mypy.overrides]]
# The training stack ships in optional extras and is imported only inside
# the function that trains an adapter.
module = ["datasets.*", "peft.*", "transformers.*", "unsloth.*"]
ignore_missing_imports = true
follow_imports = "skip"

[[tool.mypy.overrides]]
# MLflow ships in the optional governance extra and is imported lazily, only
# when a tracking URI is given; the client is used behind GovernanceStore.
Expand Down
Loading
Loading