diff --git a/CHANGELOG.md b/CHANGELOG.md index de81bbc1..6fa065b4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,23 @@ adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ## [Unreleased] +### Added + +- **NeMo Relay native plugin** — a dynamically loaded integration that runs + libsy's weighted-random, LLM-classifier, escalation, and stage-router + algorithms in process while Switchyard owns provider HTTP dispatch, + credentials, translation, retries, and fallback. Managed calls require NeMo + Relay 0.7 or newer and do not depend on `switchyard-server`. Target bindings + accept non-secret `extra_body` provider defaults, preserved requests are + re-encoded after routing mutations, and synthetic Relay gateway identities do + not become shared router session state. + +- **NeMo Relay routing-model usage marks** — classifier judges, escalation + judges and discarded weak candidates, and failed routing candidates now emit + `switchyard.routing.llm_call` ATOF marks with normalized token usage and + latency. The final serving call remains represented only by Relay's outer LLM + lifecycle event to prevent double-counting. + ### Removed - **Deprecated Python server stack** — `switchyard serve`, YAML route bundles, diff --git a/Cargo.lock b/Cargo.lock index 39871f0e..b318bdbe 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -106,6 +106,18 @@ dependencies = [ "serde_json", ] +[[package]] +name = "async-channel" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "924ed96dd52d1b75e9c1a3e6275715fd320f5f9439fb5a4a11fa51f4221158d2" +dependencies = [ + "concurrent-queue", + "event-listener-strategy", + "futures-core", + "pin-project-lite", +] + [[package]] name = "async-stream" version = "0.3.6" @@ -274,6 +286,9 @@ name = "bitflags" version = "2.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" +dependencies = [ + "serde_core", +] [[package]] name = "borrow-or-share" @@ -334,6 +349,16 @@ dependencies = [ "rand_core 0.10.1", ] +[[package]] +name = "chrono" +version = "0.4.45" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1aa79e62e7697b8e29b513a68abacf485adcd1fe8284a4316c5ae868e6633327" +dependencies = [ + "num-traits", + "serde", +] + [[package]] name = "clap" version = "4.6.2" @@ -399,6 +424,15 @@ dependencies = [ "memchr", ] +[[package]] +name = "concurrent-queue" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4ca0197aee26d1ae37445ee532fefce43251d24cc7c166799f4d46817f1d3973" +dependencies = [ + "crossbeam-utils", +] + [[package]] name = "core-foundation" version = "0.10.1" @@ -424,6 +458,12 @@ dependencies = [ "libc", ] +[[package]] +name = "crossbeam-utils" +version = "0.8.22" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "61803da095bee82a81bb1a452ecc25d3b2f1416d1897eb86430c6159ef717c17" + [[package]] name = "data-encoding" version = "2.11.1" @@ -502,6 +542,26 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "event-listener" +version = "5.4.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5a23add41df1562121a9393cb065eab5146a1242410f23a644851e90cfd669d2" +dependencies = [ + "parking", + "pin-project-lite", +] + +[[package]] +name = "event-listener-strategy" +version = "0.5.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8be9f3dfaaffdae2972880079a491a1a8bb7cbed0b8dd7a347f668b4150a3b93" +dependencies = [ + "event-listener", + "pin-project-lite", +] + [[package]] name = "fancy-regex" version = "0.18.0" @@ -1226,6 +1286,31 @@ dependencies = [ "windows-sys 0.61.2", ] +[[package]] +name = "nemo-relay-plugin" +version = "0.7.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "df37bebc79a6d757a7cb18d6c94ca0825fcb6c8e51ab4e13900e7d5853e0de8b" +dependencies = [ + "nemo-relay-types", + "serde", + "serde_json", +] + +[[package]] +name = "nemo-relay-types" +version = "0.7.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "99ea078c95f9e0804a77a0beb5d86f003ff4a9315c27e1a7afb2cee98bd4d7fd" +dependencies = [ + "bitflags", + "chrono", + "serde", + "serde_json", + "typed-builder", + "uuid", +] + [[package]] name = "nu-ansi-term" version = "0.50.3" @@ -1431,6 +1516,12 @@ version = "0.5.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1a80800c0488c3a21695ea981a54918fbb37abf04f4d0720c453632255e2ff0e" +[[package]] +name = "parking" +version = "2.2.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f38d5652c16fde515bb1ecef450ab0f6a219d619a7274976324d5e377f7dceba" + [[package]] name = "parking_lot" version = "0.12.5" @@ -2311,6 +2402,24 @@ dependencies = [ "wiremock", ] +[[package]] +name = "switchyard-nemo-relay-plugin" +version = "0.2.0" +dependencies = [ + "async-channel", + "async-trait", + "futures-util", + "http", + "nemo-relay-plugin", + "serde", + "serde_json", + "switchyard-libsy", + "switchyard-llm-client", + "switchyard-protocol", + "switchyard-translation", + "tokio", +] + [[package]] name = "switchyard-protocol" version = "0.2.0" @@ -2766,6 +2875,26 @@ version = "0.2.5" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e421abadd41a4225275504ea4d6566923418b7f05506fbc9c0fe86ba7396114b" +[[package]] +name = "typed-builder" +version = "0.23.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "31aa81521b70f94402501d848ccc0ecaa8f93c8eb6999eb9747e72287757ffda" +dependencies = [ + "typed-builder-macro", +] + +[[package]] +name = "typed-builder-macro" +version = "0.23.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "076a02dc54dd46795c2e9c8282ed40bcfb1e22747e955de9389a1de28190fb26" +dependencies = [ + "proc-macro2", + "quote", + "syn 2.0.119", +] + [[package]] name = "unicode-general-category" version = "1.1.0" @@ -2808,6 +2937,18 @@ version = "0.2.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "06abde3611657adf66d383f00b093d7faecc7fa57071cce2578660c9f1010821" +[[package]] +name = "uuid" +version = "1.18.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2f87b8aa10b915a06587d0dec516c282ff295b475d94abf425d62b57710070a2" +dependencies = [ + "getrandom 0.3.4", + "js-sys", + "serde", + "wasm-bindgen", +] + [[package]] name = "uuid-simd" version = "0.8.0" diff --git a/Cargo.toml b/Cargo.toml index 07d133bb..82caff84 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -8,6 +8,7 @@ members = [ "crates/libsy-llm-client", "crates/switchyard-py", "crates/protocol", + "crates/switchyard-nemo-relay-plugin", "crates/switchyard-server", "crates/switchyard-skill-distillation", "crates/switchyard-translation", @@ -22,6 +23,7 @@ repository = "https://github.com/NVIDIA-NeMo/Switchyard" rust-version = "1.96.1" [workspace.dependencies] +async-channel = "2" async-stream = "0.3" async-trait = "0.1" futures = "0.3" @@ -30,6 +32,7 @@ http = "1" httpdate = "1" jsonschema = { version = "0.49.4", default-features = false } jsonptr = { version = "0.8.1", default-features = false, features = ["std", "json", "resolve"] } +nemo-relay-plugin = "=0.7.0" parking_lot = "0.12" rand = "0.10" reqwest = { version = "0.13.4", default-features = false, features = ["json", "rustls", "stream"] } diff --git a/README.md b/README.md index cdc2b51d..736b4066 100644 --- a/README.md +++ b/README.md @@ -21,6 +21,7 @@ algorithm you write yourself. - **Protocol Translation**: convert between OpenAI Chat, Anthropic Messages, and OpenAI Responses formats - **Multi-Backend Routing**: random routing, LLM-as-classifier routing, signal-driven stage-router, or your own algorithm - **Operational Metrics**: Prometheus metrics cover requests, errors, latency, tokens, and routing overhead +- **NeMo Relay Plugin**: run random, classifier, escalation, or stage routing in Relay while Switchyard owns provider HTTP dispatch ## Maturity @@ -154,6 +155,7 @@ configured LLM client selects one upstream format. - **[`switchyard-libsy`](crates/libsy/README.md)**: embed routing algorithms in a Rust application - **[`switchyard-protocol`](crates/protocol/README.md)**: provider-neutral request, response, and streaming types - **[`switchyard-translation`](crates/switchyard-translation/README.md)**: request, response, and stream translation +- **[`switchyard-nemo-relay-plugin`](crates/switchyard-nemo-relay-plugin/README.md)**: install Switchyard as a native NeMo Relay plugin ## Community diff --git a/crates/switchyard-nemo-relay-plugin/Cargo.toml b/crates/switchyard-nemo-relay-plugin/Cargo.toml new file mode 100644 index 00000000..243e66cc --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/Cargo.toml @@ -0,0 +1,30 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +[package] +name = "switchyard-nemo-relay-plugin" +version.workspace = true +description = "Switchyard-owned HTTP routing plugin for NeMo Relay" +authors.workspace = true +edition.workspace = true +license.workspace = true +repository.workspace = true +rust-version.workspace = true +publish = false + +[lib] +crate-type = ["cdylib"] + +[dependencies] +async-channel.workspace = true +async-trait.workspace = true +futures-util.workspace = true +http.workspace = true +nemo-relay-plugin.workspace = true +serde.workspace = true +serde_json.workspace = true +switchyard-libsy.workspace = true +switchyard-llm-client.workspace = true +switchyard-protocol.workspace = true +switchyard-translation.workspace = true +tokio.workspace = true diff --git a/crates/switchyard-nemo-relay-plugin/README.md b/crates/switchyard-nemo-relay-plugin/README.md new file mode 100644 index 00000000..d718b4a6 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/README.md @@ -0,0 +1,400 @@ + + +# Switchyard NeMo Relay Dynamic Plugin + +This crate builds the external `nvidia.switchyard` native plugin. It embeds +`switchyard-libsy`, drives it through `switchyard-llm-client::run`, and uses +`switchyard-llm-client` for provider HTTP calls. Managed calls use Relay's +completion-based asynchronous middleware hooks and do not require a targeted +provider continuation from Relay. + +The plugin uses NeMo Relay native API v1. It depends on the small +`nemo-relay-plugin` authoring SDK, not the Relay runtime, and does not start +`switchyard-server`. Managed provider calls do not use Relay's provider +continuation. + +## Ownership boundary + +For a managed LLM call: + +1. Relay invokes the native LLM execution intercept. +2. The plugin decodes the caller JSON through `switchyard-translation`. +3. The plugin passes the configured algorithm and its target-to-client map to + `switchyard-llm-client::run`, using the library's public execution and + observation boundary. +4. For every routed call, the selected target client translates the neutral + request, applies its URL and credentials, and performs the HTTP request. +5. `switchyard-llm-client` drives libsy to its final response while the plugin + records decisions and routing-only model usage. +6. The plugin encodes the final neutral response into the caller's protocol. + +Relay still owns the outer LLM lifecycle, dynamic-plugin loading, plugin +configuration, and event substrate. Relay's downstream LLM continuation is +used only for calls whose inbound protocol is not managed by this plugin. + +```mermaid +flowchart LR + A["Caller JSON"] --> B["Relay LLM execution intercept"] + B --> C["Switchyard decode"] + C --> D["switchyard-llm-client run"] + D --> E["libsy algorithm"] + E --> F["target ClientRouter"] + F --> G["Provider HTTP endpoint"] + G --> H["Switchyard response or event decode"] + H --> I["libsy final response"] + I --> J["routing observations"] + J --> K["Switchyard encode"] + K --> A + + U["Unmanaged profile"] -.-> V["Relay v1 continuation"] +``` + +This boundary has two important consequences: + +- Managed provider calls do not traverse Relay middleware registered after the + Switchyard intercept and do not use the host's provider callback. Provider + transport activity is therefore not represented as nested Relay LLM + lifecycle events. Relay records the outer managed call and the plugin emits + Switchyard routing marks; bridging Switchyard transport spans into Relay is + future work. The adapter captures the active Relay scope before returning + `Pending`, so asynchronous routing marks retain their event parent. +- Switchyard owns provider URLs, credentials, HTTP retry behavior, and + translation for managed calls. Relay neither validates nor transports those + target details. + +## Native API v1 and asynchronous execution + +The manifest remains `compat.native_api = "1"`, but the plugin requires the +generic host-table v3 extension shipped by Relay 0.7. It registers through +v3's completion-based buffered and incremental streaming hooks, returns +`Pending` immediately, and performs libsy and provider HTTP work on a +plugin-owned Tokio runtime. Relay workers therefore do not wait synchronously +for provider I/O. + +The stream adapter forwards the plugin's bounded 32-message channel into +Relay's bounded output queue. It retries a logical event when the host queue is +full and checks cancellation between attempts. Managed HTTP work is selected +against Relay caller cancellation, so cancelling a buffered or streaming call +drops its in-flight provider future. + +Unmanaged profiles use the same v3 continuation hooks for pass-through. V3's +downstream stream callback has continue/cancel control but no asynchronous +acknowledgement, so the adapter uses a nonblocking bridge capped at 8 MiB of +queued encoded payloads and 256 events. A pass-through stream that outruns +either bound is rejected rather than consuming unbounded memory. + +This is a raw C boundary: Switchyard contains a small ownership adapter for +host strings, completion and stream handles, continuation handles, and captured +scope handles because Relay 0.7 does not expose a safe Rust facade for its +generic asynchronous surface. The HTTP, routing, and translation behavior +remains in Switchyard. NeMo Relay plans to provide the equivalent safe typed +surface in 0.8.0; once that is available, the raw-FFI compatibility adapter +should be removable. + +## Supported routers + +The plugin supports four libsy routing modes: + +- seeded, weighted `random` routing; and +- capability-based `llm_classifier` routing, where a judge selects the weak or + strong target before the final provider call; +- escalation-mode `llm_classifier` routing, where a judge evaluates the weak + model's completed turn and latches a session to the strong target after a + configured confirmation streak; and +- signal-driven `stage_router` routing, with optional handoff notes, tier + prompts, and a capability-classifier fallback for ambiguous turns. + +Unsupported algorithm kinds are rejected instead of being approximated. + +## Compatibility Matrix + +The following matrix describes the algorithm behavior implemented by the +plugin. `Conditional` means that the feature is implemented with the constraint +shown in the table; it does not mean that the feature falls back to a different +algorithm. + +| Compatibility Area | `random` | `llm_classifier` (`capability`) | `llm_classifier` (`escalation`) | `stage_router` | +|---|---|---|---|---| +| Version-2 configuration and static validation | Supported | Supported | Supported | Supported | +| Caller protocols | OpenAI Chat, OpenAI Responses, Anthropic Messages | OpenAI Chat, OpenAI Responses, Anthropic Messages | OpenAI Chat, OpenAI Responses, Anthropic Messages | OpenAI Chat, OpenAI Responses, Anthropic Messages | +| Serving-target protocols | OpenAI Chat, OpenAI Responses, Anthropic Messages | OpenAI Chat, OpenAI Responses, Anthropic Messages | OpenAI Chat, OpenAI Responses, Anthropic Messages | OpenAI Chat, OpenAI Responses, Anthropic Messages | +| Structured-output judge protocols | Not applicable | OpenAI Chat or OpenAI Responses | OpenAI Chat or OpenAI Responses | OpenAI Chat or OpenAI Responses for the optional classifier | +| Buffered responses | Supported | Supported | Supported | Supported | +| Streaming responses | Supported | Supported after the judge selects a target | Conditional: an unlatched weak stream is aggregated before the judge runs | Supported after the signal cascade selects a target | +| Retained routing state | No selection affinity; context-overflow eviction can use session identity | Optional session affinity and message-hash fallback | Confirmation streak and strong latch require stable session identity | No classifier affinity; context-overflow eviction can use session identity | +| Router-specific prompts | Not applicable | Optional judge prompt | Optional escalation-judge prompt | Optional tier prompts, handoff notes, and classifier prompt | +| Relay decision marks | Algorithm, attempt, selected target, reasoning, answer-call status, and identity | Algorithm, attempt, selected target, reasoning, answer-call status, and identity | Algorithm, attempt, selected target, reasoning, answer-call status, and identity | Algorithm, attempt, selected target, reasoning, answer-call status, and identity | +| ATOF routing-LLM usage | Not applicable unless a failed candidate is replaced | Judge calls, plus failed candidates | Judge calls and discarded weak candidates | Optional classifier judge calls, plus failed candidates | + +Anthropic Messages is supported for callers and serving targets, but not for a +structured-output judge. That restriction is intentional and fails during +static configuration loading. Same-protocol streaming preserves parsed provider +events when the router does not aggregate or replace them; raw SSE bytes and +framing are not part of the compatibility contract. + +### Known issue: OpenAI Responses structured-output judges + +OpenAI Responses targets are accepted for structured-output judges, but the +shared Responses request encoder currently emits the Chat-compatible JSON +Schema object directly under `text.format`. This places `name`, `schema`, and +`strict` under `text.format.json_schema`; conforming Responses endpoints expect +those fields directly under `text.format`. InferenceHub therefore returns HTTP +400 with `Missing required parameter: 'text.format.name'`, and the affected +router follows its existing judge-failure or fall-open path. + +OpenAI Responses remains supported as a caller and ordinary serving-target +protocol. Until the shared `switchyard-translation` encoder is corrected, +configure structured-output judges with `protocol = "openai_chat"`. Follow-up +work must add the inverse of the existing Responses-to-neutral schema conversion +plus core and process-level regression coverage for all three affected router +paths. + +Managed inner provider calls also do not re-enter Relay's downstream provider +middleware. This behavior is part of the current ownership boundary, not an +automatic compatibility fallback. + +Each completed routing-only model call emits a +`switchyard.routing.llm_call` ATOF mark. Its data identifies the algorithm, +attempt, call order, target, role (`judge` or discarded `candidate`), outcome, +latency, and normalized provider token `usage`. The +successful call that serves the caller is deliberately excluded because +Relay's outer LLM end event already records that usage. A failed call, or a +provider response that omits usage, has `usage = null`. Consumers can therefore +add these marks to the outer LLM usage to measure total request compute without +double-counting the serving model. + +The plugin owns the outer routing retry loop. Each retry starts a fresh libsy +run. Random routing draws again; an algorithm configured with persistent state, +such as classifier session affinity, may intentionally retain its assignment. +Each target's built-in HTTP retry count is set to zero to avoid retrying a +failed target invisibly before reselection. A random target with `weight = 0` +is fallback-only and is not considered by the algorithm. Trusted fallback is +attempted at most once and, for streaming responses, only before the first +caller event is emitted. Outer routing retries use exponential backoff starting +at 250 milliseconds and capped at 2 seconds. They do not currently honor +provider `Retry-After` headers because the client error contract does not expose +that metadata to the routing loop. + +## Translation and stream fidelity + +`switchyard-translation` is the only request, response, and event translation +layer. It decodes caller JSON into Switchyard's neutral protocol, encodes each +selected call for the target protocol, decodes provider results, and encodes +`ReturnToAgent` back to the caller protocol. Relay codecs are not used. + +The streaming contract carries each parsed provider JSON event in a preservation +envelope alongside its normalized `LlmResponseChunk` representation. +Same-protocol routes replay the preserved JSON unchanged, including +provider-specific fields; this preserves parsed events, not raw SSE bytes or +framing. Cross-protocol routes encode only normalized chunks, and the streaming +helpers still do not expose the buffered translation engine's reject-lossy +diagnostics, so unsupported fields may be normalized or omitted. Replacing +normalized stream content or folding a stream into an aggregate drops the +per-event preservation envelope. + +## Configuration + +The manifest declares `compat.native_api = "1"` and Relay `>=0.7.0,<0.8`, and +the Rust SDK uses the exact published `0.7.0` crate. The manifest API value +selects Relay's released native plugin contract; the binary also requires the +v3 C host table shipped on the Relay 0.7 line. Rebuild the bundle when changing +SDK versions rather than assuming Rust dynamic-library compatibility from the +manifest value alone. + +A Relay project can configure a seeded weighted-random router as follows: + +```toml +version = 1 + +[[plugins.dynamic]] +manifest = "/opt/switchyard-relay-plugin/relay-plugin.toml" + +[plugins.dynamic.config] +version = 2 +priority = 0 +max_retries = 3 + +[plugins.dynamic.config.algorithm] +kind = "random" +seed = 42 + +[plugins.dynamic.config.default_targets] +openai_chat = "fast" + +[plugins.dynamic.config.targets.fast] +model = "provider/model" +protocol = "openai_chat" +endpoint = "/v1/chat/completions" +base_url = "https://provider.example.com" +weight = 1 +drop_caller_extra_body = true + +[plugins.dynamic.config.targets.fast.header_env] +authorization = "PROVIDER_AUTHORIZATION" +``` + +Target map keys such as `fast` are stable semantic names visible to libsy. The +target binding is authoritative for the provider model, protocol, endpoint, +base URL, weight, and environment-backed headers. Each `default_targets` key +both enables that inbound protocol and names its trusted fallback. + +`header_env` is the only custom provider-header source. It resolves values in +the plugin process at registration time so literal header values never appear +in configuration. Environment values must not appear in errors, routing marks, +spans, or debug output. The plugin does not inherit caller credentials for +managed calls. Each variable supplies the complete header value, so an +`authorization` value must include its scheme, such as `Bearer`. Literal +`headers` configuration is rejected; non-secret routing or tenancy headers must +also use `header_env`. + +Relay may intercept an OpenAI SDK call before the SDK materializes its +`extra_body` option into a provider request. Targets that reject this +caller-specific wrapper can set `drop_caller_extra_body = true`. The plugin +then drops the wrapper and its contents; it does not promote those values to +top-level provider fields. The default is `false` so lossless same-format +forwarding remains unchanged for targets that consume the extension. + +`extra_body` supplies non-secret provider defaults for a target. It is useful +for provider-specific controls such as disabling reasoning on a dedicated +judge model. Fields already present on the caller's request take precedence. +Do not put credentials in `extra_body`; use `header_env` for secrets. + +For `kind = "llm_classifier"`, the classifier target must use `openai_chat` or +`openai_responses`; libsy's judge request uses a JSON-schema response format +that cannot be represented losslessly by Anthropic Messages. Omitting `mode` +selects `capability`, preserving the original version-2 configuration shape. + +Escalation mode evaluates the weak model's completed response before returning +it or replacing it with a strong-model response: + +```toml +[plugins.dynamic.config.algorithm] +kind = "llm_classifier" +mode = "escalation" +classifier_target = "judge" +weak_target = "weak" +strong_target = "strong" +prompt = "Judge whether the weak model is stuck." +max_output_tokens = 512 + +[plugins.dynamic.config.algorithm.escalation] +confirmations = 2 +recent_turn_window = 28 +window_message_chars = 500 +``` + +`judge`, `weak`, and `strong` are keys in +`plugins.dynamic.config.targets`, configured with the same model, protocol, +URL, and `header_env` fields shown above. The judge must use `openai_chat` or +`openai_responses`; the serving targets may use any supported protocol. + +Use a dedicated, non-reasoning model for the judge when possible. Providers +that expose a reasoning switch can configure it on that target, for example: + +```toml +[plugins.dynamic.config.targets.judge] +model = "provider/non-reasoning-judge" +protocol = "openai_chat" +base_url = "https://provider.example.com" +extra_body = { think = false } +``` + +The packaged escalation rubric is intentionally detailed and can consume +roughly two thousand or more input tokens depending on the tokenizer. Every +unlatched request also pays for a complete judge call. A custom `prompt` can +reduce that cost, but should be evaluated against representative trajectories +before deployment. Reasoning models may spend `max_output_tokens` on hidden or +visible reasoning before returning the structured verdict; disable reasoning +with provider-supported `extra_body` controls or raise the cap after measuring. + +An unlatched streaming escalation request is intentionally buffered. Libsy must +read the complete weak response before asking the judge, so caller first-token +delivery waits for the weak call and judge verdict. A declined escalation is +reconstructed as a stream from the aggregate response, which drops the +provider-event preservation envelope. A confirmed escalation discards that +weak response and serves the strong target. + +The default `confirmations = 2` retains a streak per Switchyard session. Callers +must send a stable `x-switchyard-session-id` header for the streak and strong +latch to survive across turns. Without session identity each request has +isolated state and a multi-confirmation escalation cannot latch. + +A full stage router can combine tool-result signals, model-specific prompts, +handoff notes, and an optional judge for ambiguous turns: + +```toml +[plugins.dynamic.config.algorithm] +kind = "stage_router" +capable_target = "strong" +efficient_target = "weak" +picker = "efficient_first" +confidence_threshold = 0.5 +recent_turn_window = 3 +capable_system_prompt = "Diagnose before editing." +efficient_system_prompt = "Follow the settled plan." + +[plugins.dynamic.config.algorithm.handoff_notes] +escalation_note = "The previous model was stalling; pick up the diagnosis." +deescalation_note = "The task is settled; continue with the mechanical work." +only_on_wrong_signal_escalation = true + +[plugins.dynamic.config.algorithm.classifier] +target = "judge" +base_threshold = 0.5 +threshold_step = 0.1 +recent_turn_window = 3 +prompt = "Estimate whether the efficient target can finish this turn." +max_output_tokens = 512 +``` + +Stage routing reads normalized tool calls and tool results from OpenAI Chat, +OpenAI Responses, and Anthropic Messages traffic. When the signals do not cross +`confidence_threshold`, the optional classifier decides; if it is absent or +cannot decide, the configured picker's default tier serves the turn. The +classifier target has the same structured-output protocol restriction as the +standalone classifier. + +Ambiguous turns that reach the optional classifier add one judge call; +decisive tool signals do not. Decision marks report the selected model, +reasoning, and answer-call status exposed by libsy's `Decision` API. + +Version-1 service configuration, decision-only execution, and observe-only +mode are rejected. + +## Build and bundle + +The crate is a non-publishable member of the Switchyard Cargo workspace. +Operators install a binary bundle rather than a Rust crate: + +```bash +cargo build --release \ + --manifest-path crates/switchyard-nemo-relay-plugin/Cargo.toml +python3 crates/switchyard-nemo-relay-plugin/scripts/package_bundle.py \ + --library target/release/libswitchyard_nemo_relay_plugin.so \ + --output build/switchyard-nemo-relay-plugin-linux-x86_64 \ + --archive dist/switchyard-nemo-relay-plugin-0.2.0-linux-x86_64.tar.gz +``` + +On macOS the library suffix is `.dylib`; Windows builds use `.dll`. The bundle +builder creates the Relay package: the shared library, a materialized manifest +with Relay's inline SHA-256 integrity digest, the JSON schema, and the project +license files. Use `.tar.gz` archives on Linux and macOS and `.zip` on Windows. +The archive's top-level directory is always `switchyard-nemo-relay-plugin`. + +The release archive convention is +`switchyard-nemo-relay-plugin--.`. A future Actions +matrix should upload each archive under the artifact name +`switchyard-nemo-relay-plugin-`, matching Switchyard's existing +platform-qualified artifact convention. + +Install the materialized bundle with Relay's normal lifecycle commands: + +```bash +nemo-relay plugins validate /opt/switchyard-relay-plugin/relay-plugin.toml +nemo-relay plugins add /opt/switchyard-relay-plugin/relay-plugin.toml +nemo-relay plugins enable nvidia.switchyard +nemo-relay plugins inspect nvidia.switchyard +``` diff --git a/crates/switchyard-nemo-relay-plugin/config.schema.json b/crates/switchyard-nemo-relay-plugin/config.schema.json new file mode 100644 index 00000000..63db7b40 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/config.schema.json @@ -0,0 +1,217 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "title": "Switchyard NeMo Relay Plugin", + "description": "In-process Switchyard routing with Switchyard-owned provider HTTP dispatch.", + "type": "object", + "additionalProperties": false, + "required": ["version", "algorithm", "targets", "default_targets"], + "properties": { + "version": { + "const": 2, + "description": "Library-only Switchyard configuration version." + }, + "priority": { + "type": "integer", + "default": 0 + }, + "max_retries": { + "type": "integer", + "minimum": 0, + "maximum": 10, + "default": 3, + "description": "Routing retries after the initial libsy run. Every retry starts a fresh run." + }, + "algorithm": { + "description": "In-process random, capability, escalation, or stage-router configuration.", + "oneOf": [ + { + "type": "object", + "additionalProperties": false, + "required": ["kind"], + "properties": { + "kind": { "const": "random" }, + "seed": { "type": ["integer", "null"], "minimum": 0 } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": [ + "kind", + "classifier_target", + "weak_target", + "strong_target", + "base_threshold" + ], + "properties": { + "kind": { "const": "llm_classifier" }, + "mode": { "const": "capability", "default": "capability" }, + "classifier_target": { + "type": "string", + "minLength": 1, + "description": "Semantic target name for the judge. The target must use openai_chat or openai_responses because the judge requires a JSON-schema response format." + }, + "weak_target": { "type": "string", "minLength": 1 }, + "strong_target": { "type": "string", "minLength": 1 }, + "base_threshold": { "type": "number", "minimum": 0, "maximum": 1 }, + "threshold_step": { "type": "number", "minimum": 0, "default": 0 }, + "recent_turn_window": { + "type": ["integer", "null"], + "minimum": 0 + }, + "max_output_tokens": { + "type": "integer", + "minimum": 1, + "default": 4096 + }, + "prompt": { "type": "string" }, + "session_affinity": { "type": "boolean", "default": false }, + "message_hash_fallback": { "type": "boolean", "default": false } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": [ + "kind", + "mode", + "classifier_target", + "weak_target", + "strong_target", + "escalation" + ], + "properties": { + "kind": { "const": "llm_classifier" }, + "mode": { "const": "escalation" }, + "classifier_target": { + "type": "string", + "minLength": 1, + "description": "Semantic target name for the trajectory judge. The target must use openai_chat or openai_responses." + }, + "weak_target": { "type": "string", "minLength": 1 }, + "strong_target": { "type": "string", "minLength": 1 }, + "prompt": { "type": "string" }, + "max_output_tokens": { + "type": "integer", + "minimum": 1, + "default": 4096 + }, + "escalation": { + "type": "object", + "additionalProperties": false, + "properties": { + "confirmations": { "type": "integer", "minimum": 1, "default": 2 }, + "recent_turn_window": { "type": "integer", "minimum": 1, "default": 28 }, + "window_message_chars": { "type": "integer", "minimum": 50, "default": 500 } + } + } + } + }, + { + "type": "object", + "additionalProperties": false, + "required": [ + "kind", + "capable_target", + "efficient_target", + "picker", + "confidence_threshold" + ], + "properties": { + "kind": { "const": "stage_router" }, + "capable_target": { "type": "string", "minLength": 1 }, + "efficient_target": { "type": "string", "minLength": 1 }, + "picker": { "enum": ["capable_first", "efficient_first"] }, + "confidence_threshold": { "type": "number", "minimum": 0, "maximum": 1 }, + "recent_turn_window": { + "type": ["integer", "null"], + "minimum": 0, + "description": "Trailing tool results scored for the current turn. Null uses libsy's default window." + }, + "capable_system_prompt": { "type": "string" }, + "efficient_system_prompt": { "type": "string" }, + "handoff_notes": { + "type": "object", + "additionalProperties": false, + "required": ["escalation_note"], + "properties": { + "escalation_note": { "type": "string", "minLength": 1 }, + "deescalation_note": { "type": ["string", "null"], "minLength": 1 }, + "only_on_wrong_signal_escalation": { "type": "boolean", "default": true } + } + }, + "classifier": { + "type": "object", + "additionalProperties": false, + "required": ["target", "base_threshold"], + "properties": { + "target": { + "type": "string", + "minLength": 1, + "description": "Judge target used only when stage signals are ambiguous. It must use openai_chat or openai_responses." + }, + "base_threshold": { "type": "number", "minimum": 0, "maximum": 1 }, + "threshold_step": { "type": "number", "minimum": 0, "default": 0 }, + "recent_turn_window": { "type": ["integer", "null"], "minimum": 0 }, + "prompt": { "type": "string" }, + "max_output_tokens": { "type": "integer", "minimum": 1, "default": 4096 } + } + } + } + } + ] + }, + "targets": { + "type": "object", + "minProperties": 1, + "additionalProperties": { + "type": "object", + "additionalProperties": false, + "required": ["model", "protocol", "base_url"], + "properties": { + "model": { "type": "string", "minLength": 1 }, + "protocol": { + "enum": ["openai_chat", "openai_responses", "anthropic_messages"] + }, + "endpoint": { + "type": "string", + "pattern": "^$|^/", + "description": "Optional provider endpoint override. The resolved URL must end in the canonical route for the selected protocol." + }, + "base_url": { + "type": "string", + "pattern": "^https?://" + }, + "weight": { "type": "number", "minimum": 0, "default": 1 }, + "drop_caller_extra_body": { + "type": "boolean", + "default": false, + "description": "Drop an intercepted OpenAI SDK extra_body wrapper instead of forwarding it to targets that reject caller-specific extensions." + }, + "extra_body": { + "type": "object", + "default": {}, + "description": "Non-secret provider request defaults, such as judge reasoning controls. Caller-provided fields take precedence.", + "additionalProperties": true + }, + "header_env": { + "type": "object", + "description": "Sole custom provider-header source. Maps header names to environment-variable names resolved by the plugin process so literal values are never stored in configuration.", + "additionalProperties": { "type": "string", "minLength": 1 } + } + } + } + }, + "default_targets": { + "type": "object", + "description": "Maps each managed inbound protocol to its trusted fallback target.", + "minProperties": 1, + "additionalProperties": false, + "properties": { + "openai_chat": { "type": "string", "minLength": 1 }, + "openai_responses": { "type": "string", "minLength": 1 }, + "anthropic_messages": { "type": "string", "minLength": 1 } + } + } + } +} diff --git a/crates/switchyard-nemo-relay-plugin/relay-plugin.toml b/crates/switchyard-nemo-relay-plugin/relay-plugin.toml new file mode 100644 index 00000000..920decc3 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/relay-plugin.toml @@ -0,0 +1,31 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +manifest_version = 1 + +[plugin] +id = "nvidia.switchyard" +kind = "rust_dynamic" + +[compat] +relay = ">=0.7.0,<0.8" +native_api = "1" + +[defaults] +enabled = false + +[capabilities] +items = ["plugin_native", "config_schema"] + +[config_schema] +path = "config.schema.json" + +[source] +artifact = "" + +[integrity] +sha256 = "sha256:" + +[load] +library = "" +symbol = "nemo_relay_register_plugin" diff --git a/crates/switchyard-nemo-relay-plugin/scripts/package_bundle.py b/crates/switchyard-nemo-relay-plugin/scripts/package_bundle.py new file mode 100644 index 00000000..c5ea9e94 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/scripts/package_bundle.py @@ -0,0 +1,99 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Materialize the minimal Relay plugin bundle from a compiled cdylib.""" + +from __future__ import annotations + +import argparse +import hashlib +import shutil +import tarfile +import zipfile +from pathlib import Path + +CRATE_ROOT = Path(__file__).resolve().parents[1] +REPOSITORY_ROOT = CRATE_ROOT.parents[1] +PACKAGE_NAME = "switchyard-nemo-relay-plugin" + + +def digest(path: Path) -> str: + """Return the lowercase SHA-256 digest for a file.""" + value = hashlib.sha256() + with path.open("rb") as stream: + for chunk in iter(lambda: stream.read(1024 * 1024), b""): + value.update(chunk) + return value.hexdigest() + + +def archive_bundle(bundle: Path, archive: Path) -> None: + """Archive a materialized bundle under the stable package directory name.""" + archive.parent.mkdir(parents=True, exist_ok=True) + if archive.exists(): + raise ValueError(f"bundle archive already exists: {archive}") + + if archive.name.endswith(".tar.gz"): + with tarfile.open(archive, "w:gz") as stream: + stream.add(bundle, arcname=PACKAGE_NAME) + return + + if archive.suffix == ".zip": + with zipfile.ZipFile(archive, "w", compression=zipfile.ZIP_DEFLATED) as stream: + for path in sorted(bundle.rglob("*")): + if path.is_file(): + stream.write(path, Path(PACKAGE_NAME) / path.relative_to(bundle)) + return + + raise ValueError("bundle archive must end in .tar.gz or .zip") + + +def main() -> None: + """Materialize a Relay-loadable plugin bundle in an empty directory.""" + parser = argparse.ArgumentParser() + parser.add_argument("--library", required=True, type=Path) + parser.add_argument("--output", required=True, type=Path) + parser.add_argument("--archive", type=Path) + args = parser.parse_args() + + library = args.library.resolve() + if not library.is_file(): + parser.error(f"compiled plugin library does not exist: {library}") + + manifest = (CRATE_ROOT / "relay-plugin.toml").read_text(encoding="utf-8") + placeholders = ("", "") + missing = [placeholder for placeholder in placeholders if placeholder not in manifest] + if missing: + parser.error(f"plugin manifest is missing placeholders: {', '.join(missing)}") + + output = args.output.resolve() + if output.exists() and not output.is_dir(): + parser.error(f"bundle output exists and is not a directory: {output}") + if output.is_dir() and any(output.iterdir()): + parser.error(f"bundle output directory must be empty: {output}") + output.mkdir(parents=True, exist_ok=True) + + artifact = output / library.name + shutil.copy2(library, artifact) + shutil.copy2(CRATE_ROOT / "config.schema.json", output / "config.schema.json") + for filename in ("LICENSE", "NOTICE"): + shutil.copy2(REPOSITORY_ROOT / filename, output / filename) + + artifact_digest = digest(artifact) + manifest = manifest.replace("", artifact.name) + manifest = manifest.replace("", artifact_digest) + (output / "relay-plugin.toml").write_text(manifest, encoding="utf-8") + + if args.archive is None: + print(output) + return + + archive = args.archive.resolve() + try: + archive_bundle(output, archive) + except ValueError as error: + parser.error(str(error)) + print(archive) + + +if __name__ == "__main__": + main() diff --git a/crates/switchyard-nemo-relay-plugin/src/client.rs b/crates/switchyard-nemo-relay-plugin/src/client.rs new file mode 100644 index 00000000..bc197bfb --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/client.rs @@ -0,0 +1,463 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +//! Switchyard-owned HTTP clients bound to one semantic routing target. + +use std::collections::BTreeMap; + +use async_trait::async_trait; +use serde_json::Value as Json; +use switchyard_llm_client::{Backend, HttpBackendConfig, ModelConfig, TranslatingLlmClient}; +use switchyard_protocol::{ + ContentBlock, Decision, LlmClientError, Message, Request, Response, Role, RoutedLlmClient, + ToolCall, ToolResult, WireFormat, +}; +use switchyard_translation::TranslationEngine; + +use crate::translation; + +/// A provider client bound to one configured Switchyard target. +/// +/// libsy routes with a stable semantic name (for example `fast`). The provider +/// still expects its own model id (for example `meta/llama-3.1-8b-instruct`). +/// Keeping that mapping here prevents an algorithm's semantic labels from +/// leaking into provider requests. +pub(crate) struct TargetClient { + provider_model: String, + target_format: WireFormat, + drop_caller_extra_body: bool, + inner: TranslatingLlmClient, + translation: TranslationEngine, +} + +impl TargetClient { + pub(crate) fn new( + provider_model: String, + target_format: WireFormat, + dispatch_url: String, + headers: BTreeMap, + extra_body: BTreeMap, + drop_caller_extra_body: bool, + ) -> Result { + let backend_config = HttpBackendConfig { + // `dispatch_url` is already resolved by configuration. Backend URL + // joining accepts a complete canonical endpoint as well as a base + // URL/prefix. + base_url: dispatch_url, + api_key: None, + extra_headers: headers, + extra_body, + // Routing retries belong to the plugin: every retry must start a + // fresh libsy run and obtain a fresh decision. + max_retries: 0, + }; + let backend = match target_format { + WireFormat::OpenAiChat => Backend::OpenAiChat(backend_config), + WireFormat::OpenAiResponses => Backend::OpenAiResponses(backend_config), + WireFormat::AnthropicMessages => Backend::Anthropic(backend_config), + }; + let model = ModelConfig::new(provider_model.clone(), backend, None); + let inner = TranslatingLlmClient::new(&[model])?; + Ok(Self { + provider_model, + target_format, + drop_caller_extra_body, + inner, + translation: TranslationEngine::default(), + }) + } + + /// Retargets only the provider-facing transport metadata. + /// + /// Correlation and agent identity remain available to libsy, while inbound + /// HTTP headers are deliberately removed. Provider credentials come solely + /// from this target's `header_env` configuration. + fn prepare_request(&self, mut request: Request, decision: &Decision) -> Request { + if !decision.is_answer_call() { + sanitize_judge_request(&mut request); + } + if decision.reasoning() == Some("escalation classifier: efficient tier") { + // Escalation always buffers this draft before judging it. Asking the + // provider for a buffered response preserves normalized usage for ATOF; + // libsy reconstructs a caller stream when the weak draft wins. + request.llm_request.stream = false; + request.llm_request.preservation.requests.clear(); + } + let metadata = request.metadata.get_or_insert_default(); + metadata.wire_format = Some(self.target_format); + metadata.http_headers = None; + if self.drop_caller_extra_body { + request.llm_request.extensions.fields.remove("extra_body"); + for preserved in request.llm_request.preservation.requests.values_mut() { + if let Some(body) = preserved.as_object_mut() { + body.remove("extra_body"); + } + } + } + request + } +} + +#[async_trait] +impl RoutedLlmClient for TargetClient { + async fn call(&self, request: Request, decision: Decision) -> Result { + let request = self.prepare_request(request, &decision); + translation::validate_target_request( + &self.translation, + self.target_format, + &request.llm_request, + ) + .map_err(LlmClientError::RequestEncoding)?; + self.inner + .call_rewrite_model(request, Some(&self.provider_model)) + .await + } +} + +/// Maximum plain-text context retained from one native tool block in a judge request. +const MAX_JUDGE_TOOL_CONTEXT_CHARS: usize = 4_096; + +/// Keep judge requests provider-neutral. Native tool turns without their original +/// definitions are rejected by some OpenAI-compatible Bedrock gateways, while the +/// text evidence is still valuable to the classifier. +fn sanitize_judge_request(request: &mut Request) { + request.llm_request.messages = request + .llm_request + .messages + .drain(..) + .map(|message| Message { + role: if message.role == Role::Tool { + Role::User + } else { + message.role + }, + content: message + .content + .into_iter() + .map(|block| match block { + ContentBlock::ToolCall(call) => ContentBlock::Text { + text: bounded_tool_context(tool_call_text(call)), + }, + ContentBlock::ToolResult(result) => ContentBlock::Text { + text: bounded_tool_context(tool_result_text(result)), + }, + ordinary => ordinary, + }) + .collect(), + }) + .collect(); + request.llm_request.tools.clear(); + request.llm_request.tool_choice = None; + if let Some(response_format) = request.llm_request.output.response_format.as_mut() { + remove_numeric_schema_bounds(response_format); + } + request.llm_request.preservation.requests.clear(); +} + +fn tool_call_text(call: ToolCall) -> String { + format!( + "[tool call]\nid: {}\nname: {}\narguments: {}", + Json::String(call.id), + Json::String(call.name), + call.arguments + ) +} + +fn tool_result_text(result: ToolResult) -> String { + let content = result + .content + .into_iter() + .map(tool_content_text) + .collect::>() + .join("\n"); + format!( + "[tool result]\ncall_id: {}\nis_error: {}\ncontent:\n{}", + Json::String(result.tool_call_id), + result + .is_error + .map_or_else(|| "unknown".to_string(), |value| value.to_string()), + content + ) +} + +fn tool_content_text(block: ContentBlock) -> String { + match block { + ContentBlock::Text { text } + | ContentBlock::Reasoning { text, .. } + | ContentBlock::Refusal { text } => text, + ContentBlock::ToolCall(call) => tool_call_text(call), + ContentBlock::ToolResult(result) => tool_result_text(result), + ContentBlock::Image { .. } => "[image omitted]".to_string(), + ContentBlock::Audio { .. } => "[audio omitted]".to_string(), + ContentBlock::Video { .. } => "[video omitted]".to_string(), + ContentBlock::File { .. } => "[file omitted]".to_string(), + ContentBlock::Unknown { provider, .. } => { + format!("[unsupported {provider} content omitted]") + } + } +} + +fn bounded_tool_context(text: String) -> String { + const TRUNCATED: &str = "\n[truncated]"; + let keep = MAX_JUDGE_TOOL_CONTEXT_CHARS - TRUNCATED.chars().count(); + let mut chars = text.chars(); + let prefix = chars.by_ref().take(keep).collect::(); + if chars.next().is_some() { + prefix + TRUNCATED + } else { + prefix + } +} + +fn remove_numeric_schema_bounds(value: &mut Json) { + match value { + Json::Object(object) => { + for key in ["minimum", "maximum", "exclusiveMinimum", "exclusiveMaximum"] { + object.remove(key); + } + for child in object.values_mut() { + remove_numeric_schema_bounds(child); + } + } + Json::Array(values) => { + for child in values { + remove_numeric_schema_bounds(child); + } + } + _ => {} + } +} + +#[cfg(test)] +mod tests { + use super::*; + use serde_json::json; + use switchyard_protocol::{ + LlmRequest, Metadata, PreservationMetadata, ProviderExtensions, ToolChoice, ToolDefinition, + }; + + fn decision() -> Decision { + Decision::new("target", None, true) + } + + fn client(format: WireFormat) -> TargetClient { + TargetClient::new( + "provider/model".into(), + format, + match format { + WireFormat::OpenAiChat => "https://provider.example/v1/chat/completions".into(), + WireFormat::OpenAiResponses => "https://provider.example/v1/responses".into(), + WireFormat::AnthropicMessages => "https://provider.example/v1/messages".into(), + }, + BTreeMap::new(), + BTreeMap::new(), + false, + ) + .unwrap() + } + + #[test] + fn target_preparation_forces_format_and_removes_inbound_headers() { + let client = client(WireFormat::AnthropicMessages); + let request = Request { + metadata: Some(Metadata { + correlation_id: Some("request-123".into()), + wire_format: Some(WireFormat::OpenAiChat), + http_headers: Some(http::HeaderMap::from_iter([ + ( + http::HeaderName::from_static("authorization"), + http::HeaderValue::from_static("Bearer caller-secret"), + ), + ( + http::HeaderName::from_static("x-caller-only"), + http::HeaderValue::from_static("must-not-forward"), + ), + ])), + ..Metadata::default() + }), + ..Request::default() + }; + + let prepared = client.prepare_request(request, &decision()); + let metadata = prepared.metadata.unwrap(); + assert_eq!(metadata.wire_format, Some(WireFormat::AnthropicMessages)); + assert_eq!(metadata.correlation_id.as_deref(), Some("request-123")); + assert!(metadata.http_headers.is_none()); + } + + #[test] + fn missing_metadata_is_created_for_the_target_format() { + let client = client(WireFormat::OpenAiResponses); + let prepared = client.prepare_request(Request::default(), &decision()); + assert_eq!( + prepared.metadata.and_then(|metadata| metadata.wire_format), + Some(WireFormat::OpenAiResponses) + ); + } + + #[test] + fn configured_target_drops_intercepted_caller_extra_body() { + let client = TargetClient::new( + "provider/model".into(), + WireFormat::OpenAiChat, + "https://provider.example/v1/chat/completions".into(), + BTreeMap::new(), + BTreeMap::new(), + true, + ) + .unwrap(); + let request = Request { + llm_request: LlmRequest { + extensions: ProviderExtensions { + fields: serde_json::Map::from_iter([( + "extra_body".into(), + json!({"reasoning": {"effort": "medium"}}), + )]), + }, + preservation: PreservationMetadata { + requests: BTreeMap::from([( + WireFormat::OpenAiChat.into(), + json!({ + "model": "route", + "messages": [{"role": "user", "content": "hello"}], + "extra_body": { + "reasoning": {"effort": "medium"}, + "session_id": "hermes-session" + } + }), + )]), + ..PreservationMetadata::default() + }, + ..LlmRequest::default() + }, + ..Request::default() + }; + + let prepared = client.prepare_request(request, &decision()); + assert!( + !prepared + .llm_request + .extensions + .fields + .contains_key("extra_body") + ); + assert!( + prepared + .llm_request + .preservation + .requests + .values() + .all(|body| body.get("extra_body").is_none()) + ); + } + + #[test] + fn judge_preparation_sanitizes_tool_history_and_schema_dialect() { + let client = client(WireFormat::OpenAiChat); + let request = Request { + llm_request: LlmRequest { + messages: vec![ + Message::text(Role::User, "inspect the workspace"), + Message { + role: Role::Assistant, + content: vec![ContentBlock::ToolCall(ToolCall { + id: "call-1".into(), + name: "terminal".into(), + arguments: json!({"command": "pwd"}), + })], + }, + Message { + role: Role::Tool, + content: vec![ContentBlock::ToolResult(ToolResult { + tool_call_id: "call-1".into(), + content: vec![ContentBlock::Text { + text: format!("result {} TAIL", "x".repeat(5_000)), + }], + is_error: Some(false), + })], + }, + ], + tools: vec![ToolDefinition { + name: "terminal".into(), + description: None, + parameters: json!({"type": "object"}), + strict: None, + }], + tool_choice: Some(ToolChoice::Required), + output: switchyard_protocol::OutputParams { + max_output_tokens: Some(64), + response_format: Some(json!({ + "type": "json_schema", + "json_schema": { + "schema": { + "properties": { + "p_solve": { + "type": "number", + "minimum": 0.0, + "maximum": 1.0 + } + } + } + } + })), + }, + ..LlmRequest::default() + }, + ..Request::default() + }; + + let prepared = client.prepare_request( + request, + &Decision::new("judge", Some("structured judge".into()), false), + ); + + assert!(prepared.llm_request.tools.is_empty()); + assert_eq!(prepared.llm_request.tool_choice, None); + assert!( + prepared + .llm_request + .messages + .iter() + .all(|message| message.role != Role::Tool) + ); + let text = prepared + .llm_request + .messages + .iter() + .filter_map(|message| message.text_content("\n")) + .collect::>() + .join("\n"); + assert!(text.contains("[tool call]")); + assert!(text.contains("terminal")); + assert!(text.contains("[tool result]")); + assert!(text.contains("[truncated]")); + assert!(!text.contains("TAIL")); + let schema = prepared.llm_request.output.response_format.unwrap(); + let p_solve = schema + .pointer("/json_schema/schema/properties/p_solve") + .unwrap(); + assert!(p_solve.get("minimum").is_none()); + assert!(p_solve.get("maximum").is_none()); + } + + #[test] + fn escalation_candidate_is_buffered_for_usage_accounting() { + let client = client(WireFormat::OpenAiChat); + let mut request = Request::default(); + request.llm_request.stream = true; + request.llm_request.preservation.requests.insert( + WireFormat::OpenAiChat.into(), + json!({"model": "route", "stream": true}), + ); + let decision = Decision::new( + "weak", + Some("escalation classifier: efficient tier".into()), + true, + ); + + let prepared = client.prepare_request(request, &decision); + + assert!(!prepared.llm_request.stream); + assert!(prepared.llm_request.preservation.requests.is_empty()); + } +} diff --git a/crates/switchyard-nemo-relay-plugin/src/config.rs b/crates/switchyard-nemo-relay-plugin/src/config.rs new file mode 100644 index 00000000..ff366815 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/config.rs @@ -0,0 +1,587 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +use std::collections::{BTreeMap, BTreeSet}; +use std::sync::Arc; + +use http::Uri; +use http::header::{HeaderName, HeaderValue}; +use serde::Deserialize; +use serde_json::Value as Json; +use switchyard_libsy::{ + Algorithm, ClassifierContractConfig, EscalationJudgeConfig, HandoffNoteConfig, + LlmClassifierConfig, LlmFallback, LlmTarget, LlmTargetSet, LlmTaskClassifier, PickerMode, + Random, StageRouter, StageRouterConfig, TargetPrompts, TaskClassifierConfig, +}; +use switchyard_protocol::{RoutedLlmClient, WireFormat}; + +use crate::client::TargetClient; + +pub(crate) fn protocol_from_call(name: &str) -> Option { + match name { + "openai.chat_completions" => Some(WireFormat::OpenAiChat), + "openai.responses" => Some(WireFormat::OpenAiResponses), + "anthropic.messages" => Some(WireFormat::AnthropicMessages), + _ => None, + } +} + +const fn default_endpoint(protocol: WireFormat) -> &'static str { + match protocol { + WireFormat::OpenAiChat => "/v1/chat/completions", + WireFormat::OpenAiResponses => "/v1/responses", + WireFormat::AnthropicMessages => "/v1/messages", + } +} + +#[derive(Deserialize)] +#[serde(deny_unknown_fields)] +struct TargetBinding { + model: String, + protocol: WireFormat, + #[serde(default)] + endpoint: String, + base_url: String, + #[serde(default = "default_weight")] + weight: f64, + #[serde(default)] + drop_caller_extra_body: bool, + #[serde(default)] + header_env: BTreeMap, + #[serde(default)] + extra_body: BTreeMap, +} + +impl TargetBinding { + fn dispatch_url(&self) -> String { + let base = self.base_url.trim_end_matches('/'); + let default = default_endpoint(self.protocol); + if self.endpoint.is_empty() && base.ends_with(default) { + return base.to_string(); + } + let endpoint = if self.endpoint.is_empty() { + default + } else { + &self.endpoint + }; + let endpoint = if base.ends_with("/v1") && endpoint.starts_with("/v1/") { + &endpoint[3..] + } else { + endpoint + }; + format!("{base}{endpoint}") + } + + fn validate(&self, name: &str) -> Result<(), String> { + if self.model.trim().is_empty() { + return Err(format!("target {name:?} model must be non-empty")); + } + if !self.endpoint.is_empty() && !self.endpoint.starts_with('/') { + return Err(format!( + "target {name:?} endpoint must be empty or begin with '/'" + )); + } + if !self.weight.is_finite() || self.weight < 0.0 { + return Err(format!( + "target {name:?} weight must be finite and nonnegative" + )); + } + validate_dispatch_url(name, self.protocol, &self.dispatch_url())?; + self.validate_headers(name) + } + + fn validate_headers(&self, target_name: &str) -> Result<(), String> { + let mut normalized = BTreeSet::new(); + for (name, variable) in &self.header_env { + let canonical = validate_header_name(name)?; + if !normalized.insert(canonical) { + return Err(format!( + "target {target_name:?} configures header {name:?} more than once (header names are case-insensitive)" + )); + } + if variable.trim().is_empty() { + return Err(format!( + "environment variable name for target header {name:?} must not be empty" + )); + } + if variable.as_bytes().contains(&b'=') || variable.as_bytes().contains(&b'\0') { + return Err(format!( + "environment variable name for target header {name:?} must not contain '=' or NUL" + )); + } + } + Ok(()) + } + + fn prepare(&self) -> Result { + let mut headers = BTreeMap::new(); + for (name, variable) in &self.header_env { + let value = std::env::var(variable) + .map_err(|_| format!("environment variable {variable:?} is not set"))?; + validate_header(name, &value)?; + headers.insert(name.clone(), value); + } + let dispatch_url = self.dispatch_url(); + let client = TargetClient::new( + self.model.clone(), + self.protocol, + dispatch_url, + headers, + self.extra_body.clone(), + self.drop_caller_extra_body, + ) + .map_err(|error| format!("failed to create target HTTP client: {error}"))?; + Ok(PreparedTargetBinding { + client: Arc::new(client), + }) + } +} + +pub(crate) struct PreparedTargetBinding { + pub(crate) client: Arc, +} + +#[derive(Clone, Copy, Default, Deserialize)] +#[serde(rename_all = "snake_case")] +enum LlmClassifierMode { + #[default] + Capability, + Escalation, +} + +#[derive(Clone, Deserialize)] +#[serde(deny_unknown_fields)] +struct LlmClassifierAlgorithmConfig { + #[serde(default)] + mode: LlmClassifierMode, + classifier_target: String, + weak_target: String, + strong_target: String, + #[serde(default)] + base_threshold: Option, + #[serde(default)] + threshold_step: Option, + #[serde(default)] + session_affinity: Option, + #[serde(default)] + message_hash_fallback: Option, + #[serde(default)] + recent_turn_window: Option, + #[serde(default)] + prompt: Option, + #[serde(default = "default_classifier_max_output_tokens")] + max_output_tokens: u64, + #[serde(default)] + escalation: Option, +} + +impl LlmClassifierAlgorithmConfig { + fn capability_config(&self) -> Result { + if self.escalation.is_some() { + return Err( + "llm_classifier capability mode does not accept escalation settings".into(), + ); + } + let base_threshold = self + .base_threshold + .ok_or_else(|| "llm_classifier capability mode requires base_threshold".to_string())?; + let mut contract = ClassifierContractConfig::default(); + if let Some(prompt) = &self.prompt { + contract = contract.with_prompt(prompt.clone()); + } + Ok(TaskClassifierConfig { + base_threshold, + threshold_step: self.threshold_step.unwrap_or_default(), + session_affinity: self.session_affinity.unwrap_or_default(), + message_hash_fallback: self.message_hash_fallback.unwrap_or_default(), + recent_turn_window: self.recent_turn_window, + contract, + max_output_tokens: self.max_output_tokens, + }) + } + + fn escalation_config( + &self, + ) -> Result<(ClassifierContractConfig, EscalationJudgeConfig), String> { + if self.base_threshold.is_some() + || self.threshold_step.is_some() + || self.session_affinity.is_some() + || self.message_hash_fallback.is_some() + || self.recent_turn_window.is_some() + { + return Err( + "llm_classifier escalation mode does not accept capability settings".into(), + ); + } + let config = self.escalation.clone().ok_or_else(|| { + "llm_classifier escalation mode requires escalation settings".to_string() + })?; + let mut contract = ClassifierContractConfig::default(); + if let Some(prompt) = &self.prompt { + contract = contract.with_prompt(prompt.clone()); + } + Ok((contract, config)) + } +} + +#[derive(Clone, Deserialize)] +#[serde(deny_unknown_fields)] +struct StageFallbackConfig { + target: String, + base_threshold: f64, + #[serde(default)] + threshold_step: f64, + #[serde(default)] + recent_turn_window: Option, + #[serde(default)] + prompt: Option, + #[serde(default = "default_classifier_max_output_tokens")] + max_output_tokens: u64, +} + +impl StageFallbackConfig { + fn classifier_config(&self) -> TaskClassifierConfig { + let mut contract = ClassifierContractConfig::default(); + if let Some(prompt) = &self.prompt { + contract = contract.with_prompt(prompt.clone()); + } + TaskClassifierConfig { + base_threshold: self.base_threshold, + threshold_step: self.threshold_step, + session_affinity: false, + message_hash_fallback: false, + recent_turn_window: self.recent_turn_window, + contract, + max_output_tokens: self.max_output_tokens, + } + } +} + +#[derive(Deserialize)] +#[serde(tag = "kind", rename_all = "snake_case", deny_unknown_fields)] +enum AlgorithmConfig { + Random { + #[serde(default)] + seed: Option, + }, + LlmClassifier { + #[serde(flatten)] + config: LlmClassifierAlgorithmConfig, + }, + StageRouter { + capable_target: String, + efficient_target: String, + picker: PickerMode, + confidence_threshold: f64, + #[serde(default)] + recent_turn_window: Option, + #[serde(default)] + capable_system_prompt: Option, + #[serde(default)] + efficient_system_prompt: Option, + #[serde(default)] + handoff_notes: Option, + #[serde(default)] + classifier: Option, + }, +} + +#[derive(Deserialize)] +#[serde(deny_unknown_fields)] +pub(crate) struct SwitchyardConfig { + version: u32, + #[serde(default)] + pub(crate) priority: i32, + #[serde(default = "default_max_retries")] + max_retries: u32, + algorithm: AlgorithmConfig, + targets: BTreeMap, + default_targets: BTreeMap, +} + +pub(crate) struct PreparedConfig { + pub(crate) max_retries: u32, + pub(crate) algorithm: Arc, + pub(crate) targets: BTreeMap, + pub(crate) default_targets: BTreeMap, +} + +impl SwitchyardConfig { + pub(crate) fn validate(&self) -> Result<(), String> { + self.validate_structure()?; + self.build_algorithm(None).map(drop) + } + + fn validate_structure(&self) -> Result<(), String> { + if self.version != 2 { + return Err(format!( + "unsupported Switchyard config version {}; version 1 used switchyard-server; migrate to version = 2", + self.version + )); + } + if self.max_retries > 10 { + return Err("max_retries must not exceed 10".into()); + } + if self.targets.is_empty() { + return Err("targets must not be empty".into()); + } + if self.default_targets.is_empty() { + return Err("default_targets must not be empty".into()); + } + for (name, target) in &self.targets { + if name.trim().is_empty() { + return Err("target names must be non-empty".into()); + } + target.validate(name)?; + } + for (protocol, fallback) in &self.default_targets { + let target = self + .targets + .get(fallback) + .ok_or_else(|| format!("default target {fallback:?} is not configured"))?; + if target.protocol != *protocol { + return Err(format!( + "default target {fallback:?} must use protocol {}", + protocol.as_str() + )); + } + } + Ok(()) + } + + pub(crate) fn prepare(self) -> Result { + self.validate_structure()?; + let targets = self + .targets + .iter() + .map(|(name, target)| target.prepare().map(|prepared| (name.clone(), prepared))) + .collect::, _>>()?; + let algorithm = self.build_algorithm(Some(&targets))?; + Ok(PreparedConfig { + max_retries: self.max_retries, + algorithm, + targets, + default_targets: self.default_targets, + }) + } + + fn build_algorithm( + &self, + prepared: Option<&BTreeMap>, + ) -> Result, String> { + let target = |name: &str| { + if !self.targets.contains_key(name) { + return Err(format!("algorithm target {name:?} is not configured")); + } + Ok(match prepared { + Some(targets) => { + targets + .get(name) + .ok_or_else(|| format!("algorithm target {name:?} was not prepared"))?; + LlmTarget { + semantic_name: name.to_string(), + } + } + None => LlmTarget { + semantic_name: name.to_string(), + }, + }) + }; + + match &self.algorithm { + AlgorithmConfig::Random { seed } => { + let routable = self + .targets + .iter() + .filter(|(_, binding)| binding.weight > 0.0) + .collect::>(); + if routable.is_empty() { + return Err( + "random routing requires at least one positive target weight".into(), + ); + } + let targets = routable + .iter() + .map(|(name, _)| target(name)) + .collect::, _>>()?; + let weights = routable + .iter() + .map(|(_, binding)| binding.weight) + .collect::>(); + Random::new(LlmTargetSet::new(targets), Some(weights), *seed) + .map(|algorithm| Arc::new(algorithm) as Arc) + .map_err(|error| error.to_string()) + } + AlgorithmConfig::LlmClassifier { config } => { + self.validate_judge_target(&config.classifier_target)?; + let algorithm = match config.mode { + LlmClassifierMode::Capability => LlmClassifierConfig::Capability { + judge_target: target(&config.classifier_target)?, + efficient_target: target(&config.weak_target)?, + capable_target: target(&config.strong_target)?, + config: config.capability_config()?, + }, + LlmClassifierMode::Escalation => { + let (contract, escalation) = config.escalation_config()?; + LlmClassifierConfig::Escalation { + judge_target: target(&config.classifier_target)?, + efficient_target: target(&config.weak_target)?, + capable_target: target(&config.strong_target)?, + contract, + config: escalation, + max_output_tokens: config.max_output_tokens, + } + } + }; + LlmTaskClassifier::new(algorithm) + .map(|algorithm| Arc::new(algorithm) as Arc) + .map_err(|error| error.to_string()) + } + AlgorithmConfig::StageRouter { + capable_target, + efficient_target, + picker, + confidence_threshold, + recent_turn_window, + capable_system_prompt, + efficient_system_prompt, + handoff_notes, + classifier, + } => { + let capable = target(capable_target)?; + let efficient = target(efficient_target)?; + let mut config = StageRouterConfig::new(*picker, *confidence_threshold); + config.recent_window = *recent_turn_window; + config.handoff_notes = handoff_notes.clone(); + let mut prompts = TargetPrompts::default(); + if let Some(prompt) = capable_system_prompt { + prompts = prompts.with(capable_target, prompt); + } + if let Some(prompt) = efficient_system_prompt { + prompts = prompts.with(efficient_target, prompt); + } + config.tier_prompts = prompts; + if let Some(classifier) = classifier { + self.validate_judge_target(&classifier.target)?; + config.llm_fallback = Some(LlmFallback { + judge_target: target(&classifier.target)?, + config: classifier.classifier_config(), + }); + } + StageRouter::new(capable, efficient, config) + .map(|algorithm| Arc::new(algorithm) as Arc) + .map_err(|error| error.to_string()) + } + } + } + + fn validate_judge_target(&self, name: &str) -> Result<(), String> { + let binding = self + .targets + .get(name) + .ok_or_else(|| format!("algorithm target {name:?} is not configured"))?; + if binding.protocol == WireFormat::AnthropicMessages { + return Err(format!( + "classifier target {name:?} uses anthropic_messages, which cannot encode the required JSON-schema response format without loss; use an openai_chat or openai_responses target" + )); + } + Ok(()) + } +} + +fn validate_dispatch_url( + target_name: &str, + protocol: WireFormat, + dispatch_url: &str, +) -> Result<(), String> { + let uri = dispatch_url + .parse::() + .map_err(|error| format!("target {target_name:?} has invalid URL: {error}"))?; + if !matches!(uri.scheme_str(), Some("http" | "https")) { + return Err(format!( + "target {target_name:?} base_url must use http or https" + )); + } + let authority = uri + .authority() + .ok_or_else(|| format!("target {target_name:?} URL must include a host"))?; + if authority.host().is_empty() { + return Err(format!("target {target_name:?} URL must include a host")); + } + if authority.as_str().contains('@') { + return Err(format!( + "target {target_name:?} URL must not contain embedded credentials" + )); + } + if uri.query().is_some() { + return Err(format!( + "target {target_name:?} URL query parameters are not supported" + )); + } + + // The current switchyard-llm-client accepts provider base URLs and complete + // canonical endpoints. Reject a custom terminal route to avoid + // allowing Backend::url() to append another provider suffix silently. + let expected_suffix = match protocol { + WireFormat::OpenAiChat => "/chat/completions", + WireFormat::OpenAiResponses => "/responses", + WireFormat::AnthropicMessages => "/v1/messages", + }; + if !uri.path().ends_with(expected_suffix) { + return Err(format!( + "target {target_name:?} endpoint must resolve to a canonical {protocol} route ending in {expected_suffix:?}" + )); + } + Ok(()) +} + +fn validate_header_name(name: &str) -> Result { + let parsed = HeaderName::from_bytes(name.as_bytes()) + .map_err(|error| format!("invalid target header name {name:?}: {error}"))?; + let canonical = parsed.as_str().to_ascii_lowercase(); + if is_forbidden_target_header(&canonical) { + return Err(format!( + "target header {name:?} is controlled by the HTTP transport and cannot be configured" + )); + } + Ok(canonical) +} + +fn validate_header(name: &str, value: &str) -> Result { + let canonical = validate_header_name(name)?; + HeaderValue::from_str(value) + .map_err(|error| format!("invalid target header value for {name:?}: {error}"))?; + Ok(canonical) +} + +fn is_forbidden_target_header(name: &str) -> bool { + matches!( + name, + "connection" + | "content-length" + | "host" + | "keep-alive" + | "proxy-connection" + | "proxy-authenticate" + | "proxy-authorization" + | "te" + | "trailer" + | "transfer-encoding" + | "upgrade" + ) || name.starts_with("x-nemo-relay-internal-") +} + +const fn default_max_retries() -> u32 { + 3 +} + +const fn default_weight() -> f64 { + 1.0 +} + +fn default_classifier_max_output_tokens() -> u64 { + TaskClassifierConfig::default().max_output_tokens +} + +#[cfg(test)] +mod tests; diff --git a/crates/switchyard-nemo-relay-plugin/src/config/tests.rs b/crates/switchyard-nemo-relay-plugin/src/config/tests.rs new file mode 100644 index 00000000..1a1153e9 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/config/tests.rs @@ -0,0 +1,545 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +use super::*; +use serde_json::{Value, json}; + +fn binding(protocol: WireFormat, model: &str) -> TargetBinding { + TargetBinding { + model: model.into(), + protocol, + endpoint: String::new(), + base_url: "https://provider.example/v1".into(), + weight: 1.0, + drop_caller_extra_body: false, + header_env: BTreeMap::new(), + extra_body: BTreeMap::new(), + } +} + +fn config() -> SwitchyardConfig { + SwitchyardConfig { + version: 2, + priority: 0, + max_retries: 3, + algorithm: AlgorithmConfig::Random { seed: Some(42) }, + targets: BTreeMap::from([ + ( + "chat".into(), + binding(WireFormat::OpenAiChat, "provider/chat"), + ), + ( + "responses".into(), + binding(WireFormat::OpenAiResponses, "provider/responses"), + ), + ( + "anthropic".into(), + binding(WireFormat::AnthropicMessages, "provider/anthropic"), + ), + ]), + default_targets: BTreeMap::from([ + (WireFormat::OpenAiChat, "chat".into()), + (WireFormat::OpenAiResponses, "responses".into()), + (WireFormat::AnthropicMessages, "anthropic".into()), + ]), + } +} + +#[test] +fn version_two_random_configuration_builds_clients_without_a_service() { + let config = config(); + config.validate().unwrap(); + let prepared = config.prepare().unwrap(); + assert_eq!(prepared.algorithm.name(), "random"); + assert_eq!(prepared.targets.len(), 3); + assert!( + prepared + .targets + .values() + .all(|target| Arc::strong_count(&target.client) == 1) + ); +} + +#[test] +fn target_endpoints_must_be_canonical_for_the_current_http_client() { + let mut config = config(); + config.targets.get_mut("chat").unwrap().endpoint = "/custom/chat".into(); + let error = config.validate().unwrap_err(); + assert!(error.contains("ending in \"/chat/completions\"")); + + config.targets.get_mut("chat").unwrap().endpoint = "/custom/chat/completions".into(); + config.validate().unwrap(); + assert_eq!( + config.targets["chat"].dispatch_url(), + "https://provider.example/v1/custom/chat/completions" + ); +} + +#[test] +fn complete_provider_endpoint_is_not_appended_twice() { + let mut config = config(); + let chat = config.targets.get_mut("chat").unwrap(); + chat.base_url = "https://provider.example/v1/chat/completions/".into(); + assert_eq!( + chat.dispatch_url(), + "https://provider.example/v1/chat/completions" + ); + config.validate().unwrap(); +} + +#[test] +fn absolute_urls_cannot_embed_credentials_or_query_parameters() { + let mut config = config(); + config.targets.get_mut("chat").unwrap().base_url = + "https://user:password@provider.example/v1".into(); + assert!( + config + .validate() + .unwrap_err() + .contains("embedded credentials") + ); + + config.targets.get_mut("chat").unwrap().base_url = + "https://provider.example/v1?api-version=1".into(); + assert!(config.validate().unwrap_err().contains("query parameters")); +} + +#[test] +fn transport_owned_and_case_duplicate_environment_headers_are_rejected() { + let mut host_header_config = config(); + let chat = host_header_config.targets.get_mut("chat").unwrap(); + chat.header_env.insert("Host".into(), "TARGET_HOST".into()); + assert!( + host_header_config + .validate() + .unwrap_err() + .contains("HTTP transport") + ); + + let mut duplicate_config = config(); + let chat = duplicate_config.targets.get_mut("chat").unwrap(); + chat.header_env + .insert("X-Tenant".into(), "TARGET_TENANT_A".into()); + chat.header_env + .insert("x-tenant".into(), "TARGET_TENANT_B".into()); + assert!( + duplicate_config + .validate() + .unwrap_err() + .contains("more than once") + ); +} + +#[test] +fn only_canonical_relay_execution_names_resolve_protocols() { + assert_eq!( + protocol_from_call("openai.chat_completions"), + Some(WireFormat::OpenAiChat) + ); + assert_eq!( + protocol_from_call("openai.responses"), + Some(WireFormat::OpenAiResponses) + ); + assert_eq!( + protocol_from_call("anthropic.messages"), + Some(WireFormat::AnthropicMessages) + ); + assert_eq!(protocol_from_call("openai_chat"), None); +} + +#[test] +fn schema_required_contract_fields_do_not_default_during_deserialization() { + let base = json!({ + "version": 2, + "algorithm": {"kind": "random"}, + "targets": { + "chat": { + "model": "provider/chat", + "protocol": "openai_chat", + "base_url": "https://provider.example/v1" + } + }, + "default_targets": {"openai_chat": "chat"} + }); + for field in ["version", "algorithm", "default_targets"] { + let mut value = base.clone(); + value.as_object_mut().unwrap().remove(field); + let error = serde_json::from_value::(value) + .err() + .expect("required field must not default"); + assert!(error.to_string().contains(field), "field={field}: {error}"); + } +} + +#[test] +fn unknown_target_fields_are_rejected() { + let value = json!({ + "version": 2, + "algorithm": {"kind": "random"}, + "targets": { + "chat": { + "model": "provider/chat", + "protocol": "openai_chat", + "base_url": "https://provider.example/v1", + "unexpected_setting": true + } + }, + "default_targets": {"openai_chat": "chat"} + }); + let error = serde_json::from_value::(value) + .err() + .expect("unknown target field must be rejected"); + assert!(error.to_string().contains("unexpected_setting")); +} + +#[test] +fn literal_target_headers_are_rejected() { + let value = json!({ + "version": 2, + "algorithm": {"kind": "random"}, + "targets": { + "chat": { + "model": "provider/chat", + "protocol": "openai_chat", + "base_url": "https://provider.example/v1", + "headers": {"x-provider-token": "plaintext-secret"} + } + }, + "default_targets": {"openai_chat": "chat"} + }); + let error = serde_json::from_value::(value) + .err() + .expect("literal target headers must be rejected") + .to_string(); + assert!(error.contains("unknown field `headers`")); + assert!(!error.contains("plaintext-secret")); +} + +#[test] +fn unknown_algorithm_fields_are_rejected() { + let error = serde_json::from_value::(json!({ + "kind": "random", + "seed": 42, + "unexpected_setting": true + })) + .err() + .expect("unknown algorithm field must be rejected"); + assert!(error.to_string().contains("unexpected_setting")); +} + +#[test] +fn classifier_prepares_clients_for_judge_and_routed_targets() { + let mut config = config(); + config.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic", + "base_threshold": 0.5, + "recent_turn_window": 4, + "max_output_tokens": 512 + })) + .unwrap(); + config.validate().unwrap(); + let prepared = config.prepare().unwrap(); + assert_eq!(prepared.algorithm.name(), "llm_task_classifier"); + assert!( + prepared + .targets + .values() + .all(|target| Arc::strong_count(&target.client) == 1) + ); +} + +#[test] +fn target_provider_defaults_are_accepted_for_judge_controls() { + let mut config = config(); + config.targets.get_mut("chat").unwrap().extra_body = + BTreeMap::from([("think".into(), json!(false))]); + + config.validate().unwrap(); + config.prepare().unwrap(); +} + +#[test] +fn classifier_rejects_anthropic_judge_targets_before_dispatch() { + let mut config = config(); + config.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "classifier_target": "anthropic", + "weak_target": "responses", + "strong_target": "chat", + "base_threshold": 0.5 + })) + .unwrap(); + + let error = config.validate().unwrap_err(); + assert!(error.contains("classifier target \"anthropic\" uses anthropic_messages")); +} + +#[test] +fn validation_does_not_resolve_environment_backed_headers() { + let mut config = config(); + config.targets.get_mut("chat").unwrap().header_env = BTreeMap::from([( + "authorization".into(), + "SWITCHYARD_TEST_ENVIRONMENT_VARIABLE_THAT_IS_NOT_SET".into(), + )]); + + config.validate().unwrap(); + let error = config + .prepare() + .err() + .expect("preparation must resolve headers"); + assert!(error.contains("SWITCHYARD_TEST_ENVIRONMENT_VARIABLE_THAT_IS_NOT_SET")); +} + +#[test] +fn invalid_environment_variable_names_are_rejected_before_resolution() { + for variable in ["INVALID=VARIABLE", "INVALID\0VARIABLE"] { + let mut config = config(); + config.targets.get_mut("chat").unwrap().header_env = + BTreeMap::from([("authorization".into(), variable.into())]); + + let error = config.validate().unwrap_err(); + assert!(error.contains("must not contain '=' or NUL")); + } +} + +#[test] +fn static_validation_preserves_algorithm_constructor_checks() { + let mut random = config(); + for target in random.targets.values_mut() { + target.weight = 0.0; + } + assert!( + random + .validate() + .unwrap_err() + .contains("at least one positive target weight") + ); + + let mut classifier = config(); + classifier.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic", + "base_threshold": 1.1 + })) + .unwrap(); + assert!( + classifier + .validate() + .unwrap_err() + .contains("base_threshold must be between 0 and 1") + ); +} + +#[test] +fn escalation_classifier_builds_with_defaulted_policy_settings() { + let mut config = config(); + config.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "mode": "escalation", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic", + "prompt": "Judge the completed trajectory.", + "max_output_tokens": 256, + "escalation": {} + })) + .unwrap(); + + config.validate().unwrap(); + let prepared = config.prepare().unwrap(); + assert_eq!(prepared.algorithm.name(), "llm_task_classifier"); + assert!( + prepared + .targets + .values() + .all(|target| Arc::strong_count(&target.client) == 1) + ); +} + +#[test] +fn classifier_modes_reject_mixed_or_missing_settings() { + let mut capability = config(); + capability.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic", + "base_threshold": 0.5, + "escalation": {} + })) + .unwrap(); + assert!( + capability + .validate() + .unwrap_err() + .contains("capability mode does not accept escalation") + ); + + let mut escalation = config(); + escalation.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "mode": "escalation", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic", + "base_threshold": 0.5, + "escalation": {} + })) + .unwrap(); + assert!( + escalation + .validate() + .unwrap_err() + .contains("escalation mode does not accept capability") + ); + + let mut missing = config(); + missing.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "mode": "escalation", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic" + })) + .unwrap(); + assert!( + missing + .validate() + .unwrap_err() + .contains("requires escalation settings") + ); +} + +#[test] +fn escalation_settings_are_validated_by_the_libsy_constructor() { + for (settings, expected) in [ + ( + json!({"confirmations": 0}), + "confirmations must be at least 1", + ), + ( + json!({"recent_turn_window": 0}), + "recent_turn_window must be at least 1", + ), + ( + json!({"window_message_chars": 49}), + "window_message_chars must be at least 50", + ), + ] { + let mut config = config(); + config.algorithm = serde_json::from_value(json!({ + "kind": "llm_classifier", + "mode": "escalation", + "classifier_target": "chat", + "weak_target": "responses", + "strong_target": "anthropic", + "escalation": settings + })) + .unwrap(); + assert!(config.validate().unwrap_err().contains(expected)); + } +} + +#[test] +fn full_stage_router_configuration_builds_all_clients() { + let mut config = config(); + config.algorithm = serde_json::from_value(json!({ + "kind": "stage_router", + "capable_target": "anthropic", + "efficient_target": "responses", + "picker": "efficient_first", + "confidence_threshold": 0.5, + "recent_turn_window": 3, + "capable_system_prompt": "Diagnose before editing.", + "efficient_system_prompt": "Follow the settled plan.", + "handoff_notes": { + "escalation_note": "The previous model was stalling.", + "deescalation_note": "The task is settled.", + "only_on_wrong_signal_escalation": true + }, + "classifier": { + "target": "chat", + "base_threshold": 0.5, + "threshold_step": 0.1, + "recent_turn_window": 3, + "prompt": "Can the efficient tier finish this turn?", + "max_output_tokens": 256 + } + })) + .unwrap(); + + config.validate().unwrap(); + let prepared = config.prepare().unwrap(); + assert_eq!(prepared.algorithm.name(), "stage_router"); + assert!( + prepared + .targets + .values() + .all(|target| Arc::strong_count(&target.client) == 1) + ); +} + +#[test] +fn stage_router_validates_threshold_targets_and_judge_protocol() { + let stage = |classifier: Value, threshold: f64| { + serde_json::from_value(json!({ + "kind": "stage_router", + "capable_target": "anthropic", + "efficient_target": "responses", + "picker": "capable_first", + "confidence_threshold": threshold, + "classifier": classifier + })) + .unwrap() + }; + + let mut invalid_threshold = config(); + invalid_threshold.algorithm = stage(Value::Null, 1.1); + assert!( + invalid_threshold + .validate() + .unwrap_err() + .contains("confidence_threshold must be between 0 and 1") + ); + + let mut missing_target = config(); + missing_target.algorithm = serde_json::from_value(json!({ + "kind": "stage_router", + "capable_target": "missing", + "efficient_target": "responses", + "picker": "capable_first", + "confidence_threshold": 0.5 + })) + .unwrap(); + assert!( + missing_target + .validate() + .unwrap_err() + .contains("algorithm target \"missing\" is not configured") + ); + + let mut anthropic_judge = config(); + anthropic_judge.algorithm = stage(json!({"target": "anthropic", "base_threshold": 0.5}), 0.5); + assert!( + anthropic_judge + .validate() + .unwrap_err() + .contains("classifier target \"anthropic\" uses anthropic_messages") + ); +} + +#[test] +fn zero_weight_random_targets_are_fallback_only() { + let mut config = config(); + config.targets.get_mut("anthropic").unwrap().weight = 0.0; + let prepared = config.prepare().unwrap(); + + assert_eq!(Arc::strong_count(&prepared.targets["anthropic"].client), 1); + assert_eq!(Arc::strong_count(&prepared.targets["chat"].client), 1); + assert_eq!(Arc::strong_count(&prepared.targets["responses"].client), 1); +} diff --git a/crates/switchyard-nemo-relay-plugin/src/executor.rs b/crates/switchyard-nemo-relay-plugin/src/executor.rs new file mode 100644 index 00000000..206c7aa9 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/executor.rs @@ -0,0 +1,137 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +use std::future::Future; +use std::sync::{Arc, Mutex, mpsc}; +use std::thread::{self, JoinHandle}; + +use tokio::runtime::{Builder, Handle}; +use tokio::sync::oneshot; +use tokio::task::AbortHandle; + +/// Plugin-owned async executor. +/// +/// The public native-plugin SDK uses synchronous Rust callbacks and pull-based +/// iterators at the dynamic-library boundary. Switchyard performs provider I/O +/// on this dedicated runtime rather than entering Relay's Tokio runtime from a +/// separately linked cdylib. +#[derive(Clone)] +pub(crate) struct PluginExecutor { + inner: Arc, +} + +struct ExecutorInner { + handle: Handle, + shutdown: Mutex>>, + thread: Mutex>>, +} + +impl PluginExecutor { + pub(crate) fn new() -> Result { + let (ready_tx, ready_rx) = mpsc::sync_channel(1); + let thread = thread::Builder::new() + .name("switchyard-relay-http".into()) + .spawn(move || { + let runtime = match Builder::new_multi_thread() + .worker_threads(2) + .thread_name("switchyard-relay-http-worker") + .enable_all() + .build() + { + Ok(runtime) => runtime, + Err(error) => { + let _ = ready_tx.send(Err(error.to_string())); + return; + } + }; + let (shutdown_tx, shutdown_rx) = oneshot::channel(); + if ready_tx + .send(Ok((runtime.handle().clone(), shutdown_tx))) + .is_err() + { + return; + } + runtime.block_on(async { + let _ = shutdown_rx.await; + }); + }) + .map_err(|error| format!("failed to start Switchyard HTTP runtime: {error}"))?; + let (handle, shutdown) = ready_rx + .recv() + .map_err(|_| "Switchyard HTTP runtime stopped during startup".to_string())??; + Ok(Self { + inner: Arc::new(ExecutorInner { + handle, + shutdown: Mutex::new(Some(shutdown)), + thread: Mutex::new(Some(thread)), + }), + }) + } + + pub(crate) fn spawn(&self, future: F) -> AbortHandle + where + F: Future + Send + 'static, + { + self.inner.handle.spawn(future).abort_handle() + } +} + +impl Drop for ExecutorInner { + fn drop(&mut self) { + if let Some(shutdown) = self + .shutdown + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .take() + { + let _ = shutdown.send(()); + } + if let Some(thread) = self + .thread + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .take() + { + if std::thread::current() + .name() + .is_some_and(|name| name.starts_with("switchyard-relay-http-worker")) + { + // The runtime owner will join this worker after the current + // task returns. Waiting here would deadlock that shutdown. + drop(thread); + } else { + let _ = thread.join(); + } + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn executor_spawns_work() { + let executor = PluginExecutor::new().unwrap(); + let (sender, receiver) = mpsc::sync_channel(1); + executor.spawn(async move { + sender.send("done").unwrap(); + }); + assert_eq!(receiver.recv().unwrap(), "done"); + } + + #[test] + fn last_reference_can_drop_on_a_worker() { + let executor = PluginExecutor::new().unwrap(); + let worker_reference = executor.clone(); + let (sender, receiver) = mpsc::sync_channel(1); + executor.spawn(async move { + drop(worker_reference); + sender.send(()).unwrap(); + }); + drop(executor); + receiver + .recv_timeout(std::time::Duration::from_secs(5)) + .expect("dropping the executor on its own worker must not deadlock"); + } +} diff --git a/crates/switchyard-nemo-relay-plugin/src/ffi.rs b/crates/switchyard-nemo-relay-plugin/src/ffi.rs new file mode 100644 index 00000000..f5ce094c --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/ffi.rs @@ -0,0 +1,444 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +//! Small ownership wrapper around Relay's generic C host-table v3 hooks. +//! +//! The plugin manifest remains native API v1. Relay 0.7 supplies the appended +//! v3 host table to rebuilt v1 plugins, which lets this crate return `Pending` +//! and settle work from its own runtime without a targeted-continuation ABI. + +use std::ffi::c_void; +use std::ptr; +use std::sync::Arc; +use std::sync::atomic::{AtomicUsize, Ordering}; +use std::time::Duration; + +use nemo_relay_plugin::{ + Json, LlmRequest, NemoRelayNativeAsyncCompletion, NemoRelayNativeAsyncNext, + NemoRelayNativeAsyncNextStreamCb, NemoRelayNativeAsyncStream, NemoRelayNativeHostApiV1, + NemoRelayNativeHostApiV3, NemoRelayNativeScopeHandle, NemoRelayNativeString, NemoRelayStatus, +}; +use serde::Serialize; +use tokio::sync::{mpsc, oneshot}; + +const BACKPRESSURE_POLL: Duration = Duration::from_millis(1); +const CANCELLATION_POLL: Duration = Duration::from_millis(10); +const MAX_PASSTHROUGH_BUFFER_BYTES: usize = 8 * 1024 * 1024; +const MAX_PASSTHROUGH_BUFFER_EVENTS: usize = 256; + +pub(crate) struct HostString { + host: NemoRelayNativeHostApiV1, + ptr: *mut NemoRelayNativeString, +} + +// Host strings are immutable allocations owned by Relay's thread-safe host table. +unsafe impl Send for HostString {} + +impl HostString { + pub(crate) fn json( + host: &NemoRelayNativeHostApiV1, + value: &impl Serialize, + ) -> Result { + let value = serde_json::to_string(value).map_err(|error| error.to_string())?; + Self::text(host, &value) + } + + pub(crate) fn text(host: &NemoRelayNativeHostApiV1, value: &str) -> Result { + let mut ptr = ptr::null_mut(); + let status = unsafe { (host.string_new)(value.as_ptr(), value.len(), &mut ptr) }; + if status == NemoRelayStatus::Ok && !ptr.is_null() { + Ok(Self { host: *host, ptr }) + } else { + Err(format!("Relay host string allocation failed: {status:?}")) + } + } + + pub(crate) fn as_ptr(&self) -> *const NemoRelayNativeString { + self.ptr + } +} + +impl Drop for HostString { + fn drop(&mut self) { + unsafe { (self.host.string_free)(self.ptr) }; + } +} + +pub(crate) fn read_string( + host: &NemoRelayNativeHostApiV1, + value: *const NemoRelayNativeString, +) -> Result { + if value.is_null() { + return Err("Relay passed a null native string".into()); + } + let len = unsafe { (host.string_len)(value) }; + let data = unsafe { (host.string_data)(value) }; + if data.is_null() && len != 0 { + return Err("Relay passed an invalid native string".into()); + } + let bytes = if len == 0 { + &[][..] + } else { + unsafe { std::slice::from_raw_parts(data, len) } + }; + std::str::from_utf8(bytes) + .map(str::to_owned) + .map_err(|error| error.to_string()) +} + +pub(crate) fn read_json( + host: &NemoRelayNativeHostApiV1, + value: *const NemoRelayNativeString, +) -> Result { + serde_json::from_str(&read_string(host, value)?).map_err(|error| error.to_string()) +} + +/// Captures the current Relay scope as an explicit event parent. +/// +/// Async plugin work runs on a plugin-owned thread, so relying on thread-local +/// scope state would orphan its marks. The host handle is a cloned scope handle +/// and remains valid until this guard is dropped. +pub(crate) struct ParentScope { + host: NemoRelayNativeHostApiV1, + ptr: *mut NemoRelayNativeScopeHandle, +} + +unsafe impl Send for ParentScope {} +unsafe impl Sync for ParentScope {} + +impl ParentScope { + pub(crate) fn capture(host: &NemoRelayNativeHostApiV1) -> Option { + let mut ptr = ptr::null_mut(); + let status = unsafe { (host.scope_get_current)(&mut ptr) }; + (status == NemoRelayStatus::Ok && !ptr.is_null()).then_some(Self { host: *host, ptr }) + } + + pub(crate) fn emit_mark(&self, name: &str, data: &Json, metadata: &Json) -> Result<(), String> { + let name = HostString::text(&self.host, name)?; + let data = HostString::json(&self.host, data)?; + let metadata = HostString::json(&self.host, metadata)?; + let status = unsafe { + (self.host.emit_mark)( + name.as_ptr(), + self.ptr, + data.as_ptr(), + metadata.as_ptr(), + ptr::null(), + ) + }; + if status == NemoRelayStatus::Ok { + Ok(()) + } else { + Err(format!( + "Relay rejected Switchyard routing mark: {status:?}" + )) + } + } +} + +impl Drop for ParentScope { + fn drop(&mut self) { + unsafe { (self.host.scope_handle_free)(self.ptr) }; + } +} + +pub(crate) fn invoke_next_buffered( + host: &NemoRelayNativeHostApiV3, + next: usize, + completion: usize, + request: &LlmRequest, +) -> Result<(), String> { + let request = HostString::json(&host.v1, request)?; + let status = unsafe { + (host.async_next_invoke)( + next as *const NemoRelayNativeAsyncNext, + request.as_ptr(), + completion as *const NemoRelayNativeAsyncCompletion, + ) + }; + if status == NemoRelayStatus::Ok { + Ok(()) + } else { + Err(format!("Relay rejected buffered pass-through: {status:?}")) + } +} + +enum DownstreamStreamItem { + Chunk { value: Json, encoded_bytes: usize }, +} + +struct DownstreamStreamState { + host: NemoRelayNativeHostApiV1, + sender: mpsc::Sender, + terminal: Option>>, + queued_bytes: Arc, +} + +pub(crate) async fn invoke_next_stream( + host: &NemoRelayNativeHostApiV3, + next: usize, + output: usize, + request: &LlmRequest, +) -> Result<(), String> { + let request = HostString::json(&host.v1, request)?; + let (sender, mut receiver) = mpsc::channel(MAX_PASSTHROUGH_BUFFER_EVENTS); + let (terminal, terminal_result) = oneshot::channel(); + let queued_bytes = Arc::new(AtomicUsize::new(0)); + let state = Box::into_raw(Box::new(DownstreamStreamState { + host: host.v1, + sender, + terminal: Some(terminal), + queued_bytes: Arc::clone(&queued_bytes), + })) + .cast::(); + let status = unsafe { + (host.async_next_invoke_stream)( + next as *const NemoRelayNativeAsyncNext, + request.as_ptr(), + output as *const NemoRelayNativeAsyncStream, + downstream_stream_result as NemoRelayNativeAsyncNextStreamCb, + state, + ) + }; + if status != NemoRelayStatus::Ok { + unsafe { drop(Box::from_raw(state.cast::())) }; + return Err(format!("Relay rejected streaming pass-through: {status:?}")); + } + + while let Some(item) = receiver.recv().await { + match item { + DownstreamStreamItem::Chunk { + value, + encoded_bytes, + } => { + let result = push_stream(host, output, &value).await; + queued_bytes.fetch_sub(encoded_bytes, Ordering::AcqRel); + result?; + } + } + } + terminal_result + .await + .unwrap_or_else(|_| Err("Relay dropped the streaming pass-through callback".into())) +} + +unsafe extern "C" fn downstream_stream_result( + user_data: *mut c_void, + chunk_json: *const NemoRelayNativeString, + error: *const NemoRelayNativeString, + done: bool, +) -> bool { + if !error.is_null() { + let state = unsafe { Box::from_raw(user_data.cast::()) }; + let error = read_string(&state.host, error) + .unwrap_or_else(|_| "Relay streaming pass-through failed".into()); + settle_downstream_stream(state, Err(error)); + return false; + } + if done { + let state = unsafe { Box::from_raw(user_data.cast::()) }; + settle_downstream_stream(state, Ok(())); + return false; + } + + let state = unsafe { &*user_data.cast::() }; + let parsed = read_string(&state.host, chunk_json).and_then(|encoded| { + let encoded_bytes = encoded.len(); + let value = serde_json::from_str(&encoded).map_err(|error| error.to_string())?; + Ok((value, encoded_bytes)) + }); + let (value, encoded_bytes) = match parsed { + Ok(parsed) => parsed, + Err(error) => { + let state = unsafe { Box::from_raw(user_data.cast::()) }; + settle_downstream_stream(state, Err(error)); + return false; + } + }; + if !reserve_buffer_bytes(&state.queued_bytes, encoded_bytes) { + let state = unsafe { Box::from_raw(user_data.cast::()) }; + settle_downstream_stream( + state, + Err(format!( + "Relay streaming pass-through exceeded its {}-byte queued payload limit", + MAX_PASSTHROUGH_BUFFER_BYTES + )), + ); + return false; + } + + match state.sender.try_send(DownstreamStreamItem::Chunk { + value, + encoded_bytes, + }) { + Ok(()) => true, + Err(error) => { + let (item, message) = match error { + mpsc::error::TrySendError::Full(item) => ( + item, + format!( + "Relay streaming pass-through exceeded its {MAX_PASSTHROUGH_BUFFER_EVENTS}-event queue" + ), + ), + mpsc::error::TrySendError::Closed(item) => ( + item, + "Relay dropped the streaming pass-through receiver".into(), + ), + }; + let encoded_bytes = item.encoded_bytes(); + state + .queued_bytes + .fetch_sub(encoded_bytes, Ordering::AcqRel); + let state = unsafe { Box::from_raw(user_data.cast::()) }; + settle_downstream_stream(state, Err(message)); + false + } + } +} + +impl DownstreamStreamItem { + fn encoded_bytes(&self) -> usize { + match self { + Self::Chunk { encoded_bytes, .. } => *encoded_bytes, + } + } +} + +fn settle_downstream_stream(mut state: Box, result: Result<(), String>) { + if let Some(terminal) = state.terminal.take() { + let _ = terminal.send(result); + } +} + +fn reserve_buffer_bytes(queued: &AtomicUsize, encoded_bytes: usize) -> bool { + queued + .fetch_update(Ordering::AcqRel, Ordering::Acquire, |current| { + current + .checked_add(encoded_bytes) + .filter(|next| *next <= MAX_PASSTHROUGH_BUFFER_BYTES) + }) + .is_ok() +} + +pub(crate) async fn wait_for_completion_cancellation( + host: &NemoRelayNativeHostApiV3, + completion: usize, +) { + while !completion_cancelled(host, completion as *const NemoRelayNativeAsyncCompletion) { + tokio::time::sleep(CANCELLATION_POLL).await; + } +} + +pub(crate) async fn wait_for_stream_cancellation(host: &NemoRelayNativeHostApiV3, stream: usize) { + while !unsafe { (host.async_stream_is_cancelled)(stream as *const NemoRelayNativeAsyncStream) } + { + tokio::time::sleep(CANCELLATION_POLL).await; + } +} + +pub(crate) fn completion_cancelled( + host: &NemoRelayNativeHostApiV3, + completion: *const NemoRelayNativeAsyncCompletion, +) -> bool { + unsafe { (host.async_completion_is_cancelled)(completion) } +} + +pub(crate) fn resolve_completion( + host: &NemoRelayNativeHostApiV3, + completion: *const NemoRelayNativeAsyncCompletion, + value: &Json, +) -> NemoRelayStatus { + match HostString::json(&host.v1, value) { + Ok(value) => unsafe { (host.async_completion_resolve_json)(completion, value.as_ptr()) }, + Err(_) => NemoRelayStatus::Internal, + } +} + +pub(crate) fn reject_completion( + host: &NemoRelayNativeHostApiV3, + completion: *const NemoRelayNativeAsyncCompletion, + message: &str, +) -> NemoRelayStatus { + match HostString::text(&host.v1, message) { + Ok(message) => unsafe { (host.async_completion_reject)(completion, message.as_ptr()) }, + Err(_) => NemoRelayStatus::Internal, + } +} + +pub(crate) async fn push_stream( + host: &NemoRelayNativeHostApiV3, + stream: usize, + value: &Json, +) -> Result<(), String> { + let value = HostString::json(&host.v1, value)?; + loop { + if unsafe { (host.async_stream_is_cancelled)(stream as *const NemoRelayNativeAsyncStream) } + { + return Err("Relay caller cancelled the output stream".into()); + } + match unsafe { + (host.async_stream_push_json)( + stream as *const NemoRelayNativeAsyncStream, + value.as_ptr(), + ) + } { + NemoRelayStatus::Ok => return Ok(()), + // Native API v1 reports its bounded queue's WouldBlock state as Internal. + NemoRelayStatus::Internal => tokio::time::sleep(BACKPRESSURE_POLL).await, + status => return Err(format!("Relay rejected output stream event: {status:?}")), + } + } +} + +pub(crate) fn finish_stream( + host: &NemoRelayNativeHostApiV3, + stream: *const NemoRelayNativeAsyncStream, +) -> NemoRelayStatus { + unsafe { (host.async_stream_finish)(stream) } +} + +pub(crate) async fn reject_stream( + host: &NemoRelayNativeHostApiV3, + stream: usize, + message: &str, +) -> NemoRelayStatus { + let Ok(message) = HostString::text(&host.v1, message) else { + return NemoRelayStatus::Internal; + }; + loop { + if unsafe { (host.async_stream_is_cancelled)(stream as *const NemoRelayNativeAsyncStream) } + { + return NemoRelayStatus::InvalidArg; + } + match unsafe { + (host.async_stream_reject)( + stream as *const NemoRelayNativeAsyncStream, + message.as_ptr(), + ) + } { + NemoRelayStatus::Internal => tokio::time::sleep(BACKPRESSURE_POLL).await, + status => return status, + } + } +} + +pub(crate) unsafe fn release_completion( + host: &NemoRelayNativeHostApiV3, + completion: *const NemoRelayNativeAsyncCompletion, +) { + unsafe { (host.async_completion_release)(completion) }; +} + +pub(crate) unsafe fn release_next( + host: &NemoRelayNativeHostApiV3, + next: *const NemoRelayNativeAsyncNext, +) { + unsafe { (host.async_next_release)(next) }; +} + +pub(crate) unsafe fn release_stream( + host: &NemoRelayNativeHostApiV3, + stream: *const NemoRelayNativeAsyncStream, +) { + unsafe { (host.async_stream_release)(stream) }; +} diff --git a/crates/switchyard-nemo-relay-plugin/src/lib.rs b/crates/switchyard-nemo-relay-plugin/src/lib.rs new file mode 100644 index 00000000..f3a2fd7c --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/lib.rs @@ -0,0 +1,442 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +mod client; +mod config; +mod executor; +mod ffi; +mod runtime; +mod translation; + +use std::ffi::c_void; +use std::mem; +use std::panic::AssertUnwindSafe; +use std::sync::Arc; + +use futures_util::FutureExt; +use nemo_relay_plugin::{ + ConfigDiagnostic, DiagnosticLevel, Json, NEMO_RELAY_NATIVE_ABI_VERSION_ASYNC_MIDDLEWARE, + NativePlugin, NemoRelayNativeAsyncCallbackState, NemoRelayNativeAsyncCompletion, + NemoRelayNativeAsyncMiddlewareKind, NemoRelayNativeAsyncNext, NemoRelayNativeAsyncStream, + NemoRelayNativeHostApiV3, NemoRelayNativeString, NemoRelayStatus, PluginContext, +}; +use serde::Deserialize; +use serde_json::Map; + +use crate::config::SwitchyardConfig; +use crate::executor::PluginExecutor; +use crate::runtime::{RoutingMark, StreamMessage, SwitchyardRuntime}; + +#[derive(Deserialize)] +struct Invocation { + name: String, + request: nemo_relay_plugin::LlmRequest, +} + +struct CallbackState { + host: NemoRelayNativeHostApiV3, + runtime: Arc, + executor: PluginExecutor, +} + +#[derive(Default)] +struct SwitchyardPlugin; + +impl NativePlugin for SwitchyardPlugin { + fn plugin_kind(&self) -> &str { + "nvidia.switchyard" + } + + fn allows_multiple_components(&self) -> bool { + false + } + + fn validate(&self, plugin_config: &Map) -> Vec { + match parse_config(plugin_config).and_then(|config| config.validate()) { + Ok(()) => Vec::new(), + Err(message) => vec![ConfigDiagnostic { + level: DiagnosticLevel::Error, + code: "switchyard.invalid_config".into(), + component: Some("nvidia.switchyard".into()), + field: Some("config".into()), + message, + }], + } + } + + fn register( + &mut self, + plugin_config: &Map, + ctx: &mut PluginContext<'_>, + ) -> nemo_relay_plugin::Result<()> { + let host_v1 = ctx.host_api(); + if host_v1.abi_version < NEMO_RELAY_NATIVE_ABI_VERSION_ASYNC_MIDDLEWARE + || host_v1.struct_size < mem::size_of::() + { + return Err( + "Switchyard requires Relay 0.7 or newer with the generic asynchronous native host table" + .into(), + ); + } + let host = unsafe { *(host_v1 as *const _ as *const NemoRelayNativeHostApiV3) }; + let config = parse_config(plugin_config)?; + let priority = config.priority; + let state = Arc::new(CallbackState { + host, + runtime: Arc::new(SwitchyardRuntime::new(config)?), + executor: PluginExecutor::new()?, + }); + + register_buffered(ctx, priority, Arc::clone(&state))?; + register_stream(ctx, priority, state)?; + Ok(()) + } +} + +fn register_buffered( + ctx: &mut PluginContext<'_>, + priority: i32, + state: Arc, +) -> Result<(), String> { + let user_data = Box::into_raw(Box::new(state)).cast::(); + let status = unsafe { + ctx.register_async_middleware_raw( + NemoRelayNativeAsyncMiddlewareKind::LlmExecutionIntercept, + "switchyard.run_stream.buffered", + priority, + false, + buffered_callback, + user_data, + Some(free_callback_state), + ) + }; + if status == NemoRelayStatus::Ok { + Ok(()) + } else { + Err(format!( + "failed to register Switchyard buffered execution: {status:?}" + )) + } +} + +fn register_stream( + ctx: &mut PluginContext<'_>, + priority: i32, + state: Arc, +) -> Result<(), String> { + let user_data = Box::into_raw(Box::new(state)).cast::(); + let status = unsafe { + ctx.register_async_stream_middleware_raw( + "switchyard.run_stream.streaming", + priority, + stream_callback, + user_data, + Some(free_callback_state), + ) + }; + if status == NemoRelayStatus::Ok { + Ok(()) + } else { + Err(format!( + "failed to register Switchyard streaming execution: {status:?}" + )) + } +} + +fn parse_config(plugin_config: &Map) -> Result { + match plugin_config.get("version").and_then(Json::as_u64) { + Some(2) => {} + Some(version) => { + return Err(format!( + "unsupported Switchyard config version {version}; version 1 used switchyard-server; migrate to version = 2" + )); + } + None => { + return Err("invalid Switchyard configuration: version must be the integer 2".into()); + } + } + serde_json::from_value(Json::Object(plugin_config.clone())) + .map_err(|error| format!("invalid Switchyard configuration: {error}")) +} + +fn emit_marks(parent: Option<&ffi::ParentScope>, marks: Vec) { + let Some(parent) = parent else { + return; + }; + for mark in marks { + if let Err(error) = parent.emit_mark(&mark.name, &mark.data, &mark.metadata) { + eprintln!( + "Switchyard could not emit routing mark {:?}: {error}", + mark.name + ); + } + } +} + +async fn execute_managed_stream( + state: &CallbackState, + output: usize, + inbound: switchyard_protocol::WireFormat, + request: switchyard_protocol::Request, + parent: Option<&ffi::ParentScope>, +) -> Result<(), String> { + let (sender, receiver) = async_channel::bounded(32); + let runtime = Arc::clone(&state.runtime); + let execution = async move { runtime.execute_stream(inbound, request, &sender).await }; + let forwarding = async { + while let Ok(message) = receiver.recv().await { + match message { + StreamMessage::Mark(mark) => emit_marks(parent, vec![mark]), + StreamMessage::Event(event) => { + ffi::push_stream(&state.host, output, &event).await? + } + } + } + Ok(()) + }; + tokio::try_join!(execution, forwarding)?; + Ok(()) +} + +unsafe extern "C" fn free_callback_state(user_data: *mut c_void) { + if !user_data.is_null() { + unsafe { drop(Box::from_raw(user_data.cast::>())) }; + } +} + +unsafe extern "C" fn buffered_callback( + user_data: *mut c_void, + invocation_json: *const NemoRelayNativeString, + next: *const NemoRelayNativeAsyncNext, + completion: *const NemoRelayNativeAsyncCompletion, +) -> u32 { + if user_data.is_null() || completion.is_null() || next.is_null() { + return NemoRelayNativeAsyncCallbackState::Complete as u32; + } + let state = unsafe { &*user_data.cast::>() }.clone(); + let invocation = ffi::read_json(&state.host.v1, invocation_json).and_then(|value| { + serde_json::from_value::(value).map_err(|error| error.to_string()) + }); + let next = next as usize; + let completion = completion as usize; + let invocation = match invocation { + Ok(invocation) => invocation, + Err(error) => { + let _ = ffi::reject_completion( + &state.host, + completion as *const NemoRelayNativeAsyncCompletion, + &format!("invalid Relay LLM invocation: {error}"), + ); + unsafe { + ffi::release_next(&state.host, next as *const NemoRelayNativeAsyncNext); + ffi::release_completion( + &state.host, + completion as *const NemoRelayNativeAsyncCompletion, + ); + } + return NemoRelayNativeAsyncCallbackState::Pending as u32; + } + }; + let Some(inbound) = state.runtime.managed_protocol(&invocation.name) else { + if let Err(error) = + ffi::invoke_next_buffered(&state.host, next, completion, &invocation.request) + { + let _ = ffi::reject_completion( + &state.host, + completion as *const NemoRelayNativeAsyncCompletion, + &error, + ); + } + unsafe { + ffi::release_next(&state.host, next as *const NemoRelayNativeAsyncNext); + ffi::release_completion( + &state.host, + completion as *const NemoRelayNativeAsyncCompletion, + ); + } + return NemoRelayNativeAsyncCallbackState::Pending as u32; + }; + let request = match state + .runtime + .decode_request(inbound, &invocation.request, false) + { + Ok(request) => request, + Err(error) => { + let _ = ffi::reject_completion( + &state.host, + completion as *const NemoRelayNativeAsyncCompletion, + &error, + ); + unsafe { + ffi::release_next(&state.host, next as *const NemoRelayNativeAsyncNext); + ffi::release_completion( + &state.host, + completion as *const NemoRelayNativeAsyncCompletion, + ); + } + return NemoRelayNativeAsyncCallbackState::Pending as u32; + } + }; + let parent = ffi::ParentScope::capture(&state.host.v1); + let task_state = Arc::clone(&state); + state.executor.spawn(async move { + let execution = AssertUnwindSafe(async { + let mut marks = Vec::new(); + let result = task_state + .runtime + .execute_buffered(inbound, request, &mut marks) + .await; + emit_marks(parent.as_ref(), marks); + result + }) + .catch_unwind(); + tokio::pin!(execution); + let result = tokio::select! { + biased; + () = ffi::wait_for_completion_cancellation(&task_state.host, completion) => None, + result = &mut execution => Some( + result.unwrap_or_else(|_| Err("Switchyard buffered execution panicked".into())) + ), + }; + + let completion_ptr = completion as *const NemoRelayNativeAsyncCompletion; + if let Some(result) = result { + match result { + Ok(response) => { + let _ = ffi::resolve_completion(&task_state.host, completion_ptr, &response); + } + Err(error) => { + let _ = ffi::reject_completion(&task_state.host, completion_ptr, &error); + } + } + } + unsafe { + ffi::release_next(&task_state.host, next as *const NemoRelayNativeAsyncNext); + ffi::release_completion(&task_state.host, completion_ptr); + } + }); + NemoRelayNativeAsyncCallbackState::Pending as u32 +} + +unsafe extern "C" fn stream_callback( + user_data: *mut c_void, + invocation_json: *const NemoRelayNativeString, + next: *const NemoRelayNativeAsyncNext, + output: *const NemoRelayNativeAsyncStream, +) -> u32 { + if user_data.is_null() || output.is_null() || next.is_null() { + return NemoRelayNativeAsyncCallbackState::Complete as u32; + } + let state = unsafe { &*user_data.cast::>() }.clone(); + let invocation = ffi::read_json(&state.host.v1, invocation_json).and_then(|value| { + serde_json::from_value::(value).map_err(|error| error.to_string()) + }); + let managed_protocol = invocation + .as_ref() + .ok() + .and_then(|invocation| state.runtime.managed_protocol(&invocation.name)); + let parent = managed_protocol.and_then(|_| ffi::ParentScope::capture(&state.host.v1)); + let next = next as usize; + let output = output as usize; + let task_state = Arc::clone(&state); + state.executor.spawn(async move { + let execution = AssertUnwindSafe(async { + match invocation { + Ok(invocation) => { + if let Some(inbound) = managed_protocol { + match task_state + .runtime + .decode_request(inbound, &invocation.request, true) + { + Ok(request) => { + execute_managed_stream( + &task_state, + output, + inbound, + request, + parent.as_ref(), + ) + .await + } + Err(error) => Err(error), + } + } else { + ffi::invoke_next_stream(&task_state.host, next, output, &invocation.request) + .await + } + } + Err(error) => Err(format!("invalid Relay LLM stream invocation: {error}")), + } + }) + .catch_unwind(); + tokio::pin!(execution); + let result = tokio::select! { + biased; + () = ffi::wait_for_stream_cancellation(&task_state.host, output) => None, + result = &mut execution => Some( + result.unwrap_or_else(|_| Err("Switchyard streaming execution panicked".into())) + ), + }; + + match result { + Some(Ok(())) => { + let _ = ffi::finish_stream( + &task_state.host, + output as *const NemoRelayNativeAsyncStream, + ); + } + Some(Err(error)) => { + let _ = ffi::reject_stream(&task_state.host, output, &error).await; + } + None => {} + } + unsafe { + ffi::release_next(&task_state.host, next as *const NemoRelayNativeAsyncNext); + ffi::release_stream( + &task_state.host, + output as *const NemoRelayNativeAsyncStream, + ); + } + }); + NemoRelayNativeAsyncCallbackState::Pending as u32 +} + +nemo_relay_plugin::nemo_relay_plugin!(nemo_relay_register_plugin, SwitchyardPlugin::default); + +#[cfg(test)] +mod tests { + use serde_json::json; + + use super::*; + + #[test] + fn version_one_service_config_gets_a_migration_error_before_v2_deserialization() { + let value = json!({ + "version": 1, + "service_url": "http://127.0.0.1:8080", + "health_endpoint": "/healthz" + }); + let plugin_config = value.as_object().unwrap(); + + let error = parse_config(plugin_config) + .err() + .expect("version one must be rejected"); + assert!(error.contains("version 1 used switchyard-server")); + assert!(error.contains("migrate to version = 2")); + assert!(!error.contains("unknown field")); + } + + #[test] + fn version_must_be_an_integer() { + let value = json!({"version": "2"}); + let plugin_config = value.as_object().unwrap(); + + let error = parse_config(plugin_config) + .err() + .expect("non-integer versions must be rejected"); + assert_eq!( + error, + "invalid Switchyard configuration: version must be the integer 2" + ); + } +} diff --git a/crates/switchyard-nemo-relay-plugin/src/runtime.rs b/crates/switchyard-nemo-relay-plugin/src/runtime.rs new file mode 100644 index 00000000..9615e91d --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/runtime.rs @@ -0,0 +1,675 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +use std::collections::{BTreeMap, HashMap}; +use std::sync::{Arc, Mutex}; +use std::time::Duration; + +use futures_util::{StreamExt, stream}; +use nemo_relay_plugin::{Json, LlmRequest as RelayRequest}; +use serde_json::{Map, json}; +use switchyard_libsy::{Algorithm, LibsyError}; +use switchyard_llm_client::{ClientRouter, LlmCallObservation, RunObservation, RunObserver, run}; +use switchyard_protocol::{ + Decision, LlmClientError, LlmResponse, Metadata, Request, Response, WireFormat, +}; +use switchyard_translation::{TranslationEngine, encode_stream}; + +use crate::config::{PreparedTargetBinding, SwitchyardConfig, protocol_from_call}; +use crate::translation; + +const INITIAL_RETRY_BACKOFF: Duration = Duration::from_millis(250); +const MAX_RETRY_BACKOFF: Duration = Duration::from_secs(2); + +#[derive(Debug)] +pub(crate) struct RoutingMark { + pub(crate) name: String, + pub(crate) data: Json, + pub(crate) metadata: Json, +} + +#[derive(Debug)] +pub(crate) enum StreamMessage { + Mark(RoutingMark), + Event(Json), +} + +pub(crate) struct SwitchyardRuntime { + max_retries: u32, + algorithm: Arc, + targets: BTreeMap, + default_targets: BTreeMap, + translation: TranslationEngine, +} + +impl SwitchyardRuntime { + pub(crate) fn new(config: SwitchyardConfig) -> Result { + let prepared = config.prepare()?; + Ok(Self { + max_retries: prepared.max_retries, + algorithm: prepared.algorithm, + targets: prepared.targets, + default_targets: prepared.default_targets, + translation: TranslationEngine::default(), + }) + } + + pub(crate) fn managed_protocol(&self, name: &str) -> Option { + protocol_from_call(name).filter(|protocol| self.default_targets.contains_key(protocol)) + } + + pub(crate) fn decode_request( + &self, + inbound: WireFormat, + request: &RelayRequest, + streaming: bool, + ) -> Result { + let mut llm_request = translation::decode_request(&self.translation, inbound, request)?; + llm_request.stream = streaming; + let headers = string_headers(&request.headers); + let mut metadata = Metadata::from_headers(&headers); + let relay_gateway_placeholder = !headers.contains_key("x-switchyard-session-id") + && headers + .get("x-nemo-relay-source") + .and_then(|value| value.to_str().ok()) + == Some("gateway") + && metadata.session_id.as_deref() == Some("gateway-gateway"); + if relay_gateway_placeholder { + metadata.session_id = None; + } + // Keep identity/routing metadata, but target clients deliberately clear + // these caller headers before HTTP dispatch. + metadata.http_headers = Some(headers); + metadata.wire_format = Some(inbound); + Ok(Request { + llm_request, + raw_request: Some(request.content.clone()), + metadata: Some(metadata), + }) + } + + pub(crate) async fn execute_buffered( + &self, + inbound: WireFormat, + request: Request, + marks: &mut Vec, + ) -> Result { + let metadata = identity_metadata(request.metadata.as_ref()); + let max_attempts = self.max_retries + 1; + let mut attempt = 1; + loop { + self.mark( + marks, + "switchyard.routing.requested", + json!({"algorithm": self.algorithm.name(), "attempt": attempt}), + &metadata, + ); + let result = self + .drive(request.clone(), attempt, marks, &metadata) + .await + .and_then(|response| { + finalize_buffered_response(&self.translation, inbound, response) + .map_err(|source| LibsyError::client_call("return_to_agent", source)) + }); + match result { + Ok(response) => return Ok(response), + Err(failure) if libsy_error_retryable(&failure) && attempt < max_attempts => { + self.mark( + marks, + "switchyard.routing.retry", + failure_mark_data(attempt, &failure), + &metadata, + ); + sleep_before_retry(attempt).await; + attempt += 1; + } + Err(failure) => { + self.mark( + marks, + "switchyard.routing.error", + failure_mark_data(attempt, &failure), + &metadata, + ); + let response = self + .fallback_response(inbound, request, marks, &metadata) + .await?; + return finalize_buffered_response(&self.translation, inbound, response) + .map_err(|error| { + public_response_failure("trusted fallback response", &error) + }); + } + } + } + } + + pub(crate) async fn execute_stream( + &self, + inbound: WireFormat, + request: Request, + output: &async_channel::Sender, + ) -> Result<(), String> { + let metadata = identity_metadata(request.metadata.as_ref()); + let max_attempts = self.max_retries + 1; + let mut attempt = 1; + let mut marks = Vec::new(); + 'attempts: loop { + self.mark( + &mut marks, + "switchyard.routing.requested", + json!({"algorithm": self.algorithm.name(), "attempt": attempt}), + &metadata, + ); + let (response, mut fallback_used) = match self + .drive(request.clone(), attempt, &mut marks, &metadata) + .await + { + Ok(response) => (response, false), + Err(failure) if libsy_error_retryable(&failure) && attempt < max_attempts => { + self.mark( + &mut marks, + "switchyard.routing.retry", + failure_mark_data(attempt, &failure), + &metadata, + ); + send_marks(output, &mut marks).await?; + sleep_before_retry(attempt).await; + attempt += 1; + continue; + } + Err(failure) => { + self.mark( + &mut marks, + "switchyard.routing.error", + failure_mark_data(attempt, &failure), + &metadata, + ); + let fallback = self + .fallback_response(inbound, request.clone(), &mut marks, &metadata) + .await; + send_marks(output, &mut marks).await?; + (fallback?, true) + } + }; + send_marks(output, &mut marks).await?; + + let mut events = match returned_events(response, inbound).await { + Ok(events) => events, + Err(failure) + if !fallback_used + && libsy_error_retryable(&failure) + && attempt < max_attempts => + { + self.mark( + &mut marks, + "switchyard.routing.retry", + failure_mark_data(attempt, &failure), + &metadata, + ); + send_marks(output, &mut marks).await?; + sleep_before_retry(attempt).await; + attempt += 1; + continue; + } + Err(failure) if !fallback_used => { + self.mark( + &mut marks, + "switchyard.routing.error", + failure_mark_data(attempt, &failure), + &metadata, + ); + fallback_used = true; + let fallback = self + .fallback_response(inbound, request.clone(), &mut marks, &metadata) + .await; + send_marks(output, &mut marks).await?; + let fallback = fallback?; + returned_events(fallback, inbound) + .await + .map_err(|error| public_libsy_failure("trusted fallback stream", &error))? + } + Err(failure) => { + return Err(public_libsy_failure("trusted fallback stream", &failure)); + } + }; + + let mut committed = false; + while let Some(item) = events.next().await { + match item { + Ok(event) => { + send_event(output, event).await?; + committed = true; + } + Err(failure) + if !fallback_used + && !committed + && libsy_error_retryable(&failure) + && attempt < max_attempts => + { + self.mark( + &mut marks, + "switchyard.routing.retry", + failure_mark_data(attempt, &failure), + &metadata, + ); + send_marks(output, &mut marks).await?; + sleep_before_retry(attempt).await; + attempt += 1; + continue 'attempts; + } + Err(failure) if !fallback_used && !committed => { + self.mark( + &mut marks, + "switchyard.routing.error", + failure_mark_data(attempt, &failure), + &metadata, + ); + let fallback = self + .fallback_response(inbound, request.clone(), &mut marks, &metadata) + .await; + send_marks(output, &mut marks).await?; + let fallback = fallback?; + let mut fallback = + returned_events(fallback, inbound).await.map_err(|error| { + public_libsy_failure("trusted fallback stream", &error) + })?; + while let Some(item) = fallback.next().await { + let event = item.map_err(|error| { + public_libsy_failure("trusted fallback stream", &error) + })?; + send_event(output, event).await?; + } + return Ok(()); + } + Err(failure) if !committed => { + return Err(public_libsy_failure("trusted fallback stream", &failure)); + } + Err(failure) => { + self.mark( + &mut marks, + "switchyard.routing.error", + failure_mark_data(attempt, &failure), + &metadata, + ); + send_marks(output, &mut marks).await?; + return Err(public_libsy_failure( + "Switchyard stream failed after response commitment", + &failure, + )); + } + } + } + if committed { + return Ok(()); + } + return Err("Switchyard response stream produced no caller events".into()); + } + } + + async fn drive( + &self, + request: Request, + attempt: u32, + marks: &mut Vec, + mark_metadata: &Json, + ) -> Result { + let observations = Arc::new(Mutex::new(Vec::new())); + let observed_calls = observations.clone(); + let observer: RunObserver = Arc::new(move |observation| { + if let RunObservation::LlmCall(call) = observation { + observed_calls + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .push(call); + } + }); + let clients = ClientRouter::new( + self.targets + .iter() + .map(|(name, target)| (name.clone(), target.client.clone())) + .collect::>(), + ); + match run(self.algorithm.clone(), clients, request, Some(observer)).await { + Ok((decisions, response)) => { + for decision in decisions { + self.emit_decision(marks, &decision, attempt, mark_metadata); + } + self.emit_routing_llm_calls( + marks, + take_observed_calls(&observations), + attempt, + mark_metadata, + true, + ); + Ok(response) + } + Err(error) => { + self.emit_routing_llm_calls( + marks, + take_observed_calls(&observations), + attempt, + mark_metadata, + false, + ); + Err(error) + } + } + } + + async fn fallback_response( + &self, + inbound: WireFormat, + request: Request, + marks: &mut Vec, + metadata: &Json, + ) -> Result { + let target_name = self.default_target(inbound)?; + let target = self.target(target_name)?; + self.mark( + marks, + "switchyard.routing.fallback", + json!({"selected_target": target_name}), + metadata, + ); + let decision = Decision::new(target_name, Some("trusted fallback target".into()), true); + target + .client + .call(request, decision) + .await + .map_err(|error| public_client_failure("trusted fallback", &error)) + } + + fn target(&self, name: &str) -> Result<&PreparedTargetBinding, String> { + self.targets + .get(name) + .ok_or_else(|| format!("libsy selected unknown target {name:?}")) + } + + fn default_target(&self, protocol: WireFormat) -> Result<&str, String> { + self.default_targets + .get(&protocol) + .map(String::as_str) + .ok_or_else(|| format!("managed protocol {protocol} has no default target")) + } + + fn mark(&self, marks: &mut Vec, name: &str, data: Json, metadata: &Json) { + marks.push(RoutingMark { + name: name.to_string(), + data, + metadata: metadata.clone(), + }); + } + + fn emit_decision( + &self, + marks: &mut Vec, + decision: &Decision, + attempt: u32, + metadata: &Json, + ) { + self.mark( + marks, + "switchyard.routing.decision", + json!({ + "algorithm": self.algorithm.name(), + "attempt": attempt, + "selected_target": decision.selected_model_id(), + "reasoning": decision.reasoning(), + "is_answer_call": decision.is_answer_call(), + }), + metadata, + ); + } + + fn emit_routing_llm_calls( + &self, + marks: &mut Vec, + mut calls: Vec, + attempt: u32, + metadata: &Json, + successful_run: bool, + ) { + // The last successful routed call produced the response represented by Relay's + // outer LLM lifecycle event. Keep it out of these marks so consumers can add + // routing overhead without counting the serving call twice. Earlier routed calls + // are discarded candidates (for example, escalation's weak draft). + if successful_run + && let Some(position) = calls + .iter() + .rposition(|call| call.is_answer_call && call.is_success) + { + calls.remove(position); + } + + for (index, call) in calls.into_iter().enumerate() { + self.mark( + marks, + "switchyard.routing.llm_call", + json!({ + "algorithm": self.algorithm.name(), + "attempt": attempt, + "call_index": index + 1, + "selected_target": call.selected_model, + "call_role": if call.is_answer_call { "candidate" } else { "judge" }, + "outcome": if call.is_success { "ok" } else { "error" }, + "latency_ms": call.duration.as_secs_f64() * 1_000.0, + "usage": call.usage, + "contributes_to_routing_overhead": true, + }), + metadata, + ); + } + } +} + +fn take_observed_calls(observations: &Mutex>) -> Vec { + std::mem::take( + &mut *observations + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()), + ) +} + +async fn send_marks( + output: &async_channel::Sender, + marks: &mut Vec, +) -> Result<(), String> { + for mark in marks.drain(..) { + output + .send(StreamMessage::Mark(mark)) + .await + .map_err(|_| "Relay cancelled the Switchyard response stream".to_string())?; + } + Ok(()) +} + +async fn send_event( + output: &async_channel::Sender, + event: Json, +) -> Result<(), String> { + output + .send(StreamMessage::Event(event)) + .await + .map_err(|_| "Relay cancelled the Switchyard response stream".to_string()) +} + +type ReturnedEventStream = + std::pin::Pin> + Send>>; + +fn finalize_buffered_response( + translation_engine: &TranslationEngine, + inbound: WireFormat, + response: Response, +) -> Result { + let LlmResponse::Agg(response) = response.llm_response else { + return Err(LlmClientError::InvalidResponse { + source: Box::new(std::io::Error::other( + "libsy returned a stream for a buffered request", + )), + }); + }; + translation::encode_response(translation_engine, inbound, &response) + .map_err(LlmClientError::ResponseTranslation) +} + +async fn returned_events( + response: Response, + inbound: WireFormat, +) -> Result { + let chunks = match response.llm_response { + LlmResponse::Agg(response) => response.into_stream(), + LlmResponse::Stream(mut chunks) => { + let Some(first) = chunks.next().await else { + return Err(LibsyError::client_call( + "return_to_agent", + LlmClientError::InvalidResponse { + source: Box::new(std::io::Error::new( + std::io::ErrorKind::UnexpectedEof, + "provider returned an empty stream", + )), + }, + )); + }; + Box::pin(stream::once(async move { first }).chain(chunks)) + } + }; + let events = encode_stream(chunks, inbound, None) + .map_err(|error| LibsyError::client_call("return_to_agent", error))?; + Ok(Box::pin(events.map(|item| { + item.map_err(|source| match source.downcast::() { + Ok(source) => LibsyError::client_call("return_to_agent", *source), + Err(source) => LibsyError::client_call( + "return_to_agent", + LlmClientError::ResponseTranslation(source.to_string()), + ), + }) + }))) +} + +fn libsy_error_retryable(error: &LibsyError) -> bool { + let LibsyError::ClientCall { source, .. } = error else { + return false; + }; + match source { + LlmClientError::UpstreamHttp { status, .. } => { + matches!(*status, 408 | 425 | 429 | 500 | 502 | 503 | 504) + } + LlmClientError::Transport { .. } | LlmClientError::Timeout { .. } => true, + _ => false, + } +} + +fn retry_backoff(attempt: u32) -> Duration { + let exponent = attempt.saturating_sub(1).min(3); + INITIAL_RETRY_BACKOFF + .saturating_mul(1_u32 << exponent) + .min(MAX_RETRY_BACKOFF) +} + +async fn sleep_before_retry(attempt: u32) { + tokio::time::sleep(retry_backoff(attempt)).await; +} + +fn failure_mark_data(attempt: u32, failure: &LibsyError) -> Json { + let mut data = Map::from_iter([ + ("attempt".into(), Json::from(attempt)), + ( + "retryable".into(), + Json::from(libsy_error_retryable(failure)), + ), + ]); + match failure { + LibsyError::ClientCall { + source: LlmClientError::UpstreamHttp { status, .. }, + .. + } => { + data.insert("failure_kind".into(), Json::from("http")); + data.insert("http_status".into(), Json::from(*status)); + } + LibsyError::ClientCall { source, .. } => { + data.insert("failure_kind".into(), Json::from("non_http")); + data.insert( + "non_http_kind".into(), + Json::from(client_error_label(source)), + ); + } + _ => { + data.insert("failure_kind".into(), Json::from("algorithm")); + } + } + Json::Object(data) +} + +fn client_error_label(error: &LlmClientError) -> &'static str { + match error { + LlmClientError::InvalidRequest { .. } => "invalid_request", + LlmClientError::RequestTranslation(_) => "request_translation", + LlmClientError::RequestEncoding(_) => "request_encoding", + LlmClientError::ResponseTranslation(_) => "response_translation", + LlmClientError::Configuration { .. } => "configuration", + LlmClientError::Transport { .. } => "transport", + LlmClientError::Timeout { .. } => "timeout", + LlmClientError::ContextWindowExceeded { .. } => "context_window_exceeded", + LlmClientError::UpstreamHttp { .. } => "http", + LlmClientError::InvalidResponse { .. } => "invalid_response", + LlmClientError::Ffi { .. } => "ffi", + LlmClientError::General(_) => "general", + _ => "unknown", + } +} + +fn public_libsy_failure(prefix: &str, error: &LibsyError) -> String { + match error { + LibsyError::ClientCall { source, .. } => public_client_failure(prefix, source), + _ => format!("{prefix}: Switchyard algorithm failure"), + } +} + +fn public_response_failure(prefix: &str, error: &LlmClientError) -> String { + match error { + LlmClientError::InvalidResponse { .. } => format!("{prefix}: invalid response"), + LlmClientError::ResponseTranslation(_) => { + format!("{prefix}: response translation failure") + } + _ => format!("{prefix}: response finalization failure"), + } +} + +fn public_client_failure(prefix: &str, error: &LlmClientError) -> String { + match error { + LlmClientError::UpstreamHttp { status, .. } => { + format!("{prefix}: provider returned HTTP {status}") + } + _ => format!("{prefix}: provider {} failure", client_error_label(error)), + } +} + +fn string_headers(headers: &Map) -> http::HeaderMap { + let mut parsed = http::HeaderMap::with_capacity(headers.len()); + for (name, value) in headers { + let Some(value) = value.as_str() else { + continue; + }; + let (Ok(name), Ok(value)) = ( + http::HeaderName::from_bytes(name.as_bytes()), + http::HeaderValue::from_str(value), + ) else { + continue; + }; + parsed.insert(name, value); + } + parsed +} + +fn identity_metadata(metadata: Option<&Metadata>) -> Json { + json!({ + "session_id": metadata.and_then(|value| value.session_id.as_deref()), + "agent_id": metadata.and_then(|value| value.agent_id.as_deref()), + "parent_agent_id": metadata.and_then(|value| value.parent_agent_id.as_deref()), + "task_id": metadata.and_then(|value| value.task_id.as_deref()), + "turn_id": metadata.and_then(|value| value.turn_id.as_deref()), + "correlation_id": metadata.and_then(|value| value.correlation_id.as_deref()), + }) +} + +#[cfg(test)] +mod tests; diff --git a/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs b/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs new file mode 100644 index 00000000..54d38327 --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/runtime/tests.rs @@ -0,0 +1,874 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +use std::sync::atomic::{AtomicUsize, Ordering}; + +use switchyard_libsy::{ + ClassifierContractConfig, EscalationJudgeConfig, LlmClassifierConfig, LlmFallback, LlmTarget, + LlmTaskClassifier, Passthrough, PickerMode, StageRouter, StageRouterConfig, + TaskClassifierConfig, +}; +use switchyard_protocol::{LlmResponseStream, RoutedLlmClient, Usage, text_request, text_response}; + +use super::*; + +enum ScriptedBehavior { + Text(&'static str), + EmptyBuffered, + EmptyStream, + FailingStream, + TransportFailure(&'static str), +} + +struct ScriptedClient { + behavior: ScriptedBehavior, + calls: AtomicUsize, +} + +fn scripted(behavior: ScriptedBehavior) -> Arc { + Arc::new(ScriptedClient { + behavior, + calls: AtomicUsize::new(0), + }) +} + +#[async_trait::async_trait] +impl RoutedLlmClient for ScriptedClient { + async fn call( + &self, + request: Request, + _decision: Decision, + ) -> Result { + self.calls.fetch_add(1, Ordering::Relaxed); + match self.behavior { + ScriptedBehavior::Text(text) => { + let mut response = text_response(None, text); + response.usage = Usage { + input_tokens: Some(11), + output_tokens: Some(7), + total_tokens: Some(18), + ..Usage::default() + }; + Ok(Response { + llm_response: LlmResponse::Agg(response), + metadata: request.metadata, + }) + } + ScriptedBehavior::EmptyBuffered => Ok(Response { + llm_response: LlmResponse::Agg(Default::default()), + metadata: None, + }), + ScriptedBehavior::EmptyStream => Ok(Response { + llm_response: LlmResponse::Stream(Box::pin(stream::empty())), + metadata: None, + }), + ScriptedBehavior::FailingStream => { + let stream: LlmResponseStream = Box::pin(stream::once(async { + Err(LlmClientError::Transport { + source: Box::new(std::io::Error::other("fallback stream failed")), + }) + })); + Ok(Response { + llm_response: LlmResponse::Stream(stream), + metadata: None, + }) + } + ScriptedBehavior::TransportFailure(message) => Err(LlmClientError::Transport { + source: Box::new(std::io::Error::other(message)), + }), + } + } +} + +fn fixed_target(name: &str) -> LlmTarget { + LlmTarget { + semantic_name: name.to_string(), + } +} + +fn runtime_with_algorithm( + algorithm: Arc, + fallback: Arc, + protocol: WireFormat, +) -> SwitchyardRuntime { + runtime_with_algorithm_clients(algorithm, fallback, protocol, Vec::new()) +} + +fn runtime_with_algorithm_clients( + algorithm: Arc, + fallback: Arc, + protocol: WireFormat, + clients: Vec<(&str, Arc)>, +) -> SwitchyardRuntime { + let mut targets = BTreeMap::from([( + "fallback".into(), + PreparedTargetBinding { + client: fallback as Arc, + }, + )]); + for (name, client) in clients { + targets.insert( + name.to_string(), + PreparedTargetBinding { + client: client as Arc, + }, + ); + } + SwitchyardRuntime { + max_retries: 0, + algorithm, + targets, + default_targets: BTreeMap::from([(protocol, "fallback".into())]), + translation: TranslationEngine::default(), + } +} + +fn request_with_session(protocol: WireFormat, session: Option<&str>) -> Request { + Request { + llm_request: text_request(Some("auto".into()), "fix the build"), + raw_request: None, + metadata: Some(Metadata { + wire_format: Some(protocol), + session_id: session.map(str::to_string), + ..Metadata::default() + }), + } +} + +fn stage_signal_relay_request(protocol: WireFormat) -> RelayRequest { + let content = match protocol { + WireFormat::OpenAiChat => json!({ + "model": "auto", + "messages": [ + {"role": "user", "content": "fix the build"}, + { + "role": "assistant", + "content": null, + "tool_calls": [{ + "id": "call-1", + "type": "function", + "function": { + "name": "bash", + "arguments": "{\"cmd\":\"cargo test\"}" + } + }] + }, + { + "role": "tool", + "tool_call_id": "call-1", + "content": "fatal runtime error: out of memory" + } + ] + }), + WireFormat::OpenAiResponses => json!({ + "model": "auto", + "input": [ + {"type": "message", "role": "user", "content": "fix the build"}, + { + "type": "function_call", + "call_id": "call-1", + "name": "bash", + "arguments": "{\"cmd\":\"cargo test\"}" + }, + { + "type": "function_call_output", + "call_id": "call-1", + "output": "fatal runtime error: out of memory" + } + ] + }), + WireFormat::AnthropicMessages => json!({ + "model": "auto", + "max_tokens": 128, + "messages": [ + {"role": "user", "content": "fix the build"}, + { + "role": "assistant", + "content": [{ + "type": "tool_use", + "id": "call-1", + "name": "bash", + "input": {"cmd": "cargo test"} + }] + }, + { + "role": "user", + "content": [{ + "type": "tool_result", + "tool_use_id": "call-1", + "content": "fatal runtime error: out of memory", + "is_error": true + }] + } + ] + }), + }; + + RelayRequest { + headers: Map::from_iter([( + "x-switchyard-session-id".into(), + json!(format!("stage-{}", protocol.as_str())), + )]), + content, + } +} + +#[test] +fn relay_gateway_placeholder_session_is_not_retained() { + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let runtime = runtime_with_algorithm( + Arc::new(Passthrough::new(LlmTarget { + semantic_name: "selected".into(), + })), + fallback, + WireFormat::OpenAiChat, + ); + let request = RelayRequest { + headers: Map::from_iter([ + ("x-nemo-relay-source".into(), json!("gateway")), + ("x-nemo-relay-session-id".into(), json!("gateway-gateway")), + ("x-dynamo-session-id".into(), json!("gateway-gateway")), + ]), + content: json!({ + "model": "router", + "messages": [{"role": "user", "content": "hello"}] + }), + }; + + let decoded = runtime + .decode_request(WireFormat::OpenAiChat, &request, false) + .unwrap(); + + assert_eq!(decoded.metadata.unwrap().session_id, None); +} + +#[test] +fn explicit_switchyard_session_overrides_relay_gateway_placeholder() { + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let runtime = runtime_with_algorithm( + Arc::new(Passthrough::new(LlmTarget { + semantic_name: "selected".into(), + })), + fallback, + WireFormat::OpenAiChat, + ); + let request = RelayRequest { + headers: Map::from_iter([ + ("x-switchyard-session-id".into(), json!("caller-session")), + ("x-nemo-relay-source".into(), json!("gateway")), + ("x-nemo-relay-session-id".into(), json!("gateway-gateway")), + ]), + content: json!({ + "model": "router", + "messages": [{"role": "user", "content": "hello"}] + }), + }; + + let decoded = runtime + .decode_request(WireFormat::OpenAiChat, &request, false) + .unwrap(); + + assert_eq!( + decoded.metadata.unwrap().session_id.as_deref(), + Some("caller-session") + ); +} + +#[tokio::test] +async fn buffered_finalization_failure_uses_fallback_once() { + let selected = scripted(ScriptedBehavior::EmptyStream); + let fallback = scripted(ScriptedBehavior::EmptyBuffered); + let runtime = SwitchyardRuntime { + max_retries: 1, + algorithm: Arc::new(Passthrough::new(LlmTarget { + semantic_name: "selected".into(), + })), + targets: BTreeMap::from([ + ( + "selected".into(), + PreparedTargetBinding { + client: selected.clone(), + }, + ), + ( + "fallback".into(), + PreparedTargetBinding { + client: fallback.clone(), + }, + ), + ]), + default_targets: BTreeMap::from([(WireFormat::OpenAiChat, "fallback".into())]), + translation: TranslationEngine::default(), + }; + let mut marks = Vec::new(); + + let response = runtime + .execute_buffered(WireFormat::OpenAiChat, Request::default(), &mut marks) + .await + .expect("the buffered fallback response should be encoded"); + + assert!(response.is_object()); + assert_eq!(selected.calls.load(Ordering::Relaxed), 1); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 1); + assert!( + !marks + .iter() + .any(|mark| mark.name == "switchyard.routing.retry") + ); + let error = marks + .iter() + .find(|mark| mark.name == "switchyard.routing.error") + .expect("finalization failure should emit an error mark"); + assert_eq!(error.data["retryable"], false); + assert_eq!(error.data["non_http_kind"], "invalid_response"); + assert_eq!( + marks + .iter() + .filter(|mark| mark.name == "switchyard.routing.fallback") + .count(), + 1 + ); +} + +#[tokio::test] +async fn returned_events_replays_preserved_openai_chat_without_duplicate_terminal() { + let content = json!({ + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "model": "gpt-4o", + "system_fingerprint": "fp_provider_specific", + "choices": [{ + "index": 0, + "delta": {"content": "Hi"}, + "finish_reason": null + }] + }); + let terminal = json!({ + "id": "chatcmpl-test", + "object": "chat.completion.chunk", + "model": "gpt-4o", + "choices": [{ + "index": 0, + "delta": {}, + "finish_reason": "stop" + }] + }); + let body = format!("data: {content}\n\ndata: {terminal}\n\ndata: [DONE]\n\n").into_bytes(); + let stream = switchyard_translation::decode_stream( + stream::once(async move { Ok::<_, LlmClientError>(body) }), + WireFormat::OpenAiChat, + ) + .expect("provider SSE should decode"); + let response = Response { + llm_response: LlmResponse::Stream(stream), + metadata: None, + }; + + let replayed = returned_events(response, WireFormat::OpenAiChat) + .await + .expect("return stream should encode") + .collect::>() + .await + .into_iter() + .collect::, _>>() + .expect("return stream should not fail"); + + assert_eq!(replayed, vec![content, terminal]); +} + +#[tokio::test] +async fn invalid_selected_stream_does_not_invoke_failing_fallback_twice() { + let selected = scripted(ScriptedBehavior::EmptyStream); + let fallback = scripted(ScriptedBehavior::FailingStream); + let runtime = SwitchyardRuntime { + max_retries: 0, + algorithm: Arc::new(Passthrough::new(LlmTarget { + semantic_name: "selected".into(), + })), + targets: BTreeMap::from([ + ( + "selected".into(), + PreparedTargetBinding { + client: selected.clone(), + }, + ), + ( + "fallback".into(), + PreparedTargetBinding { + client: fallback.clone(), + }, + ), + ]), + default_targets: BTreeMap::from([(WireFormat::OpenAiChat, "fallback".into())]), + translation: TranslationEngine::default(), + }; + let (output, _messages) = async_channel::bounded(32); + + let error = runtime + .execute_stream(WireFormat::OpenAiChat, Request::default(), &output) + .await + .expect_err("the failing fallback stream must fail the request"); + + assert_eq!(error, "trusted fallback stream: provider transport failure"); + assert_eq!(selected.calls.load(Ordering::Relaxed), 1); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 1); +} + +#[tokio::test] +async fn failing_fallback_call_flushes_error_and_fallback_marks() { + let selected = scripted(ScriptedBehavior::EmptyStream); + let fallback = scripted(ScriptedBehavior::TransportFailure("fallback call failed")); + let runtime = SwitchyardRuntime { + max_retries: 0, + algorithm: Arc::new(Passthrough::new(LlmTarget { + semantic_name: "selected".into(), + })), + targets: BTreeMap::from([ + ( + "selected".into(), + PreparedTargetBinding { + client: selected.clone(), + }, + ), + ( + "fallback".into(), + PreparedTargetBinding { + client: fallback.clone(), + }, + ), + ]), + default_targets: BTreeMap::from([(WireFormat::OpenAiChat, "fallback".into())]), + translation: TranslationEngine::default(), + }; + let (output, messages) = async_channel::bounded(32); + + let error = runtime + .execute_stream(WireFormat::OpenAiChat, Request::default(), &output) + .await + .expect_err("the failing fallback call must fail the request"); + + assert_eq!(error, "trusted fallback: provider transport failure"); + assert_eq!(selected.calls.load(Ordering::Relaxed), 1); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 1); + let mut terminal_marks = Vec::new(); + while let Ok(message) = messages.try_recv() { + if let StreamMessage::Mark(mark) = message + && matches!( + mark.name.as_str(), + "switchyard.routing.error" | "switchyard.routing.fallback" + ) + { + terminal_marks.push(mark.name); + } + } + assert_eq!( + terminal_marks, + ["switchyard.routing.error", "switchyard.routing.fallback"] + ); +} + +#[test] +fn retry_backoff_increases_exponentially_and_is_capped() { + assert_eq!(retry_backoff(1), Duration::from_millis(250)); + assert_eq!(retry_backoff(2), Duration::from_millis(500)); + assert_eq!(retry_backoff(3), Duration::from_secs(1)); + assert_eq!(retry_backoff(4), Duration::from_secs(2)); + assert_eq!(retry_backoff(u32::MAX), Duration::from_secs(2)); +} + +#[tokio::test] +async fn capability_classifier_emits_judge_usage_without_serving_usage() { + let weak = scripted(ScriptedBehavior::Text("weak answer")); + let strong = scripted(ScriptedBehavior::Text("strong answer")); + let judge = scripted(ScriptedBehavior::Text( + r#"{"crux":"bounded","primary_rule":"SUP-1","capability_boundary":"supported","p_solve":0.9}"#, + )); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let algorithm = LlmTaskClassifier::new(LlmClassifierConfig::Capability { + judge_target: fixed_target("judge"), + efficient_target: fixed_target("weak"), + capable_target: fixed_target("strong"), + config: TaskClassifierConfig { + base_threshold: 0.5, + ..TaskClassifierConfig::default() + }, + }) + .unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback, + WireFormat::OpenAiChat, + vec![ + ("weak", weak.clone()), + ("strong", strong.clone()), + ("judge", judge.clone()), + ], + ); + let mut marks = Vec::new(); + + runtime + .execute_buffered( + WireFormat::OpenAiChat, + request_with_session(WireFormat::OpenAiChat, Some("capability")), + &mut marks, + ) + .await + .unwrap(); + + assert_eq!(judge.calls.load(Ordering::Relaxed), 1); + assert_eq!(weak.calls.load(Ordering::Relaxed), 1); + assert_eq!(strong.calls.load(Ordering::Relaxed), 0); + let routing_calls = marks + .iter() + .filter(|mark| mark.name == "switchyard.routing.llm_call") + .collect::>(); + assert_eq!(routing_calls.len(), 1); + assert_eq!(routing_calls[0].data["selected_target"], "judge"); + assert_eq!(routing_calls[0].data["usage"]["total_tokens"], 18); +} + +#[tokio::test] +async fn escalation_buffers_weak_stream_then_latches_the_session_to_strong() { + let weak = scripted(ScriptedBehavior::Text("weak draft")); + let strong = scripted(ScriptedBehavior::Text("strong answer")); + let judge = scripted(ScriptedBehavior::Text( + r#"{"escalate":true,"reason":"stuck"}"#, + )); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let algorithm = LlmTaskClassifier::new(LlmClassifierConfig::Escalation { + judge_target: fixed_target("judge"), + efficient_target: fixed_target("weak"), + capable_target: fixed_target("strong"), + contract: ClassifierContractConfig::default(), + config: EscalationJudgeConfig { + confirmations: 1, + ..EscalationJudgeConfig::default() + }, + max_output_tokens: 128, + }) + .unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback.clone(), + WireFormat::OpenAiChat, + vec![ + ("weak", weak.clone()), + ("strong", strong.clone()), + ("judge", judge.clone()), + ], + ); + + let mut first = request_with_session(WireFormat::OpenAiChat, Some("session-1")); + first.llm_request.stream = true; + let (output, messages) = async_channel::bounded(32); + runtime + .execute_stream(WireFormat::OpenAiChat, first, &output) + .await + .unwrap(); + let mut streamed = Vec::new(); + let mut routing_calls = Vec::new(); + while let Ok(message) = messages.try_recv() { + match message { + StreamMessage::Event(event) => streamed.push(event), + StreamMessage::Mark(mark) if mark.name == "switchyard.routing.llm_call" => { + routing_calls.push(mark.data) + } + StreamMessage::Mark(_) => {} + } + } + assert!(!streamed.is_empty()); + assert!( + streamed + .iter() + .any(|event| event.to_string().contains("strong answer")) + ); + assert_eq!(routing_calls.len(), 2); + assert_eq!(routing_calls[0]["selected_target"], "weak"); + assert_eq!(routing_calls[0]["call_role"], "candidate"); + assert_eq!(routing_calls[0]["usage"]["total_tokens"], 18); + assert_eq!(routing_calls[1]["selected_target"], "judge"); + assert_eq!(routing_calls[1]["call_role"], "judge"); + assert_eq!(routing_calls[1]["usage"]["total_tokens"], 18); + assert!( + routing_calls + .iter() + .all(|call| call["selected_target"] != "strong") + ); + + let mut marks = Vec::new(); + let response = runtime + .execute_buffered( + WireFormat::OpenAiChat, + request_with_session(WireFormat::OpenAiChat, Some("session-1")), + &mut marks, + ) + .await + .unwrap(); + assert!(response.to_string().contains("strong answer")); + assert_eq!(weak.calls.load(Ordering::Relaxed), 1); + assert_eq!(judge.calls.load(Ordering::Relaxed), 1); + assert_eq!(strong.calls.load(Ordering::Relaxed), 2); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 0); + assert!( + !marks + .iter() + .any(|mark| mark.name == "switchyard.routing.llm_call") + ); + assert!(marks.iter().any(|mark| { + mark.name == "switchyard.routing.decision" + && mark.data["selected_target"] == "strong" + && mark.data["is_answer_call"] == true + && mark.data["reasoning"].is_string() + && mark.metadata["session_id"] == "session-1" + })); +} + +#[tokio::test] +async fn escalation_judge_failure_falls_open_to_the_buffered_weak_response() { + let weak = scripted(ScriptedBehavior::Text("weak answer")); + let strong = scripted(ScriptedBehavior::Text("strong answer")); + let judge = scripted(ScriptedBehavior::TransportFailure("scripted failure")); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let algorithm = LlmTaskClassifier::new(LlmClassifierConfig::Escalation { + judge_target: fixed_target("judge"), + efficient_target: fixed_target("weak"), + capable_target: fixed_target("strong"), + contract: ClassifierContractConfig::default(), + config: EscalationJudgeConfig::default(), + max_output_tokens: 128, + }) + .unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback.clone(), + WireFormat::OpenAiChat, + vec![ + ("weak", weak.clone()), + ("strong", strong.clone()), + ("judge", judge.clone()), + ], + ); + let mut marks = Vec::new(); + + let response = runtime + .execute_buffered( + WireFormat::OpenAiChat, + request_with_session(WireFormat::OpenAiChat, Some("session-1")), + &mut marks, + ) + .await + .unwrap(); + + assert!(response.to_string().contains("weak answer")); + assert_eq!(weak.calls.load(Ordering::Relaxed), 1); + assert_eq!(judge.calls.load(Ordering::Relaxed), 1); + assert_eq!(strong.calls.load(Ordering::Relaxed), 0); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 0); + let routing_calls = marks + .iter() + .filter(|mark| mark.name == "switchyard.routing.llm_call") + .collect::>(); + assert_eq!(routing_calls.len(), 1); + assert_eq!(routing_calls[0].data["selected_target"], "judge"); + assert_eq!(routing_calls[0].data["call_role"], "judge"); + assert_eq!(routing_calls[0].data["outcome"], "error"); + assert!(routing_calls[0].data["usage"].is_null()); +} + +#[tokio::test] +async fn escalation_without_session_identity_cannot_accumulate_confirmations() { + let weak = scripted(ScriptedBehavior::Text("weak answer")); + let strong = scripted(ScriptedBehavior::Text("strong answer")); + let judge = scripted(ScriptedBehavior::Text( + r#"{"escalate":true,"reason":"stuck"}"#, + )); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let algorithm = LlmTaskClassifier::new(LlmClassifierConfig::Escalation { + judge_target: fixed_target("judge"), + efficient_target: fixed_target("weak"), + capable_target: fixed_target("strong"), + contract: ClassifierContractConfig::default(), + config: EscalationJudgeConfig { + confirmations: 2, + ..EscalationJudgeConfig::default() + }, + max_output_tokens: 128, + }) + .unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback.clone(), + WireFormat::OpenAiChat, + vec![ + ("weak", weak.clone()), + ("strong", strong.clone()), + ("judge", judge.clone()), + ], + ); + + for _ in 0..2 { + let mut marks = Vec::new(); + let response = runtime + .execute_buffered( + WireFormat::OpenAiChat, + request_with_session(WireFormat::OpenAiChat, None), + &mut marks, + ) + .await + .unwrap(); + assert!(response.to_string().contains("weak answer")); + } + assert_eq!(weak.calls.load(Ordering::Relaxed), 2); + assert_eq!(judge.calls.load(Ordering::Relaxed), 2); + assert_eq!(strong.calls.load(Ordering::Relaxed), 0); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 0); +} + +#[tokio::test] +async fn stage_router_uses_tool_signals_for_every_managed_protocol() { + for protocol in [ + WireFormat::OpenAiChat, + WireFormat::OpenAiResponses, + WireFormat::AnthropicMessages, + ] { + let capable = scripted(ScriptedBehavior::Text("capable answer")); + let efficient = scripted(ScriptedBehavior::Text("efficient answer")); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let algorithm = StageRouter::new( + fixed_target("strong"), + fixed_target("weak"), + StageRouterConfig::new(PickerMode::EfficientFirst, 0.5), + ) + .unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback.clone(), + protocol, + vec![("strong", capable.clone()), ("weak", efficient.clone())], + ); + let mut marks = Vec::new(); + let relay_request = stage_signal_relay_request(protocol); + let request = runtime + .decode_request(protocol, &relay_request, false) + .unwrap(); + + let response = runtime + .execute_buffered(protocol, request, &mut marks) + .await + .unwrap(); + + assert!(response.to_string().contains("capable answer")); + assert_eq!(capable.calls.load(Ordering::Relaxed), 1); + assert_eq!(efficient.calls.load(Ordering::Relaxed), 0); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 0); + assert!(marks.iter().any(|mark| { + mark.name == "switchyard.routing.decision" + && mark.data["algorithm"] == "stage_router" + && mark.data["attempt"] == 1 + && mark.data["selected_target"] == "strong" + && mark.data["reasoning"].is_string() + && mark.data["is_answer_call"] == true + && mark.metadata["session_id"] == format!("stage-{}", protocol.as_str()) + })); + } +} + +#[tokio::test] +async fn stage_router_falls_open_to_each_picker_default_without_tool_history() { + for (picker, expected) in [ + (PickerMode::CapableFirst, "strong"), + (PickerMode::EfficientFirst, "weak"), + ] { + let capable = scripted(ScriptedBehavior::Text("strong")); + let efficient = scripted(ScriptedBehavior::Text("weak")); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let algorithm = StageRouter::new( + fixed_target("strong"), + fixed_target("weak"), + StageRouterConfig::new(picker, 0.5), + ) + .unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback, + WireFormat::OpenAiChat, + vec![("strong", capable), ("weak", efficient)], + ); + let mut marks = Vec::new(); + + runtime + .execute_buffered( + WireFormat::OpenAiChat, + request_with_session(WireFormat::OpenAiChat, None), + &mut marks, + ) + .await + .unwrap(); + + assert!(marks.iter().any(|mark| { + mark.name == "switchyard.routing.decision" + && mark.data["selected_target"] == expected + && mark.data["reasoning"].is_string() + && mark.data["is_answer_call"] == true + })); + } +} + +#[tokio::test] +async fn stage_router_classifier_resolves_an_ambiguous_turn() { + let capable = scripted(ScriptedBehavior::Text("strong")); + let efficient = scripted(ScriptedBehavior::Text("weak")); + let judge = scripted(ScriptedBehavior::Text( + r#"{"crux":"bounded","primary_rule":"SUP-1","capability_boundary":"supported","p_solve":0.9}"#, + )); + let fallback = scripted(ScriptedBehavior::Text("fallback")); + let mut config = StageRouterConfig::new(PickerMode::CapableFirst, 0.5); + config.llm_fallback = Some(LlmFallback { + judge_target: fixed_target("judge"), + config: TaskClassifierConfig { + base_threshold: 0.5, + ..TaskClassifierConfig::default() + }, + }); + let algorithm = StageRouter::new(fixed_target("strong"), fixed_target("weak"), config).unwrap(); + let runtime = runtime_with_algorithm_clients( + Arc::new(algorithm), + fallback.clone(), + WireFormat::OpenAiChat, + vec![ + ("strong", capable.clone()), + ("weak", efficient.clone()), + ("judge", judge.clone()), + ], + ); + let mut marks = Vec::new(); + + runtime + .execute_buffered( + WireFormat::OpenAiChat, + request_with_session(WireFormat::OpenAiChat, Some("stage-classifier")), + &mut marks, + ) + .await + .unwrap(); + + assert_eq!(judge.calls.load(Ordering::Relaxed), 1); + assert_eq!(efficient.calls.load(Ordering::Relaxed), 1); + assert_eq!(capable.calls.load(Ordering::Relaxed), 0); + assert_eq!(fallback.calls.load(Ordering::Relaxed), 0); + let routing_calls = marks + .iter() + .filter(|mark| mark.name == "switchyard.routing.llm_call") + .collect::>(); + assert_eq!(routing_calls.len(), 1); + assert_eq!(routing_calls[0].data["selected_target"], "judge"); + assert_eq!(routing_calls[0].data["call_role"], "judge"); + assert_eq!(routing_calls[0].data["outcome"], "ok"); + assert_eq!(routing_calls[0].data["usage"]["total_tokens"], 18); + assert!(marks.iter().any(|mark| { + mark.name == "switchyard.routing.decision" + && mark.data["selected_target"] == "weak" + && mark.data["reasoning"].is_string() + && mark.data["is_answer_call"] == true + })); +} diff --git a/crates/switchyard-nemo-relay-plugin/src/translation.rs b/crates/switchyard-nemo-relay-plugin/src/translation.rs new file mode 100644 index 00000000..6f6f965f --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/src/translation.rs @@ -0,0 +1,116 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +use nemo_relay_plugin::LlmRequest as RelayRequest; +use serde_json::Value as Json; +use switchyard_protocol::{AggLlmResponse, LlmRequest, WireFormat}; +use switchyard_translation::{ + DeterministicIdPolicy, DiagnosticSeverity, LossyConversionPolicy, PreservationPolicy, + TargetCapabilities, TranslationDiagnostic, TranslationEngine, TranslationPolicy, + UnknownFieldPolicy, +}; + +pub(crate) fn decode_request( + engine: &TranslationEngine, + protocol: WireFormat, + request: &RelayRequest, +) -> Result { + let output = engine + .decode_request(protocol, &request.content, &policy()) + .map_err(error)?; + safe(&output.diagnostics)?; + Ok(output.request) +} + +pub(crate) fn validate_target_request( + engine: &TranslationEngine, + protocol: WireFormat, + request: &LlmRequest, +) -> Result<(), String> { + let output = engine + .encode_request(protocol, request, &request_policy(protocol)) + .map_err(error)?; + safe(&output.diagnostics) +} + +pub(crate) fn encode_response( + engine: &TranslationEngine, + protocol: WireFormat, + response: &AggLlmResponse, +) -> Result { + let output = engine + .encode_response(protocol, response, &policy()) + .map_err(error)?; + safe(&output.diagnostics)?; + Ok(output.body) +} + +fn policy() -> TranslationPolicy { + TranslationPolicy { + unknown_field_policy: UnknownFieldPolicy::Preserve, + lossy_conversion_policy: LossyConversionPolicy::Reject, + deterministic_ids: DeterministicIdPolicy::GenerateStable { + prefix: "relay".into(), + }, + preservation: PreservationPolicy::InMemory, + target_capabilities: TargetCapabilities::default(), + } +} + +fn request_policy(protocol: WireFormat) -> TranslationPolicy { + let mut policy = policy(); + if protocol == WireFormat::AnthropicMessages { + policy + .target_capabilities + .supports_json_schema_response_format = Some(false); + } + policy +} + +fn safe(diagnostics: &[TranslationDiagnostic]) -> Result<(), String> { + let unsafe_diagnostics = diagnostics + .iter() + .filter(|diagnostic| diagnostic.severity != DiagnosticSeverity::Info) + .collect::>(); + if unsafe_diagnostics.is_empty() { + Ok(()) + } else { + Err(format!( + "Switchyard translation was not lossless: {unsafe_diagnostics:?}" + )) + } +} + +fn error(error: switchyard_translation::TranslationError) -> String { + format!("Switchyard translation failed: {error}") +} + +#[cfg(test)] +mod tests { + use serde_json::{Map, json}; + + use super::*; + + #[test] + fn same_protocol_request_preserves_unknown_fields() { + let request = RelayRequest { + headers: Map::new(), + content: json!({ + "model": "route", + "messages": [{"role": "user", "content": "hello"}], + "provider_extension": {"exact": true} + }), + }; + let engine = TranslationEngine::default(); + let decoded = decode_request(&engine, WireFormat::OpenAiChat, &request).unwrap(); + validate_target_request(&engine, WireFormat::OpenAiChat, &decoded).unwrap(); + assert_eq!( + decoded + .preservation + .requests + .get(&WireFormat::OpenAiChat.into()) + .and_then(|body| body.get("provider_extension")), + Some(&json!({"exact": true})) + ); + } +} diff --git a/crates/switchyard-nemo-relay-plugin/tests/test_package_bundle.py b/crates/switchyard-nemo-relay-plugin/tests/test_package_bundle.py new file mode 100644 index 00000000..296cbc4b --- /dev/null +++ b/crates/switchyard-nemo-relay-plugin/tests/test_package_bundle.py @@ -0,0 +1,76 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +"""Tests for the native Relay plugin bundle packager.""" + +from __future__ import annotations + +import hashlib +import subprocess +import sys +import tarfile +import tempfile +import unittest +import zipfile +from pathlib import Path + +CRATE_ROOT = Path(__file__).resolve().parents[1] +PACKAGER = CRATE_ROOT / "scripts" / "package_bundle.py" +PACKAGE_NAME = "switchyard-nemo-relay-plugin" + + +class PackageBundleTest(unittest.TestCase): + """Verify materialized and archived plugin bundle contents.""" + + def test_materializes_and_archives_supported_formats(self) -> None: + for archive_suffix in (".tar.gz", ".zip"): + with self.subTest(archive_suffix=archive_suffix), tempfile.TemporaryDirectory() as tmp: + root = Path(tmp) + library = root / "libswitchyard_nemo_relay_plugin.so" + library.write_bytes(b"compiled plugin") + output = root / "bundle" + archive = root / f"{PACKAGE_NAME}-0.2.0-linux-x86_64{archive_suffix}" + + subprocess.run( + [ + sys.executable, + str(PACKAGER), + "--library", + str(library), + "--output", + str(output), + "--archive", + str(archive), + ], + check=True, + capture_output=True, + text=True, + ) + + expected = { + "LICENSE", + "NOTICE", + "config.schema.json", + library.name, + "relay-plugin.toml", + } + self.assertEqual({path.name for path in output.iterdir()}, expected) + manifest = (output / "relay-plugin.toml").read_text(encoding="utf-8") + self.assertIn(f'artifact = "{library.name}"', manifest) + self.assertIn(hashlib.sha256(library.read_bytes()).hexdigest(), manifest) + self.assertNotIn("", manifest) + self.assertNotIn("", manifest) + self.assertEqual(self.archive_members(archive), {f"{PACKAGE_NAME}/{name}" for name in expected}) + + @staticmethod + def archive_members(archive: Path) -> set[str]: + """Return regular-file paths from a supported bundle archive.""" + if archive.name.endswith(".tar.gz"): + with tarfile.open(archive) as stream: + return {member.name for member in stream.getmembers() if member.isfile()} + with zipfile.ZipFile(archive) as stream: + return {member.filename for member in stream.infolist() if not member.is_dir()} + + +if __name__ == "__main__": + unittest.main() diff --git a/docs/index.md b/docs/index.md index 1dc5056c..93814e9c 100644 --- a/docs/index.md +++ b/docs/index.md @@ -10,6 +10,7 @@ It supports OpenAI Chat Completions, OpenAI Responses, and Anthropic Messages. | Run Claude Code, Codex, or OpenClaw through Switchyard | Launcher Path | [Install and launch an agent](getting_started.md#launcher-path) | | Run Switchyard as a standalone proxy for API clients | Server Path | [Build and run the Rust server](getting_started.md#server-path) | | Add Switchyard routing to a Rust application | Library Path | [`switchyard-libsy`](../crates/libsy/README.md) | +| Add Switchyard routing to NeMo Relay | Native Plugin Path | [`switchyard-nemo-relay-plugin`](../crates/switchyard-nemo-relay-plugin/README.md) | The Launcher Path installs the `switchyard` CLI and hosts the native Rust server through its packaged PyO3 binding. The Server Path builds and runs the @@ -30,3 +31,4 @@ standalone `switchyard-server` binary. - [`switchyard-libsy`](reference/rust_api.md#switchyard-libsy): embeddable routing algorithms - [`switchyard-protocol`](reference/rust_api.md#switchyard-protocol): provider-neutral API types - [`switchyard-translation`](../crates/switchyard-translation/README.md): protocol translation +- [`switchyard-nemo-relay-plugin`](../crates/switchyard-nemo-relay-plugin/README.md): native NeMo Relay integration