From 508184a5358f61ca90560e634a82dbef05ca2ec2 Mon Sep 17 00:00:00 2001 From: Elyas Mehtabuddin Date: Thu, 1 Oct 2026 13:28:16 -0700 Subject: [PATCH 1/5] feat(server): let a forwarding route mix OpenAI and Anthropic clients on one host Signed-off-by: Elyas Mehtabuddin --- crates/libsy-llm-client/README.md | 5 +- crates/libsy-llm-client/src/backend.rs | 3 +- crates/libsy-llm-client/src/client.rs | 5 +- crates/switchyard-nemo-relay-plugin/README.md | 4 +- crates/switchyard-runner/src/config.rs | 109 ++++++++++- crates/switchyard-server/README.md | 12 +- crates/switchyard-server/tests/server.rs | 170 ++++++++++++++++++ docs/getting_started.md | 12 +- docs/integrations/nemo_relay.md | 4 +- docs/reference/toml_schema.md | 19 +- 10 files changed, 317 insertions(+), 26 deletions(-) diff --git a/crates/libsy-llm-client/README.md b/crates/libsy-llm-client/README.md index 4a880d3a0..8450c0b84 100644 --- a/crates/libsy-llm-client/README.md +++ b/crates/libsy-llm-client/README.md @@ -242,7 +242,10 @@ fn build_multi_format_client( normalize `authorization`, `chatgpt-account-id`, and `x-openai-fedramp`. Anthropic backends forward `authorization` or `x-api-key`; they also keep `oauth-*` values from `anthropic-beta` and remove other caller-supplied beta values. - All backends reachable through a forwarding route must use the same provider. + All backends reachable through a forwarding route must use one credential family + (OpenAI or Anthropic) unless they all use the same scheme, host, and port. Such a + route serves Chat Completions and Responses callers and forwards their bearer + token to every backend. Headers owned by other providers are preserved as application headers. - Per-backend custom headers go in `HttpBackendConfig::extra_headers`. Set credentials with `api_key`. OpenAI backends reject `Authorization`; Anthropic backends reject `x-api-key` diff --git a/crates/libsy-llm-client/src/backend.rs b/crates/libsy-llm-client/src/backend.rs index bbe23b0d1..fa6f4064e 100644 --- a/crates/libsy-llm-client/src/backend.rs +++ b/crates/libsy-llm-client/src/backend.rs @@ -52,7 +52,8 @@ pub struct HttpBackendConfig { pub api_key: Option, /// Whether this backend forwards the caller's provider credential and application headers. /// - /// All backends reachable through a forwarding route must use the same provider. + /// All backends reachable through a forwarding route must use one credential family + /// (OpenAI or Anthropic) unless they all use the same scheme, host, and port. pub forward_auth: bool, /// Custom headers added to every outbound call to this backend. /// diff --git a/crates/libsy-llm-client/src/client.rs b/crates/libsy-llm-client/src/client.rs index 6ba089d9c..ab11c7259 100644 --- a/crates/libsy-llm-client/src/client.rs +++ b/crates/libsy-llm-client/src/client.rs @@ -640,8 +640,9 @@ impl TranslatingLlmClient { /// `http_headers` are carried through as the request's /// [`Metadata::http_headers`]. Backends with `forward_auth` disabled forward only /// allowed metadata headers; `forward_auth` backends forward all application - /// headers. All backends reachable through a forwarding route must use the same - /// provider. Transport headers are always rebuilt. Pass `None` to forward nothing. + /// headers. All backends reachable through a forwarding route must use one credential + /// family unless they all use the same scheme, host, and port. Transport headers are + /// always rebuilt. Pass `None` to forward nothing. pub async fn call_rewrite_model_raw( &self, raw_http_request: Value, diff --git a/crates/switchyard-nemo-relay-plugin/README.md b/crates/switchyard-nemo-relay-plugin/README.md index 0bf2103ad..0b156070c 100644 --- a/crates/switchyard-nemo-relay-plugin/README.md +++ b/crates/switchyard-nemo-relay-plugin/README.md @@ -216,7 +216,9 @@ and configure `api_key_env` on each authenticated client. The two options cannot be enabled together. If each caller must use its own provider credential, use standalone `switchyard-server`. Standalone forwarding requires the caller and target to use the same credential family: OpenAI-compatible (Chat Completions -and Responses) or Anthropic (Messages). +and Responses) or Anthropic (Messages). A route may mix both families when all +of its forwarding clients use the same scheme, host, and port; such a route +serves Chat Completions and Responses callers. Support for provider-specific fields depends on the source and target formats. Test any fields that your application relies on before deploying a translated diff --git a/crates/switchyard-runner/src/config.rs b/crates/switchyard-runner/src/config.rs index 1c976a47b..fcfdcfb90 100644 --- a/crates/switchyard-runner/src/config.rs +++ b/crates/switchyard-runner/src/config.rs @@ -363,6 +363,8 @@ impl DeploymentConfig { let mut by_model = HashMap::new(); let mut targets_by_model: HashMap<&str, (&str, &TargetConfig)> = HashMap::new(); let mut caller_auth = None; + let mut mixes_families = false; + let mut forwarding_origins = BTreeSet::new(); for name in route.callable_target_names() { let target = self.targets.get(name).ok_or_else(|| { RunnerError::configuration(format!("route references unknown target {name}")) @@ -386,16 +388,26 @@ impl DeploymentConfig { })?; if client_config.forward_auth { let target_auth = client_config.format.caller_auth_kind(); - if caller_auth.is_some_and(|kind| kind != target_auth) { - return Err(RunnerError::configuration(format!( - "route {route_name} cannot forward both Anthropic and OpenAI caller credentials" - ))); - } + mixes_families |= caller_auth.is_some_and(|kind| kind != target_auth); caller_auth = Some(target_auth); + forwarding_origins.insert(client_config.base_url.0.origin().ascii_serialization()); } let client: Arc = client.clone(); by_model.insert(target.id.clone(), client); } + // A route may mix credential families only when all of its forwarding clients use the + // same scheme, host, and port, so the caller's credential reaches only that host. Such a + // route serves Chat Completions and Responses callers because Anthropic clients forward + // the caller's `authorization` header unchanged. + if mixes_families { + if forwarding_origins.len() > 1 { + let origins = Vec::from_iter(forwarding_origins).join(", "); + return Err(RunnerError::configuration(format!( + "route {route_name} cannot forward both Anthropic and OpenAI caller credentials to different hosts ({origins}); point all of its forwarding clients at one host" + ))); + } + caller_auth = Some(CallerAuthKind::OpenAi); + } let completion_targets = route .routing_target_names() .into_iter() @@ -1905,6 +1917,93 @@ confidence_threshold = 0.5 } } + // Builds a classifier route whose judge uses a Responses client and whose tiers use a + // Messages client, plus a passthrough route that uses only the Messages client. + fn mixed_forwarding_config(messages_url: &str) -> String { + format!( + r#" +schema_version = 1 + +[llm_clients.responses] +format = "openai_responses" +base_url = "https://gateway.example.test/v1" +forward_auth = true + +[llm_clients.messages] +format = "anthropic_messages" +base_url = "{messages_url}" +forward_auth = true + +[targets.judge] +id = "judge/model" +llm_client = "responses" + +[targets.capable] +id = "capable/model" +llm_client = "messages" + +[targets.efficient] +id = "efficient/model" +llm_client = "messages" + +[routes.hub] +id = "switchyard/hub" +type = "llm_classifier" +classifier_target = "judge" +strong_target = "capable" +weak_target = "efficient" +base_threshold = 0.5 + +[routes.claude] +id = "switchyard/claude" +type = "passthrough" +target = "capable" +"# + ) + } + + #[test] + fn forwarding_route_mixes_credential_families_only_on_one_host() -> RunnerResult<()> { + // The URL path does not count, so both base URLs name one host. + for messages_url in [ + "https://gateway.example.test", + "https://gateway.example.test/v1", + ] { + let runner = runner_from_toml(&mixed_forwarding_config(messages_url))?; + let caller_auth = |id| runner.route(id).and_then(Route::caller_auth); + assert_eq!(caller_auth("switchyard/hub"), Some(CallerAuthKind::OpenAi)); + assert_eq!( + caller_auth("switchyard/claude"), + Some(CallerAuthKind::Anthropic) + ); + } + + // A different host, port, or scheme is a different origin. + for (messages_url, origins) in [ + ( + "https://api.anthropic.test", + "https://api.anthropic.test, https://gateway.example.test", + ), + ( + "https://gateway.example.test:8443", + "https://gateway.example.test, https://gateway.example.test:8443", + ), + ( + "http://gateway.example.test", + "http://gateway.example.test, https://gateway.example.test", + ), + ] { + let error = error_message(&mixed_forwarding_config(messages_url)); + assert!( + error.contains(&format!( + "route hub cannot forward both Anthropic and OpenAI caller credentials to different hosts ({origins}); point all of its forwarding clients at one host" + )), + "{error}" + ); + } + Ok(()) + } + const ADVISOR_CONFIG: &str = r#" schema_version = 1 diff --git a/crates/switchyard-server/README.md b/crates/switchyard-server/README.md index 8bb64b91f..cccb3ccb3 100644 --- a/crates/switchyard-server/README.md +++ b/crates/switchyard-server/README.md @@ -91,10 +91,14 @@ A client can set `forward_auth = true` instead of `api_key_env` to send the caller's credential to the configured upstream. OpenAI clients forward `authorization`, `chatgpt-account-id`, and `x-openai-fedramp`. Anthropic clients forward `authorization` or `x-api-key`. Enable this only when every forwarding -client's `base_url` should receive the caller's login. All backends reachable -through the route must use the same provider. Other application headers are -preserved and may contain provider-specific credentials. A forwarding route -must be called through the matching provider API. +client's `base_url` should receive the caller's login. Other application headers +are preserved and may contain provider-specific credentials. A route's +forwarding clients must use one credential family: all OpenAI formats or all +`anthropic_messages`. A route may mix the two families only when all of its +forwarding clients use the same scheme, host, and port. Such a route serves +Chat Completions and Responses callers and forwards the caller's bearer token to +every client. +The server returns 400 to a caller whose API the route does not serve. Target-level `extra_body` values are shallow-merged into the upstream request when the request does not already contain that key. Target-level `system_prompt` values are prepended when that target serves a completion. diff --git a/crates/switchyard-server/tests/server.rs b/crates/switchyard-server/tests/server.rs index 42e6a6911..63a012886 100644 --- a/crates/switchyard-server/tests/server.rs +++ b/crates/switchyard-server/tests/server.rs @@ -3283,6 +3283,176 @@ target = "openai" Ok(()) } +/// Serves `/v1/responses` and `/v1/messages` for a stub gateway on one host. For each call, it +/// records every value of the `authorization`, `x-api-key`, `chatgpt-account-id`, and +/// `anthropic-version` headers, so a header sent twice shows up as two values. Every +/// `/v1/responses` call returns a judge verdict that picks the efficient tier. +async fn upstream_gateway_records_auth( + State(calls): State>>>, + uri: Uri, + headers: HeaderMap, + Json(body): Json, +) -> HttpResponse { + let header = |name: &str| { + headers + .get_all(name) + .iter() + .filter_map(|value| value.to_str().ok()) + .collect::>() + }; + calls.lock().await.push(json!({ + "path": uri.path(), + "authorization": header("authorization"), + "x_api_key": header("x-api-key"), + "chatgpt_account_id": header("chatgpt-account-id"), + "anthropic_version": header("anthropic-version"), + })); + let model = body["model"].as_str().unwrap_or_default(); + if uri.path() == "/v1/responses" { + let verdict = json!({ + "crux": "bounded task", "primary_rule": "SUP-1", + "capability_boundary": "supported", "p_solve": 0.9, + }); + return Json(responses_body("resp_judge", model, &verdict.to_string())).into_response(); + } + Json(json!({ + "id": "msg_gateway", "type": "message", "role": "assistant", "model": model, + "content": [{"type": "text", "text": "ok"}], + "stop_reason": "end_turn", "stop_sequence": null, + "usage": {"input_tokens": 1, "output_tokens": 1} + })) + .into_response() +} + +#[tokio::test] +async fn route_on_one_host_forwards_the_bearer_token_to_responses_and_messages() -> TestResult { + let calls = Arc::new(Mutex::new(Vec::new())); + let gateway = Router::new() + .route("/v1/responses", post(upstream_gateway_records_auth)) + .route("/v1/messages", post(upstream_gateway_records_auth)) + .with_state(Arc::clone(&calls)); + let listener = TcpListener::bind("127.0.0.1:0").await?; + let base_url = format!("http://{}/v1", listener.local_addr()?); + tokio::spawn(async move { axum::serve(listener, gateway).await }); + let state = load_test_config(&format!( + r#" +schema_version = 1 + +[llm_clients.gateway_responses] +format = "openai_responses" +base_url = "{base_url}" +forward_auth = true +max_retries = 0 + +[llm_clients.gateway_messages] +format = "anthropic_messages" +base_url = "{base_url}" +forward_auth = true +max_retries = 0 + +[targets] +judge = {{ id = "model/judge", llm_client = "gateway_responses" }} +capable = {{ id = "model/capable", llm_client = "gateway_messages" }} +efficient = {{ id = "model/efficient", llm_client = "gateway_messages" }} + +[routes.agent] +id = "switchyard/agent" +type = "composite" +classifier = {{ target = "judge", base_threshold = 0.5, classify_trigger = "user_turn" }} +stage = {{ capable_target = "capable", efficient_target = "efficient", confidence_threshold = 0.5 }} + +[routes.claude] +id = "switchyard/claude" +type = "passthrough" +target = "efficient" +"# + ))?; + let app = build_switchyard_router(state); + let bearer = [ + ("authorization", "Bearer gateway-key"), + ("chatgpt-account-id", "account-1"), + ]; + + for (path, body) in [ + ( + "/v1/chat/completions", + json!({"model": "switchyard/agent", "messages": [{"role": "user", "content": "hello"}]}), + ), + ( + "/v1/responses", + json!({"model": "switchyard/agent", "input": "hi there"}), + ), + ] { + let response = send_with_headers(&app, "POST", path, Some(body), &bearer).await?; + assert_eq!( + response.status, + StatusCode::OK, + "{path}: {}", + response.text()? + ); + assert_eq!( + response.headers["x-model-router-selected-model"], + "model/efficient" + ); + } + // Both endpoints receive exactly one copy of the caller's bearer token and no `x-api-key`. + // Only the Messages call gets `anthropic-version`. + let judge = json!({ + "path": "/v1/responses", "authorization": ["Bearer gateway-key"], "x_api_key": [], + "chatgpt_account_id": ["account-1"], "anthropic_version": [] + }); + let answer = json!({ + "path": "/v1/messages", "authorization": ["Bearer gateway-key"], "x_api_key": [], + "chatgpt_account_id": ["account-1"], "anthropic_version": ["2023-06-01"] + }); + assert_eq!( + *calls.lock().await, + [judge.clone(), answer.clone(), judge, answer] + ); + + // The mixed route serves OpenAI callers only, so a Messages caller gets 400 before any call. + let messages_body = |model: &str| { + json!({ + "model": model, + "max_tokens": 16, + "messages": [{"role": "user", "content": "hello"}] + }) + }; + let wrong_api = send_with_headers( + &app, + "POST", + "/v1/messages", + Some(messages_body("switchyard/agent")), + &bearer, + ) + .await?; + assert_eq!(wrong_api.status, StatusCode::BAD_REQUEST); + assert_eq!( + wrong_api.json()?["error"]["message"], + "route switchyard/agent forwards an OpenAI login; call it through /v1/chat/completions or /v1/responses" + ); + assert_eq!(calls.lock().await.len(), 4); + + // A route that uses only the Messages client still serves Messages callers. + let claude = send_with_headers( + &app, + "POST", + "/v1/messages", + Some(messages_body("switchyard/claude")), + &[("authorization", "Bearer gateway-key")], + ) + .await?; + assert_eq!(claude.status, StatusCode::OK, "{}", claude.text()?); + assert_eq!( + calls.lock().await[4..], + [json!({ + "path": "/v1/messages", "authorization": ["Bearer gateway-key"], "x_api_key": [], + "chatgpt_account_id": [], "anthropic_version": ["2023-06-01"] + })] + ); + Ok(()) +} + #[tokio::test] async fn routes_dispatch_and_discovery_endpoints_are_stable() -> TestResult { let (upstream, app) = test_app(&[ diff --git a/docs/getting_started.md b/docs/getting_started.md index f8865b1ed..bb961c7a7 100644 --- a/docs/getting_started.md +++ b/docs/getting_started.md @@ -109,10 +109,14 @@ A client can set `forward_auth = true` instead of `api_key_env` to send each caller's credential to that upstream. OpenAI clients forward `authorization`, `chatgpt-account-id`, and `x-openai-fedramp`. Anthropic clients forward `authorization` or `x-api-key`. Enable this only for an upstream that should -receive the caller's login. All backends reachable through the route, including -efficient and capable targets, must use the same provider. Other application -headers are preserved, so they may contain provider-specific credentials. The -server rejects a forwarding route called through the other provider's API. +receive the caller's login. Other application headers are preserved, so they +may contain provider-specific credentials. The forwarding clients in a route, +including efficient and capable targets, must use one credential family: all +OpenAI formats or all `anthropic_messages`. A route may mix the two families +only when all of its forwarding clients use the same scheme, host, and port. +Such a route serves Chat Completions and Responses callers and forwards the +caller's bearer token to every client. The server returns 400 to a caller whose +API the route does not serve. ### Run the server diff --git a/docs/integrations/nemo_relay.md b/docs/integrations/nemo_relay.md index 31ce7a640..24f8070b6 100644 --- a/docs/integrations/nemo_relay.md +++ b/docs/integrations/nemo_relay.md @@ -222,7 +222,9 @@ and configure `api_key_env` on each authenticated client. The two options cannot be enabled together. If each caller must use its own provider credential, use standalone `switchyard-server`. Standalone forwarding requires the caller and target to use the same credential family: OpenAI-compatible (Chat Completions -and Responses) or Anthropic (Messages). +and Responses) or Anthropic (Messages). A route may mix both families when all +of its forwarding clients use the same scheme, host, and port; such a route +serves Chat Completions and Responses callers. Support for provider-specific fields depends on the source and target formats. Test any fields that your application relies on before deploying a translated diff --git a/docs/reference/toml_schema.md b/docs/reference/toml_schema.md index c77c6c89e..fc340b819 100644 --- a/docs/reference/toml_schema.md +++ b/docs/reference/toml_schema.md @@ -51,7 +51,7 @@ route reaches no upstream. A file without a `[targets]` table is rejected with | `format` | Yes | — | `openai_chat`, `openai_responses`, or `anthropic_messages`. | | `base_url` | Yes | — | Upstream base URL. | | `api_key_env` | No | unset | Name of the environment variable holding the key. Omit to send no authentication. | -| `forward_auth` | No | `false` | Forward the caller's provider credential and application headers. All backends reachable through the route must use the same provider. | +| `forward_auth` | No | `false` | Forward the caller's provider credential and application headers. A route's forwarding clients must use one credential family unless they all use the same scheme, host, and port. | | `extra_headers` | No | `{}` | Custom HTTP headers sent to the model server. Set credentials with `api_key_env` or `forward_auth`; the server rejects headers owned by the selected auth mode. Header names are case-insensitive. | | `max_retries` | No | `2` | Retry budget, `0`–`10`. | | `failure_cooldown_ms` | No | `5000` (5 seconds) | Skip a backend for this many milliseconds after an exhausted transient completion failure. Zero disables it. | @@ -106,13 +106,18 @@ values. This setting gives `base_url` the caller's login. Enable it only when that upstream should receive the credential, and use HTTPS unless the upstream runs -on loopback. All backends reachable through the route must use the same -provider because other application headers are preserved and may contain +on loopback. Other application headers are preserved and may contain provider-specific credentials. Forwarding clients do not follow HTTP redirects. -Check every forwarding client used by a route, including classifier and judge -targets. The server rejects an Anthropic forwarding route called through an -OpenAI endpoint, or an OpenAI forwarding route called through an Anthropic -endpoint, before it calls an upstream. + +All forwarding clients in a route, including classifier and judge targets, must +use one credential family: OpenAI (`openai_chat` and `openai_responses` clients, +which serve Chat Completions and Responses callers) or Anthropic +(`anthropic_messages` clients, which serve Messages callers). A route may mix +the two families only when all of its forwarding clients use the same host: the +same scheme, host, and port in `base_url`, though the path may differ. Such a +route serves Chat Completions and Responses callers and forwards the caller's +bearer token to every client. The server returns `400` without calling an +upstream when a caller uses an API that the route does not serve. ## `[targets.]` From 09148f761e7396fe14da24f34dbb7b4a3f6e45c8 Mon Sep 17 00:00:00 2001 From: Elyas Mehtabuddin Date: Fri, 2 Oct 2026 14:04:49 -0700 Subject: [PATCH 2/5] feat(server): explain the gateway key case in docs and the different-hosts error Signed-off-by: Elyas Mehtabuddin --- crates/libsy-llm-client/README.md | 9 ++-- crates/libsy-llm-client/src/backend.rs | 5 ++- crates/libsy-llm-client/src/client.rs | 6 +-- crates/switchyard-nemo-relay-plugin/README.md | 4 +- crates/switchyard-runner/src/config.rs | 28 +++++++++---- crates/switchyard-server/README.md | 11 +++-- docs/getting_started.md | 13 +++--- docs/integrations/nemo_relay.md | 4 +- docs/reference/toml_schema.md | 42 ++++++++++++++----- 9 files changed, 83 insertions(+), 39 deletions(-) diff --git a/crates/libsy-llm-client/README.md b/crates/libsy-llm-client/README.md index 8450c0b84..81ff8ecab 100644 --- a/crates/libsy-llm-client/README.md +++ b/crates/libsy-llm-client/README.md @@ -242,10 +242,11 @@ fn build_multi_format_client( normalize `authorization`, `chatgpt-account-id`, and `x-openai-fedramp`. Anthropic backends forward `authorization` or `x-api-key`; they also keep `oauth-*` values from `anthropic-beta` and remove other caller-supplied beta values. - All backends reachable through a forwarding route must use one credential family - (OpenAI or Anthropic) unless they all use the same scheme, host, and port. Such a - route serves Chat Completions and Responses callers and forwards their bearer - token to every backend. + The forwarding backends in one route must use one credential family (OpenAI or + Anthropic) unless they all use the same scheme, host, and port, such as one LLM + gateway that accepts the caller's gateway key on every endpoint. Such a route + serves Chat Completions and Responses callers and forwards their bearer token + to every backend. Backends that send a configured key are not restricted. Headers owned by other providers are preserved as application headers. - Per-backend custom headers go in `HttpBackendConfig::extra_headers`. Set credentials with `api_key`. OpenAI backends reject `Authorization`; Anthropic backends reject `x-api-key` diff --git a/crates/libsy-llm-client/src/backend.rs b/crates/libsy-llm-client/src/backend.rs index fa6f4064e..0a28ef6ce 100644 --- a/crates/libsy-llm-client/src/backend.rs +++ b/crates/libsy-llm-client/src/backend.rs @@ -52,8 +52,9 @@ pub struct HttpBackendConfig { pub api_key: Option, /// Whether this backend forwards the caller's provider credential and application headers. /// - /// All backends reachable through a forwarding route must use one credential family - /// (OpenAI or Anthropic) unless they all use the same scheme, host, and port. + /// The forwarding backends in one route must use one credential family (OpenAI or + /// Anthropic) unless they all use the same scheme, host, and port, such as one LLM + /// gateway. Backends that send a configured key are not restricted. pub forward_auth: bool, /// Custom headers added to every outbound call to this backend. /// diff --git a/crates/libsy-llm-client/src/client.rs b/crates/libsy-llm-client/src/client.rs index ab11c7259..7348012f7 100644 --- a/crates/libsy-llm-client/src/client.rs +++ b/crates/libsy-llm-client/src/client.rs @@ -640,9 +640,9 @@ impl TranslatingLlmClient { /// `http_headers` are carried through as the request's /// [`Metadata::http_headers`]. Backends with `forward_auth` disabled forward only /// allowed metadata headers; `forward_auth` backends forward all application - /// headers. All backends reachable through a forwarding route must use one credential - /// family unless they all use the same scheme, host, and port. Transport headers are - /// always rebuilt. Pass `None` to forward nothing. + /// headers. The forwarding backends in one route must use one credential family + /// unless they all use the same scheme, host, and port. Transport headers are always + /// rebuilt. Pass `None` to forward nothing. pub async fn call_rewrite_model_raw( &self, raw_http_request: Value, diff --git a/crates/switchyard-nemo-relay-plugin/README.md b/crates/switchyard-nemo-relay-plugin/README.md index 0b156070c..058796082 100644 --- a/crates/switchyard-nemo-relay-plugin/README.md +++ b/crates/switchyard-nemo-relay-plugin/README.md @@ -217,8 +217,8 @@ be enabled together. If each caller must use its own provider credential, use standalone `switchyard-server`. Standalone forwarding requires the caller and target to use the same credential family: OpenAI-compatible (Chat Completions and Responses) or Anthropic (Messages). A route may mix both families when all -of its forwarding clients use the same scheme, host, and port; such a route -serves Chat Completions and Responses callers. +of its forwarding clients use the same scheme, host, and port, such as one LLM +gateway; such a route serves Chat Completions and Responses callers. Support for provider-specific fields depends on the source and target formats. Test any fields that your application relies on before deploying a translated diff --git a/crates/switchyard-runner/src/config.rs b/crates/switchyard-runner/src/config.rs index fcfdcfb90..a24e96ea7 100644 --- a/crates/switchyard-runner/src/config.rs +++ b/crates/switchyard-runner/src/config.rs @@ -395,15 +395,19 @@ impl DeploymentConfig { let client: Arc = client.clone(); by_model.insert(target.id.clone(), client); } - // A route may mix credential families only when all of its forwarding clients use the - // same scheme, host, and port, so the caller's credential reaches only that host. Such a - // route serves Chat Completions and Responses callers because Anthropic clients forward - // the caller's `authorization` header unchanged. + // Only forwarding clients count here. A client with api_key_env sends the server's own + // key, so a route can mix formats and providers through such clients. A forwarded + // credential belongs to the service that issued it, so forwarding clients must share one + // credential family unless they all use the same scheme, host, and port. One host that + // serves both formats is a single service, such as an LLM gateway that accepts the + // caller's gateway key on its OpenAI and Anthropic endpoints. Such a route serves Chat + // Completions and Responses callers because Anthropic clients forward the caller's + // `authorization` header unchanged. if mixes_families { if forwarding_origins.len() > 1 { let origins = Vec::from_iter(forwarding_origins).join(", "); return Err(RunnerError::configuration(format!( - "route {route_name} cannot forward both Anthropic and OpenAI caller credentials to different hosts ({origins}); point all of its forwarding clients at one host" + "route {route_name} cannot forward both Anthropic and OpenAI caller credentials to different hosts ({origins}); point all of its forwarding clients at one host, such as an LLM gateway, or set api_key_env instead of forward_auth on one provider's clients" ))); } caller_auth = Some(CallerAuthKind::OpenAi); @@ -1963,7 +1967,7 @@ target = "capable" } #[test] - fn forwarding_route_mixes_credential_families_only_on_one_host() -> RunnerResult<()> { + fn mixed_formats_need_one_host_only_when_forwarding() -> RunnerResult<()> { // The URL path does not count, so both base URLs name one host. for messages_url in [ "https://gateway.example.test", @@ -1996,11 +2000,21 @@ target = "capable" let error = error_message(&mixed_forwarding_config(messages_url)); assert!( error.contains(&format!( - "route hub cannot forward both Anthropic and OpenAI caller credentials to different hosts ({origins}); point all of its forwarding clients at one host" + "route hub cannot forward both Anthropic and OpenAI caller credentials to different hosts ({origins}); point all of its forwarding clients at one host, such as an LLM gateway, or set api_key_env instead of forward_auth on one provider's clients" )), "{error}" ); } + + // Clients that send the server's own key never limit a route, even on two hosts, so the + // route serves callers of every API. + let server_keys = mixed_forwarding_config("https://api.anthropic.test") + .replace("forward_auth = true", "api_key_env = \"PATH\""); + let runner = runner_from_toml(&server_keys)?; + assert_eq!( + runner.route("switchyard/hub").and_then(Route::caller_auth), + None + ); Ok(()) } diff --git a/crates/switchyard-server/README.md b/crates/switchyard-server/README.md index cccb3ccb3..4a6c5ea89 100644 --- a/crates/switchyard-server/README.md +++ b/crates/switchyard-server/README.md @@ -94,11 +94,14 @@ forward `authorization` or `x-api-key`. Enable this only when every forwarding client's `base_url` should receive the caller's login. Other application headers are preserved and may contain provider-specific credentials. A route's forwarding clients must use one credential family: all OpenAI formats or all -`anthropic_messages`. A route may mix the two families only when all of its -forwarding clients use the same scheme, host, and port. Such a route serves -Chat Completions and Responses callers and forwards the caller's bearer token to -every client. +`anthropic_messages`. The exception is one host that serves both formats, such +as an LLM gateway that accepts each caller's gateway key on every endpoint: a +route may mix the two families when all of its forwarding clients use the same +scheme, host, and port. Such a route serves Chat Completions and Responses +callers and forwards the caller's bearer token to every client. The server returns 400 to a caller whose API the route does not serve. +Clients that use `api_key_env` send the server's own key, so these limits do +not apply to them. Target-level `extra_body` values are shallow-merged into the upstream request when the request does not already contain that key. Target-level `system_prompt` values are prepended when that target serves a completion. diff --git a/docs/getting_started.md b/docs/getting_started.md index bb961c7a7..acfa1c240 100644 --- a/docs/getting_started.md +++ b/docs/getting_started.md @@ -112,11 +112,14 @@ caller's credential to that upstream. OpenAI clients forward `authorization`, receive the caller's login. Other application headers are preserved, so they may contain provider-specific credentials. The forwarding clients in a route, including efficient and capable targets, must use one credential family: all -OpenAI formats or all `anthropic_messages`. A route may mix the two families -only when all of its forwarding clients use the same scheme, host, and port. -Such a route serves Chat Completions and Responses callers and forwards the -caller's bearer token to every client. The server returns 400 to a caller whose -API the route does not serve. +OpenAI formats or all `anthropic_messages`. The exception is one host that +serves both formats, such as an LLM gateway that accepts each caller's gateway +key on every endpoint: a route may mix the two families when all of its +forwarding clients use the same scheme, host, and port. Such a route serves +Chat Completions and Responses callers and forwards the caller's bearer token +to every client. The server returns 400 to a caller whose API the route does +not serve. Clients that use `api_key_env` send the server's own key, so these +limits do not apply to them. ### Run the server diff --git a/docs/integrations/nemo_relay.md b/docs/integrations/nemo_relay.md index 24f8070b6..5339ac565 100644 --- a/docs/integrations/nemo_relay.md +++ b/docs/integrations/nemo_relay.md @@ -223,8 +223,8 @@ be enabled together. If each caller must use its own provider credential, use standalone `switchyard-server`. Standalone forwarding requires the caller and target to use the same credential family: OpenAI-compatible (Chat Completions and Responses) or Anthropic (Messages). A route may mix both families when all -of its forwarding clients use the same scheme, host, and port; such a route -serves Chat Completions and Responses callers. +of its forwarding clients use the same scheme, host, and port, such as one LLM +gateway; such a route serves Chat Completions and Responses callers. Support for provider-specific fields depends on the source and target formats. Test any fields that your application relies on before deploying a translated diff --git a/docs/reference/toml_schema.md b/docs/reference/toml_schema.md index fc340b819..3bb92e6b6 100644 --- a/docs/reference/toml_schema.md +++ b/docs/reference/toml_schema.md @@ -51,7 +51,7 @@ route reaches no upstream. A file without a `[targets]` table is rejected with | `format` | Yes | — | `openai_chat`, `openai_responses`, or `anthropic_messages`. | | `base_url` | Yes | — | Upstream base URL. | | `api_key_env` | No | unset | Name of the environment variable holding the key. Omit to send no authentication. | -| `forward_auth` | No | `false` | Forward the caller's provider credential and application headers. A route's forwarding clients must use one credential family unless they all use the same scheme, host, and port. | +| `forward_auth` | No | `false` | Forward the caller's provider credential and application headers. A route's forwarding clients must use one credential family unless they all use the same scheme, host, and port, such as one LLM gateway. | | `extra_headers` | No | `{}` | Custom HTTP headers sent to the model server. Set credentials with `api_key_env` or `forward_auth`; the server rejects headers owned by the selected auth mode. Header names are case-insensitive. | | `max_retries` | No | `2` | Retry budget, `0`–`10`. | | `failure_cooldown_ms` | No | `5000` (5 seconds) | Skip a backend for this many milliseconds after an exhausted transient completion failure. Zero disables it. | @@ -109,15 +109,37 @@ upstream should receive the credential, and use HTTPS unless the upstream runs on loopback. Other application headers are preserved and may contain provider-specific credentials. Forwarding clients do not follow HTTP redirects. -All forwarding clients in a route, including classifier and judge targets, must -use one credential family: OpenAI (`openai_chat` and `openai_responses` clients, -which serve Chat Completions and Responses callers) or Anthropic -(`anthropic_messages` clients, which serve Messages callers). A route may mix -the two families only when all of its forwarding clients use the same host: the -same scheme, host, and port in `base_url`, though the path may differ. Such a -route serves Chat Completions and Responses callers and forwards the caller's -bearer token to every client. The server returns `400` without calling an -upstream when a caller uses an API that the route does not serve. +A forwarded credential belongs to the service that issued it. A ChatGPT login, +for example, must never reach Anthropic. So all forwarding clients in a route, +including classifier and judge targets, must use one credential family: OpenAI +(`openai_chat` and `openai_responses` clients, which serve Chat Completions and +Responses callers) or Anthropic (`anthropic_messages` clients, which serve +Messages callers). + +The exception is one host that serves both formats, such as an LLM gateway that +accepts each caller's gateway key on its OpenAI and Anthropic endpoints. A route +may mix the two families when all of its forwarding clients use the same scheme, +host, and port in `base_url`; the path may differ: + +```toml +[llm_clients.gateway_responses] +format = "openai_responses" +base_url = "https://gateway.example.com/v1" +forward_auth = true + +[llm_clients.gateway_messages] +format = "anthropic_messages" +base_url = "https://gateway.example.com" +forward_auth = true +``` + +Such a route serves Chat Completions and Responses callers and forwards the +caller's bearer token to every client. The server returns `400` without calling +an upstream when a caller uses an API that the route does not serve. + +These limits apply only to forwarded credentials. A client with `api_key_env` +sends the server's own key, so a route can mix formats and providers through +such clients. ## `[targets.]` From a976376a4319822f970c7c2c53bda5736668b240 Mon Sep 17 00:00:00 2001 From: Elyas Mehtabuddin Date: Fri, 2 Oct 2026 14:16:09 -0700 Subject: [PATCH 3/5] test: shorten the one-host forwarding tests Signed-off-by: Elyas Mehtabuddin --- crates/switchyard-runner/src/config.rs | 106 +++++++------------ crates/switchyard-server/tests/server.rs | 128 +++++------------------ 2 files changed, 61 insertions(+), 173 deletions(-) diff --git a/crates/switchyard-runner/src/config.rs b/crates/switchyard-runner/src/config.rs index a24e96ea7..ed9027ab0 100644 --- a/crates/switchyard-runner/src/config.rs +++ b/crates/switchyard-runner/src/config.rs @@ -395,14 +395,11 @@ impl DeploymentConfig { let client: Arc = client.clone(); by_model.insert(target.id.clone(), client); } - // Only forwarding clients count here. A client with api_key_env sends the server's own - // key, so a route can mix formats and providers through such clients. A forwarded - // credential belongs to the service that issued it, so forwarding clients must share one - // credential family unless they all use the same scheme, host, and port. One host that - // serves both formats is a single service, such as an LLM gateway that accepts the - // caller's gateway key on its OpenAI and Anthropic endpoints. Such a route serves Chat - // Completions and Responses callers because Anthropic clients forward the caller's - // `authorization` header unchanged. + // Only forwarding clients count: an api_key_env client sends the server's own key. + // Forwarding clients must share one credential family unless they all use the same + // scheme, host, and port, such as one LLM gateway that accepts the caller's gateway key + // on every endpoint. Such a route serves Chat Completions and Responses callers because + // Anthropic clients forward the caller's `authorization` header unchanged. if mixes_families { if forwarding_origins.len() > 1 { let origins = Vec::from_iter(forwarding_origins).join(", "); @@ -1921,8 +1918,8 @@ confidence_threshold = 0.5 } } - // Builds a classifier route whose judge uses a Responses client and whose tiers use a - // Messages client, plus a passthrough route that uses only the Messages client. + // A route that mixes a GPT target on a Responses client with a Claude target on a Messages + // client, plus a route that uses only the Messages client. fn mixed_forwarding_config(messages_url: &str) -> String { format!( r#" @@ -1938,83 +1935,50 @@ format = "anthropic_messages" base_url = "{messages_url}" forward_auth = true -[targets.judge] -id = "judge/model" -llm_client = "responses" - -[targets.capable] -id = "capable/model" -llm_client = "messages" - -[targets.efficient] -id = "efficient/model" -llm_client = "messages" +[targets] +gpt = {{ id = "gpt/model", llm_client = "responses" }} +claude = {{ id = "claude/model", llm_client = "messages" }} -[routes.hub] -id = "switchyard/hub" -type = "llm_classifier" -classifier_target = "judge" -strong_target = "capable" -weak_target = "efficient" -base_threshold = 0.5 +[routes.mixed] +id = "switchyard/mixed" +type = "random" +targets = ["gpt", "claude"] [routes.claude] id = "switchyard/claude" type = "passthrough" -target = "capable" +target = "claude" "# ) } #[test] - fn mixed_formats_need_one_host_only_when_forwarding() -> RunnerResult<()> { - // The URL path does not count, so both base URLs name one host. - for messages_url in [ - "https://gateway.example.test", - "https://gateway.example.test/v1", - ] { - let runner = runner_from_toml(&mixed_forwarding_config(messages_url))?; - let caller_auth = |id| runner.route(id).and_then(Route::caller_auth); - assert_eq!(caller_auth("switchyard/hub"), Some(CallerAuthKind::OpenAi)); - assert_eq!( - caller_auth("switchyard/claude"), - Some(CallerAuthKind::Anthropic) - ); - } + fn forwarding_route_mixes_formats_only_on_one_host() -> RunnerResult<()> { + // Same host, different paths: the mixed route serves OpenAI callers, and the route that + // uses only the Messages client keeps serving Messages callers. + let runner = runner_from_toml(&mixed_forwarding_config("https://gateway.example.test"))?; + let caller_auth = |id| runner.route(id).and_then(Route::caller_auth); + assert_eq!( + caller_auth("switchyard/mixed"), + Some(CallerAuthKind::OpenAi) + ); + assert_eq!( + caller_auth("switchyard/claude"), + Some(CallerAuthKind::Anthropic) + ); - // A different host, port, or scheme is a different origin. - for (messages_url, origins) in [ - ( - "https://api.anthropic.test", - "https://api.anthropic.test, https://gateway.example.test", - ), - ( - "https://gateway.example.test:8443", - "https://gateway.example.test, https://gateway.example.test:8443", - ), - ( - "http://gateway.example.test", - "http://gateway.example.test, https://gateway.example.test", - ), + // A different host, port, or scheme fails, and the error names it. + for other in [ + "https://api.anthropic.test", + "https://gateway.example.test:8443", + "http://gateway.example.test", ] { - let error = error_message(&mixed_forwarding_config(messages_url)); + let error = error_message(&mixed_forwarding_config(other)); assert!( - error.contains(&format!( - "route hub cannot forward both Anthropic and OpenAI caller credentials to different hosts ({origins}); point all of its forwarding clients at one host, such as an LLM gateway, or set api_key_env instead of forward_auth on one provider's clients" - )), + error.contains("different hosts") && error.contains(other), "{error}" ); } - - // Clients that send the server's own key never limit a route, even on two hosts, so the - // route serves callers of every API. - let server_keys = mixed_forwarding_config("https://api.anthropic.test") - .replace("forward_auth = true", "api_key_env = \"PATH\""); - let runner = runner_from_toml(&server_keys)?; - assert_eq!( - runner.route("switchyard/hub").and_then(Route::caller_auth), - None - ); Ok(()) } diff --git a/crates/switchyard-server/tests/server.rs b/crates/switchyard-server/tests/server.rs index 63a012886..3fe9582e4 100644 --- a/crates/switchyard-server/tests/server.rs +++ b/crates/switchyard-server/tests/server.rs @@ -3283,30 +3283,23 @@ target = "openai" Ok(()) } -/// Serves `/v1/responses` and `/v1/messages` for a stub gateway on one host. For each call, it -/// records every value of the `authorization`, `x-api-key`, `chatgpt-account-id`, and -/// `anthropic-version` headers, so a header sent twice shows up as two values. Every -/// `/v1/responses` call returns a judge verdict that picks the efficient tier. +/// Serves `/v1/responses` and `/v1/messages` for a stub gateway on one host and records the +/// path and `authorization` values of each call. `/v1/responses` calls get a judge verdict. async fn upstream_gateway_records_auth( State(calls): State>>>, uri: Uri, headers: HeaderMap, Json(body): Json, ) -> HttpResponse { - let header = |name: &str| { - headers - .get_all(name) - .iter() - .filter_map(|value| value.to_str().ok()) - .collect::>() - }; - calls.lock().await.push(json!({ - "path": uri.path(), - "authorization": header("authorization"), - "x_api_key": header("x-api-key"), - "chatgpt_account_id": header("chatgpt-account-id"), - "anthropic_version": header("anthropic-version"), - })); + let authorization: Vec<_> = headers + .get_all("authorization") + .iter() + .filter_map(|value| value.to_str().ok()) + .collect(); + calls + .lock() + .await + .push(json!({"path": uri.path(), "authorization": authorization})); let model = body["model"].as_str().unwrap_or_default(); if uri.path() == "/v1/responses" { let verdict = json!({ @@ -3332,7 +3325,7 @@ async fn route_on_one_host_forwards_the_bearer_token_to_responses_and_messages() .route("/v1/messages", post(upstream_gateway_records_auth)) .with_state(Arc::clone(&calls)); let listener = TcpListener::bind("127.0.0.1:0").await?; - let base_url = format!("http://{}/v1", listener.local_addr()?); + let host = format!("http://{}", listener.local_addr()?); tokio::spawn(async move { axum::serve(listener, gateway).await }); let state = load_test_config(&format!( r#" @@ -3340,13 +3333,13 @@ schema_version = 1 [llm_clients.gateway_responses] format = "openai_responses" -base_url = "{base_url}" +base_url = "{host}/v1" forward_auth = true max_retries = 0 [llm_clients.gateway_messages] format = "anthropic_messages" -base_url = "{base_url}" +base_url = "{host}" forward_auth = true max_retries = 0 @@ -3360,95 +3353,26 @@ id = "switchyard/agent" type = "composite" classifier = {{ target = "judge", base_threshold = 0.5, classify_trigger = "user_turn" }} stage = {{ capable_target = "capable", efficient_target = "efficient", confidence_threshold = 0.5 }} - -[routes.claude] -id = "switchyard/claude" -type = "passthrough" -target = "efficient" "# ))?; - let app = build_switchyard_router(state); - let bearer = [ - ("authorization", "Bearer gateway-key"), - ("chatgpt-account-id", "account-1"), - ]; - - for (path, body) in [ - ( - "/v1/chat/completions", - json!({"model": "switchyard/agent", "messages": [{"role": "user", "content": "hello"}]}), - ), - ( - "/v1/responses", - json!({"model": "switchyard/agent", "input": "hi there"}), - ), - ] { - let response = send_with_headers(&app, "POST", path, Some(body), &bearer).await?; - assert_eq!( - response.status, - StatusCode::OK, - "{path}: {}", - response.text()? - ); - assert_eq!( - response.headers["x-model-router-selected-model"], - "model/efficient" - ); - } - // Both endpoints receive exactly one copy of the caller's bearer token and no `x-api-key`. - // Only the Messages call gets `anthropic-version`. - let judge = json!({ - "path": "/v1/responses", "authorization": ["Bearer gateway-key"], "x_api_key": [], - "chatgpt_account_id": ["account-1"], "anthropic_version": [] - }); - let answer = json!({ - "path": "/v1/messages", "authorization": ["Bearer gateway-key"], "x_api_key": [], - "chatgpt_account_id": ["account-1"], "anthropic_version": ["2023-06-01"] - }); - assert_eq!( - *calls.lock().await, - [judge.clone(), answer.clone(), judge, answer] - ); - - // The mixed route serves OpenAI callers only, so a Messages caller gets 400 before any call. - let messages_body = |model: &str| { - json!({ - "model": model, - "max_tokens": 16, - "messages": [{"role": "user", "content": "hello"}] - }) - }; - let wrong_api = send_with_headers( - &app, - "POST", - "/v1/messages", - Some(messages_body("switchyard/agent")), - &bearer, - ) - .await?; - assert_eq!(wrong_api.status, StatusCode::BAD_REQUEST); - assert_eq!( - wrong_api.json()?["error"]["message"], - "route switchyard/agent forwards an OpenAI login; call it through /v1/chat/completions or /v1/responses" - ); - assert_eq!(calls.lock().await.len(), 4); - // A route that uses only the Messages client still serves Messages callers. - let claude = send_with_headers( - &app, + let response = send_with_headers( + &build_switchyard_router(state), "POST", - "/v1/messages", - Some(messages_body("switchyard/claude")), + "/v1/responses", + Some(json!({"model": "switchyard/agent", "input": "hi there"})), &[("authorization", "Bearer gateway-key")], ) .await?; - assert_eq!(claude.status, StatusCode::OK, "{}", claude.text()?); + assert_eq!(response.status, StatusCode::OK, "{}", response.text()?); + + // The judge call and the answer call each carry the caller's bearer token once, unchanged. assert_eq!( - calls.lock().await[4..], - [json!({ - "path": "/v1/messages", "authorization": ["Bearer gateway-key"], "x_api_key": [], - "chatgpt_account_id": [], "anthropic_version": ["2023-06-01"] - })] + *calls.lock().await, + [ + json!({"path": "/v1/responses", "authorization": ["Bearer gateway-key"]}), + json!({"path": "/v1/messages", "authorization": ["Bearer gateway-key"]}), + ] ); Ok(()) } From 434d07b6e090c31f03333ea87ce0abb8a1519981 Mon Sep 17 00:00:00 2001 From: Elyas Mehtabuddin Date: Fri, 2 Oct 2026 14:45:31 -0700 Subject: [PATCH 4/5] refactor(runner): rename mixes_families to has_mixed_families and add doc comments Signed-off-by: Elyas Mehtabuddin --- crates/switchyard-runner/src/config.rs | 14 +++++++++----- crates/switchyard-server/tests/server.rs | 2 ++ 2 files changed, 11 insertions(+), 5 deletions(-) diff --git a/crates/switchyard-runner/src/config.rs b/crates/switchyard-runner/src/config.rs index ed9027ab0..2395ead5f 100644 --- a/crates/switchyard-runner/src/config.rs +++ b/crates/switchyard-runner/src/config.rs @@ -350,6 +350,9 @@ impl DeploymentConfig { .collect() } + /// Builds the client router for one route. The second value is the caller credential family + /// that the route's forwarding clients need, or `None` when no client forwards the caller's + /// credential. A request through the other family's APIs fails before any upstream call. fn build_route_clients( &self, route_name: &str, @@ -363,7 +366,7 @@ impl DeploymentConfig { let mut by_model = HashMap::new(); let mut targets_by_model: HashMap<&str, (&str, &TargetConfig)> = HashMap::new(); let mut caller_auth = None; - let mut mixes_families = false; + let mut has_mixed_families = false; let mut forwarding_origins = BTreeSet::new(); for name in route.callable_target_names() { let target = self.targets.get(name).ok_or_else(|| { @@ -388,7 +391,7 @@ impl DeploymentConfig { })?; if client_config.forward_auth { let target_auth = client_config.format.caller_auth_kind(); - mixes_families |= caller_auth.is_some_and(|kind| kind != target_auth); + has_mixed_families |= caller_auth.is_some_and(|kind| kind != target_auth); caller_auth = Some(target_auth); forwarding_origins.insert(client_config.base_url.0.origin().ascii_serialization()); } @@ -400,7 +403,7 @@ impl DeploymentConfig { // scheme, host, and port, such as one LLM gateway that accepts the caller's gateway key // on every endpoint. Such a route serves Chat Completions and Responses callers because // Anthropic clients forward the caller's `authorization` header unchanged. - if mixes_families { + if has_mixed_families { if forwarding_origins.len() > 1 { let origins = Vec::from_iter(forwarding_origins).join(", "); return Err(RunnerError::configuration(format!( @@ -1918,8 +1921,8 @@ confidence_threshold = 0.5 } } - // A route that mixes a GPT target on a Responses client with a Claude target on a Messages - // client, plus a route that uses only the Messages client. + /// Returns a config whose `mixed` route has a GPT target on a Responses client and a Claude + /// target on a Messages client at `messages_url`. Its `claude` route uses Claude only. fn mixed_forwarding_config(messages_url: &str) -> String { format!( r#" @@ -1952,6 +1955,7 @@ target = "claude" ) } + /// A forwarding route can mix OpenAI and Anthropic clients only when they all use one host. #[test] fn forwarding_route_mixes_formats_only_on_one_host() -> RunnerResult<()> { // Same host, different paths: the mixed route serves OpenAI callers, and the route that diff --git a/crates/switchyard-server/tests/server.rs b/crates/switchyard-server/tests/server.rs index 3fe9582e4..b8f5bffe4 100644 --- a/crates/switchyard-server/tests/server.rs +++ b/crates/switchyard-server/tests/server.rs @@ -3317,6 +3317,8 @@ async fn upstream_gateway_records_auth( .into_response() } +/// A route that mixes formats on one gateway forwards the caller's bearer token to both of the +/// gateway's APIs: `/v1/responses` for the judge and `/v1/messages` for the answer. #[tokio::test] async fn route_on_one_host_forwards_the_bearer_token_to_responses_and_messages() -> TestResult { let calls = Arc::new(Mutex::new(Vec::new())); From 66d05e3f291cd9c7dc4a1045ff9b26f2599b27aa Mon Sep 17 00:00:00 2001 From: Elyas Mehtabuddin Date: Fri, 2 Oct 2026 14:52:59 -0700 Subject: [PATCH 5/5] fix(runner): list origins in the forwarding error and say only forwarding clients get the token Signed-off-by: Elyas Mehtabuddin --- crates/libsy-llm-client/README.md | 2 +- crates/switchyard-runner/src/config.rs | 7 ++++--- crates/switchyard-server/README.md | 2 +- docs/getting_started.md | 6 +++--- docs/reference/toml_schema.md | 4 ++-- 5 files changed, 11 insertions(+), 10 deletions(-) diff --git a/crates/libsy-llm-client/README.md b/crates/libsy-llm-client/README.md index 81ff8ecab..19a423c16 100644 --- a/crates/libsy-llm-client/README.md +++ b/crates/libsy-llm-client/README.md @@ -246,7 +246,7 @@ fn build_multi_format_client( Anthropic) unless they all use the same scheme, host, and port, such as one LLM gateway that accepts the caller's gateway key on every endpoint. Such a route serves Chat Completions and Responses callers and forwards their bearer token - to every backend. Backends that send a configured key are not restricted. + to every forwarding backend. Backends that send a configured key are not restricted. Headers owned by other providers are preserved as application headers. - Per-backend custom headers go in `HttpBackendConfig::extra_headers`. Set credentials with `api_key`. OpenAI backends reject `Authorization`; Anthropic backends reject `x-api-key` diff --git a/crates/switchyard-runner/src/config.rs b/crates/switchyard-runner/src/config.rs index 2395ead5f..51105bd42 100644 --- a/crates/switchyard-runner/src/config.rs +++ b/crates/switchyard-runner/src/config.rs @@ -407,7 +407,7 @@ impl DeploymentConfig { if forwarding_origins.len() > 1 { let origins = Vec::from_iter(forwarding_origins).join(", "); return Err(RunnerError::configuration(format!( - "route {route_name} cannot forward both Anthropic and OpenAI caller credentials to different hosts ({origins}); point all of its forwarding clients at one host, such as an LLM gateway, or set api_key_env instead of forward_auth on one provider's clients" + "route {route_name} cannot forward both Anthropic and OpenAI caller credentials to different origins ({origins}); point all of its forwarding clients at one origin (same scheme, host, and port), such as an LLM gateway, or set api_key_env instead of forward_auth on one provider's clients" ))); } caller_auth = Some(CallerAuthKind::OpenAi); @@ -1955,7 +1955,8 @@ target = "claude" ) } - /// A forwarding route can mix OpenAI and Anthropic clients only when they all use one host. + /// A forwarding route can mix OpenAI and Anthropic clients only when they all use the same + /// scheme, host, and port. #[test] fn forwarding_route_mixes_formats_only_on_one_host() -> RunnerResult<()> { // Same host, different paths: the mixed route serves OpenAI callers, and the route that @@ -1979,7 +1980,7 @@ target = "claude" ] { let error = error_message(&mixed_forwarding_config(other)); assert!( - error.contains("different hosts") && error.contains(other), + error.contains("different origins") && error.contains(other), "{error}" ); } diff --git a/crates/switchyard-server/README.md b/crates/switchyard-server/README.md index 4a6c5ea89..08335197f 100644 --- a/crates/switchyard-server/README.md +++ b/crates/switchyard-server/README.md @@ -98,7 +98,7 @@ forwarding clients must use one credential family: all OpenAI formats or all as an LLM gateway that accepts each caller's gateway key on every endpoint: a route may mix the two families when all of its forwarding clients use the same scheme, host, and port. Such a route serves Chat Completions and Responses -callers and forwards the caller's bearer token to every client. +callers and forwards the caller's bearer token to every forwarding client. The server returns 400 to a caller whose API the route does not serve. Clients that use `api_key_env` send the server's own key, so these limits do not apply to them. diff --git a/docs/getting_started.md b/docs/getting_started.md index acfa1c240..5d524765e 100644 --- a/docs/getting_started.md +++ b/docs/getting_started.md @@ -117,9 +117,9 @@ serves both formats, such as an LLM gateway that accepts each caller's gateway key on every endpoint: a route may mix the two families when all of its forwarding clients use the same scheme, host, and port. Such a route serves Chat Completions and Responses callers and forwards the caller's bearer token -to every client. The server returns 400 to a caller whose API the route does -not serve. Clients that use `api_key_env` send the server's own key, so these -limits do not apply to them. +to every forwarding client. The server returns 400 to a caller whose API the +route does not serve. Clients that use `api_key_env` send the server's own key, +so these limits do not apply to them. ### Run the server diff --git a/docs/reference/toml_schema.md b/docs/reference/toml_schema.md index 3bb92e6b6..9f820e72f 100644 --- a/docs/reference/toml_schema.md +++ b/docs/reference/toml_schema.md @@ -134,8 +134,8 @@ forward_auth = true ``` Such a route serves Chat Completions and Responses callers and forwards the -caller's bearer token to every client. The server returns `400` without calling -an upstream when a caller uses an API that the route does not serve. +caller's bearer token to every forwarding client. The server returns `400` without +calling an upstream when a caller uses an API that the route does not serve. These limits apply only to forwarded credentials. A client with `api_key_env` sends the server's own key, so a route can mix formats and providers through