diff --git a/crates/libsy-llm-client/README.md b/crates/libsy-llm-client/README.md index 4a880d3a0..19a423c16 100644 --- a/crates/libsy-llm-client/README.md +++ b/crates/libsy-llm-client/README.md @@ -242,7 +242,11 @@ fn build_multi_format_client( normalize `authorization`, `chatgpt-account-id`, and `x-openai-fedramp`. Anthropic backends forward `authorization` or `x-api-key`; they also keep `oauth-*` values from `anthropic-beta` and remove other caller-supplied beta values. - All backends reachable through a forwarding route must use the same provider. + The forwarding backends in one route must use one credential family (OpenAI or + Anthropic) unless they all use the same scheme, host, and port, such as one LLM + gateway that accepts the caller's gateway key on every endpoint. Such a route + serves Chat Completions and Responses callers and forwards their bearer token + to every forwarding backend. Backends that send a configured key are not restricted. Headers owned by other providers are preserved as application headers. - Per-backend custom headers go in `HttpBackendConfig::extra_headers`. Set credentials with `api_key`. OpenAI backends reject `Authorization`; Anthropic backends reject `x-api-key` diff --git a/crates/libsy-llm-client/src/backend.rs b/crates/libsy-llm-client/src/backend.rs index bbe23b0d1..0a28ef6ce 100644 --- a/crates/libsy-llm-client/src/backend.rs +++ b/crates/libsy-llm-client/src/backend.rs @@ -52,7 +52,9 @@ pub struct HttpBackendConfig { pub api_key: Option, /// Whether this backend forwards the caller's provider credential and application headers. /// - /// All backends reachable through a forwarding route must use the same provider. + /// The forwarding backends in one route must use one credential family (OpenAI or + /// Anthropic) unless they all use the same scheme, host, and port, such as one LLM + /// gateway. Backends that send a configured key are not restricted. pub forward_auth: bool, /// Custom headers added to every outbound call to this backend. /// diff --git a/crates/libsy-llm-client/src/client.rs b/crates/libsy-llm-client/src/client.rs index 6ba089d9c..7348012f7 100644 --- a/crates/libsy-llm-client/src/client.rs +++ b/crates/libsy-llm-client/src/client.rs @@ -640,8 +640,9 @@ impl TranslatingLlmClient { /// `http_headers` are carried through as the request's /// [`Metadata::http_headers`]. Backends with `forward_auth` disabled forward only /// allowed metadata headers; `forward_auth` backends forward all application - /// headers. All backends reachable through a forwarding route must use the same - /// provider. Transport headers are always rebuilt. Pass `None` to forward nothing. + /// headers. The forwarding backends in one route must use one credential family + /// unless they all use the same scheme, host, and port. Transport headers are always + /// rebuilt. Pass `None` to forward nothing. pub async fn call_rewrite_model_raw( &self, raw_http_request: Value, diff --git a/crates/switchyard-nemo-relay-plugin/README.md b/crates/switchyard-nemo-relay-plugin/README.md index 0bf2103ad..058796082 100644 --- a/crates/switchyard-nemo-relay-plugin/README.md +++ b/crates/switchyard-nemo-relay-plugin/README.md @@ -216,7 +216,9 @@ and configure `api_key_env` on each authenticated client. The two options cannot be enabled together. If each caller must use its own provider credential, use standalone `switchyard-server`. Standalone forwarding requires the caller and target to use the same credential family: OpenAI-compatible (Chat Completions -and Responses) or Anthropic (Messages). +and Responses) or Anthropic (Messages). A route may mix both families when all +of its forwarding clients use the same scheme, host, and port, such as one LLM +gateway; such a route serves Chat Completions and Responses callers. Support for provider-specific fields depends on the source and target formats. Test any fields that your application relies on before deploying a translated diff --git a/crates/switchyard-runner/src/config.rs b/crates/switchyard-runner/src/config.rs index 1c976a47b..51105bd42 100644 --- a/crates/switchyard-runner/src/config.rs +++ b/crates/switchyard-runner/src/config.rs @@ -350,6 +350,9 @@ impl DeploymentConfig { .collect() } + /// Builds the client router for one route. The second value is the caller credential family + /// that the route's forwarding clients need, or `None` when no client forwards the caller's + /// credential. A request through the other family's APIs fails before any upstream call. fn build_route_clients( &self, route_name: &str, @@ -363,6 +366,8 @@ impl DeploymentConfig { let mut by_model = HashMap::new(); let mut targets_by_model: HashMap<&str, (&str, &TargetConfig)> = HashMap::new(); let mut caller_auth = None; + let mut has_mixed_families = false; + let mut forwarding_origins = BTreeSet::new(); for name in route.callable_target_names() { let target = self.targets.get(name).ok_or_else(|| { RunnerError::configuration(format!("route references unknown target {name}")) @@ -386,16 +391,27 @@ impl DeploymentConfig { })?; if client_config.forward_auth { let target_auth = client_config.format.caller_auth_kind(); - if caller_auth.is_some_and(|kind| kind != target_auth) { - return Err(RunnerError::configuration(format!( - "route {route_name} cannot forward both Anthropic and OpenAI caller credentials" - ))); - } + has_mixed_families |= caller_auth.is_some_and(|kind| kind != target_auth); caller_auth = Some(target_auth); + forwarding_origins.insert(client_config.base_url.0.origin().ascii_serialization()); } let client: Arc = client.clone(); by_model.insert(target.id.clone(), client); } + // Only forwarding clients count: an api_key_env client sends the server's own key. + // Forwarding clients must share one credential family unless they all use the same + // scheme, host, and port, such as one LLM gateway that accepts the caller's gateway key + // on every endpoint. Such a route serves Chat Completions and Responses callers because + // Anthropic clients forward the caller's `authorization` header unchanged. + if has_mixed_families { + if forwarding_origins.len() > 1 { + let origins = Vec::from_iter(forwarding_origins).join(", "); + return Err(RunnerError::configuration(format!( + "route {route_name} cannot forward both Anthropic and OpenAI caller credentials to different origins ({origins}); point all of its forwarding clients at one origin (same scheme, host, and port), such as an LLM gateway, or set api_key_env instead of forward_auth on one provider's clients" + ))); + } + caller_auth = Some(CallerAuthKind::OpenAi); + } let completion_targets = route .routing_target_names() .into_iter() @@ -1905,6 +1921,72 @@ confidence_threshold = 0.5 } } + /// Returns a config whose `mixed` route has a GPT target on a Responses client and a Claude + /// target on a Messages client at `messages_url`. Its `claude` route uses Claude only. + fn mixed_forwarding_config(messages_url: &str) -> String { + format!( + r#" +schema_version = 1 + +[llm_clients.responses] +format = "openai_responses" +base_url = "https://gateway.example.test/v1" +forward_auth = true + +[llm_clients.messages] +format = "anthropic_messages" +base_url = "{messages_url}" +forward_auth = true + +[targets] +gpt = {{ id = "gpt/model", llm_client = "responses" }} +claude = {{ id = "claude/model", llm_client = "messages" }} + +[routes.mixed] +id = "switchyard/mixed" +type = "random" +targets = ["gpt", "claude"] + +[routes.claude] +id = "switchyard/claude" +type = "passthrough" +target = "claude" +"# + ) + } + + /// A forwarding route can mix OpenAI and Anthropic clients only when they all use the same + /// scheme, host, and port. + #[test] + fn forwarding_route_mixes_formats_only_on_one_host() -> RunnerResult<()> { + // Same host, different paths: the mixed route serves OpenAI callers, and the route that + // uses only the Messages client keeps serving Messages callers. + let runner = runner_from_toml(&mixed_forwarding_config("https://gateway.example.test"))?; + let caller_auth = |id| runner.route(id).and_then(Route::caller_auth); + assert_eq!( + caller_auth("switchyard/mixed"), + Some(CallerAuthKind::OpenAi) + ); + assert_eq!( + caller_auth("switchyard/claude"), + Some(CallerAuthKind::Anthropic) + ); + + // A different host, port, or scheme fails, and the error names it. + for other in [ + "https://api.anthropic.test", + "https://gateway.example.test:8443", + "http://gateway.example.test", + ] { + let error = error_message(&mixed_forwarding_config(other)); + assert!( + error.contains("different origins") && error.contains(other), + "{error}" + ); + } + Ok(()) + } + const ADVISOR_CONFIG: &str = r#" schema_version = 1 diff --git a/crates/switchyard-server/README.md b/crates/switchyard-server/README.md index 8bb64b91f..08335197f 100644 --- a/crates/switchyard-server/README.md +++ b/crates/switchyard-server/README.md @@ -91,10 +91,17 @@ A client can set `forward_auth = true` instead of `api_key_env` to send the caller's credential to the configured upstream. OpenAI clients forward `authorization`, `chatgpt-account-id`, and `x-openai-fedramp`. Anthropic clients forward `authorization` or `x-api-key`. Enable this only when every forwarding -client's `base_url` should receive the caller's login. All backends reachable -through the route must use the same provider. Other application headers are -preserved and may contain provider-specific credentials. A forwarding route -must be called through the matching provider API. +client's `base_url` should receive the caller's login. Other application headers +are preserved and may contain provider-specific credentials. A route's +forwarding clients must use one credential family: all OpenAI formats or all +`anthropic_messages`. The exception is one host that serves both formats, such +as an LLM gateway that accepts each caller's gateway key on every endpoint: a +route may mix the two families when all of its forwarding clients use the same +scheme, host, and port. Such a route serves Chat Completions and Responses +callers and forwards the caller's bearer token to every forwarding client. +The server returns 400 to a caller whose API the route does not serve. +Clients that use `api_key_env` send the server's own key, so these limits do +not apply to them. Target-level `extra_body` values are shallow-merged into the upstream request when the request does not already contain that key. Target-level `system_prompt` values are prepended when that target serves a completion. diff --git a/crates/switchyard-server/tests/server.rs b/crates/switchyard-server/tests/server.rs index 42e6a6911..b8f5bffe4 100644 --- a/crates/switchyard-server/tests/server.rs +++ b/crates/switchyard-server/tests/server.rs @@ -3283,6 +3283,102 @@ target = "openai" Ok(()) } +/// Serves `/v1/responses` and `/v1/messages` for a stub gateway on one host and records the +/// path and `authorization` values of each call. `/v1/responses` calls get a judge verdict. +async fn upstream_gateway_records_auth( + State(calls): State>>>, + uri: Uri, + headers: HeaderMap, + Json(body): Json, +) -> HttpResponse { + let authorization: Vec<_> = headers + .get_all("authorization") + .iter() + .filter_map(|value| value.to_str().ok()) + .collect(); + calls + .lock() + .await + .push(json!({"path": uri.path(), "authorization": authorization})); + let model = body["model"].as_str().unwrap_or_default(); + if uri.path() == "/v1/responses" { + let verdict = json!({ + "crux": "bounded task", "primary_rule": "SUP-1", + "capability_boundary": "supported", "p_solve": 0.9, + }); + return Json(responses_body("resp_judge", model, &verdict.to_string())).into_response(); + } + Json(json!({ + "id": "msg_gateway", "type": "message", "role": "assistant", "model": model, + "content": [{"type": "text", "text": "ok"}], + "stop_reason": "end_turn", "stop_sequence": null, + "usage": {"input_tokens": 1, "output_tokens": 1} + })) + .into_response() +} + +/// A route that mixes formats on one gateway forwards the caller's bearer token to both of the +/// gateway's APIs: `/v1/responses` for the judge and `/v1/messages` for the answer. +#[tokio::test] +async fn route_on_one_host_forwards_the_bearer_token_to_responses_and_messages() -> TestResult { + let calls = Arc::new(Mutex::new(Vec::new())); + let gateway = Router::new() + .route("/v1/responses", post(upstream_gateway_records_auth)) + .route("/v1/messages", post(upstream_gateway_records_auth)) + .with_state(Arc::clone(&calls)); + let listener = TcpListener::bind("127.0.0.1:0").await?; + let host = format!("http://{}", listener.local_addr()?); + tokio::spawn(async move { axum::serve(listener, gateway).await }); + let state = load_test_config(&format!( + r#" +schema_version = 1 + +[llm_clients.gateway_responses] +format = "openai_responses" +base_url = "{host}/v1" +forward_auth = true +max_retries = 0 + +[llm_clients.gateway_messages] +format = "anthropic_messages" +base_url = "{host}" +forward_auth = true +max_retries = 0 + +[targets] +judge = {{ id = "model/judge", llm_client = "gateway_responses" }} +capable = {{ id = "model/capable", llm_client = "gateway_messages" }} +efficient = {{ id = "model/efficient", llm_client = "gateway_messages" }} + +[routes.agent] +id = "switchyard/agent" +type = "composite" +classifier = {{ target = "judge", base_threshold = 0.5, classify_trigger = "user_turn" }} +stage = {{ capable_target = "capable", efficient_target = "efficient", confidence_threshold = 0.5 }} +"# + ))?; + + let response = send_with_headers( + &build_switchyard_router(state), + "POST", + "/v1/responses", + Some(json!({"model": "switchyard/agent", "input": "hi there"})), + &[("authorization", "Bearer gateway-key")], + ) + .await?; + assert_eq!(response.status, StatusCode::OK, "{}", response.text()?); + + // The judge call and the answer call each carry the caller's bearer token once, unchanged. + assert_eq!( + *calls.lock().await, + [ + json!({"path": "/v1/responses", "authorization": ["Bearer gateway-key"]}), + json!({"path": "/v1/messages", "authorization": ["Bearer gateway-key"]}), + ] + ); + Ok(()) +} + #[tokio::test] async fn routes_dispatch_and_discovery_endpoints_are_stable() -> TestResult { let (upstream, app) = test_app(&[ diff --git a/docs/getting_started.md b/docs/getting_started.md index f8865b1ed..5d524765e 100644 --- a/docs/getting_started.md +++ b/docs/getting_started.md @@ -109,10 +109,17 @@ A client can set `forward_auth = true` instead of `api_key_env` to send each caller's credential to that upstream. OpenAI clients forward `authorization`, `chatgpt-account-id`, and `x-openai-fedramp`. Anthropic clients forward `authorization` or `x-api-key`. Enable this only for an upstream that should -receive the caller's login. All backends reachable through the route, including -efficient and capable targets, must use the same provider. Other application -headers are preserved, so they may contain provider-specific credentials. The -server rejects a forwarding route called through the other provider's API. +receive the caller's login. Other application headers are preserved, so they +may contain provider-specific credentials. The forwarding clients in a route, +including efficient and capable targets, must use one credential family: all +OpenAI formats or all `anthropic_messages`. The exception is one host that +serves both formats, such as an LLM gateway that accepts each caller's gateway +key on every endpoint: a route may mix the two families when all of its +forwarding clients use the same scheme, host, and port. Such a route serves +Chat Completions and Responses callers and forwards the caller's bearer token +to every forwarding client. The server returns 400 to a caller whose API the +route does not serve. Clients that use `api_key_env` send the server's own key, +so these limits do not apply to them. ### Run the server diff --git a/docs/integrations/nemo_relay.md b/docs/integrations/nemo_relay.md index 31ce7a640..5339ac565 100644 --- a/docs/integrations/nemo_relay.md +++ b/docs/integrations/nemo_relay.md @@ -222,7 +222,9 @@ and configure `api_key_env` on each authenticated client. The two options cannot be enabled together. If each caller must use its own provider credential, use standalone `switchyard-server`. Standalone forwarding requires the caller and target to use the same credential family: OpenAI-compatible (Chat Completions -and Responses) or Anthropic (Messages). +and Responses) or Anthropic (Messages). A route may mix both families when all +of its forwarding clients use the same scheme, host, and port, such as one LLM +gateway; such a route serves Chat Completions and Responses callers. Support for provider-specific fields depends on the source and target formats. Test any fields that your application relies on before deploying a translated diff --git a/docs/reference/toml_schema.md b/docs/reference/toml_schema.md index c77c6c89e..9f820e72f 100644 --- a/docs/reference/toml_schema.md +++ b/docs/reference/toml_schema.md @@ -51,7 +51,7 @@ route reaches no upstream. A file without a `[targets]` table is rejected with | `format` | Yes | — | `openai_chat`, `openai_responses`, or `anthropic_messages`. | | `base_url` | Yes | — | Upstream base URL. | | `api_key_env` | No | unset | Name of the environment variable holding the key. Omit to send no authentication. | -| `forward_auth` | No | `false` | Forward the caller's provider credential and application headers. All backends reachable through the route must use the same provider. | +| `forward_auth` | No | `false` | Forward the caller's provider credential and application headers. A route's forwarding clients must use one credential family unless they all use the same scheme, host, and port, such as one LLM gateway. | | `extra_headers` | No | `{}` | Custom HTTP headers sent to the model server. Set credentials with `api_key_env` or `forward_auth`; the server rejects headers owned by the selected auth mode. Header names are case-insensitive. | | `max_retries` | No | `2` | Retry budget, `0`–`10`. | | `failure_cooldown_ms` | No | `5000` (5 seconds) | Skip a backend for this many milliseconds after an exhausted transient completion failure. Zero disables it. | @@ -106,13 +106,40 @@ values. This setting gives `base_url` the caller's login. Enable it only when that upstream should receive the credential, and use HTTPS unless the upstream runs -on loopback. All backends reachable through the route must use the same -provider because other application headers are preserved and may contain +on loopback. Other application headers are preserved and may contain provider-specific credentials. Forwarding clients do not follow HTTP redirects. -Check every forwarding client used by a route, including classifier and judge -targets. The server rejects an Anthropic forwarding route called through an -OpenAI endpoint, or an OpenAI forwarding route called through an Anthropic -endpoint, before it calls an upstream. + +A forwarded credential belongs to the service that issued it. A ChatGPT login, +for example, must never reach Anthropic. So all forwarding clients in a route, +including classifier and judge targets, must use one credential family: OpenAI +(`openai_chat` and `openai_responses` clients, which serve Chat Completions and +Responses callers) or Anthropic (`anthropic_messages` clients, which serve +Messages callers). + +The exception is one host that serves both formats, such as an LLM gateway that +accepts each caller's gateway key on its OpenAI and Anthropic endpoints. A route +may mix the two families when all of its forwarding clients use the same scheme, +host, and port in `base_url`; the path may differ: + +```toml +[llm_clients.gateway_responses] +format = "openai_responses" +base_url = "https://gateway.example.com/v1" +forward_auth = true + +[llm_clients.gateway_messages] +format = "anthropic_messages" +base_url = "https://gateway.example.com" +forward_auth = true +``` + +Such a route serves Chat Completions and Responses callers and forwards the +caller's bearer token to every forwarding client. The server returns `400` without +calling an upstream when a caller uses an API that the route does not serve. + +These limits apply only to forwarded credentials. A client with `api_key_env` +sends the server's own key, so a route can mix formats and providers through +such clients. ## `[targets.]`