From 35bb96f136d2a552f0dd69f8fa126f4f8167f92d Mon Sep 17 00:00:00 2001 From: ChethanUK Date: Wed, 30 Sep 2026 14:37:32 +0200 Subject: [PATCH] feat(runner): accept `subagents` on an llm_classifier route Signed-off-by: ChethanUK --- crates/switchyard-runner/src/algorithm.rs | 86 +++++++----- crates/switchyard-runner/src/config.rs | 143 ++++++++++++++++---- crates/switchyard-server/tests/server.rs | 62 +++++++++ docs/reference/toml_schema.md | 1 + docs/routing_algorithms/overview.md | 2 +- docs/routing_algorithms/subagent_routing.md | 4 +- 6 files changed, 237 insertions(+), 61 deletions(-) diff --git a/crates/switchyard-runner/src/algorithm.rs b/crates/switchyard-runner/src/algorithm.rs index e742d48ba..781e4fe26 100644 --- a/crates/switchyard-runner/src/algorithm.rs +++ b/crates/switchyard-runner/src/algorithm.rs @@ -267,7 +267,7 @@ pub struct LlmClassifierRouteConfig { } /// Routing policy applied only to delegated sub-agent work, nested inside a -/// `passthrough` or `stage_router` route. +/// `passthrough`, `llm_classifier`, `stage_router` or `composite` route. #[derive(Clone, Debug, Deserialize)] #[serde(tag = "type", rename_all = "snake_case", deny_unknown_fields)] pub enum SubagentRouteConfig { @@ -351,6 +351,9 @@ pub enum AlgorithmSpec { /// Judge and tier settings, written directly in the route table. #[serde(flatten)] config: LlmClassifierRouteConfig, + /// Separate policy for delegated sub-agent work. + #[serde(default)] + subagents: Option, }, /// Picks a tier per turn by scoring signals from recent tool results. StageRouter { @@ -554,25 +557,33 @@ impl AlgorithmSpec { efficient_target, .. } => vec![capable_target.as_str(), efficient_target.as_str()], - Self::LlmClassifier { config, .. } => match config.classifier_mode() { - ClassifierMode::Capability => config - .weak_target - .iter() - .chain(&config.strong_target) - .map(String::as_str) - .collect(), - ClassifierMode::Escalation => config - .strong_target - .iter() - .chain(&config.weak_target) - .map(String::as_str) - .collect(), - ClassifierMode::Custom => config - .models - .as_ref() - .map(CategoryModelConfig::routing_names) - .unwrap_or_default(), - }, + Self::LlmClassifier { + config, subagents, .. + } => { + let mut names: Vec<&str> = match config.classifier_mode() { + ClassifierMode::Capability => config + .weak_target + .iter() + .chain(&config.strong_target) + .map(String::as_str) + .collect(), + ClassifierMode::Escalation => config + .strong_target + .iter() + .chain(&config.weak_target) + .map(String::as_str) + .collect(), + ClassifierMode::Custom => config + .models + .as_ref() + .map(CategoryModelConfig::routing_names) + .unwrap_or_default(), + }; + if let Some(subagents) = subagents { + names.extend(subagents.routing_target_names()); + } + names + } Self::StageRouter { tiers, subagents, .. } => { @@ -649,6 +660,10 @@ impl AlgorithmSpec { subagents: Some(subagents), .. } + | Self::LlmClassifier { + subagents: Some(subagents), + .. + } | Self::StageRouter { subagents: Some(subagents), .. @@ -688,7 +703,7 @@ impl AlgorithmSpec { vec![capable_target.clone(), efficient_target.clone()], ), ]), - Self::LlmClassifier { config } => { + Self::LlmClassifier { config, .. } => { classifier_runtime_model_names(config.validated_classifier_mode(route_name)?) } Self::StageRouter { @@ -742,6 +757,7 @@ impl AlgorithmSpec { let subagents = match self { Self::Passthrough { subagents, .. } + | Self::LlmClassifier { subagents, .. } | Self::StageRouter { subagents, .. } | Self::Composite { subagents, .. } => subagents.as_ref(), _ => None, @@ -755,22 +771,25 @@ impl AlgorithmSpec { Ok(RuntimeModelNames { parent, subagent }) } - /// Response target and routing-only dependency for routers that answer while routing. - pub(crate) fn routing_response_and_dependency(&self) -> Option<(&str, &str)> { + /// Response target and routing-only dependencies for routers that answer while routing. + pub(crate) fn routing_response_and_dependencies(&self) -> Option<(&str, Vec<&str>)> { match self { - Self::LlmClassifier { config, .. } - if matches!(config.classifier_mode(), ClassifierMode::Escalation) => - { - Some(( - config.weak_target.as_deref()?, - config.classifier_target.as_str(), - )) + Self::LlmClassifier { + config, subagents, .. + } if matches!(config.classifier_mode(), ClassifierMode::Escalation) => { + // A sub-agent judge is routing-only too; on the answer model it would get + // that model's system_prompt. + let mut dependencies = vec![config.classifier_target.as_str()]; + if let Some(subagents) = subagents { + dependencies.extend(subagents.judge_target_names()); + } + Some((config.weak_target.as_deref()?, dependencies)) } Self::Advisor { executor_target, advisor_target, .. - } => Some((executor_target, advisor_target)), + } => Some((executor_target, vec![advisor_target])), Self::Noop { .. } | Self::Random { .. } | Self::Passthrough { .. } @@ -1223,7 +1242,7 @@ fn build_algorithm( } AlgorithmSpec::LlmClassifier { config: classifier_config, - .. + subagents, } => { let mode = classifier_config.validated_classifier_mode(route_name)?; let algorithm = match mode { @@ -1283,7 +1302,8 @@ fn build_algorithm( error, ) })?; - Ok(Arc::new(algorithm)) + let parent: Arc = Arc::new(algorithm); + attach_subagent_router(route_name, parent, subagents.as_ref(), targets) } AlgorithmSpec::StageRouter { tiers, diff --git a/crates/switchyard-runner/src/config.rs b/crates/switchyard-runner/src/config.rs index f37b4cda9..86c9f4dad 100644 --- a/crates/switchyard-runner/src/config.rs +++ b/crates/switchyard-runner/src/config.rs @@ -440,8 +440,8 @@ impl DeploymentConfig { prompts, routing_answer_target: None, }; - let Some((response_name, dependency_name)) = - route.algorithm.routing_response_and_dependency() + let Some((response_name, dependency_names)) = + route.algorithm.routing_response_and_dependencies() else { return Ok(policy); }; @@ -451,14 +451,18 @@ impl DeploymentConfig { if !policy.prompts.contains_key(&response.id) { return Ok(policy); } - let dependency = self.targets.get(dependency_name).ok_or_else(|| { - RunnerError::configuration(format!("route references unknown target {dependency_name}")) - })?; - if response.id == dependency.id { - return Err(RunnerError::configuration(format!( - "route {route_name} cannot apply system_prompt to target {response_name}: model {} is also used by routing-only target {dependency_name}", - response.id, - ))); + for dependency_name in dependency_names { + let dependency = self.targets.get(dependency_name).ok_or_else(|| { + RunnerError::configuration(format!( + "route references unknown target {dependency_name}" + )) + })?; + if response.id == dependency.id { + return Err(RunnerError::configuration(format!( + "route {route_name} cannot apply system_prompt to target {response_name}: model {} is also used by routing-only target {dependency_name}", + response.id, + ))); + } } policy.routing_answer_target = Some(response.id.clone()); Ok(policy) @@ -930,20 +934,27 @@ classify_trigger = "new_session""#, // The sub-agent target is also the parent's capable tier. Merged into one group it // would be indistinguishable from that tier, and delegated work would follow the // parent's ordering instead of its own configured target. - let runner = runner_from_toml(&with_subagent_passthrough(&stage_config(), "stage"))?; - let models = runner - .route("switchyard/stage") - .expect("stage route should exist") - .models(); - - assert_eq!( - models.subagent_models_for(&Category::Any), - [ModelId::from("strong/model")] - ); - assert_eq!( - models.models_for(&Category::Any), - [ModelId::from("strong/model"), ModelId::from("weak/model")] - ); + for (route, parent_any) in [ + ("stage", ["strong", "weak"]), + ("classifier", ["weak", "strong"]), + ] { + let runner = runner_from_toml(&with_subagent_passthrough(&stage_config(), route))?; + let models = runner + .route(&format!("switchyard/{route}")) + .expect("route should exist") + .models(); + + assert_eq!( + models.subagent_models_for(&Category::Any), + [ModelId::from("strong/model")], + "{route}" + ); + assert_eq!( + models.models_for(&Category::Any), + parent_any.map(|name| ModelId::from(format!("{name}/model"))), + "{route}" + ); + } Ok(()) } @@ -1067,7 +1078,7 @@ new = ["send_message"] } #[test] - fn passthrough_and_stage_accept_subagent_routing() -> RunnerResult<()> { + fn parent_routes_accept_subagent_routing() -> RunnerResult<()> { let stage = stage_config(); let stage_with_classifier = with_subagent_llm_classifier(&stage, "stage", ""); let parsed: DeploymentConfig = toml::from_str(&stage_with_classifier).map_err(|error| { @@ -1081,17 +1092,84 @@ new = ["send_message"] assert!(callable_targets.contains(&expected)); } + // An llm_classifier parent ends up with two judges once it nests a sub-agent route: + // its own and the child's. The child also gets a target of its own (`worker`), which + // only appears in the parent's callable targets if the child's targets are included. + let base = VALID_CONFIG.replace( + "classifier_target = \"classifier\"", + "classifier_target = \"parent_judge\"", + ) + "\n[targets.parent_judge]\nid = \"parent-judge/model\"\nllm_client = \"primary\"\n\n[targets.worker]\nid = \"worker/model\"\nllm_client = \"primary\"\n"; + let classifier_with_classifier = with_subagent_llm_classifier(&base, "classifier", "") + .replace("capable = [\"strong\"]", "capable = [\"worker\"]") + .replace( + "any = [\"strong\", \"weak\"]", + "any = [\"worker\", \"weak\"]", + ); + let parsed: DeploymentConfig = + toml::from_str(&classifier_with_classifier).map_err(|error| { + RunnerError::configuration(format!("failed to parse classifier config: {error}")) + })?; + let Some(classifier_route) = parsed.routes.get("classifier") else { + return Err(RunnerError::configuration("classifier route is missing")); + }; + let callable_targets = classifier_route.callable_target_names(); + for expected in ["weak", "strong", "parent_judge", "classifier", "worker"] { + assert!(callable_targets.contains(&expected), "{expected}"); + } + for configured in [ with_subagent_llm_classifier(VALID_CONFIG, "passthrough", ""), with_subagent_passthrough(VALID_CONFIG, "passthrough"), stage_with_classifier, with_subagent_passthrough(&stage, "stage"), + classifier_with_classifier, + // One llm_classifier-child row per parent shape (capability above, escalation, + // custom); a passthrough child has no judge, so it adds no branch here. + // Escalation answers while routing; a prompted weak target is fine while the + // child's judge runs on a different model. + with_subagent_llm_classifier(&prompted_escalation_config(), "classifier", ""), + with_subagent_llm_classifier(&custom_classifier_parent_config(), "custom", ""), ] { runner_from_toml(&configured)?; } Ok(()) } + fn prompted_escalation_config() -> String { + let escalation = + VALID_CONFIG.replace("base_threshold = 0.5", "escalation = { confirmations = 1 }"); + assert!( + escalation.contains("escalation = "), + "escalation replace missed" + ); + let prompted = escalation.replace( + "id = \"weak/model\"\nllm_client = \"anthropic\"", + "id = \"weak/model\"\nllm_client = \"anthropic\"\nsystem_prompt = \"answer prompt\"", + ); + assert!( + prompted.contains("system_prompt = "), + "system_prompt replace missed" + ); + prompted + } + + fn custom_classifier_parent_config() -> String { + format!( + r#"{VALID_CONFIG} +[routes.custom] +id = "switchyard/custom" +type = "llm_classifier" +mode = "custom" +models = {{ judge = ["classifier"], capable = ["strong"], efficient = ["weak"], any = ["strong", "weak"] }} +default_target = "efficient" +prompt = "Select a target." +response_schema = '{{"type":"object","properties":{{"target":{{"type":"string","enum":["capable","efficient"]}}}},"required":["target"],"additionalProperties":false}}' +policy = {{ type = "target_selector", selector = "/target" }} +classify_trigger = "new_session" +"# + ) + } + #[test] fn aliased_completion_targets_reject_prompt_conflicts() { let configured = stage_config() @@ -1464,6 +1542,8 @@ classifier_magic = true ), "message_hash_fallback requires classify_trigger = new_session", ), + // Sub-agent affinity needs the harness child identity, so nested classifier + // routes reject message hash fallback. ( with_subagent_llm_classifier( VALID_CONFIG, @@ -1472,6 +1552,19 @@ classifier_magic = true ), "cannot use message_hash_fallback", ), + ( + with_subagent_llm_classifier( + VALID_CONFIG, + "classifier", + "\nmessage_hash_fallback = true", + ), + "cannot use message_hash_fallback", + ), + ( + with_subagent_llm_classifier(&prompted_escalation_config(), "classifier", "") + .replace("judge = [\"classifier\"]", "judge = [\"weak\"]"), + "cannot apply system_prompt to target weak: model weak/model is also used by routing-only target weak", + ), ( with_subagent_llm_classifier(VALID_CONFIG, "passthrough", "") .replace("mode = \"custom\"", "mode = \"capability\""), diff --git a/crates/switchyard-server/tests/server.rs b/crates/switchyard-server/tests/server.rs index cd21d3c91..46ca122b9 100644 --- a/crates/switchyard-server/tests/server.rs +++ b/crates/switchyard-server/tests/server.rs @@ -3753,6 +3753,68 @@ selector = "/decision/target" Ok(()) } +// An llm_classifier parent sends delegated work to its sub-agent route without calling +// its own judge, and keeps judging its own traffic. +#[tokio::test] +async fn llm_classifier_parent_sends_subagent_work_to_the_child_route() -> TestResult { + for mode in [ + "base_threshold = 0.5", + "mode = \"escalation\"\nescalation = { confirmations = 1 }", + ] { + let upstream = MockUpstream::start().await?; + let app = build_switchyard_router(load_test_config(&format!( + r#" +schema_version = 1 +[llm_clients.upstream] +format = "openai_chat" +base_url = "{base_url}" +[targets] +classifier = {{ id = "model/classifier", llm_client = "upstream" }} +strong = {{ id = "model/strong", llm_client = "upstream" }} +weak = {{ id = "model/weak", llm_client = "upstream" }} +worker = {{ id = "model/worker", llm_client = "upstream" }} +[routes.agent] +id = "agent" +type = "llm_classifier" +classifier_target = "classifier" +strong_target = "strong" +weak_target = "weak" +{mode} +[routes.agent.subagents] +type = "passthrough" +target = "worker" +"#, + base_url = upstream.base_url + ))?); + let body = json!({"model":"agent","messages":[{"role":"user","content":"bounded task"}]}); + + let child = [ + ("x-claude-code-session-id", "root-session"), + ("x-claude-code-agent-id", "child-agent"), + ]; + let response = send_with_headers( + &app, + "POST", + "/v1/chat/completions", + Some(body.clone()), + &child, + ) + .await?; + assert_eq!(response.status, StatusCode::OK, "{mode}"); + assert_eq!(upstream.models().await, ["model/worker"], "{mode}"); + + upstream.calls.lock().await.clear(); + let parent = [("x-claude-code-session-id", "root-session")]; + let response = + send_with_headers(&app, "POST", "/v1/chat/completions", Some(body), &parent).await?; + assert_eq!(response.status, StatusCode::OK, "{mode}"); + let mut models = upstream.models().await; + models.sort(); + assert_eq!(models, ["model/classifier", "model/weak"], "{mode}"); + } + Ok(()) +} + // Codex's structured kind takes precedence over the flat `collab_spawn` header: // maintenance uses the parent classifier; delegated work uses the subagent route. #[tokio::test] diff --git a/docs/reference/toml_schema.md b/docs/reference/toml_schema.md index 8bf382a08..91f458128 100644 --- a/docs/reference/toml_schema.md +++ b/docs/reference/toml_schema.md @@ -230,6 +230,7 @@ Runs one of three judge-backed modes: `capability`, `escalation`, or `custom`. | `classifier_target` | Capability, escalation | — | Target the judge is called through. Not a routing destination. Custom mode uses `models.judge`. | | `max_output_tokens` | No | `4096` | Maximum completion tokens for the judge verdict. Must be at least `1`. | | `response_format_type` | No | `json_schema` | Structured-output mode for capability and escalation judges. Use `json_object` when the provider does not support JSON Schema; Switchyard adds the schema to the prompt and validates the verdict locally. Custom mode always uses its configured JSON Schema. | +| `subagents` | No | unset | Nested `passthrough` or custom `llm_classifier` policy used only for delegated sub-agent work. See [Sub-Agent-Aware Routing](../routing_algorithms/subagent_routing.md). | Capability mode classifies before serving. See [LLM Classifier Routing](../routing_algorithms/llm_classifier_routing.md). diff --git a/docs/routing_algorithms/overview.md b/docs/routing_algorithms/overview.md index 1a1eac7f0..7712e1120 100644 --- a/docs/routing_algorithms/overview.md +++ b/docs/routing_algorithms/overview.md @@ -45,7 +45,7 @@ These options remain available when you need a different routing policy. | [Escalation](escalation_router_routing.md) | Start on the efficient model and escalate when an LLM judge detects trouble. | `llm_classifier` with `mode = "escalation"` | | [Custom](llm_classifier_routing.md#custom-multi-target-routing) | Route among two or more models using your own classification schema and rules. | `llm_classifier` with `mode = "custom"` | | [Advisor Gate](advisor_gate_routing.md) | Keep one executor model and have a stronger advisor review its plans and completion claims. | `advisor` | -| [Sub-Agent-Aware Routing](subagent_routing.md) | Delegated sub-agents should use a separate routing policy from the parent agent. | `passthrough`, `stage_router`, or `composite` with `subagents` | +| [Sub-Agent-Aware Routing](subagent_routing.md) | Delegated sub-agents should use a separate routing policy from the parent agent. | `passthrough`, `llm_classifier`, `stage_router`, or `composite` with `subagents` | | [Random Routing](random_routing.md) | You need a fixed traffic split for A/B tests, baselines, or cost experiments. | `random` | | [Fixed Model](#direct-model-routes) | Send every request to one target without a routing decision. | `passthrough` | diff --git a/docs/routing_algorithms/subagent_routing.md b/docs/routing_algorithms/subagent_routing.md index 618331236..b4760ce37 100644 --- a/docs/routing_algorithms/subagent_routing.md +++ b/docs/routing_algorithms/subagent_routing.md @@ -2,8 +2,8 @@ Sub-agent-aware routing leaves parent-agent traffic with its configured routing algorithm while routing delegated sub-agent work separately. It is available on -`passthrough`, `stage_router`, and `composite` routes through the optional -`subagents` table. +`passthrough`, `llm_classifier`, `stage_router`, and `composite` routes through +the optional `subagents` table. ```toml schema_version = 1