Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
292 changes: 266 additions & 26 deletions crates/libsy/src/algorithms/llm_class.rs

Large diffs are not rendered by default.

2 changes: 1 addition & 1 deletion crates/libsy/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,7 @@ pub use algorithms::advisor_gate::{AdvisorGate, AdvisorGateConfig, GateTrigger};
pub use algorithms::composite::{CompositeRouter, CompositeRouterConfig};
pub use algorithms::llm_class::{
CustomClassifierConfig, CustomClassifierPolicy, LlmClassifierConfig, LlmTaskClassifier,
TaskClassifierConfig,
TaskAnchor, TaskClassifierConfig,
};
pub use algorithms::noop::Noop;
pub use algorithms::passthrough::Passthrough;
Expand Down
4 changes: 3 additions & 1 deletion crates/switchyard-py/src/libsy_bindings.rs
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,8 @@ use switchyard_libsy::{
CustomClassifierConfig, CustomClassifierPolicy, DeescalationConfig, EscalationJudgeConfig,
HandoffNoteConfig, LibsyError as RustLibsyError, LlmClassifierConfig, LlmFallback,
LlmTaskClassifier, Noop, PickerMode, Random, RoutingOutcome, RuntimeModels, StageRouter,
StageRouterConfig, Step as RustStep, StepStream, TaskClassifierConfig, ToolSemantics,
StageRouterConfig, Step as RustStep, StepStream, TaskAnchor, TaskClassifierConfig,
ToolSemantics,
};
use switchyard_protocol::{
Category, LlmClientError, LlmResponse, LlmResponseStream, LlmResponseStreamEvent, Metadata,
Expand Down Expand Up @@ -333,6 +334,7 @@ impl PyTaskClassifierConfig {
classify_trigger: classify_trigger(session_affinity),
message_hash_fallback,
recent_turn_window,
task_anchor: TaskAnchor::default(),
contract: classifier_contract(prompt, response_format_type)?,
max_output_tokens,
},
Expand Down
28 changes: 22 additions & 6 deletions crates/switchyard-runner/src/algorithm.rs
Original file line number Diff line number Diff line change
Expand Up @@ -15,7 +15,7 @@ use libsy::{
CustomClassifierPolicy, EscalationJudgeConfig, GateTrigger, HandoffNoteConfig,
LlmClassifierConfig, LlmFallback, LlmTaskClassifier, Noop, Passthrough, PickerMode,
PlanExecute, PlanExecuteConfig, Random, StageRouter, StageRouterConfig, SubagentRouter,
SubagentRouterConfig, TaskClassifierConfig, ToolSemantics,
SubagentRouterConfig, TaskAnchor, TaskClassifierConfig, ToolSemantics,
};
use serde::Deserialize;
use switchyard_protocol::{Category, ModelId};
Expand Down Expand Up @@ -106,6 +106,7 @@ struct CapabilityClassifierRouteConfig {
classify_trigger: ClassifyTrigger,
message_hash_fallback: bool,
recent_turn_window: Option<usize>,
task_anchor: TaskAnchor,
prompt: Option<String>,
response_format_type: ClassifierResponseFormat,
max_output_tokens: u64,
Expand All @@ -132,6 +133,7 @@ struct CustomClassifierRouteConfig {
classify_trigger: ClassifyTrigger,
message_hash_fallback: bool,
recent_turn_window: Option<usize>,
task_anchor: TaskAnchor,
max_output_tokens: u64,
}

Expand Down Expand Up @@ -245,6 +247,9 @@ pub struct LlmClassifierRouteConfig {
/// How many trailing turns the judge sees. Unset shows it the opening task
/// and the latest user follow-up only.
pub recent_turn_window: Option<usize>,
/// Which user message is the task: the opening one (default) or the latest.
/// With `latest_user_turn`, `recent_turn_window` counts the messages before it.
pub task_anchor: Option<TaskAnchor>,
/// Replaces the packaged judge prompt. Required in custom mode.
pub prompt: Option<String>,
/// How the judge is asked for structured output. Use `json_object` when the
Expand Down Expand Up @@ -527,6 +532,7 @@ impl StageClassifierConfig {
classify_trigger: self.classify_trigger,
message_hash_fallback: self.message_hash_fallback,
recent_turn_window: self.recent_turn_window,
task_anchor: TaskAnchor::default(),
contract: classifier_contract(self.prompt.as_deref())
.with_response_format_type(self.response_format_type),
max_output_tokens: self.max_output_tokens,
Expand Down Expand Up @@ -881,6 +887,7 @@ impl LlmClassifierRouteConfig {
classify_trigger,
message_hash_fallback,
recent_turn_window,
task_anchor,
prompt,
response_format_type,
max_output_tokens,
Expand Down Expand Up @@ -936,6 +943,7 @@ impl LlmClassifierRouteConfig {
classify_trigger: *classify_trigger,
message_hash_fallback: *message_hash_fallback,
recent_turn_window: *recent_turn_window,
task_anchor: task_anchor.unwrap_or_default(),
prompt: prompt.clone(),
response_format_type: *response_format_type,
max_output_tokens: *max_output_tokens,
Expand All @@ -956,11 +964,15 @@ impl LlmClassifierRouteConfig {
"llm_classifier route {route_name} mode escalation cannot use classify_trigger"
)));
}
if mode.is_some()
&& (base_threshold.is_some()
|| threshold_step.is_some()
|| *message_hash_fallback
|| recent_turn_window.is_some())
// `task_anchor` is new, so no existing escalation configuration carries it:
// reject it even when the mode is implied by `escalation`. The older
// capability keys stay tolerated in that implicit form for compatibility.
if task_anchor.is_some()
|| (mode.is_some()
&& (base_threshold.is_some()
|| threshold_step.is_some()
|| *message_hash_fallback
|| recent_turn_window.is_some()))
{
return Err(AlgorithmConfigError::new(format!(
"llm_classifier route {route_name} mode escalation cannot use capability routing settings"
Expand Down Expand Up @@ -1031,6 +1043,7 @@ impl LlmClassifierRouteConfig {
classify_trigger: *classify_trigger,
message_hash_fallback: *message_hash_fallback,
recent_turn_window: *recent_turn_window,
task_anchor: task_anchor.unwrap_or_default(),
max_output_tokens: *max_output_tokens,
},
))
Expand Down Expand Up @@ -1117,6 +1130,7 @@ fn build_subagent_router_config(
config.policy.into_libsy(),
);
classifier_config.recent_turn_window = config.recent_turn_window;
classifier_config.task_anchor = config.task_anchor;
classifier_config.max_output_tokens = config.max_output_tokens;
let classifier = Arc::new(
LlmTaskClassifier::new(LlmClassifierConfig::Custom {
Expand Down Expand Up @@ -1234,6 +1248,7 @@ fn build_algorithm(
classify_trigger: config.classify_trigger,
message_hash_fallback: config.message_hash_fallback,
recent_turn_window: config.recent_turn_window,
task_anchor: config.task_anchor,
contract: classifier_contract(config.prompt.as_deref())
.with_response_format_type(config.response_format_type),
max_output_tokens: config.max_output_tokens,
Expand Down Expand Up @@ -1270,6 +1285,7 @@ fn build_algorithm(
classifier_config.classify_trigger = config.classify_trigger;
classifier_config.message_hash_fallback = config.message_hash_fallback;
classifier_config.recent_turn_window = config.recent_turn_window;
classifier_config.task_anchor = config.task_anchor;
classifier_config.max_output_tokens = config.max_output_tokens;
LlmTaskClassifier::new(LlmClassifierConfig::Custom {
default_target: config.default_target,
Expand Down
28 changes: 28 additions & 0 deletions crates/switchyard-runner/src/config.rs
Original file line number Diff line number Diff line change
Expand Up @@ -1183,6 +1183,34 @@ new = ["send_message"]
Ok(())
}

#[test]
fn task_anchor_is_a_capability_setting() -> RunnerResult<()> {
// Accepted on a capability route; unknown values fail at parse time.
runner_from_toml(&VALID_CONFIG.replace(
"base_threshold = 0.5",
"base_threshold = 0.5\ntask_anchor = \"latest_user_turn\"",
))?;
assert!(
error_message(&VALID_CONFIG.replace(
"base_threshold = 0.5",
"base_threshold = 0.5\ntask_anchor = \"middle\"",
))
.contains("unknown variant")
);

// Rejected on an escalation route, including the implicit form where the
// `escalation` table alone selects the mode: the key is new, so no existing
// configuration relies on it being ignored there.
assert!(
error_message(&VALID_CONFIG.replace(
"base_threshold = 0.5",
"base_threshold = 0.5\ntask_anchor = \"latest_user_turn\"\nescalation = { confirmations = 2 }",
))
.contains("mode escalation cannot use capability routing settings")
);
Ok(())
}

#[test]
fn a_target_reasoning_effort_parses_and_is_rejected_where_unsupported() -> RunnerResult<()> {
let strong = "[targets.strong]\nid = \"strong/model\"\nllm_client = \"responses\"";
Expand Down
1 change: 1 addition & 0 deletions crates/switchyard-server/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -132,6 +132,7 @@ routes to `weak_target` or `strong_target`. Beyond the three targets it accepts
| `threshold_step` | `0.0` | Finite, non-negative amount added once for uncertain or unmatched verdicts and twice for unsupported verdicts. `base_threshold + 2 * threshold_step` must be at most `1`. |
| `classify_trigger` | `every_request` | When the judge runs. `every_request` judges every request including tool continuations, `user_turn` judges each new user message and holds that target across the tool calls between, `new_session` judges once and reuses that target for the session. |
| `message_hash_fallback` | `false` | Extends affinity to clients that send no session header, keying on the first user message. Requires `classify_trigger = "new_session"` or `"user_turn"`. |
| `task_anchor` | `opening_task` | Which user message the judge treats as the task. `latest_user_turn` judges the newest user message, which suits `user_turn` in long sessions; `recent_turn_window` then counts the messages before it. |

Session affinity retains a decision for the process lifetime, including a `strong_target`
fallback produced while the judge was unreachable. `message_hash_fallback` keys on request
Expand Down
2 changes: 2 additions & 0 deletions docs/reference/toml_schema.md
Original file line number Diff line number Diff line change
Expand Up @@ -243,6 +243,7 @@ Capability mode classifies before serving. See
| `classify_trigger` | No | `every_request` | When the judge runs. `every_request` judges every request, tool continuations included. `user_turn` judges each new user message and retains that target across intervening tool calls only when requests carry a session ID; without a session ID, it behaves like `every_request`. `new_session` judges once and reuses that target for the session. |
| `message_hash_fallback` | No | `false` | Retains the target against a hash of the first user message when a request carries no session ID. Requires `classify_trigger = "new_session"` or `"user_turn"`. |
| `recent_turn_window` | No | unset | When unset, the judge sees the opening task and latest user follow-up, when present. When set, it also sees trailing turns. |
| `task_anchor` | No | `opening_task` | Which user message is the task. `opening_task` keeps the behavior above. `latest_user_turn` judges the newest ordinary user message; with `recent_turn_window`, the window then counts the messages before that turn and the turn is sent last. |
| `prompt` | No | packaged prompt | Replaces the capability prompt. The packaged schema is sent separately as structured-output configuration. |

Escalation mode serves the weak target first and judges the completed turn. See
Expand Down Expand Up @@ -289,6 +290,7 @@ how one route chooses between more than two models.
| `classify_trigger` | No | `every_request` | When the judge runs. `every_request` judges every request, tool continuations included. `user_turn` judges each new user message and retains that target across intervening tool calls only when requests carry a session ID; without a session ID, it behaves like `every_request`. `new_session` judges once and reuses that target for the session. |
| `message_hash_fallback` | No | `false` | Retains the target against a hash of the first user message when a request carries no session ID. Requires `classify_trigger = "new_session"` or `"user_turn"`. |
| `recent_turn_window` | No | unset | When unset, the judge sees the opening task and latest user follow-up, when present. When set, it also sees trailing turns. |
| `task_anchor` | No | `opening_task` | Which user message is the task. `opening_task` keeps the behavior above. `latest_user_turn` judges the newest ordinary user message; with `recent_turn_window`, the window then counts the messages before that turn and the turn is sent last. |

The selected JSON label must name a configured group. A label naming a target
rather than a group, or a group you did not configure, falls back to
Expand Down
15 changes: 15 additions & 0 deletions docs/routing_algorithms/llm_classifier_routing.md
Original file line number Diff line number Diff line change
Expand Up @@ -120,6 +120,7 @@ for the server merge behavior.
| `base_threshold` | required | Lowest `p_solve` that routes a supported task to `weak_target`. Must be between `0` and `1`. |
| `threshold_step` | `0.0` | Amount added for each boundary step. Must be finite and non-negative, and `base_threshold + 2 * threshold_step` must not exceed `1`. |
| `recent_turn_window` | unset | When unset, the judge sees the opening user task and the latest user message when they differ. When set to `N`, it sees the opening user task and the last `N` conversation messages after that task. `0` keeps only the opening task. Client system and developer instructions are not shown to the judge. |
| `task_anchor` | `opening_task` | Which user message is the task. `opening_task` is the behavior described above. `latest_user_turn` judges the newest ordinary user message; without a window it is sent alone, and with `recent_turn_window = N` the judge sees the `N` messages before that turn and then the turn itself. |
| `classify_trigger` | `every_request` | When the judge runs. `every_request` judges every request, tool continuations included. `user_turn` judges each new user message and holds that target across the tool calls between. `new_session` judges once and reuses that target for the session. |
| `message_hash_fallback` | `false` | When session metadata is absent, keys affinity from the first user-message text. Requires `classify_trigger = "new_session"` or `"user_turn"`. |
| `prompt` | packaged capability prompt | Replaces the classifier's system prompt. The packaged verdict schema and routing policy remain active. |
Expand Down Expand Up @@ -257,6 +258,20 @@ Tool results are the agent continuing work the user already asked for, so they
do not re-open the decision. A failed or unusable verdict keeps the current
target. When no target has been selected yet, the next request is judged again.

By default each of those decisions is still anchored to the conversation's
opening task, with the new user message shown as a follow-up. In a long session
where later user messages open different jobs, set `task_anchor =
"latest_user_turn"` so the judge treats the message that triggered the decision
as the task, with `recent_turn_window` supplying the messages before it as
context:

```toml
[routes.smart]
classify_trigger = "user_turn"
task_anchor = "latest_user_turn"
recent_turn_window = 4
```

`new_session` judges once and reuses that target for the rest of the session,
including `strong_target` when it was selected as the fallback for an unusable
verdict. There is no warmup period, and later requests skip the judge entirely.
Expand Down
Loading