From 16d99ccd8a7006151e6057575b87e33e5b819cdc Mon Sep 17 00:00:00 2001 From: Hiroshi Morishige Date: Fri, 25 Sep 2026 15:06:19 +0900 Subject: [PATCH 1/2] feat(libsy): let classifier routes judge the latest user turn as the task Add `task_anchor` (`opening_task`, the default, or `latest_user_turn`) to capability and custom `llm_classifier` routes. With `latest_user_turn` the judge treats the newest ordinary user message as the task: alone without a window, or after the `recent_turn_window` messages that precede it, with tool pairs kept whole. `classify_trigger = "user_turn"` re-decides on every user message, but until now every decision was anchored to the conversation's opening task, so later jobs in a long coding session were judged against a request the user had finished long ago (#848). Escalation mode rejects the key like the other capability-only settings; stage, composite and the Python bindings keep the default. Signed-off-by: Hiroshi Morishige --- crates/libsy/src/algorithms/llm_class.rs | 240 ++++++++++++++++-- crates/libsy/src/lib.rs | 2 +- crates/switchyard-py/src/libsy_bindings.rs | 4 +- crates/switchyard-runner/src/algorithm.rs | 28 +- crates/switchyard-runner/src/config.rs | 28 ++ crates/switchyard-server/README.md | 1 + docs/reference/toml_schema.md | 2 + .../llm_classifier_routing.md | 15 ++ 8 files changed, 286 insertions(+), 34 deletions(-) diff --git a/crates/libsy/src/algorithms/llm_class.rs b/crates/libsy/src/algorithms/llm_class.rs index 488bc6585..e0438cfb6 100644 --- a/crates/libsy/src/algorithms/llm_class.rs +++ b/crates/libsy/src/algorithms/llm_class.rs @@ -148,18 +148,30 @@ fn window_start(tail: &[&Message], recent_turn_window: usize) -> usize { counted } +/// Ordinary user content. Decoders also use the user role for tool results, so a +/// message made only of tool results is never a task. +fn is_task_content(block: &ContentBlock) -> bool { + !matches!( + block, + ContentBlock::ToolCall(_) | ContentBlock::ToolResult(_) | ContentBlock::Reasoning { .. } + ) +} + +/// The ordinary user content of `message`, as the judge's task line. +fn task_only(message: &Message) -> Message { + Message { + role: Role::User, + content: message + .content + .iter() + .filter(|block| is_task_content(block)) + .cloned() + .collect(), + } +} + /// Keeps the opening task and the latest user follow-up when they differ. fn task_messages(messages: &[Message]) -> Vec { - // Decoders also use the user role for tool results. Select ordinary user content - // first, so a tool result cannot replace the opening task or latest follow-up. - let is_task_content = |block: &ContentBlock| { - !matches!( - block, - ContentBlock::ToolCall(_) - | ContentBlock::ToolResult(_) - | ContentBlock::Reasoning { .. } - ) - }; let mut user_messages = messages.iter().filter(|message| { message.role == Role::User && message.content.iter().any(is_task_content) }); @@ -169,30 +181,75 @@ fn task_messages(messages: &[Message]) -> Vec { [Some(opening_task), user_messages.next_back()] .into_iter() .flatten() - .map(|message| Message { - role: Role::User, - content: message - .content - .iter() - .filter(|block| is_task_content(block)) - .cloned() - .collect(), - }) + .map(task_only) .collect() } +/// Index of the newest message that carries ordinary user content. +fn latest_user_turn(messages: &[Message]) -> Option { + messages.iter().rposition(|message| { + message.role == Role::User && message.content.iter().any(is_task_content) + }) +} + +/// Keeps the latest user turn alone. +fn latest_task_message(messages: &[Message]) -> Vec { + latest_user_turn(messages) + .map(|task| vec![task_only(&messages[task])]) + .unwrap_or_default() +} + +/// Keeps the last `recent_turn_window` messages before the latest user turn, then that +/// turn last, so the judge reads the context first and the task it must route at the end. +/// +/// The window is counted over the messages before the task and keeps tool pairs whole +/// the same way [`trim_messages`] does for the trailing window. +fn trim_messages_before_latest(messages: &[Message], recent_turn_window: usize) -> Vec { + let is_instruction = |message: &Message| matches!(message.role, Role::System | Role::Developer); + let mut kept: Vec<&Message> = messages.iter().filter(|m| is_instruction(m)).collect(); + let Some(task) = latest_user_turn(messages) else { + return kept.into_iter().cloned().collect(); + }; + let head: Vec<&Message> = messages[..task] + .iter() + .filter(|m| !is_instruction(m)) + .collect(); + kept.extend(&head[window_start(&head, recent_turn_window)..]); + kept.push(&messages[task]); + kept.into_iter().cloned().collect() +} + +/// Which user message a capability or custom classifier treats as the task. +#[derive(Clone, Copy, Debug, Default, Deserialize, PartialEq, Eq)] +#[serde(rename_all = "snake_case")] +pub enum TaskAnchor { + /// The conversation's first ordinary user message; later user messages are + /// follow-ups to it. + #[default] + OpeningTask, + /// The newest ordinary user message. Suits `classify_trigger = "user_turn"` in + /// long sessions where each user message may open a different job. + LatestUserTurn, +} + /// Selects the task messages shown to capability and custom-schema classifiers. struct TaskInput { recent_turn_window: Option, + task_anchor: TaskAnchor, } impl ClassifierInput for TaskInput { fn build_messages(&self, _state: &State, request: &Request) -> Vec { - // The default preserves the whole-task anchor and latest user update. A - // configured window widens that to the surrounding conversation. - let mut messages = match self.recent_turn_window { - Some(window) => trim_messages(&request.llm_request.messages, window), - None => task_messages(&request.llm_request.messages), + // The anchor picks the task; a configured window widens the judge's view to the + // surrounding conversation: after the opening task, or before the latest turn. + let conversation = &request.llm_request.messages; + let mut messages = match (self.task_anchor, self.recent_turn_window) { + (TaskAnchor::OpeningTask, Some(window)) => trim_messages(conversation, window), + (TaskAnchor::OpeningTask, None) => task_messages(conversation), + (TaskAnchor::LatestUserTurn, Some(window)) => { + trim_messages_before_latest(conversation, window) + } + (TaskAnchor::LatestUserTurn, None) => latest_task_message(conversation), }; // Reasoning is provider-private and not required to classify the task. Some // upstreams also reject an unsigned reasoning item replayed without the @@ -318,6 +375,11 @@ pub struct TaskClassifierConfig { /// `Some(n)` widens that to the client instructions, the opening task, and /// the last `n` turns after it. pub recent_turn_window: Option, + /// Which user message is the task: the opening one (default) or the latest. + /// + /// With `LatestUserTurn`, `recent_turn_window` counts the messages before the + /// task instead of after it. + pub task_anchor: TaskAnchor, /// Prompt and verdict contract settings for the classifier judge. pub contract: ClassifierContractConfig, /// Maximum completion tokens available to the classifier verdict. @@ -338,6 +400,8 @@ struct TaskClassifierConfigWire { #[serde(default)] recent_turn_window: Option, #[serde(default)] + task_anchor: TaskAnchor, + #[serde(default)] prompt: Option, #[serde(default)] response_format_type: ClassifierResponseFormat, @@ -362,6 +426,7 @@ impl<'de> Deserialize<'de> for TaskClassifierConfig { classify_trigger: wire.classify_trigger, message_hash_fallback: wire.message_hash_fallback, recent_turn_window: wire.recent_turn_window, + task_anchor: wire.task_anchor, contract, max_output_tokens: wire.max_output_tokens, }) @@ -380,6 +445,7 @@ impl Default for TaskClassifierConfig { classify_trigger: ClassifyTrigger::default(), message_hash_fallback: false, recent_turn_window: None, + task_anchor: TaskAnchor::default(), contract: ClassifierContractConfig::default(), max_output_tokens: DEFAULT_JUDGE_MAX_OUTPUT_TOKENS, } @@ -464,6 +530,8 @@ pub struct CustomClassifierConfig { pub message_hash_fallback: bool, /// Trailing conversation turns shown to the classifier judge. pub recent_turn_window: Option, + /// Which user message is the task: the opening one (default) or the latest. + pub task_anchor: TaskAnchor, /// Maximum completion tokens available to the classifier verdict. pub max_output_tokens: u64, } @@ -482,6 +550,7 @@ impl CustomClassifierConfig { classify_trigger: ClassifyTrigger::default(), message_hash_fallback: false, recent_turn_window: None, + task_anchor: TaskAnchor::default(), max_output_tokens: DEFAULT_JUDGE_MAX_OUTPUT_TOKENS, } } @@ -637,6 +706,7 @@ impl LlmTaskClassifier { StructuredJudge::new( TaskInput { recent_turn_window: config.recent_turn_window, + task_anchor: config.task_anchor, }, contract, SerdeDecoder::new(), @@ -665,6 +735,7 @@ impl LlmTaskClassifier { classify_trigger, message_hash_fallback, recent_turn_window, + task_anchor, max_output_tokens, } = config; let contract = ClassifierContract::from_inner_schema(&prompt, response_schema)?; @@ -675,7 +746,10 @@ impl LlmTaskClassifier { }; let classifier: Arc> = Arc::new(JudgeClassifier::new( StructuredJudge::new( - TaskInput { recent_turn_window }, + TaskInput { + recent_turn_window, + task_anchor, + }, contract, JsonSchemaDecoder::new(), JudgeRuntimeConfig::new(max_output_tokens)?, @@ -1382,7 +1456,10 @@ mod tests { /// The no-window case is covered by `capability_judge_builds_a_structured_request`. fn capability_judge(recent_turn_window: Option) -> Result { Ok(StructuredJudge::new( - TaskInput { recent_turn_window }, + TaskInput { + recent_turn_window, + task_anchor: TaskAnchor::default(), + }, LlmTaskClassifier::load_capability_contract(&ClassifierContractConfig::default())?, SerdeDecoder::new(), JudgeRuntimeConfig::new(DEFAULT_JUDGE_MAX_OUTPUT_TOKENS)?, @@ -1470,6 +1547,7 @@ mod tests { }); let input = TaskInput { recent_turn_window: None, + task_anchor: TaskAnchor::default(), }; let mut request = Request { llm_request: LlmRequest { @@ -1646,6 +1724,7 @@ mod tests { let built = TaskInput { recent_turn_window: Some(10), + task_anchor: TaskAnchor::default(), } .build_messages(&State::default(), &request); @@ -1700,6 +1779,114 @@ mod tests { Ok(()) } + /// `latest_user_turn` judges the newest ordinary user message as the task. A user + /// message made only of tool results does not count, so an Anthropic-style tool + /// continuation never displaces the request the user actually typed. + #[test] + fn latest_user_turn_anchor_sends_the_newest_user_message_alone() { + let mut continuation = tool_result("call-9"); + continuation.role = Role::User; + let request = Request { + llm_request: LlmRequest { + messages: vec![ + Message::text(Role::System, "client instructions"), + Message::text(Role::User, "add caching"), + Message::text(Role::Assistant, "done"), + Message::text(Role::User, "now write the migration"), + continuation, + ], + ..LlmRequest::default() + }, + raw_request: None, + metadata: None, + }; + + let built = TaskInput { + recent_turn_window: None, + task_anchor: TaskAnchor::LatestUserTurn, + } + .build_messages(&State::default(), &request); + + assert_eq!(built.len(), 1, "{built:?}"); + assert_eq!( + built[0].text_content("\n").as_deref(), + Some("now write the migration") + ); + } + + /// With a window, the context precedes the task and tool pairs inside it stay whole. + #[test] + fn latest_user_turn_anchor_window_precedes_the_task_and_keeps_tool_pairs_whole() { + let request = Request { + llm_request: LlmRequest { + messages: vec![ + Message::text(Role::User, "add caching"), + Message::text(Role::Assistant, "plan"), + tool_call("call-1"), + tool_result("call-1"), + Message::text(Role::Assistant, "cached"), + Message::text(Role::User, "now write the migration"), + ], + ..LlmRequest::default() + }, + raw_request: None, + metadata: None, + }; + let build = |window: usize| { + TaskInput { + recent_turn_window: Some(window), + task_anchor: TaskAnchor::LatestUserTurn, + } + .build_messages(&State::default(), &request) + }; + let texts = |built: &[Message]| -> Vec { + built + .iter() + .filter_map(|message| message.text_content("\n")) + .collect() + }; + + // A window of 0 is the task alone, plus the trailing routing instruction. + assert_eq!( + texts(&build(0)), + vec![ + "now write the migration".to_string(), + TRAILING_ROUTING_INSTRUCTION.to_string(), + ] + ); + // A window of 1 counts back from the message before the task. + assert_eq!( + texts(&build(1)), + vec![ + "cached".to_string(), + "now write the migration".to_string(), + TRAILING_ROUTING_INSTRUCTION.to_string(), + ] + ); + // A window of 2 would open on the tool result, so it widens to include its call. + let built = build(2); + assert!(built.contains(&tool_call("call-1"))); + assert!(built.contains(&tool_result("call-1"))); + let two = texts(&built); + assert!(!two.contains(&"add caching".to_string()), "{two:?}"); + assert_eq!(two[two.len() - 2], "now write the migration"); + } + + #[test] + fn task_anchor_parses_and_defaults_to_the_opening_task() { + let parsed: TaskClassifierConfig = + serde_json::from_value(serde_json::json!({ "base_threshold": 0.5 })) + .expect("config without task_anchor parses"); + assert_eq!(parsed.task_anchor, TaskAnchor::OpeningTask); + + let parsed: TaskClassifierConfig = serde_json::from_value(serde_json::json!({ + "base_threshold": 0.5, + "task_anchor": "latest_user_turn", + })) + .expect("config with task_anchor parses"); + assert_eq!(parsed.task_anchor, TaskAnchor::LatestUserTurn); + } + #[test] fn capability_judge_builds_a_structured_request() -> Result<()> { let judge = capability_judge(None)?; @@ -1800,6 +1987,7 @@ mod tests { let judge: CapabilityJudge = StructuredJudge::new( TaskInput { recent_turn_window: None, + task_anchor: TaskAnchor::default(), }, contract, SerdeDecoder::new(), diff --git a/crates/libsy/src/lib.rs b/crates/libsy/src/lib.rs index a1f1ef022..21c61b8df 100644 --- a/crates/libsy/src/lib.rs +++ b/crates/libsy/src/lib.rs @@ -21,7 +21,7 @@ pub use algorithms::advisor_gate::{AdvisorGate, AdvisorGateConfig, GateTrigger}; pub use algorithms::composite::{CompositeRouter, CompositeRouterConfig}; pub use algorithms::llm_class::{ CustomClassifierConfig, CustomClassifierPolicy, LlmClassifierConfig, LlmTaskClassifier, - TaskClassifierConfig, + TaskAnchor, TaskClassifierConfig, }; pub use algorithms::noop::Noop; pub use algorithms::passthrough::Passthrough; diff --git a/crates/switchyard-py/src/libsy_bindings.rs b/crates/switchyard-py/src/libsy_bindings.rs index 23c90d7ad..f37e32038 100644 --- a/crates/switchyard-py/src/libsy_bindings.rs +++ b/crates/switchyard-py/src/libsy_bindings.rs @@ -17,7 +17,8 @@ use switchyard_libsy::{ CustomClassifierConfig, CustomClassifierPolicy, DeescalationConfig, EscalationJudgeConfig, HandoffNoteConfig, LibsyError as RustLibsyError, LlmClassifierConfig, LlmFallback, LlmTaskClassifier, Noop, PickerMode, Random, RoutingOutcome, RuntimeModels, StageRouter, - StageRouterConfig, Step as RustStep, StepStream, TaskClassifierConfig, ToolSemantics, + StageRouterConfig, Step as RustStep, StepStream, TaskAnchor, TaskClassifierConfig, + ToolSemantics, }; use switchyard_protocol::{ Category, LlmClientError, LlmResponse, LlmResponseStream, LlmResponseStreamEvent, Metadata, @@ -333,6 +334,7 @@ impl PyTaskClassifierConfig { classify_trigger: classify_trigger(session_affinity), message_hash_fallback, recent_turn_window, + task_anchor: TaskAnchor::default(), contract: classifier_contract(prompt, response_format_type)?, max_output_tokens, }, diff --git a/crates/switchyard-runner/src/algorithm.rs b/crates/switchyard-runner/src/algorithm.rs index e742d48ba..7579f1819 100644 --- a/crates/switchyard-runner/src/algorithm.rs +++ b/crates/switchyard-runner/src/algorithm.rs @@ -15,7 +15,7 @@ use libsy::{ CustomClassifierPolicy, EscalationJudgeConfig, GateTrigger, HandoffNoteConfig, LlmClassifierConfig, LlmFallback, LlmTaskClassifier, Noop, Passthrough, PickerMode, PlanExecute, PlanExecuteConfig, Random, StageRouter, StageRouterConfig, SubagentRouter, - SubagentRouterConfig, TaskClassifierConfig, ToolSemantics, + SubagentRouterConfig, TaskAnchor, TaskClassifierConfig, ToolSemantics, }; use serde::Deserialize; use switchyard_protocol::{Category, ModelId}; @@ -106,6 +106,7 @@ struct CapabilityClassifierRouteConfig { classify_trigger: ClassifyTrigger, message_hash_fallback: bool, recent_turn_window: Option, + task_anchor: TaskAnchor, prompt: Option, response_format_type: ClassifierResponseFormat, max_output_tokens: u64, @@ -132,6 +133,7 @@ struct CustomClassifierRouteConfig { classify_trigger: ClassifyTrigger, message_hash_fallback: bool, recent_turn_window: Option, + task_anchor: TaskAnchor, max_output_tokens: u64, } @@ -245,6 +247,9 @@ pub struct LlmClassifierRouteConfig { /// How many trailing turns the judge sees. Unset shows it the opening task /// and the latest user follow-up only. pub recent_turn_window: Option, + /// Which user message is the task: the opening one (default) or the latest. + /// With `latest_user_turn`, `recent_turn_window` counts the messages before it. + pub task_anchor: Option, /// Replaces the packaged judge prompt. Required in custom mode. pub prompt: Option, /// How the judge is asked for structured output. Use `json_object` when the @@ -527,6 +532,7 @@ impl StageClassifierConfig { classify_trigger: self.classify_trigger, message_hash_fallback: self.message_hash_fallback, recent_turn_window: self.recent_turn_window, + task_anchor: TaskAnchor::default(), contract: classifier_contract(self.prompt.as_deref()) .with_response_format_type(self.response_format_type), max_output_tokens: self.max_output_tokens, @@ -881,6 +887,7 @@ impl LlmClassifierRouteConfig { classify_trigger, message_hash_fallback, recent_turn_window, + task_anchor, prompt, response_format_type, max_output_tokens, @@ -936,6 +943,7 @@ impl LlmClassifierRouteConfig { classify_trigger: *classify_trigger, message_hash_fallback: *message_hash_fallback, recent_turn_window: *recent_turn_window, + task_anchor: task_anchor.unwrap_or_default(), prompt: prompt.clone(), response_format_type: *response_format_type, max_output_tokens: *max_output_tokens, @@ -956,11 +964,15 @@ impl LlmClassifierRouteConfig { "llm_classifier route {route_name} mode escalation cannot use classify_trigger" ))); } - if mode.is_some() - && (base_threshold.is_some() - || threshold_step.is_some() - || *message_hash_fallback - || recent_turn_window.is_some()) + // `task_anchor` is new, so no existing escalation configuration carries it: + // reject it even when the mode is implied by `escalation`. The older + // capability keys stay tolerated in that implicit form for compatibility. + if task_anchor.is_some() + || (mode.is_some() + && (base_threshold.is_some() + || threshold_step.is_some() + || *message_hash_fallback + || recent_turn_window.is_some())) { return Err(AlgorithmConfigError::new(format!( "llm_classifier route {route_name} mode escalation cannot use capability routing settings" @@ -1031,6 +1043,7 @@ impl LlmClassifierRouteConfig { classify_trigger: *classify_trigger, message_hash_fallback: *message_hash_fallback, recent_turn_window: *recent_turn_window, + task_anchor: task_anchor.unwrap_or_default(), max_output_tokens: *max_output_tokens, }, )) @@ -1117,6 +1130,7 @@ fn build_subagent_router_config( config.policy.into_libsy(), ); classifier_config.recent_turn_window = config.recent_turn_window; + classifier_config.task_anchor = config.task_anchor; classifier_config.max_output_tokens = config.max_output_tokens; let classifier = Arc::new( LlmTaskClassifier::new(LlmClassifierConfig::Custom { @@ -1234,6 +1248,7 @@ fn build_algorithm( classify_trigger: config.classify_trigger, message_hash_fallback: config.message_hash_fallback, recent_turn_window: config.recent_turn_window, + task_anchor: config.task_anchor, contract: classifier_contract(config.prompt.as_deref()) .with_response_format_type(config.response_format_type), max_output_tokens: config.max_output_tokens, @@ -1270,6 +1285,7 @@ fn build_algorithm( classifier_config.classify_trigger = config.classify_trigger; classifier_config.message_hash_fallback = config.message_hash_fallback; classifier_config.recent_turn_window = config.recent_turn_window; + classifier_config.task_anchor = config.task_anchor; classifier_config.max_output_tokens = config.max_output_tokens; LlmTaskClassifier::new(LlmClassifierConfig::Custom { default_target: config.default_target, diff --git a/crates/switchyard-runner/src/config.rs b/crates/switchyard-runner/src/config.rs index f37b4cda9..e52138571 100644 --- a/crates/switchyard-runner/src/config.rs +++ b/crates/switchyard-runner/src/config.rs @@ -1183,6 +1183,34 @@ new = ["send_message"] Ok(()) } + #[test] + fn task_anchor_is_a_capability_setting() -> RunnerResult<()> { + // Accepted on a capability route; unknown values fail at parse time. + runner_from_toml(&VALID_CONFIG.replace( + "base_threshold = 0.5", + "base_threshold = 0.5\ntask_anchor = \"latest_user_turn\"", + ))?; + assert!( + error_message(&VALID_CONFIG.replace( + "base_threshold = 0.5", + "base_threshold = 0.5\ntask_anchor = \"middle\"", + )) + .contains("unknown variant") + ); + + // Rejected on an escalation route, including the implicit form where the + // `escalation` table alone selects the mode: the key is new, so no existing + // configuration relies on it being ignored there. + assert!( + error_message(&VALID_CONFIG.replace( + "base_threshold = 0.5", + "base_threshold = 0.5\ntask_anchor = \"latest_user_turn\"\nescalation = { confirmations = 2 }", + )) + .contains("mode escalation cannot use capability routing settings") + ); + Ok(()) + } + #[test] fn a_target_reasoning_effort_parses_and_is_rejected_where_unsupported() -> RunnerResult<()> { let strong = "[targets.strong]\nid = \"strong/model\"\nllm_client = \"responses\""; diff --git a/crates/switchyard-server/README.md b/crates/switchyard-server/README.md index 152dbfe02..8bf23ad60 100644 --- a/crates/switchyard-server/README.md +++ b/crates/switchyard-server/README.md @@ -132,6 +132,7 @@ routes to `weak_target` or `strong_target`. Beyond the three targets it accepts | `threshold_step` | `0.0` | Finite, non-negative amount added once for uncertain or unmatched verdicts and twice for unsupported verdicts. `base_threshold + 2 * threshold_step` must be at most `1`. | | `classify_trigger` | `every_request` | When the judge runs. `every_request` judges every request including tool continuations, `user_turn` judges each new user message and holds that target across the tool calls between, `new_session` judges once and reuses that target for the session. | | `message_hash_fallback` | `false` | Extends affinity to clients that send no session header, keying on the first user message. Requires `classify_trigger = "new_session"` or `"user_turn"`. | +| `task_anchor` | `opening_task` | Which user message the judge treats as the task. `latest_user_turn` judges the newest user message, which suits `user_turn` in long sessions; `recent_turn_window` then counts the messages before it. | Session affinity retains a decision for the process lifetime, including a `strong_target` fallback produced while the judge was unreachable. `message_hash_fallback` keys on request diff --git a/docs/reference/toml_schema.md b/docs/reference/toml_schema.md index 8bf382a08..73b39d27d 100644 --- a/docs/reference/toml_schema.md +++ b/docs/reference/toml_schema.md @@ -243,6 +243,7 @@ Capability mode classifies before serving. See | `classify_trigger` | No | `every_request` | When the judge runs. `every_request` judges every request, tool continuations included. `user_turn` judges each new user message and retains that target across intervening tool calls only when requests carry a session ID; without a session ID, it behaves like `every_request`. `new_session` judges once and reuses that target for the session. | | `message_hash_fallback` | No | `false` | Retains the target against a hash of the first user message when a request carries no session ID. Requires `classify_trigger = "new_session"` or `"user_turn"`. | | `recent_turn_window` | No | unset | When unset, the judge sees the opening task and latest user follow-up, when present. When set, it also sees trailing turns. | +| `task_anchor` | No | `opening_task` | Which user message is the task. `opening_task` keeps the behavior above. `latest_user_turn` judges the newest ordinary user message; with `recent_turn_window`, the window then counts the messages before that turn and the turn is sent last. | | `prompt` | No | packaged prompt | Replaces the capability prompt. The packaged schema is sent separately as structured-output configuration. | Escalation mode serves the weak target first and judges the completed turn. See @@ -289,6 +290,7 @@ how one route chooses between more than two models. | `classify_trigger` | No | `every_request` | When the judge runs. `every_request` judges every request, tool continuations included. `user_turn` judges each new user message and retains that target across intervening tool calls only when requests carry a session ID; without a session ID, it behaves like `every_request`. `new_session` judges once and reuses that target for the session. | | `message_hash_fallback` | No | `false` | Retains the target against a hash of the first user message when a request carries no session ID. Requires `classify_trigger = "new_session"` or `"user_turn"`. | | `recent_turn_window` | No | unset | When unset, the judge sees the opening task and latest user follow-up, when present. When set, it also sees trailing turns. | +| `task_anchor` | No | `opening_task` | Which user message is the task. `opening_task` keeps the behavior above. `latest_user_turn` judges the newest ordinary user message; with `recent_turn_window`, the window then counts the messages before that turn and the turn is sent last. | The selected JSON label must name a configured group. A label naming a target rather than a group, or a group you did not configure, falls back to diff --git a/docs/routing_algorithms/llm_classifier_routing.md b/docs/routing_algorithms/llm_classifier_routing.md index de35015ff..7b7e1ed43 100644 --- a/docs/routing_algorithms/llm_classifier_routing.md +++ b/docs/routing_algorithms/llm_classifier_routing.md @@ -120,6 +120,7 @@ for the server merge behavior. | `base_threshold` | required | Lowest `p_solve` that routes a supported task to `weak_target`. Must be between `0` and `1`. | | `threshold_step` | `0.0` | Amount added for each boundary step. Must be finite and non-negative, and `base_threshold + 2 * threshold_step` must not exceed `1`. | | `recent_turn_window` | unset | When unset, the judge sees the opening user task and the latest user message when they differ. When set to `N`, it sees the opening user task and the last `N` conversation messages after that task. `0` keeps only the opening task. Client system and developer instructions are not shown to the judge. | +| `task_anchor` | `opening_task` | Which user message is the task. `opening_task` is the behavior described above. `latest_user_turn` judges the newest ordinary user message; without a window it is sent alone, and with `recent_turn_window = N` the judge sees the `N` messages before that turn and then the turn itself. | | `classify_trigger` | `every_request` | When the judge runs. `every_request` judges every request, tool continuations included. `user_turn` judges each new user message and holds that target across the tool calls between. `new_session` judges once and reuses that target for the session. | | `message_hash_fallback` | `false` | When session metadata is absent, keys affinity from the first user-message text. Requires `classify_trigger = "new_session"` or `"user_turn"`. | | `prompt` | packaged capability prompt | Replaces the classifier's system prompt. The packaged verdict schema and routing policy remain active. | @@ -257,6 +258,20 @@ Tool results are the agent continuing work the user already asked for, so they do not re-open the decision. A failed or unusable verdict keeps the current target. When no target has been selected yet, the next request is judged again. +By default each of those decisions is still anchored to the conversation's +opening task, with the new user message shown as a follow-up. In a long session +where later user messages open different jobs, set `task_anchor = +"latest_user_turn"` so the judge treats the message that triggered the decision +as the task, with `recent_turn_window` supplying the messages before it as +context: + +```toml +[routes.smart] +classify_trigger = "user_turn" +task_anchor = "latest_user_turn" +recent_turn_window = 4 +``` + `new_session` judges once and reuses that target for the rest of the session, including `strong_target` when it was selected as the fallback for an unusable verdict. There is no warmup period, and later requests skip the judge entirely. From a1415b8dcbb40c09861551cd9dbe2e50edb70154 Mon Sep 17 00:00:00 2001 From: Hiroshi Morishige Date: Fri, 25 Sep 2026 15:15:45 +0900 Subject: [PATCH 2/2] fix(libsy): send the latest-turn task without its tool results With `task_anchor = "latest_user_turn"` and a window, the task message was appended whole, so a tool result decoded into the same user message could reach the judge without the call that introduced it when the window did not cover that call. The task now carries its ordinary content only, the same projection the no-window path already applies; complete tool pairs stay in the surrounding context. Adds a test for the zero-window mixed message. Signed-off-by: Hiroshi Morishige --- crates/libsy/src/algorithms/llm_class.rs | 58 ++++++++++++++++++++++-- 1 file changed, 55 insertions(+), 3 deletions(-) diff --git a/crates/libsy/src/algorithms/llm_class.rs b/crates/libsy/src/algorithms/llm_class.rs index e0438cfb6..55f2b390a 100644 --- a/crates/libsy/src/algorithms/llm_class.rs +++ b/crates/libsy/src/algorithms/llm_class.rs @@ -203,7 +203,9 @@ fn latest_task_message(messages: &[Message]) -> Vec { /// turn last, so the judge reads the context first and the task it must route at the end. /// /// The window is counted over the messages before the task and keeps tool pairs whole -/// the same way [`trim_messages`] does for the trailing window. +/// the same way [`trim_messages`] does for the trailing window. The task itself is sent +/// with its ordinary content only, so a tool result decoded into the same user message +/// never travels without the call that introduced it. fn trim_messages_before_latest(messages: &[Message], recent_turn_window: usize) -> Vec { let is_instruction = |message: &Message| matches!(message.role, Role::System | Role::Developer); let mut kept: Vec<&Message> = messages.iter().filter(|m| is_instruction(m)).collect(); @@ -215,8 +217,11 @@ fn trim_messages_before_latest(messages: &[Message], recent_turn_window: usize) .filter(|m| !is_instruction(m)) .collect(); kept.extend(&head[window_start(&head, recent_turn_window)..]); - kept.push(&messages[task]); - kept.into_iter().cloned().collect() + let task_message = task_only(&messages[task]); + kept.into_iter() + .cloned() + .chain(std::iter::once(task_message)) + .collect() } /// Which user message a capability or custom classifier treats as the task. @@ -1872,6 +1877,53 @@ mod tests { assert_eq!(two[two.len() - 2], "now write the migration"); } + /// A tool result decoded into the same user message as the task stays out of the + /// task line, so a zero window never sends a result whose call is not in the window. + #[test] + fn latest_user_turn_anchor_sends_the_task_without_its_tool_results() { + let mut mixed = tool_result("call-1"); + mixed.role = Role::User; + mixed.content.push(ContentBlock::Text { + text: "now write the migration".to_string(), + }); + let request = Request { + llm_request: LlmRequest { + messages: vec![ + Message::text(Role::User, "add caching"), + tool_call("call-1"), + mixed, + ], + ..LlmRequest::default() + }, + raw_request: None, + metadata: None, + }; + + let built = TaskInput { + recent_turn_window: Some(0), + task_anchor: TaskAnchor::LatestUserTurn, + } + .build_messages(&State::default(), &request); + + assert!( + !built + .iter() + .flat_map(|message| &message.content) + .any(|block| matches!(block, ContentBlock::ToolResult(_))), + "{built:?}" + ); + assert_eq!( + built + .iter() + .filter_map(|message| message.text_content("\n")) + .collect::>(), + vec![ + "now write the migration".to_string(), + TRAILING_ROUTING_INSTRUCTION.to_string(), + ] + ); + } + #[test] fn task_anchor_parses_and_defaults_to_the_opening_task() { let parsed: TaskClassifierConfig =