Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

1 change: 1 addition & 0 deletions crates/libsy-llm-client/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -29,6 +29,7 @@ parking_lot.workspace = true
http.workspace = true
httpdate.workspace = true
serde_json.workspace = true
serde.workspace = true
tokio.workspace = true
tracing.workspace = true
tracing-opentelemetry.workspace = true
Expand Down
4 changes: 4 additions & 0 deletions crates/libsy-llm-client/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -252,6 +252,10 @@ fn build_multi_format_client(
transport failures, timeouts, HTTP 408/429, and 5xx responses. Buffered body
transport failures are retried; streaming body failures are not replayed after
the response has been returned.
- `ModelConfig::with_responses_reasoning` controls Responses reasoning replay.
Every Responses model defaults to `PreserveEncrypted`: plaintext is removed,
encrypted provider state is retained. Set `Drop` explicitly for a backend
that cannot consume encrypted reasoning. Messages and tool history are retained.
- `HttpBackendConfig::timeout` bounds one complete response, including retries,
retry delays, and every stream read. Expiry returns `LlmClientError::Timeout`,
either from the call or from the returned stream, which then ends. `None` leaves
Expand Down
220 changes: 220 additions & 0 deletions crates/libsy-llm-client/src/client.rs
Original file line number Diff line number Diff line change
Expand Up @@ -63,6 +63,7 @@ pub struct ModelConfig {
model_name: ModelId,
default_backend: Backend,
other_backends: Option<Vec<Backend>>,
responses_reasoning: crate::ResponsesReasoningPolicy,
}

impl ModelConfig {
Expand All @@ -77,8 +78,16 @@ impl ModelConfig {
model_name: model_name.into(),
default_backend,
other_backends,
responses_reasoning: crate::ResponsesReasoningPolicy::default(),
}
}

/// Sets how Responses reasoning items are replayed to this model.
#[must_use]
pub fn with_responses_reasoning(mut self, policy: crate::ResponsesReasoningPolicy) -> Self {
self.responses_reasoning = policy;
self
}
}

/// A model-bearing provider operation outside the normal completion endpoint.
Expand Down Expand Up @@ -262,6 +271,13 @@ impl TranslatingLlmClient {
}
omit_configured_body_fields(&mut body, backend.omit_body_fields());
merge_extra_body(&mut body, backend.extra_body());
if matches!(backend, Backend::OpenAiResponses(_)) {
self.model_to_config
.get(model)
.map(|config| config.responses_reasoning)
.unwrap_or_default()
.normalize(&mut body);
Comment thread
deepujain marked this conversation as resolved.
}
// After the merge on purpose: the effort override must win over both the caller's
// value and any `reasoning` default a target set through `extra_body`.
apply_reasoning_effort(&mut body, backend);
Expand Down Expand Up @@ -1446,6 +1462,13 @@ mod tests {
)]
}

fn local_responses_map(base_url: &str) -> Vec<ModelConfig> {
vec![
ModelConfig::new("local", Backend::OpenAiResponses(config(base_url)), None)
.with_responses_reasoning(crate::ResponsesReasoningPolicy::Drop),
]
}

fn chat_map_with_retries(base_url: &str, max_retries: u32) -> Vec<ModelConfig> {
vec![ModelConfig::new(
"gpt",
Expand Down Expand Up @@ -2192,6 +2215,203 @@ mod tests {
Ok(())
}

#[tokio::test]
async fn responses_policy_normalizes_input_replaced_by_target_defaults()
-> std::result::Result<(), Box<dyn Error + Sync + Send + 'static>> {
for (policy, expected_reasoning) in [
(crate::ResponsesReasoningPolicy::PreserveEncrypted, 1),
(crate::ResponsesReasoningPolicy::Drop, 0),
] {
let server = MockServer::start().await;
Mock::given(method("POST"))
.and(path("/v1/responses"))
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
"id": "resp_1", "object": "response", "model": "gpt",
"status": "completed", "output": [],
"usage": {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2}
})))
.expect(1)
.mount(&server)
.await;
let mut backend = config(&format!("{}/v1", server.uri()));
backend.omit_body_fields.insert("input".to_string());
backend.extra_body.insert("input".to_string(), json!([
{"type": "message", "role": "user", "content": "replacement"},
{"type": "reasoning", "encrypted_content": "opaque", "content": [{"type": "reasoning_text", "text": "private"}]},
{"type": "reasoning", "encrypted_content": null, "content": [{"type": "reasoning_text", "text": "local"}]}
]));
let client = TranslatingLlmClient::new(&[ModelConfig::new(
"gpt",
Backend::OpenAiResponses(backend),
None,
)
.with_responses_reasoning(policy)])?;
client
.call_rewrite_model_raw(
json!({"model": "gpt", "input": "original"}),
None,
Some(&ModelId::from("gpt")),
WireFormat::OpenAiResponses,
)
.await?;
let requests = server.received_requests().await.expect("recorded requests");
let body: Value = serde_json::from_slice(&requests[0].body)?;
let input = body["input"].as_array().expect("replacement input");
assert_eq!(input[0]["content"], "replacement");
let reasoning: Vec<_> = input
.iter()
.filter(|item| item["type"] == "reasoning")
.collect();
assert_eq!(reasoning.len(), expected_reasoning);
for item in reasoning {
assert_eq!(item["encrypted_content"], "opaque");
assert_eq!(item["content"], json!([]));
}
}
Ok(())
}

// Strict Responses upstreams retain encrypted state without replaying plaintext.
#[tokio::test]
async fn responses_requests_drop_unsigned_reasoning_items()
-> std::result::Result<(), Box<dyn Error + Sync + Send + 'static>> {
let server = MockServer::start().await;
Mock::given(method("POST"))
.and(path("/v1/responses"))
.and(|request: &wiremock::Request| {
let body: Value = serde_json::from_slice(&request.body).unwrap_or(Value::Null);
let Some(input) = body.get("input").and_then(Value::as_array) else {
return false;
};
let reasoning: Vec<&Value> = input
.iter()
.filter(|item| item.get("type").and_then(Value::as_str) == Some("reasoning"))
.collect();
reasoning.len() == 1
&& reasoning[0]
.get("encrypted_content")
.and_then(Value::as_str)
== Some("encrypted")
&& reasoning[0].get("content") == Some(&json!([]))
&& input.iter().any(|item| {
item.get("type").and_then(Value::as_str) == Some("function_call")
})
&& input.iter().any(|item| {
item.get("type").and_then(Value::as_str) == Some("function_call_output")
})
})
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
"id": "resp_1",
"object": "response",
"model": "gpt",
"status": "completed",
"output": [{
"type": "message",
"role": "assistant",
"content": [{"type": "output_text", "text": "ok"}]
}],
"usage": {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2}
})))
.mount(&server)
.await;

let client = TranslatingLlmClient::new(&responses_map(&format!("{}/v1", server.uri())))?;
let raw = json!({
"model": "client-facing",
"input": [
{"type": "message", "role": "user", "content": "inspect the repo"},
{
"type": "reasoning",
"content": [{"type": "reasoning_text", "text": "private local reasoning"}],
"encrypted_content": ""
},
{"type": "function_call", "call_id": "call_1", "name": "shell", "arguments": "{}"},
{"type": "function_call_output", "call_id": "call_1", "output": "ok"},
{
"type": "reasoning",
"content": [{"type": "reasoning_text", "text": "must not be replayed"}],
"encrypted_content": "encrypted"
}
]
});

client
.call_rewrite_model_raw(
raw,
None,
Some(&ModelId::from("gpt")),
WireFormat::OpenAiResponses,
)
.await?;
Ok(())
}

// Encrypted hosted reasoning is opaque to a local Responses backend. Dropping
// it avoids llama.cpp rejecting a missing or empty `content` array while
// retaining the conversation and tool-call history it can consume.
#[tokio::test]
async fn local_responses_requests_drop_encrypted_reasoning_items()
-> std::result::Result<(), Box<dyn Error + Sync + Send + 'static>> {
let server = MockServer::start().await;
Mock::given(method("POST"))
.and(path("/v1/responses"))
.and(|request: &wiremock::Request| {
let body: Value = serde_json::from_slice(&request.body).unwrap_or(Value::Null);
let Some(input) = body.get("input").and_then(Value::as_array) else {
return false;
};
input
.iter()
.all(|item| item.get("type").and_then(Value::as_str) != Some("reasoning"))
&& input.iter().any(|item| {
item.get("type").and_then(Value::as_str) == Some("function_call")
})
&& input.iter().any(|item| {
item.get("type").and_then(Value::as_str) == Some("function_call_output")
})
})
.respond_with(ResponseTemplate::new(200).set_body_json(json!({
"id": "resp_local",
"object": "response",
"model": "local",
"status": "completed",
"output": [{
"type": "message",
"role": "assistant",
"content": [{"type": "output_text", "text": "ok"}]
}],
"usage": {"input_tokens": 1, "output_tokens": 1, "total_tokens": 2}
})))
.mount(&server)
.await;

let client =
TranslatingLlmClient::new(&local_responses_map(&format!("{}/v1", server.uri())))?;
let raw = json!({
"model": "client-facing",
"input": [
{"type": "message", "role": "user", "content": [{"type": "input_text", "text": "inspect"}]},
{
"type": "reasoning",
"encrypted_content": "opaque-provider-reasoning"
},
{"type": "function_call", "call_id": "call_1", "name": "shell", "arguments": "{}"},
{"type": "function_call_output", "call_id": "call_1", "output": "ok"},
{"type": "message", "role": "user", "content": [{"type": "input_text", "text": "continue"}]}
]
});

client
.call_rewrite_model_raw(
raw,
None,
Some(&ModelId::from("local")),
WireFormat::OpenAiResponses,
)
.await?;
Ok(())
}

// A router can serve earlier turns from an OpenAI target and later turns from
// an Anthropic one, so the Anthropic leg must drop OpenAI-only fields the
// caller keeps sending or the upstream rejects the whole request.
Expand Down
2 changes: 2 additions & 0 deletions crates/libsy-llm-client/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -23,13 +23,15 @@ pub mod metrics;
mod observability;
mod observation;
pub mod raw;
mod responses_reasoning;
pub mod run;

pub use backend::{Backend, DEFAULT_MAX_RETRIES, HttpBackendConfig};
pub use client::{AuxiliaryOperation, ModelConfig, TranslatingLlmClient};
pub use error::{LlmClientError, Result};
pub use observation::{LlmCallObservation, RunObservation, RunObserver};
pub use raw::RawResponse;
pub use responses_reasoning::ResponsesReasoningPolicy;
pub use run::{ClientRouter, decide, run};
pub use switchyard_translation::RawEventStream;

Expand Down
Loading
Loading