Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions crates/libsy-llm-client/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -248,6 +248,12 @@ fn build_multi_format_client(
and `anthropic-version`. Header names are case-insensitive.
- Per-target top-level request defaults go in `HttpBackendConfig::extra_body`.
The merge is shallow and fields already present in the request take precedence.
- Anthropic reconstruction preserves the caller's native `thinking` settings, including
disabled thinking and manual budgets. `output_config.effort` is kept separately.
Effort implies adaptive thinking only when native `thinking` is absent.
To replace either top-level field for a target, list it in `omit_body_fields` and
supply its replacement in `extra_body`. Omission runs before defaults are merged.
Target `reasoning_effort` overrides apply to OpenAI backends.
- `HttpBackendConfig::max_retries` controls additional attempts after retryable
transport failures, timeouts, HTTP 408/429, and 5xx responses. Buffered body
transport failures are retried; streaming body failures are not replayed after
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -291,8 +291,16 @@ impl FormatCodec for AnthropicMessagesCodec {
if request.stream {
body.insert("stream".to_string(), Value::Bool(true));
}
// Native thinking controls the mode independently of output effort.
// Other codecs store their own provider's reasoning object in `raw`.
if is_anthropic_request(request)
&& let Some(thinking) = &request.reasoning.raw
{
body.insert("thinking".to_string(), thinking.clone());
}
if let Some(effort) = &request.reasoning.effort {
body.insert("thinking".to_string(), json!({"type": "adaptive"}));
body.entry("thinking".to_string())
.or_insert_with(|| json!({"type": "adaptive"}));
body.insert("output_config".to_string(), json!({"effort": effort}));
}
if let Some(response_format) = &request.output.response_format
Expand Down
32 changes: 32 additions & 0 deletions crates/switchyard-translation/tests/request_translation.rs
Original file line number Diff line number Diff line change
Expand Up @@ -484,6 +484,38 @@ fn anthropic_thinking_to_responses_uses_normalized_effort() -> TestResult {
Ok(())
}

#[test]
fn anthropic_reconstruction_preserves_thinking() -> TestResult {
let engine = TranslationEngine::default();
let policy = normalized_policy();
for (thinking, effort) in [
(json!({"type": "disabled"}), Some("high")),
(json!({"type": "enabled", "budget_tokens": 2048}), None),
(json!({"type": "adaptive"}), Some("high")),
] {
let mut body = json!({
"model": "caller", "max_tokens": 4096,
"messages": [{"role": "user", "content": "hi"}],
"thinking": thinking
});
if let Some(effort) = effort {
body["output_config"] = json!({"effort": effort});
}
let mut request = engine
.decode_request(WireFormat::AnthropicMessages, &body, &policy)?
.request;
prepare_request_for_target(&mut request, &"target/model".into(), Some("target prompt"));
let output = engine
.encode_request(WireFormat::AnthropicMessages, &request, &policy)?
.body;

body["model"] = json!("target/model");
body["system"] = json!("target prompt");
assert_eq!(output, body);
}
Ok(())
}

#[test]
fn anthropic_target_prompt_preserves_native_request_fields() -> TestResult {
let engine = TranslationEngine::default();
Expand Down
Loading