diff --git a/crates/libsy-llm-client/README.md b/crates/libsy-llm-client/README.md index 34ec8a989..8b0ef430d 100644 --- a/crates/libsy-llm-client/README.md +++ b/crates/libsy-llm-client/README.md @@ -248,6 +248,12 @@ fn build_multi_format_client( and `anthropic-version`. Header names are case-insensitive. - Per-target top-level request defaults go in `HttpBackendConfig::extra_body`. The merge is shallow and fields already present in the request take precedence. +- Anthropic reconstruction preserves the caller's native `thinking` settings, including + disabled thinking and manual budgets. `output_config.effort` is kept separately. + Effort implies adaptive thinking only when native `thinking` is absent. + To replace either top-level field for a target, list it in `omit_body_fields` and + supply its replacement in `extra_body`. Omission runs before defaults are merged. + Target `reasoning_effort` overrides apply to OpenAI backends. - `HttpBackendConfig::max_retries` controls additional attempts after retryable transport failures, timeouts, HTTP 408/429, and 5xx responses. Buffered body transport failures are retried; streaming body failures are not replayed after diff --git a/crates/switchyard-translation/src/codecs/anthropic/buffered.rs b/crates/switchyard-translation/src/codecs/anthropic/buffered.rs index c6cdcfba5..231d97b84 100644 --- a/crates/switchyard-translation/src/codecs/anthropic/buffered.rs +++ b/crates/switchyard-translation/src/codecs/anthropic/buffered.rs @@ -291,8 +291,16 @@ impl FormatCodec for AnthropicMessagesCodec { if request.stream { body.insert("stream".to_string(), Value::Bool(true)); } + // Native thinking controls the mode independently of output effort. + // Other codecs store their own provider's reasoning object in `raw`. + if is_anthropic_request(request) + && let Some(thinking) = &request.reasoning.raw + { + body.insert("thinking".to_string(), thinking.clone()); + } if let Some(effort) = &request.reasoning.effort { - body.insert("thinking".to_string(), json!({"type": "adaptive"})); + body.entry("thinking".to_string()) + .or_insert_with(|| json!({"type": "adaptive"})); body.insert("output_config".to_string(), json!({"effort": effort})); } if let Some(response_format) = &request.output.response_format diff --git a/crates/switchyard-translation/tests/request_translation.rs b/crates/switchyard-translation/tests/request_translation.rs index 2af80f9d3..eceb54693 100644 --- a/crates/switchyard-translation/tests/request_translation.rs +++ b/crates/switchyard-translation/tests/request_translation.rs @@ -484,6 +484,38 @@ fn anthropic_thinking_to_responses_uses_normalized_effort() -> TestResult { Ok(()) } +#[test] +fn anthropic_reconstruction_preserves_thinking() -> TestResult { + let engine = TranslationEngine::default(); + let policy = normalized_policy(); + for (thinking, effort) in [ + (json!({"type": "disabled"}), Some("high")), + (json!({"type": "enabled", "budget_tokens": 2048}), None), + (json!({"type": "adaptive"}), Some("high")), + ] { + let mut body = json!({ + "model": "caller", "max_tokens": 4096, + "messages": [{"role": "user", "content": "hi"}], + "thinking": thinking + }); + if let Some(effort) = effort { + body["output_config"] = json!({"effort": effort}); + } + let mut request = engine + .decode_request(WireFormat::AnthropicMessages, &body, &policy)? + .request; + prepare_request_for_target(&mut request, &"target/model".into(), Some("target prompt")); + let output = engine + .encode_request(WireFormat::AnthropicMessages, &request, &policy)? + .body; + + body["model"] = json!("target/model"); + body["system"] = json!("target prompt"); + assert_eq!(output, body); + } + Ok(()) +} + #[test] fn anthropic_target_prompt_preserves_native_request_fields() -> TestResult { let engine = TranslationEngine::default();