From c1aba3d021047300ca8068c97e6fe1e028bb7f1c Mon Sep 17 00:00:00 2001 From: Graham King Date: Wed, 30 Sep 2026 12:38:56 -0400 Subject: [PATCH] fix(translation): preserve native Anthropic thinking settings `AnthropicMessagesCodec::encode_request` (crates/switchyard-translation/src/codecs/anthropic/buffered.rs:294-308) now re-emits the caller's native thinking object on the reconstructed path, and only falls back to `{"type": "adaptive"}` when effort is set and no native thinking exists. Previously, any `reasoning.effort` clobbered thinking with adaptive, so reconstruction disagreed with verbatim replay (a caller sending thinking: disabled + output_config.effort replayed as disabled but reconstructed as adaptive). GLM 5.3's review: > Real bug, correctly fixed. README claims match the code, including omit-before-merge ordering and reasoning_effort only applying to OpenAI backends. Fixes: https://linear.app/nvidia/issue/SWITCH-1490/ Assisted-by: Pi:GPT 6 Astra medium Reviewed-by: Pi:GLM 5.3 high Signed-off-by: Graham King --- crates/libsy-llm-client/README.md | 6 ++++ .../src/codecs/anthropic/buffered.rs | 10 +++++- .../tests/request_translation.rs | 32 +++++++++++++++++++ 3 files changed, 47 insertions(+), 1 deletion(-) diff --git a/crates/libsy-llm-client/README.md b/crates/libsy-llm-client/README.md index 34ec8a989..8b0ef430d 100644 --- a/crates/libsy-llm-client/README.md +++ b/crates/libsy-llm-client/README.md @@ -248,6 +248,12 @@ fn build_multi_format_client( and `anthropic-version`. Header names are case-insensitive. - Per-target top-level request defaults go in `HttpBackendConfig::extra_body`. The merge is shallow and fields already present in the request take precedence. +- Anthropic reconstruction preserves the caller's native `thinking` settings, including + disabled thinking and manual budgets. `output_config.effort` is kept separately. + Effort implies adaptive thinking only when native `thinking` is absent. + To replace either top-level field for a target, list it in `omit_body_fields` and + supply its replacement in `extra_body`. Omission runs before defaults are merged. + Target `reasoning_effort` overrides apply to OpenAI backends. - `HttpBackendConfig::max_retries` controls additional attempts after retryable transport failures, timeouts, HTTP 408/429, and 5xx responses. Buffered body transport failures are retried; streaming body failures are not replayed after diff --git a/crates/switchyard-translation/src/codecs/anthropic/buffered.rs b/crates/switchyard-translation/src/codecs/anthropic/buffered.rs index c6cdcfba5..231d97b84 100644 --- a/crates/switchyard-translation/src/codecs/anthropic/buffered.rs +++ b/crates/switchyard-translation/src/codecs/anthropic/buffered.rs @@ -291,8 +291,16 @@ impl FormatCodec for AnthropicMessagesCodec { if request.stream { body.insert("stream".to_string(), Value::Bool(true)); } + // Native thinking controls the mode independently of output effort. + // Other codecs store their own provider's reasoning object in `raw`. + if is_anthropic_request(request) + && let Some(thinking) = &request.reasoning.raw + { + body.insert("thinking".to_string(), thinking.clone()); + } if let Some(effort) = &request.reasoning.effort { - body.insert("thinking".to_string(), json!({"type": "adaptive"})); + body.entry("thinking".to_string()) + .or_insert_with(|| json!({"type": "adaptive"})); body.insert("output_config".to_string(), json!({"effort": effort})); } if let Some(response_format) = &request.output.response_format diff --git a/crates/switchyard-translation/tests/request_translation.rs b/crates/switchyard-translation/tests/request_translation.rs index 2af80f9d3..eceb54693 100644 --- a/crates/switchyard-translation/tests/request_translation.rs +++ b/crates/switchyard-translation/tests/request_translation.rs @@ -484,6 +484,38 @@ fn anthropic_thinking_to_responses_uses_normalized_effort() -> TestResult { Ok(()) } +#[test] +fn anthropic_reconstruction_preserves_thinking() -> TestResult { + let engine = TranslationEngine::default(); + let policy = normalized_policy(); + for (thinking, effort) in [ + (json!({"type": "disabled"}), Some("high")), + (json!({"type": "enabled", "budget_tokens": 2048}), None), + (json!({"type": "adaptive"}), Some("high")), + ] { + let mut body = json!({ + "model": "caller", "max_tokens": 4096, + "messages": [{"role": "user", "content": "hi"}], + "thinking": thinking + }); + if let Some(effort) = effort { + body["output_config"] = json!({"effort": effort}); + } + let mut request = engine + .decode_request(WireFormat::AnthropicMessages, &body, &policy)? + .request; + prepare_request_for_target(&mut request, &"target/model".into(), Some("target prompt")); + let output = engine + .encode_request(WireFormat::AnthropicMessages, &request, &policy)? + .body; + + body["model"] = json!("target/model"); + body["system"] = json!("target prompt"); + assert_eq!(output, body); + } + Ok(()) +} + #[test] fn anthropic_target_prompt_preserves_native_request_fields() -> TestResult { let engine = TranslationEngine::default();