From 5b238491da463c1b8cd18c3c8365003309bf091d Mon Sep 17 00:00:00 2001 From: Colin McNamara Date: Tue, 29 Sep 2026 21:10:44 -0500 Subject: [PATCH 1/2] fix(translation): return HTTP 502 for buffered Chat abort and error finish reasons Signed-off-by: Colin McNamara --- crates/switchyard-server/tests/server.rs | 95 +++++++++++++++++++ .../src/codecs/openai_chat/buffered.rs | 12 +++ .../tests/response_translation.rs | 25 +++++ 3 files changed, 132 insertions(+) diff --git a/crates/switchyard-server/tests/server.rs b/crates/switchyard-server/tests/server.rs index cd21d3c91..2f9976853 100644 --- a/crates/switchyard-server/tests/server.rs +++ b/crates/switchyard-server/tests/server.rs @@ -342,6 +342,22 @@ async fn upstream_chat( .into_response(); } + // A generation that did not complete (`abort` from vLLM or SGLang, `error` from OpenRouter). + if let Some(reason @ ("abort" | "error")) = model.strip_prefix("model/finish-") { + return Json(json!({ + "id": "chatcmpl-finish", + "object": "chat.completion", + "model": model, + "choices": [{ + "index": 0, + "message": {"role": "assistant", "content": "partial"}, + "finish_reason": reason + }], + "usage": {"prompt_tokens": 3, "completion_tokens": 1, "total_tokens": 4} + })) + .into_response(); + } + // Buffered tool call, the non-streaming counterpart of the branch above. if prompt == "mcp-tool-call" { let called = body["tool_choice"]["function"]["name"] @@ -1134,6 +1150,85 @@ async fn failed_responses_return_errors_and_try_fallback_across_endpoints() -> T Ok(()) } +// A Chat backend that ends a generation with `abort` or `error` did not complete +// it: every client path gets a 502, and a fallback route moves on. +#[tokio::test] +async fn chat_abort_and_error_finish_reasons_return_errors_and_try_fallback() -> TestResult { + let upstream = MockUpstream::start().await?; + let requests = [ + ( + "/v1/chat/completions", + json!({ + "model": ROUTE_MODEL, "messages": [{"role": "user", "content": "hello"}] + }), + ), + ( + "/v1/messages", + json!({ + "model": ROUTE_MODEL, "max_tokens": 16, + "messages": [{"role": "user", "content": "hello"}] + }), + ), + ( + "/v1/responses", + json!({"model": ROUTE_MODEL, "input": "hello"}), + ), + ]; + for reason in ["abort", "error"] { + let model = format!("model/finish-{reason}"); + let message = format!("provider finished with finish_reason \"{reason}\""); + for fallback in [false, true] { + let route = if fallback { + "type = \"random\"\ntargets = [\"first\", \"second\"]\nweights = [1000, 1]\nseed = 17" + } else { + "type = \"passthrough\"\ntarget = \"first\"" + }; + let app = build_switchyard_router(load_test_config(&format!( + r#" +schema_version = 1 +[llm_clients.mock] +format = "openai_chat" +base_url = "{base_url}" +max_retries = 0 +[targets] +first = {{ id = "{model}", llm_client = "mock" }} +second = {{ id = "model/fallback", llm_client = "mock" }} +[routes.chat] +id = "{ROUTE_MODEL}" +{route} +"#, + base_url = upstream.base_url, + ))?); + for (path, body) in &requests { + let previous_calls = upstream.models().await.len(); + let response = send(&app, "POST", path, Some(body.clone())).await?; + let calls = upstream.models().await[previous_calls..].to_vec(); + if fallback { + assert_eq!(response.status, StatusCode::OK, "{model}: {path}"); + assert_eq!( + response.json()?["model"], + "model/fallback", + "{model}: {path}" + ); + assert_eq!(calls, [model.as_str(), "model/fallback"], "{model}: {path}"); + continue; + } + assert_eq!(response.status, StatusCode::BAD_GATEWAY, "{model}: {path}"); + let expected = if *path == "/v1/messages" { + json!({"type": "error", "error": {"type": "api_error", "message": message}}) + } else { + json!({"error": { + "type": "upstream_error", "code": "upstream_error", "message": message + }}) + }; + assert_eq!(response.json()?, expected, "{model}: {path}"); + assert_eq!(calls, [model.as_str()], "{model}: {path}"); + } + } + } + Ok(()) +} + #[tokio::test] async fn stats_reset_returns_confirmation_and_clears_all_stats() -> TestResult { let (_upstream, app) = test_app(&[(ROUTE_MODEL, &["model/a"])]).await?; diff --git a/crates/switchyard-translation/src/codecs/openai_chat/buffered.rs b/crates/switchyard-translation/src/codecs/openai_chat/buffered.rs index d90423518..74635eeda 100644 --- a/crates/switchyard-translation/src/codecs/openai_chat/buffered.rs +++ b/crates/switchyard-translation/src/codecs/openai_chat/buffered.rs @@ -269,6 +269,18 @@ impl FormatCodec for OpenAiChatCodec { _policy: &TranslationPolicy, ) -> Result { let object = object(body, "$")?; + // `abort` and `error` mean the generation did not complete. + if let Some(reason @ ("abort" | "error")) = object + .get("choices") + .and_then(Value::as_array) + .and_then(|choices| choices.first()) + .and_then(|choice| choice.get("finish_reason")) + .and_then(Value::as_str) + { + return Err(TranslationError::UpstreamFailure { + error: json!({ "message": format!("provider finished with finish_reason \"{reason}\"") }), + }); + } let mut response = AggLlmResponse { id: object .get("id") diff --git a/crates/switchyard-translation/tests/response_translation.rs b/crates/switchyard-translation/tests/response_translation.rs index 4280fa3f1..afc0d1f43 100644 --- a/crates/switchyard-translation/tests/response_translation.rs +++ b/crates/switchyard-translation/tests/response_translation.rs @@ -1057,6 +1057,31 @@ fn failed_responses_return_upstream_failure_with_provider_message() -> TestResul Ok(()) } +// Verifies Chat `abort` and `error` finish reasons fail the turn. +#[test] +fn abort_and_error_finish_reasons_return_upstream_failure() -> TestResult { + let engine = TranslationEngine::default(); + for reason in ["abort", "error"] { + let body = json!({ + "id": "chatcmpl-test", "object": "chat.completion", "model": "gpt-4o", + "choices": [{"index": 0, "finish_reason": reason, + "message": {"role": "assistant", "content": "partial"}}] + }); + let error = engine + .translate_response( + WireFormat::OpenAiChat, + WireFormat::AnthropicMessages, + &body, + &TranslationPolicy::default(), + ) + .err() + .ok_or_else(|| format!("accepted finish_reason {reason}"))?; + assert_eq!(error.kind(), "UpstreamFailure"); + assert!(error.to_string().contains(reason), "{error}"); + } + Ok(()) +} + // Verifies a moderation stop stays distinguishable from a normal turn in both // directions, and that a named refusal category survives re-encoding. #[test] From 5c9c39e1ea65688f66655c82d4e7cfa853a2f410 Mon Sep 17 00:00:00 2001 From: Colin McNamara Date: Thu, 1 Oct 2026 07:15:00 -0500 Subject: [PATCH 2/2] fix(translation): return HTTP 502 for buffered Chat repetition finish reason Signed-off-by: Colin McNamara --- crates/switchyard-server/tests/server.rs | 13 +++++++------ .../src/codecs/openai_chat/buffered.rs | 4 ++-- .../tests/response_translation.rs | 6 +++--- 3 files changed, 12 insertions(+), 11 deletions(-) diff --git a/crates/switchyard-server/tests/server.rs b/crates/switchyard-server/tests/server.rs index 2f9976853..3612ce9da 100644 --- a/crates/switchyard-server/tests/server.rs +++ b/crates/switchyard-server/tests/server.rs @@ -342,8 +342,9 @@ async fn upstream_chat( .into_response(); } - // A generation that did not complete (`abort` from vLLM or SGLang, `error` from OpenRouter). - if let Some(reason @ ("abort" | "error")) = model.strip_prefix("model/finish-") { + // A generation that did not complete (`abort` from vLLM or SGLang, `error` from OpenRouter, + // `repetition` from vLLM). + if let Some(reason @ ("abort" | "error" | "repetition")) = model.strip_prefix("model/finish-") { return Json(json!({ "id": "chatcmpl-finish", "object": "chat.completion", @@ -1150,10 +1151,10 @@ async fn failed_responses_return_errors_and_try_fallback_across_endpoints() -> T Ok(()) } -// A Chat backend that ends a generation with `abort` or `error` did not complete -// it: every client path gets a 502, and a fallback route moves on. +// A Chat backend that ends a generation with `abort`, `error` or `repetition` did +// not complete it: every client path gets a 502, and a fallback route moves on. #[tokio::test] -async fn chat_abort_and_error_finish_reasons_return_errors_and_try_fallback() -> TestResult { +async fn chat_failed_finish_reasons_return_errors_and_try_fallback() -> TestResult { let upstream = MockUpstream::start().await?; let requests = [ ( @@ -1174,7 +1175,7 @@ async fn chat_abort_and_error_finish_reasons_return_errors_and_try_fallback() -> json!({"model": ROUTE_MODEL, "input": "hello"}), ), ]; - for reason in ["abort", "error"] { + for reason in ["abort", "error", "repetition"] { let model = format!("model/finish-{reason}"); let message = format!("provider finished with finish_reason \"{reason}\""); for fallback in [false, true] { diff --git a/crates/switchyard-translation/src/codecs/openai_chat/buffered.rs b/crates/switchyard-translation/src/codecs/openai_chat/buffered.rs index 74635eeda..c522e3aef 100644 --- a/crates/switchyard-translation/src/codecs/openai_chat/buffered.rs +++ b/crates/switchyard-translation/src/codecs/openai_chat/buffered.rs @@ -269,8 +269,8 @@ impl FormatCodec for OpenAiChatCodec { _policy: &TranslationPolicy, ) -> Result { let object = object(body, "$")?; - // `abort` and `error` mean the generation did not complete. - if let Some(reason @ ("abort" | "error")) = object + // `abort`, `error` and `repetition` mean the generation did not complete. + if let Some(reason @ ("abort" | "error" | "repetition")) = object .get("choices") .and_then(Value::as_array) .and_then(|choices| choices.first()) diff --git a/crates/switchyard-translation/tests/response_translation.rs b/crates/switchyard-translation/tests/response_translation.rs index afc0d1f43..20e6dfb30 100644 --- a/crates/switchyard-translation/tests/response_translation.rs +++ b/crates/switchyard-translation/tests/response_translation.rs @@ -1057,11 +1057,11 @@ fn failed_responses_return_upstream_failure_with_provider_message() -> TestResul Ok(()) } -// Verifies Chat `abort` and `error` finish reasons fail the turn. +// Verifies Chat `abort`, `error` and `repetition` finish reasons fail the turn. #[test] -fn abort_and_error_finish_reasons_return_upstream_failure() -> TestResult { +fn failed_finish_reasons_return_upstream_failure() -> TestResult { let engine = TranslationEngine::default(); - for reason in ["abort", "error"] { + for reason in ["abort", "error", "repetition"] { let body = json!({ "id": "chatcmpl-test", "object": "chat.completion", "model": "gpt-4o", "choices": [{"index": 0, "finish_reason": reason,