From 86a2ae2f3b59b42f66089155b88ad728728256d2 Mon Sep 17 00:00:00 2001 From: Dragos Boca Date: Thu, 17 Sep 2026 19:08:22 +0300 Subject: [PATCH] fix(gateway): pass the invalid_guard_verdict worker code through Guard models emit invalid_guard_verdict when their first output positions carry no complete Yes/No distribution (#285), and the SDK READMEs tell callers to expect that code. The gateway's worker error allowlist predates it, so the terminal collapsed to a generic inference_error with the "internal error during generation" message on both the buffered and the streaming path. Admit the code so the typed terminal and its message reach the client. It settles like empty_model_output: terminal, non-retryable, server_error. --- packages/sie_gateway/src/http_error.rs | 4 ++++ packages/sie_gateway/src/queue/streaming.rs | 2 ++ 2 files changed, 6 insertions(+) diff --git a/packages/sie_gateway/src/http_error.rs b/packages/sie_gateway/src/http_error.rs index 4b08ad52e..a38bbffcc 100644 --- a/packages/sie_gateway/src/http_error.rs +++ b/packages/sie_gateway/src/http_error.rs @@ -149,6 +149,10 @@ pub mod openai_code { /// the whole budget, #3104/#3136). Terminal and non-retryable: tokens were /// genuinely consumed, so it settles exactly like ``inference_error``. pub const EMPTY_MODEL_OUTPUT: &str = "empty_model_output"; + /// Worker terminal for a guard model whose first output positions carry + /// no complete Yes/No verdict distribution. Terminal and non-retryable; + /// settles like ``empty_model_output``. + pub const INVALID_GUARD_VERDICT: &str = "invalid_guard_verdict"; } /// OpenAI-shaped error body: diff --git a/packages/sie_gateway/src/queue/streaming.rs b/packages/sie_gateway/src/queue/streaming.rs index bed8d98a1..f283e74f9 100644 --- a/packages/sie_gateway/src/queue/streaming.rs +++ b/packages/sie_gateway/src/queue/streaming.rs @@ -276,6 +276,7 @@ pub(crate) fn client_safe_worker_error_code(code: &str) -> &'static str { openai_code::CANCELLED => openai_code::CANCELLED, openai_code::CONTEXT_EXCEEDED => openai_code::CONTEXT_EXCEEDED, openai_code::EMPTY_MODEL_OUTPUT => openai_code::EMPTY_MODEL_OUTPUT, + openai_code::INVALID_GUARD_VERDICT => openai_code::INVALID_GUARD_VERDICT, "grammar_invalid" => "grammar_invalid", "invalid_request" => "invalid_request", "parallel_tool_calls_violated" => "parallel_tool_calls_violated", @@ -1392,6 +1393,7 @@ mod tests { "cancelled", "context_exceeded", "empty_model_output", + "invalid_guard_verdict", "grammar_invalid", "invalid_request", "parallel_tool_calls_violated",