mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-05 04:24:39 -04:00
fix(cloud-proxy): surface Anthropic refusals instead of empty replies (#12424)
When an Anthropic model declines to answer, the Messages API returns HTTP 200 with `stop_reason: "refusal"` and empty `content`. The translate mode mapped that to a normal reply with no content, so the OpenAI-compatible response looked like a successful completion (`finish_reason: "stop"`, empty message). Routers, agents and UIs could not tell "the model declined" from "the model had nothing to say", and no fallback was triggered. Return an explicit error for `stop_reason: "refusal"` in both the non-streaming path and the streaming path (`message_delta`). Regular replies, including empty `end_turn` replies, are unchanged. Assisted-by: Claude:claude-opus-5-5 Signed-off-by: Stefan Walcz <stefan.walcz@walcz.de>
This commit is contained in:
1 parent
985df46d24
commit
1056c62f4c
3 files changed
+96
-3
No files matched your search
@@ -5,6 +5,7 @@ import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
@@ -30,8 +31,8 @@ import (
|
||||
// message_stop (terminates the stream). Others are ignored.
|
||||
|
||||
type anthropicRequest struct {
|
||||
Model string `json:"model"`
|
||||
MaxTokens int32 `json:"max_tokens"`
|
||||
Model string `json:"model"`
|
||||
MaxTokens int32 `json:"max_tokens"`
|
||||
// System is `any`: a bare string normally, or []anthropicSystemBlock
|
||||
// when cache_prompt is on (the block form carries cache_control).
|
||||
System any `json:"system,omitempty"`
|
||||
@@ -115,7 +116,10 @@ type anthropicResponse struct {
|
||||
Role string `json:"role"`
|
||||
Content []anthropicContentBlock `json:"content"`
|
||||
Model string `json:"model"`
|
||||
Usage *anthropicUsage `json:"usage,omitempty"`
|
||||
// StopReason is "end_turn", "max_tokens", "tool_use", ... or
|
||||
// "refusal" when the model declines to answer.
|
||||
StopReason string `json:"stop_reason,omitempty"`
|
||||
Usage *anthropicUsage `json:"usage,omitempty"`
|
||||
}
|
||||
|
||||
type anthropicUsage struct {
|
||||
@@ -140,8 +144,20 @@ type anthropicStreamDelta struct {
|
||||
Type string `json:"type,omitempty"`
|
||||
Text string `json:"text,omitempty"`
|
||||
PartialJSON string `json:"partial_json,omitempty"`
|
||||
// StopReason is set on message_delta events.
|
||||
StopReason string `json:"stop_reason,omitempty"`
|
||||
}
|
||||
|
||||
// anthropicStopRefusal is the stop_reason Anthropic returns when the
|
||||
// model declines to answer. Such a response carries no (or only partial)
|
||||
// content. Passing it through as a normal reply makes a refusal look like
|
||||
// an empty, successful completion (finish_reason "stop", no content), so
|
||||
// routers and agents cannot tell "declined" from "nothing to say" and
|
||||
// never fall back. Surface it as an error instead.
|
||||
const anthropicStopRefusal = "refusal"
|
||||
|
||||
var errAnthropicRefusal = errors.New("cloud-proxy: upstream model refused to answer (stop_reason=refusal)")
|
||||
|
||||
// Anthropic requires max_tokens. If the caller didn't set it, use a
|
||||
// generous-but-bounded default so the request doesn't 400.
|
||||
const anthropicDefaultMaxTokens int32 = 4096
|
||||
@@ -438,6 +454,9 @@ func (c *CloudProxy) predictAnthropicRich(ctx context.Context, cfg *proxyConfig,
|
||||
if err := json.NewDecoder(resp.Body).Decode(&parsed); err != nil {
|
||||
return nil, fmt.Errorf("cloud-proxy: decode response: %w", err)
|
||||
}
|
||||
if parsed.StopReason == anthropicStopRefusal {
|
||||
return nil, errAnthropicRefusal
|
||||
}
|
||||
|
||||
reply := &pb.Reply{}
|
||||
if parsed.Usage != nil {
|
||||
@@ -552,6 +571,9 @@ func (c *CloudProxy) predictAnthropicStreamRich(ctx context.Context, cfg *proxyC
|
||||
}
|
||||
}
|
||||
case "message_delta":
|
||||
if ev.Delta != nil && ev.Delta.StopReason == anthropicStopRefusal {
|
||||
return errAnthropicRefusal
|
||||
}
|
||||
// Anthropic sends final usage in message_delta.usage. Emit
|
||||
// a usage-only Reply so the consumer can record totals.
|
||||
if ev.Usage != nil {
|
||||
|
||||
@@ -387,3 +387,72 @@ func TestPredict_Anthropic_PromptCache(t *testing.T) {
|
||||
g.Expect(off).NotTo(ContainSubstring("cache_control"))
|
||||
g.Expect(off).To(ContainSubstring(`"system":"be brief"`))
|
||||
}
|
||||
|
||||
// A refusal must not look like an empty, successful completion.
|
||||
func TestPredict_Anthropic_RefusalIsAnError(t *testing.T) {
|
||||
g := NewWithT(t)
|
||||
srv, _ := fakeAnthropicUpstream(t, func(_ anthropicRequest) (int, string, string) {
|
||||
return 200, `{"type":"message","role":"assistant","content":[],"stop_reason":"refusal","usage":{"input_tokens":9,"output_tokens":0}}`, "application/json"
|
||||
})
|
||||
defer srv.Close()
|
||||
cp := newAnthropicTranslateCloudProxy(t, srv.URL)
|
||||
|
||||
_, err := cp.Predict(&pb.PredictOptions{Messages: []*pb.Message{{Role: "user", Content: "x"}}, Tokens: 16})
|
||||
g.Expect(err).To(MatchError(errAnthropicRefusal))
|
||||
}
|
||||
|
||||
// Counter-check: an empty end_turn reply stays a normal, successful reply.
|
||||
func TestPredict_Anthropic_EmptyEndTurnIsNotAnError(t *testing.T) {
|
||||
g := NewWithT(t)
|
||||
srv, _ := fakeAnthropicUpstream(t, func(_ anthropicRequest) (int, string, string) {
|
||||
return 200, `{"type":"message","role":"assistant","content":[],"stop_reason":"end_turn"}`, "application/json"
|
||||
})
|
||||
defer srv.Close()
|
||||
cp := newAnthropicTranslateCloudProxy(t, srv.URL)
|
||||
|
||||
got, err := cp.Predict(&pb.PredictOptions{Messages: []*pb.Message{{Role: "user", Content: "x"}}, Tokens: 16})
|
||||
g.Expect(err).NotTo(HaveOccurred())
|
||||
g.Expect(got).To(Equal(""))
|
||||
}
|
||||
|
||||
func streamAnthropic(t *testing.T, frames []string) ([]string, error) {
|
||||
t.Helper()
|
||||
srv, _ := fakeAnthropicUpstream(t, func(_ anthropicRequest) (int, string, string) {
|
||||
return 200, strings.Join(frames, ""), "text/event-stream"
|
||||
})
|
||||
defer srv.Close()
|
||||
cp := newAnthropicTranslateCloudProxy(t, srv.URL)
|
||||
results := make(chan string, 8)
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
done <- cp.PredictStream(&pb.PredictOptions{Messages: []*pb.Message{{Role: "user", Content: "hi"}}, Tokens: 16}, results)
|
||||
}()
|
||||
var got []string
|
||||
for s := range results {
|
||||
got = append(got, s)
|
||||
}
|
||||
return got, <-done
|
||||
}
|
||||
|
||||
func TestPredictStream_Anthropic_RefusalIsAnError(t *testing.T) {
|
||||
g := NewWithT(t)
|
||||
_, err := streamAnthropic(t, []string{
|
||||
"event: message_start\ndata: {\"type\":\"message_start\"}\n\n",
|
||||
"event: message_delta\ndata: {\"type\":\"message_delta\",\"delta\":{\"stop_reason\":\"refusal\"},\"usage\":{\"output_tokens\":0}}\n\n",
|
||||
"event: message_stop\ndata: {\"type\":\"message_stop\"}\n\n",
|
||||
})
|
||||
g.Expect(err).To(MatchError(errAnthropicRefusal))
|
||||
}
|
||||
|
||||
// Counter-check: a regular stream ending with stop_reason end_turn succeeds.
|
||||
func TestPredictStream_Anthropic_EndTurnIsNotAnError(t *testing.T) {
|
||||
g := NewWithT(t)
|
||||
got, err := streamAnthropic(t, []string{
|
||||
"event: message_start\ndata: {\"type\":\"message_start\"}\n\n",
|
||||
"event: content_block_delta\ndata: {\"type\":\"content_block_delta\",\"index\":0,\"delta\":{\"type\":\"text_delta\",\"text\":\"ok\"}}\n\n",
|
||||
"event: message_delta\ndata: {\"type\":\"message_delta\",\"delta\":{\"stop_reason\":\"end_turn\"},\"usage\":{\"output_tokens\":1}}\n\n",
|
||||
"event: message_stop\ndata: {\"type\":\"message_stop\"}\n\n",
|
||||
})
|
||||
g.Expect(err).NotTo(HaveOccurred())
|
||||
g.Expect(strings.Join(got, "")).To(Equal("ok"))
|
||||
}
|
||||
@@ -198,6 +198,8 @@ image blocks, and per-request usage tokens are dropped through the
|
||||
internal `Predict()` signature. Use passthrough mode when your clients need
|
||||
the upstream's full feature set.
|
||||
|
||||
In translate mode, an Anthropic response with `stop_reason: "refusal"` is returned to the client as an error instead of an empty successful reply, for both non-streaming and streaming requests. A streamed response may already have delivered partial content when the refusal arrives. Responses that end normally (`end_turn`) are unaffected, even when their content is empty.
|
||||
|
||||
#### Anthropic prompt caching
|
||||
|
||||
`proxy.cache_prompt: true` makes the translator add Anthropic
|
||||
|
||||
Reference in new issue
Block a user