mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-05 12:34:43 -04:00
fix(llama-cpp): do not stream the error text as content on pre-stream failures (#12425)
When the first result of a streamed request is an error (for example a prompt that exceeds the context), PredictStream wrote the error message as a Reply and only then returned the error status. LocalAI treated that Reply as the first token: it sent the assistant role chunk and the error text as `content` on an HTTP 200 stream. Because a chunk had already been written, the pre-stream HTTP error path from #12204 never triggered, so streaming clients still got a 200 with the error as model output, while the same request without streaming correctly returns a 400. Return the error only as the gRPC status. The e2e backend suite gets a `context_overflow` capability (enabled for llama-cpp) that streams an over-long prompt and asserts an error status with no content. Assisted-by: Claude:claude-opus-5-5 Signed-off-by: Stefan Walcz <stefan.walcz@walcz.de>
This commit is contained in:
1 parent
1056c62f4c
commit
c3bea567fe
4 files changed
+44
-4
No files matched your search
@@ -747,7 +747,7 @@ test-extra-backend: protogen-go
|
||||
## Convenience wrappers: build the image, then exercise it.
|
||||
test-extra-backend-llama-cpp: docker-build-llama-cpp
|
||||
BACKEND_IMAGE=local-ai-backend:llama-cpp \
|
||||
BACKEND_TEST_CAPS=health,load,predict,stream,logprobs,logit_bias \
|
||||
BACKEND_TEST_CAPS=health,load,predict,stream,logprobs,logit_bias,context_overflow \
|
||||
$(MAKE) test-extra-backend
|
||||
|
||||
## Raw llama.cpp embeddings are required by Go-side pooling. This exercises the
|
||||
|
||||
@@ -2176,10 +2176,10 @@ public:
|
||||
// connection is closed
|
||||
return grpc::Status(grpc::StatusCode::CANCELLED, "Request cancelled by client");
|
||||
} else if (first_result->is_error()) {
|
||||
// Return the error only as the status. Writing it as a Reply first
|
||||
// made it the first content chunk: LocalAI streamed the error text
|
||||
// as assistant output on an HTTP 200 instead of failing the request.
|
||||
json error_json = first_result->to_json();
|
||||
backend::Reply reply;
|
||||
reply.set_message(error_json.value("message", ""));
|
||||
writer->Write(reply);
|
||||
return grpc::Status(grpc::StatusCode::INTERNAL, error_json.value("message", "Error occurred"));
|
||||
}
|
||||
|
||||
|
||||
@@ -33,6 +33,8 @@ Available additional parameters: `top_p`, `top_k`, `max_tokens`
|
||||
|
||||
Reasoning models return their thinking in the `reasoning` field. When a model reasons and calls a tool in the same turn, see [Interleaved Thinking with Tool Calls]({{%relref "features/interleaved-thinking" %}}).
|
||||
|
||||
When `stream: true` is set and the llama.cpp backend fails before the first chunk, for example because the prompt exceeds the context size, the request fails with an HTTP error. The error message is not streamed as assistant content. An error after streaming has started is reported inside the stream.
|
||||
|
||||
### Edit completions
|
||||
|
||||
https://platform.openai.com/docs/api-reference/edits
|
||||
|
||||
@@ -62,6 +62,10 @@ import (
|
||||
// model output into ChatDelta.tool_calls.
|
||||
// "image" exercises the GenerateImage RPC and asserts a
|
||||
// non-empty file is written to the requested dst path.
|
||||
// "context_overflow" streams a prompt longer than the
|
||||
// context and asserts the backend fails the stream with
|
||||
// an error status WITHOUT first sending the error text as
|
||||
// a content chunk (which clients would read as model output).
|
||||
// "long_prefill" sends a prompt long enough to span more
|
||||
// than one prefill batch and asserts the answer still
|
||||
// reflects the prompt. Catches GPU backends whose kernels
|
||||
@@ -98,6 +102,7 @@ const (
|
||||
capLoad = "load"
|
||||
capPredict = "predict"
|
||||
capStream = "stream"
|
||||
capCtxOverflow = "context_overflow"
|
||||
capEmbeddings = "embeddings"
|
||||
capTools = "tools"
|
||||
capTranscription = "transcription"
|
||||
@@ -540,6 +545,39 @@ var _ = Describe("Backend container", Ordered, func() {
|
||||
GinkgoWriter.Printf("Stream: %d chunks, combined=%q\n", chunks, combined)
|
||||
})
|
||||
|
||||
It("fails a stream whose prompt exceeds the context without emitting content", func() {
|
||||
if !caps[capCtxOverflow] {
|
||||
Skip("context_overflow capability not enabled")
|
||||
}
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 120*time.Second)
|
||||
defer cancel()
|
||||
// Far more tokens than any test context (default 512).
|
||||
stream, err := client.PredictStream(ctx, &pb.PredictOptions{
|
||||
Prompt: strings.Repeat("overflow ", 8*int(envInt32("BACKEND_TEST_CTX_SIZE", 512))+64),
|
||||
Tokens: 8,
|
||||
})
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
|
||||
var content string
|
||||
var streamErr error
|
||||
for {
|
||||
msg, err := stream.Recv()
|
||||
if err == io.EOF {
|
||||
break
|
||||
}
|
||||
if err != nil {
|
||||
streamErr = err
|
||||
break
|
||||
}
|
||||
content += string(msg.GetMessage())
|
||||
}
|
||||
Expect(streamErr).To(HaveOccurred(), "an over-long prompt must fail the stream")
|
||||
Expect(streamErr.Error()).To(ContainSubstring("exceeds the available context size"))
|
||||
// Before the fix the backend wrote the error text as a Reply message
|
||||
// first. LocalAI forwarded it as assistant content on a 200 stream.
|
||||
Expect(content).To(BeEmpty(), "error text was streamed as content: %q", content)
|
||||
})
|
||||
|
||||
// Logprobs: backends that wire OpenAI-compatible logprobs return a
|
||||
// JSON-encoded payload in Reply.logprobs (see backend.proto). The exact
|
||||
// shape is backend-specific; we only assert that the field is populated
|
||||
|
||||
Reference in new issue
Block a user