diff --git a/Makefile b/Makefile index 49812a8d7..7c05fda6d 100644 --- a/Makefile +++ b/Makefile @@ -747,7 +747,7 @@ test-extra-backend: protogen-go ## Convenience wrappers: build the image, then exercise it. test-extra-backend-llama-cpp: docker-build-llama-cpp BACKEND_IMAGE=local-ai-backend:llama-cpp \ - BACKEND_TEST_CAPS=health,load,predict,stream,logprobs,logit_bias \ + BACKEND_TEST_CAPS=health,load,predict,stream,logprobs,logit_bias,context_overflow \ $(MAKE) test-extra-backend ## Raw llama.cpp embeddings are required by Go-side pooling. This exercises the diff --git a/backend/cpp/llama-cpp/grpc-server.cpp b/backend/cpp/llama-cpp/grpc-server.cpp index bf1c43d39..3b908736e 100644 --- a/backend/cpp/llama-cpp/grpc-server.cpp +++ b/backend/cpp/llama-cpp/grpc-server.cpp @@ -2176,10 +2176,10 @@ public: // connection is closed return grpc::Status(grpc::StatusCode::CANCELLED, "Request cancelled by client"); } else if (first_result->is_error()) { + // Return the error only as the status. Writing it as a Reply first + // made it the first content chunk: LocalAI streamed the error text + // as assistant output on an HTTP 200 instead of failing the request. json error_json = first_result->to_json(); - backend::Reply reply; - reply.set_message(error_json.value("message", "")); - writer->Write(reply); return grpc::Status(grpc::StatusCode::INTERNAL, error_json.value("message", "Error occurred")); } diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md index d1af3b7fb..4945c8b6d 100644 --- a/docs/content/features/text-generation.md +++ b/docs/content/features/text-generation.md @@ -33,6 +33,8 @@ Available additional parameters: `top_p`, `top_k`, `max_tokens` Reasoning models return their thinking in the `reasoning` field. When a model reasons and calls a tool in the same turn, see [Interleaved Thinking with Tool Calls]({{%relref "features/interleaved-thinking" %}}). +When `stream: true` is set and the llama.cpp backend fails before the first chunk, for example because the prompt exceeds the context size, the request fails with an HTTP error. The error message is not streamed as assistant content. An error after streaming has started is reported inside the stream. + ### Edit completions https://platform.openai.com/docs/api-reference/edits diff --git a/tests/e2e-backends/backend_test.go b/tests/e2e-backends/backend_test.go index 9ebd044c8..4afda646a 100644 --- a/tests/e2e-backends/backend_test.go +++ b/tests/e2e-backends/backend_test.go @@ -62,6 +62,10 @@ import ( // model output into ChatDelta.tool_calls. // "image" exercises the GenerateImage RPC and asserts a // non-empty file is written to the requested dst path. +// "context_overflow" streams a prompt longer than the +// context and asserts the backend fails the stream with +// an error status WITHOUT first sending the error text as +// a content chunk (which clients would read as model output). // "long_prefill" sends a prompt long enough to span more // than one prefill batch and asserts the answer still // reflects the prompt. Catches GPU backends whose kernels @@ -98,6 +102,7 @@ const ( capLoad = "load" capPredict = "predict" capStream = "stream" + capCtxOverflow = "context_overflow" capEmbeddings = "embeddings" capTools = "tools" capTranscription = "transcription" @@ -540,6 +545,39 @@ var _ = Describe("Backend container", Ordered, func() { GinkgoWriter.Printf("Stream: %d chunks, combined=%q\n", chunks, combined) }) + It("fails a stream whose prompt exceeds the context without emitting content", func() { + if !caps[capCtxOverflow] { + Skip("context_overflow capability not enabled") + } + ctx, cancel := context.WithTimeout(context.Background(), 120*time.Second) + defer cancel() + // Far more tokens than any test context (default 512). + stream, err := client.PredictStream(ctx, &pb.PredictOptions{ + Prompt: strings.Repeat("overflow ", 8*int(envInt32("BACKEND_TEST_CTX_SIZE", 512))+64), + Tokens: 8, + }) + Expect(err).NotTo(HaveOccurred()) + + var content string + var streamErr error + for { + msg, err := stream.Recv() + if err == io.EOF { + break + } + if err != nil { + streamErr = err + break + } + content += string(msg.GetMessage()) + } + Expect(streamErr).To(HaveOccurred(), "an over-long prompt must fail the stream") + Expect(streamErr.Error()).To(ContainSubstring("exceeds the available context size")) + // Before the fix the backend wrote the error text as a Reply message + // first. LocalAI forwarded it as assistant content on a 200 stream. + Expect(content).To(BeEmpty(), "error text was streamed as content: %q", content) + }) + // Logprobs: backends that wire OpenAI-compatible logprobs return a // JSON-encoded payload in Reply.logprobs (see backend.proto). The exact // shape is backend-specific; we only assert that the field is populated