mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-28 17:15:02 -04:00
fix: return correct HTTP status codes for saturation and no-nodes-available (#12113)
* feat: return 429 when backends are saturated When backends are at capacity (per-model max_concurrent or the process-wide --max-concurrent-backend-requests ceiling), the response was 503. The OpenAI SDK, litellm, and most agent harnesses key on 429 for rate-limit backoff and treat 503 as a hard error. Both saturation paths now return 429 with the existing Retry-After header and type: "rate_limit_error" in the JSON body. The per-model admission middleware keeps admission_rejected as the code field so existing alerts that match on it still fire. Non-saturation 503s are unchanged: model cold-loading (with progress body), model-load failure cooldown, PII detector fail-closed, and classifier unavailable. These mean "not ready" rather than "busy". Assisted-by: AGENT:regolo/glm5.2 [TOOL] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * fix: return 503 when scheduler has no available nodes When the scheduler cannot find any healthy node to serve a model — all nodes are full and eviction cannot free a slot, or a node_selector excludes every candidate — the error fell through to 500. A 500 tells clients something is broken when the condition is transient and retryable. The router now wraps these errors with a new ErrNoAvailableNodes sentinel. The HTTP error handler maps it to 503 via applyNoAvailableNodes, following the same pattern as applyBackendAdmission (429). Unrelated scheduler errors (DB timeouts, registry lookups) still return 500. Three return sites are wrapped: - resolveSelectorCandidates: selector matches zero healthy nodes - scheduleNewModel eviction-busy: all models have in-flight requests - scheduleNewModel eviction-failed: eviction itself errored The existing scheduleAndLoad wrapper ("no available nodes: %w") preserves the sentinel through the chain via errors.Is, as does ModelRouterAdapter. Assisted-by: AGENT:regolo/glm5.2 [TOOL] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * test(http): use Ginkgo for admission tests Replace forbidden testing.T calls with Ginkgo and Gomega so the lint check accepts the admission handler tests. Assisted-by: Codex:GPT-6 forbidigo * fix(middleware): show the recorded status for admission rejections The admission audit row now records 429, but the Middleware page still printed a hard-coded 503. Read the status from the event, and update the two package comments that still said 503. Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Assisted-by: Claude:claude-opus-5-5 [Claude Code] --------- Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com>
This commit is contained in:
13 files changed
+144
-19
No files matched your search
@@ -11,8 +11,9 @@ import (
|
||||
)
|
||||
|
||||
// BackendAdmissionError reports that the process-wide backend execution
|
||||
// ceiling is full. HTTP callers map it to 503; internal callers receive the
|
||||
// same typed error instead of silently queueing and growing in-flight state.
|
||||
// ceiling is full. HTTP callers map it to 429 (Too Many Requests) with a
|
||||
// Retry-After header; internal callers receive the same typed error instead
|
||||
// of silently queueing and growing in-flight state.
|
||||
type BackendAdmissionError struct {
|
||||
Limit int
|
||||
RetryAfter time.Duration
|
||||
|
||||
@@ -0,0 +1,70 @@
|
||||
package http
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"time"
|
||||
|
||||
"github.com/labstack/echo/v4"
|
||||
corebackend "github.com/mudler/LocalAI/core/backend"
|
||||
"github.com/mudler/LocalAI/core/services/nodes"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("Backend admission", func() {
|
||||
It("maps BackendAdmissionError to 429 with Retry-After", func() {
|
||||
e := echo.New()
|
||||
req := httptest.NewRequest(http.MethodPost, "/", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
c := e.NewContext(req, rec)
|
||||
|
||||
err := &corebackend.BackendAdmissionError{Limit: 4, RetryAfter: 3 * time.Second}
|
||||
code := applyBackendAdmission(err, http.StatusInternalServerError, c)
|
||||
|
||||
Expect(code).To(Equal(http.StatusTooManyRequests))
|
||||
Expect(rec.Header().Get("Retry-After")).To(Equal("3"))
|
||||
})
|
||||
|
||||
It("passes through non-admission errors unchanged", func() {
|
||||
e := echo.New()
|
||||
req := httptest.NewRequest(http.MethodPost, "/", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
c := e.NewContext(req, rec)
|
||||
|
||||
code := applyBackendAdmission(errors.New("some other error"), http.StatusInternalServerError, c)
|
||||
Expect(code).To(Equal(http.StatusInternalServerError))
|
||||
Expect(rec.Header().Get("Retry-After")).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("No available nodes", func() {
|
||||
It("maps ErrNoAvailableNodes to 503", func() {
|
||||
// The scheduler wraps the sentinel in fmt.Errorf chains and via
|
||||
// errors.Join — errors.Is must still find it.
|
||||
wrapped := fmt.Errorf("routing model foo: %w",
|
||||
fmt.Errorf("no available nodes: %w",
|
||||
fmt.Errorf("no healthy nodes available: %w",
|
||||
errors.Join(nodes.ErrEvictionBusy, nodes.ErrNoAvailableNodes))))
|
||||
|
||||
code := applyNoAvailableNodes(wrapped, http.StatusInternalServerError)
|
||||
Expect(code).To(Equal(http.StatusServiceUnavailable))
|
||||
})
|
||||
|
||||
It("maps selector-mismatch chain to 503", func() {
|
||||
wrapped := fmt.Errorf("routing model bar: %w",
|
||||
fmt.Errorf("no available nodes: %w",
|
||||
fmt.Errorf("no healthy nodes match selector for model bar: {\"gpu.vendor\":\"tpu\"}: %w",
|
||||
nodes.ErrNoAvailableNodes)))
|
||||
|
||||
code := applyNoAvailableNodes(wrapped, http.StatusInternalServerError)
|
||||
Expect(code).To(Equal(http.StatusServiceUnavailable))
|
||||
})
|
||||
|
||||
It("passes through unrelated errors unchanged", func() {
|
||||
code := applyNoAvailableNodes(errors.New("database timeout"), http.StatusInternalServerError)
|
||||
Expect(code).To(Equal(http.StatusInternalServerError))
|
||||
})
|
||||
})
|
||||
+22
-2
@@ -85,7 +85,20 @@ func applyBackendAdmission(err error, code int, c echo.Context) int {
|
||||
return code
|
||||
}
|
||||
c.Response().Header().Set("Retry-After", strconv.Itoa(int(capacityErr.RetryAfter.Seconds())))
|
||||
return http.StatusServiceUnavailable
|
||||
return http.StatusTooManyRequests
|
||||
}
|
||||
|
||||
// applyNoAvailableNodes maps scheduler "no available nodes" errors to 503.
|
||||
// When the cluster has no healthy node to serve a model — all are full, a
|
||||
// node selector excludes every candidate, or eviction could not free a slot —
|
||||
// the request is retryable, not a server bug. Without this the error fell
|
||||
// through to 500, which tells clients something is broken when they just
|
||||
// need to wait for a node.
|
||||
func applyNoAvailableNodes(err error, code int) int {
|
||||
if errors.Is(err, nodes.ErrNoAvailableNodes) {
|
||||
return http.StatusServiceUnavailable
|
||||
}
|
||||
return code
|
||||
}
|
||||
|
||||
// respondModelLoading answers a request whose model is still cold-loading with
|
||||
@@ -208,6 +221,7 @@ func API(application *application.Application) (*echo.Echo, error) {
|
||||
}
|
||||
code = applyModelLoadCooldown(err, code, c)
|
||||
code = applyBackendAdmission(err, code, c)
|
||||
code = applyNoAvailableNodes(err, code)
|
||||
|
||||
// Handle 404 errors: serve React SPA for HTML requests, JSON otherwise
|
||||
if code == http.StatusNotFound {
|
||||
@@ -224,8 +238,13 @@ func API(application *application.Application) (*echo.Echo, error) {
|
||||
}
|
||||
|
||||
// Send custom error page
|
||||
errType := ""
|
||||
var capErr *corebackend.BackendAdmissionError
|
||||
if errors.As(err, &capErr) {
|
||||
errType = "rate_limit_error"
|
||||
}
|
||||
c.JSON(code, schema.ErrorResponse{
|
||||
Error: &schema.APIError{Message: err.Error(), Code: code},
|
||||
Error: &schema.APIError{Message: err.Error(), Code: code, Type: errType},
|
||||
})
|
||||
}
|
||||
} else {
|
||||
@@ -240,6 +259,7 @@ func API(application *application.Application) (*echo.Echo, error) {
|
||||
// Opaque errors deliberately withhold the body, so a still-loading
|
||||
// model gets the status and Retry-After but no progress detail.
|
||||
code = applyModelLoading(err, code, c)
|
||||
code = applyNoAvailableNodes(err, code)
|
||||
c.NoContent(code)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -20,7 +20,7 @@ import (
|
||||
// SERVED model — a router fanout that lands on a saturated downstream
|
||||
// model gets rejected even though the requested router-model has slack.
|
||||
//
|
||||
// On reject: HTTP 503, Retry-After header, error JSON. An audit row
|
||||
// On reject: HTTP 429, Retry-After header, error JSON. An audit row
|
||||
// goes into the shared event store under KindAdmission so admins see
|
||||
// rejection rates alongside PII and proxy events.
|
||||
//
|
||||
@@ -39,9 +39,10 @@ func AdmissionControl(limiter *admission.Limiter, events pii.EventStore) echo.Mi
|
||||
retryAfter := admission.RetryAfter(cfg.Limits.RetryAfterSeconds)
|
||||
recordAdmissionRejection(events, cfg.Name, retryAfter)
|
||||
c.Response().Header().Set("Retry-After", strconv.Itoa(int(retryAfter.Seconds())))
|
||||
return c.JSON(http.StatusServiceUnavailable, map[string]any{
|
||||
return c.JSON(http.StatusTooManyRequests, map[string]any{
|
||||
"error": map[string]any{
|
||||
"type": "admission_rejected",
|
||||
"type": "rate_limit_error",
|
||||
"code": "admission_rejected",
|
||||
"message": fmt.Sprintf("model %q is at capacity (max_concurrent=%d); retry after %s", cfg.Name, max, retryAfter),
|
||||
},
|
||||
})
|
||||
@@ -61,7 +62,7 @@ func recordAdmissionRejection(events pii.EventStore, modelName string, retryAfte
|
||||
if events == nil {
|
||||
return
|
||||
}
|
||||
statusCode := http.StatusServiceUnavailable
|
||||
statusCode := http.StatusTooManyRequests
|
||||
durMS := retryAfter.Milliseconds()
|
||||
id := fmt.Sprintf("adm_%d_%s", admissionEventSeq.Add(1), randHex(4))
|
||||
_ = events.Record(context.Background(), pii.PIIEvent{
|
||||
|
||||
@@ -60,7 +60,7 @@ var _ = Describe("Admission", func() {
|
||||
|
||||
It("rejects when full", func() {
|
||||
// Saturate the limiter outside the middleware, then a request
|
||||
// at the same model gets 503 with a Retry-After header.
|
||||
// at the same model gets 429 with a Retry-After header.
|
||||
lim := admission.New()
|
||||
release, ok := lim.Acquire("busy", 1)
|
||||
Expect(ok).To(BeTrue(), "setup acquire should succeed")
|
||||
@@ -75,7 +75,7 @@ var _ = Describe("Admission", func() {
|
||||
return c.String(http.StatusOK, "ok")
|
||||
})
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(rec.Code).To(Equal(http.StatusServiceUnavailable))
|
||||
Expect(rec.Code).To(Equal(http.StatusTooManyRequests))
|
||||
Expect(rec.Header().Get("Retry-After")).To(Equal("3"))
|
||||
Expect(handlerCalled).To(BeFalse(), "handler should not run when admission rejects")
|
||||
Expect(rec.Body.String()).To(ContainSubstring("admission_rejected"))
|
||||
|
||||
@@ -931,7 +931,8 @@ function eventDetails(e) {
|
||||
}
|
||||
case 'admission': {
|
||||
const retry = e.duration_ms != null ? `retry-after ${Math.round(e.duration_ms / 1000)}s` : ''
|
||||
return `HTTP 503 rejected · ${retry}`
|
||||
// Older audit rows were recorded as 503; newer ones as 429.
|
||||
return `HTTP ${e.status_code || 429} rejected · ${retry}`
|
||||
}
|
||||
default: {
|
||||
const len = e.length != null ? `len ${e.length}` : ''
|
||||
|
||||
@@ -955,7 +955,7 @@ func (r *SmartRouter) resolveSelectorCandidates(ctx context.Context, modelID str
|
||||
return nil, fmt.Errorf("looking up nodes for selector %s: %w", sched.NodeSelector, err)
|
||||
}
|
||||
if len(candidates) == 0 {
|
||||
return nil, fmt.Errorf("no healthy nodes match selector for model %s: %s", modelID, sched.NodeSelector)
|
||||
return nil, fmt.Errorf("no healthy nodes match selector for model %s: %s: %w", modelID, sched.NodeSelector, ErrNoAvailableNodes)
|
||||
}
|
||||
return extractNodeIDs(candidates), nil
|
||||
}
|
||||
@@ -1167,9 +1167,9 @@ func (r *SmartRouter) scheduleNewModel(ctx context.Context, backendType, modelID
|
||||
evictedNode, evictErr := r.evictLRUAndFreeNodeFrom(ctx, candidateNodeIDs)
|
||||
if evictErr != nil {
|
||||
if errors.Is(evictErr, ErrEvictionBusy) {
|
||||
return nil, "", 0, fmt.Errorf("no healthy nodes available: %w", evictErr)
|
||||
return nil, "", 0, fmt.Errorf("no healthy nodes available: %w", errors.Join(evictErr, ErrNoAvailableNodes))
|
||||
}
|
||||
return nil, "", 0, fmt.Errorf("no healthy nodes available and eviction failed: %w", evictErr)
|
||||
return nil, "", 0, fmt.Errorf("no healthy nodes available and eviction failed: %w", errors.Join(evictErr, ErrNoAvailableNodes))
|
||||
}
|
||||
node = evictedNode
|
||||
}
|
||||
@@ -2059,6 +2059,13 @@ func (r *SmartRouter) EvictLRU(ctx context.Context, nodeID string) (string, erro
|
||||
// and none can be evicted to make room.
|
||||
var ErrEvictionBusy = errors.New("all models busy, cannot evict")
|
||||
|
||||
// ErrNoAvailableNodes is returned when the scheduler cannot find any healthy
|
||||
// node to serve a model — all nodes are full and eviction cannot free a slot,
|
||||
// or a node selector excludes every candidate. The HTTP layer maps this to
|
||||
// 503 so clients treat it as a transient condition rather than a server bug
|
||||
// (which is what 500 would imply).
|
||||
var ErrNoAvailableNodes = errors.New("no available nodes")
|
||||
|
||||
// evictLRUAndFreeNode finds the globally least-recently-used model with zero in-flight,
|
||||
// unloads it, and returns its node for reuse. If all models are busy, retries briefly.
|
||||
//
|
||||
|
||||
@@ -843,6 +843,26 @@ var _ = Describe("SmartRouter", func() {
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(err.Error()).To(ContainSubstring("no available nodes"))
|
||||
})
|
||||
|
||||
It("wraps ErrNoAvailableNodes when all nodes are full and eviction cannot help", func() {
|
||||
// gorm.ErrRecordNotFound is the registry's verdict that no node
|
||||
// matches — the scheduler then falls through to eviction. With
|
||||
// DB nil, eviction returns ErrEvictionBusy, and the scheduler
|
||||
// wraps the error with ErrNoAvailableNodes so the HTTP layer can
|
||||
// map it to 503 instead of 500.
|
||||
reg.findIdleErr = errors.New("no idle")
|
||||
reg.findLeastLoadedErr = gorm.ErrRecordNotFound
|
||||
|
||||
router := NewSmartRouter(reg, SmartRouterOptions{
|
||||
Unloader: unloader,
|
||||
ClientFactory: factory,
|
||||
})
|
||||
|
||||
_, err := router.Route(context.Background(), "m5", "models/m5.gguf", "llama-cpp", "", nil, false)
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue())
|
||||
Expect(errors.Is(err, ErrEvictionBusy)).To(BeTrue())
|
||||
})
|
||||
})
|
||||
|
||||
Describe("UnloadModel (mock-based)", func() {
|
||||
@@ -970,6 +990,7 @@ var _ = Describe("SmartRouter", func() {
|
||||
_, err := router.Route(context.Background(), "aliased-model", "models/aliased.gguf", "llama-cpp", "", nil, false)
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(err.Error()).To(ContainSubstring("no healthy nodes match selector"))
|
||||
Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue())
|
||||
})
|
||||
|
||||
It("returns error when no nodes match selector", func() {
|
||||
@@ -988,6 +1009,7 @@ var _ = Describe("SmartRouter", func() {
|
||||
_, err := router.Route(context.Background(), "no-match-model", "models/nomatch.gguf", "llama-cpp", "", nil, false)
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(err.Error()).To(ContainSubstring("no healthy nodes match selector"))
|
||||
Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue())
|
||||
})
|
||||
|
||||
It("uses regular methods when model has no scheduling config", func() {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
// Package admission is routing-module subsystem 5: per-model
|
||||
// concurrency control + audit. The middleware acquires a slot
|
||||
// before the handler runs; on full, the request gets 503 with
|
||||
// before the handler runs; on full, the request gets 429 with
|
||||
// Retry-After so clients back off rather than pile on. The audit
|
||||
// row goes into the shared event store alongside PII and proxy
|
||||
// rows so admins see a single timeline of routing pressure.
|
||||
|
||||
@@ -109,7 +109,7 @@ const (
|
||||
// model's MaxConcurrent ceiling is full. The Host field carries
|
||||
// the model name (overloading the existing column rather than
|
||||
// adding a new one — admins read it as "the thing that was
|
||||
// busy"); StatusCode is 503.
|
||||
// busy"); StatusCode is 429.
|
||||
KindAdmission EventKind = "admission"
|
||||
)
|
||||
|
||||
|
||||
@@ -88,7 +88,9 @@ The `/v1/responses` endpoint returns errors with this structure:
|
||||
| 404 | Not Found | Model or resource does not exist |
|
||||
| 409 | Conflict | Resource already exists (e.g., duplicate token) |
|
||||
| 422 | Unprocessable Entity | Validation failed (e.g., invalid parameter range) |
|
||||
| 429 | Too Many Requests | All backends are saturated (per-model `max_concurrent` or process-wide `--max-concurrent-backend-requests` ceiling reached). Includes a `Retry-After` header and `type: "rate_limit_error"` so OpenAI-compatible clients and harnesses back off automatically |
|
||||
| 500 | Internal Server Error | Backend inference failure, unexpected server errors |
|
||||
| 503 | Service Unavailable | No healthy node available to serve the model (cluster is full, eviction cannot free a slot, or a `node_selector` excludes all candidates). Also used during model-load cooldown and while a model is still cold-loading. Retryable |
|
||||
|
||||
## Global Error Handling
|
||||
|
||||
|
||||
@@ -95,7 +95,7 @@ For more information on VRAM management, see [VRAM and Memory Management]({{%rel
|
||||
| Parameter | Default | Description | Environment Variable |
|
||||
|-----------|---------|-------------|----------------------|
|
||||
| `--address` | `:8080` | Bind address for the API server | `$LOCALAI_ADDRESS`, `$ADDRESS` |
|
||||
| `--max-concurrent-backend-requests` | `1024` | Process-wide ceiling for concurrent backend inference operations. Excess inference receives HTTP 503 with `Retry-After`; UI and administrative endpoints remain available | `$LOCALAI_MAX_CONCURRENT_BACKEND_REQUESTS`, `$MAX_CONCURRENT_BACKEND_REQUESTS` |
|
||||
| `--max-concurrent-backend-requests` | `1024` | Process-wide ceiling for concurrent backend inference operations. Excess inference receives HTTP 429 with `Retry-After`; UI and administrative endpoints remain available | `$LOCALAI_MAX_CONCURRENT_BACKEND_REQUESTS`, `$MAX_CONCURRENT_BACKEND_REQUESTS` |
|
||||
| `--cors` | `false` | Enable CORS (Cross-Origin Resource Sharing) | `$LOCALAI_CORS`, `$CORS` |
|
||||
| `--cors-allow-origins` | | Comma-separated list of allowed CORS origins | `$LOCALAI_CORS_ALLOW_ORIGINS`, `$CORS_ALLOW_ORIGINS` |
|
||||
| `--disable-csrf` | `false` | Disable CSRF middleware (enabled by default) | `$LOCALAI_DISABLE_CSRF` |
|
||||
|
||||
@@ -21,8 +21,9 @@ The left column is the literal string as it appears in the LocalAI server log (o
|
||||
| `grpc service not ready` | The backend process was spawned but its gRPC server did not become healthy in time (slow start, crash on startup, or the process died while loading). When a local backend has already exited, the error includes its exit code and last stderr line. | Use the included stderr diagnostic when present; otherwise check the log lines just above. A crash here often means out of memory, a missing shared library, or an incompatible CPU (see `SIGILL`). Increase available RAM/VRAM or pick a smaller quantization. |
|
||||
| `failed to load model: ...` | Returned by the load endpoints and several feature paths (voice, realtime, audio transform) when the model config could not be resolved or the backend load failed. | Confirm the model name exists (`local-ai models list`) and its YAML is valid. The trailing text carries the specific reason. |
|
||||
| HTTP `503` with a `Retry-After` header, after a load failed | Model-load failure cooldown. After a model fails to load, LocalAI refuses new load attempts for that model for a short window so a client that keeps polling a broken model does not respawn a crashing backend on every request. The window starts at `--model-load-failure-cooldown` (default `10s`) and doubles per consecutive failure up to 5m; it resets on the first success. | Fix the underlying load failure (see the rows above), then wait out the `Retry-After` seconds before retrying, or restart LocalAI to clear the cooldown. Set `--model-load-failure-cooldown 0` (or `LOCALAI_MODEL_LOAD_FAILURE_COOLDOWN=0`) to disable the cooldown entirely. See {{% relref "reference/cli-reference" %}}. |
|
||||
| HTTP `503` with a `Retry-After` header, under load | Per-model concurrency limit reached. When a model config sets a `MaxConcurrent` limit, extra requests are rejected with `503` and a `Retry-After` (whole seconds, floor 1) instead of queueing. | Retry after the advised delay, raise the model's concurrency limit, or run more replicas. |
|
||||
| HTTP `503` when backend inference is saturated | The process-wide `--max-concurrent-backend-requests` backend-execution ceiling is full. This protects inference and in-flight backend-trace memory without blocking UI or administrative endpoints. | Retry after the advised delay, reduce inference concurrency, raise the limit if the host has capacity, or add replicas. |
|
||||
| HTTP `503` with `no available nodes` or `no healthy nodes match selector` | The scheduler could not find any healthy node to serve the model. All nodes are full and eviction cannot free a slot, or a `node_selector` in the model's scheduling config excludes every candidate. | Retry after a node becomes available or an in-flight request completes and frees a slot. In a cluster, add nodes or replicas. If a selector is set, confirm at least one healthy node matches it. |
|
||||
| HTTP `429` with a `Retry-After` header, under load | Per-model concurrency limit reached. When a model config sets a `MaxConcurrent` limit, extra requests are rejected with `429` and a `Retry-After` (whole seconds, floor 1) instead of queueing. | Retry after the advised delay, raise the model's concurrency limit, or run more replicas. |
|
||||
| HTTP `429` when backend inference is saturated | The process-wide `--max-concurrent-backend-requests` backend-execution ceiling is full. This protects inference and in-flight backend-trace memory without blocking UI or administrative endpoints. | Retry after the advised delay, reduce inference concurrency, raise the limit if the host has capacity, or add replicas. |
|
||||
| `invalid pitch` (with CUDA) | The prompt exceeded the model's context size. | Reduce the prompt length, or raise the model's context size (`context_size:` in the model YAML). |
|
||||
| `SIGILL` (illegal instruction) on startup | The prebuilt backend binary uses CPU instructions your CPU does not have (for example AVX512, AVX2, F16C, FMA). | Rebuild the backend for your CPU. In a container, set `REBUILD=true` and disable the unsupported instructions, for example `CMAKE_ARGS="-DGGML_F16C=OFF -DGGML_AVX512=OFF -DGGML_AVX2=OFF -DGGML_FMA=OFF" make build`. |
|
||||
| CUDA / VRAM out of memory (backend log shows `out of memory`, `CUDA error: out of memory`, or the process is killed loading) | The model plus its KV cache does not fit in GPU memory. | Use a smaller quantization, reduce `context_size:`, offload fewer layers to the GPU (lower `gpu_layers:`), or free VRAM held by other processes. On multi-GPU hosts, confirm the model is not trying to load entirely onto one device. |
|
||||
|
||||
Reference in new issue
Block a user