diff --git a/core/backend/global_admission.go b/core/backend/global_admission.go index 9b216ea83..7ee69aeb1 100644 --- a/core/backend/global_admission.go +++ b/core/backend/global_admission.go @@ -11,8 +11,9 @@ import ( ) // BackendAdmissionError reports that the process-wide backend execution -// ceiling is full. HTTP callers map it to 503; internal callers receive the -// same typed error instead of silently queueing and growing in-flight state. +// ceiling is full. HTTP callers map it to 429 (Too Many Requests) with a +// Retry-After header; internal callers receive the same typed error instead +// of silently queueing and growing in-flight state. type BackendAdmissionError struct { Limit int RetryAfter time.Duration diff --git a/core/http/admission_handler_test.go b/core/http/admission_handler_test.go new file mode 100644 index 000000000..6a0a2201d --- /dev/null +++ b/core/http/admission_handler_test.go @@ -0,0 +1,70 @@ +package http + +import ( + "errors" + "fmt" + "net/http" + "net/http/httptest" + "time" + + "github.com/labstack/echo/v4" + corebackend "github.com/mudler/LocalAI/core/backend" + "github.com/mudler/LocalAI/core/services/nodes" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("Backend admission", func() { + It("maps BackendAdmissionError to 429 with Retry-After", func() { + e := echo.New() + req := httptest.NewRequest(http.MethodPost, "/", nil) + rec := httptest.NewRecorder() + c := e.NewContext(req, rec) + + err := &corebackend.BackendAdmissionError{Limit: 4, RetryAfter: 3 * time.Second} + code := applyBackendAdmission(err, http.StatusInternalServerError, c) + + Expect(code).To(Equal(http.StatusTooManyRequests)) + Expect(rec.Header().Get("Retry-After")).To(Equal("3")) + }) + + It("passes through non-admission errors unchanged", func() { + e := echo.New() + req := httptest.NewRequest(http.MethodPost, "/", nil) + rec := httptest.NewRecorder() + c := e.NewContext(req, rec) + + code := applyBackendAdmission(errors.New("some other error"), http.StatusInternalServerError, c) + Expect(code).To(Equal(http.StatusInternalServerError)) + Expect(rec.Header().Get("Retry-After")).To(BeEmpty()) + }) +}) + +var _ = Describe("No available nodes", func() { + It("maps ErrNoAvailableNodes to 503", func() { + // The scheduler wraps the sentinel in fmt.Errorf chains and via + // errors.Join — errors.Is must still find it. + wrapped := fmt.Errorf("routing model foo: %w", + fmt.Errorf("no available nodes: %w", + fmt.Errorf("no healthy nodes available: %w", + errors.Join(nodes.ErrEvictionBusy, nodes.ErrNoAvailableNodes)))) + + code := applyNoAvailableNodes(wrapped, http.StatusInternalServerError) + Expect(code).To(Equal(http.StatusServiceUnavailable)) + }) + + It("maps selector-mismatch chain to 503", func() { + wrapped := fmt.Errorf("routing model bar: %w", + fmt.Errorf("no available nodes: %w", + fmt.Errorf("no healthy nodes match selector for model bar: {\"gpu.vendor\":\"tpu\"}: %w", + nodes.ErrNoAvailableNodes))) + + code := applyNoAvailableNodes(wrapped, http.StatusInternalServerError) + Expect(code).To(Equal(http.StatusServiceUnavailable)) + }) + + It("passes through unrelated errors unchanged", func() { + code := applyNoAvailableNodes(errors.New("database timeout"), http.StatusInternalServerError) + Expect(code).To(Equal(http.StatusInternalServerError)) + }) +}) diff --git a/core/http/app.go b/core/http/app.go index f03522a45..c94b323e4 100644 --- a/core/http/app.go +++ b/core/http/app.go @@ -85,7 +85,20 @@ func applyBackendAdmission(err error, code int, c echo.Context) int { return code } c.Response().Header().Set("Retry-After", strconv.Itoa(int(capacityErr.RetryAfter.Seconds()))) - return http.StatusServiceUnavailable + return http.StatusTooManyRequests +} + +// applyNoAvailableNodes maps scheduler "no available nodes" errors to 503. +// When the cluster has no healthy node to serve a model — all are full, a +// node selector excludes every candidate, or eviction could not free a slot — +// the request is retryable, not a server bug. Without this the error fell +// through to 500, which tells clients something is broken when they just +// need to wait for a node. +func applyNoAvailableNodes(err error, code int) int { + if errors.Is(err, nodes.ErrNoAvailableNodes) { + return http.StatusServiceUnavailable + } + return code } // respondModelLoading answers a request whose model is still cold-loading with @@ -208,6 +221,7 @@ func API(application *application.Application) (*echo.Echo, error) { } code = applyModelLoadCooldown(err, code, c) code = applyBackendAdmission(err, code, c) + code = applyNoAvailableNodes(err, code) // Handle 404 errors: serve React SPA for HTML requests, JSON otherwise if code == http.StatusNotFound { @@ -224,8 +238,13 @@ func API(application *application.Application) (*echo.Echo, error) { } // Send custom error page + errType := "" + var capErr *corebackend.BackendAdmissionError + if errors.As(err, &capErr) { + errType = "rate_limit_error" + } c.JSON(code, schema.ErrorResponse{ - Error: &schema.APIError{Message: err.Error(), Code: code}, + Error: &schema.APIError{Message: err.Error(), Code: code, Type: errType}, }) } } else { @@ -240,6 +259,7 @@ func API(application *application.Application) (*echo.Echo, error) { // Opaque errors deliberately withhold the body, so a still-loading // model gets the status and Retry-After but no progress detail. code = applyModelLoading(err, code, c) + code = applyNoAvailableNodes(err, code) c.NoContent(code) } } diff --git a/core/http/middleware/admission.go b/core/http/middleware/admission.go index c79066925..d6134b026 100644 --- a/core/http/middleware/admission.go +++ b/core/http/middleware/admission.go @@ -20,7 +20,7 @@ import ( // SERVED model — a router fanout that lands on a saturated downstream // model gets rejected even though the requested router-model has slack. // -// On reject: HTTP 503, Retry-After header, error JSON. An audit row +// On reject: HTTP 429, Retry-After header, error JSON. An audit row // goes into the shared event store under KindAdmission so admins see // rejection rates alongside PII and proxy events. // @@ -39,9 +39,10 @@ func AdmissionControl(limiter *admission.Limiter, events pii.EventStore) echo.Mi retryAfter := admission.RetryAfter(cfg.Limits.RetryAfterSeconds) recordAdmissionRejection(events, cfg.Name, retryAfter) c.Response().Header().Set("Retry-After", strconv.Itoa(int(retryAfter.Seconds()))) - return c.JSON(http.StatusServiceUnavailable, map[string]any{ + return c.JSON(http.StatusTooManyRequests, map[string]any{ "error": map[string]any{ - "type": "admission_rejected", + "type": "rate_limit_error", + "code": "admission_rejected", "message": fmt.Sprintf("model %q is at capacity (max_concurrent=%d); retry after %s", cfg.Name, max, retryAfter), }, }) @@ -61,7 +62,7 @@ func recordAdmissionRejection(events pii.EventStore, modelName string, retryAfte if events == nil { return } - statusCode := http.StatusServiceUnavailable + statusCode := http.StatusTooManyRequests durMS := retryAfter.Milliseconds() id := fmt.Sprintf("adm_%d_%s", admissionEventSeq.Add(1), randHex(4)) _ = events.Record(context.Background(), pii.PIIEvent{ diff --git a/core/http/middleware/admission_test.go b/core/http/middleware/admission_test.go index 841a2dd47..1e8649c2f 100644 --- a/core/http/middleware/admission_test.go +++ b/core/http/middleware/admission_test.go @@ -60,7 +60,7 @@ var _ = Describe("Admission", func() { It("rejects when full", func() { // Saturate the limiter outside the middleware, then a request - // at the same model gets 503 with a Retry-After header. + // at the same model gets 429 with a Retry-After header. lim := admission.New() release, ok := lim.Acquire("busy", 1) Expect(ok).To(BeTrue(), "setup acquire should succeed") @@ -75,7 +75,7 @@ var _ = Describe("Admission", func() { return c.String(http.StatusOK, "ok") }) Expect(err).NotTo(HaveOccurred()) - Expect(rec.Code).To(Equal(http.StatusServiceUnavailable)) + Expect(rec.Code).To(Equal(http.StatusTooManyRequests)) Expect(rec.Header().Get("Retry-After")).To(Equal("3")) Expect(handlerCalled).To(BeFalse(), "handler should not run when admission rejects") Expect(rec.Body.String()).To(ContainSubstring("admission_rejected")) diff --git a/core/http/react-ui/src/pages/Middleware.jsx b/core/http/react-ui/src/pages/Middleware.jsx index 34d55d049..363b35f61 100644 --- a/core/http/react-ui/src/pages/Middleware.jsx +++ b/core/http/react-ui/src/pages/Middleware.jsx @@ -931,7 +931,8 @@ function eventDetails(e) { } case 'admission': { const retry = e.duration_ms != null ? `retry-after ${Math.round(e.duration_ms / 1000)}s` : '' - return `HTTP 503 rejected · ${retry}` + // Older audit rows were recorded as 503; newer ones as 429. + return `HTTP ${e.status_code || 429} rejected · ${retry}` } default: { const len = e.length != null ? `len ${e.length}` : '' diff --git a/core/services/nodes/router.go b/core/services/nodes/router.go index d024b5621..36eee4a47 100644 --- a/core/services/nodes/router.go +++ b/core/services/nodes/router.go @@ -955,7 +955,7 @@ func (r *SmartRouter) resolveSelectorCandidates(ctx context.Context, modelID str return nil, fmt.Errorf("looking up nodes for selector %s: %w", sched.NodeSelector, err) } if len(candidates) == 0 { - return nil, fmt.Errorf("no healthy nodes match selector for model %s: %s", modelID, sched.NodeSelector) + return nil, fmt.Errorf("no healthy nodes match selector for model %s: %s: %w", modelID, sched.NodeSelector, ErrNoAvailableNodes) } return extractNodeIDs(candidates), nil } @@ -1167,9 +1167,9 @@ func (r *SmartRouter) scheduleNewModel(ctx context.Context, backendType, modelID evictedNode, evictErr := r.evictLRUAndFreeNodeFrom(ctx, candidateNodeIDs) if evictErr != nil { if errors.Is(evictErr, ErrEvictionBusy) { - return nil, "", 0, fmt.Errorf("no healthy nodes available: %w", evictErr) + return nil, "", 0, fmt.Errorf("no healthy nodes available: %w", errors.Join(evictErr, ErrNoAvailableNodes)) } - return nil, "", 0, fmt.Errorf("no healthy nodes available and eviction failed: %w", evictErr) + return nil, "", 0, fmt.Errorf("no healthy nodes available and eviction failed: %w", errors.Join(evictErr, ErrNoAvailableNodes)) } node = evictedNode } @@ -2059,6 +2059,13 @@ func (r *SmartRouter) EvictLRU(ctx context.Context, nodeID string) (string, erro // and none can be evicted to make room. var ErrEvictionBusy = errors.New("all models busy, cannot evict") +// ErrNoAvailableNodes is returned when the scheduler cannot find any healthy +// node to serve a model — all nodes are full and eviction cannot free a slot, +// or a node selector excludes every candidate. The HTTP layer maps this to +// 503 so clients treat it as a transient condition rather than a server bug +// (which is what 500 would imply). +var ErrNoAvailableNodes = errors.New("no available nodes") + // evictLRUAndFreeNode finds the globally least-recently-used model with zero in-flight, // unloads it, and returns its node for reuse. If all models are busy, retries briefly. // diff --git a/core/services/nodes/router_test.go b/core/services/nodes/router_test.go index 3735ec5e4..7577e2846 100644 --- a/core/services/nodes/router_test.go +++ b/core/services/nodes/router_test.go @@ -843,6 +843,26 @@ var _ = Describe("SmartRouter", func() { Expect(err).To(HaveOccurred()) Expect(err.Error()).To(ContainSubstring("no available nodes")) }) + + It("wraps ErrNoAvailableNodes when all nodes are full and eviction cannot help", func() { + // gorm.ErrRecordNotFound is the registry's verdict that no node + // matches — the scheduler then falls through to eviction. With + // DB nil, eviction returns ErrEvictionBusy, and the scheduler + // wraps the error with ErrNoAvailableNodes so the HTTP layer can + // map it to 503 instead of 500. + reg.findIdleErr = errors.New("no idle") + reg.findLeastLoadedErr = gorm.ErrRecordNotFound + + router := NewSmartRouter(reg, SmartRouterOptions{ + Unloader: unloader, + ClientFactory: factory, + }) + + _, err := router.Route(context.Background(), "m5", "models/m5.gguf", "llama-cpp", "", nil, false) + Expect(err).To(HaveOccurred()) + Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue()) + Expect(errors.Is(err, ErrEvictionBusy)).To(BeTrue()) + }) }) Describe("UnloadModel (mock-based)", func() { @@ -970,6 +990,7 @@ var _ = Describe("SmartRouter", func() { _, err := router.Route(context.Background(), "aliased-model", "models/aliased.gguf", "llama-cpp", "", nil, false) Expect(err).To(HaveOccurred()) Expect(err.Error()).To(ContainSubstring("no healthy nodes match selector")) + Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue()) }) It("returns error when no nodes match selector", func() { @@ -988,6 +1009,7 @@ var _ = Describe("SmartRouter", func() { _, err := router.Route(context.Background(), "no-match-model", "models/nomatch.gguf", "llama-cpp", "", nil, false) Expect(err).To(HaveOccurred()) Expect(err.Error()).To(ContainSubstring("no healthy nodes match selector")) + Expect(errors.Is(err, ErrNoAvailableNodes)).To(BeTrue()) }) It("uses regular methods when model has no scheduling config", func() { diff --git a/core/services/routing/admission/admission.go b/core/services/routing/admission/admission.go index 168248181..f37273373 100644 --- a/core/services/routing/admission/admission.go +++ b/core/services/routing/admission/admission.go @@ -1,6 +1,6 @@ // Package admission is routing-module subsystem 5: per-model // concurrency control + audit. The middleware acquires a slot -// before the handler runs; on full, the request gets 503 with +// before the handler runs; on full, the request gets 429 with // Retry-After so clients back off rather than pile on. The audit // row goes into the shared event store alongside PII and proxy // rows so admins see a single timeline of routing pressure. diff --git a/core/services/routing/pii/types.go b/core/services/routing/pii/types.go index c2e2510df..ad15ea462 100644 --- a/core/services/routing/pii/types.go +++ b/core/services/routing/pii/types.go @@ -109,7 +109,7 @@ const ( // model's MaxConcurrent ceiling is full. The Host field carries // the model name (overloading the existing column rather than // adding a new one — admins read it as "the thing that was - // busy"); StatusCode is 503. + // busy"); StatusCode is 429. KindAdmission EventKind = "admission" ) diff --git a/docs/content/reference/api-errors.md b/docs/content/reference/api-errors.md index 9bd9dea02..20f63786f 100644 --- a/docs/content/reference/api-errors.md +++ b/docs/content/reference/api-errors.md @@ -88,7 +88,9 @@ The `/v1/responses` endpoint returns errors with this structure: | 404 | Not Found | Model or resource does not exist | | 409 | Conflict | Resource already exists (e.g., duplicate token) | | 422 | Unprocessable Entity | Validation failed (e.g., invalid parameter range) | +| 429 | Too Many Requests | All backends are saturated (per-model `max_concurrent` or process-wide `--max-concurrent-backend-requests` ceiling reached). Includes a `Retry-After` header and `type: "rate_limit_error"` so OpenAI-compatible clients and harnesses back off automatically | | 500 | Internal Server Error | Backend inference failure, unexpected server errors | +| 503 | Service Unavailable | No healthy node available to serve the model (cluster is full, eviction cannot free a slot, or a `node_selector` excludes all candidates). Also used during model-load cooldown and while a model is still cold-loading. Retryable | ## Global Error Handling diff --git a/docs/content/reference/cli-reference.md b/docs/content/reference/cli-reference.md index 8cf47e74a..73fa07722 100644 --- a/docs/content/reference/cli-reference.md +++ b/docs/content/reference/cli-reference.md @@ -95,7 +95,7 @@ For more information on VRAM management, see [VRAM and Memory Management]({{%rel | Parameter | Default | Description | Environment Variable | |-----------|---------|-------------|----------------------| | `--address` | `:8080` | Bind address for the API server | `$LOCALAI_ADDRESS`, `$ADDRESS` | -| `--max-concurrent-backend-requests` | `1024` | Process-wide ceiling for concurrent backend inference operations. Excess inference receives HTTP 503 with `Retry-After`; UI and administrative endpoints remain available | `$LOCALAI_MAX_CONCURRENT_BACKEND_REQUESTS`, `$MAX_CONCURRENT_BACKEND_REQUESTS` | +| `--max-concurrent-backend-requests` | `1024` | Process-wide ceiling for concurrent backend inference operations. Excess inference receives HTTP 429 with `Retry-After`; UI and administrative endpoints remain available | `$LOCALAI_MAX_CONCURRENT_BACKEND_REQUESTS`, `$MAX_CONCURRENT_BACKEND_REQUESTS` | | `--cors` | `false` | Enable CORS (Cross-Origin Resource Sharing) | `$LOCALAI_CORS`, `$CORS` | | `--cors-allow-origins` | | Comma-separated list of allowed CORS origins | `$LOCALAI_CORS_ALLOW_ORIGINS`, `$CORS_ALLOW_ORIGINS` | | `--disable-csrf` | `false` | Disable CSRF middleware (enabled by default) | `$LOCALAI_DISABLE_CSRF` | diff --git a/docs/content/reference/runtime-errors.md b/docs/content/reference/runtime-errors.md index 1e14dc0b5..fd5ef2bd8 100644 --- a/docs/content/reference/runtime-errors.md +++ b/docs/content/reference/runtime-errors.md @@ -21,8 +21,9 @@ The left column is the literal string as it appears in the LocalAI server log (o | `grpc service not ready` | The backend process was spawned but its gRPC server did not become healthy in time (slow start, crash on startup, or the process died while loading). When a local backend has already exited, the error includes its exit code and last stderr line. | Use the included stderr diagnostic when present; otherwise check the log lines just above. A crash here often means out of memory, a missing shared library, or an incompatible CPU (see `SIGILL`). Increase available RAM/VRAM or pick a smaller quantization. | | `failed to load model: ...` | Returned by the load endpoints and several feature paths (voice, realtime, audio transform) when the model config could not be resolved or the backend load failed. | Confirm the model name exists (`local-ai models list`) and its YAML is valid. The trailing text carries the specific reason. | | HTTP `503` with a `Retry-After` header, after a load failed | Model-load failure cooldown. After a model fails to load, LocalAI refuses new load attempts for that model for a short window so a client that keeps polling a broken model does not respawn a crashing backend on every request. The window starts at `--model-load-failure-cooldown` (default `10s`) and doubles per consecutive failure up to 5m; it resets on the first success. | Fix the underlying load failure (see the rows above), then wait out the `Retry-After` seconds before retrying, or restart LocalAI to clear the cooldown. Set `--model-load-failure-cooldown 0` (or `LOCALAI_MODEL_LOAD_FAILURE_COOLDOWN=0`) to disable the cooldown entirely. See {{% relref "reference/cli-reference" %}}. | -| HTTP `503` with a `Retry-After` header, under load | Per-model concurrency limit reached. When a model config sets a `MaxConcurrent` limit, extra requests are rejected with `503` and a `Retry-After` (whole seconds, floor 1) instead of queueing. | Retry after the advised delay, raise the model's concurrency limit, or run more replicas. | -| HTTP `503` when backend inference is saturated | The process-wide `--max-concurrent-backend-requests` backend-execution ceiling is full. This protects inference and in-flight backend-trace memory without blocking UI or administrative endpoints. | Retry after the advised delay, reduce inference concurrency, raise the limit if the host has capacity, or add replicas. | +| HTTP `503` with `no available nodes` or `no healthy nodes match selector` | The scheduler could not find any healthy node to serve the model. All nodes are full and eviction cannot free a slot, or a `node_selector` in the model's scheduling config excludes every candidate. | Retry after a node becomes available or an in-flight request completes and frees a slot. In a cluster, add nodes or replicas. If a selector is set, confirm at least one healthy node matches it. | +| HTTP `429` with a `Retry-After` header, under load | Per-model concurrency limit reached. When a model config sets a `MaxConcurrent` limit, extra requests are rejected with `429` and a `Retry-After` (whole seconds, floor 1) instead of queueing. | Retry after the advised delay, raise the model's concurrency limit, or run more replicas. | +| HTTP `429` when backend inference is saturated | The process-wide `--max-concurrent-backend-requests` backend-execution ceiling is full. This protects inference and in-flight backend-trace memory without blocking UI or administrative endpoints. | Retry after the advised delay, reduce inference concurrency, raise the limit if the host has capacity, or add replicas. | | `invalid pitch` (with CUDA) | The prompt exceeded the model's context size. | Reduce the prompt length, or raise the model's context size (`context_size:` in the model YAML). | | `SIGILL` (illegal instruction) on startup | The prebuilt backend binary uses CPU instructions your CPU does not have (for example AVX512, AVX2, F16C, FMA). | Rebuild the backend for your CPU. In a container, set `REBUILD=true` and disable the unsupported instructions, for example `CMAKE_ARGS="-DGGML_F16C=OFF -DGGML_AVX512=OFF -DGGML_AVX2=OFF -DGGML_FMA=OFF" make build`. | | CUDA / VRAM out of memory (backend log shows `out of memory`, `CUDA error: out of memory`, or the process is killed loading) | The model plus its KV cache does not fit in GPU memory. | Use a smaller quantization, reduce `context_size:`, offload fewer layers to the GPU (lower `gpu_layers:`), or free VRAM held by other processes. On multi-GPU hosts, confirm the model is not trying to load entirely onto one device. |