mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-03 19:44:39 -04:00
* feat: return 429 when backends are saturated When backends are at capacity (per-model max_concurrent or the process-wide --max-concurrent-backend-requests ceiling), the response was 503. The OpenAI SDK, litellm, and most agent harnesses key on 429 for rate-limit backoff and treat 503 as a hard error. Both saturation paths now return 429 with the existing Retry-After header and type: "rate_limit_error" in the JSON body. The per-model admission middleware keeps admission_rejected as the code field so existing alerts that match on it still fire. Non-saturation 503s are unchanged: model cold-loading (with progress body), model-load failure cooldown, PII detector fail-closed, and classifier unavailable. These mean "not ready" rather than "busy". Assisted-by: AGENT:regolo/glm5.2 [TOOL] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * fix: return 503 when scheduler has no available nodes When the scheduler cannot find any healthy node to serve a model — all nodes are full and eviction cannot free a slot, or a node_selector excludes every candidate — the error fell through to 500. A 500 tells clients something is broken when the condition is transient and retryable. The router now wraps these errors with a new ErrNoAvailableNodes sentinel. The HTTP error handler maps it to 503 via applyNoAvailableNodes, following the same pattern as applyBackendAdmission (429). Unrelated scheduler errors (DB timeouts, registry lookups) still return 500. Three return sites are wrapped: - resolveSelectorCandidates: selector matches zero healthy nodes - scheduleNewModel eviction-busy: all models have in-flight requests - scheduleNewModel eviction-failed: eviction itself errored The existing scheduleAndLoad wrapper ("no available nodes: %w") preserves the sentinel through the chain via errors.Is, as does ModelRouterAdapter. Assisted-by: AGENT:regolo/glm5.2 [TOOL] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * test(http): use Ginkgo for admission tests Replace forbidden testing.T calls with Ginkgo and Gomega so the lint check accepts the admission handler tests. Assisted-by: Codex:GPT-6 forbidigo * fix(middleware): show the recorded status for admission rejections The admission audit row now records 429, but the Middleware page still printed a hard-coded 503. Read the status from the event, and update the two package comments that still said 503. Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Assisted-by: Claude:claude-opus-5-5 [Claude Code] --------- Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com>
119 lines
4.0 KiB
Go
119 lines
4.0 KiB
Go
package middleware_test
|
|
|
|
import (
|
|
"context"
|
|
"net/http"
|
|
"net/http/httptest"
|
|
"strings"
|
|
"sync"
|
|
|
|
"github.com/labstack/echo/v4"
|
|
"github.com/mudler/LocalAI/core/config"
|
|
. "github.com/mudler/LocalAI/core/http/middleware"
|
|
"github.com/mudler/LocalAI/core/services/routing/admission"
|
|
"github.com/mudler/LocalAI/core/services/routing/pii"
|
|
|
|
. "github.com/onsi/ginkgo/v2"
|
|
. "github.com/onsi/gomega"
|
|
)
|
|
|
|
// recordingStore captures admission rows so the test can assert
|
|
// the audit trail without standing up the full pii event store.
|
|
type recordingStore struct {
|
|
mu sync.Mutex
|
|
events []pii.PIIEvent
|
|
}
|
|
|
|
func (r *recordingStore) Record(_ context.Context, e pii.PIIEvent) error {
|
|
r.mu.Lock()
|
|
defer r.mu.Unlock()
|
|
r.events = append(r.events, e)
|
|
return nil
|
|
}
|
|
func (r *recordingStore) List(_ context.Context, _ pii.ListQuery) ([]pii.PIIEvent, error) {
|
|
return nil, nil
|
|
}
|
|
func (r *recordingStore) Count(_ context.Context) (int, error) { return 0, nil }
|
|
func (r *recordingStore) Close() error { return nil }
|
|
|
|
func runAdmission(lim *admission.Limiter, store *recordingStore, cfg *config.ModelConfig, handler echo.HandlerFunc) (*httptest.ResponseRecorder, error) {
|
|
req := httptest.NewRequest(http.MethodPost, "/v1/chat/completions", strings.NewReader("{}"))
|
|
rec := httptest.NewRecorder()
|
|
c := echo.New().NewContext(req, rec)
|
|
c.Set(CONTEXT_LOCALS_KEY_MODEL_CONFIG, cfg)
|
|
mw := AdmissionControl(lim, store)
|
|
err := mw(handler)(c)
|
|
return rec, err
|
|
}
|
|
|
|
var _ = Describe("Admission", func() {
|
|
It("allows when under limit", func() {
|
|
lim := admission.New()
|
|
cfg := &config.ModelConfig{Limits: config.LimitsConfig{MaxConcurrent: 2}}
|
|
cfg.Name = "m"
|
|
rec, err := runAdmission(lim, &recordingStore{}, cfg, func(c echo.Context) error {
|
|
return c.String(http.StatusOK, "ok")
|
|
})
|
|
Expect(err).NotTo(HaveOccurred())
|
|
Expect(rec.Code).To(Equal(http.StatusOK))
|
|
})
|
|
|
|
It("rejects when full", func() {
|
|
// Saturate the limiter outside the middleware, then a request
|
|
// at the same model gets 429 with a Retry-After header.
|
|
lim := admission.New()
|
|
release, ok := lim.Acquire("busy", 1)
|
|
Expect(ok).To(BeTrue(), "setup acquire should succeed")
|
|
defer release()
|
|
|
|
cfg := &config.ModelConfig{Limits: config.LimitsConfig{MaxConcurrent: 1, RetryAfterSeconds: 3}}
|
|
cfg.Name = "busy"
|
|
store := &recordingStore{}
|
|
handlerCalled := false
|
|
rec, err := runAdmission(lim, store, cfg, func(c echo.Context) error {
|
|
handlerCalled = true
|
|
return c.String(http.StatusOK, "ok")
|
|
})
|
|
Expect(err).NotTo(HaveOccurred())
|
|
Expect(rec.Code).To(Equal(http.StatusTooManyRequests))
|
|
Expect(rec.Header().Get("Retry-After")).To(Equal("3"))
|
|
Expect(handlerCalled).To(BeFalse(), "handler should not run when admission rejects")
|
|
Expect(rec.Body.String()).To(ContainSubstring("admission_rejected"))
|
|
Expect(store.events).To(HaveLen(1))
|
|
Expect(store.events[0].Kind).To(Equal(pii.KindAdmission))
|
|
Expect(store.events[0].Host).To(Equal("busy"), "audit row carries the model name")
|
|
})
|
|
|
|
It("no limit configured is no-op", func() {
|
|
// MaxConcurrent=0 means unlimited — handler always runs and no
|
|
// audit row is written even after many calls.
|
|
lim := admission.New()
|
|
cfg := &config.ModelConfig{}
|
|
cfg.Name = "open"
|
|
store := &recordingStore{}
|
|
for i := 0; i < 10; i++ {
|
|
rec, err := runAdmission(lim, store, cfg, func(c echo.Context) error {
|
|
return c.String(http.StatusOK, "ok")
|
|
})
|
|
Expect(err).NotTo(HaveOccurred())
|
|
Expect(rec.Code).To(Equal(http.StatusOK))
|
|
}
|
|
Expect(store.events).To(BeEmpty())
|
|
})
|
|
|
|
It("releases after handler", func() {
|
|
// One slot, two SEQUENTIAL requests: the second succeeds because
|
|
// the first's release runs on handler return.
|
|
lim := admission.New()
|
|
cfg := &config.ModelConfig{Limits: config.LimitsConfig{MaxConcurrent: 1}}
|
|
cfg.Name = "tight"
|
|
for i := 0; i < 3; i++ {
|
|
rec, err := runAdmission(lim, &recordingStore{}, cfg, func(c echo.Context) error {
|
|
return c.String(http.StatusOK, "ok")
|
|
})
|
|
Expect(err).NotTo(HaveOccurred())
|
|
Expect(rec.Code).To(Equal(http.StatusOK))
|
|
}
|
|
})
|
|
})
|