Files
LocalAI/core/backend/global_admission.go
Richard Palethorpe 799cc9f211 feat: bound global admission and expose running backend traces (#11560)
feat: bound backend admission and expose running traces

Add process-wide backend execution admission without blocking UI or administrative HTTP work. Represent backend operations while they are in flight, surface running traces with immediate log links, and tie streaming admission leases to the gRPC receive lifecycle.

Assisted-by: OpenAI Codex: GPT-5

Signed-off-by: Richard Palethorpe <io@richiejp.com>
2026-08-18 08:56:59 +02:00

73 lines
2.1 KiB
Go

// SPDX-License-Identifier: MIT
package backend
import (
"fmt"
"sync"
"time"
"github.com/mudler/LocalAI/core/config"
)
// BackendAdmissionError reports that the process-wide backend execution
// ceiling is full. HTTP callers map it to 503; internal callers receive the
// same typed error instead of silently queueing and growing in-flight state.
type BackendAdmissionError struct {
Limit int
RetryAfter time.Duration
}
func (e *BackendAdmissionError) Error() string {
return fmt.Sprintf("backend inference capacity reached (max_concurrent=%d); retry after %s", e.Limit, e.RetryAfter)
}
var backendAdmission = struct {
sync.RWMutex
limit int
slots chan struct{}
}{}
// ConfigureGlobalBackendAdmission sets the process-wide ceiling. It is called
// during application construction, before backend work can begin.
func ConfigureGlobalBackendAdmission(limit int) {
if limit <= 0 {
limit = config.DefaultMaxConcurrentBackendRequests
}
backendAdmission.Lock()
backendAdmission.limit = limit
backendAdmission.slots = make(chan struct{}, limit)
backendAdmission.Unlock()
}
// AcquireGlobalBackendSlot admits one backend operation without queueing.
// Callers must invoke release on every completion path.
func AcquireGlobalBackendSlot() (release func(), err error) {
backendAdmission.RLock()
limit, slots := backendAdmission.limit, backendAdmission.slots
backendAdmission.RUnlock()
if slots == nil {
backendAdmission.Lock()
if backendAdmission.slots == nil {
backendAdmission.limit = config.DefaultMaxConcurrentBackendRequests
backendAdmission.slots = make(chan struct{}, backendAdmission.limit)
}
limit, slots = backendAdmission.limit, backendAdmission.slots
backendAdmission.Unlock()
}
select {
case slots <- struct{}{}:
var once sync.Once
return func() { once.Do(func() { <-slots }) }, nil
default:
return nil, &BackendAdmissionError{Limit: limit, RetryAfter: time.Second}
}
}
// GlobalBackendInFlight is the current number of admitted backend operations.
func GlobalBackendInFlight() int {
backendAdmission.RLock()
defer backendAdmission.RUnlock()
return len(backendAdmission.slots)
}