mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-29 09:35:02 -04:00
Assisted-by: Claude:claude-opus-5-5 Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
60 lines
2.5 KiB
Go
60 lines
2.5 KiB
Go
package application
|
|
|
|
import (
|
|
"github.com/mudler/LocalAI/core/backend"
|
|
"github.com/mudler/LocalAI/core/config"
|
|
"github.com/mudler/LocalAI/core/services/failover"
|
|
"github.com/mudler/LocalAI/pkg/grpc"
|
|
"github.com/mudler/LocalAI/pkg/model"
|
|
"github.com/mudler/xlog"
|
|
)
|
|
|
|
// preloadModelByName is a seam over backend.PreloadModelByName so tests can
|
|
// substitute a controllable loader instead of touching real models/disk.
|
|
var preloadModelByName = backend.PreloadModelByName
|
|
|
|
// applyFailoverWarmTargets pins warm failover targets in the watchdog and
|
|
// loads them, so a switch does not wait for a cold load.
|
|
//
|
|
// SyncPinnedModelsToWatchdog runs synchronously: it is a cheap in-memory
|
|
// update, and the pin must land before the watchdog can evict a target that
|
|
// is about to become (or stay) a chain's active path. Preloading is not
|
|
// cheap — it can download or load a multi-GB model — and this callback runs
|
|
// on the failover manager's single scheduler goroutine (Sync, called from
|
|
// Tick, called from Run), before that tick's probes fire. A slow or hung
|
|
// load here would freeze probing and fail-back for every chain, so it runs
|
|
// in its own goroutine instead of blocking the scheduler loop.
|
|
//
|
|
// In distributed mode only the probe leader preloads: every frontend would
|
|
// otherwise ask the workers for the same load. Followers still pin, so their
|
|
// own watchdog never evicts a warm target they happen to hold.
|
|
func (a *Application) applyFailoverWarmTargets(warm []string) {
|
|
a.SyncPinnedModelsToWatchdog()
|
|
if a.IsDistributed() && !a.failoverManager.IsLeader() {
|
|
return
|
|
}
|
|
go func() {
|
|
for _, name := range warm {
|
|
if _, err := preloadModelByName(a.ApplicationConfig().Context, a.ModelConfigLoader(), a.ModelLoader(), a.ApplicationConfig(), name); err != nil {
|
|
xlog.Warn("failover: could not preload warm target", "model", name, "error", err)
|
|
}
|
|
}
|
|
}()
|
|
}
|
|
|
|
// failoverLoadedBackend gives the failover prober the running backend of a
|
|
// local target without ever loading it. CheckIsLoaded may run the loader's
|
|
// own health check and drop a dead process; the target is then "not loaded"
|
|
// and the next real request loads and judges it.
|
|
func failoverLoadedBackend(ml *model.ModelLoader) failover.LoadedFunc {
|
|
return func(cfg config.ModelConfig) grpc.Backend {
|
|
m := ml.CheckIsLoaded(cfg.ModelID())
|
|
if m == nil {
|
|
return nil
|
|
}
|
|
// Load always enables parallel requests; match it in case this is
|
|
// the first client built for the model.
|
|
return m.GRPC(true, ml.GetWatchDog())
|
|
}
|
|
}
|