mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-29 01:25:03 -04:00
fix(failover): warn about warm on a remote target, align the spec
The spec promised a load-time warning when a chain marks a remote target warm, where the flag does nothing; the loader now logs it. The remote-backend test moves into ModelConfig.IsRemoteProxy so the loader and the failover manager agree on what is remote. The spec now says what ships: a load blocked by pinned warm targets proceeds over the limit after eviction retries, without an error that names them. Assisted-by: Claude:claude-opus-5-5 Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
This commit is contained in:
1 parent
8dbe7a0e3f
commit
f7f12f8203
6 files changed
+53
-3
No files matched your search
@@ -49,6 +49,17 @@ const (
|
||||
// IsFailover reports whether this config is a failover chain.
|
||||
func (c ModelConfig) IsFailover() bool { return c.Failover != nil }
|
||||
|
||||
// IsRemoteProxy reports whether the model is served by a remote upstream
|
||||
// through a proxy backend. Failover probes such a target over HTTP, and
|
||||
// `warm` has no effect on it.
|
||||
func (c ModelConfig) IsRemoteProxy() bool {
|
||||
switch c.Backend {
|
||||
case "cloud-proxy", "localai-proxy":
|
||||
return true
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func (f FailoverConfig) ProbeInterval() time.Duration {
|
||||
return durationOr(f.Probe.Interval, DefaultFailoverProbeInterval)
|
||||
}
|
||||
|
||||
@@ -543,6 +543,28 @@ func validateFailoverTargets(cfg *ModelConfig, lookup func(string) (ModelConfig,
|
||||
return nil
|
||||
}
|
||||
|
||||
// failoverWarmRemoteTargets lists the chain's targets marked warm that are
|
||||
// remote. Warm only keeps a local model loaded, so the flag does nothing there.
|
||||
func failoverWarmRemoteTargets(cfg *ModelConfig, lookup func(string) (ModelConfig, bool)) []string {
|
||||
if cfg == nil || !cfg.IsFailover() {
|
||||
return nil
|
||||
}
|
||||
var out []string
|
||||
for _, t := range cfg.Failover.Targets {
|
||||
if !t.Warm {
|
||||
continue
|
||||
}
|
||||
target, ok := lookup(t.Model)
|
||||
if ok && target.IsAlias() {
|
||||
target, ok = lookup(target.Alias)
|
||||
}
|
||||
if ok && target.IsRemoteProxy() {
|
||||
out = append(out, t.Model)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func failoverTargetsShareUsecase(cfg *ModelConfig, lookup func(string) (ModelConfig, bool)) bool {
|
||||
if cfg == nil || !cfg.IsFailover() {
|
||||
return true
|
||||
@@ -931,6 +953,9 @@ func (bcl *ModelConfigLoader) loadModelConfigsFromPath(path string, strict bool,
|
||||
if !failoverTargetsShareUsecase(&c, lookup) {
|
||||
xlog.Warn("failover chain targets share no known usecase", "model", name)
|
||||
}
|
||||
if remote := failoverWarmRemoteTargets(&c, lookup); len(remote) > 0 {
|
||||
xlog.Warn("failover chain: warm has no effect on remote targets", "model", name, "targets", remote)
|
||||
}
|
||||
}
|
||||
|
||||
return nil
|
||||
|
||||
@@ -441,4 +441,14 @@ var _ = Describe("ModelConfigLoader failover validation", func() {
|
||||
Expect(loader.FailoverTargetsShareUsecase(chain("a", "b"))).To(BeTrue())
|
||||
Expect(loader.FailoverTargetsShareUsecase(chain("a", "tts"))).To(BeFalse())
|
||||
})
|
||||
It("finds warm targets that are remote, where warm has no effect", func() {
|
||||
loader.configs["remote"] = ModelConfig{Name: "remote", Backend: "cloud-proxy"}
|
||||
loader.configs["alias-remote"] = ModelConfig{Name: "alias-remote", Alias: "remote"}
|
||||
c := chain("remote", "alias-remote", "a")
|
||||
for i := range c.Failover.Targets {
|
||||
c.Failover.Targets[i].Warm = true
|
||||
}
|
||||
Expect(failoverWarmRemoteTargets(c, loader.GetModelConfig)).To(Equal([]string{"remote", "alias-remote"}))
|
||||
Expect(failoverWarmRemoteTargets(chain("remote", "a"), loader.GetModelConfig)).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
@@ -87,8 +87,7 @@ type ChainStatus struct {
|
||||
// KindOf decides how a target is probed: proxy backends forward to another
|
||||
// server and are checked over HTTP, everything else runs in this instance.
|
||||
func KindOf(cfg config.ModelConfig) Kind {
|
||||
switch cfg.Backend {
|
||||
case "cloud-proxy", "localai-proxy":
|
||||
if cfg.IsRemoteProxy() {
|
||||
return KindRemote
|
||||
}
|
||||
return KindLocal
|
||||
|
||||
@@ -107,6 +107,9 @@ count toward the active backend limit (`--max-active-backends`) like any
|
||||
pinned model: LocalAI never evicts them to make room, and if they fill the
|
||||
limit, a new model still loads rather than being blocked.
|
||||
|
||||
`warm` applies only to local targets. On a remote (`cloud-proxy`) target it
|
||||
has no effect, and LocalAI logs a warning when it loads the chain.
|
||||
|
||||
## Realtime pipelines
|
||||
|
||||
A pipeline stage can name a chain:
|
||||
|
||||
@@ -190,7 +190,9 @@ this.
|
||||
The manager loads `warm: true` targets at startup and marks them pinned in the
|
||||
watchdog, so LRU and idle eviction skip them. They still count toward the
|
||||
active backend limit. When pinned warm targets leave no room for another load,
|
||||
that load fails with an error that names them. The docs state this.
|
||||
the loader never evicts them: it retries eviction and then loads the model
|
||||
anyway, over the limit, with no error that names the warm targets. The docs
|
||||
state this.
|
||||
|
||||
### Events
|
||||
|
||||
|
||||
Reference in new issue
Block a user