Files
LocalAI/core/application/failover_test.go
T
Ettore Di Giacinto 5d2b90b848 fix(failover): never load a warm target inside a probe
A warm target's liveness probe called ModelLoader.Load, which blocked
until the model finished loading (while the warm preload loaded it
too). Tick waited for every probe, so all probing froze, and the probe
then ran HealthCheck on an expired context and tripped the target at
every startup.

The prober now takes a function that returns the running backend
without loading it. A target that is not loaded passes liveness; its
recovery is neither confirmed nor failed and it returns to healthy
after min_dwell, like a cold target. Tick no longer waits for probes:
each probe applies its own result and a target whose probe is running
is skipped.

Assisted-by: Claude:claude-opus-5-5
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
2026-09-27 07:42:20 +00:00

64 lines
2.4 KiB
Go

package application
import (
"context"
"time"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/pkg/grpc"
"github.com/mudler/LocalAI/pkg/model"
"github.com/mudler/LocalAI/pkg/system"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("applyFailoverWarmTargets", func() {
It("returns promptly even when the preload call blocks", func() {
// Guards against a regression to a synchronous preload loop: onWarm
// runs on the failover manager's single scheduler goroutine, so a
// blocking loader here must not block the caller.
started := make(chan struct{})
release := make(chan struct{})
orig := preloadModelByName
preloadModelByName = func(ctx context.Context, cl *config.ModelConfigLoader, ml *model.ModelLoader, appConfig *config.ApplicationConfig, name string) ([]string, error) {
close(started)
<-release // never released within the test's timeout
return nil, nil
}
DeferCleanup(func() { preloadModelByName = orig })
DeferCleanup(func() { close(release) })
app := &Application{applicationConfig: &config.ApplicationConfig{Context: context.Background()}}
callReturned := make(chan struct{})
go func() {
defer GinkgoRecover()
app.applyFailoverWarmTargets([]string{"warm-a"})
close(callReturned)
}()
Eventually(callReturned, time.Second).Should(BeClosed(), "applyFailoverWarmTargets must not wait on the preload goroutine")
Eventually(started, time.Second).Should(BeClosed(), "the preload goroutine should still run in the background")
})
})
type healthyBackend struct{ grpc.Backend }
func (healthyBackend) HealthCheck(context.Context) (bool, error) { return true, nil }
var _ = Describe("failoverLoadedBackend", func() {
It("returns the running backend and never loads a model that is not loaded", func() {
ml := model.NewModelLoader(&system.SystemState{Model: system.Model{ModelsPath: GinkgoT().TempDir()}})
store := model.NewInMemoryModelStore()
ml.SetModelStore(store)
loaded := failoverLoadedBackend(ml)
Expect(loaded(config.ModelConfig{Name: "gemma", Backend: "llama-cpp"})).To(BeNil())
Expect(ml.ListLoadedModels()).To(BeEmpty(), "the lookup must not start a load")
client := healthyBackend{}
store.Set("gemma", model.NewModelWithClient("gemma", "127.0.0.1:0", client))
Expect(loaded(config.ModelConfig{Name: "gemma", Backend: "llama-cpp"})).To(Equal(client))
})
})