mirror of
https://github.com/mudler/LocalAI.git
synced 2026-07-30 09:57:57 -04:00
fix(upgrade-check): don't filter upgrade candidates by controller capability (#11024)
CheckUpgradesAgainst resolved gallery entries through AvailableBackends, which drops every entry the *local* host cannot run. In distributed mode the host running the check is a CPU-only controller while the GPU backends live on worker nodes, so FindGalleryElement returned nil for every cuda/rocm/l4t entry and those backends were silently skipped. Measured on a live cluster: GET /backends reported 48 installed backends, POST /backends/upgrades/check evaluated 5 — all of them plain or cpu-prefixed. The 43 skipped were all hardware-specific builds. As a result cuda13-nvidia-l4t-arm64-longcat-video-development stayed at sha256:0b8dc851 while the registry tag held sha256:38dae6ff, and a cuDNN packaging fix sat unnoticed on a GPU worker for two days. Every name looked up here is already installed somewhere in the cluster, so hardware compatibility was decided at install time; re-deciding it against the controller is wrong. Switch both CheckUpgradesAgainst and UpgradeBackend to AvailableBackendsUnfiltered. Assisted-by: Claude Code:claude-opus-4-8[1m] [Read] [Edit] [Bash] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
This commit is contained in:
committed by
GitHub
parent
3584e0776d
commit
2b61e4bc1d
@@ -58,7 +58,15 @@ func CheckBackendUpgrades(ctx context.Context, galleries []config.Gallery, syste
|
||||
// row is flagged upgradeable regardless of whether any node matches the gallery
|
||||
// — next Upgrade All realigns the cluster. NodeDrift lists the outliers.
|
||||
func CheckUpgradesAgainst(ctx context.Context, galleries []config.Gallery, systemState *system.SystemState, installedBackends SystemBackends) (map[string]UpgradeInfo, error) {
|
||||
galleryBackends, err := AvailableBackends(galleries, systemState)
|
||||
// Unfiltered on purpose. AvailableBackends drops every gallery entry the
|
||||
// *local* host cannot run, and in distributed mode the host running this
|
||||
// check is a CPU-only controller while the GPU backends live on worker
|
||||
// nodes. Filtering there made FindGalleryElement return nil for every
|
||||
// cuda/rocm/l4t entry, so those backends were silently skipped and never
|
||||
// reported an upgrade. Every name we look up here is already installed
|
||||
// somewhere in the cluster, so hardware compatibility has been decided at
|
||||
// install time and re-deciding it against the controller is wrong.
|
||||
galleryBackends, err := AvailableBackendsUnfiltered(galleries, systemState)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to list available backends: %w", err)
|
||||
}
|
||||
@@ -254,8 +262,10 @@ func UpgradeBackend(ctx context.Context, systemState *system.SystemState, modelL
|
||||
return UpgradeBackend(ctx, systemState, modelLoader, galleries, installed.Metadata.MetaBackendFor, downloadStatus, requireIntegrity)
|
||||
}
|
||||
|
||||
// Find the gallery entry
|
||||
galleryBackends, err := AvailableBackends(galleries, systemState)
|
||||
// Find the gallery entry. Unfiltered for the same reason as the check
|
||||
// above: backendName is already installed here, so the capability filter
|
||||
// can only reject an entry we must be able to resolve.
|
||||
galleryBackends, err := AvailableBackendsUnfiltered(galleries, systemState)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to list available backends: %w", err)
|
||||
}
|
||||
|
||||
@@ -359,6 +359,47 @@ var _ = Describe("Upgrade Detection and Execution", func() {
|
||||
Expect(upgrades).To(HaveKey("my-backend-development"))
|
||||
Expect(upgrades).NotTo(HaveKey("my-alias"))
|
||||
})
|
||||
|
||||
// Hardware-specific backends live on GPU worker nodes while the
|
||||
// controller is typically a CPU-only pod. The gallery candidate set
|
||||
// must therefore NOT be filtered by the controller's own capability:
|
||||
// doing so drops every cuda/rocm/l4t entry, FindGalleryElement
|
||||
// returns nil, and the backend is silently skipped. Observed live:
|
||||
// 48 installed backends, only 5 (all CPU) ever evaluated, while
|
||||
// cuda13-nvidia-l4t-arm64-longcat-video-development sat two versions
|
||||
// behind on a worker.
|
||||
It("flags a GPU-only backend installed on a worker even though the controller is CPU-only", func() {
|
||||
cpuOnlyState := system.NewCapabilityState("default",
|
||||
system.WithBackendPath(backendsPath))
|
||||
|
||||
writeGalleryYAML([]GalleryBackend{
|
||||
{
|
||||
Metadata: Metadata{Name: "cuda13-nvidia-l4t-arm64-longcat-video-development"},
|
||||
URI: filepath.Join(tempDir, "gpu-source"),
|
||||
Version: "2.0.0",
|
||||
},
|
||||
})
|
||||
|
||||
installed := SystemBackends{
|
||||
"cuda13-nvidia-l4t-arm64-longcat-video-development": SystemBackend{
|
||||
Name: "cuda13-nvidia-l4t-arm64-longcat-video-development",
|
||||
Metadata: &BackendMetadata{
|
||||
Name: "cuda13-nvidia-l4t-arm64-longcat-video-development",
|
||||
Version: "1.0.0",
|
||||
},
|
||||
Nodes: []NodeBackendRef{
|
||||
{NodeID: "a", NodeName: "gpu-worker-1", Version: "1.0.0"},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
upgrades, err := CheckUpgradesAgainst(context.Background(), galleries, cpuOnlyState, installed)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(upgrades).To(HaveKey("cuda13-nvidia-l4t-arm64-longcat-video-development"))
|
||||
info := upgrades["cuda13-nvidia-l4t-arm64-longcat-video-development"]
|
||||
Expect(info.InstalledVersion).To(Equal("1.0.0"))
|
||||
Expect(info.AvailableVersion).To(Equal("2.0.0"))
|
||||
})
|
||||
})
|
||||
|
||||
Describe("UpgradeBackend", func() {
|
||||
|
||||
Reference in New Issue
Block a user