mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-30 18:14:32 -04:00
fix(distributed): backend discovery hid worker-installed backends Backend discovery endpoints filter on installed-state, which on a distributed controller derives from the controller's own filesystem. A backend lives on the worker node that runs it, so every backend an admin installed on a GPU worker read as "not installed" and vanished from the listing. #10947 fixed the sibling capability filter on the same endpoints, so a fine-tuning-capable GPU worker now made the backend listable while the installed-state filter still dropped it: the dropdown stayed empty. The controller cannot derive this locally, but it already aggregates the per-node view that GET /backends renders, so discovery reuses the active BackendManager rather than growing a second path. Three surfaces shared the root cause and route through the same helper now: - GET /backends/available (Installed is now cluster-wide) - GET /api/fine-tuning/backends - GET /api/quantization/backends The response stays a boolean rather than an installed-on-N-of-M count: per-node install state is already served by GET /backends nodes[], and per-node control by POST /api/nodes/:id/backends/install, so a summary is all these dropdowns need. A nil provider (single-node) leaves the local filesystem as the only source and reproduces today's listing exactly, and a registry error degrades to that same listing instead of blanking the catalog. Assisted-by: Claude:claude-opus-4-8 golangci-lint Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
53 lines
1.9 KiB
Go
53 lines
1.9 KiB
Go
package routes
|
|
|
|
import (
|
|
"context"
|
|
|
|
"github.com/mudler/LocalAI/core/application"
|
|
"github.com/mudler/LocalAI/core/http/endpoints/localai"
|
|
)
|
|
|
|
// ClusterCapabilityProviderFor returns the capability source backing every
|
|
// capability-filtered backend discovery endpoint, or nil in single-node mode.
|
|
//
|
|
// A nil provider makes those endpoints filter against the local system exactly
|
|
// as they always have. In distributed mode the controller is typically a
|
|
// GPU-less pod, so discovery instead unions the capabilities of the healthy
|
|
// worker nodes that would actually run the backend.
|
|
func ClusterCapabilityProviderFor(app *application.Application) localai.ClusterCapabilityProvider {
|
|
if app == nil || !app.IsDistributed() || app.Distributed().Registry == nil {
|
|
return nil
|
|
}
|
|
return app.Distributed().Registry.HealthyBackendCapabilities
|
|
}
|
|
|
|
// ClusterInstalledProviderFor returns the install-state source backing every
|
|
// backend discovery endpoint that filters on installed backends, or nil in
|
|
// single-node mode.
|
|
//
|
|
// A nil provider leaves those endpoints reading the local filesystem exactly as
|
|
// they always have. In distributed mode a backend lives on the worker that runs
|
|
// it, so the controller's own disk cannot answer the question; the active
|
|
// BackendManager already aggregates the per-node view that GET /backends
|
|
// renders, and discovery reuses it rather than growing a second path.
|
|
func ClusterInstalledProviderFor(app *application.Application) localai.ClusterInstalledProvider {
|
|
if app == nil || !app.IsDistributed() || app.GalleryService() == nil {
|
|
return nil
|
|
}
|
|
return func(ctx context.Context) ([]string, error) {
|
|
manager := app.GalleryService().BackendManager()
|
|
if manager == nil {
|
|
return nil, nil
|
|
}
|
|
backends, err := manager.ListBackends()
|
|
if err != nil {
|
|
return nil, err
|
|
}
|
|
names := make([]string, 0, len(backends))
|
|
for name := range backends {
|
|
names = append(names, name)
|
|
}
|
|
return names, nil
|
|
}
|
|
}
|