mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-29 01:25:03 -04:00
* fix(distributed): stage every shard of a split GGUF A split GGUF is configured by its first shard only. llama.cpp opens the other "-0000N-of-0000M.gguf" files from the same directory by name. The router staged only the configured path, so the worker received shard 1 and the load failed with "failed to load GGUF split". The router now stages the remaining shards next to the first one. A missing shard fails the load and names the file. The file count for progress and the payload size also include all shards. The payload size feeds the load deadline and the disk-headroom check. For a 111 GB model whose first shard is 10 MB, both were sized for less than 1 GB. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * fix(distributed): stage the files a model install declares Replace the split GGUF file name matching with the model's own file list. A gallery install or an import records every file of the model in ._gallery_<name>.yaml (files:), and a config can list more under download_files:. The router now stages all of these files, not only the files that the config's path fields name. This includes the other shards of a split GGUF, which llama.cpp opens by name. The application gives the router a resolver that reads the two lists. The resolver looks up the files by model name when it stages them, so a replica that the reconciler loads from saved load options gets the same files. backend.proto does not change. A declared file that is missing on the frontend is skipped with a warning. The load deadline and the disk headroom check include the declared files. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> --------- Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
30 lines
989 B
Go
30 lines
989 B
Go
package application
|
|
|
|
import (
|
|
"path/filepath"
|
|
|
|
"github.com/mudler/LocalAI/core/config"
|
|
"github.com/mudler/LocalAI/core/gallery"
|
|
"github.com/mudler/LocalAI/pkg/utils"
|
|
)
|
|
|
|
// declaredModelFiles resolves the files a model needs on disk beyond the ones
|
|
// its config names: what its gallery install or import declared under
|
|
// `files:`, and what the config itself lists under download_files. The
|
|
// distributed router stages these to workers, which cannot see the frontend's
|
|
// models directory.
|
|
func declaredModelFiles(configLoader *config.ModelConfigLoader, modelsPath string) func(modelName string) []string {
|
|
return func(modelName string) []string {
|
|
files := gallery.InstalledModelFiles(modelsPath, modelName)
|
|
if cfg, ok := configLoader.GetModelConfig(modelName); ok {
|
|
for _, f := range cfg.DownloadFiles {
|
|
if utils.VerifyPath(f.Filename, modelsPath) != nil {
|
|
continue
|
|
}
|
|
files = append(files, filepath.Join(modelsPath, f.Filename))
|
|
}
|
|
}
|
|
return files
|
|
}
|
|
}
|