mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-29 01:25:03 -04:00
* fix(distributed): stage every shard of a split GGUF A split GGUF is configured by its first shard only. llama.cpp opens the other "-0000N-of-0000M.gguf" files from the same directory by name. The router staged only the configured path, so the worker received shard 1 and the load failed with "failed to load GGUF split". The router now stages the remaining shards next to the first one. A missing shard fails the load and names the file. The file count for progress and the payload size also include all shards. The payload size feeds the load deadline and the disk-headroom check. For a 111 GB model whose first shard is 10 MB, both were sized for less than 1 GB. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * fix(distributed): stage the files a model install declares Replace the split GGUF file name matching with the model's own file list. A gallery install or an import records every file of the model in ._gallery_<name>.yaml (files:), and a config can list more under download_files:. The router now stages all of these files, not only the files that the config's path fields name. This includes the other shards of a split GGUF, which llama.cpp opens by name. The application gives the router a resolver that reads the two lists. The resolver looks up the files by model name when it stages them, so a replica that the reconciler loads from saved load options gets the same files. backend.proto does not change. A declared file that is missing on the frontend is skipped with a warning. The load deadline and the disk headroom check include the declared files. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto <mudler@localai.io> --------- Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
42 lines
1.2 KiB
Go
42 lines
1.2 KiB
Go
package application
|
|
|
|
import (
|
|
"os"
|
|
"path/filepath"
|
|
|
|
. "github.com/onsi/ginkgo/v2"
|
|
. "github.com/onsi/gomega"
|
|
|
|
"github.com/mudler/LocalAI/core/config"
|
|
"github.com/mudler/LocalAI/core/gallery"
|
|
)
|
|
|
|
var _ = Describe("declaredModelFiles", func() {
|
|
It("combines the gallery install's files with the config's download_files", func() {
|
|
modelsPath := GinkgoT().TempDir()
|
|
Expect(os.WriteFile(filepath.Join(modelsPath, "big.yaml"), []byte(`
|
|
name: big
|
|
backend: llama-cpp
|
|
parameters:
|
|
model: big/Big-00001-of-00002.gguf
|
|
download_files:
|
|
- filename: big/extra.bin
|
|
uri: https://example.com/extra.bin
|
|
`), 0o644)).To(Succeed())
|
|
Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("big")), []byte(`
|
|
files:
|
|
- filename: big/Big-00001-of-00002.gguf
|
|
- filename: big/Big-00002-of-00002.gguf
|
|
`), 0o644)).To(Succeed())
|
|
|
|
loader := config.NewModelConfigLoader(modelsPath)
|
|
Expect(loader.LoadModelConfigsFromPath(modelsPath)).To(Succeed())
|
|
|
|
Expect(declaredModelFiles(loader, modelsPath)("big")).To(ConsistOf(
|
|
filepath.Join(modelsPath, "big/Big-00001-of-00002.gguf"),
|
|
filepath.Join(modelsPath, "big/Big-00002-of-00002.gguf"),
|
|
filepath.Join(modelsPath, "big/extra.bin"),
|
|
))
|
|
})
|
|
})
|