mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-28 09:05:05 -04:00
* fix(quantization): pin the producing backend on imported quantized models (#11875) ImportModel hands the copied GGUF to importers.ImportLocalPath, which detects the file format and defaults every GGUF to `backend: llama-cpp`. For a model this service just produced with a backend stock llama.cpp cannot read, the generated config names an engine that cannot load the file, and the import silently registers an unloadable model. Correcting `backend:` by hand makes the same file work. The job record already carries the backend that served StartQuantization, so carry it into the config instead of keeping the detected default. The gallery publishes a quantizer as a release channel of the engine that runs its output ("llama-cpp-quantization" is llama.cpp's quantizer, whose GGUF is served by "llama-cpp"), so the channel suffix is stripped to get the serving backend. A backend that both quantizes and serves ("rocmfp4") carries no suffix and passes through unchanged, as do pinned hardware variants ("rocm-rocmfp4"), which are valid values for a config's backend field. An empty job backend leaves the detected default in place. Also replace the importer's generic "Fine-tuned model (GGUF)" description for this path: the model was quantized, not fine-tuned, and the job knows the type. Signed-off-by: Tai An <antai12232931@outlook.com> * style: restore trailing newline in service.go for gofmt Signed-off-by: Anai Guo <antai12232931@outlook.com> * style(quantization): restore trailing newline in service.go gofmt requires the file to end with a newline; the previous style commit did not actually add it. Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Assisted-by: Claude:claude-opus-5-5 [Claude Code] --------- Signed-off-by: Tai An <antai12232931@outlook.com> Signed-off-by: Anai Guo <antai12232931@outlook.com> Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
This commit is contained in:
1 parent
afebecc63d
commit
5b10d36b5f
2 files changed
+48
No files matched your search
@@ -621,6 +621,21 @@ func sanitizeQuantModelName(s string) string {
|
||||
return strings.ToLower(s)
|
||||
}
|
||||
|
||||
// inferenceBackendFor returns the backend that can load what a quantization
|
||||
// backend produced.
|
||||
//
|
||||
// The gallery publishes a quantizer as a release channel of the engine that
|
||||
// runs its output: "llama-cpp-quantization" is llama.cpp's quantizer, and the
|
||||
// GGUF it writes is served by "llama-cpp". The suffix is a channel marker and
|
||||
// carries no engine information, so stripping it yields the backend to pin in
|
||||
// the imported model's config. Names that carry no channel suffix (a backend
|
||||
// that both quantizes and serves, such as "rocmfp4") are already the engine
|
||||
// name and pass through unchanged, as do pinned hardware variants
|
||||
// ("rocm-rocmfp4"), which are valid values for a config's `backend:`.
|
||||
func inferenceBackendFor(quantBackend string) string {
|
||||
return strings.TrimSuffix(config.NormalizeBackendName(quantBackend), "-quantization")
|
||||
}
|
||||
|
||||
// ImportModel imports a quantized model into LocalAI asynchronously.
|
||||
func (s *QuantizationService) ImportModel(ctx context.Context, userID, jobID string, req schema.QuantizationImportRequest) (string, error) {
|
||||
s.mu.Lock()
|
||||
@@ -719,6 +734,17 @@ func (s *QuantizationService) ImportModel(ctx context.Context, userID, jobID str
|
||||
|
||||
cfg.Name = modelName
|
||||
|
||||
// The importer detects the file format and defaults to llama-cpp for any
|
||||
// GGUF. That is wrong for a model this service just quantized with a
|
||||
// backend stock llama.cpp cannot read: the job knows which backend
|
||||
// produced the file, so pin that one instead of the detected default.
|
||||
if backend := inferenceBackendFor(job.Backend); backend != "" {
|
||||
cfg.Backend = backend
|
||||
}
|
||||
if job.QuantizationType != "" {
|
||||
cfg.Description = "Quantized model (" + job.QuantizationType + ", GGUF)"
|
||||
}
|
||||
|
||||
// Write YAML config
|
||||
yamlData, err := yaml.Marshal(cfg)
|
||||
if err != nil {
|
||||
|
||||
@@ -329,6 +329,28 @@ var _ = Describe("QuantizationService", func() {
|
||||
})
|
||||
})
|
||||
|
||||
Describe("imported model backend", func() {
|
||||
It("strips the quantization channel suffix so the config pins the serving engine", func() {
|
||||
Expect(inferenceBackendFor("llama-cpp-quantization")).To(Equal("llama-cpp"))
|
||||
})
|
||||
|
||||
It("leaves a backend that both quantizes and serves unchanged", func() {
|
||||
Expect(inferenceBackendFor("rocmfp4")).To(Equal("rocmfp4"))
|
||||
})
|
||||
|
||||
It("keeps a pinned hardware variant, which is a valid backend value", func() {
|
||||
Expect(inferenceBackendFor("rocm-rocmfp4-quantization")).To(Equal("rocm-rocmfp4"))
|
||||
})
|
||||
|
||||
It("normalizes dots the way gallery names are written", func() {
|
||||
Expect(inferenceBackendFor("llama.cpp-quantization")).To(Equal("llama-cpp"))
|
||||
})
|
||||
|
||||
It("returns empty for an unset backend so the detected default is kept", func() {
|
||||
Expect(inferenceBackendFor("")).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
|
||||
Describe("compile-time adapter contract", func() {
|
||||
It("satisfies syncstate.Store for *distributed.QuantStore", func() {
|
||||
// Guards against drift between the adapter and the component interface;
|
||||
|
||||
Reference in new issue
Block a user