mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-28 00:55:06 -04:00
fix(ollama): report on-disk size for /api/tags and /api/ps (#11989)
Hardcoding size/size_vram as 0 made Ollama clients treat loaded models as free. Prefer ModelFileName+ModelPath Stat when available, and omit size_vram (and size) when the value is unknown instead of emitting literal zeros. Resolve each listed model by its stored ID so tagged variants use their own weights. Fixes #11969 Signed-off-by: lei_lei <imleilei123@gmail.com>
This commit is contained in:
1 parent
eb97d5f549
commit
afebecc63d
4 files changed
+216
-15
No files matched your search
@@ -3,6 +3,8 @@ package ollama
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
@@ -37,7 +39,7 @@ func ListModelsEndpoint(bcl *config.ModelConfigLoader, ml *model.ModelLoader) ec
|
||||
Name: ollamaName,
|
||||
Model: ollamaName,
|
||||
ModifiedAt: time.Now().UTC(),
|
||||
Size: 0,
|
||||
Size: modelOnDiskSize(bcl, ml, name),
|
||||
Digest: digest,
|
||||
Details: details,
|
||||
Capabilities: caps,
|
||||
@@ -101,13 +103,15 @@ func ListRunningEndpoint(bcl *config.ModelConfigLoader, ml *model.ModelLoader) e
|
||||
|
||||
details, caps := modelMetaFromConfig(bcl, name)
|
||||
entry := schema.OllamaPsEntry{
|
||||
Name: ollamaName,
|
||||
Model: ollamaName,
|
||||
Size: 0,
|
||||
Digest: fmt.Sprintf("sha256:%x", sha256.Sum256([]byte(name))),
|
||||
Details: details,
|
||||
ExpiresAt: time.Now().Add(24 * time.Hour).UTC(),
|
||||
SizeVRAM: 0,
|
||||
Name: ollamaName,
|
||||
Model: ollamaName,
|
||||
Size: modelOnDiskSize(bcl, ml, name),
|
||||
Digest: fmt.Sprintf("sha256:%x", sha256.Sum256([]byte(name))),
|
||||
Details: details,
|
||||
ExpiresAt: time.Now().Add(24 * time.Hour).UTC(),
|
||||
// SizeVRAM is left unset: LocalAI has no authoritative per-model
|
||||
// VRAM figure to report, and a literal 0 is worse than omitting
|
||||
// the field (clients treat 0 as "costs nothing").
|
||||
Capabilities: caps,
|
||||
}
|
||||
models = append(models, entry)
|
||||
@@ -143,6 +147,37 @@ func modelMetaFromConfig(bcl *config.ModelConfigLoader, name string) (schema.Oll
|
||||
return modelDetailsFromModelConfig(&cfg), modelCapabilities(&cfg)
|
||||
}
|
||||
|
||||
// modelOnDiskSize returns the on-disk byte size of a model's primary weight
|
||||
// file when it can be resolved via ModelConfig.ModelFileName() + ModelPath.
|
||||
// Returns nil when the size is unknown so callers omit the JSON field instead
|
||||
// of emitting an authoritative 0 (issue #11969).
|
||||
func modelOnDiskSize(bcl *config.ModelConfigLoader, ml *model.ModelLoader, name string) *int64 {
|
||||
if ml == nil || ml.ModelPath == "" {
|
||||
return nil
|
||||
}
|
||||
|
||||
// List endpoints pass the stored model ID, including any configured tag.
|
||||
configName := name
|
||||
rel := configName
|
||||
if bcl != nil {
|
||||
if cfg, exists := bcl.GetModelConfig(configName); exists {
|
||||
if fileName := cfg.ModelFileName(); fileName != "" {
|
||||
rel = fileName
|
||||
}
|
||||
}
|
||||
}
|
||||
if rel == "" {
|
||||
return nil
|
||||
}
|
||||
|
||||
info, err := os.Stat(filepath.Join(ml.ModelPath, rel))
|
||||
if err != nil || !info.Mode().IsRegular() || info.Size() <= 0 {
|
||||
return nil
|
||||
}
|
||||
size := info.Size()
|
||||
return &size
|
||||
}
|
||||
|
||||
func modelDetailsFromModelConfig(cfg *config.ModelConfig) schema.OllamaModelDetails {
|
||||
family := cfg.Backend
|
||||
details := schema.OllamaModelDetails{
|
||||
|
||||
@@ -13,6 +13,8 @@ import (
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
"github.com/mudler/LocalAI/core/http/endpoints/ollama"
|
||||
"github.com/mudler/LocalAI/core/schema"
|
||||
"github.com/mudler/LocalAI/pkg/model"
|
||||
"github.com/mudler/LocalAI/pkg/system"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
@@ -163,8 +165,165 @@ parameters:
|
||||
})
|
||||
|
||||
Describe("ListModelsEndpoint", func() {
|
||||
It("includes capabilities and details for each listed model in /api/tags", func() {
|
||||
Skip("covered by per-entry tests; integration smoke test")
|
||||
var (
|
||||
tmpDir string
|
||||
bcl *config.ModelConfigLoader
|
||||
ml *model.ModelLoader
|
||||
)
|
||||
|
||||
BeforeEach(func() {
|
||||
var err error
|
||||
tmpDir, err = os.MkdirTemp("", "ollama-tags-test-*")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
|
||||
systemState, err := system.GetSystemState(system.WithModelPath(tmpDir))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
ml = model.NewModelLoader(systemState)
|
||||
bcl = config.NewModelConfigLoader(tmpDir)
|
||||
})
|
||||
|
||||
AfterEach(func() {
|
||||
_ = os.RemoveAll(tmpDir)
|
||||
})
|
||||
|
||||
writeConfig := func(name, yaml string) {
|
||||
path := filepath.Join(tmpDir, name+".yaml")
|
||||
Expect(os.WriteFile(path, []byte(yaml), 0o644)).To(Succeed())
|
||||
Expect(bcl.ReadModelConfig(path)).To(Succeed())
|
||||
}
|
||||
|
||||
callTags := func() (schema.OllamaListResponse, []byte) {
|
||||
req := httptest.NewRequest(http.MethodGet, "/api/tags", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
c := e.NewContext(req, rec)
|
||||
|
||||
handler := ollama.ListModelsEndpoint(bcl, ml)
|
||||
Expect(handler(c)).To(Succeed())
|
||||
Expect(rec.Code).To(Equal(http.StatusOK))
|
||||
|
||||
var resp schema.OllamaListResponse
|
||||
Expect(json.Unmarshal(rec.Body.Bytes(), &resp)).To(Succeed())
|
||||
return resp, rec.Body.Bytes()
|
||||
}
|
||||
|
||||
It("uses the exact configured name when a model has a tag", func() {
|
||||
Expect(os.WriteFile(filepath.Join(tmpDir, "base.gguf"), []byte("base"), 0o644)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(tmpDir, "tagged.gguf"), []byte("tagged-weights"), 0o644)).To(Succeed())
|
||||
writeConfig("chat", "name: chat\nparameters:\n model: base.gguf\n")
|
||||
writeConfig("tagged", "name: chat:q8\nparameters:\n model: tagged.gguf\n")
|
||||
|
||||
resp, _ := callTags()
|
||||
var tagged *int64
|
||||
for _, entry := range resp.Models {
|
||||
if entry.Name == "chat:q8" {
|
||||
tagged = entry.Size
|
||||
}
|
||||
}
|
||||
Expect(tagged).ToNot(BeNil())
|
||||
Expect(*tagged).To(Equal(int64(len("tagged-weights"))))
|
||||
})
|
||||
|
||||
It("reports on-disk size from ModelFileName+ModelPath and omits size when unknown", func() {
|
||||
weight := []byte("fake-gguf-weights-0123456789")
|
||||
Expect(os.WriteFile(filepath.Join(tmpDir, "Llama-3-8B-Q4_K_M.gguf"), weight, 0o644)).To(Succeed())
|
||||
writeConfig("chat", `
|
||||
name: chat
|
||||
backend: llama-cpp
|
||||
template:
|
||||
chat: "{{ .Input }}"
|
||||
parameters:
|
||||
model: Llama-3-8B-Q4_K_M.gguf
|
||||
`)
|
||||
writeConfig("missing-weights", `
|
||||
name: missing-weights
|
||||
backend: llama-cpp
|
||||
template:
|
||||
chat: "{{ .Input }}"
|
||||
parameters:
|
||||
model: does-not-exist.gguf
|
||||
`)
|
||||
|
||||
resp, raw := callTags()
|
||||
Expect(resp.Models).To(HaveLen(2))
|
||||
|
||||
byName := map[string]schema.OllamaModelEntry{}
|
||||
for _, m := range resp.Models {
|
||||
byName[m.Name] = m
|
||||
}
|
||||
|
||||
chat := byName["chat:latest"]
|
||||
Expect(chat.Size).ToNot(BeNil())
|
||||
Expect(*chat.Size).To(Equal(int64(len(weight))))
|
||||
Expect(chat.Capabilities).To(ContainElement("completion"))
|
||||
Expect(chat.Details.QuantizationLevel).To(Equal("Q4_K_M"))
|
||||
|
||||
missing := byName["missing-weights:latest"]
|
||||
Expect(missing.Size).To(BeNil())
|
||||
Expect(string(raw)).ToNot(ContainSubstring(`"size":0`))
|
||||
})
|
||||
})
|
||||
|
||||
Describe("ListRunningEndpoint", func() {
|
||||
var (
|
||||
tmpDir string
|
||||
bcl *config.ModelConfigLoader
|
||||
ml *model.ModelLoader
|
||||
)
|
||||
|
||||
BeforeEach(func() {
|
||||
var err error
|
||||
tmpDir, err = os.MkdirTemp("", "ollama-ps-test-*")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
|
||||
systemState, err := system.GetSystemState(system.WithModelPath(tmpDir))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
ml = model.NewModelLoader(systemState)
|
||||
bcl = config.NewModelConfigLoader(tmpDir)
|
||||
})
|
||||
|
||||
AfterEach(func() {
|
||||
_ = os.RemoveAll(tmpDir)
|
||||
})
|
||||
|
||||
It("reports on-disk size for loaded models and omits size_vram when unknown", func() {
|
||||
weight := []byte("loaded-model-weights-abcdef")
|
||||
Expect(os.WriteFile(filepath.Join(tmpDir, "granite-Q4_K_M.gguf"), weight, 0o644)).To(Succeed())
|
||||
|
||||
cfgPath := filepath.Join(tmpDir, "granite.yaml")
|
||||
Expect(os.WriteFile(cfgPath, []byte(`
|
||||
name: granite
|
||||
backend: llama-cpp
|
||||
template:
|
||||
chat: "{{ .Input }}"
|
||||
parameters:
|
||||
model: granite-Q4_K_M.gguf
|
||||
`), 0o644)).To(Succeed())
|
||||
Expect(bcl.ReadModelConfig(cfgPath)).To(Succeed())
|
||||
|
||||
store := model.NewInMemoryModelStore()
|
||||
store.Set("granite", model.NewModel("granite", "addr", nil))
|
||||
ml.SetModelStore(store)
|
||||
|
||||
req := httptest.NewRequest(http.MethodGet, "/api/ps", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
c := e.NewContext(req, rec)
|
||||
|
||||
handler := ollama.ListRunningEndpoint(bcl, ml)
|
||||
Expect(handler(c)).To(Succeed())
|
||||
Expect(rec.Code).To(Equal(http.StatusOK))
|
||||
|
||||
raw := rec.Body.String()
|
||||
Expect(raw).ToNot(ContainSubstring(`"size":0`))
|
||||
Expect(raw).ToNot(ContainSubstring(`"size_vram"`))
|
||||
|
||||
var resp schema.OllamaPsResponse
|
||||
Expect(json.Unmarshal(rec.Body.Bytes(), &resp)).To(Succeed())
|
||||
Expect(resp.Models).To(HaveLen(1))
|
||||
Expect(resp.Models[0].Name).To(Equal("granite:latest"))
|
||||
Expect(resp.Models[0].Size).ToNot(BeNil())
|
||||
Expect(*resp.Models[0].Size).To(Equal(int64(len(weight))))
|
||||
Expect(resp.Models[0].SizeVRAM).To(BeNil())
|
||||
Expect(resp.Models[0].Details.QuantizationLevel).To(Equal("Q4_K_M"))
|
||||
})
|
||||
})
|
||||
})
|
||||
+10
-5
@@ -293,12 +293,14 @@ type OllamaModelDetails struct {
|
||||
QuantizationLevel string `json:"quantization_level,omitempty"`
|
||||
}
|
||||
|
||||
// OllamaModelEntry represents a model in the list response
|
||||
// OllamaModelEntry represents a model in the list response.
|
||||
// Size is a pointer so an unknown on-disk size can be omitted instead of
|
||||
// serializing as the misleading literal 0 (see issue #11969).
|
||||
type OllamaModelEntry struct {
|
||||
Name string `json:"name"`
|
||||
Model string `json:"model"`
|
||||
ModifiedAt time.Time `json:"modified_at"`
|
||||
Size int64 `json:"size"`
|
||||
Size *int64 `json:"size,omitempty"`
|
||||
Digest string `json:"digest"`
|
||||
Details OllamaModelDetails `json:"details"`
|
||||
Capabilities []string `json:"capabilities,omitempty"`
|
||||
@@ -309,15 +311,18 @@ type OllamaListResponse struct {
|
||||
Models []OllamaModelEntry `json:"models"`
|
||||
}
|
||||
|
||||
// OllamaPsEntry represents a running model in the ps response
|
||||
// OllamaPsEntry represents a running model in the ps response.
|
||||
// Size and SizeVRAM are pointers so unknown values are omitted rather than
|
||||
// reported as authoritative zeros (see issue #11969). SizeVRAM is only set
|
||||
// when the runtime can provide a real VRAM figure.
|
||||
type OllamaPsEntry struct {
|
||||
Name string `json:"name"`
|
||||
Model string `json:"model"`
|
||||
Size int64 `json:"size"`
|
||||
Size *int64 `json:"size,omitempty"`
|
||||
Digest string `json:"digest"`
|
||||
Details OllamaModelDetails `json:"details"`
|
||||
ExpiresAt time.Time `json:"expires_at"`
|
||||
SizeVRAM int64 `json:"size_vram"`
|
||||
SizeVRAM *int64 `json:"size_vram,omitempty"`
|
||||
Capabilities []string `json:"capabilities,omitempty"`
|
||||
}
|
||||
|
||||
|
||||
@@ -391,6 +391,8 @@ See the [Model Configuration]({{% relref "advanced/model-configuration" %}}) gui
|
||||
|
||||
### List Installed Models
|
||||
|
||||
Ollama clients can list configured models with `GET /api/tags` and loaded models with `GET /api/ps`. Each entry includes `size` in bytes when LocalAI can resolve a non-empty primary weights file on disk. This is the size of that file, not the total size of a multi-file model or its memory use. Unknown sizes are omitted; `/api/ps` also omits `size_vram` because per-model VRAM use is not available.
|
||||
|
||||
```bash
|
||||
# Via API
|
||||
curl http://localhost:8080/v1/models
|
||||
|
||||
Reference in new issue
Block a user