fix(ollama): report on-disk size for /api/tags and /api/ps (#11989)

Hardcoding size/size_vram as 0 made Ollama clients treat loaded models as
free. Prefer ModelFileName+ModelPath Stat when available, and omit size_vram
(and size) when the value is unknown instead of emitting literal zeros.
Resolve each listed model by its stored ID so tagged variants use their
own weights.

Fixes #11969

Signed-off-by: lei_lei <imleilei123@gmail.com>
This commit is contained in:
leilei3167 authored and GitHub committed 2026-09-28 00:11:41 +02:00
1 parent eb97d5f549
commit afebecc63d
4 files changed
+216 -15

No files matched your search

+43 -8
View File
@@ -3,6 +3,8 @@ package ollama
import (
"crypto/sha256"
"fmt"
"os"
"path/filepath"
"strings"
"time"
@@ -37,7 +39,7 @@ func ListModelsEndpoint(bcl *config.ModelConfigLoader, ml *model.ModelLoader) ec
Name: ollamaName,
Model: ollamaName,
ModifiedAt: time.Now().UTC(),
Size: 0,
Size: modelOnDiskSize(bcl, ml, name),
Digest: digest,
Details: details,
Capabilities: caps,
@@ -101,13 +103,15 @@ func ListRunningEndpoint(bcl *config.ModelConfigLoader, ml *model.ModelLoader) e
details, caps := modelMetaFromConfig(bcl, name)
entry := schema.OllamaPsEntry{
Name: ollamaName,
Model: ollamaName,
Size: 0,
Digest: fmt.Sprintf("sha256:%x", sha256.Sum256([]byte(name))),
Details: details,
ExpiresAt: time.Now().Add(24 * time.Hour).UTC(),
SizeVRAM: 0,
Name: ollamaName,
Model: ollamaName,
Size: modelOnDiskSize(bcl, ml, name),
Digest: fmt.Sprintf("sha256:%x", sha256.Sum256([]byte(name))),
Details: details,
ExpiresAt: time.Now().Add(24 * time.Hour).UTC(),
// SizeVRAM is left unset: LocalAI has no authoritative per-model
// VRAM figure to report, and a literal 0 is worse than omitting
// the field (clients treat 0 as "costs nothing").
Capabilities: caps,
}
models = append(models, entry)
@@ -143,6 +147,37 @@ func modelMetaFromConfig(bcl *config.ModelConfigLoader, name string) (schema.Oll
return modelDetailsFromModelConfig(&cfg), modelCapabilities(&cfg)
}
// modelOnDiskSize returns the on-disk byte size of a model's primary weight
// file when it can be resolved via ModelConfig.ModelFileName() + ModelPath.
// Returns nil when the size is unknown so callers omit the JSON field instead
// of emitting an authoritative 0 (issue #11969).
func modelOnDiskSize(bcl *config.ModelConfigLoader, ml *model.ModelLoader, name string) *int64 {
if ml == nil || ml.ModelPath == "" {
return nil
}
// List endpoints pass the stored model ID, including any configured tag.
configName := name
rel := configName
if bcl != nil {
if cfg, exists := bcl.GetModelConfig(configName); exists {
if fileName := cfg.ModelFileName(); fileName != "" {
rel = fileName
}
}
}
if rel == "" {
return nil
}
info, err := os.Stat(filepath.Join(ml.ModelPath, rel))
if err != nil || !info.Mode().IsRegular() || info.Size() <= 0 {
return nil
}
size := info.Size()
return &size
}
func modelDetailsFromModelConfig(cfg *config.ModelConfig) schema.OllamaModelDetails {
family := cfg.Backend
details := schema.OllamaModelDetails{
+161 -2
View File
@@ -13,6 +13,8 @@ import (
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/http/endpoints/ollama"
"github.com/mudler/LocalAI/core/schema"
"github.com/mudler/LocalAI/pkg/model"
"github.com/mudler/LocalAI/pkg/system"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
@@ -163,8 +165,165 @@ parameters:
})
Describe("ListModelsEndpoint", func() {
It("includes capabilities and details for each listed model in /api/tags", func() {
Skip("covered by per-entry tests; integration smoke test")
var (
tmpDir string
bcl *config.ModelConfigLoader
ml *model.ModelLoader
)
BeforeEach(func() {
var err error
tmpDir, err = os.MkdirTemp("", "ollama-tags-test-*")
Expect(err).ToNot(HaveOccurred())
systemState, err := system.GetSystemState(system.WithModelPath(tmpDir))
Expect(err).ToNot(HaveOccurred())
ml = model.NewModelLoader(systemState)
bcl = config.NewModelConfigLoader(tmpDir)
})
AfterEach(func() {
_ = os.RemoveAll(tmpDir)
})
writeConfig := func(name, yaml string) {
path := filepath.Join(tmpDir, name+".yaml")
Expect(os.WriteFile(path, []byte(yaml), 0o644)).To(Succeed())
Expect(bcl.ReadModelConfig(path)).To(Succeed())
}
callTags := func() (schema.OllamaListResponse, []byte) {
req := httptest.NewRequest(http.MethodGet, "/api/tags", nil)
rec := httptest.NewRecorder()
c := e.NewContext(req, rec)
handler := ollama.ListModelsEndpoint(bcl, ml)
Expect(handler(c)).To(Succeed())
Expect(rec.Code).To(Equal(http.StatusOK))
var resp schema.OllamaListResponse
Expect(json.Unmarshal(rec.Body.Bytes(), &resp)).To(Succeed())
return resp, rec.Body.Bytes()
}
It("uses the exact configured name when a model has a tag", func() {
Expect(os.WriteFile(filepath.Join(tmpDir, "base.gguf"), []byte("base"), 0o644)).To(Succeed())
Expect(os.WriteFile(filepath.Join(tmpDir, "tagged.gguf"), []byte("tagged-weights"), 0o644)).To(Succeed())
writeConfig("chat", "name: chat\nparameters:\n model: base.gguf\n")
writeConfig("tagged", "name: chat:q8\nparameters:\n model: tagged.gguf\n")
resp, _ := callTags()
var tagged *int64
for _, entry := range resp.Models {
if entry.Name == "chat:q8" {
tagged = entry.Size
}
}
Expect(tagged).ToNot(BeNil())
Expect(*tagged).To(Equal(int64(len("tagged-weights"))))
})
It("reports on-disk size from ModelFileName+ModelPath and omits size when unknown", func() {
weight := []byte("fake-gguf-weights-0123456789")
Expect(os.WriteFile(filepath.Join(tmpDir, "Llama-3-8B-Q4_K_M.gguf"), weight, 0o644)).To(Succeed())
writeConfig("chat", `
name: chat
backend: llama-cpp
template:
chat: "{{ .Input }}"
parameters:
model: Llama-3-8B-Q4_K_M.gguf
`)
writeConfig("missing-weights", `
name: missing-weights
backend: llama-cpp
template:
chat: "{{ .Input }}"
parameters:
model: does-not-exist.gguf
`)
resp, raw := callTags()
Expect(resp.Models).To(HaveLen(2))
byName := map[string]schema.OllamaModelEntry{}
for _, m := range resp.Models {
byName[m.Name] = m
}
chat := byName["chat:latest"]
Expect(chat.Size).ToNot(BeNil())
Expect(*chat.Size).To(Equal(int64(len(weight))))
Expect(chat.Capabilities).To(ContainElement("completion"))
Expect(chat.Details.QuantizationLevel).To(Equal("Q4_K_M"))
missing := byName["missing-weights:latest"]
Expect(missing.Size).To(BeNil())
Expect(string(raw)).ToNot(ContainSubstring(`"size":0`))
})
})
Describe("ListRunningEndpoint", func() {
var (
tmpDir string
bcl *config.ModelConfigLoader
ml *model.ModelLoader
)
BeforeEach(func() {
var err error
tmpDir, err = os.MkdirTemp("", "ollama-ps-test-*")
Expect(err).ToNot(HaveOccurred())
systemState, err := system.GetSystemState(system.WithModelPath(tmpDir))
Expect(err).ToNot(HaveOccurred())
ml = model.NewModelLoader(systemState)
bcl = config.NewModelConfigLoader(tmpDir)
})
AfterEach(func() {
_ = os.RemoveAll(tmpDir)
})
It("reports on-disk size for loaded models and omits size_vram when unknown", func() {
weight := []byte("loaded-model-weights-abcdef")
Expect(os.WriteFile(filepath.Join(tmpDir, "granite-Q4_K_M.gguf"), weight, 0o644)).To(Succeed())
cfgPath := filepath.Join(tmpDir, "granite.yaml")
Expect(os.WriteFile(cfgPath, []byte(`
name: granite
backend: llama-cpp
template:
chat: "{{ .Input }}"
parameters:
model: granite-Q4_K_M.gguf
`), 0o644)).To(Succeed())
Expect(bcl.ReadModelConfig(cfgPath)).To(Succeed())
store := model.NewInMemoryModelStore()
store.Set("granite", model.NewModel("granite", "addr", nil))
ml.SetModelStore(store)
req := httptest.NewRequest(http.MethodGet, "/api/ps", nil)
rec := httptest.NewRecorder()
c := e.NewContext(req, rec)
handler := ollama.ListRunningEndpoint(bcl, ml)
Expect(handler(c)).To(Succeed())
Expect(rec.Code).To(Equal(http.StatusOK))
raw := rec.Body.String()
Expect(raw).ToNot(ContainSubstring(`"size":0`))
Expect(raw).ToNot(ContainSubstring(`"size_vram"`))
var resp schema.OllamaPsResponse
Expect(json.Unmarshal(rec.Body.Bytes(), &resp)).To(Succeed())
Expect(resp.Models).To(HaveLen(1))
Expect(resp.Models[0].Name).To(Equal("granite:latest"))
Expect(resp.Models[0].Size).ToNot(BeNil())
Expect(*resp.Models[0].Size).To(Equal(int64(len(weight))))
Expect(resp.Models[0].SizeVRAM).To(BeNil())
Expect(resp.Models[0].Details.QuantizationLevel).To(Equal("Q4_K_M"))
})
})
})
+10 -5
View File
@@ -293,12 +293,14 @@ type OllamaModelDetails struct {
QuantizationLevel string `json:"quantization_level,omitempty"`
}
// OllamaModelEntry represents a model in the list response
// OllamaModelEntry represents a model in the list response.
// Size is a pointer so an unknown on-disk size can be omitted instead of
// serializing as the misleading literal 0 (see issue #11969).
type OllamaModelEntry struct {
Name string `json:"name"`
Model string `json:"model"`
ModifiedAt time.Time `json:"modified_at"`
Size int64 `json:"size"`
Size *int64 `json:"size,omitempty"`
Digest string `json:"digest"`
Details OllamaModelDetails `json:"details"`
Capabilities []string `json:"capabilities,omitempty"`
@@ -309,15 +311,18 @@ type OllamaListResponse struct {
Models []OllamaModelEntry `json:"models"`
}
// OllamaPsEntry represents a running model in the ps response
// OllamaPsEntry represents a running model in the ps response.
// Size and SizeVRAM are pointers so unknown values are omitted rather than
// reported as authoritative zeros (see issue #11969). SizeVRAM is only set
// when the runtime can provide a real VRAM figure.
type OllamaPsEntry struct {
Name string `json:"name"`
Model string `json:"model"`
Size int64 `json:"size"`
Size *int64 `json:"size,omitempty"`
Digest string `json:"digest"`
Details OllamaModelDetails `json:"details"`
ExpiresAt time.Time `json:"expires_at"`
SizeVRAM int64 `json:"size_vram"`
SizeVRAM *int64 `json:"size_vram,omitempty"`
Capabilities []string `json:"capabilities,omitempty"`
}
+2
View File
@@ -391,6 +391,8 @@ See the [Model Configuration]({{% relref "advanced/model-configuration" %}}) gui
### List Installed Models
Ollama clients can list configured models with `GET /api/tags` and loaded models with `GET /api/ps`. Each entry includes `size` in bytes when LocalAI can resolve a non-empty primary weights file on disk. This is the size of that file, not the total size of a multi-file model or its memory use. Unknown sizes are omitted; `/api/ps` also omits `size_vram` because per-model VRAM use is not available.
```bash
# Via API
curl http://localhost:8080/v1/models