Files
LocalAI/core/http/endpoints/ollama/models.go
T
leilei3167 afebecc63d fix(ollama): report on-disk size for /api/tags and /api/ps (#11989)
Hardcoding size/size_vram as 0 made Ollama clients treat loaded models as
free. Prefer ModelFileName+ModelPath Stat when available, and omit size_vram
(and size) when the value is unknown instead of emitting literal zeros.
Resolve each listed model by its stored ID so tagged variants use their
own weights.

Fixes #11969

Signed-off-by: lei_lei <imleilei123@gmail.com>
2026-09-28 00:11:41 +02:00

212 lines
6.1 KiB
Go

package ollama
import (
"crypto/sha256"
"fmt"
"os"
"path/filepath"
"strings"
"time"
"github.com/labstack/echo/v4"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/schema"
"github.com/mudler/LocalAI/core/services/galleryop"
"github.com/mudler/LocalAI/pkg/model"
)
const ollamaCompatVersion = "0.9.0"
// ListModelsEndpoint handles Ollama-compatible GET /api/tags
func ListModelsEndpoint(bcl *config.ModelConfigLoader, ml *model.ModelLoader) echo.HandlerFunc {
return func(c echo.Context) error {
modelNames, err := galleryop.ListModels(bcl, ml, nil, galleryop.SKIP_IF_CONFIGURED)
if err != nil {
return ollamaError(c, 500, fmt.Sprintf("failed to list models: %v", err))
}
var models []schema.OllamaModelEntry
for _, name := range modelNames {
ollamaName := name
if !strings.Contains(ollamaName, ":") {
ollamaName += ":latest"
}
digest := fmt.Sprintf("sha256:%x", sha256.Sum256([]byte(name)))
details, caps := modelMetaFromConfig(bcl, name)
entry := schema.OllamaModelEntry{
Name: ollamaName,
Model: ollamaName,
ModifiedAt: time.Now().UTC(),
Size: modelOnDiskSize(bcl, ml, name),
Digest: digest,
Details: details,
Capabilities: caps,
}
models = append(models, entry)
}
return c.JSON(200, schema.OllamaListResponse{Models: models})
}
}
// ShowModelEndpoint handles Ollama-compatible POST /api/show
func ShowModelEndpoint(bcl *config.ModelConfigLoader) echo.HandlerFunc {
return func(c echo.Context) error {
var req schema.OllamaShowRequest
if err := c.Bind(&req); err != nil {
return ollamaError(c, 400, "invalid request body")
}
name := req.Name
if name == "" {
name = req.Model
}
if name == "" {
return ollamaError(c, 400, "name is required")
}
// Strip tag suffix for config lookup
configName := strings.Split(name, ":")[0]
cfg, exists := bcl.GetModelConfig(configName)
if !exists {
return ollamaError(c, 404, fmt.Sprintf("model '%s' not found", name))
}
resp := schema.OllamaShowResponse{
Modelfile: fmt.Sprintf("FROM %s", cfg.Model),
Parameters: "",
Template: cfg.TemplateConfig.Chat,
Details: modelDetailsFromModelConfig(&cfg),
ModelInfo: modelInfoFromModelConfig(&cfg),
Capabilities: modelCapabilities(&cfg),
}
return c.JSON(200, resp)
}
}
// ListRunningEndpoint handles Ollama-compatible GET /api/ps
func ListRunningEndpoint(bcl *config.ModelConfigLoader, ml *model.ModelLoader) echo.HandlerFunc {
return func(c echo.Context) error {
loadedModels := ml.ListLoadedModels()
var models []schema.OllamaPsEntry
for _, m := range loadedModels {
name := m.ID
ollamaName := name
if !strings.Contains(ollamaName, ":") {
ollamaName += ":latest"
}
details, caps := modelMetaFromConfig(bcl, name)
entry := schema.OllamaPsEntry{
Name: ollamaName,
Model: ollamaName,
Size: modelOnDiskSize(bcl, ml, name),
Digest: fmt.Sprintf("sha256:%x", sha256.Sum256([]byte(name))),
Details: details,
ExpiresAt: time.Now().Add(24 * time.Hour).UTC(),
// SizeVRAM is left unset: LocalAI has no authoritative per-model
// VRAM figure to report, and a literal 0 is worse than omitting
// the field (clients treat 0 as "costs nothing").
Capabilities: caps,
}
models = append(models, entry)
}
return c.JSON(200, schema.OllamaPsResponse{Models: models})
}
}
// VersionEndpoint handles Ollama-compatible GET /api/version
func VersionEndpoint() echo.HandlerFunc {
return func(c echo.Context) error {
return c.JSON(200, schema.OllamaVersionResponse{Version: ollamaCompatVersion})
}
}
// HeartbeatEndpoint handles the Ollama root health check
func HeartbeatEndpoint() echo.HandlerFunc {
return func(c echo.Context) error {
return c.String(200, "Ollama is running")
}
}
// modelMetaFromConfig fetches the ModelConfig for `name` and derives both the
// Ollama details block and capability list. Returns zero values when the model
// is not configured.
func modelMetaFromConfig(bcl *config.ModelConfigLoader, name string) (schema.OllamaModelDetails, []string) {
configName := strings.Split(name, ":")[0]
cfg, exists := bcl.GetModelConfig(configName)
if !exists {
return schema.OllamaModelDetails{}, nil
}
return modelDetailsFromModelConfig(&cfg), modelCapabilities(&cfg)
}
// modelOnDiskSize returns the on-disk byte size of a model's primary weight
// file when it can be resolved via ModelConfig.ModelFileName() + ModelPath.
// Returns nil when the size is unknown so callers omit the JSON field instead
// of emitting an authoritative 0 (issue #11969).
func modelOnDiskSize(bcl *config.ModelConfigLoader, ml *model.ModelLoader, name string) *int64 {
if ml == nil || ml.ModelPath == "" {
return nil
}
// List endpoints pass the stored model ID, including any configured tag.
configName := name
rel := configName
if bcl != nil {
if cfg, exists := bcl.GetModelConfig(configName); exists {
if fileName := cfg.ModelFileName(); fileName != "" {
rel = fileName
}
}
}
if rel == "" {
return nil
}
info, err := os.Stat(filepath.Join(ml.ModelPath, rel))
if err != nil || !info.Mode().IsRegular() || info.Size() <= 0 {
return nil
}
size := info.Size()
return &size
}
func modelDetailsFromModelConfig(cfg *config.ModelConfig) schema.OllamaModelDetails {
family := cfg.Backend
details := schema.OllamaModelDetails{
Format: "gguf",
Family: family,
ParameterSize: extractParameterSize(cfg.Model),
QuantizationLevel: extractQuantizationLevel(cfg.Model),
}
if family != "" {
details.Families = []string{family}
}
return details
}
// modelInfoFromModelConfig returns a small map of model_info entries derived
// from the LocalAI ModelConfig. Ollama clients use this map for architecture
// and context-length information; we expose what we can without loading the
// model.
func modelInfoFromModelConfig(cfg *config.ModelConfig) map[string]any {
info := map[string]any{}
if cfg.Backend != "" {
info["general.architecture"] = cfg.Backend
}
if cfg.ContextSize != nil && *cfg.ContextSize > 0 {
info["general.context_length"] = *cfg.ContextSize
}
if len(info) == 0 {
return nil
}
return info
}