feat(system): report per-model DRM VRAM (#12026)

* feat(system): report per-model DRM VRAM

Expose optional resident device memory for local backend process trees.
Deduplicate DRM clients and omit unsupported or incomplete readings.
Document accounting limits and preserve a measured zero in JSON.

Closes #11970.

Assisted-by: Codex:gpt-6

* fix(system): document trusted procfs reads

Scope G304 annotations to paths built from the fixed procfs root,
integer process IDs, and kernel directory entries. These reads accept
no user-controlled path components.

Assisted-by: Codex:GPT-6 gosec

---------

Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com>
This commit is contained in:
localai-org-maint-botandlocalai-org-maint-bot authored and GitHub committed 2026-09-27 21:18:34 +02:00
1 parent 5794495a37
commit f154bd990a
11 files changed
+399

No files matched your search

+6
View File
@@ -8,6 +8,7 @@ import (
"github.com/mudler/LocalAI/core/schema"
"github.com/mudler/LocalAI/core/services/monitoring"
"github.com/mudler/LocalAI/pkg/model"
"github.com/mudler/LocalAI/pkg/xsysinfo"
)
// SystemInformations returns the system informations
@@ -42,6 +43,11 @@ func SystemInformations(cl *config.ModelConfigLoader, ml *model.ModelLoader, app
entry.Process = proc
}
}
if pid, ok := localPID(m); ok {
if used, ok := xsysinfo.ProcessVRAM(int(pid)); ok {
entry.SizeVRAM = &used
}
}
sysmodels = append(sysmodels, entry)
}
if sampler != nil {
@@ -0,0 +1,49 @@
// SPDX-License-Identifier: MIT
package localai_test
import (
"encoding/json"
"net/http"
"net/http/httptest"
"os"
"path/filepath"
"github.com/labstack/echo/v4"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/http/endpoints/localai"
"github.com/mudler/LocalAI/pkg/model"
"github.com/mudler/LocalAI/pkg/system"
process "github.com/mudler/go-processmanager"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("SystemInformations memory", func() {
It("keeps model metadata and omits VRAM for remote or stopped backends", func() {
path, err := os.MkdirTemp("", "system-info-")
Expect(err).NotTo(HaveOccurred())
DeferCleanup(os.RemoveAll, path)
configFile := filepath.Join(path, "remote.yaml")
Expect(os.WriteFile(configFile, []byte("name: remote\nbackend: llama-cpp\n"), 0600)).To(Succeed())
cl := config.NewModelConfigLoader(path)
Expect(cl.ReadModelConfig(configFile)).To(Succeed())
ml := model.NewModelLoader(&system.SystemState{})
store := model.NewInMemoryModelStore()
store.Set("remote", model.NewModel("remote", "worker:50051", nil))
store.Set("stopped", model.NewModel("stopped", "", &process.Process{}))
ml.SetModelStore(store)
app := echo.New()
app.GET("/system", localai.SystemInformations(cl, ml, &config.ApplicationConfig{}, nil))
rec := httptest.NewRecorder()
app.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/system", nil))
Expect(rec.Code).To(Equal(http.StatusOK))
var response struct {
Models []map[string]any `json:"loaded_models"`
}
Expect(json.Unmarshal(rec.Body.Bytes(), &response)).To(Succeed())
Expect(response.Models).To(ConsistOf(
map[string]any{"id": "remote", "backend": "llama-cpp"},
map[string]any{"id": "stopped"},
))
})
})
+3
View File
@@ -208,6 +208,9 @@ type SysInfoModel struct {
// when the model has no local process (a distributed worker holds it) or
// the process could not be read.
Process *SysInfoProcess `json:"process,omitempty"`
// SizeVRAM is DRM-accounted resident device memory in bytes. Nil means
// the backend process tree has no complete supported reading.
SizeVRAM *uint64 `json:"size_vram,omitempty"`
}
// SysInfoProcess is a point-in-time reading of one backend process.
+25
View File
@@ -0,0 +1,25 @@
// SPDX-License-Identifier: MIT
package schema_test
import (
"encoding/json"
"github.com/mudler/LocalAI/core/schema"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("SysInfoModel memory", func() {
It("omits unavailable VRAM while preserving a measured zero", func() {
entry := schema.SysInfoModel{ID: "model"}
encoded, err := json.Marshal(entry)
Expect(err).NotTo(HaveOccurred())
Expect(string(encoded)).To(MatchJSON(`{"id":"model"}`))
zero := uint64(0)
entry.SizeVRAM = &zero
encoded, err = json.Marshal(entry)
Expect(err).NotTo(HaveOccurred())
Expect(string(encoded)).To(MatchJSON(`{"id":"model","size_vram":0}`))
})
})