mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-28 17:15:02 -04:00
feat(system): report per-model DRM VRAM (#12026)
* feat(system): report per-model DRM VRAM Expose optional resident device memory for local backend process trees. Deduplicate DRM clients and omit unsupported or incomplete readings. Document accounting limits and preserve a measured zero in JSON. Closes #11970. Assisted-by: Codex:gpt-6 * fix(system): document trusted procfs reads Scope G304 annotations to paths built from the fixed procfs root, integer process IDs, and kernel directory entries. These reads accept no user-controlled path components. Assisted-by: Codex:GPT-6 gosec --------- Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com>
This commit is contained in:
1 parent
5794495a37
commit
f154bd990a
11 files changed
+399
No files matched your search
@@ -8,6 +8,7 @@ import (
|
||||
"github.com/mudler/LocalAI/core/schema"
|
||||
"github.com/mudler/LocalAI/core/services/monitoring"
|
||||
"github.com/mudler/LocalAI/pkg/model"
|
||||
"github.com/mudler/LocalAI/pkg/xsysinfo"
|
||||
)
|
||||
|
||||
// SystemInformations returns the system informations
|
||||
@@ -42,6 +43,11 @@ func SystemInformations(cl *config.ModelConfigLoader, ml *model.ModelLoader, app
|
||||
entry.Process = proc
|
||||
}
|
||||
}
|
||||
if pid, ok := localPID(m); ok {
|
||||
if used, ok := xsysinfo.ProcessVRAM(int(pid)); ok {
|
||||
entry.SizeVRAM = &used
|
||||
}
|
||||
}
|
||||
sysmodels = append(sysmodels, entry)
|
||||
}
|
||||
if sampler != nil {
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
package localai_test
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
"github.com/labstack/echo/v4"
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
"github.com/mudler/LocalAI/core/http/endpoints/localai"
|
||||
"github.com/mudler/LocalAI/pkg/model"
|
||||
"github.com/mudler/LocalAI/pkg/system"
|
||||
process "github.com/mudler/go-processmanager"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("SystemInformations memory", func() {
|
||||
It("keeps model metadata and omits VRAM for remote or stopped backends", func() {
|
||||
path, err := os.MkdirTemp("", "system-info-")
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
DeferCleanup(os.RemoveAll, path)
|
||||
configFile := filepath.Join(path, "remote.yaml")
|
||||
Expect(os.WriteFile(configFile, []byte("name: remote\nbackend: llama-cpp\n"), 0600)).To(Succeed())
|
||||
cl := config.NewModelConfigLoader(path)
|
||||
Expect(cl.ReadModelConfig(configFile)).To(Succeed())
|
||||
ml := model.NewModelLoader(&system.SystemState{})
|
||||
store := model.NewInMemoryModelStore()
|
||||
store.Set("remote", model.NewModel("remote", "worker:50051", nil))
|
||||
store.Set("stopped", model.NewModel("stopped", "", &process.Process{}))
|
||||
ml.SetModelStore(store)
|
||||
app := echo.New()
|
||||
app.GET("/system", localai.SystemInformations(cl, ml, &config.ApplicationConfig{}, nil))
|
||||
rec := httptest.NewRecorder()
|
||||
app.ServeHTTP(rec, httptest.NewRequest(http.MethodGet, "/system", nil))
|
||||
Expect(rec.Code).To(Equal(http.StatusOK))
|
||||
var response struct {
|
||||
Models []map[string]any `json:"loaded_models"`
|
||||
}
|
||||
Expect(json.Unmarshal(rec.Body.Bytes(), &response)).To(Succeed())
|
||||
Expect(response.Models).To(ConsistOf(
|
||||
map[string]any{"id": "remote", "backend": "llama-cpp"},
|
||||
map[string]any{"id": "stopped"},
|
||||
))
|
||||
})
|
||||
})
|
||||
@@ -208,6 +208,9 @@ type SysInfoModel struct {
|
||||
// when the model has no local process (a distributed worker holds it) or
|
||||
// the process could not be read.
|
||||
Process *SysInfoProcess `json:"process,omitempty"`
|
||||
// SizeVRAM is DRM-accounted resident device memory in bytes. Nil means
|
||||
// the backend process tree has no complete supported reading.
|
||||
SizeVRAM *uint64 `json:"size_vram,omitempty"`
|
||||
}
|
||||
|
||||
// SysInfoProcess is a point-in-time reading of one backend process.
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
package schema_test
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
|
||||
"github.com/mudler/LocalAI/core/schema"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("SysInfoModel memory", func() {
|
||||
It("omits unavailable VRAM while preserving a measured zero", func() {
|
||||
entry := schema.SysInfoModel{ID: "model"}
|
||||
encoded, err := json.Marshal(entry)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(string(encoded)).To(MatchJSON(`{"id":"model"}`))
|
||||
|
||||
zero := uint64(0)
|
||||
entry.SizeVRAM = &zero
|
||||
encoded, err = json.Marshal(entry)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(string(encoded)).To(MatchJSON(`{"id":"model","size_vram":0}`))
|
||||
})
|
||||
})
|
||||
Reference in new issue
Block a user