mirror of
https://github.com/mudler/LocalAI.git
synced 2026-07-30 18:09:05 -04:00
* feat(vram): add vrambudget primitive for per-node VRAM caps Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(vram): apply default VRAM budget in xsysinfo aggregate getters Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(vram): wire LOCALAI_VRAM_BUDGET flag to xsysinfo default budget Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(vram): persist VRAM budget via runtime settings with live apply Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * test(vram): reset process-global VRAM budget after runtime-settings spec Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(vram): add VRAM budget field to Settings page Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(vram): store and enforce per-node VRAM budget in the node registry Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(vram): apply per-node VRAM budget in router hardware defaults Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(vram): report worker VRAM budget in node registration The distributed worker now reports its operator-set VRAM budget string (LOCALAI_VRAM_BUDGET) to the server on registration. The worker keeps reporting RAW total/available VRAM and never sets the xsysinfo process-global budget (that stays standalone-only); the server resolves and enforces the budget uniformly (Task 6). Also closes a Task 6 gap: on re-registration, a struct Updates zero-skips an empty budget, so a worker that dropped LOCALAI_VRAM_BUDGET left the stale cap in place. For non-admin-override nodes the budget columns are now force-written (map Updates) even when empty, so removing the env var clears the cap; admin overrides are preserved unchanged. Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * style(vram): drop em dash from worker-clear comment Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(vram): add node VRAM budget admin endpoints Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(vram): add node VRAM budget control to the node UI Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(vram): expose set_node_vram_budget MCP admin tool Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * docs(vram): document LOCALAI_VRAM_BUDGET and node VRAM budget UI Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * fix(vram): avoid double-applying VRAM budget in GetResourceAggregateInfo The GPU-branch aggregate returned by GetResourceInfo is sourced from GetGPUAggregateInfo, which already caps total/free/used against the process-wide VRAM budget. GetResourceAggregateInfo then applied the budget a second time. For an absolute budget this is idempotent, but for a percentage budget b.Apply resolves the ceiling as a fraction of its input total, so a second pass yields P*(P*T) instead of P*T and distorts UsagePercent (read by the memory reclaimer in pkg/model/watchdog.go). Remove the redundant second application so the budget is applied exactly once, against the raw physical totals, upstream in GetGPUAggregateInfo. Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * fix(vram): implement SetNodeVRAMBudget on mcp assistant test stub The LocalAIClient interface gained SetNodeVRAMBudget; the stubClient in core/http/endpoints/mcp used by the assistant tests is a separate implementer and needs the method too (broke golangci-lint typecheck and both test jobs). Signed-off-by: Ettore Di Giacinto <mudler@localai.io> --------- Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
55 lines
2.3 KiB
Go
55 lines
2.3 KiB
Go
package localaitools
|
|
|
|
// Tool names exposed by the LocalAI Assistant MCP server. Use these
|
|
// constants — never bare strings — when registering tools, asserting the
|
|
// catalog in tests, or referencing tool names from other packages. The
|
|
// embedded skill prompts under prompts/ keep the bare strings because
|
|
// go:embed-ed markdown can't reference Go constants; TestPromptsContain
|
|
// SafetyAnchors guards that those strings stay aligned.
|
|
const (
|
|
// Read-only tools.
|
|
ToolGallerySearch = "gallery_search"
|
|
ToolListInstalledModels = "list_installed_models"
|
|
ToolListGalleries = "list_galleries"
|
|
ToolGetJobStatus = "get_job_status"
|
|
ToolGetModelConfig = "get_model_config"
|
|
ToolListBackends = "list_backends"
|
|
ToolListKnownBackends = "list_known_backends"
|
|
ToolSystemInfo = "system_info"
|
|
ToolListNodes = "list_nodes"
|
|
ToolVRAMEstimate = "vram_estimate"
|
|
ToolGetBranding = "get_branding"
|
|
ToolGetUsageStats = "get_usage_stats"
|
|
ToolGetPIIEvents = "get_pii_events"
|
|
ToolGetMiddlewareStatus = "get_middleware_status"
|
|
ToolGetRouterDecisions = "get_router_decisions"
|
|
ToolListVoiceProfiles = "list_voice_profiles"
|
|
|
|
// Mutating tools — guarded by Options.DisableMutating and the
|
|
// LLM-side safety prompt (see prompts/10_safety.md).
|
|
ToolInstallModel = "install_model"
|
|
ToolImportModelURI = "import_model_uri"
|
|
ToolDeleteModel = "delete_model"
|
|
ToolEditModelConfig = "edit_model_config"
|
|
ToolReloadModels = "reload_models"
|
|
ToolLoadModel = "load_model"
|
|
ToolInstallBackend = "install_backend"
|
|
ToolUpgradeBackend = "upgrade_backend"
|
|
ToolToggleModelState = "toggle_model_state"
|
|
ToolToggleModelPinned = "toggle_model_pinned"
|
|
ToolSetBranding = "set_branding"
|
|
ToolSetAlias = "set_alias"
|
|
ToolCreateVoiceProfile = "create_voice_profile"
|
|
ToolDeleteVoiceProfile = "delete_voice_profile"
|
|
ToolSetNodeVRAMBudget = "set_node_vram_budget"
|
|
|
|
// ToolListAliases is read-only but lives here so the alias tools stay
|
|
// grouped; the catalog tests assert its read-only placement.
|
|
ToolListAliases = "list_aliases"
|
|
)
|
|
|
|
// DefaultServerName is the MCP Implementation.Name surfaced when
|
|
// Options.ServerName is empty. Use the constant when you want a stable
|
|
// reference across packages (e.g. test fixtures, CLI defaults).
|
|
const DefaultServerName = "localai-admin"
|