chore: merge master into distributed transport PR

Keep the version-reporting import from master and omit the unused
sanitize import after the transport changes.

Assisted-by: Codex:GPT-6
This commit is contained in:
localai-org-maint-bot committed 2026-09-28 12:02:15 +00:00
commit 4151ceda85
51 files changed
+1785 -72

No files matched your search

+1 -1
View File
@@ -9,7 +9,7 @@
# recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean
# rebuild and so the bump bot can see the pin.
AUDIO_CPP_VERSION?=94bd4656399180befc141b17bd6696bf84df0a9f
AUDIO_CPP_VERSION?=77491a33c589c53ff18add050095cf35647c8213
AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp
CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST))))
+1 -1
View File
@@ -1,5 +1,5 @@
IK_LLAMA_VERSION?=cdf232cc17e410e60c1bc3b85516c4a41199b662
IK_LLAMA_VERSION?=ed27bf7ed25e637692e89cd341d802522a2cee8a
LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp
CMAKE_ARGS?=
+5 -3
View File
@@ -539,8 +539,10 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
// Initialize ctx_shift to false by default (can be overridden by options)
params.ctx_shift = false;
// Initialize cache_ram_mib to -1 by default (no limit, can be overridden by options)
params.cache_ram_mib = -1;
// cache_ram_mib keeps llama.cpp's own default (8192 MiB) unless overridden by
// options. It used to be forced to -1 (no limit): since kv_unified and
// cache_idle_slots are on by default, every distinct prompt then leaves its
// slot state in host RAM and the backend grows without bound.
// Initialize n_parallel to 1 by default (can be overridden by options)
params.n_parallel = 1;
// Initialize grpc_servers to empty (can be overridden by options)
@@ -656,7 +658,7 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
try {
params.cache_ram_mib = std::stoi(optval_str);
} catch (const std::exception& e) {
// If conversion fails, keep default value (-1)
// If conversion fails, keep the default value
}
}
} else if (!strcmp(optname, "parallel") || !strcmp(optname, "n_parallel")) {
+1 -1
View File
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
# CrispASR version (release tag)
CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR
CRISPASR_VERSION?=013ae1624dc40ecf059065d577180722439f804e
CRISPASR_VERSION?=ec98831d0776ec8a16ccaf93955693eb7ecfbec3
SO_TARGET?=libgocrispasr.so
CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
+1 -1
View File
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
# omnivoice.cpp version
OMNIVOICE_REPO?=https://github.com/ServeurpersoCom/omnivoice.cpp
OMNIVOICE_VERSION?=8ab42195a05a9d48a3942b17568c1f3a876e133a
OMNIVOICE_VERSION?=ead199a2bc4c53a57cac90095ae049a111d9e98d
SO_TARGET?=libgomnivoicecpp.so
CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
+1 -1
View File
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
# stablediffusion.cpp (ggml)
STABLEDIFFUSION_GGML_REPO?=https://github.com/leejet/stable-diffusion.cpp
STABLEDIFFUSION_GGML_VERSION?=2f886889e6e8b78738d6b87f7191f6018557c551
STABLEDIFFUSION_GGML_VERSION?=3f8527a46c54ecf4cb4ed6003da8e8982283c73c
CMAKE_ARGS+=-DGGML_MAX_NAME=128
+1 -1
View File
@@ -137,8 +137,8 @@ func (sd *SDGGML) Load(opts *pb.ModelOptions) error {
// If it's an option path, we resolve absolute path from the model path
if strings.Contains(op, ":") && strings.Contains(op, "path") {
data := strings.Split(op, ":")
data[1] = filepath.Join(opts.ModelPath, data[1])
if err := utils.VerifyPath(data[1], opts.ModelPath); err == nil {
data[1] = filepath.Join(opts.ModelPath, data[1])
oo = append(oo, strings.Join(data, ":"))
}
} else {
+1 -1
View File
@@ -161,10 +161,10 @@ func resolveModels(modelFile, modelPath string, options []string) (modelSet, err
continue
}
if !filepath.IsAbs(value) {
value = filepath.Join(modelPath, value)
if err := utils.VerifyPath(value, modelPath); err != nil {
return modelSet{}, fmt.Errorf("option %s: %w", key, err)
}
value = filepath.Join(modelPath, value)
}
overrides[key] = value
}
+4 -1
View File
@@ -134,9 +134,12 @@ var _ = Describe("resolveModels", func() {
It("rejects option paths escaping the model directory", func() {
touch(dir, fullSet...)
// The escaping file exists, so only the containment check can
// reject it; a missing file would fail for an unrelated reason.
touch(filepath.Dir(dir), "outside.gguf")
_, err := resolveModels("ss_flow_f16.gguf", dir, []string{"dino_path:../outside.gguf"})
Expect(err).To(HaveOccurred())
Expect(err).To(MatchError(ContainSubstring("outside of trusted root")))
})
})
+146 -11
View File
@@ -90,6 +90,19 @@ except Exception:
_SEED_KEY = "sampling_seed"
# Engine.async_generate() only grew a require_reasoning keyword in sglang
# 0.5.13. The CPU build compiles v0.5.11 from source and the other profiles
# only set a >=0.5.11 floor, and async_generate() takes no **kwargs, so
# passing the keyword unconditionally fails every request with TypeError.
try:
import inspect as _inspect
_ASYNC_GENERATE_HAS_REQUIRE_REASONING = (
"require_reasoning" in _inspect.signature(Engine.async_generate).parameters
)
except Exception:
_ASYNC_GENERATE_HAS_REQUIRE_REASONING = False
_ONE_DAY_IN_SECONDS = 60 * 60 * 24
# proto3 has no field presence, so an explicit 0 is indistinguishable from
@@ -105,6 +118,12 @@ MAX_WORKERS = int(os.environ.get('PYTHON_GRPC_MAX_WORKERS', '1'))
class BackendServicer(backend_pb2_grpc.BackendServicer):
"""gRPC servicer implementing the Backend service for sglang."""
# Class-level default so a servicer used before LoadModel (e.g. in unit
# tests that construct it directly) doesn't AttributeError in
# _build_sampling_params.
thinking_budget: Optional[int] = None
reasoning_default: Optional[str] = None
def _parse_options(self, options_list) -> Dict[str, str]:
opts: Dict[str, str] = {}
for opt in options_list:
@@ -114,6 +133,49 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
opts[key.strip()] = value.strip()
return opts
@staticmethod
def _parse_thinking_budget(value) -> Optional[int]:
"""Turn the `thinking_budget` model option into a positive int, or None.
Options arrive as strings from the YAML `options:` list, but a value
like "5000.0" is a plausible thing to write, and a crash here would
take down LoadModel for the whole model. So: integral numbers are
accepted in any spelling ("512", "512.0"), anything else is ignored
with a warning instead of raising. Zero and negative budgets are
ignored too: sglang gives them no defined meaning, and turning
reasoning off is what `reasoning_default: off` is for.
"""
if value is None or str(value).strip() == "":
return None
raw = str(value).strip()
try:
number = float(raw)
except ValueError:
print(f"thinking_budget {raw!r} is not a number, ignoring it", file=sys.stderr)
return None
if not number.is_integer():
print(f"thinking_budget {raw!r} is not a whole number of tokens, ignoring it", file=sys.stderr)
return None
if number <= 0:
print(
f"thinking_budget {raw!r} must be positive, ignoring it "
"(use reasoning_default:off to disable reasoning)",
file=sys.stderr,
)
return None
return int(number)
@staticmethod
def _strict_thinking_warning(thinking_budget: Optional[int], engine_kwargs: dict) -> Optional[str]:
"""sglang only enforces the budget with enable_strict_thinking on; without
it the budget is silently ignored, so say so at load time."""
if thinking_budget is not None and not engine_kwargs.get("enable_strict_thinking"):
return (
f"thinking_budget={thinking_budget} is set but enable_strict_thinking is not "
"in engine_args; sglang will ignore the budget"
)
return None
def _apply_engine_args(self, engine_kwargs: dict, engine_args_json: str) -> dict:
"""Merge user-supplied engine_args (JSON object) into the kwargs dict
that will be forwarded to ``sglang.Engine`` (which constructs a
@@ -230,6 +292,35 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
self.tool_parser_name: Optional[str] = opts.get("tool_parser") or None
self.reasoning_parser_name: Optional[str] = opts.get("reasoning_parser") or None
# Fixed reasoning-length budget for every request on this model, in
# tokens. There is no protobuf field to carry a per-request
# custom_params blob, so this rides the same model-level `options:`
# mechanism as tool_parser/reasoning_parser above — mirroring how
# sglang's own `--preferred-sampling-params` is a server-wide
# default, not a per-request choice. Requires `enable_strict_thinking`
# in `engine_args:` (sglang >=0.5.12); without it sglang has no
# tokenizer-derived budget mechanism to enforce this against.
self.thinking_budget: Optional[int] = self._parse_thinking_budget(
opts.get("thinking_budget")
)
# Model-level default for whether the chat template opens a reasoning
# block, as "off" or "on". Rides the same `options:` mechanism as
# thinking_budget above.
#
# Why this is needed even though `reasoning_effort` exists: that one
# only reaches this backend when a *caller* sets it per request (the
# Go side turns it into Metadata["enable_thinking"]). As a model-level
# `parameters:` default it is silently dropped, so a config reading
# `reasoning_effort: none` still produces full reasoning on every
# request - the config says one thing and the model does another.
#
# A per-request value always wins; this only fills in the gap when the
# request says nothing.
self.reasoning_default: Optional[str] = (
opts.get("reasoning_default") or ""
).lower() or None
# Also hand the parser names to sglang's engine so its HTTP/OAI
# paths work identically if someone hits the engine directly.
if self.tool_parser_name:
@@ -247,6 +338,10 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
print(f"engine_args error: {err}", file=sys.stderr)
return backend_pb2.Result(success=False, message=str(err))
warning = self._strict_thinking_warning(self.thinking_budget, engine_kwargs)
if warning:
print(warning, file=sys.stderr)
try:
self.llm = Engine(**engine_kwargs)
except Exception as err:
@@ -362,8 +457,28 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
except json.JSONDecodeError:
sampling_params["ebnf"] = grammar
if self.thinking_budget is not None:
sampling_params["custom_params"] = {"thinking_budget": self.thinking_budget}
return sampling_params
def _thinking_default(self, request) -> Optional[bool]:
"""Whether this request should render with reasoning on, off, or unset.
Per-request ``Metadata["enable_thinking"]`` wins; the model-level
``reasoning_default`` option fills in when the request is silent.
Returns None when neither says anything, leaving template behaviour
untouched.
"""
wanted = request.Metadata.get("enable_thinking", "").lower()
if wanted in ("true", "false"):
return wanted == "true"
if self.reasoning_default == "off":
return False
if self.reasoning_default == "on":
return True
return None
def _build_prompt(self, request) -> str:
prompt = request.Prompt
if prompt or not request.UseTokenizerTemplate or not request.Messages:
@@ -384,9 +499,9 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
template_kwargs["tools"] = json.loads(request.Tools)
except json.JSONDecodeError:
pass
_thinking = request.Metadata.get("enable_thinking", "").lower()
if _thinking in ("true", "false"):
template_kwargs["enable_thinking"] = (_thinking == "true")
_thinking = self._thinking_default(request)
if _thinking is not None:
template_kwargs["enable_thinking"] = _thinking
# sglang locates the attached images/videos by scanning the rendered
# prompt for the model's own media token, so the template has to be
@@ -438,12 +553,19 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
there files the answer as reasoning and leaves content empty. sglang's
own server keeps the two apart for the same reason — its grammar
backend owns the reasoning prefix when a reasoning parser is set.
Returns a ``(parser, forced)`` pair. ``forced`` is also the signal
``_predict`` passes as ``Engine.async_generate(require_reasoning=...)``:
sglang's own OpenAI server derives that flag from per-template
config (``ChatServing._get_reasoning_from_request``); this backend
has no template manager, so the same prompt-suffix heuristic that
already decides parser forcing doubles as that signal.
"""
if grammar_constrained:
prompt = ""
if not (HAS_REASONING_PARSERS and self.reasoning_parser_name):
return None
return None, False
kwargs = {
"model_type": self.reasoning_parser_name,
@@ -453,10 +575,12 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
parser = ReasoningParser(**kwargs)
except Exception as e:
print(f"ReasoningParser init failed: {e!r}", file=sys.stderr)
return None
return None, False
forced = False
start = getattr(getattr(parser, "detector", None), "think_start_token", None)
if start and prompt and prompt.rstrip().endswith(start):
forced = True
try:
parser = ReasoningParser(force_reasoning=True, **kwargs)
except TypeError:
@@ -469,10 +593,16 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
file=sys.stderr,
)
return parser
return parser, forced
def _make_parsers(self, request, prompt: str = ""):
"""Construct fresh per-request parser instances (stateful)."""
"""Construct fresh per-request parser instances (stateful).
Also returns ``require_reasoning`` (see ``_new_reasoning_parser``),
which ``_predict`` forwards to ``Engine.async_generate()`` so
sglang's ``--enable-strict-thinking`` grammar backend knows this
request is in a reasoning block.
"""
tool_parser = None
if HAS_TOOL_PARSERS and self.tool_parser_name and request.Tools:
@@ -485,23 +615,27 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
except Exception as e:
print(f"FunctionCallParser init failed: {e!r}", file=sys.stderr)
reasoning_parser = self._new_reasoning_parser(
reasoning_parser, require_reasoning = self._new_reasoning_parser(
True, prompt, bool(getattr(request, "Grammar", "")),
)
return tool_parser, reasoning_parser
return tool_parser, reasoning_parser, require_reasoning
async def _predict(self, request, context, streaming: bool = False):
sampling_params = self._build_sampling_params(request)
prompt = self._build_prompt(request)
tool_parser, reasoning_parser = self._make_parsers(request, prompt)
tool_parser, reasoning_parser, require_reasoning = self._make_parsers(request, prompt)
image_data = list(request.Images) if request.Images else None
video_data = list(request.Videos) if request.Videos else None
# Kick off streaming generation. We always use stream=True so the
# non-stream path still gets parser coverage on the final text.
generate_kwargs = {}
if _ASYNC_GENERATE_HAS_REQUIRE_REASONING:
generate_kwargs["require_reasoning"] = require_reasoning
try:
iterator = await self.llm.async_generate(
prompt=prompt,
@@ -509,6 +643,7 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
image_data=image_data,
video_data=video_data,
stream=True,
**generate_kwargs,
)
except Exception as e:
print(f"sglang async_generate failed: {e!r}", file=sys.stderr)
@@ -591,7 +726,7 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
final_tool_calls: List[backend_pb2.ToolCallDelta] = []
if not streaming:
final_reasoning_parser = self._new_reasoning_parser(
final_reasoning_parser, _ = self._new_reasoning_parser(
False, prompt, bool(getattr(request, "Grammar", "")),
)
+121 -5
View File
@@ -9,6 +9,13 @@ because ``_apply_engine_args`` validates keys against ``ServerArgs``
import unittest
def _request(metadata=None):
"""Minimal stand-in for a PredictOptions request in reasoning tests."""
from types import SimpleNamespace
return SimpleNamespace(Metadata=metadata or {})
class TestSglangHelpers(unittest.TestCase):
"""Tests for the pure helpers on BackendServicer (no gRPC, no engine)."""
@@ -170,13 +177,17 @@ class TestSglangHelpers(unittest.TestCase):
# What the model actually emits when the prompt ends in "<think>".
completion = "adding two and two</think>4"
forced = servicer._new_reasoning_parser(False, prompt="user: hi\n<think>\n")
forced, require_reasoning = servicer._new_reasoning_parser(
False, prompt="user: hi\n<think>\n"
)
self.assertTrue(require_reasoning)
reasoning, content = forced.parse_non_stream(completion)
self.assertEqual(reasoning, "adding two and two")
self.assertEqual(content, "4")
# No prefilled tag in the prompt: detector default, unchanged behaviour.
unforced = servicer._new_reasoning_parser(False, prompt="user: hi\n")
unforced, require_reasoning = servicer._new_reasoning_parser(False, prompt="user: hi\n")
self.assertFalse(require_reasoning)
reasoning, content = unforced.parse_non_stream(completion)
self.assertFalse(reasoning)
self.assertEqual(content, completion)
@@ -187,7 +198,8 @@ class TestSglangHelpers(unittest.TestCase):
servicer = self._servicer()
servicer.reasoning_parser_name = "qwen3"
parser = servicer._new_reasoning_parser(False, prompt="user: primes?\n")
parser, require_reasoning = servicer._new_reasoning_parser(False, prompt="user: primes?\n")
self.assertFalse(require_reasoning)
reasoning, content = parser.parse_non_stream("2,3,5,7,11")
self.assertFalse(reasoning)
self.assertEqual(content, "2,3,5,7,11")
@@ -200,9 +212,10 @@ class TestSglangHelpers(unittest.TestCase):
servicer.reasoning_parser_name = "qwen3"
schema_out = '{"findings": [{"line": 42, "issue": "off-by-one"}]}'
parser = servicer._new_reasoning_parser(
parser, require_reasoning = servicer._new_reasoning_parser(
False, prompt="audit this\n<think>\n", grammar_constrained=True,
)
self.assertFalse(require_reasoning)
reasoning, content = parser.parse_non_stream(schema_out)
self.assertFalse(reasoning)
self.assertEqual(content, schema_out)
@@ -210,7 +223,110 @@ class TestSglangHelpers(unittest.TestCase):
def test_reasoning_parser_absent_without_configured_parser(self):
servicer = self._servicer()
servicer.reasoning_parser_name = None
self.assertIsNone(servicer._new_reasoning_parser(False, prompt="<think>"))
parser, require_reasoning = servicer._new_reasoning_parser(False, prompt="<think>")
self.assertIsNone(parser)
self.assertFalse(require_reasoning)
def test_reasoning_default_off_applies_when_request_is_silent(self):
"""A model configured with reasoning_default:off must render with
thinking disabled even when the request carries no enable_thinking -
that is the whole point: `parameters: reasoning_effort:` never
reaches this backend, so without this the config lies about the
default."""
servicer = self._servicer()
servicer.reasoning_default = "off"
self.assertIs(servicer._thinking_default(_request(metadata={})), False)
def test_request_metadata_overrides_reasoning_default(self):
"""A per-request value always wins over the model-level default -
in both directions."""
servicer = self._servicer()
servicer.reasoning_default = "off"
self.assertIs(
servicer._thinking_default(_request(metadata={"enable_thinking": "true"})),
True,
)
servicer.reasoning_default = "on"
self.assertIs(
servicer._thinking_default(_request(metadata={"enable_thinking": "false"})),
False,
)
def test_no_reasoning_default_leaves_template_untouched(self):
"""Unconfigured must stay unconfigured: returning None means the
backend adds no enable_thinking kwarg at all, so the template keeps
whatever default it ships with."""
servicer = self._servicer()
self.assertIsNone(servicer._thinking_default(_request(metadata={})))
def test_thinking_budget_added_to_sampling_params_as_custom_params(self):
"""The model-level thinking_budget option (set from LoadModel's
Options, mirroring tool_parser/reasoning_parser) must ride along as
sampling_params['custom_params']['thinking_budget'] on every
request — that's the only field sglang's --enable-strict-thinking
grammar backend reads to bound the reasoning length."""
from types import SimpleNamespace
servicer = self._servicer()
servicer.thinking_budget = 512
request = SimpleNamespace(
Temperature=0.7, N=0, PresencePenalty=0, FrequencyPenalty=0,
RepetitionPenalty=0, TopP=0, TopK=0, MinP=0, Seed=0,
StopPrompts=[], StopTokenIds=[], IgnoreEOS=False, Tokens=0,
MinTokens=0, SkipSpecialTokens=False, Grammar="",
)
params = servicer._build_sampling_params(request)
self.assertEqual(params["custom_params"], {"thinking_budget": 512})
def test_no_thinking_budget_means_no_custom_params_key(self):
"""Unconfigured is unconfigured: no thinking_budget option must not
add an empty/None custom_params that could clobber a sglang-side
--preferred-sampling-params default (see sglang#40634)."""
from types import SimpleNamespace
servicer = self._servicer()
request = SimpleNamespace(
Temperature=0.7, N=0, PresencePenalty=0, FrequencyPenalty=0,
RepetitionPenalty=0, TopP=0, TopK=0, MinP=0, Seed=0,
StopPrompts=[], StopTokenIds=[], IgnoreEOS=False, Tokens=0,
MinTokens=0, SkipSpecialTokens=False, Grammar="",
)
params = servicer._build_sampling_params(request)
self.assertNotIn("custom_params", params)
def test_thinking_budget_accepts_integral_spellings(self):
"""YAML options arrive as strings; "512" and "512.0" both mean 512."""
servicer = self._servicer()
self.assertEqual(servicer._parse_thinking_budget("512"), 512)
self.assertEqual(servicer._parse_thinking_budget("512.0"), 512)
self.assertEqual(servicer._parse_thinking_budget(" 64 "), 64)
self.assertEqual(servicer._parse_thinking_budget(256), 256)
def test_thinking_budget_unset_is_none(self):
servicer = self._servicer()
self.assertIsNone(servicer._parse_thinking_budget(None))
self.assertIsNone(servicer._parse_thinking_budget(""))
def test_thinking_budget_zero_and_negative_are_ignored(self):
"""No defined meaning in sglang -- ignored, not passed through."""
servicer = self._servicer()
self.assertIsNone(servicer._parse_thinking_budget("0"))
self.assertIsNone(servicer._parse_thinking_budget("-100"))
def test_thinking_budget_non_integer_does_not_raise(self):
"""A bad value must not crash LoadModel for the whole model."""
servicer = self._servicer()
self.assertIsNone(servicer._parse_thinking_budget("12.5"))
self.assertIsNone(servicer._parse_thinking_budget("lots"))
def test_warns_when_budget_set_without_strict_thinking(self):
servicer = self._servicer()
self.assertIn(
"enable_strict_thinking",
servicer._strict_thinking_warning(512, {"model_path": "x"}),
)
self.assertIsNone(servicer._strict_thinking_warning(512, {"enable_strict_thinking": True}))
self.assertIsNone(servicer._strict_thinking_warning(None, {}))
def test_explicit_zero_temperature_and_seed_are_preserved(self):
"""Temperature=0 is greedy decoding and 0 is a valid seed — neither is
+5
View File
@@ -623,9 +623,13 @@ func initDistributed(cfg *config.ApplicationConfig, authDB *gorm.DB, configLoade
// All dependencies ready — build SmartRouter with all options at once
var conflictResolver nodes.ConcurrencyConflictResolver
var pinnedResolver nodes.PinnedModelResolver
var modelFiles func(string) []string
if configLoader != nil {
conflictResolver = configLoader
pinnedResolver = configLoader
if cfg.SystemState != nil {
modelFiles = declaredModelFiles(configLoader, cfg.SystemState.Model.ModelsPath)
}
}
modelCleanup := nodes.NewModelCleanupService(registry, remoteUnloader)
// Absence is stamped on by distributedSchedulerOptions rather than written
@@ -645,6 +649,7 @@ func initDistributed(cfg *config.ApplicationConfig, authDB *gorm.DB, configLoade
DataPath: cfg.DataPath,
ConflictResolver: conflictResolver,
PinnedResolver: pinnedResolver,
ModelFiles: modelFiles,
PrefixProvider: prefixProvider,
PrefixConfig: prefixCfg,
Pressure: pressure,
+29
View File
@@ -0,0 +1,29 @@
package application
import (
"path/filepath"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/gallery"
"github.com/mudler/LocalAI/pkg/utils"
)
// declaredModelFiles resolves the files a model needs on disk beyond the ones
// its config names: what its gallery install or import declared under
// `files:`, and what the config itself lists under download_files. The
// distributed router stages these to workers, which cannot see the frontend's
// models directory.
func declaredModelFiles(configLoader *config.ModelConfigLoader, modelsPath string) func(modelName string) []string {
return func(modelName string) []string {
files := gallery.InstalledModelFiles(modelsPath, modelName)
if cfg, ok := configLoader.GetModelConfig(modelName); ok {
for _, f := range cfg.DownloadFiles {
if utils.VerifyPath(f.Filename, modelsPath) != nil {
continue
}
files = append(files, filepath.Join(modelsPath, f.Filename))
}
}
return files
}
}
+41
View File
@@ -0,0 +1,41 @@
package application
import (
"os"
"path/filepath"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/gallery"
)
var _ = Describe("declaredModelFiles", func() {
It("combines the gallery install's files with the config's download_files", func() {
modelsPath := GinkgoT().TempDir()
Expect(os.WriteFile(filepath.Join(modelsPath, "big.yaml"), []byte(`
name: big
backend: llama-cpp
parameters:
model: big/Big-00001-of-00002.gguf
download_files:
- filename: big/extra.bin
uri: https://example.com/extra.bin
`), 0o644)).To(Succeed())
Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("big")), []byte(`
files:
- filename: big/Big-00001-of-00002.gguf
- filename: big/Big-00002-of-00002.gguf
`), 0o644)).To(Succeed())
loader := config.NewModelConfigLoader(modelsPath)
Expect(loader.LoadModelConfigsFromPath(modelsPath)).To(Succeed())
Expect(declaredModelFiles(loader, modelsPath)("big")).To(ConsistOf(
filepath.Join(modelsPath, "big/Big-00001-of-00002.gguf"),
filepath.Join(modelsPath, "big/Big-00002-of-00002.gguf"),
filepath.Join(modelsPath, "big/extra.bin"),
))
})
})
+2 -2
View File
@@ -88,7 +88,7 @@ func ModelTTS(
// a FS path
mp := filepath.Join(loader.ModelPath, modelConfig.Model)
if _, err := os.Stat(mp); err == nil {
if err := utils.VerifyPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil {
if err := utils.VerifyResolvedPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil {
return "", nil, err
}
modelPath = mp
@@ -189,7 +189,7 @@ func ModelTTSStream(
// a FS path
mp := filepath.Join(loader.ModelPath, modelConfig.Model)
if _, err := os.Stat(mp); err == nil {
if err := utils.VerifyPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil {
if err := utils.VerifyResolvedPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil {
return err
}
modelPath = mp
+3
View File
@@ -20,6 +20,7 @@ import (
"github.com/mudler/LocalAI/core/services/jobs"
mcpRemote "github.com/mudler/LocalAI/core/services/mcp"
"github.com/mudler/LocalAI/core/services/messaging"
"github.com/mudler/LocalAI/internal"
"github.com/mudler/cogito"
"github.com/mudler/cogito/clients"
"github.com/mudler/xlog"
@@ -163,6 +164,8 @@ func (cmd *AgentWorkerCMD) Run(ctx *cliContext.Context) error {
registrationBody := map[string]any{
"name": nodeName,
"node_type": "agent",
"version": internal.Version,
"commit": internal.Commit,
}
if cmd.RegistrationToken != "" {
registrationBody["token"] = cmd.RegistrationToken
+72
View File
@@ -0,0 +1,72 @@
package gallery_test
import (
"os"
"path/filepath"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"github.com/mudler/LocalAI/core/gallery"
"github.com/mudler/LocalAI/pkg/system"
)
// DeleteModelFromSystem removes files named by a model name and by the
// model's gallery file. Neither may reach outside the models directory: the
// name can come from an API caller or from an assistant tool call, and the
// gallery file is a YAML file on disk.
var _ = Describe("DeleteModelFromSystem path containment", func() {
var (
root string
modelsPath string
outside string
state *system.SystemState
)
BeforeEach(func() {
root = GinkgoT().TempDir()
modelsPath = filepath.Join(root, "models")
outside = filepath.Join(root, "outside")
Expect(os.MkdirAll(modelsPath, 0o755)).To(Succeed())
Expect(os.MkdirAll(outside, 0o755)).To(Succeed())
var err error
state, err = system.GetSystemState(system.WithModelPath(modelsPath))
Expect(err).ToNot(HaveOccurred())
})
It("refuses a model name that escapes the models directory", func() {
victim := filepath.Join(outside, "victim.yaml")
Expect(os.WriteFile(victim, []byte("name: victim\n"), 0o644)).To(Succeed())
Expect(gallery.DeleteModelFromSystem(state, "../outside/victim")).ToNot(Succeed())
Expect(victim).To(BeARegularFile())
})
It("does not remove gallery-declared files outside the models directory", func() {
secret := filepath.Join(outside, "secret.bin")
Expect(os.WriteFile(secret, []byte("x"), 0o644)).To(Succeed())
Expect(os.WriteFile(filepath.Join(modelsPath, "m.yaml"), []byte("name: m\n"), 0o644)).To(Succeed())
Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("m")), []byte(`
files:
- filename: ../outside/secret.bin
`), 0o644)).To(Succeed())
_ = gallery.DeleteModelFromSystem(state, "m")
Expect(secret).To(BeARegularFile())
})
It("still deletes a normal model and its declared files", func() {
weights := filepath.Join(modelsPath, "m", "w.gguf")
Expect(os.MkdirAll(filepath.Dir(weights), 0o755)).To(Succeed())
Expect(os.WriteFile(weights, []byte("w"), 0o644)).To(Succeed())
Expect(os.WriteFile(filepath.Join(modelsPath, "m.yaml"), []byte("name: m\nparameters:\n model: m/w.gguf\n"), 0o644)).To(Succeed())
Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("m")), []byte(`
files:
- filename: m/w.gguf
`), 0o644)).To(Succeed())
Expect(gallery.DeleteModelFromSystem(state, "m")).To(Succeed())
Expect(weights).ToNot(BeAnExistingFile())
Expect(filepath.Join(modelsPath, "m.yaml")).ToNot(BeAnExistingFile())
})
})
+46
View File
@@ -0,0 +1,46 @@
package gallery
import (
"os"
"path/filepath"
"strings"
"github.com/mudler/LocalAI/pkg/utils"
"github.com/mudler/xlog"
)
// InstalledModelFiles returns the absolute paths of the files that the install
// of model name declared (the entry's `files:`), as recorded in its gallery
// file. A model config names only the file a backend opens first, while a
// backend can read more by itself (llama.cpp opens the other shards of a split
// GGUF by name), so this is the complete list of what the model needs on disk.
// It returns nil for a model that was not installed from a gallery or import.
func InstalledModelFiles(modelsPath, name string) []string {
// Model names can hold path separators; the gallery file flattens them
// the same way listModelFiles does.
rel := galleryFileName(strings.ReplaceAll(name, string(os.PathSeparator), "__"))
if err := utils.VerifyPath(rel, modelsPath); err != nil {
return nil
}
galleryFile := filepath.Join(modelsPath, rel)
if _, err := os.Stat(galleryFile); err != nil {
return nil
}
cfg, err := ReadConfigFile[ModelConfig](galleryFile)
if err != nil {
xlog.Warn("Failed to read gallery file for installed model files", "model", name, "file", galleryFile, "error", err)
return nil
}
files := make([]string, 0, len(cfg.Files))
for _, f := range cfg.Files {
// VerifyPath joins its argument onto modelsPath itself, so it must
// get the relative name; an absolute path would always pass.
if err := utils.VerifyPath(f.Filename, modelsPath); err != nil {
xlog.Warn("Ignoring declared model file outside the models path", "model", name, "file", f.Filename)
continue
}
files = append(files, filepath.Join(modelsPath, f.Filename))
}
return files
}
+53
View File
@@ -0,0 +1,53 @@
package gallery_test
import (
"os"
"path/filepath"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"github.com/mudler/LocalAI/core/gallery"
)
var _ = Describe("InstalledModelFiles", func() {
var modelsPath string
BeforeEach(func() {
modelsPath = GinkgoT().TempDir()
})
writeGalleryFile := func(name, body string) {
Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName(name)), []byte(body), 0o644)).To(Succeed())
}
It("returns the files the install declared, under the models path", func() {
writeGalleryFile("big", `
name: big
files:
- filename: llama-cpp/models/big/Big-00001-of-00002.gguf
uri: huggingface://org/repo/Big-00001-of-00002.gguf
- filename: llama-cpp/models/big/Big-00002-of-00002.gguf
uri: huggingface://org/repo/Big-00002-of-00002.gguf
`)
Expect(gallery.InstalledModelFiles(modelsPath, "big")).To(Equal([]string{
filepath.Join(modelsPath, "llama-cpp/models/big/Big-00001-of-00002.gguf"),
filepath.Join(modelsPath, "llama-cpp/models/big/Big-00002-of-00002.gguf"),
}))
})
It("drops entries that escape the models path", func() {
writeGalleryFile("evil", `
files:
- filename: ../outside.gguf
- filename: ok.gguf
`)
Expect(gallery.InstalledModelFiles(modelsPath, "evil")).To(Equal([]string{
filepath.Join(modelsPath, "ok.gguf"),
}))
})
It("returns nothing for a model that was not installed from a gallery", func() {
Expect(gallery.InstalledModelFiles(modelsPath, "handwritten")).To(BeEmpty())
})
})
+7 -4
View File
@@ -808,8 +808,11 @@ func GetLocalModelConfiguration(basePath string, name string) (*ModelConfig, err
func listModelFiles(systemState *system.SystemState, name string) ([]string, error) {
// VerifyPath joins its argument onto the models path itself, so every
// check below passes the relative name: an already-joined absolute path
// always lands inside the base and the check would pass anything.
configFile := filepath.Join(systemState.Model.ModelsPath, fmt.Sprintf("%s.yaml", name))
if err := utils.VerifyPath(configFile, systemState.Model.ModelsPath); err != nil {
if err := utils.VerifyPath(fmt.Sprintf("%s.yaml", name), systemState.Model.ModelsPath); err != nil {
return nil, fmt.Errorf("failed to verify path %s: %w", configFile, err)
}
@@ -817,7 +820,7 @@ func listModelFiles(systemState *system.SystemState, name string) ([]string, err
name = strings.ReplaceAll(name, string(os.PathSeparator), "__")
galleryFile := filepath.Join(systemState.Model.ModelsPath, galleryFileName(name))
if err := utils.VerifyPath(galleryFile, systemState.Model.ModelsPath); err != nil {
if err := utils.VerifyPath(galleryFileName(name), systemState.Model.ModelsPath); err != nil {
return nil, fmt.Errorf("failed to verify path %s: %w", galleryFile, err)
}
@@ -847,7 +850,7 @@ func listModelFiles(systemState *system.SystemState, name string) ([]string, err
if err == nil && galleryconfig != nil {
for _, f := range galleryconfig.Files {
fullPath := filepath.Join(systemState.Model.ModelsPath, f.Filename)
if err := utils.VerifyPath(fullPath, systemState.Model.ModelsPath); err != nil {
if err := utils.VerifyPath(f.Filename, systemState.Model.ModelsPath); err != nil {
return allFiles, fmt.Errorf("failed to verify path %s: %w", fullPath, err)
}
allFiles = append(allFiles, fullPath)
@@ -858,7 +861,7 @@ func listModelFiles(systemState *system.SystemState, name string) ([]string, err
for _, f := range additionalFiles {
fullPath := filepath.Join(filepath.Join(systemState.Model.ModelsPath, f))
if err := utils.VerifyPath(fullPath, systemState.Model.ModelsPath); err != nil {
if err := utils.VerifyPath(f, systemState.Model.ModelsPath); err != nil {
return allFiles, fmt.Errorf("failed to verify path %s: %w", fullPath, err)
}
allFiles = append(allFiles, fullPath)
+7
View File
@@ -114,6 +114,11 @@ type RegisterNodeRequest struct {
// VRAMBudget is the worker's operator-set VRAM cap ("80%" or "12GB"). The
// registry resolves and enforces it against the raw reported VRAM.
VRAMBudget string `json:"vram_budget,omitempty"`
// Version is the LocalAI build version reported by the worker at
// registration. Empty for workers registered before this field existed.
Version string `json:"version,omitempty"`
// Commit is the git commit hash the worker binary was built from.
Commit string `json:"commit,omitempty"`
}
// RegisterNodeEndpoint registers a new backend node.
@@ -191,6 +196,8 @@ func RegisterNodeEndpoint(registry *nodes.NodeRegistry, expectedToken string, au
Capability: req.Capability,
MaxReplicasPerModel: maxReplicasPerModel,
VRAMBudget: req.VRAMBudget,
Version: req.Version,
Commit: req.Commit,
}
ctx := c.Request().Context()
+7
View File
@@ -9966,6 +9966,13 @@ button.collapsible-header:focus-visible {
.node-inspector__actions .btn { justify-content: center; min-width: 0; }
.node-inspector__back { align-items: center; background: transparent; border: 0; color: var(--color-primary); cursor: pointer; display: flex; font: inherit; font-size: var(--text-xs); gap: 6px; max-width: 285px; overflow: hidden; padding: 3px 0; text-overflow: ellipsis; white-space: nowrap; }
.node-inspector__back:focus-visible { border-radius: var(--radius-sm); outline: 2px solid var(--color-primary); outline-offset: 3px; }
.node-inspector__models { margin: 10px 0 0; }
.node-inspector__models > dd { margin: 0; }
.node-inspector__model-list { display: grid; gap: 4px; list-style: none; margin: 6px 0 0; padding: 0; }
.node-inspector__model-row { align-items: center; display: flex; flex-wrap: wrap; gap: 6px; font-size: var(--text-xs); }
.node-inspector__model-row .cell-mono { font-family: var(--font-mono); font-size: .625rem; overflow-wrap: anywhere; }
.node-inspector__model-row .state-pill { border-radius: var(--radius-full); font-size: .5625rem; font-weight: 600; padding: 1px 7px; text-transform: capitalize; }
.node-inspector__model-row .text-muted { font-size: .5625rem; }
.model-inspector__backends { margin-top: 10px; }
.model-inspector__nodes { display: grid; gap: 9px; }
.model-inspector__node { background: var(--color-bg-tertiary); border: 1px solid var(--color-border-subtle); border-radius: var(--radius-md); padding: 10px; }
@@ -1,6 +1,6 @@
import { useEffect, useRef, useState } from 'react'
import StatusPill from './StatusPill'
import { formatBytes, formatCapacity, timeAgo } from './nodeStatus'
import { formatBytes, formatCapacity, timeAgo, modelStateConfig } from './nodeStatus'
import { nodesApi } from '../../utils/api'
import { capacityReading, nodeLifecycleAction } from '../../utils/nodeFleet'
import useInspectorDrawer from './useInspectorDrawer'
@@ -24,6 +24,8 @@ function ResourceBar({ label, total, available, tone }) {
export default function NodeInspector({ node, open, onClose, onApprove, onDrain, onResume, onBack, backLabel }) {
const [backends, setBackends] = useState(null)
const [backendError, setBackendError] = useState('')
const [models, setModels] = useState(null)
const [modelError, setModelError] = useState('')
const nodeId = node?.id
const backRef = useRef(null)
const closeRef = useRef(null)
@@ -48,6 +50,19 @@ export default function NodeInspector({ node, open, onClose, onApprove, onDrain,
return () => { current = false }
}, [open, nodeId])
useEffect(() => {
if (!open || !nodeId) return undefined
let current = true
setModels(null)
setModelError('')
nodesApi.getModels(nodeId).then(data => {
if (current) setModels(Array.isArray(data) ? data : [])
}).catch(error => {
if (current) setModelError(error.message || 'Unable to load models')
})
return () => { current = false }
}, [open, nodeId])
if (!open || !node) return null
const cpuKnown = node.cpu_logical_cores > 0 && Number.isFinite(node.cpu_usage_percent) && Number.isFinite(node.cpu_load_1)
const disk = capacityReading(node.total_disk, node.available_disk)
@@ -74,6 +89,7 @@ export default function NodeInspector({ node, open, onClose, onApprove, onDrain,
<h3>Node</h3>
<dl className="node-inspector__metrics">
<InspectorMetric label="Address"><span className="node-inspector__address">{node.address || 'No address reported'}</span></InspectorMetric>
<InspectorMetric label="Version">{node.version || '—'}</InspectorMetric>
<InspectorMetric label="Heartbeat">{timeAgo(node.last_heartbeat)}</InspectorMetric>
</dl>
<div className="node-inspector__labels" aria-label="Node labels">{Object.keys(node.labels || {}).length ? Object.entries(node.labels).map(([key, value]) => <span key={key}>{key}={value}</span>) : <span className="text-muted">No labels</span>}</div>
@@ -94,6 +110,26 @@ export default function NodeInspector({ node, open, onClose, onApprove, onDrain,
<InspectorMetric label="Backends">{backendError ? <span className="text-error">{backendError}</span> : backends === null ? 'Loading…' : `${backends.length} backend${backends.length === 1 ? '' : 's'}`}</InspectorMetric>
<InspectorMetric label="In-flight work">{node.in_flight_count ?? 0}</InspectorMetric>
</dl>
<div className="node-inspector__models">
<dt className="drawer-eyebrow">Running models</dt>
<dd>
{modelError ? <span className="text-error">{modelError}</span>
: models === null ? <span className="text-muted">Loading…</span>
: models.length === 0 ? <span className="text-muted">No models loaded</span>
: <ul className="node-inspector__model-list">
{models.map(model => {
const stCfg = modelStateConfig[model.state] || modelStateConfig.idle
return (
<li key={model.id || `${model.model_name}#${model.replica_index}`} className="node-inspector__model-row">
<span className="cell-mono">{model.model_name}</span>
<span className="state-pill" style={{ background: stCfg.bg, color: stCfg.color, border: `1px solid ${stCfg.border}` }}>{model.state}</span>
<span className="text-muted">{model.in_flight ?? 0} in flight</span>
</li>
)
})}
</ul>}
</dd>
</div>
</section>
</div>
<footer className="node-inspector__actions">
@@ -138,6 +138,10 @@ export default function NodeDetail() {
<div className="drawer-eyebrow">In-flight</div>
<span className="cell-mono">{node.in_flight_count || 0}</span>
</div>
<div>
<div className="drawer-eyebrow">Version</div>
<span className="cell-mono">{node.version || '—'}</span>
</div>
<div>
<div className="drawer-eyebrow">Heartbeat</div>
<span>{timeAgo(node.last_heartbeat)}</span>
+4 -4
View File
@@ -93,7 +93,7 @@ func (s *ConfigService) GetConfig(_ context.Context, name string) (*ConfigView,
if configPath == "" {
return nil, ErrConfigFileMissing
}
if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil {
if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil {
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
}
data, err := os.ReadFile(configPath)
@@ -137,7 +137,7 @@ func (s *ConfigService) patchConfig(ctx context.Context, name string, patch map[
return nil, fmt.Errorf("%w: PATCH cannot rename model %q to %q; use the model edit endpoint", ErrInvalidConfig, name, patchedName)
}
configPath := cfg.GetModelConfigFile()
if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil {
if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil {
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
}
diskYAML, err := os.ReadFile(configPath)
@@ -289,7 +289,7 @@ func (s *ConfigService) editYAML(ctx context.Context, name string, body []byte)
configPath := existing.GetModelConfigFile()
modelsPath := s.modelsPath()
if err := utils.VerifyPath(configPath, modelsPath); err != nil {
if err := utils.VerifyResolvedPath(configPath, modelsPath); err != nil {
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
}
@@ -304,7 +304,7 @@ func (s *ConfigService) editYAML(ctx context.Context, name string, body []byte)
}
newConfigPath := filepath.Join(modelsPath, req.Name+".yaml")
paths = append(paths, newConfigPath, filepath.Join(modelsPath, gallery.GalleryFileName(name)), filepath.Join(modelsPath, gallery.GalleryFileName(req.Name)))
if err := utils.VerifyPath(newConfigPath, modelsPath); err != nil {
if err := utils.VerifyPath(req.Name+".yaml", modelsPath); err != nil {
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
}
if _, err := os.Stat(newConfigPath); err == nil {
@@ -0,0 +1,58 @@
package modeladmin
import (
"context"
"os"
"path/filepath"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
// A model config can be loaded from outside the models directory (for
// example with --config-file). The admin mutations write the config file
// back, so they must refuse a file outside the models directory rather than
// write wherever the loader found it.
var _ = Describe("ConfigService config file containment", func() {
var (
svc *ConfigService
ctx context.Context
outside string
orig []byte
)
BeforeEach(func() {
svc, _ = newTestService()
ctx = context.Background()
outside = filepath.Join(GinkgoT().TempDir(), "external.yaml")
orig = []byte("name: external\nbackend: llama-cpp\n")
Expect(os.WriteFile(outside, orig, 0o644)).To(Succeed())
Expect(svc.Loader.ReadModelConfig(outside, svc.AppConfig.ToConfigLoaderOptions()...)).To(Succeed())
cfg, ok := svc.Loader.GetModelConfig("external")
Expect(ok).To(BeTrue())
Expect(cfg.GetModelConfigFile()).To(Equal(outside))
})
It("refuses to pin a model whose config file is outside the models directory", func() {
_, err := svc.TogglePinned(ctx, "external", ActionPin, nil)
Expect(err).To(MatchError(ErrPathNotTrusted))
Expect(os.ReadFile(outside)).To(Equal(orig))
})
It("refuses to toggle the state of such a model", func() {
_, err := svc.ToggleState(ctx, "external", ActionDisable)
Expect(err).To(MatchError(ErrPathNotTrusted))
Expect(os.ReadFile(outside)).To(Equal(orig))
})
It("refuses to patch such a model", func() {
_, err := svc.PatchConfig(ctx, "external", map[string]any{"context_size": 4096})
Expect(err).To(MatchError(ErrPathNotTrusted))
Expect(os.ReadFile(outside)).To(Equal(orig))
})
It("refuses to read such a model's config", func() {
_, err := svc.GetConfig(ctx, "external")
Expect(err).To(MatchError(ErrPathNotTrusted))
})
})
+1 -1
View File
@@ -29,7 +29,7 @@ func (s *ConfigService) TogglePinned(_ context.Context, name string, action Acti
if configPath == "" {
return nil, ErrConfigFileMissing
}
if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil {
if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil {
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
}
if err := mutateYAMLBoolFlag(configPath, "pinned", action == ActionPin); err != nil {
+1 -1
View File
@@ -49,7 +49,7 @@ func (s *ConfigService) toggleState(ctx context.Context, name string, action Act
if configPath == "" {
return nil, ErrConfigFileMissing
}
if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil {
if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil {
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
}
var result *ToggleResult
+78
View File
@@ -0,0 +1,78 @@
package nodes
import (
"os"
"path/filepath"
"strings"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
"github.com/mudler/xlog"
)
// declaredExtraFiles returns the files the model's install declared that the
// path fields of opts do not already stage: neither named by a field nor
// inside a directory a field names. It must run on the local paths, before
// staging rewrites the fields to remote ones.
func (r *SmartRouter) declaredExtraFiles(trackingKey string, opts *pb.ModelOptions) []string {
if r.modelFiles == nil || opts == nil || trackingKey == "" {
return nil
}
covered := append([]string{
opts.ModelFile, opts.MMProj, opts.LoraAdapter, opts.DraftModel,
opts.CLIPModel, opts.Tokenizer, opts.AudioPath, opts.LoraBase,
}, opts.LoraAdapters...)
seen := map[string]struct{}{}
var extra []string
for _, p := range r.modelFiles(trackingKey) {
p = filepath.Clean(p)
if _, dup := seen[p]; dup || coveredByField(p, covered) {
continue
}
seen[p] = struct{}{}
extra = append(extra, p)
}
return extra
}
func coveredByField(path string, fields []string) bool {
for _, f := range fields {
if f == "" {
continue
}
f = filepath.Clean(f)
if path == f || strings.HasPrefix(path, f+string(filepath.Separator)) {
return true
}
}
return false
}
// existingFiles drops declared files that are not on the frontend. An install
// can declare files that are gone by load time (an archive unpacked and then
// removed, say), so a missing one is not a reason to refuse the load; the
// backend reports it if it really needed it.
func existingFiles(paths []string, nodeName, trackingKey string) []string {
out := paths[:0:0]
for _, p := range paths {
if _, err := os.Stat(p); err != nil {
xlog.Warn("Skipping staging for declared model file that is not on the frontend", "path", p, "node", nodeName, "model", trackingKey, "error", err)
continue
}
out = append(out, p)
}
return out
}
// stagingPayloadBytes totals the on-disk size of everything staging uploads
// for a model: the path fields plus the declared files they do not cover. The
// first shard of a split GGUF can be a few MB of metadata while the weights
// sit in the others, so sizing the fields alone starves the load budget and
// the disk-headroom check.
func (r *SmartRouter) stagingPayloadBytes(trackingKey string, opts *pb.ModelOptions) int64 {
total := modelPayloadBytes(opts)
for _, p := range r.declaredExtraFiles(trackingKey, opts) {
total += pathBytes(p)
}
return total
}
+5
View File
@@ -124,6 +124,11 @@ type BackendNode struct {
// worker's re-registration value does not clobber it (mirrors
// MaxReplicasPerModelManuallySet).
VRAMBudgetManuallySet bool `gorm:"column:vram_budget_manually_set;default:false" json:"vram_budget_manually_set"`
// Version is the LocalAI build version reported by the worker at
// registration. Empty for workers registered before this field existed.
Version string `gorm:"column:version;size:64" json:"version,omitempty"`
// Commit is the git commit hash the worker binary was built from.
Commit string `gorm:"column:commit;size:64" json:"commit,omitempty"`
APIKeyID string `gorm:"size:36" json:"-"` // auto-provisioned API key ID (for cleanup)
AuthUserID string `gorm:"size:36" json:"-"` // auto-provisioned user ID (for cleanup)
LastHeartbeat time.Time `gorm:"column:last_heartbeat" json:"last_heartbeat"`
+32 -2
View File
@@ -73,6 +73,12 @@ type SmartRouterOptions struct {
// nil disables the exclusion. Deliberate teardown (UnloadModel, admin
// endpoints, node drain) is unaffected.
PinnedResolver PinnedModelResolver
// ModelFiles, when set, returns the absolute local paths of every file a
// model's install declared (gallery `files:`, config `download_files`).
// The path fields of a load request name only what the backend opens
// first; this is how staging learns about the rest, such as the other
// shards of a split GGUF. nil stages the path fields alone.
ModelFiles func(modelName string) []string
// PrefixProvider, when set, enables prefix-cache-aware routing: requests
// carrying a prompt prefix chain (distributedhdr.PrefixChain) are biased
// toward the node that already holds the longest matching prefix, subject
@@ -189,6 +195,9 @@ type SmartRouter struct {
// pinnedResolver feeds the eviction paths the set of pinned model names
// (see SmartRouterOptions.PinnedResolver). nil disables the exclusion.
pinnedResolver PinnedModelResolver
// modelFiles resolves a model's declared files (see
// SmartRouterOptions.ModelFiles). nil stages the path fields alone.
modelFiles func(modelName string) []string
// prefixProvider is the prefix-cache routing seam (nil disables it; see
// SmartRouterOptions.PrefixProvider). prefixConfig holds the global policy
// and thresholds.
@@ -283,6 +292,7 @@ func NewSmartRouter(registry ModelRouter, opts SmartRouterOptions) *SmartRouter
stagingTracker: NewStagingTracker(),
conflictResolver: opts.ConflictResolver,
pinnedResolver: opts.PinnedResolver,
modelFiles: opts.ModelFiles,
probeCache: newProbeCache(probeCacheTTL),
prefixProvider: opts.PrefixProvider,
prefixConfig: opts.PrefixConfig,
@@ -425,7 +435,7 @@ func (r *SmartRouter) scheduleAndLoad(ctx context.Context, backendType, tracking
// Size the remote load budget BEFORE staging: stageModelFiles rewrites the
// path fields to their remote equivalents on a clone, and only the local
// paths can be stat'ed here.
payloadBytes := modelPayloadBytes(modelOpts)
payloadBytes := r.stagingPayloadBytes(trackingKey, modelOpts)
loadTimeout := r.loadTimeoutFor(payloadBytes)
// Pre-stage model files via FileStager before loading
@@ -1367,7 +1377,7 @@ func (r *SmartRouter) narrowByDiskHeadroom(ctx context.Context, modelID string,
return candidateNodeIDs, nil
}
requiredDisk := DiskRequirementFor(modelPayloadBytes(modelOpts))
requiredDisk := DiskRequirementFor(r.stagingPayloadBytes(modelID, modelOpts))
diskCandidates, diskErr := r.registry.NarrowByDiskHeadroom(ctx, candidateNodeIDs, requiredDisk)
// The check runs even when disabled. "Disabled" means do not BLOCK, not do
@@ -1568,6 +1578,10 @@ func (r *SmartRouter) stageModelFiles(ctx context.Context, node *BackendNode, op
localModelDir = filepath.Dir(opts.ModelFile)
}
// Resolved before the path fields are rewritten to remote paths below,
// since that is what tells which declared files the fields already cover.
declared := existingFiles(r.declaredExtraFiles(trackingKey, opts), node.Name, trackingKey)
// keyMapper generates storage keys namespaced under trackingKey, preserving
// subdirectory structure relative to frontendModelsDir. This ensures:
// 1. All files for a model land in one directory on the worker for clean deletion
@@ -1614,6 +1628,7 @@ func (r *SmartRouter) stageModelFiles(ctx context.Context, node *BackendNode, op
totalFiles++
}
}
totalFiles += len(declared)
// Start tracking staging progress
r.stagingTracker.Start(trackingKey, node.Name, totalFiles)
@@ -1757,6 +1772,21 @@ func (r *SmartRouter) stageModelFiles(ctx context.Context, node *BackendNode, op
}
}
for _, localPath := range declared {
fileIdx++
fileName := filepath.Base(localPath)
stageCtx := r.withStagingCallback(ctx, trackingKey, fileName, fileIdx, totalFiles)
xlog.Info("Staging declared model file", "model", trackingKey, "node", node.Name, "file", fileName, "fileIndex", fileIdx, "totalFiles", totalFiles)
if _, err := r.fileStager.EnsureRemote(stageCtx, node.ID, localPath, keyMapper.Key(localPath)); err != nil {
// The install declared it, so the backend may read it: loading
// without it fails later with a less useful error.
xlog.Error("Failed to stage declared model file for remote node", "node", node.Name, "path", localPath, "error", err)
return nil, fmt.Errorf("staging declared model file %s: %w", localPath, err)
}
r.stagingTracker.FileComplete(trackingKey, fileIdx, totalFiles)
}
// Stage file paths referenced in generic Options (key:value pairs where values
// are file paths). Options stay as relative paths — backends resolve them via ModelPath.
for _, options := range [][]string{opts.Options, opts.Overrides} {
@@ -0,0 +1,126 @@
package nodes
import (
"context"
"os"
"path/filepath"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
)
// A model's config names only the file the backend opens first, but its
// install can declare more that the backend reads by itself: llama.cpp opens
// the "-0000N-of-0000M" shards of a split GGUF from the directory of the first
// one. The worker has no view of the frontend's models directory, so every
// declared file must be staged, or the load fails with "failed to load GGUF
// split".
var _ = Describe("stageModelFiles declared model files", func() {
var (
stager *fakeFileStager
router *SmartRouter
node *BackendNode
modelDir string
shards []string
mmproj string
declared map[string][]string
)
BeforeEach(func() {
stager = &fakeFileStager{}
declared = map[string][]string{}
router = &SmartRouter{
fileStager: stager,
stagingTracker: NewStagingTracker(),
modelFiles: func(name string) []string { return declared[name] },
}
node = &BackendNode{ID: "node-1", Name: "node-1", Address: "10.0.0.1:50051"}
root := GinkgoT().TempDir()
modelDir = filepath.Join(root, "llama-cpp", "models", "big")
Expect(os.MkdirAll(modelDir, 0o755)).To(Succeed())
shards = nil
for _, name := range []string{
"Big-Q4_K_M-00001-of-00003.gguf",
"Big-Q4_K_M-00002-of-00003.gguf",
"Big-Q4_K_M-00003-of-00003.gguf",
} {
p := filepath.Join(modelDir, name)
Expect(os.WriteFile(p, []byte("shard "+name), 0o644)).To(Succeed())
shards = append(shards, p)
}
mmproj = filepath.Join(root, "llama-cpp", "mmproj", "big", "mmproj.gguf")
Expect(os.MkdirAll(filepath.Dir(mmproj), 0o755)).To(Succeed())
Expect(os.WriteFile(mmproj, []byte("mmproj"), 0o644)).To(Succeed())
})
opts := func() *pb.ModelOptions {
return &pb.ModelOptions{
Model: "llama-cpp/models/big/Big-Q4_K_M-00001-of-00003.gguf",
ModelFile: shards[0],
MMProj: mmproj,
}
}
stagedPaths := func() []string {
out := make([]string, 0, len(stager.ensureCalls))
for _, c := range stager.ensureCalls {
out = append(out, c.localPath)
}
return out
}
It("stages every declared file once, beside the ones the config names", func() {
declared["big"] = append(append([]string{}, shards...), mmproj)
staged, err := router.stageModelFiles(context.Background(), node, opts(), "big")
Expect(err).ToNot(HaveOccurred())
Expect(stagedPaths()).To(ConsistOf(shards[0], mmproj, shards[1], shards[2]))
// llama.cpp derives the other shards' paths from the first one, so
// they must land in the same remote directory.
for _, c := range stager.ensureCalls {
if c.localPath != mmproj {
Expect(filepath.Dir(c.key)).To(Equal(filepath.Dir(stager.ensureCalls[0].key)))
}
}
Expect(staged.ModelFile).To(Equal("/remote/" + stager.ensureCalls[0].key))
})
It("sizes declared files for the load budget and disk check", func() {
declared["big"] = append(append([]string{}, shards...), mmproj)
var want int64
for _, p := range append(append([]string{}, shards...), mmproj) {
fi, err := os.Stat(p)
Expect(err).ToNot(HaveOccurred())
want += fi.Size()
}
Expect(router.stagingPayloadBytes("big", opts())).To(Equal(want))
})
It("skips a declared file that is missing locally instead of failing", func() {
declared["big"] = append(append([]string{}, shards...), filepath.Join(modelDir, "gone.bin"))
_, err := router.stageModelFiles(context.Background(), node, opts(), "big")
Expect(err).ToNot(HaveOccurred())
Expect(stagedPaths()).To(ConsistOf(shards[0], mmproj, shards[1], shards[2]))
})
It("does not stage a declared file twice when a directory field covers it", func() {
declared["dir"] = []string{shards[1]}
_, err := router.stageModelFiles(context.Background(), node,
&pb.ModelOptions{Model: "llama-cpp/models/big", ModelFile: modelDir}, "dir")
Expect(err).ToNot(HaveOccurred())
Expect(stagedPaths()).To(ConsistOf(shards[0], shards[1], shards[2]))
})
It("stages only the named files for a model that declares none", func() {
_, err := router.stageModelFiles(context.Background(), node, opts(), "handwritten")
Expect(err).ToNot(HaveOccurred())
Expect(stagedPaths()).To(ConsistOf(shards[0], mmproj))
})
})
+3
View File
@@ -8,6 +8,7 @@ import (
"strconv"
"strings"
"github.com/mudler/LocalAI/internal"
"github.com/mudler/LocalAI/pkg/system"
"github.com/mudler/LocalAI/pkg/xsysinfo"
"github.com/mudler/xlog"
@@ -154,6 +155,8 @@ func (cfg *Config) registrationBody() map[string]any {
"gpu_compute_capability": gpuComputeCap,
"capability": capability,
"max_replicas_per_model": maxReplicas,
"version": internal.Version,
"commit": internal.Commit,
}
// Report free space on the filesystem that backs the MODELS directory.
+2 -2
View File
@@ -41,7 +41,7 @@ func InstallModelsWithOptions(ctx context.Context, galleryService *galleryop.Gal
// Check if it's a model gallery, or print a warning
e, found := installModel(ctx, galleries, backendGalleries, url, systemState, modelLoader, downloadStatus, enforceScan, autoloadBackendGalleries, requireBackendIntegrity, installOptions...)
if e != nil && found {
xlog.Error("[startup] failed installing model", "error", err, "model", url)
xlog.Error("[startup] failed installing model", "error", e, "model", url)
err = errors.Join(err, e)
} else if !found {
xlog.Debug("[startup] model not found in the gallery", "model", url)
@@ -54,7 +54,7 @@ func InstallModelsWithOptions(ctx context.Context, galleryService *galleryop.Gal
modelConfig, discoverErr := importers.DiscoverModelConfig(url, json.RawMessage{})
if discoverErr != nil {
xlog.Error("[startup] failed to discover model config", "error", discoverErr, "model", url)
err = errors.Join(discoverErr, fmt.Errorf("failed to discover model config: %w", err))
err = errors.Join(err, fmt.Errorf("failed to discover model config: %w", discoverErr))
continue
}
+1 -1
View File
@@ -41,7 +41,7 @@ services:
# Here we can specify a list of models to run (see quickstart https://localai.io/basics/getting_started/#running-models )
# or an URL pointing to a YAML configuration file, for example:
# - https://gist.githubusercontent.com/mudler/ad601a0488b497b69ec549150d9edd18/raw/a8a8869ef1bb7e3830bf5c0bae29a0cce991ff8d/phi-2.yaml
- phi-2
- phi-2-chat
# For NVIDIA GPU support with CDI (recommended for NVIDIA Container Toolkit 1.14+):
# Uncomment the following deploy section and use driver: nvidia.com/gpu.
# Include `utility` in capabilities so nvidia-smi / NVML are available —
@@ -74,6 +74,8 @@ When using `--models-config-file`, you can define multiple models as a list:
backend: llama-cpp
```
LocalAI changes only config files that are inside the models directory. If the file from `--models-config-file` is outside the models directory, you cannot view, edit, pin, enable or disable its models from the web UI or the model admin API. Edit the file directly, then restart LocalAI.
## Core Configuration Fields
### Basic Model Settings
+1 -1
View File
@@ -243,7 +243,7 @@ The devices in the following list have been tested with `hipblas` images.
1. Check your GPU LLVM target is compatible with the version of ROCm. This can be found in the [LLVM Docs](https://llvm.org/docs/AMDGPUUsage.html).
2. Check which ROCm version is compatible with your LLVM target and your chosen OS (pay special attention to supported kernel versions). See the [ROCm compatibility matrix](https://rocm.docs.amd.com/en/latest/compatibility/compatibility-matrix.html).
3. Install your chosen version of the `dkms` and `rocm` (it is recommended that the native package manager be used for this process for any OS as version changes are executed more easily via this method if updates are required). Take care to restart after installing `amdgpu-dkms` and before installing `rocm`, for details regarding this see the [ROCm installation documentation](https://rocm.docs.amd.com/projects/install-on-linux/en/latest/how-to/native-install/index.html).
3. Install your chosen version of the `dkms` and `rocm` (it is recommended that the native package manager be used for this process for any OS as version changes are executed more easily via this method if updates are required). Take care to restart after installing `amdgpu-dkms` and before installing `rocm`, for details regarding this see the [ROCm installation documentation](https://rocm.docs.amd.com/projects/install-on-linux/en/latest/install/install-methods/package-manager-index.html).
4. Deploy. Yes it's that easy.
#### Setup Example (Docker/containerd)
+13
View File
@@ -740,6 +740,19 @@ Set `LOCALAI_DISTRIBUTED_SHARED_MODELS=true` (or `--distributed-shared-models`)
This flag is a contract you assert: all nodes must mount identical paths. Leave it off (the default) when workers have independent models directories - the frontend stages files to them over HTTP (or S3) as described above.
### Which files are staged
The frontend stages the files that the model config names (`parameters.model`, `mmproj`, draft model, LoRA adapters and similar fields). It also stages every other file that the model declares:
- The `files:` of the gallery entry or `/import-model` import that installed the model. LocalAI records these in `._gallery_<name>.yaml` next to the model config.
- The `download_files:` of the model config.
A backend can read files that the config does not name. For example, llama.cpp opens all shards of a split GGUF (`<name>-00002-of-00004.gguf` and the rest) from the directory of the first shard. The worker cannot see the frontend's models directory, so it gets only the files that the frontend stages.
If you write a model config by hand and the model has files like these, list them under `download_files:`. If you do not, the worker gets only the first shard and the load fails with `failed to load GGUF split`.
The file sizes used for the load deadline and for the disk headroom check include all of these files.
### Model artifact staging
For managed Hugging Face artifacts, the controller resolves the repository and
@@ -98,7 +98,7 @@ LLAMACPP_GRPC_SERVERS="address1:port,address2:port" local-ai run
```
The workload on the LocalAI server will then be distributed across the specified nodes.
Alternatively, you can build the RPC workers/server following the llama.cpp [README](https://github.com/ggerganov/llama.cpp/blob/master/examples/rpc/README.md), which is compatible with LocalAI.
Alternatively, you can build the RPC workers/server following the llama.cpp [README](https://github.com/ggml-org/llama.cpp/blob/master/tools/rpc/README.md), which is compatible with LocalAI.
## Manual example (worker)
+1 -1
View File
@@ -373,7 +373,7 @@ curl $LOCALAI/models/apply -H "Content-Type: application/json" -d '{
where:
- `localai` is the repository. It is optional and can be omitted. If the repository is omitted LocalAI will search the model by name in all the repositories. In the case the same model name is present in both galleries the first match wins.
- `bert-embeddings` is the model name in the gallery
(read its [config here](https://github.com/mudler/LocalAI/tree/master/gallery/blob/main/bert-embeddings.yaml)).
(read its [config here](https://github.com/mudler/LocalAI/blob/master/gallery/index.yaml)).
### Model variants
+39 -2
View File
@@ -587,7 +587,7 @@ The `llama.cpp` backend supports additional configuration options that can be sp
|--------|------|-------------|---------|
| `use_jinja` or `jinja` | boolean | Enable Jinja2 template processing for chat templates. When enabled, the backend uses Jinja2-based chat templates from the model for formatting messages. | `use_jinja:true` |
| `context_shift` | boolean | Enable context shifting, which allows the model to dynamically adjust context window usage. | `context_shift:true` |
| `cache_ram` | integer | Size budget in MiB for the **server-side prompt cache** (a host-RAM store of idle slot KV states that's reloaded on a prompt-prefix hit, see [upstream PR #16391](https://github.com/ggml-org/llama.cpp/pull/16391)). Default: `-1` (no limit). `0` disables the prompt cache entirely. Together with `kv_unified` and `cache_idle_slots` this is what makes a repeated system prompt skip prefill on subsequent calls. | `cache_ram:4096` |
| `cache_ram` | integer | Size budget in MiB for the **server-side prompt cache** (a host-RAM store of idle slot KV states that's reloaded on a prompt-prefix hit, see [upstream PR #16391](https://github.com/ggml-org/llama.cpp/pull/16391)). Default: `8192` MiB (llama.cpp default). `-1` removes the limit. `0` disables the prompt cache entirely. Together with `kv_unified` and `cache_idle_slots` this is what makes a repeated system prompt skip prefill on subsequent calls. | `cache_ram:4096` |
| `parallel` or `n_parallel` | integer | Enable parallel request processing. When set to a value greater than 1, enables continuous batching for handling multiple requests concurrently. | `parallel:4` |
| `grpc_servers` or `rpc_servers` | string | Comma-separated list of gRPC server addresses for distributed inference. Allows distributing workload across multiple llama.cpp workers. | `grpc_servers:localhost:50051,localhost:50052` |
| `fit_params` or `fit` | boolean | Enable auto-adjustment of model/context parameters to fit available device memory. Default: `true`. | `fit_params:true` |
@@ -643,7 +643,7 @@ Agents, coding assistants, and Anthropic/OpenAI-compatible CLIs typically resend
| Setting | Default | Role |
|---|---|---|
| `cache_ram:N` | `-1` (no limit) | Allocates the host-side prompt cache. `0` disables it. |
| `cache_ram:N` | `8192` (llama.cpp default) | Allocates the host-side prompt cache. `0` disables it. |
| `kv_unified:true` | `true` | Single unified KV buffer (**prerequisite** for idle-slot saving). |
| `cache_idle_slots:true` | `true` | Persists the idle slot's KV into the prompt cache on task switch. |
@@ -658,6 +658,8 @@ options:
Set `cache_ram:0` to opt out of the prompt cache entirely (saves host RAM at the cost of re-prefilling repeated prompts).
`cache_ram:-1` removes the limit. With idle-slot saving on, every distinct prompt then leaves its slot state in host RAM, so a workload with many different prompts (classification, ingestion) grows the backend by roughly the KV size of each prompt until the host runs out of memory.
#### Reference
- [llama](https://github.com/ggerganov/llama.cpp)
@@ -988,6 +990,41 @@ options:
The full list of registered parsers lives in `sglang.srt.function_call`
and `sglang.srt.parser.reasoning_parser`.
#### Reasoning defaults and token budgets
Set SGLang reasoning options in the model's `options:` list:
```yaml
options:
- reasoning_parser:qwen3
- thinking_budget:512
- reasoning_default:on
engine_args:
enable_strict_thinking: true
```
`thinking_budget` sets a positive integer token budget for reasoning on each request.
Invalid, zero, and negative values produce a warning and leave the budget unset.
SGLang requires `engine_args.enable_strict_thinking: true` to enforce the budget.
LocalAI warns if you configure a budget without that engine option.
Keep the budget well below the `max_tokens` of your requests: if `max_tokens` is reached first,
the budget never triggers and the whole reply can be spent on reasoning, leaving the answer empty.
`reasoning_default:on` or `reasoning_default:off` sets the default for LocalAI's tokenizer chat template.
Request metadata `enable_thinking` set to `"true"` or `"false"` overrides this default.
An explicit prompt bypasses tokenizer template rendering.
When no default or request override is set, the template keeps its own behavior.
LocalAI signals required reasoning when the rendered prompt ends with the configured parser's opening reasoning token.
An explicit output grammar disables this detection.
Configure a reasoning parser that matches your model.
The backend reads these options when it loads the model.
`POST /models/reload` rereads model configuration files but does not update options in an already loaded backend.
Restarting only the backend does not reread configuration files.
Restart LocalAI after changing these options to reload both the configuration and the backend.
### vllm.cpp
[vllm.cpp](https://github.com/mudler/vllm.cpp) is the LocalAI team's C++ port of
@@ -108,6 +108,8 @@ docker run -ti --name local-ai -p 8080:8080 --runtime nvidia --gpus all localai/
## Using Compose
The repository's `docker-compose.yaml` installs `phi-2-chat` from the model gallery by default. Change its `command` list to select a different gallery model.
For a more manageable setup, especially with persistent volumes, use Docker Compose or Podman Compose:
### Using CDI (Container Device Interface) - Recommended for NVIDIA Container Toolkit 1.14+
@@ -25,17 +25,17 @@ Here's an example to initiate the **phi-2** model:
docker run -p 8080:8080 localai/localai:{{< version >}} https://gist.githubusercontent.com/mudler/ad601a0488b497b69ec549150d9edd18/raw/a8a8869ef1bb7e3830bf5c0bae29a0cce991ff8d/phi-2.yaml
```
You can also check all the embedded models configurations [here](https://github.com/mudler/LocalAI/tree/master/embedded/models).
You can also check all the embedded models configurations [here](https://github.com/mudler/LocalAI/tree/master/gallery).
{{% notice tip %}}
The model configurations used in the quickstart are accessible here: [https://github.com/mudler/LocalAI/tree/master/embedded/models](https://github.com/mudler/LocalAI/tree/master/embedded/models). Contributions are welcome; please feel free to submit a Pull Request.
The model configurations used in the quickstart are accessible here: [https://github.com/mudler/LocalAI/tree/master/gallery](https://github.com/mudler/LocalAI/tree/master/gallery). Contributions are welcome; please feel free to submit a Pull Request.
The `phi-2` model configuration from the quickstart is expanded from [https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml](https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml).
The `phi-2` model configuration from the quickstart is expanded from [https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml](https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml).
{{% /notice %}}
## Example: Customizing the Prompt Template
To modify the prompt template, create a Github gist or a Pastebin file, and copy the content from [https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml](https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml). Alter the fields as needed:
To modify the prompt template, create a Github gist or a Pastebin file, and copy the content from [https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml](https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml). Alter the fields as needed:
```yaml
name: phi-2
+1 -1
View File
@@ -98,7 +98,7 @@ availability may lag upstream releases.
- [AnythingLLM](https://github.com/Mintplex-Labs/anything-llm)
- [Logseq GPT3 OpenAI plugin](https://github.com/briansunter/logseq-plugin-gpt3-openai)
- [CodeGPT (JetBrains)](https://plugins.jetbrains.com/plugin/21056-codegpt) - Custom OpenAI-compatible endpoints
- [Wave Terminal](https://docs.waveterm.dev/features/supportedLLMs/localai) - Native LocalAI support
- [Wave Terminal](https://docs.waveterm.dev/ai-presets) - Native LocalAI support
- [Obsidian BMO Chatbot](https://github.com/longy2k/obsidian-bmo-chatbot)
- [spark](https://github.com/cedriking/spark)
- [openops (Mattermost)](https://github.com/mattermost/openops)
@@ -45,7 +45,7 @@ All backends listed here can be installed on demand from the [Backend Gallery]({
| [moonshine](https://github.com/moonshine-ai/moonshine) | Ultra-fast transcription for low-end devices (ONNX) | CPU, CUDA 12/13, Metal |
| [parakeet.cpp](https://github.com/mudler/parakeet.cpp) | C++/GGML port of NVIDIA NeMo Parakeet (tdt/ctc/rnnt/hybrid), with cache-aware streaming | CPU, CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, Jetson L4T |
| [CrispASR](https://github.com/CrispStrobe/CrispASR) | Unified speech engine (whisper.cpp fork) supporting Parakeet, Canary, and many ASR architectures, plus TTS | CPU, CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, Jetson L4T |
| [voxtral](https://github.com/mudler/voxtral.c) | Voxtral Realtime 4B speech-to-text in pure C | CPU, Metal |
| [voxtral](https://github.com/antirez/voxtral.c) | Voxtral Realtime 4B speech-to-text in pure C | CPU, Metal |
| [Qwen3-ASR](https://github.com/QwenLM/Qwen3-ASR) | Qwen3 automatic speech recognition | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal, Jetson L4T |
| [NeMo](https://github.com/NVIDIA/NeMo) | NVIDIA NeMo ASR toolkit | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal |
| [sherpa-onnx](https://k2-fsa.github.io/sherpa/onnx/) | Sherpa-ONNX ASR (Whisper, Paraformer, SenseVoice) and TTS | CPU, CUDA 12, Metal |
@@ -70,10 +70,10 @@ All backends listed here can be installed on demand from the [Backend Gallery]({
| [OmniVoice](https://github.com/ServeurpersoCom/omnivoice.cpp) | Native C++/GGML TTS with voice cloning, voice design, and streaming | CPU, CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, Jetson L4T |
| [fish-speech](https://github.com/fishaudio/fish-speech) | High-quality TTS with voice cloning | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal, Jetson L4T |
| [Pocket TTS](https://github.com/kyutai-labs/pocket-tts) | Lightweight CPU-efficient TTS with voice cloning | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal, Jetson L4T |
| [OuteTTS](https://github.com/OuteAI/outetts) | TTS with custom speaker voices | CPU, CUDA 12 |
| [OuteTTS](https://github.com/edwko/OuteTTS) | TTS with custom speaker voices | CPU, CUDA 12 |
| [faster-qwen3-tts](https://github.com/andimarafioti/faster-qwen3-tts) | Real-time Qwen3-TTS with CUDA graph capture | CPU, CUDA 12/13, Jetson L4T |
| [NeuTTS Air](https://github.com/neuphonic/neutts-air) | Instant voice cloning, on-device TTS | CPU, CUDA 12, ROCm |
| [VoxCPM](https://github.com/ModelBest/VoxCPM) | Expressive end-to-end TTS | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal |
| [VoxCPM](https://github.com/OpenBMB/VoxCPM) | Expressive end-to-end TTS | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal |
| [Kitten TTS](https://github.com/KittenML/KittenTTS) | Kitten TTS model | CPU, Metal |
| [Supertonic](https://github.com/supertone-inc/supertonic) | Lightning-fast on-device multilingual TTS via ONNX | CPU |
| [MLX-Audio](https://github.com/Blaizzy/mlx-audio) | Audio models on Apple Silicon | CPU, CUDA 12/13, Metal, Jetson L4T |
+4 -7
View File
@@ -297,7 +297,7 @@
files:
- filename: ds4flash.gguf
uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF
sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf
sha256: f33633d55f5379e8571db06674bf7a07a2ea7bb7b44287e1a9bdc3686d69a5c6
- name: "qwopus3.8-27b-flash-v2"
variants:
- model: qwopus3.8-27b-flash-v2-q8
@@ -6420,7 +6420,6 @@
- filename: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
sha256: 8e5601dbd18fbc2b731cf674a040dd32f3ec2d09a312f4e0f3c4d7bc92998837
- name: sharp-spark-x2.5-4b-q5
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
@@ -6457,7 +6456,6 @@
- filename: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
sha256: f445f1a57e58b70ea85078e1edcd29763843f71f154bac2efc57eea1b8333a26
- name: sharp-spark-x2.5-4b-q6
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
@@ -6494,7 +6492,6 @@
- filename: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
sha256: 793e673f34d2dde9674d24d277c25dbf03b89290333835aa31b7ee1d62e20dfc
- &spark-x2-5-4b
name: "spark-x2.5-4b-q4"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
@@ -9309,7 +9306,7 @@
files:
- filename: llama-cpp/models/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-APEX/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-APEX.gguf
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-APEX.gguf
sha256: 35026b978de6ee6ff63870d3d68be90ab0797a9a6cd5c83332e1f9c510c6a695
sha256: 97602082e8639b1e6660b36de0193861742e035711b865faf76abf54a5a2be09
- filename: llama-cpp/mmproj/Hermes3.6-35B-A3B-Uncensored-Genesis-Final/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
sha256: 5129bb5eb19e4346c0f2071f1ce8e1b0a076e0ab08d57a77d6033ce01235252c
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
@@ -9399,7 +9396,7 @@
files:
- filename: llama-cpp/models/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX.gguf
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX.gguf
sha256: 612561952f0698539a133a479e1cf18ffce85bb6b4311a5d05e50d6160970c75
sha256: 3efbc83f38ffa48251ff4c072fbbefce63c995df3cba20fd6bf0d8d7ab965cd9
- filename: llama-cpp/mmproj/Hermes3.6-35B-A3B-Uncensored-Genesis-Final/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
sha256: 5129bb5eb19e4346c0f2071f1ce8e1b0a076e0ab08d57a77d6033ce01235252c
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
@@ -9451,7 +9448,7 @@
files:
- filename: llama-cpp/models/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX-Compact/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX-Compact.gguf
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX-Compact.gguf
sha256: 7a17aaff5ec81ba34e33d3a8b032a699abf4231d6cbe9930d1c9a159db3e87dc
sha256: 27a2edec6f66e585fbf02eabe397f694378474786711362d9f78cd129a0e8313
- filename: llama-cpp/mmproj/Hermes3.6-35B-A3B-Uncensored-Genesis-Final/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
sha256: 5129bb5eb19e4346c0f2071f1ce8e1b0a076e0ab08d57a77d6033ce01235252c
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
+18 -4
View File
@@ -13,21 +13,35 @@ func ExistsInPath(path string, s string) bool {
}
func InTrustedRoot(path string, trustedRoot string) error {
for path != "/" {
path = filepath.Dir(path)
for {
parent := filepath.Dir(path)
// Dir stops changing at "/" for an absolute path and at "." for a
// relative one; waiting for "/" alone spins forever on the latter.
if parent == path {
return fmt.Errorf("path is outside of trusted root")
}
path = parent
if path == trustedRoot {
return nil
}
}
return fmt.Errorf("path is outside of trusted root")
}
// VerifyPath verifies that path is based in basePath.
// VerifyPath verifies that path, taken relative to basePath, is based in
// basePath. It joins path onto basePath first, so an absolute path is read as
// relative to the base as well: give it the untrusted relative name, never a
// path that has already been joined. For a full path use VerifyResolvedPath.
func VerifyPath(path, basePath string) error {
c := filepath.Clean(filepath.Join(basePath, path))
return InTrustedRoot(c, filepath.Clean(basePath))
}
// VerifyResolvedPath verifies that path, a full path rather than one relative
// to basePath, is based in basePath.
func VerifyResolvedPath(path, basePath string) error {
return InTrustedRoot(filepath.Clean(path), filepath.Clean(basePath))
}
// SanitizeFileName sanitizes the given filename
func SanitizeFileName(fileName string) string {
// filepath.Clean to clean the path
+35
View File
@@ -3,6 +3,7 @@ package utils_test
import (
"os"
"path/filepath"
"time"
. "github.com/mudler/LocalAI/pkg/utils"
. "github.com/onsi/ginkgo/v2"
@@ -71,6 +72,25 @@ var _ = Describe("utils/path tests", func() {
})
})
Describe("VerifyResolvedPath", func() {
It("accepts a full path inside the base", func() {
Expect(VerifyResolvedPath("/srv/models/a/model.yaml", "/srv/models")).To(Succeed())
})
It("rejects a full path outside the base", func() {
// VerifyPath would join this onto the base and accept it.
Expect(VerifyResolvedPath("/etc/passwd", "/srv/models")).ToNot(Succeed())
})
It("rejects a joined path that climbed out of the base", func() {
Expect(VerifyResolvedPath(filepath.Join("/srv/models", "../other/x"), "/srv/models")).ToNot(Succeed())
})
It("cleans both paths before comparing", func() {
Expect(VerifyResolvedPath("/srv/models/./a/../b.yaml", "/srv/models/")).To(Succeed())
})
})
Describe("InTrustedRoot", func() {
It("accepts a strict descendant of the trusted root", func() {
Expect(InTrustedRoot("/srv/models/file", "/srv/models")).To(Succeed())
@@ -93,6 +113,21 @@ var _ = Describe("utils/path tests", func() {
It("rejects an unrelated absolute path", func() {
Expect(InTrustedRoot("/etc/passwd", "/srv/models")).ToNot(Succeed())
})
It("rejects a relative path outside a relative root instead of looping", func() {
// Walking up a relative path ends at ".", never at "/", so the
// walk must stop when it stops making progress.
done := make(chan error, 1)
go func() { done <- InTrustedRoot("x", "models") }()
Eventually(done).WithTimeout(2 * time.Second).Should(Receive(HaveOccurred()))
go func() { done <- VerifyPath("../x", "models") }()
Eventually(done).WithTimeout(2 * time.Second).Should(Receive(HaveOccurred()))
})
It("accepts a relative descendant of a relative root", func() {
Expect(InTrustedRoot("models/a/file", "models")).To(Succeed())
})
})
Describe("SanitizeFileName", func() {
+279 -1
View File
@@ -3770,6 +3770,90 @@ const docTemplate = `{
}
}
},
"/v1/systemone": {
"post": {
"description": "Runs zero-shot NER over the supplied state and answers each question. Question types: noul (binary entity presence), choice (pick one option), score (pick one level).",
"tags": [
"systemone"
],
"summary": "Answer structured-extraction questions over state text.",
"parameters": [
{
"description": "state + questions",
"name": "request",
"in": "body",
"required": true,
"schema": {
"$ref": "#/definitions/schema.SystemOneRequest"
}
}
],
"responses": {
"200": {
"description": "OK",
"schema": {
"$ref": "#/definitions/schema.SystemOneResponse"
}
}
}
}
},
"/v1/systemone/permute": {
"post": {
"description": "Re-runs one choice question under n_perm option orders. Reports per-order probabilities, argmax stability, and spread.",
"tags": [
"systemone"
],
"summary": "Re-run a choice question under multiple option orders.",
"parameters": [
{
"description": "request + question + n_perm + seed",
"name": "request",
"in": "body",
"required": true,
"schema": {
"$ref": "#/definitions/schema.SystemOnePermuteRequest"
}
}
],
"responses": {
"200": {
"description": "OK",
"schema": {
"$ref": "#/definitions/schema.SystemOnePermuteResponse"
}
}
}
}
},
"/v1/systemone/separate": {
"post": {
"description": "Runs N independent NER passes, one per question, against the same state. Response shape matches /v1/systemone.",
"tags": [
"systemone"
],
"summary": "Answer each question in a separate NER pass.",
"parameters": [
{
"description": "state + questions",
"name": "request",
"in": "body",
"required": true,
"schema": {
"$ref": "#/definitions/schema.SystemOneRequest"
}
}
],
"responses": {
"200": {
"description": "OK",
"schema": {
"$ref": "#/definitions/schema.SystemOneResponse"
}
}
}
}
},
"/v1/text-to-speech/{voice-id}": {
"post": {
"tags": [
@@ -4093,8 +4177,16 @@ const docTemplate = `{
"config.Gallery": {
"type": "object",
"properties": {
"artifact_verification": {
"description": "ArtifactVerification overrides Verification only for the gallery OCI artifact.\nBackend images keep their separate Verification policy.",
"allOf": [
{
"$ref": "#/definitions/config.GalleryVerification"
}
]
},
"mirrors": {
"description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://).",
"description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://, oci://).",
"type": "array",
"items": {
"type": "string"
@@ -4129,6 +4221,10 @@ const docTemplate = `{
"not_before": {
"description": "NotBefore is an RFC3339 timestamp. Empty disables the time check.",
"type": "string"
},
"source_repository": {
"description": "SourceRepository is an https URL compared exactly against the\ncertificate's source-repository extension. Empty skips the check.",
"type": "string"
}
}
},
@@ -7811,12 +7907,42 @@ const docTemplate = `{
"id": {
"type": "string"
},
"process": {
"description": "Process is the backend process serving the model on this host. Absent\nwhen the model has no local process (a distributed worker holds it) or\nthe process could not be read.",
"allOf": [
{
"$ref": "#/definitions/schema.SysInfoProcess"
}
]
},
"size_vram": {
"description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
"type": "integer"
}
}
},
"schema.SysInfoProcess": {
"type": "object",
"properties": {
"cpu_percent": {
"description": "CPUPercent is the share of the whole host's CPU used since the previous\nreading, 0-100. Absent on the first reading of a process.",
"type": "number"
},
"memory_percent": {
"type": "number"
},
"pid": {
"type": "integer"
},
"rss_bytes": {
"description": "RSSBytes is resident host memory. Weights offloaded to a GPU are not\nin it.",
"type": "integer"
},
"started_at": {
"type": "string"
}
}
},
"schema.SystemInformationResponse": {
"type": "object",
"properties": {
@@ -7836,6 +7962,158 @@ const docTemplate = `{
}
}
},
"schema.SystemOneAnswer": {
"type": "object",
"properties": {
"choice": {
"type": "string"
},
"confidence": {
"type": "number"
},
"entities": {
"type": "array",
"items": {
"$ref": "#/definitions/schema.SystemOneEntity"
}
},
"legend": {
"type": "object",
"additionalProperties": {
"type": "string"
}
},
"noul": {
"type": "number"
},
"probabilities": {
"type": "object",
"additionalProperties": {
"type": "number",
"format": "float64"
}
},
"score": {
"type": "number"
},
"type": {
"type": "string"
}
}
},
"schema.SystemOneEntity": {
"type": "object",
"properties": {
"confidence": {
"type": "number"
},
"end": {
"type": "integer"
},
"start": {
"type": "integer"
},
"text": {
"type": "string"
}
}
},
"schema.SystemOnePermuteRequest": {
"type": "object",
"properties": {
"n_perm": {
"type": "integer"
},
"question": {
"type": "string"
},
"request": {
"$ref": "#/definitions/schema.SystemOneRequest"
},
"seed": {
"type": "integer"
}
}
},
"schema.SystemOnePermuteResponse": {
"type": "object",
"properties": {
"argmax_stable": {
"type": "boolean"
},
"runs": {
"type": "array",
"items": {
"$ref": "#/definitions/schema.SystemOnePermuteRun"
}
},
"spread": {
"type": "object",
"additionalProperties": {
"type": "number",
"format": "float64"
}
}
}
},
"schema.SystemOnePermuteRun": {
"type": "object",
"properties": {
"choice": {
"type": "string"
},
"latency_ms": {
"type": "number"
},
"order": {
"type": "array",
"items": {
"type": "string"
}
},
"probabilities": {
"type": "object",
"additionalProperties": {
"type": "number",
"format": "float64"
}
}
}
},
"schema.SystemOneRequest": {
"type": "object"
},
"schema.SystemOneResponse": {
"type": "object",
"properties": {
"answers": {
"type": "object",
"additionalProperties": {
"$ref": "#/definitions/schema.SystemOneAnswer"
}
},
"latency_ms": {
"type": "number"
},
"model": {
"type": "string"
},
"usage": {
"$ref": "#/definitions/schema.SystemOneUsage"
}
}
},
"schema.SystemOneUsage": {
"type": "object",
"properties": {
"input_tokens": {
"type": "integer"
},
"output_tokens": {
"type": "integer"
}
}
},
"schema.TTSRequest": {
"description": "TTS request body",
"type": "object",
+279 -1
View File
@@ -3767,6 +3767,90 @@
}
}
},
"/v1/systemone": {
"post": {
"description": "Runs zero-shot NER over the supplied state and answers each question. Question types: noul (binary entity presence), choice (pick one option), score (pick one level).",
"tags": [
"systemone"
],
"summary": "Answer structured-extraction questions over state text.",
"parameters": [
{
"description": "state + questions",
"name": "request",
"in": "body",
"required": true,
"schema": {
"$ref": "#/definitions/schema.SystemOneRequest"
}
}
],
"responses": {
"200": {
"description": "OK",
"schema": {
"$ref": "#/definitions/schema.SystemOneResponse"
}
}
}
}
},
"/v1/systemone/permute": {
"post": {
"description": "Re-runs one choice question under n_perm option orders. Reports per-order probabilities, argmax stability, and spread.",
"tags": [
"systemone"
],
"summary": "Re-run a choice question under multiple option orders.",
"parameters": [
{
"description": "request + question + n_perm + seed",
"name": "request",
"in": "body",
"required": true,
"schema": {
"$ref": "#/definitions/schema.SystemOnePermuteRequest"
}
}
],
"responses": {
"200": {
"description": "OK",
"schema": {
"$ref": "#/definitions/schema.SystemOnePermuteResponse"
}
}
}
}
},
"/v1/systemone/separate": {
"post": {
"description": "Runs N independent NER passes, one per question, against the same state. Response shape matches /v1/systemone.",
"tags": [
"systemone"
],
"summary": "Answer each question in a separate NER pass.",
"parameters": [
{
"description": "state + questions",
"name": "request",
"in": "body",
"required": true,
"schema": {
"$ref": "#/definitions/schema.SystemOneRequest"
}
}
],
"responses": {
"200": {
"description": "OK",
"schema": {
"$ref": "#/definitions/schema.SystemOneResponse"
}
}
}
}
},
"/v1/text-to-speech/{voice-id}": {
"post": {
"tags": [
@@ -4090,8 +4174,16 @@
"config.Gallery": {
"type": "object",
"properties": {
"artifact_verification": {
"description": "ArtifactVerification overrides Verification only for the gallery OCI artifact.\nBackend images keep their separate Verification policy.",
"allOf": [
{
"$ref": "#/definitions/config.GalleryVerification"
}
]
},
"mirrors": {
"description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://).",
"description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://, oci://).",
"type": "array",
"items": {
"type": "string"
@@ -4126,6 +4218,10 @@
"not_before": {
"description": "NotBefore is an RFC3339 timestamp. Empty disables the time check.",
"type": "string"
},
"source_repository": {
"description": "SourceRepository is an https URL compared exactly against the\ncertificate's source-repository extension. Empty skips the check.",
"type": "string"
}
}
},
@@ -7808,12 +7904,42 @@
"id": {
"type": "string"
},
"process": {
"description": "Process is the backend process serving the model on this host. Absent\nwhen the model has no local process (a distributed worker holds it) or\nthe process could not be read.",
"allOf": [
{
"$ref": "#/definitions/schema.SysInfoProcess"
}
]
},
"size_vram": {
"description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
"type": "integer"
}
}
},
"schema.SysInfoProcess": {
"type": "object",
"properties": {
"cpu_percent": {
"description": "CPUPercent is the share of the whole host's CPU used since the previous\nreading, 0-100. Absent on the first reading of a process.",
"type": "number"
},
"memory_percent": {
"type": "number"
},
"pid": {
"type": "integer"
},
"rss_bytes": {
"description": "RSSBytes is resident host memory. Weights offloaded to a GPU are not\nin it.",
"type": "integer"
},
"started_at": {
"type": "string"
}
}
},
"schema.SystemInformationResponse": {
"type": "object",
"properties": {
@@ -7833,6 +7959,158 @@
}
}
},
"schema.SystemOneAnswer": {
"type": "object",
"properties": {
"choice": {
"type": "string"
},
"confidence": {
"type": "number"
},
"entities": {
"type": "array",
"items": {
"$ref": "#/definitions/schema.SystemOneEntity"
}
},
"legend": {
"type": "object",
"additionalProperties": {
"type": "string"
}
},
"noul": {
"type": "number"
},
"probabilities": {
"type": "object",
"additionalProperties": {
"type": "number",
"format": "float64"
}
},
"score": {
"type": "number"
},
"type": {
"type": "string"
}
}
},
"schema.SystemOneEntity": {
"type": "object",
"properties": {
"confidence": {
"type": "number"
},
"end": {
"type": "integer"
},
"start": {
"type": "integer"
},
"text": {
"type": "string"
}
}
},
"schema.SystemOnePermuteRequest": {
"type": "object",
"properties": {
"n_perm": {
"type": "integer"
},
"question": {
"type": "string"
},
"request": {
"$ref": "#/definitions/schema.SystemOneRequest"
},
"seed": {
"type": "integer"
}
}
},
"schema.SystemOnePermuteResponse": {
"type": "object",
"properties": {
"argmax_stable": {
"type": "boolean"
},
"runs": {
"type": "array",
"items": {
"$ref": "#/definitions/schema.SystemOnePermuteRun"
}
},
"spread": {
"type": "object",
"additionalProperties": {
"type": "number",
"format": "float64"
}
}
}
},
"schema.SystemOnePermuteRun": {
"type": "object",
"properties": {
"choice": {
"type": "string"
},
"latency_ms": {
"type": "number"
},
"order": {
"type": "array",
"items": {
"type": "string"
}
},
"probabilities": {
"type": "object",
"additionalProperties": {
"type": "number",
"format": "float64"
}
}
}
},
"schema.SystemOneRequest": {
"type": "object"
},
"schema.SystemOneResponse": {
"type": "object",
"properties": {
"answers": {
"type": "object",
"additionalProperties": {
"$ref": "#/definitions/schema.SystemOneAnswer"
}
},
"latency_ms": {
"type": "number"
},
"model": {
"type": "string"
},
"usage": {
"$ref": "#/definitions/schema.SystemOneUsage"
}
}
},
"schema.SystemOneUsage": {
"type": "object",
"properties": {
"input_tokens": {
"type": "integer"
},
"output_tokens": {
"type": "integer"
}
}
},
"schema.TTSRequest": {
"description": "TTS request body",
"type": "object",
+196 -1
View File
@@ -2,13 +2,19 @@ basePath: /
definitions:
config.Gallery:
properties:
artifact_verification:
allOf:
- $ref: '#/definitions/config.GalleryVerification'
description: |-
ArtifactVerification overrides Verification only for the gallery OCI artifact.
Backend images keep their separate Verification policy.
mirrors:
description: |-
Mirrors are tried in order when URL cannot be fetched. They are a
fallback for availability, not a load-balancing pool: the primary is
always preferred, and a mirror is only consulted after the one before
it fails. Any URI the gallery loader understands works here
(https://, github:, file://).
(https://, github:, file://, oci://).
items:
type: string
type: array
@@ -32,6 +38,11 @@ definitions:
not_before:
description: NotBefore is an RFC3339 timestamp. Empty disables the time check.
type: string
source_repository:
description: |-
SourceRepository is an https URL compared exactly against the
certificate's source-repository extension. Empty skips the check.
type: string
type: object
config.TTSVoice:
properties:
@@ -2654,12 +2665,38 @@ definitions:
type: string
id:
type: string
process:
allOf:
- $ref: '#/definitions/schema.SysInfoProcess'
description: |-
Process is the backend process serving the model on this host. Absent
when the model has no local process (a distributed worker holds it) or
the process could not be read.
size_vram:
description: |-
SizeVRAM is DRM-accounted resident device memory in bytes. Nil means
the backend process tree has no complete supported reading.
type: integer
type: object
schema.SysInfoProcess:
properties:
cpu_percent:
description: |-
CPUPercent is the share of the whole host's CPU used since the previous
reading, 0-100. Absent on the first reading of a process.
type: number
memory_percent:
type: number
pid:
type: integer
rss_bytes:
description: |-
RSSBytes is resident host memory. Weights offloaded to a GPU are not
in it.
type: integer
started_at:
type: string
type: object
schema.SystemInformationResponse:
properties:
backends:
@@ -2673,6 +2710,106 @@ definitions:
$ref: '#/definitions/schema.SysInfoModel'
type: array
type: object
schema.SystemOneAnswer:
properties:
choice:
type: string
confidence:
type: number
entities:
items:
$ref: '#/definitions/schema.SystemOneEntity'
type: array
legend:
additionalProperties:
type: string
type: object
noul:
type: number
probabilities:
additionalProperties:
format: float64
type: number
type: object
score:
type: number
type:
type: string
type: object
schema.SystemOneEntity:
properties:
confidence:
type: number
end:
type: integer
start:
type: integer
text:
type: string
type: object
schema.SystemOnePermuteRequest:
properties:
n_perm:
type: integer
question:
type: string
request:
$ref: '#/definitions/schema.SystemOneRequest'
seed:
type: integer
type: object
schema.SystemOnePermuteResponse:
properties:
argmax_stable:
type: boolean
runs:
items:
$ref: '#/definitions/schema.SystemOnePermuteRun'
type: array
spread:
additionalProperties:
format: float64
type: number
type: object
type: object
schema.SystemOnePermuteRun:
properties:
choice:
type: string
latency_ms:
type: number
order:
items:
type: string
type: array
probabilities:
additionalProperties:
format: float64
type: number
type: object
type: object
schema.SystemOneRequest:
type: object
schema.SystemOneResponse:
properties:
answers:
additionalProperties:
$ref: '#/definitions/schema.SystemOneAnswer'
type: object
latency_ms:
type: number
model:
type: string
usage:
$ref: '#/definitions/schema.SystemOneUsage'
type: object
schema.SystemOneUsage:
properties:
input_tokens:
type: integer
output_tokens:
type: integer
type: object
schema.TTSRequest:
description: TTS request body
properties:
@@ -5672,6 +5809,64 @@ paths:
summary: Generates audio from the input text.
tags:
- audio
/v1/systemone:
post:
description: 'Runs zero-shot NER over the supplied state and answers each question.
Question types: noul (binary entity presence), choice (pick one option), score
(pick one level).'
parameters:
- description: state + questions
in: body
name: request
required: true
schema:
$ref: '#/definitions/schema.SystemOneRequest'
responses:
"200":
description: OK
schema:
$ref: '#/definitions/schema.SystemOneResponse'
summary: Answer structured-extraction questions over state text.
tags:
- systemone
/v1/systemone/permute:
post:
description: Re-runs one choice question under n_perm option orders. Reports
per-order probabilities, argmax stability, and spread.
parameters:
- description: request + question + n_perm + seed
in: body
name: request
required: true
schema:
$ref: '#/definitions/schema.SystemOnePermuteRequest'
responses:
"200":
description: OK
schema:
$ref: '#/definitions/schema.SystemOnePermuteResponse'
summary: Re-run a choice question under multiple option orders.
tags:
- systemone
/v1/systemone/separate:
post:
description: Runs N independent NER passes, one per question, against the same
state. Response shape matches /v1/systemone.
parameters:
- description: state + questions
in: body
name: request
required: true
schema:
$ref: '#/definitions/schema.SystemOneRequest'
responses:
"200":
description: OK
schema:
$ref: '#/definitions/schema.SystemOneResponse'
summary: Answer each question in a separate NER pass.
tags:
- systemone
/v1/text-to-speech/{voice-id}:
post:
parameters: