From 0e4347037fdb3d9f685ced9dc8c55c5a8f8a7d27 Mon Sep 17 00:00:00 2001 From: leilei3167 Date: Mon, 28 Sep 2026 10:49:32 +0800 Subject: [PATCH 01/15] fix: point docker-compose default at gallery phi-2-chat (#11987) The quickstart compose file still requested phi-2, which is no longer in the gallery. Use phi-2-chat instead and fix model preload error wrapping so discover/install failures report the real error instead of %!w(). Keep earlier model failures when discovery fails for another model. Document the Compose gallery default. Fixes #11974 Signed-off-by: lei_lei --- core/startup/model_preload.go | 4 ++-- docker-compose.yaml | 2 +- docs/content/getting-started/containers.md | 2 ++ 3 files changed, 5 insertions(+), 3 deletions(-) diff --git a/core/startup/model_preload.go b/core/startup/model_preload.go index 4f3bb1683..8dd167828 100644 --- a/core/startup/model_preload.go +++ b/core/startup/model_preload.go @@ -41,7 +41,7 @@ func InstallModelsWithOptions(ctx context.Context, galleryService *galleryop.Gal // Check if it's a model gallery, or print a warning e, found := installModel(ctx, galleries, backendGalleries, url, systemState, modelLoader, downloadStatus, enforceScan, autoloadBackendGalleries, requireBackendIntegrity, installOptions...) if e != nil && found { - xlog.Error("[startup] failed installing model", "error", err, "model", url) + xlog.Error("[startup] failed installing model", "error", e, "model", url) err = errors.Join(err, e) } else if !found { xlog.Debug("[startup] model not found in the gallery", "model", url) @@ -54,7 +54,7 @@ func InstallModelsWithOptions(ctx context.Context, galleryService *galleryop.Gal modelConfig, discoverErr := importers.DiscoverModelConfig(url, json.RawMessage{}) if discoverErr != nil { xlog.Error("[startup] failed to discover model config", "error", discoverErr, "model", url) - err = errors.Join(discoverErr, fmt.Errorf("failed to discover model config: %w", err)) + err = errors.Join(err, fmt.Errorf("failed to discover model config: %w", discoverErr)) continue } diff --git a/docker-compose.yaml b/docker-compose.yaml index 82b3c18b6..50eccaaf4 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -41,7 +41,7 @@ services: # Here we can specify a list of models to run (see quickstart https://localai.io/basics/getting_started/#running-models ) # or an URL pointing to a YAML configuration file, for example: # - https://gist.githubusercontent.com/mudler/ad601a0488b497b69ec549150d9edd18/raw/a8a8869ef1bb7e3830bf5c0bae29a0cce991ff8d/phi-2.yaml - - phi-2 + - phi-2-chat # For NVIDIA GPU support with CDI (recommended for NVIDIA Container Toolkit 1.14+): # Uncomment the following deploy section and use driver: nvidia.com/gpu. # Include `utility` in capabilities so nvidia-smi / NVML are available — diff --git a/docs/content/getting-started/containers.md b/docs/content/getting-started/containers.md index 6e6ca9526..d5616b201 100644 --- a/docs/content/getting-started/containers.md +++ b/docs/content/getting-started/containers.md @@ -108,6 +108,8 @@ docker run -ti --name local-ai -p 8080:8080 --runtime nvidia --gpus all localai/ ## Using Compose +The repository's `docker-compose.yaml` installs `phi-2-chat` from the model gallery by default. Change its `command` list to select a different gallery model. + For a more manageable setup, especially with persistent volumes, use Docker Compose or Podman Compose: ### Using CDI (Container Device Interface) - Recommended for NVIDIA Container Toolkit 1.14+ From 61f4f67b756510c15963812b095d3072bfa2cd0d Mon Sep 17 00:00:00 2001 From: pos-ei-don <1822533+pos-ei-don@users.noreply.github.com> Date: Mon, 28 Sep 2026 04:49:37 +0200 Subject: [PATCH 02/15] sglang backend: pass through thinking_budget + require_reasoning (#12193) * sglang backend: pass through thinking_budget + require_reasoning sglang's raw Engine.async_generate() API (which this backend calls directly, bypassing sglang's own OpenAI server) supports a precise, tokenizer-derived reasoning-length budget via sampling_params["custom_params"]["thinking_budget"] plus require_reasoning=True, gated behind --enable-strict-thinking. Neither was reachable through LocalAI: this backend built sampling_params only from a fixed field mapping (temperature, top_p, ...) with no custom_params key, and never passed require_reasoning to async_generate at all. - LoadModel now reads a model-level "thinking_budget" option (same mechanism as the existing tool_parser/reasoning_parser options), and _build_sampling_params adds it as custom_params.thinking_budget on every request when configured. - _new_reasoning_parser already derives, from the rendered prompt, whether the model's chat template pre-opened a reasoning block (Qwen3-style templates append to the prompt instead of letting the model emit it) -- the same signal sglang's own OpenAI server computes from per-template config to decide require_reasoning. This backend has no template manager, so it now returns that signal too and _predict forwards it to async_generate(require_reasoning=...). Verified against production (NVFP4, sm_121, Qwen3.6-35B-A3B) via a raw Engine.async_generate() call bypassing this backend: 301 reasoning tokens against a 300-token budget, clean completion, ~27s. Not yet verified through this backend's own gRPC path end-to-end (no local CUDA/sglang environment available here) -- existing + new unit tests in test.py cover the pure-Python merge/passthrough logic only. Scope note: require_reasoning is derived only from the existing prompt-suffix heuristic, not sglang's full per-template _get_reasoning_from_request decision tree (minimax-m3/hunyuan special cases etc.) -- this backend has no template manager to evaluate that tree against, and the prompt-suffix check is the one heuristic already validated in this file (test_reasoning_parser_forced_when_template_prefills_think_tag). Signed-off-by: pos-ei-don <1822533+pos-ei-don@users.noreply.github.com> * sglang backend: honour a model-level reasoning_default A model YAML can already carry "parameters: reasoning_effort:", but that value only reaches this backend when a *caller* sets it per request (the Go side turns it into Metadata["enable_thinking"]). As a model-level default it is silently dropped: a config reading "reasoning_effort: none" still produces full reasoning on every request, so the config says one thing and the model does another. That gap is expensive in practice. On a self-hosted Qwen3.6-35B-A3B the reasoning phase consumed the entire max_tokens budget before any content was produced - 90% of code completions came back empty at max_tokens=768, and the server log filled with "backend produced only reasoning, retrying". The config looked like reasoning was off the whole time. This adds "reasoning_default:off" (or ":on") on the same model-level options: mechanism as thinking_budget. A per-request value always wins; the default only fills in when the request is silent. Measured on the stack above (sglang 0.5.20, NVFP4, GB10/sm_121) after applying it: default (nothing set) -> 0 chars reasoning, 27 tokens "reasoning_effort": "none" -> 0 chars reasoning, 27 tokens metadata enable_thinking=true -> capped at the 512-token thinking_budget, 541 tokens total, finish_reason stop Tests: three cases added to backend/python/sglang/test.py covering the default, per-request override in both directions, and the unconfigured case (which must leave the template untouched). Signed-off-by: pos-ei-don <1822533+pos-ei-don@users.noreply.github.com> * sglang backend: validate thinking_budget instead of crashing LoadModel Addresses the review on this PR: - `int(thinking_budget)` raised on values like "5000.0" or "abc" and took LoadModel down. The option is now parsed by _parse_thinking_budget(): integral numbers in any spelling are accepted, anything else is ignored with a warning on stderr. - Zero and negative budgets are ignored with a warning instead of being passed to sglang, where they have no defined meaning. Turning reasoning off is what reasoning_default:off is for. - A load-time warning when thinking_budget is set but enable_strict_thinking is not in engine_args, since sglang then ignores the budget silently. - Tests for integral spellings, unset, zero, negative, non-integer and the strict-thinking warning. Signed-off-by: pos-ei-don <1822533+pos-ei-don@users.noreply.github.com> * docs(sglang): explain reasoning options Document the reasoning budget, strict-thinking requirement, and precedence of request metadata over the model-level default. Also note that the budget has to stay well below max_tokens (otherwise it never triggers and the reply can end up empty), and that POST /models/reload or a backend-only restart does not pick up changed options; LocalAI itself has to be restarted. Assisted-by: Codex:GPT-6 Signed-off-by: pos-ei-don <1822533+pos-ei-don@users.noreply.github.com> * docs(sglang): clarify configuration reloads Distinguish rereading model configuration from updating a running backend. Keep the full LocalAI restart recommendation for changed reasoning options. Assisted-by: Codex:GPT-6 * sglang backend: only pass require_reasoning when sglang supports it Engine.async_generate() gained the require_reasoning keyword in sglang 0.5.13 and takes no **kwargs. The CPU profile builds v0.5.11 from source and the other profiles only set a >=0.5.11 floor, so passing the keyword unconditionally made every request fail with TypeError. Detect support once at import time, as the file already does for sampling_seed. enable_strict_thinking first appears in sglang 0.5.12; fix the comment. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --------- Signed-off-by: pos-ei-don <1822533+pos-ei-don@users.noreply.github.com> Signed-off-by: Ettore Di Giacinto Co-authored-by: localai-org-maint-bot Co-authored-by: Ettore Di Giacinto --- backend/python/sglang/backend.py | 157 +++++++++++++++++++++-- backend/python/sglang/test.py | 126 +++++++++++++++++- docs/content/features/text-generation.md | 35 +++++ 3 files changed, 302 insertions(+), 16 deletions(-) diff --git a/backend/python/sglang/backend.py b/backend/python/sglang/backend.py index 838d394bb..7936ddcf2 100644 --- a/backend/python/sglang/backend.py +++ b/backend/python/sglang/backend.py @@ -90,6 +90,19 @@ except Exception: _SEED_KEY = "sampling_seed" +# Engine.async_generate() only grew a require_reasoning keyword in sglang +# 0.5.13. The CPU build compiles v0.5.11 from source and the other profiles +# only set a >=0.5.11 floor, and async_generate() takes no **kwargs, so +# passing the keyword unconditionally fails every request with TypeError. +try: + import inspect as _inspect + _ASYNC_GENERATE_HAS_REQUIRE_REASONING = ( + "require_reasoning" in _inspect.signature(Engine.async_generate).parameters + ) +except Exception: + _ASYNC_GENERATE_HAS_REQUIRE_REASONING = False + + _ONE_DAY_IN_SECONDS = 60 * 60 * 24 # proto3 has no field presence, so an explicit 0 is indistinguishable from @@ -105,6 +118,12 @@ MAX_WORKERS = int(os.environ.get('PYTHON_GRPC_MAX_WORKERS', '1')) class BackendServicer(backend_pb2_grpc.BackendServicer): """gRPC servicer implementing the Backend service for sglang.""" + # Class-level default so a servicer used before LoadModel (e.g. in unit + # tests that construct it directly) doesn't AttributeError in + # _build_sampling_params. + thinking_budget: Optional[int] = None + reasoning_default: Optional[str] = None + def _parse_options(self, options_list) -> Dict[str, str]: opts: Dict[str, str] = {} for opt in options_list: @@ -114,6 +133,49 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): opts[key.strip()] = value.strip() return opts + @staticmethod + def _parse_thinking_budget(value) -> Optional[int]: + """Turn the `thinking_budget` model option into a positive int, or None. + + Options arrive as strings from the YAML `options:` list, but a value + like "5000.0" is a plausible thing to write, and a crash here would + take down LoadModel for the whole model. So: integral numbers are + accepted in any spelling ("512", "512.0"), anything else is ignored + with a warning instead of raising. Zero and negative budgets are + ignored too: sglang gives them no defined meaning, and turning + reasoning off is what `reasoning_default: off` is for. + """ + if value is None or str(value).strip() == "": + return None + raw = str(value).strip() + try: + number = float(raw) + except ValueError: + print(f"thinking_budget {raw!r} is not a number, ignoring it", file=sys.stderr) + return None + if not number.is_integer(): + print(f"thinking_budget {raw!r} is not a whole number of tokens, ignoring it", file=sys.stderr) + return None + if number <= 0: + print( + f"thinking_budget {raw!r} must be positive, ignoring it " + "(use reasoning_default:off to disable reasoning)", + file=sys.stderr, + ) + return None + return int(number) + + @staticmethod + def _strict_thinking_warning(thinking_budget: Optional[int], engine_kwargs: dict) -> Optional[str]: + """sglang only enforces the budget with enable_strict_thinking on; without + it the budget is silently ignored, so say so at load time.""" + if thinking_budget is not None and not engine_kwargs.get("enable_strict_thinking"): + return ( + f"thinking_budget={thinking_budget} is set but enable_strict_thinking is not " + "in engine_args; sglang will ignore the budget" + ) + return None + def _apply_engine_args(self, engine_kwargs: dict, engine_args_json: str) -> dict: """Merge user-supplied engine_args (JSON object) into the kwargs dict that will be forwarded to ``sglang.Engine`` (which constructs a @@ -230,6 +292,35 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): self.tool_parser_name: Optional[str] = opts.get("tool_parser") or None self.reasoning_parser_name: Optional[str] = opts.get("reasoning_parser") or None + # Fixed reasoning-length budget for every request on this model, in + # tokens. There is no protobuf field to carry a per-request + # custom_params blob, so this rides the same model-level `options:` + # mechanism as tool_parser/reasoning_parser above — mirroring how + # sglang's own `--preferred-sampling-params` is a server-wide + # default, not a per-request choice. Requires `enable_strict_thinking` + # in `engine_args:` (sglang >=0.5.12); without it sglang has no + # tokenizer-derived budget mechanism to enforce this against. + self.thinking_budget: Optional[int] = self._parse_thinking_budget( + opts.get("thinking_budget") + ) + + # Model-level default for whether the chat template opens a reasoning + # block, as "off" or "on". Rides the same `options:` mechanism as + # thinking_budget above. + # + # Why this is needed even though `reasoning_effort` exists: that one + # only reaches this backend when a *caller* sets it per request (the + # Go side turns it into Metadata["enable_thinking"]). As a model-level + # `parameters:` default it is silently dropped, so a config reading + # `reasoning_effort: none` still produces full reasoning on every + # request - the config says one thing and the model does another. + # + # A per-request value always wins; this only fills in the gap when the + # request says nothing. + self.reasoning_default: Optional[str] = ( + opts.get("reasoning_default") or "" + ).lower() or None + # Also hand the parser names to sglang's engine so its HTTP/OAI # paths work identically if someone hits the engine directly. if self.tool_parser_name: @@ -247,6 +338,10 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): print(f"engine_args error: {err}", file=sys.stderr) return backend_pb2.Result(success=False, message=str(err)) + warning = self._strict_thinking_warning(self.thinking_budget, engine_kwargs) + if warning: + print(warning, file=sys.stderr) + try: self.llm = Engine(**engine_kwargs) except Exception as err: @@ -362,8 +457,28 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): except json.JSONDecodeError: sampling_params["ebnf"] = grammar + if self.thinking_budget is not None: + sampling_params["custom_params"] = {"thinking_budget": self.thinking_budget} + return sampling_params + def _thinking_default(self, request) -> Optional[bool]: + """Whether this request should render with reasoning on, off, or unset. + + Per-request ``Metadata["enable_thinking"]`` wins; the model-level + ``reasoning_default`` option fills in when the request is silent. + Returns None when neither says anything, leaving template behaviour + untouched. + """ + wanted = request.Metadata.get("enable_thinking", "").lower() + if wanted in ("true", "false"): + return wanted == "true" + if self.reasoning_default == "off": + return False + if self.reasoning_default == "on": + return True + return None + def _build_prompt(self, request) -> str: prompt = request.Prompt if prompt or not request.UseTokenizerTemplate or not request.Messages: @@ -384,9 +499,9 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): template_kwargs["tools"] = json.loads(request.Tools) except json.JSONDecodeError: pass - _thinking = request.Metadata.get("enable_thinking", "").lower() - if _thinking in ("true", "false"): - template_kwargs["enable_thinking"] = (_thinking == "true") + _thinking = self._thinking_default(request) + if _thinking is not None: + template_kwargs["enable_thinking"] = _thinking # sglang locates the attached images/videos by scanning the rendered # prompt for the model's own media token, so the template has to be @@ -438,12 +553,19 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): there files the answer as reasoning and leaves content empty. sglang's own server keeps the two apart for the same reason — its grammar backend owns the reasoning prefix when a reasoning parser is set. + + Returns a ``(parser, forced)`` pair. ``forced`` is also the signal + ``_predict`` passes as ``Engine.async_generate(require_reasoning=...)``: + sglang's own OpenAI server derives that flag from per-template + config (``ChatServing._get_reasoning_from_request``); this backend + has no template manager, so the same prompt-suffix heuristic that + already decides parser forcing doubles as that signal. """ if grammar_constrained: prompt = "" if not (HAS_REASONING_PARSERS and self.reasoning_parser_name): - return None + return None, False kwargs = { "model_type": self.reasoning_parser_name, @@ -453,10 +575,12 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): parser = ReasoningParser(**kwargs) except Exception as e: print(f"ReasoningParser init failed: {e!r}", file=sys.stderr) - return None + return None, False + forced = False start = getattr(getattr(parser, "detector", None), "think_start_token", None) if start and prompt and prompt.rstrip().endswith(start): + forced = True try: parser = ReasoningParser(force_reasoning=True, **kwargs) except TypeError: @@ -469,10 +593,16 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): file=sys.stderr, ) - return parser + return parser, forced def _make_parsers(self, request, prompt: str = ""): - """Construct fresh per-request parser instances (stateful).""" + """Construct fresh per-request parser instances (stateful). + + Also returns ``require_reasoning`` (see ``_new_reasoning_parser``), + which ``_predict`` forwards to ``Engine.async_generate()`` so + sglang's ``--enable-strict-thinking`` grammar backend knows this + request is in a reasoning block. + """ tool_parser = None if HAS_TOOL_PARSERS and self.tool_parser_name and request.Tools: @@ -485,23 +615,27 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): except Exception as e: print(f"FunctionCallParser init failed: {e!r}", file=sys.stderr) - reasoning_parser = self._new_reasoning_parser( + reasoning_parser, require_reasoning = self._new_reasoning_parser( True, prompt, bool(getattr(request, "Grammar", "")), ) - return tool_parser, reasoning_parser + return tool_parser, reasoning_parser, require_reasoning async def _predict(self, request, context, streaming: bool = False): sampling_params = self._build_sampling_params(request) prompt = self._build_prompt(request) - tool_parser, reasoning_parser = self._make_parsers(request, prompt) + tool_parser, reasoning_parser, require_reasoning = self._make_parsers(request, prompt) image_data = list(request.Images) if request.Images else None video_data = list(request.Videos) if request.Videos else None # Kick off streaming generation. We always use stream=True so the # non-stream path still gets parser coverage on the final text. + generate_kwargs = {} + if _ASYNC_GENERATE_HAS_REQUIRE_REASONING: + generate_kwargs["require_reasoning"] = require_reasoning + try: iterator = await self.llm.async_generate( prompt=prompt, @@ -509,6 +643,7 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): image_data=image_data, video_data=video_data, stream=True, + **generate_kwargs, ) except Exception as e: print(f"sglang async_generate failed: {e!r}", file=sys.stderr) @@ -591,7 +726,7 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): final_tool_calls: List[backend_pb2.ToolCallDelta] = [] if not streaming: - final_reasoning_parser = self._new_reasoning_parser( + final_reasoning_parser, _ = self._new_reasoning_parser( False, prompt, bool(getattr(request, "Grammar", "")), ) diff --git a/backend/python/sglang/test.py b/backend/python/sglang/test.py index 51158791c..1b1a10d23 100644 --- a/backend/python/sglang/test.py +++ b/backend/python/sglang/test.py @@ -9,6 +9,13 @@ because ``_apply_engine_args`` validates keys against ``ServerArgs`` import unittest + +def _request(metadata=None): + """Minimal stand-in for a PredictOptions request in reasoning tests.""" + from types import SimpleNamespace + + return SimpleNamespace(Metadata=metadata or {}) + class TestSglangHelpers(unittest.TestCase): """Tests for the pure helpers on BackendServicer (no gRPC, no engine).""" @@ -170,13 +177,17 @@ class TestSglangHelpers(unittest.TestCase): # What the model actually emits when the prompt ends in "". completion = "adding two and two4" - forced = servicer._new_reasoning_parser(False, prompt="user: hi\n\n") + forced, require_reasoning = servicer._new_reasoning_parser( + False, prompt="user: hi\n\n" + ) + self.assertTrue(require_reasoning) reasoning, content = forced.parse_non_stream(completion) self.assertEqual(reasoning, "adding two and two") self.assertEqual(content, "4") # No prefilled tag in the prompt: detector default, unchanged behaviour. - unforced = servicer._new_reasoning_parser(False, prompt="user: hi\n") + unforced, require_reasoning = servicer._new_reasoning_parser(False, prompt="user: hi\n") + self.assertFalse(require_reasoning) reasoning, content = unforced.parse_non_stream(completion) self.assertFalse(reasoning) self.assertEqual(content, completion) @@ -187,7 +198,8 @@ class TestSglangHelpers(unittest.TestCase): servicer = self._servicer() servicer.reasoning_parser_name = "qwen3" - parser = servicer._new_reasoning_parser(False, prompt="user: primes?\n") + parser, require_reasoning = servicer._new_reasoning_parser(False, prompt="user: primes?\n") + self.assertFalse(require_reasoning) reasoning, content = parser.parse_non_stream("2,3,5,7,11") self.assertFalse(reasoning) self.assertEqual(content, "2,3,5,7,11") @@ -200,9 +212,10 @@ class TestSglangHelpers(unittest.TestCase): servicer.reasoning_parser_name = "qwen3" schema_out = '{"findings": [{"line": 42, "issue": "off-by-one"}]}' - parser = servicer._new_reasoning_parser( + parser, require_reasoning = servicer._new_reasoning_parser( False, prompt="audit this\n\n", grammar_constrained=True, ) + self.assertFalse(require_reasoning) reasoning, content = parser.parse_non_stream(schema_out) self.assertFalse(reasoning) self.assertEqual(content, schema_out) @@ -210,7 +223,110 @@ class TestSglangHelpers(unittest.TestCase): def test_reasoning_parser_absent_without_configured_parser(self): servicer = self._servicer() servicer.reasoning_parser_name = None - self.assertIsNone(servicer._new_reasoning_parser(False, prompt="")) + parser, require_reasoning = servicer._new_reasoning_parser(False, prompt="") + self.assertIsNone(parser) + self.assertFalse(require_reasoning) + + def test_reasoning_default_off_applies_when_request_is_silent(self): + """A model configured with reasoning_default:off must render with + thinking disabled even when the request carries no enable_thinking - + that is the whole point: `parameters: reasoning_effort:` never + reaches this backend, so without this the config lies about the + default.""" + servicer = self._servicer() + servicer.reasoning_default = "off" + self.assertIs(servicer._thinking_default(_request(metadata={})), False) + + def test_request_metadata_overrides_reasoning_default(self): + """A per-request value always wins over the model-level default - + in both directions.""" + servicer = self._servicer() + servicer.reasoning_default = "off" + self.assertIs( + servicer._thinking_default(_request(metadata={"enable_thinking": "true"})), + True, + ) + servicer.reasoning_default = "on" + self.assertIs( + servicer._thinking_default(_request(metadata={"enable_thinking": "false"})), + False, + ) + + def test_no_reasoning_default_leaves_template_untouched(self): + """Unconfigured must stay unconfigured: returning None means the + backend adds no enable_thinking kwarg at all, so the template keeps + whatever default it ships with.""" + servicer = self._servicer() + self.assertIsNone(servicer._thinking_default(_request(metadata={}))) + + def test_thinking_budget_added_to_sampling_params_as_custom_params(self): + """The model-level thinking_budget option (set from LoadModel's + Options, mirroring tool_parser/reasoning_parser) must ride along as + sampling_params['custom_params']['thinking_budget'] on every + request — that's the only field sglang's --enable-strict-thinking + grammar backend reads to bound the reasoning length.""" + from types import SimpleNamespace + + servicer = self._servicer() + servicer.thinking_budget = 512 + request = SimpleNamespace( + Temperature=0.7, N=0, PresencePenalty=0, FrequencyPenalty=0, + RepetitionPenalty=0, TopP=0, TopK=0, MinP=0, Seed=0, + StopPrompts=[], StopTokenIds=[], IgnoreEOS=False, Tokens=0, + MinTokens=0, SkipSpecialTokens=False, Grammar="", + ) + params = servicer._build_sampling_params(request) + self.assertEqual(params["custom_params"], {"thinking_budget": 512}) + + def test_no_thinking_budget_means_no_custom_params_key(self): + """Unconfigured is unconfigured: no thinking_budget option must not + add an empty/None custom_params that could clobber a sglang-side + --preferred-sampling-params default (see sglang#40634).""" + from types import SimpleNamespace + + servicer = self._servicer() + request = SimpleNamespace( + Temperature=0.7, N=0, PresencePenalty=0, FrequencyPenalty=0, + RepetitionPenalty=0, TopP=0, TopK=0, MinP=0, Seed=0, + StopPrompts=[], StopTokenIds=[], IgnoreEOS=False, Tokens=0, + MinTokens=0, SkipSpecialTokens=False, Grammar="", + ) + params = servicer._build_sampling_params(request) + self.assertNotIn("custom_params", params) + + def test_thinking_budget_accepts_integral_spellings(self): + """YAML options arrive as strings; "512" and "512.0" both mean 512.""" + servicer = self._servicer() + self.assertEqual(servicer._parse_thinking_budget("512"), 512) + self.assertEqual(servicer._parse_thinking_budget("512.0"), 512) + self.assertEqual(servicer._parse_thinking_budget(" 64 "), 64) + self.assertEqual(servicer._parse_thinking_budget(256), 256) + + def test_thinking_budget_unset_is_none(self): + servicer = self._servicer() + self.assertIsNone(servicer._parse_thinking_budget(None)) + self.assertIsNone(servicer._parse_thinking_budget("")) + + def test_thinking_budget_zero_and_negative_are_ignored(self): + """No defined meaning in sglang -- ignored, not passed through.""" + servicer = self._servicer() + self.assertIsNone(servicer._parse_thinking_budget("0")) + self.assertIsNone(servicer._parse_thinking_budget("-100")) + + def test_thinking_budget_non_integer_does_not_raise(self): + """A bad value must not crash LoadModel for the whole model.""" + servicer = self._servicer() + self.assertIsNone(servicer._parse_thinking_budget("12.5")) + self.assertIsNone(servicer._parse_thinking_budget("lots")) + + def test_warns_when_budget_set_without_strict_thinking(self): + servicer = self._servicer() + self.assertIn( + "enable_strict_thinking", + servicer._strict_thinking_warning(512, {"model_path": "x"}), + ) + self.assertIsNone(servicer._strict_thinking_warning(512, {"enable_strict_thinking": True})) + self.assertIsNone(servicer._strict_thinking_warning(None, {})) def test_explicit_zero_temperature_and_seed_are_preserved(self): """Temperature=0 is greedy decoding and 0 is a valid seed — neither is diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md index dadb5e9af..c28bf9213 100644 --- a/docs/content/features/text-generation.md +++ b/docs/content/features/text-generation.md @@ -987,6 +987,41 @@ options: The full list of registered parsers lives in `sglang.srt.function_call` and `sglang.srt.parser.reasoning_parser`. +#### Reasoning defaults and token budgets + +Set SGLang reasoning options in the model's `options:` list: + +```yaml +options: + - reasoning_parser:qwen3 + - thinking_budget:512 + - reasoning_default:on +engine_args: + enable_strict_thinking: true +``` + +`thinking_budget` sets a positive integer token budget for reasoning on each request. +Invalid, zero, and negative values produce a warning and leave the budget unset. +SGLang requires `engine_args.enable_strict_thinking: true` to enforce the budget. +LocalAI warns if you configure a budget without that engine option. +Keep the budget well below the `max_tokens` of your requests: if `max_tokens` is reached first, +the budget never triggers and the whole reply can be spent on reasoning, leaving the answer empty. + +`reasoning_default:on` or `reasoning_default:off` sets the default for LocalAI's tokenizer chat template. +Request metadata `enable_thinking` set to `"true"` or `"false"` overrides this default. +An explicit prompt bypasses tokenizer template rendering. +When no default or request override is set, the template keeps its own behavior. + +LocalAI signals required reasoning when the rendered prompt ends with the configured parser's opening reasoning token. +An explicit output grammar disables this detection. +Configure a reasoning parser that matches your model. + +The backend reads these options when it loads the model. +`POST /models/reload` rereads model configuration files but does not update options in an already loaded backend. +Restarting only the backend does not reread configuration files. +Restart LocalAI after changing these options to reload both the configuration and the backend. + + ### vllm.cpp [vllm.cpp](https://github.com/mudler/vllm.cpp) is the LocalAI team's C++ port of From ea10f26f57e7298b1dbb3cecb313fb2fa514d492 Mon Sep 17 00:00:00 2001 From: Pratik Gandhi Date: Sun, 27 Sep 2026 23:33:15 -0700 Subject: [PATCH 03/15] docs: fix 11 dead links in the documentation (#12320) - compatibility table: voxtral.c, OuteTTS and VoxCPM repos live under antirez, edwko and OpenBMB - distributed inferencing: llama.cpp RPC README moved to tools/rpc under ggml-org - customize-model: the phi-2 example config moved to LocalAI-examples, and embedded/models was replaced by the gallery - model-gallery: malformed URL; link the gallery index - GPU acceleration: ROCm install guide moved - integrations: Wave Terminal docs page moved to ai-presets Assisted-by: Claude:claude-fable-5-1 Signed-off-by: Pratik Gandhi --- docs/content/features/GPU-acceleration.md | 2 +- docs/content/features/distributed_inferencing.md | 2 +- docs/content/features/model-gallery.md | 2 +- docs/content/getting-started/customize-model.md | 8 ++++---- docs/content/integrations.md | 2 +- docs/content/reference/compatibility-table.md | 6 +++--- 6 files changed, 11 insertions(+), 11 deletions(-) diff --git a/docs/content/features/GPU-acceleration.md b/docs/content/features/GPU-acceleration.md index 207860ec6..fb939a47a 100644 --- a/docs/content/features/GPU-acceleration.md +++ b/docs/content/features/GPU-acceleration.md @@ -243,7 +243,7 @@ The devices in the following list have been tested with `hipblas` images. 1. Check your GPU LLVM target is compatible with the version of ROCm. This can be found in the [LLVM Docs](https://llvm.org/docs/AMDGPUUsage.html). 2. Check which ROCm version is compatible with your LLVM target and your chosen OS (pay special attention to supported kernel versions). See the [ROCm compatibility matrix](https://rocm.docs.amd.com/en/latest/compatibility/compatibility-matrix.html). -3. Install your chosen version of the `dkms` and `rocm` (it is recommended that the native package manager be used for this process for any OS as version changes are executed more easily via this method if updates are required). Take care to restart after installing `amdgpu-dkms` and before installing `rocm`, for details regarding this see the [ROCm installation documentation](https://rocm.docs.amd.com/projects/install-on-linux/en/latest/how-to/native-install/index.html). +3. Install your chosen version of the `dkms` and `rocm` (it is recommended that the native package manager be used for this process for any OS as version changes are executed more easily via this method if updates are required). Take care to restart after installing `amdgpu-dkms` and before installing `rocm`, for details regarding this see the [ROCm installation documentation](https://rocm.docs.amd.com/projects/install-on-linux/en/latest/install/install-methods/package-manager-index.html). 4. Deploy. Yes it's that easy. #### Setup Example (Docker/containerd) diff --git a/docs/content/features/distributed_inferencing.md b/docs/content/features/distributed_inferencing.md index f30b97a60..61617bbf4 100644 --- a/docs/content/features/distributed_inferencing.md +++ b/docs/content/features/distributed_inferencing.md @@ -98,7 +98,7 @@ LLAMACPP_GRPC_SERVERS="address1:port,address2:port" local-ai run ``` The workload on the LocalAI server will then be distributed across the specified nodes. -Alternatively, you can build the RPC workers/server following the llama.cpp [README](https://github.com/ggerganov/llama.cpp/blob/master/examples/rpc/README.md), which is compatible with LocalAI. +Alternatively, you can build the RPC workers/server following the llama.cpp [README](https://github.com/ggml-org/llama.cpp/blob/master/tools/rpc/README.md), which is compatible with LocalAI. ## Manual example (worker) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index 0bee9c890..09063fdd7 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -373,7 +373,7 @@ curl $LOCALAI/models/apply -H "Content-Type: application/json" -d '{ where: - `localai` is the repository. It is optional and can be omitted. If the repository is omitted LocalAI will search the model by name in all the repositories. In the case the same model name is present in both galleries the first match wins. - `bert-embeddings` is the model name in the gallery - (read its [config here](https://github.com/mudler/LocalAI/tree/master/gallery/blob/main/bert-embeddings.yaml)). + (read its [config here](https://github.com/mudler/LocalAI/blob/master/gallery/index.yaml)). ### Model variants diff --git a/docs/content/getting-started/customize-model.md b/docs/content/getting-started/customize-model.md index 751a2e6fa..172e6051e 100644 --- a/docs/content/getting-started/customize-model.md +++ b/docs/content/getting-started/customize-model.md @@ -25,17 +25,17 @@ Here's an example to initiate the **phi-2** model: docker run -p 8080:8080 localai/localai:{{< version >}} https://gist.githubusercontent.com/mudler/ad601a0488b497b69ec549150d9edd18/raw/a8a8869ef1bb7e3830bf5c0bae29a0cce991ff8d/phi-2.yaml ``` -You can also check all the embedded models configurations [here](https://github.com/mudler/LocalAI/tree/master/embedded/models). +You can also check all the embedded models configurations [here](https://github.com/mudler/LocalAI/tree/master/gallery). {{% notice tip %}} -The model configurations used in the quickstart are accessible here: [https://github.com/mudler/LocalAI/tree/master/embedded/models](https://github.com/mudler/LocalAI/tree/master/embedded/models). Contributions are welcome; please feel free to submit a Pull Request. +The model configurations used in the quickstart are accessible here: [https://github.com/mudler/LocalAI/tree/master/gallery](https://github.com/mudler/LocalAI/tree/master/gallery). Contributions are welcome; please feel free to submit a Pull Request. -The `phi-2` model configuration from the quickstart is expanded from [https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml](https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml). +The `phi-2` model configuration from the quickstart is expanded from [https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml](https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml). {{% /notice %}} ## Example: Customizing the Prompt Template -To modify the prompt template, create a Github gist or a Pastebin file, and copy the content from [https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml](https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml). Alter the fields as needed: +To modify the prompt template, create a Github gist or a Pastebin file, and copy the content from [https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml](https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml). Alter the fields as needed: ```yaml name: phi-2 diff --git a/docs/content/integrations.md b/docs/content/integrations.md index 8d8bd4104..6fb84a250 100644 --- a/docs/content/integrations.md +++ b/docs/content/integrations.md @@ -98,7 +98,7 @@ availability may lag upstream releases. - [AnythingLLM](https://github.com/Mintplex-Labs/anything-llm) - [Logseq GPT3 OpenAI plugin](https://github.com/briansunter/logseq-plugin-gpt3-openai) - [CodeGPT (JetBrains)](https://plugins.jetbrains.com/plugin/21056-codegpt) - Custom OpenAI-compatible endpoints -- [Wave Terminal](https://docs.waveterm.dev/features/supportedLLMs/localai) - Native LocalAI support +- [Wave Terminal](https://docs.waveterm.dev/ai-presets) - Native LocalAI support - [Obsidian BMO Chatbot](https://github.com/longy2k/obsidian-bmo-chatbot) - [spark](https://github.com/cedriking/spark) - [openops (Mattermost)](https://github.com/mattermost/openops) diff --git a/docs/content/reference/compatibility-table.md b/docs/content/reference/compatibility-table.md index 7e6bfd73d..40c6ae4f3 100644 --- a/docs/content/reference/compatibility-table.md +++ b/docs/content/reference/compatibility-table.md @@ -45,7 +45,7 @@ All backends listed here can be installed on demand from the [Backend Gallery]({ | [moonshine](https://github.com/moonshine-ai/moonshine) | Ultra-fast transcription for low-end devices (ONNX) | CPU, CUDA 12/13, Metal | | [parakeet.cpp](https://github.com/mudler/parakeet.cpp) | C++/GGML port of NVIDIA NeMo Parakeet (tdt/ctc/rnnt/hybrid), with cache-aware streaming | CPU, CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, Jetson L4T | | [CrispASR](https://github.com/CrispStrobe/CrispASR) | Unified speech engine (whisper.cpp fork) supporting Parakeet, Canary, and many ASR architectures, plus TTS | CPU, CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, Jetson L4T | -| [voxtral](https://github.com/mudler/voxtral.c) | Voxtral Realtime 4B speech-to-text in pure C | CPU, Metal | +| [voxtral](https://github.com/antirez/voxtral.c) | Voxtral Realtime 4B speech-to-text in pure C | CPU, Metal | | [Qwen3-ASR](https://github.com/QwenLM/Qwen3-ASR) | Qwen3 automatic speech recognition | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal, Jetson L4T | | [NeMo](https://github.com/NVIDIA/NeMo) | NVIDIA NeMo ASR toolkit | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal | | [sherpa-onnx](https://k2-fsa.github.io/sherpa/onnx/) | Sherpa-ONNX ASR (Whisper, Paraformer, SenseVoice) and TTS | CPU, CUDA 12, Metal | @@ -70,10 +70,10 @@ All backends listed here can be installed on demand from the [Backend Gallery]({ | [OmniVoice](https://github.com/ServeurpersoCom/omnivoice.cpp) | Native C++/GGML TTS with voice cloning, voice design, and streaming | CPU, CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, Jetson L4T | | [fish-speech](https://github.com/fishaudio/fish-speech) | High-quality TTS with voice cloning | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal, Jetson L4T | | [Pocket TTS](https://github.com/kyutai-labs/pocket-tts) | Lightweight CPU-efficient TTS with voice cloning | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal, Jetson L4T | -| [OuteTTS](https://github.com/OuteAI/outetts) | TTS with custom speaker voices | CPU, CUDA 12 | +| [OuteTTS](https://github.com/edwko/OuteTTS) | TTS with custom speaker voices | CPU, CUDA 12 | | [faster-qwen3-tts](https://github.com/andimarafioti/faster-qwen3-tts) | Real-time Qwen3-TTS with CUDA graph capture | CPU, CUDA 12/13, Jetson L4T | | [NeuTTS Air](https://github.com/neuphonic/neutts-air) | Instant voice cloning, on-device TTS | CPU, CUDA 12, ROCm | -| [VoxCPM](https://github.com/ModelBest/VoxCPM) | Expressive end-to-end TTS | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal | +| [VoxCPM](https://github.com/OpenBMB/VoxCPM) | Expressive end-to-end TTS | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal | | [Kitten TTS](https://github.com/KittenML/KittenTTS) | Kitten TTS model | CPU, Metal | | [Supertonic](https://github.com/supertone-inc/supertonic) | Lightning-fast on-device multilingual TTS via ONNX | CPU | | [MLX-Audio](https://github.com/Blaizzy/mlx-audio) | Audio models on Apple Silicon | CPU, CUDA 12/13, Metal, Jetson L4T | From 81348581da1924fef3f24cfad8cf6d9841f95cd1 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Mon, 28 Sep 2026 08:34:31 +0200 Subject: [PATCH 04/15] chore(model-gallery): :arrow_up: update checksum (#12310) :arrow_up: Checksum updates in gallery/index.yaml Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- gallery/index.yaml | 11 ++++------- 1 file changed, 4 insertions(+), 7 deletions(-) diff --git a/gallery/index.yaml b/gallery/index.yaml index f6fa9361c..c2298226b 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -297,7 +297,7 @@ files: - filename: ds4flash.gguf uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF - sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf + sha256: f33633d55f5379e8571db06674bf7a07a2ea7bb7b44287e1a9bdc3686d69a5c6 - name: "qwopus3.8-27b-flash-v2" variants: - model: qwopus3.8-27b-flash-v2-q8 @@ -6420,7 +6420,6 @@ - filename: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q4_K_XL.gguf sha256: 8e5601dbd18fbc2b731cf674a040dd32f3ec2d09a312f4e0f3c4d7bc92998837 - - name: sharp-spark-x2.5-4b-q5 url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: @@ -6457,7 +6456,6 @@ - filename: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q5_K_XL.gguf sha256: f445f1a57e58b70ea85078e1edcd29763843f71f154bac2efc57eea1b8333a26 - - name: sharp-spark-x2.5-4b-q6 url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: @@ -6494,7 +6492,6 @@ - filename: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q6_K_XL.gguf sha256: 793e673f34d2dde9674d24d277c25dbf03b89290333835aa31b7ee1d62e20dfc - - &spark-x2-5-4b name: "spark-x2.5-4b-q4" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" @@ -9309,7 +9306,7 @@ files: - filename: llama-cpp/models/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-APEX/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-APEX.gguf uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-APEX.gguf - sha256: 35026b978de6ee6ff63870d3d68be90ab0797a9a6cd5c83332e1f9c510c6a695 + sha256: 97602082e8639b1e6660b36de0193861742e035711b865faf76abf54a5a2be09 - filename: llama-cpp/mmproj/Hermes3.6-35B-A3B-Uncensored-Genesis-Final/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf sha256: 5129bb5eb19e4346c0f2071f1ce8e1b0a076e0ab08d57a77d6033ce01235252c uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf @@ -9399,7 +9396,7 @@ files: - filename: llama-cpp/models/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX.gguf uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX.gguf - sha256: 612561952f0698539a133a479e1cf18ffce85bb6b4311a5d05e50d6160970c75 + sha256: 3efbc83f38ffa48251ff4c072fbbefce63c995df3cba20fd6bf0d8d7ab965cd9 - filename: llama-cpp/mmproj/Hermes3.6-35B-A3B-Uncensored-Genesis-Final/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf sha256: 5129bb5eb19e4346c0f2071f1ce8e1b0a076e0ab08d57a77d6033ce01235252c uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf @@ -9451,7 +9448,7 @@ files: - filename: llama-cpp/models/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX-Compact/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX-Compact.gguf uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX-Compact.gguf - sha256: 7a17aaff5ec81ba34e33d3a8b032a699abf4231d6cbe9930d1c9a159db3e87dc + sha256: 27a2edec6f66e585fbf02eabe397f694378474786711362d9f78cd129a0e8313 - filename: llama-cpp/mmproj/Hermes3.6-35B-A3B-Uncensored-Genesis-Final/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf sha256: 5129bb5eb19e4346c0f2071f1ce8e1b0a076e0ab08d57a77d6033ce01235252c uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf From 97ad8f1d7033bc88d4a0fdaead721a7008c96929 Mon Sep 17 00:00:00 2001 From: mudler-agent Date: Mon, 28 Sep 2026 08:34:56 +0200 Subject: [PATCH 05/15] fix(distributed): stage the files a model install declares (#12309) * fix(distributed): stage every shard of a split GGUF A split GGUF is configured by its first shard only. llama.cpp opens the other "-0000N-of-0000M.gguf" files from the same directory by name. The router staged only the configured path, so the worker received shard 1 and the load failed with "failed to load GGUF split". The router now stages the remaining shards next to the first one. A missing shard fails the load and names the file. The file count for progress and the payload size also include all shards. The payload size feeds the load deadline and the disk-headroom check. For a 111 GB model whose first shard is 10 MB, both were sized for less than 1 GB. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(distributed): stage the files a model install declares Replace the split GGUF file name matching with the model's own file list. A gallery install or an import records every file of the model in ._gallery_.yaml (files:), and a config can list more under download_files:. The router now stages all of these files, not only the files that the config's path fields name. This includes the other shards of a split GGUF, which llama.cpp opens by name. The application gives the router a resolver that reads the two lists. The resolver looks up the files by model name when it stages them, so a replica that the reconciler loads from saved load options gets the same files. backend.proto does not change. A declared file that is missing on the frontend is skipped with a warning. The load deadline and the disk headroom check include the declared files. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto --------- Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto --- core/application/distributed.go | 5 + core/application/model_files.go | 29 ++++ core/application/model_files_test.go | 41 ++++++ core/gallery/installed_files.go | 46 +++++++ core/gallery/installed_files_test.go | 53 ++++++++ core/services/nodes/declared_files.go | 78 +++++++++++ core/services/nodes/router.go | 34 ++++- .../nodes/router_declared_files_stage_test.go | 126 ++++++++++++++++++ docs/content/features/distributed-mode.md | 13 ++ 9 files changed, 423 insertions(+), 2 deletions(-) create mode 100644 core/application/model_files.go create mode 100644 core/application/model_files_test.go create mode 100644 core/gallery/installed_files.go create mode 100644 core/gallery/installed_files_test.go create mode 100644 core/services/nodes/declared_files.go create mode 100644 core/services/nodes/router_declared_files_stage_test.go diff --git a/core/application/distributed.go b/core/application/distributed.go index 4c6214e57..bcafcb26e 100644 --- a/core/application/distributed.go +++ b/core/application/distributed.go @@ -379,9 +379,13 @@ func initDistributed(cfg *config.ApplicationConfig, authDB *gorm.DB, configLoade // All dependencies ready — build SmartRouter with all options at once var conflictResolver nodes.ConcurrencyConflictResolver var pinnedResolver nodes.PinnedModelResolver + var modelFiles func(string) []string if configLoader != nil { conflictResolver = configLoader pinnedResolver = configLoader + if cfg.SystemState != nil { + modelFiles = declaredModelFiles(configLoader, cfg.SystemState.Model.ModelsPath) + } } modelCleanup := nodes.NewModelCleanupService(registry, remoteUnloader) router := nodes.NewSmartRouter(registry, nodes.SmartRouterOptions{ @@ -393,6 +397,7 @@ func initDistributed(cfg *config.ApplicationConfig, authDB *gorm.DB, configLoade DB: authDB, ConflictResolver: conflictResolver, PinnedResolver: pinnedResolver, + ModelFiles: modelFiles, PrefixProvider: prefixProvider, PrefixConfig: prefixCfg, Pressure: pressure, diff --git a/core/application/model_files.go b/core/application/model_files.go new file mode 100644 index 000000000..53bd069df --- /dev/null +++ b/core/application/model_files.go @@ -0,0 +1,29 @@ +package application + +import ( + "path/filepath" + + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/gallery" + "github.com/mudler/LocalAI/pkg/utils" +) + +// declaredModelFiles resolves the files a model needs on disk beyond the ones +// its config names: what its gallery install or import declared under +// `files:`, and what the config itself lists under download_files. The +// distributed router stages these to workers, which cannot see the frontend's +// models directory. +func declaredModelFiles(configLoader *config.ModelConfigLoader, modelsPath string) func(modelName string) []string { + return func(modelName string) []string { + files := gallery.InstalledModelFiles(modelsPath, modelName) + if cfg, ok := configLoader.GetModelConfig(modelName); ok { + for _, f := range cfg.DownloadFiles { + if utils.VerifyPath(f.Filename, modelsPath) != nil { + continue + } + files = append(files, filepath.Join(modelsPath, f.Filename)) + } + } + return files + } +} diff --git a/core/application/model_files_test.go b/core/application/model_files_test.go new file mode 100644 index 000000000..f0ab2bf5e --- /dev/null +++ b/core/application/model_files_test.go @@ -0,0 +1,41 @@ +package application + +import ( + "os" + "path/filepath" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/gallery" +) + +var _ = Describe("declaredModelFiles", func() { + It("combines the gallery install's files with the config's download_files", func() { + modelsPath := GinkgoT().TempDir() + Expect(os.WriteFile(filepath.Join(modelsPath, "big.yaml"), []byte(` +name: big +backend: llama-cpp +parameters: + model: big/Big-00001-of-00002.gguf +download_files: + - filename: big/extra.bin + uri: https://example.com/extra.bin +`), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("big")), []byte(` +files: + - filename: big/Big-00001-of-00002.gguf + - filename: big/Big-00002-of-00002.gguf +`), 0o644)).To(Succeed()) + + loader := config.NewModelConfigLoader(modelsPath) + Expect(loader.LoadModelConfigsFromPath(modelsPath)).To(Succeed()) + + Expect(declaredModelFiles(loader, modelsPath)("big")).To(ConsistOf( + filepath.Join(modelsPath, "big/Big-00001-of-00002.gguf"), + filepath.Join(modelsPath, "big/Big-00002-of-00002.gguf"), + filepath.Join(modelsPath, "big/extra.bin"), + )) + }) +}) diff --git a/core/gallery/installed_files.go b/core/gallery/installed_files.go new file mode 100644 index 000000000..c7f6773dc --- /dev/null +++ b/core/gallery/installed_files.go @@ -0,0 +1,46 @@ +package gallery + +import ( + "os" + "path/filepath" + "strings" + + "github.com/mudler/LocalAI/pkg/utils" + "github.com/mudler/xlog" +) + +// InstalledModelFiles returns the absolute paths of the files that the install +// of model name declared (the entry's `files:`), as recorded in its gallery +// file. A model config names only the file a backend opens first, while a +// backend can read more by itself (llama.cpp opens the other shards of a split +// GGUF by name), so this is the complete list of what the model needs on disk. +// It returns nil for a model that was not installed from a gallery or import. +func InstalledModelFiles(modelsPath, name string) []string { + // Model names can hold path separators; the gallery file flattens them + // the same way listModelFiles does. + rel := galleryFileName(strings.ReplaceAll(name, string(os.PathSeparator), "__")) + if err := utils.VerifyPath(rel, modelsPath); err != nil { + return nil + } + galleryFile := filepath.Join(modelsPath, rel) + if _, err := os.Stat(galleryFile); err != nil { + return nil + } + cfg, err := ReadConfigFile[ModelConfig](galleryFile) + if err != nil { + xlog.Warn("Failed to read gallery file for installed model files", "model", name, "file", galleryFile, "error", err) + return nil + } + + files := make([]string, 0, len(cfg.Files)) + for _, f := range cfg.Files { + // VerifyPath joins its argument onto modelsPath itself, so it must + // get the relative name; an absolute path would always pass. + if err := utils.VerifyPath(f.Filename, modelsPath); err != nil { + xlog.Warn("Ignoring declared model file outside the models path", "model", name, "file", f.Filename) + continue + } + files = append(files, filepath.Join(modelsPath, f.Filename)) + } + return files +} diff --git a/core/gallery/installed_files_test.go b/core/gallery/installed_files_test.go new file mode 100644 index 000000000..d376ef1e6 --- /dev/null +++ b/core/gallery/installed_files_test.go @@ -0,0 +1,53 @@ +package gallery_test + +import ( + "os" + "path/filepath" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + "github.com/mudler/LocalAI/core/gallery" +) + +var _ = Describe("InstalledModelFiles", func() { + var modelsPath string + + BeforeEach(func() { + modelsPath = GinkgoT().TempDir() + }) + + writeGalleryFile := func(name, body string) { + Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName(name)), []byte(body), 0o644)).To(Succeed()) + } + + It("returns the files the install declared, under the models path", func() { + writeGalleryFile("big", ` +name: big +files: + - filename: llama-cpp/models/big/Big-00001-of-00002.gguf + uri: huggingface://org/repo/Big-00001-of-00002.gguf + - filename: llama-cpp/models/big/Big-00002-of-00002.gguf + uri: huggingface://org/repo/Big-00002-of-00002.gguf +`) + Expect(gallery.InstalledModelFiles(modelsPath, "big")).To(Equal([]string{ + filepath.Join(modelsPath, "llama-cpp/models/big/Big-00001-of-00002.gguf"), + filepath.Join(modelsPath, "llama-cpp/models/big/Big-00002-of-00002.gguf"), + })) + }) + + It("drops entries that escape the models path", func() { + writeGalleryFile("evil", ` +files: + - filename: ../outside.gguf + - filename: ok.gguf +`) + Expect(gallery.InstalledModelFiles(modelsPath, "evil")).To(Equal([]string{ + filepath.Join(modelsPath, "ok.gguf"), + })) + }) + + It("returns nothing for a model that was not installed from a gallery", func() { + Expect(gallery.InstalledModelFiles(modelsPath, "handwritten")).To(BeEmpty()) + }) +}) diff --git a/core/services/nodes/declared_files.go b/core/services/nodes/declared_files.go new file mode 100644 index 000000000..4df3b000a --- /dev/null +++ b/core/services/nodes/declared_files.go @@ -0,0 +1,78 @@ +package nodes + +import ( + "os" + "path/filepath" + "strings" + + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + "github.com/mudler/xlog" +) + +// declaredExtraFiles returns the files the model's install declared that the +// path fields of opts do not already stage: neither named by a field nor +// inside a directory a field names. It must run on the local paths, before +// staging rewrites the fields to remote ones. +func (r *SmartRouter) declaredExtraFiles(trackingKey string, opts *pb.ModelOptions) []string { + if r.modelFiles == nil || opts == nil || trackingKey == "" { + return nil + } + covered := append([]string{ + opts.ModelFile, opts.MMProj, opts.LoraAdapter, opts.DraftModel, + opts.CLIPModel, opts.Tokenizer, opts.AudioPath, opts.LoraBase, + }, opts.LoraAdapters...) + + seen := map[string]struct{}{} + var extra []string + for _, p := range r.modelFiles(trackingKey) { + p = filepath.Clean(p) + if _, dup := seen[p]; dup || coveredByField(p, covered) { + continue + } + seen[p] = struct{}{} + extra = append(extra, p) + } + return extra +} + +func coveredByField(path string, fields []string) bool { + for _, f := range fields { + if f == "" { + continue + } + f = filepath.Clean(f) + if path == f || strings.HasPrefix(path, f+string(filepath.Separator)) { + return true + } + } + return false +} + +// existingFiles drops declared files that are not on the frontend. An install +// can declare files that are gone by load time (an archive unpacked and then +// removed, say), so a missing one is not a reason to refuse the load; the +// backend reports it if it really needed it. +func existingFiles(paths []string, nodeName, trackingKey string) []string { + out := paths[:0:0] + for _, p := range paths { + if _, err := os.Stat(p); err != nil { + xlog.Warn("Skipping staging for declared model file that is not on the frontend", "path", p, "node", nodeName, "model", trackingKey, "error", err) + continue + } + out = append(out, p) + } + return out +} + +// stagingPayloadBytes totals the on-disk size of everything staging uploads +// for a model: the path fields plus the declared files they do not cover. The +// first shard of a split GGUF can be a few MB of metadata while the weights +// sit in the others, so sizing the fields alone starves the load budget and +// the disk-headroom check. +func (r *SmartRouter) stagingPayloadBytes(trackingKey string, opts *pb.ModelOptions) int64 { + total := modelPayloadBytes(opts) + for _, p := range r.declaredExtraFiles(trackingKey, opts) { + total += pathBytes(p) + } + return total +} diff --git a/core/services/nodes/router.go b/core/services/nodes/router.go index 36eee4a47..84cad29a6 100644 --- a/core/services/nodes/router.go +++ b/core/services/nodes/router.go @@ -59,6 +59,12 @@ type SmartRouterOptions struct { // nil disables the exclusion. Deliberate teardown (UnloadModel, admin // endpoints, node drain) is unaffected. PinnedResolver PinnedModelResolver + // ModelFiles, when set, returns the absolute local paths of every file a + // model's install declared (gallery `files:`, config `download_files`). + // The path fields of a load request name only what the backend opens + // first; this is how staging learns about the rest, such as the other + // shards of a split GGUF. nil stages the path fields alone. + ModelFiles func(modelName string) []string // PrefixProvider, when set, enables prefix-cache-aware routing: requests // carrying a prompt prefix chain (distributedhdr.PrefixChain) are biased // toward the node that already holds the longest matching prefix, subject @@ -170,6 +176,9 @@ type SmartRouter struct { // pinnedResolver feeds the eviction paths the set of pinned model names // (see SmartRouterOptions.PinnedResolver). nil disables the exclusion. pinnedResolver PinnedModelResolver + // modelFiles resolves a model's declared files (see + // SmartRouterOptions.ModelFiles). nil stages the path fields alone. + modelFiles func(modelName string) []string // prefixProvider is the prefix-cache routing seam (nil disables it; see // SmartRouterOptions.PrefixProvider). prefixConfig holds the global policy // and thresholds. @@ -254,6 +263,7 @@ func NewSmartRouter(registry ModelRouter, opts SmartRouterOptions) *SmartRouter stagingTracker: NewStagingTracker(), conflictResolver: opts.ConflictResolver, pinnedResolver: opts.PinnedResolver, + modelFiles: opts.ModelFiles, probeCache: newProbeCache(probeCacheTTL), prefixProvider: opts.PrefixProvider, prefixConfig: opts.PrefixConfig, @@ -382,7 +392,7 @@ func (r *SmartRouter) scheduleAndLoad(ctx context.Context, backendType, tracking // Size the remote load budget BEFORE staging: stageModelFiles rewrites the // path fields to their remote equivalents on a clone, and only the local // paths can be stat'ed here. - payloadBytes := modelPayloadBytes(modelOpts) + payloadBytes := r.stagingPayloadBytes(trackingKey, modelOpts) loadTimeout := r.loadTimeoutFor(payloadBytes) // Pre-stage model files via FileStager before loading @@ -1261,7 +1271,7 @@ func (r *SmartRouter) narrowByDiskHeadroom(ctx context.Context, modelID string, return candidateNodeIDs, nil } - requiredDisk := DiskRequirementFor(modelPayloadBytes(modelOpts)) + requiredDisk := DiskRequirementFor(r.stagingPayloadBytes(modelID, modelOpts)) diskCandidates, diskErr := r.registry.NarrowByDiskHeadroom(ctx, candidateNodeIDs, requiredDisk) // The check runs even when disabled. "Disabled" means do not BLOCK, not do @@ -1435,6 +1445,10 @@ func (r *SmartRouter) stageModelFiles(ctx context.Context, node *BackendNode, op localModelDir = filepath.Dir(opts.ModelFile) } + // Resolved before the path fields are rewritten to remote paths below, + // since that is what tells which declared files the fields already cover. + declared := existingFiles(r.declaredExtraFiles(trackingKey, opts), node.Name, trackingKey) + // keyMapper generates storage keys namespaced under trackingKey, preserving // subdirectory structure relative to frontendModelsDir. This ensures: // 1. All files for a model land in one directory on the worker for clean deletion @@ -1480,6 +1494,7 @@ func (r *SmartRouter) stageModelFiles(ctx context.Context, node *BackendNode, op totalFiles++ } } + totalFiles += len(declared) // Start tracking staging progress r.stagingTracker.Start(trackingKey, node.Name, totalFiles) @@ -1610,6 +1625,21 @@ func (r *SmartRouter) stageModelFiles(ctx context.Context, node *BackendNode, op } } + for _, localPath := range declared { + fileIdx++ + fileName := filepath.Base(localPath) + stageCtx := r.withStagingCallback(ctx, trackingKey, fileName, fileIdx, totalFiles) + + xlog.Info("Staging declared model file", "model", trackingKey, "node", node.Name, "file", fileName, "fileIndex", fileIdx, "totalFiles", totalFiles) + if _, err := r.fileStager.EnsureRemote(stageCtx, node.ID, localPath, keyMapper.Key(localPath)); err != nil { + // The install declared it, so the backend may read it: loading + // without it fails later with a less useful error. + xlog.Error("Failed to stage declared model file for remote node", "node", node.Name, "path", localPath, "error", err) + return nil, fmt.Errorf("staging declared model file %s: %w", localPath, err) + } + r.stagingTracker.FileComplete(trackingKey, fileIdx, totalFiles) + } + // Stage file paths referenced in generic Options (key:value pairs where values // are file paths). Options stay as relative paths — backends resolve them via ModelPath. for _, options := range [][]string{opts.Options, opts.Overrides} { diff --git a/core/services/nodes/router_declared_files_stage_test.go b/core/services/nodes/router_declared_files_stage_test.go new file mode 100644 index 000000000..e1c00e8c5 --- /dev/null +++ b/core/services/nodes/router_declared_files_stage_test.go @@ -0,0 +1,126 @@ +package nodes + +import ( + "context" + "os" + "path/filepath" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + pb "github.com/mudler/LocalAI/pkg/grpc/proto" +) + +// A model's config names only the file the backend opens first, but its +// install can declare more that the backend reads by itself: llama.cpp opens +// the "-0000N-of-0000M" shards of a split GGUF from the directory of the first +// one. The worker has no view of the frontend's models directory, so every +// declared file must be staged, or the load fails with "failed to load GGUF +// split". +var _ = Describe("stageModelFiles declared model files", func() { + var ( + stager *fakeFileStager + router *SmartRouter + node *BackendNode + modelDir string + shards []string + mmproj string + declared map[string][]string + ) + + BeforeEach(func() { + stager = &fakeFileStager{} + declared = map[string][]string{} + router = &SmartRouter{ + fileStager: stager, + stagingTracker: NewStagingTracker(), + modelFiles: func(name string) []string { return declared[name] }, + } + node = &BackendNode{ID: "node-1", Name: "node-1", Address: "10.0.0.1:50051"} + root := GinkgoT().TempDir() + modelDir = filepath.Join(root, "llama-cpp", "models", "big") + Expect(os.MkdirAll(modelDir, 0o755)).To(Succeed()) + + shards = nil + for _, name := range []string{ + "Big-Q4_K_M-00001-of-00003.gguf", + "Big-Q4_K_M-00002-of-00003.gguf", + "Big-Q4_K_M-00003-of-00003.gguf", + } { + p := filepath.Join(modelDir, name) + Expect(os.WriteFile(p, []byte("shard "+name), 0o644)).To(Succeed()) + shards = append(shards, p) + } + mmproj = filepath.Join(root, "llama-cpp", "mmproj", "big", "mmproj.gguf") + Expect(os.MkdirAll(filepath.Dir(mmproj), 0o755)).To(Succeed()) + Expect(os.WriteFile(mmproj, []byte("mmproj"), 0o644)).To(Succeed()) + }) + + opts := func() *pb.ModelOptions { + return &pb.ModelOptions{ + Model: "llama-cpp/models/big/Big-Q4_K_M-00001-of-00003.gguf", + ModelFile: shards[0], + MMProj: mmproj, + } + } + + stagedPaths := func() []string { + out := make([]string, 0, len(stager.ensureCalls)) + for _, c := range stager.ensureCalls { + out = append(out, c.localPath) + } + return out + } + + It("stages every declared file once, beside the ones the config names", func() { + declared["big"] = append(append([]string{}, shards...), mmproj) + + staged, err := router.stageModelFiles(context.Background(), node, opts(), "big") + Expect(err).ToNot(HaveOccurred()) + Expect(stagedPaths()).To(ConsistOf(shards[0], mmproj, shards[1], shards[2])) + + // llama.cpp derives the other shards' paths from the first one, so + // they must land in the same remote directory. + for _, c := range stager.ensureCalls { + if c.localPath != mmproj { + Expect(filepath.Dir(c.key)).To(Equal(filepath.Dir(stager.ensureCalls[0].key))) + } + } + Expect(staged.ModelFile).To(Equal("/remote/" + stager.ensureCalls[0].key)) + }) + + It("sizes declared files for the load budget and disk check", func() { + declared["big"] = append(append([]string{}, shards...), mmproj) + + var want int64 + for _, p := range append(append([]string{}, shards...), mmproj) { + fi, err := os.Stat(p) + Expect(err).ToNot(HaveOccurred()) + want += fi.Size() + } + Expect(router.stagingPayloadBytes("big", opts())).To(Equal(want)) + }) + + It("skips a declared file that is missing locally instead of failing", func() { + declared["big"] = append(append([]string{}, shards...), filepath.Join(modelDir, "gone.bin")) + + _, err := router.stageModelFiles(context.Background(), node, opts(), "big") + Expect(err).ToNot(HaveOccurred()) + Expect(stagedPaths()).To(ConsistOf(shards[0], mmproj, shards[1], shards[2])) + }) + + It("does not stage a declared file twice when a directory field covers it", func() { + declared["dir"] = []string{shards[1]} + + _, err := router.stageModelFiles(context.Background(), node, + &pb.ModelOptions{Model: "llama-cpp/models/big", ModelFile: modelDir}, "dir") + Expect(err).ToNot(HaveOccurred()) + Expect(stagedPaths()).To(ConsistOf(shards[0], shards[1], shards[2])) + }) + + It("stages only the named files for a model that declares none", func() { + _, err := router.stageModelFiles(context.Background(), node, opts(), "handwritten") + Expect(err).ToNot(HaveOccurred()) + Expect(stagedPaths()).To(ConsistOf(shards[0], mmproj)) + }) +}) diff --git a/docs/content/features/distributed-mode.md b/docs/content/features/distributed-mode.md index 86ad8b14e..07d1d7378 100644 --- a/docs/content/features/distributed-mode.md +++ b/docs/content/features/distributed-mode.md @@ -224,6 +224,19 @@ Set `LOCALAI_DISTRIBUTED_SHARED_MODELS=true` (or `--distributed-shared-models`) This flag is a contract you assert: all nodes must mount identical paths. Leave it off (the default) when workers have independent models directories - the frontend stages files to them over HTTP (or S3) as described above. +### Which files are staged + +The frontend stages the files that the model config names (`parameters.model`, `mmproj`, draft model, LoRA adapters and similar fields). It also stages every other file that the model declares: + +- The `files:` of the gallery entry or `/import-model` import that installed the model. LocalAI records these in `._gallery_.yaml` next to the model config. +- The `download_files:` of the model config. + +A backend can read files that the config does not name. For example, llama.cpp opens all shards of a split GGUF (`-00002-of-00004.gguf` and the rest) from the directory of the first shard. The worker cannot see the frontend's models directory, so it gets only the files that the frontend stages. + +If you write a model config by hand and the model has files like these, list them under `download_files:`. If you do not, the worker gets only the first shard and the load fails with `failed to load GGUF split`. + +The file sizes used for the load deadline and for the disk headroom check include all of these files. + ### Model artifact staging For managed Hugging Face artifacts, the controller resolves the repository and From 10f3cd80150bcd1aee4d04080cc4e954f7df8dc2 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Mon, 28 Sep 2026 08:35:19 +0200 Subject: [PATCH 06/15] feat(swagger): update swagger (#12308) Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- swagger/docs.go | 280 ++++++++++++++++++++++++++++++++++++++++++- swagger/swagger.json | 280 ++++++++++++++++++++++++++++++++++++++++++- swagger/swagger.yaml | 197 +++++++++++++++++++++++++++++- 3 files changed, 754 insertions(+), 3 deletions(-) diff --git a/swagger/docs.go b/swagger/docs.go index c447043ae..0950fdb6c 100644 --- a/swagger/docs.go +++ b/swagger/docs.go @@ -3770,6 +3770,90 @@ const docTemplate = `{ } } }, + "/v1/systemone": { + "post": { + "description": "Runs zero-shot NER over the supplied state and answers each question. Question types: noul (binary entity presence), choice (pick one option), score (pick one level).", + "tags": [ + "systemone" + ], + "summary": "Answer structured-extraction questions over state text.", + "parameters": [ + { + "description": "state + questions", + "name": "request", + "in": "body", + "required": true, + "schema": { + "$ref": "#/definitions/schema.SystemOneRequest" + } + } + ], + "responses": { + "200": { + "description": "OK", + "schema": { + "$ref": "#/definitions/schema.SystemOneResponse" + } + } + } + } + }, + "/v1/systemone/permute": { + "post": { + "description": "Re-runs one choice question under n_perm option orders. Reports per-order probabilities, argmax stability, and spread.", + "tags": [ + "systemone" + ], + "summary": "Re-run a choice question under multiple option orders.", + "parameters": [ + { + "description": "request + question + n_perm + seed", + "name": "request", + "in": "body", + "required": true, + "schema": { + "$ref": "#/definitions/schema.SystemOnePermuteRequest" + } + } + ], + "responses": { + "200": { + "description": "OK", + "schema": { + "$ref": "#/definitions/schema.SystemOnePermuteResponse" + } + } + } + } + }, + "/v1/systemone/separate": { + "post": { + "description": "Runs N independent NER passes, one per question, against the same state. Response shape matches /v1/systemone.", + "tags": [ + "systemone" + ], + "summary": "Answer each question in a separate NER pass.", + "parameters": [ + { + "description": "state + questions", + "name": "request", + "in": "body", + "required": true, + "schema": { + "$ref": "#/definitions/schema.SystemOneRequest" + } + } + ], + "responses": { + "200": { + "description": "OK", + "schema": { + "$ref": "#/definitions/schema.SystemOneResponse" + } + } + } + } + }, "/v1/text-to-speech/{voice-id}": { "post": { "tags": [ @@ -4093,8 +4177,16 @@ const docTemplate = `{ "config.Gallery": { "type": "object", "properties": { + "artifact_verification": { + "description": "ArtifactVerification overrides Verification only for the gallery OCI artifact.\nBackend images keep their separate Verification policy.", + "allOf": [ + { + "$ref": "#/definitions/config.GalleryVerification" + } + ] + }, "mirrors": { - "description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://).", + "description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://, oci://).", "type": "array", "items": { "type": "string" @@ -4129,6 +4221,10 @@ const docTemplate = `{ "not_before": { "description": "NotBefore is an RFC3339 timestamp. Empty disables the time check.", "type": "string" + }, + "source_repository": { + "description": "SourceRepository is an https URL compared exactly against the\ncertificate's source-repository extension. Empty skips the check.", + "type": "string" } } }, @@ -7811,12 +7907,42 @@ const docTemplate = `{ "id": { "type": "string" }, + "process": { + "description": "Process is the backend process serving the model on this host. Absent\nwhen the model has no local process (a distributed worker holds it) or\nthe process could not be read.", + "allOf": [ + { + "$ref": "#/definitions/schema.SysInfoProcess" + } + ] + }, "size_vram": { "description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.", "type": "integer" } } }, + "schema.SysInfoProcess": { + "type": "object", + "properties": { + "cpu_percent": { + "description": "CPUPercent is the share of the whole host's CPU used since the previous\nreading, 0-100. Absent on the first reading of a process.", + "type": "number" + }, + "memory_percent": { + "type": "number" + }, + "pid": { + "type": "integer" + }, + "rss_bytes": { + "description": "RSSBytes is resident host memory. Weights offloaded to a GPU are not\nin it.", + "type": "integer" + }, + "started_at": { + "type": "string" + } + } + }, "schema.SystemInformationResponse": { "type": "object", "properties": { @@ -7836,6 +7962,158 @@ const docTemplate = `{ } } }, + "schema.SystemOneAnswer": { + "type": "object", + "properties": { + "choice": { + "type": "string" + }, + "confidence": { + "type": "number" + }, + "entities": { + "type": "array", + "items": { + "$ref": "#/definitions/schema.SystemOneEntity" + } + }, + "legend": { + "type": "object", + "additionalProperties": { + "type": "string" + } + }, + "noul": { + "type": "number" + }, + "probabilities": { + "type": "object", + "additionalProperties": { + "type": "number", + "format": "float64" + } + }, + "score": { + "type": "number" + }, + "type": { + "type": "string" + } + } + }, + "schema.SystemOneEntity": { + "type": "object", + "properties": { + "confidence": { + "type": "number" + }, + "end": { + "type": "integer" + }, + "start": { + "type": "integer" + }, + "text": { + "type": "string" + } + } + }, + "schema.SystemOnePermuteRequest": { + "type": "object", + "properties": { + "n_perm": { + "type": "integer" + }, + "question": { + "type": "string" + }, + "request": { + "$ref": "#/definitions/schema.SystemOneRequest" + }, + "seed": { + "type": "integer" + } + } + }, + "schema.SystemOnePermuteResponse": { + "type": "object", + "properties": { + "argmax_stable": { + "type": "boolean" + }, + "runs": { + "type": "array", + "items": { + "$ref": "#/definitions/schema.SystemOnePermuteRun" + } + }, + "spread": { + "type": "object", + "additionalProperties": { + "type": "number", + "format": "float64" + } + } + } + }, + "schema.SystemOnePermuteRun": { + "type": "object", + "properties": { + "choice": { + "type": "string" + }, + "latency_ms": { + "type": "number" + }, + "order": { + "type": "array", + "items": { + "type": "string" + } + }, + "probabilities": { + "type": "object", + "additionalProperties": { + "type": "number", + "format": "float64" + } + } + } + }, + "schema.SystemOneRequest": { + "type": "object" + }, + "schema.SystemOneResponse": { + "type": "object", + "properties": { + "answers": { + "type": "object", + "additionalProperties": { + "$ref": "#/definitions/schema.SystemOneAnswer" + } + }, + "latency_ms": { + "type": "number" + }, + "model": { + "type": "string" + }, + "usage": { + "$ref": "#/definitions/schema.SystemOneUsage" + } + } + }, + "schema.SystemOneUsage": { + "type": "object", + "properties": { + "input_tokens": { + "type": "integer" + }, + "output_tokens": { + "type": "integer" + } + } + }, "schema.TTSRequest": { "description": "TTS request body", "type": "object", diff --git a/swagger/swagger.json b/swagger/swagger.json index b4e49b347..06d6c2c8d 100644 --- a/swagger/swagger.json +++ b/swagger/swagger.json @@ -3767,6 +3767,90 @@ } } }, + "/v1/systemone": { + "post": { + "description": "Runs zero-shot NER over the supplied state and answers each question. Question types: noul (binary entity presence), choice (pick one option), score (pick one level).", + "tags": [ + "systemone" + ], + "summary": "Answer structured-extraction questions over state text.", + "parameters": [ + { + "description": "state + questions", + "name": "request", + "in": "body", + "required": true, + "schema": { + "$ref": "#/definitions/schema.SystemOneRequest" + } + } + ], + "responses": { + "200": { + "description": "OK", + "schema": { + "$ref": "#/definitions/schema.SystemOneResponse" + } + } + } + } + }, + "/v1/systemone/permute": { + "post": { + "description": "Re-runs one choice question under n_perm option orders. Reports per-order probabilities, argmax stability, and spread.", + "tags": [ + "systemone" + ], + "summary": "Re-run a choice question under multiple option orders.", + "parameters": [ + { + "description": "request + question + n_perm + seed", + "name": "request", + "in": "body", + "required": true, + "schema": { + "$ref": "#/definitions/schema.SystemOnePermuteRequest" + } + } + ], + "responses": { + "200": { + "description": "OK", + "schema": { + "$ref": "#/definitions/schema.SystemOnePermuteResponse" + } + } + } + } + }, + "/v1/systemone/separate": { + "post": { + "description": "Runs N independent NER passes, one per question, against the same state. Response shape matches /v1/systemone.", + "tags": [ + "systemone" + ], + "summary": "Answer each question in a separate NER pass.", + "parameters": [ + { + "description": "state + questions", + "name": "request", + "in": "body", + "required": true, + "schema": { + "$ref": "#/definitions/schema.SystemOneRequest" + } + } + ], + "responses": { + "200": { + "description": "OK", + "schema": { + "$ref": "#/definitions/schema.SystemOneResponse" + } + } + } + } + }, "/v1/text-to-speech/{voice-id}": { "post": { "tags": [ @@ -4090,8 +4174,16 @@ "config.Gallery": { "type": "object", "properties": { + "artifact_verification": { + "description": "ArtifactVerification overrides Verification only for the gallery OCI artifact.\nBackend images keep their separate Verification policy.", + "allOf": [ + { + "$ref": "#/definitions/config.GalleryVerification" + } + ] + }, "mirrors": { - "description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://).", + "description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://, oci://).", "type": "array", "items": { "type": "string" @@ -4126,6 +4218,10 @@ "not_before": { "description": "NotBefore is an RFC3339 timestamp. Empty disables the time check.", "type": "string" + }, + "source_repository": { + "description": "SourceRepository is an https URL compared exactly against the\ncertificate's source-repository extension. Empty skips the check.", + "type": "string" } } }, @@ -7808,12 +7904,42 @@ "id": { "type": "string" }, + "process": { + "description": "Process is the backend process serving the model on this host. Absent\nwhen the model has no local process (a distributed worker holds it) or\nthe process could not be read.", + "allOf": [ + { + "$ref": "#/definitions/schema.SysInfoProcess" + } + ] + }, "size_vram": { "description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.", "type": "integer" } } }, + "schema.SysInfoProcess": { + "type": "object", + "properties": { + "cpu_percent": { + "description": "CPUPercent is the share of the whole host's CPU used since the previous\nreading, 0-100. Absent on the first reading of a process.", + "type": "number" + }, + "memory_percent": { + "type": "number" + }, + "pid": { + "type": "integer" + }, + "rss_bytes": { + "description": "RSSBytes is resident host memory. Weights offloaded to a GPU are not\nin it.", + "type": "integer" + }, + "started_at": { + "type": "string" + } + } + }, "schema.SystemInformationResponse": { "type": "object", "properties": { @@ -7833,6 +7959,158 @@ } } }, + "schema.SystemOneAnswer": { + "type": "object", + "properties": { + "choice": { + "type": "string" + }, + "confidence": { + "type": "number" + }, + "entities": { + "type": "array", + "items": { + "$ref": "#/definitions/schema.SystemOneEntity" + } + }, + "legend": { + "type": "object", + "additionalProperties": { + "type": "string" + } + }, + "noul": { + "type": "number" + }, + "probabilities": { + "type": "object", + "additionalProperties": { + "type": "number", + "format": "float64" + } + }, + "score": { + "type": "number" + }, + "type": { + "type": "string" + } + } + }, + "schema.SystemOneEntity": { + "type": "object", + "properties": { + "confidence": { + "type": "number" + }, + "end": { + "type": "integer" + }, + "start": { + "type": "integer" + }, + "text": { + "type": "string" + } + } + }, + "schema.SystemOnePermuteRequest": { + "type": "object", + "properties": { + "n_perm": { + "type": "integer" + }, + "question": { + "type": "string" + }, + "request": { + "$ref": "#/definitions/schema.SystemOneRequest" + }, + "seed": { + "type": "integer" + } + } + }, + "schema.SystemOnePermuteResponse": { + "type": "object", + "properties": { + "argmax_stable": { + "type": "boolean" + }, + "runs": { + "type": "array", + "items": { + "$ref": "#/definitions/schema.SystemOnePermuteRun" + } + }, + "spread": { + "type": "object", + "additionalProperties": { + "type": "number", + "format": "float64" + } + } + } + }, + "schema.SystemOnePermuteRun": { + "type": "object", + "properties": { + "choice": { + "type": "string" + }, + "latency_ms": { + "type": "number" + }, + "order": { + "type": "array", + "items": { + "type": "string" + } + }, + "probabilities": { + "type": "object", + "additionalProperties": { + "type": "number", + "format": "float64" + } + } + } + }, + "schema.SystemOneRequest": { + "type": "object" + }, + "schema.SystemOneResponse": { + "type": "object", + "properties": { + "answers": { + "type": "object", + "additionalProperties": { + "$ref": "#/definitions/schema.SystemOneAnswer" + } + }, + "latency_ms": { + "type": "number" + }, + "model": { + "type": "string" + }, + "usage": { + "$ref": "#/definitions/schema.SystemOneUsage" + } + } + }, + "schema.SystemOneUsage": { + "type": "object", + "properties": { + "input_tokens": { + "type": "integer" + }, + "output_tokens": { + "type": "integer" + } + } + }, "schema.TTSRequest": { "description": "TTS request body", "type": "object", diff --git a/swagger/swagger.yaml b/swagger/swagger.yaml index 34fcfc8d7..1337afa9a 100644 --- a/swagger/swagger.yaml +++ b/swagger/swagger.yaml @@ -2,13 +2,19 @@ basePath: / definitions: config.Gallery: properties: + artifact_verification: + allOf: + - $ref: '#/definitions/config.GalleryVerification' + description: |- + ArtifactVerification overrides Verification only for the gallery OCI artifact. + Backend images keep their separate Verification policy. mirrors: description: |- Mirrors are tried in order when URL cannot be fetched. They are a fallback for availability, not a load-balancing pool: the primary is always preferred, and a mirror is only consulted after the one before it fails. Any URI the gallery loader understands works here - (https://, github:, file://). + (https://, github:, file://, oci://). items: type: string type: array @@ -32,6 +38,11 @@ definitions: not_before: description: NotBefore is an RFC3339 timestamp. Empty disables the time check. type: string + source_repository: + description: |- + SourceRepository is an https URL compared exactly against the + certificate's source-repository extension. Empty skips the check. + type: string type: object config.TTSVoice: properties: @@ -2654,12 +2665,38 @@ definitions: type: string id: type: string + process: + allOf: + - $ref: '#/definitions/schema.SysInfoProcess' + description: |- + Process is the backend process serving the model on this host. Absent + when the model has no local process (a distributed worker holds it) or + the process could not be read. size_vram: description: |- SizeVRAM is DRM-accounted resident device memory in bytes. Nil means the backend process tree has no complete supported reading. type: integer type: object + schema.SysInfoProcess: + properties: + cpu_percent: + description: |- + CPUPercent is the share of the whole host's CPU used since the previous + reading, 0-100. Absent on the first reading of a process. + type: number + memory_percent: + type: number + pid: + type: integer + rss_bytes: + description: |- + RSSBytes is resident host memory. Weights offloaded to a GPU are not + in it. + type: integer + started_at: + type: string + type: object schema.SystemInformationResponse: properties: backends: @@ -2673,6 +2710,106 @@ definitions: $ref: '#/definitions/schema.SysInfoModel' type: array type: object + schema.SystemOneAnswer: + properties: + choice: + type: string + confidence: + type: number + entities: + items: + $ref: '#/definitions/schema.SystemOneEntity' + type: array + legend: + additionalProperties: + type: string + type: object + noul: + type: number + probabilities: + additionalProperties: + format: float64 + type: number + type: object + score: + type: number + type: + type: string + type: object + schema.SystemOneEntity: + properties: + confidence: + type: number + end: + type: integer + start: + type: integer + text: + type: string + type: object + schema.SystemOnePermuteRequest: + properties: + n_perm: + type: integer + question: + type: string + request: + $ref: '#/definitions/schema.SystemOneRequest' + seed: + type: integer + type: object + schema.SystemOnePermuteResponse: + properties: + argmax_stable: + type: boolean + runs: + items: + $ref: '#/definitions/schema.SystemOnePermuteRun' + type: array + spread: + additionalProperties: + format: float64 + type: number + type: object + type: object + schema.SystemOnePermuteRun: + properties: + choice: + type: string + latency_ms: + type: number + order: + items: + type: string + type: array + probabilities: + additionalProperties: + format: float64 + type: number + type: object + type: object + schema.SystemOneRequest: + type: object + schema.SystemOneResponse: + properties: + answers: + additionalProperties: + $ref: '#/definitions/schema.SystemOneAnswer' + type: object + latency_ms: + type: number + model: + type: string + usage: + $ref: '#/definitions/schema.SystemOneUsage' + type: object + schema.SystemOneUsage: + properties: + input_tokens: + type: integer + output_tokens: + type: integer + type: object schema.TTSRequest: description: TTS request body properties: @@ -5672,6 +5809,64 @@ paths: summary: Generates audio from the input text. tags: - audio + /v1/systemone: + post: + description: 'Runs zero-shot NER over the supplied state and answers each question. + Question types: noul (binary entity presence), choice (pick one option), score + (pick one level).' + parameters: + - description: state + questions + in: body + name: request + required: true + schema: + $ref: '#/definitions/schema.SystemOneRequest' + responses: + "200": + description: OK + schema: + $ref: '#/definitions/schema.SystemOneResponse' + summary: Answer structured-extraction questions over state text. + tags: + - systemone + /v1/systemone/permute: + post: + description: Re-runs one choice question under n_perm option orders. Reports + per-order probabilities, argmax stability, and spread. + parameters: + - description: request + question + n_perm + seed + in: body + name: request + required: true + schema: + $ref: '#/definitions/schema.SystemOnePermuteRequest' + responses: + "200": + description: OK + schema: + $ref: '#/definitions/schema.SystemOnePermuteResponse' + summary: Re-run a choice question under multiple option orders. + tags: + - systemone + /v1/systemone/separate: + post: + description: Runs N independent NER passes, one per question, against the same + state. Response shape matches /v1/systemone. + parameters: + - description: state + questions + in: body + name: request + required: true + schema: + $ref: '#/definitions/schema.SystemOneRequest' + responses: + "200": + description: OK + schema: + $ref: '#/definitions/schema.SystemOneResponse' + summary: Answer each question in a separate NER pass. + tags: + - systemone /v1/text-to-speech/{voice-id}: post: parameters: From bc01ef235043c87a832f8cee7fe2831eeb6d9812 Mon Sep 17 00:00:00 2001 From: mudler-agent Date: Mon, 28 Sep 2026 09:28:04 +0200 Subject: [PATCH 07/15] fix(gallery): keep model deletion inside the models directory (#12324) listModelFiles gave utils.VerifyPath paths that it had already joined onto the models directory. VerifyPath joins its argument onto the base again, so an absolute path always passes and none of the four checks could fail. Model deletion then removed files outside the models directory: - A model name such as "../outside/victim" removed outside/victim.yaml. The in-process MCP delete_model tool passes the name from the tool call without a check. - A gallery file that lists a files: entry with "../" removed that file. listModelFiles now gives VerifyPath the relative names. InTrustedRoot also looped forever when a relative path was outside a relative root. filepath.Dir stops at "." for a relative path, and the loop waited for "/". The loop now stops when Dir returns its input. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto --- core/gallery/delete_model_paths_test.go | 72 +++++++++++++++++++++++++ core/gallery/models.go | 11 ++-- pkg/utils/path.go | 11 ++-- pkg/utils/path_test.go | 16 ++++++ 4 files changed, 103 insertions(+), 7 deletions(-) create mode 100644 core/gallery/delete_model_paths_test.go diff --git a/core/gallery/delete_model_paths_test.go b/core/gallery/delete_model_paths_test.go new file mode 100644 index 000000000..a61cfe636 --- /dev/null +++ b/core/gallery/delete_model_paths_test.go @@ -0,0 +1,72 @@ +package gallery_test + +import ( + "os" + "path/filepath" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + "github.com/mudler/LocalAI/core/gallery" + "github.com/mudler/LocalAI/pkg/system" +) + +// DeleteModelFromSystem removes files named by a model name and by the +// model's gallery file. Neither may reach outside the models directory: the +// name can come from an API caller or from an assistant tool call, and the +// gallery file is a YAML file on disk. +var _ = Describe("DeleteModelFromSystem path containment", func() { + var ( + root string + modelsPath string + outside string + state *system.SystemState + ) + + BeforeEach(func() { + root = GinkgoT().TempDir() + modelsPath = filepath.Join(root, "models") + outside = filepath.Join(root, "outside") + Expect(os.MkdirAll(modelsPath, 0o755)).To(Succeed()) + Expect(os.MkdirAll(outside, 0o755)).To(Succeed()) + var err error + state, err = system.GetSystemState(system.WithModelPath(modelsPath)) + Expect(err).ToNot(HaveOccurred()) + }) + + It("refuses a model name that escapes the models directory", func() { + victim := filepath.Join(outside, "victim.yaml") + Expect(os.WriteFile(victim, []byte("name: victim\n"), 0o644)).To(Succeed()) + + Expect(gallery.DeleteModelFromSystem(state, "../outside/victim")).ToNot(Succeed()) + Expect(victim).To(BeARegularFile()) + }) + + It("does not remove gallery-declared files outside the models directory", func() { + secret := filepath.Join(outside, "secret.bin") + Expect(os.WriteFile(secret, []byte("x"), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelsPath, "m.yaml"), []byte("name: m\n"), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("m")), []byte(` +files: + - filename: ../outside/secret.bin +`), 0o644)).To(Succeed()) + + _ = gallery.DeleteModelFromSystem(state, "m") + Expect(secret).To(BeARegularFile()) + }) + + It("still deletes a normal model and its declared files", func() { + weights := filepath.Join(modelsPath, "m", "w.gguf") + Expect(os.MkdirAll(filepath.Dir(weights), 0o755)).To(Succeed()) + Expect(os.WriteFile(weights, []byte("w"), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelsPath, "m.yaml"), []byte("name: m\nparameters:\n model: m/w.gguf\n"), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("m")), []byte(` +files: + - filename: m/w.gguf +`), 0o644)).To(Succeed()) + + Expect(gallery.DeleteModelFromSystem(state, "m")).To(Succeed()) + Expect(weights).ToNot(BeAnExistingFile()) + Expect(filepath.Join(modelsPath, "m.yaml")).ToNot(BeAnExistingFile()) + }) +}) diff --git a/core/gallery/models.go b/core/gallery/models.go index c787dcde4..6d4ceb346 100644 --- a/core/gallery/models.go +++ b/core/gallery/models.go @@ -808,8 +808,11 @@ func GetLocalModelConfiguration(basePath string, name string) (*ModelConfig, err func listModelFiles(systemState *system.SystemState, name string) ([]string, error) { + // VerifyPath joins its argument onto the models path itself, so every + // check below passes the relative name: an already-joined absolute path + // always lands inside the base and the check would pass anything. configFile := filepath.Join(systemState.Model.ModelsPath, fmt.Sprintf("%s.yaml", name)) - if err := utils.VerifyPath(configFile, systemState.Model.ModelsPath); err != nil { + if err := utils.VerifyPath(fmt.Sprintf("%s.yaml", name), systemState.Model.ModelsPath); err != nil { return nil, fmt.Errorf("failed to verify path %s: %w", configFile, err) } @@ -817,7 +820,7 @@ func listModelFiles(systemState *system.SystemState, name string) ([]string, err name = strings.ReplaceAll(name, string(os.PathSeparator), "__") galleryFile := filepath.Join(systemState.Model.ModelsPath, galleryFileName(name)) - if err := utils.VerifyPath(galleryFile, systemState.Model.ModelsPath); err != nil { + if err := utils.VerifyPath(galleryFileName(name), systemState.Model.ModelsPath); err != nil { return nil, fmt.Errorf("failed to verify path %s: %w", galleryFile, err) } @@ -847,7 +850,7 @@ func listModelFiles(systemState *system.SystemState, name string) ([]string, err if err == nil && galleryconfig != nil { for _, f := range galleryconfig.Files { fullPath := filepath.Join(systemState.Model.ModelsPath, f.Filename) - if err := utils.VerifyPath(fullPath, systemState.Model.ModelsPath); err != nil { + if err := utils.VerifyPath(f.Filename, systemState.Model.ModelsPath); err != nil { return allFiles, fmt.Errorf("failed to verify path %s: %w", fullPath, err) } allFiles = append(allFiles, fullPath) @@ -858,7 +861,7 @@ func listModelFiles(systemState *system.SystemState, name string) ([]string, err for _, f := range additionalFiles { fullPath := filepath.Join(filepath.Join(systemState.Model.ModelsPath, f)) - if err := utils.VerifyPath(fullPath, systemState.Model.ModelsPath); err != nil { + if err := utils.VerifyPath(f, systemState.Model.ModelsPath); err != nil { return allFiles, fmt.Errorf("failed to verify path %s: %w", fullPath, err) } allFiles = append(allFiles, fullPath) diff --git a/pkg/utils/path.go b/pkg/utils/path.go index 1ae11d123..df3af9072 100644 --- a/pkg/utils/path.go +++ b/pkg/utils/path.go @@ -13,13 +13,18 @@ func ExistsInPath(path string, s string) bool { } func InTrustedRoot(path string, trustedRoot string) error { - for path != "/" { - path = filepath.Dir(path) + for { + parent := filepath.Dir(path) + // Dir stops changing at "/" for an absolute path and at "." for a + // relative one; waiting for "/" alone spins forever on the latter. + if parent == path { + return fmt.Errorf("path is outside of trusted root") + } + path = parent if path == trustedRoot { return nil } } - return fmt.Errorf("path is outside of trusted root") } // VerifyPath verifies that path is based in basePath. diff --git a/pkg/utils/path_test.go b/pkg/utils/path_test.go index 79c415cd4..1c9d18b2c 100644 --- a/pkg/utils/path_test.go +++ b/pkg/utils/path_test.go @@ -3,6 +3,7 @@ package utils_test import ( "os" "path/filepath" + "time" . "github.com/mudler/LocalAI/pkg/utils" . "github.com/onsi/ginkgo/v2" @@ -93,6 +94,21 @@ var _ = Describe("utils/path tests", func() { It("rejects an unrelated absolute path", func() { Expect(InTrustedRoot("/etc/passwd", "/srv/models")).ToNot(Succeed()) }) + + It("rejects a relative path outside a relative root instead of looping", func() { + // Walking up a relative path ends at ".", never at "/", so the + // walk must stop when it stops making progress. + done := make(chan error, 1) + go func() { done <- InTrustedRoot("x", "models") }() + Eventually(done).WithTimeout(2 * time.Second).Should(Receive(HaveOccurred())) + + go func() { done <- VerifyPath("../x", "models") }() + Eventually(done).WithTimeout(2 * time.Second).Should(Receive(HaveOccurred())) + }) + + It("accepts a relative descendant of a relative root", func() { + Expect(InTrustedRoot("models/a/file", "models")).To(Succeed()) + }) }) Describe("SanitizeFileName", func() { From 50c284fdcc77bface613b8a838c36cdd2208f4f5 Mon Sep 17 00:00:00 2001 From: mudler-agent Date: Mon, 28 Sep 2026 10:04:22 +0200 Subject: [PATCH 08/15] fix: make the remaining VerifyPath checks effective (#12326) utils.VerifyPath joins its argument onto the base path, so a path that the caller already joined always passes. Several callers gave it joined paths, and their checks could not fail: - modeladmin (config view, patch, edit, pin and state): the config file path from the loader. A config loaded from outside the models directory (--models-config-file) could be pinned, and the pin wrote the outside file. The patch and state paths stopped later, in the mutation snapshot, with a different error. - core/backend/tts.go: the model path joined onto the models path. - The trellis2cpp and stablediffusion-ggml backends: option paths (*_path) joined onto the model path. A "../" value outside the model directory was accepted. Add utils.VerifyResolvedPath for a full path. modeladmin and tts use it. The backends now check the relative option value before they join it. A rename in modeladmin checks the new relative name. For models from a config file outside the models directory, the admin API and web UI now return ErrPathNotTrusted for view, edit, pin, and enable or disable. The docs describe this. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto --- backend/go/stablediffusion-ggml/gosd.go | 2 +- backend/go/trellis2cpp/trellis2.go | 2 +- backend/go/trellis2cpp/trellis2_test.go | 5 +- core/backend/tts.go | 4 +- core/services/modeladmin/config.go | 8 +-- .../modeladmin/config_path_trust_test.go | 58 +++++++++++++++++++ core/services/modeladmin/pinned.go | 2 +- core/services/modeladmin/state.go | 2 +- docs/content/advanced/model-configuration.md | 2 + pkg/utils/path.go | 11 +++- pkg/utils/path_test.go | 19 ++++++ 11 files changed, 103 insertions(+), 12 deletions(-) create mode 100644 core/services/modeladmin/config_path_trust_test.go diff --git a/backend/go/stablediffusion-ggml/gosd.go b/backend/go/stablediffusion-ggml/gosd.go index e1567bd18..60f2efede 100644 --- a/backend/go/stablediffusion-ggml/gosd.go +++ b/backend/go/stablediffusion-ggml/gosd.go @@ -137,8 +137,8 @@ func (sd *SDGGML) Load(opts *pb.ModelOptions) error { // If it's an option path, we resolve absolute path from the model path if strings.Contains(op, ":") && strings.Contains(op, "path") { data := strings.Split(op, ":") - data[1] = filepath.Join(opts.ModelPath, data[1]) if err := utils.VerifyPath(data[1], opts.ModelPath); err == nil { + data[1] = filepath.Join(opts.ModelPath, data[1]) oo = append(oo, strings.Join(data, ":")) } } else { diff --git a/backend/go/trellis2cpp/trellis2.go b/backend/go/trellis2cpp/trellis2.go index addbf1bb0..c83480aa4 100644 --- a/backend/go/trellis2cpp/trellis2.go +++ b/backend/go/trellis2cpp/trellis2.go @@ -161,10 +161,10 @@ func resolveModels(modelFile, modelPath string, options []string) (modelSet, err continue } if !filepath.IsAbs(value) { - value = filepath.Join(modelPath, value) if err := utils.VerifyPath(value, modelPath); err != nil { return modelSet{}, fmt.Errorf("option %s: %w", key, err) } + value = filepath.Join(modelPath, value) } overrides[key] = value } diff --git a/backend/go/trellis2cpp/trellis2_test.go b/backend/go/trellis2cpp/trellis2_test.go index d492756e4..b221f00c5 100644 --- a/backend/go/trellis2cpp/trellis2_test.go +++ b/backend/go/trellis2cpp/trellis2_test.go @@ -134,9 +134,12 @@ var _ = Describe("resolveModels", func() { It("rejects option paths escaping the model directory", func() { touch(dir, fullSet...) + // The escaping file exists, so only the containment check can + // reject it; a missing file would fail for an unrelated reason. + touch(filepath.Dir(dir), "outside.gguf") _, err := resolveModels("ss_flow_f16.gguf", dir, []string{"dino_path:../outside.gguf"}) - Expect(err).To(HaveOccurred()) + Expect(err).To(MatchError(ContainSubstring("outside of trusted root"))) }) }) diff --git a/core/backend/tts.go b/core/backend/tts.go index 3451eb471..1172fb8f2 100644 --- a/core/backend/tts.go +++ b/core/backend/tts.go @@ -88,7 +88,7 @@ func ModelTTS( // a FS path mp := filepath.Join(loader.ModelPath, modelConfig.Model) if _, err := os.Stat(mp); err == nil { - if err := utils.VerifyPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil { + if err := utils.VerifyResolvedPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil { return "", nil, err } modelPath = mp @@ -189,7 +189,7 @@ func ModelTTSStream( // a FS path mp := filepath.Join(loader.ModelPath, modelConfig.Model) if _, err := os.Stat(mp); err == nil { - if err := utils.VerifyPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil { + if err := utils.VerifyResolvedPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil { return err } modelPath = mp diff --git a/core/services/modeladmin/config.go b/core/services/modeladmin/config.go index 2cadfcaed..f85f29ee9 100644 --- a/core/services/modeladmin/config.go +++ b/core/services/modeladmin/config.go @@ -93,7 +93,7 @@ func (s *ConfigService) GetConfig(_ context.Context, name string) (*ConfigView, if configPath == "" { return nil, ErrConfigFileMissing } - if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil { + if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil { return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err) } data, err := os.ReadFile(configPath) @@ -137,7 +137,7 @@ func (s *ConfigService) patchConfig(ctx context.Context, name string, patch map[ return nil, fmt.Errorf("%w: PATCH cannot rename model %q to %q; use the model edit endpoint", ErrInvalidConfig, name, patchedName) } configPath := cfg.GetModelConfigFile() - if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil { + if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil { return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err) } diskYAML, err := os.ReadFile(configPath) @@ -289,7 +289,7 @@ func (s *ConfigService) editYAML(ctx context.Context, name string, body []byte) configPath := existing.GetModelConfigFile() modelsPath := s.modelsPath() - if err := utils.VerifyPath(configPath, modelsPath); err != nil { + if err := utils.VerifyResolvedPath(configPath, modelsPath); err != nil { return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err) } @@ -304,7 +304,7 @@ func (s *ConfigService) editYAML(ctx context.Context, name string, body []byte) } newConfigPath := filepath.Join(modelsPath, req.Name+".yaml") paths = append(paths, newConfigPath, filepath.Join(modelsPath, gallery.GalleryFileName(name)), filepath.Join(modelsPath, gallery.GalleryFileName(req.Name))) - if err := utils.VerifyPath(newConfigPath, modelsPath); err != nil { + if err := utils.VerifyPath(req.Name+".yaml", modelsPath); err != nil { return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err) } if _, err := os.Stat(newConfigPath); err == nil { diff --git a/core/services/modeladmin/config_path_trust_test.go b/core/services/modeladmin/config_path_trust_test.go new file mode 100644 index 000000000..5b375b699 --- /dev/null +++ b/core/services/modeladmin/config_path_trust_test.go @@ -0,0 +1,58 @@ +package modeladmin + +import ( + "context" + "os" + "path/filepath" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +// A model config can be loaded from outside the models directory (for +// example with --config-file). The admin mutations write the config file +// back, so they must refuse a file outside the models directory rather than +// write wherever the loader found it. +var _ = Describe("ConfigService config file containment", func() { + var ( + svc *ConfigService + ctx context.Context + outside string + orig []byte + ) + + BeforeEach(func() { + svc, _ = newTestService() + ctx = context.Background() + outside = filepath.Join(GinkgoT().TempDir(), "external.yaml") + orig = []byte("name: external\nbackend: llama-cpp\n") + Expect(os.WriteFile(outside, orig, 0o644)).To(Succeed()) + Expect(svc.Loader.ReadModelConfig(outside, svc.AppConfig.ToConfigLoaderOptions()...)).To(Succeed()) + cfg, ok := svc.Loader.GetModelConfig("external") + Expect(ok).To(BeTrue()) + Expect(cfg.GetModelConfigFile()).To(Equal(outside)) + }) + + It("refuses to pin a model whose config file is outside the models directory", func() { + _, err := svc.TogglePinned(ctx, "external", ActionPin, nil) + Expect(err).To(MatchError(ErrPathNotTrusted)) + Expect(os.ReadFile(outside)).To(Equal(orig)) + }) + + It("refuses to toggle the state of such a model", func() { + _, err := svc.ToggleState(ctx, "external", ActionDisable) + Expect(err).To(MatchError(ErrPathNotTrusted)) + Expect(os.ReadFile(outside)).To(Equal(orig)) + }) + + It("refuses to patch such a model", func() { + _, err := svc.PatchConfig(ctx, "external", map[string]any{"context_size": 4096}) + Expect(err).To(MatchError(ErrPathNotTrusted)) + Expect(os.ReadFile(outside)).To(Equal(orig)) + }) + + It("refuses to read such a model's config", func() { + _, err := svc.GetConfig(ctx, "external") + Expect(err).To(MatchError(ErrPathNotTrusted)) + }) +}) diff --git a/core/services/modeladmin/pinned.go b/core/services/modeladmin/pinned.go index b9ef45724..17c4a2f39 100644 --- a/core/services/modeladmin/pinned.go +++ b/core/services/modeladmin/pinned.go @@ -29,7 +29,7 @@ func (s *ConfigService) TogglePinned(_ context.Context, name string, action Acti if configPath == "" { return nil, ErrConfigFileMissing } - if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil { + if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil { return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err) } if err := mutateYAMLBoolFlag(configPath, "pinned", action == ActionPin); err != nil { diff --git a/core/services/modeladmin/state.go b/core/services/modeladmin/state.go index d37368d9d..cee6d7855 100644 --- a/core/services/modeladmin/state.go +++ b/core/services/modeladmin/state.go @@ -49,7 +49,7 @@ func (s *ConfigService) toggleState(ctx context.Context, name string, action Act if configPath == "" { return nil, ErrConfigFileMissing } - if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil { + if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil { return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err) } var result *ToggleResult diff --git a/docs/content/advanced/model-configuration.md b/docs/content/advanced/model-configuration.md index 1c2ade4de..5cfa74ccd 100644 --- a/docs/content/advanced/model-configuration.md +++ b/docs/content/advanced/model-configuration.md @@ -74,6 +74,8 @@ When using `--models-config-file`, you can define multiple models as a list: backend: llama-cpp ``` +LocalAI changes only config files that are inside the models directory. If the file from `--models-config-file` is outside the models directory, you cannot view, edit, pin, enable or disable its models from the web UI or the model admin API. Edit the file directly, then restart LocalAI. + ## Core Configuration Fields ### Basic Model Settings diff --git a/pkg/utils/path.go b/pkg/utils/path.go index df3af9072..16c29de15 100644 --- a/pkg/utils/path.go +++ b/pkg/utils/path.go @@ -27,12 +27,21 @@ func InTrustedRoot(path string, trustedRoot string) error { } } -// VerifyPath verifies that path is based in basePath. +// VerifyPath verifies that path, taken relative to basePath, is based in +// basePath. It joins path onto basePath first, so an absolute path is read as +// relative to the base as well: give it the untrusted relative name, never a +// path that has already been joined. For a full path use VerifyResolvedPath. func VerifyPath(path, basePath string) error { c := filepath.Clean(filepath.Join(basePath, path)) return InTrustedRoot(c, filepath.Clean(basePath)) } +// VerifyResolvedPath verifies that path, a full path rather than one relative +// to basePath, is based in basePath. +func VerifyResolvedPath(path, basePath string) error { + return InTrustedRoot(filepath.Clean(path), filepath.Clean(basePath)) +} + // SanitizeFileName sanitizes the given filename func SanitizeFileName(fileName string) string { // filepath.Clean to clean the path diff --git a/pkg/utils/path_test.go b/pkg/utils/path_test.go index 1c9d18b2c..e970a45ab 100644 --- a/pkg/utils/path_test.go +++ b/pkg/utils/path_test.go @@ -72,6 +72,25 @@ var _ = Describe("utils/path tests", func() { }) }) + Describe("VerifyResolvedPath", func() { + It("accepts a full path inside the base", func() { + Expect(VerifyResolvedPath("/srv/models/a/model.yaml", "/srv/models")).To(Succeed()) + }) + + It("rejects a full path outside the base", func() { + // VerifyPath would join this onto the base and accept it. + Expect(VerifyResolvedPath("/etc/passwd", "/srv/models")).ToNot(Succeed()) + }) + + It("rejects a joined path that climbed out of the base", func() { + Expect(VerifyResolvedPath(filepath.Join("/srv/models", "../other/x"), "/srv/models")).ToNot(Succeed()) + }) + + It("cleans both paths before comparing", func() { + Expect(VerifyResolvedPath("/srv/models/./a/../b.yaml", "/srv/models/")).To(Succeed()) + }) + }) + Describe("InTrustedRoot", func() { It("accepts a strict descendant of the trusted root", func() { Expect(InTrustedRoot("/srv/models/file", "/srv/models")).To(Succeed()) From bc1d9924decb3ff253b8443d847325f6659dc40f Mon Sep 17 00:00:00 2001 From: Stefan Walcz Date: Mon, 28 Sep 2026 11:37:24 +0200 Subject: [PATCH 09/15] fix(llama-cpp): keep llama.cpp's default cache_ram instead of no limit (#12297) grpc-server.cpp forced params.cache_ram_mib = -1 (no limit) since #7009. Since v4.3 kv_unified and cache_idle_slots are on by default, so every distinct prompt now leaves its slot KV state in the host-side prompt cache, and without a limit the backend grows until the host runs out of memory. Measured on gfx1151 (Strix Halo, 128 GB), llama-cpp backend, one request at a time, 100 distinct prompts of ~2000 characters plus a fixed system prompt, max_tokens 200: model cache_ram RSS loaded -> after 100 gemma-4-26B-A4B (q8_0 KV) -1 (default) 1.4 GB -> 25.3 GB Qwen3.6-35B-A3B (q8_0 KV) -1 (default) 1.1 GB -> 19.5 GB gemma-4-26B-A4B -1, same prompt 100x 1.4 GB -> 1.6 GB gemma-4-26B-A4B 4096 1.4 GB -> 5.4 GB (flat from request 20 on, same latency) Qwen3.6-35B-A3B 4096 1.1 GB -> 5.1 GB (flat) The memory is not released when idle. In production a document classification pass pushed the daily chat model to 34 GB RSS overnight. Drop the override so llama.cpp's own default (8192 MiB) applies; the cache_ram option still accepts -1 for users who want no limit. Update both docs tables (the option reference and the prompt-cache table) and note what -1 does. Assisted-by: Claude:claude-opus-5-5 Assisted-by: Codex:GPT-6 Signed-off-by: Stefan Walcz --- backend/cpp/llama-cpp/grpc-server.cpp | 8 +++++--- docs/content/features/text-generation.md | 6 ++++-- 2 files changed, 9 insertions(+), 5 deletions(-) diff --git a/backend/cpp/llama-cpp/grpc-server.cpp b/backend/cpp/llama-cpp/grpc-server.cpp index 9a2240881..8b206612c 100644 --- a/backend/cpp/llama-cpp/grpc-server.cpp +++ b/backend/cpp/llama-cpp/grpc-server.cpp @@ -539,8 +539,10 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt // Initialize ctx_shift to false by default (can be overridden by options) params.ctx_shift = false; - // Initialize cache_ram_mib to -1 by default (no limit, can be overridden by options) - params.cache_ram_mib = -1; + // cache_ram_mib keeps llama.cpp's own default (8192 MiB) unless overridden by + // options. It used to be forced to -1 (no limit): since kv_unified and + // cache_idle_slots are on by default, every distinct prompt then leaves its + // slot state in host RAM and the backend grows without bound. // Initialize n_parallel to 1 by default (can be overridden by options) params.n_parallel = 1; // Initialize grpc_servers to empty (can be overridden by options) @@ -656,7 +658,7 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt try { params.cache_ram_mib = std::stoi(optval_str); } catch (const std::exception& e) { - // If conversion fails, keep default value (-1) + // If conversion fails, keep the default value } } } else if (!strcmp(optname, "parallel") || !strcmp(optname, "n_parallel")) { diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md index c28bf9213..19dcf64b0 100644 --- a/docs/content/features/text-generation.md +++ b/docs/content/features/text-generation.md @@ -586,7 +586,7 @@ The `llama.cpp` backend supports additional configuration options that can be sp |--------|------|-------------|---------| | `use_jinja` or `jinja` | boolean | Enable Jinja2 template processing for chat templates. When enabled, the backend uses Jinja2-based chat templates from the model for formatting messages. | `use_jinja:true` | | `context_shift` | boolean | Enable context shifting, which allows the model to dynamically adjust context window usage. | `context_shift:true` | -| `cache_ram` | integer | Size budget in MiB for the **server-side prompt cache** (a host-RAM store of idle slot KV states that's reloaded on a prompt-prefix hit, see [upstream PR #16391](https://github.com/ggml-org/llama.cpp/pull/16391)). Default: `-1` (no limit). `0` disables the prompt cache entirely. Together with `kv_unified` and `cache_idle_slots` this is what makes a repeated system prompt skip prefill on subsequent calls. | `cache_ram:4096` | +| `cache_ram` | integer | Size budget in MiB for the **server-side prompt cache** (a host-RAM store of idle slot KV states that's reloaded on a prompt-prefix hit, see [upstream PR #16391](https://github.com/ggml-org/llama.cpp/pull/16391)). Default: `8192` MiB (llama.cpp default). `-1` removes the limit. `0` disables the prompt cache entirely. Together with `kv_unified` and `cache_idle_slots` this is what makes a repeated system prompt skip prefill on subsequent calls. | `cache_ram:4096` | | `parallel` or `n_parallel` | integer | Enable parallel request processing. When set to a value greater than 1, enables continuous batching for handling multiple requests concurrently. | `parallel:4` | | `grpc_servers` or `rpc_servers` | string | Comma-separated list of gRPC server addresses for distributed inference. Allows distributing workload across multiple llama.cpp workers. | `grpc_servers:localhost:50051,localhost:50052` | | `fit_params` or `fit` | boolean | Enable auto-adjustment of model/context parameters to fit available device memory. Default: `true`. | `fit_params:true` | @@ -642,7 +642,7 @@ Agents, coding assistants, and Anthropic/OpenAI-compatible CLIs typically resend | Setting | Default | Role | |---|---|---| -| `cache_ram:N` | `-1` (no limit) | Allocates the host-side prompt cache. `0` disables it. | +| `cache_ram:N` | `8192` (llama.cpp default) | Allocates the host-side prompt cache. `0` disables it. | | `kv_unified:true` | `true` | Single unified KV buffer (**prerequisite** for idle-slot saving). | | `cache_idle_slots:true` | `true` | Persists the idle slot's KV into the prompt cache on task switch. | @@ -657,6 +657,8 @@ options: Set `cache_ram:0` to opt out of the prompt cache entirely (saves host RAM at the cost of re-prefilling repeated prompts). +`cache_ram:-1` removes the limit. With idle-slot saving on, every distinct prompt then leaves its slot state in host RAM, so a workload with many different prompts (classification, ingestion) grows the backend by roughly the KV size of each prompt until the host runs out of memory. + #### Reference - [llama](https://github.com/ggerganov/llama.cpp) From adbff0a44c347914cff3e60a8d99cc9017a3b319 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Mon, 28 Sep 2026 13:09:15 +0200 Subject: [PATCH 10/15] chore: :arrow_up: Update ikawrakow/ik_llama.cpp to `ed27bf7ed25e637692e89cd341d802522a2cee8a` (#12313) :arrow_up: Update ikawrakow/ik_llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/ik-llama-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/ik-llama-cpp/Makefile b/backend/cpp/ik-llama-cpp/Makefile index d6bfd7490..d32465ec9 100644 --- a/backend/cpp/ik-llama-cpp/Makefile +++ b/backend/cpp/ik-llama-cpp/Makefile @@ -1,5 +1,5 @@ -IK_LLAMA_VERSION?=cdf232cc17e410e60c1bc3b85516c4a41199b662 +IK_LLAMA_VERSION?=ed27bf7ed25e637692e89cd341d802522a2cee8a LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp CMAKE_ARGS?= From 7f139c8add7e68567285db6b174a6627a5db94dd Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Mon, 28 Sep 2026 13:09:32 +0200 Subject: [PATCH 11/15] chore: :arrow_up: Update CrispStrobe/CrispASR to `ec98831d0776ec8a16ccaf93955693eb7ecfbec3` (#12314) :arrow_up: Update CrispStrobe/CrispASR Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/crispasr/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/crispasr/Makefile b/backend/go/crispasr/Makefile index f2155ffb9..bbdfef58c 100644 --- a/backend/go/crispasr/Makefile +++ b/backend/go/crispasr/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # CrispASR version (release tag) CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR -CRISPASR_VERSION?=013ae1624dc40ecf059065d577180722439f804e +CRISPASR_VERSION?=ec98831d0776ec8a16ccaf93955693eb7ecfbec3 SO_TARGET?=libgocrispasr.so CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF From b449ad3828b0949ecbcc00db75626ff89a1002c0 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Mon, 28 Sep 2026 13:09:57 +0200 Subject: [PATCH 12/15] chore: :arrow_up: Update 0xShug0/audio.cpp to `77491a33c589c53ff18add050095cf35647c8213` (#12315) :arrow_up: Update 0xShug0/audio.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/audio-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/audio-cpp/Makefile b/backend/cpp/audio-cpp/Makefile index e8d27eb52..64638b0c1 100644 --- a/backend/cpp/audio-cpp/Makefile +++ b/backend/cpp/audio-cpp/Makefile @@ -9,7 +9,7 @@ # recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean # rebuild and so the bump bot can see the pin. -AUDIO_CPP_VERSION?=94bd4656399180befc141b17bd6696bf84df0a9f +AUDIO_CPP_VERSION?=77491a33c589c53ff18add050095cf35647c8213 AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST)))) From 197311f033a6cec247a990aeb240aaabf715267f Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Mon, 28 Sep 2026 13:10:11 +0200 Subject: [PATCH 13/15] chore: :arrow_up: Update ServeurpersoCom/omnivoice.cpp to `ead199a2bc4c53a57cac90095ae049a111d9e98d` (#12316) :arrow_up: Update ServeurpersoCom/omnivoice.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/omnivoice-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/omnivoice-cpp/Makefile b/backend/go/omnivoice-cpp/Makefile index 6d36ea9be..6c8539bd5 100644 --- a/backend/go/omnivoice-cpp/Makefile +++ b/backend/go/omnivoice-cpp/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # omnivoice.cpp version OMNIVOICE_REPO?=https://github.com/ServeurpersoCom/omnivoice.cpp -OMNIVOICE_VERSION?=8ab42195a05a9d48a3942b17568c1f3a876e133a +OMNIVOICE_VERSION?=ead199a2bc4c53a57cac90095ae049a111d9e98d SO_TARGET?=libgomnivoicecpp.so CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF From 102fbe0c78113d7fabd61d81b81c5d1a92de98d1 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Mon, 28 Sep 2026 13:10:28 +0200 Subject: [PATCH 14/15] chore: :arrow_up: Update leejet/stable-diffusion.cpp to `3f8527a46c54ecf4cb4ed6003da8e8982283c73c` (#12317) :arrow_up: Update leejet/stable-diffusion.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/stablediffusion-ggml/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/stablediffusion-ggml/Makefile b/backend/go/stablediffusion-ggml/Makefile index 6c5e70484..b2d9c9049 100644 --- a/backend/go/stablediffusion-ggml/Makefile +++ b/backend/go/stablediffusion-ggml/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # stablediffusion.cpp (ggml) STABLEDIFFUSION_GGML_REPO?=https://github.com/leejet/stable-diffusion.cpp -STABLEDIFFUSION_GGML_VERSION?=2f886889e6e8b78738d6b87f7191f6018557c551 +STABLEDIFFUSION_GGML_VERSION?=3f8527a46c54ecf4cb4ed6003da8e8982283c73c CMAKE_ARGS+=-DGGML_MAX_NAME=128 From 590512d9ebfbaaaaf9b57a18b844b16a3e446761 Mon Sep 17 00:00:00 2001 From: mudler-agent Date: Mon, 28 Sep 2026 13:13:31 +0200 Subject: [PATCH 15/15] feat(distributed): report worker version and show models in node inspector (#12328) * ci: bump Hugo from 0.146.3 to 0.166.0 The hugo-theme-relearn submodule was bumped to 9.1.x in #12096, which requires Hugo >= 0.165.0. The pinned 0.146.3 broke the docs site build with a template error in alias.html that could not evaluate the Locale field on langs.Language. Bump HUGO_VERSION to 0.166.0 (latest stable) to satisfy the theme minimum and resolve the alias.html template error. Assisted-by: nib:claude-sonnet-4.5 [bash] [read] [edit] * feat(distributed): report worker version and show models in node inspector Workers now send their LocalAI build version and git commit at registration. The controller stores them on BackendNode and exposes them through the existing node list/detail API responses. The node inspector side pane now fetches and renders the list of loaded models (name, state, in-flight) instead of showing only a count, matching what the node detail page already displays. --------- Co-authored-by: Ettore Di Giacinto --- core/cli/agent_worker.go | 3 ++ core/http/endpoints/localai/nodes.go | 7 ++++ core/http/react-ui/src/App.css | 7 ++++ .../src/components/nodes/NodeInspector.jsx | 38 ++++++++++++++++++- core/http/react-ui/src/pages/NodeDetail.jsx | 4 ++ core/services/nodes/registry.go | 5 +++ core/services/worker/registration.go | 3 ++ 7 files changed, 66 insertions(+), 1 deletion(-) diff --git a/core/cli/agent_worker.go b/core/cli/agent_worker.go index ced179e50..11f515c51 100644 --- a/core/cli/agent_worker.go +++ b/core/cli/agent_worker.go @@ -19,6 +19,7 @@ import ( "github.com/mudler/LocalAI/core/services/jobs" mcpRemote "github.com/mudler/LocalAI/core/services/mcp" "github.com/mudler/LocalAI/core/services/messaging" + "github.com/mudler/LocalAI/internal" "github.com/mudler/LocalAI/pkg/sanitize" "github.com/mudler/cogito" "github.com/mudler/cogito/clients" @@ -94,6 +95,8 @@ func (cmd *AgentWorkerCMD) Run(ctx *cliContext.Context) error { registrationBody := map[string]any{ "name": nodeName, "node_type": "agent", + "version": internal.Version, + "commit": internal.Commit, } if cmd.RegistrationToken != "" { registrationBody["token"] = cmd.RegistrationToken diff --git a/core/http/endpoints/localai/nodes.go b/core/http/endpoints/localai/nodes.go index 331c7203b..2036f3d6e 100644 --- a/core/http/endpoints/localai/nodes.go +++ b/core/http/endpoints/localai/nodes.go @@ -109,6 +109,11 @@ type RegisterNodeRequest struct { // VRAMBudget is the worker's operator-set VRAM cap ("80%" or "12GB"). The // registry resolves and enforces it against the raw reported VRAM. VRAMBudget string `json:"vram_budget,omitempty"` + // Version is the LocalAI build version reported by the worker at + // registration. Empty for workers registered before this field existed. + Version string `json:"version,omitempty"` + // Commit is the git commit hash the worker binary was built from. + Commit string `json:"commit,omitempty"` } // RegisterNodeEndpoint registers a new backend node. @@ -195,6 +200,8 @@ func RegisterNodeEndpoint(registry *nodes.NodeRegistry, expectedToken string, au Capability: req.Capability, MaxReplicasPerModel: maxReplicasPerModel, VRAMBudget: req.VRAMBudget, + Version: req.Version, + Commit: req.Commit, } ctx := c.Request().Context() diff --git a/core/http/react-ui/src/App.css b/core/http/react-ui/src/App.css index c6ea6e71d..5bafd472a 100644 --- a/core/http/react-ui/src/App.css +++ b/core/http/react-ui/src/App.css @@ -9965,6 +9965,13 @@ button.collapsible-header:focus-visible { .node-inspector__actions .btn { justify-content: center; min-width: 0; } .node-inspector__back { align-items: center; background: transparent; border: 0; color: var(--color-primary); cursor: pointer; display: flex; font: inherit; font-size: var(--text-xs); gap: 6px; max-width: 285px; overflow: hidden; padding: 3px 0; text-overflow: ellipsis; white-space: nowrap; } .node-inspector__back:focus-visible { border-radius: var(--radius-sm); outline: 2px solid var(--color-primary); outline-offset: 3px; } +.node-inspector__models { margin: 10px 0 0; } +.node-inspector__models > dd { margin: 0; } +.node-inspector__model-list { display: grid; gap: 4px; list-style: none; margin: 6px 0 0; padding: 0; } +.node-inspector__model-row { align-items: center; display: flex; flex-wrap: wrap; gap: 6px; font-size: var(--text-xs); } +.node-inspector__model-row .cell-mono { font-family: var(--font-mono); font-size: .625rem; overflow-wrap: anywhere; } +.node-inspector__model-row .state-pill { border-radius: var(--radius-full); font-size: .5625rem; font-weight: 600; padding: 1px 7px; text-transform: capitalize; } +.node-inspector__model-row .text-muted { font-size: .5625rem; } .model-inspector__backends { margin-top: 10px; } .model-inspector__nodes { display: grid; gap: 9px; } .model-inspector__node { background: var(--color-bg-tertiary); border: 1px solid var(--color-border-subtle); border-radius: var(--radius-md); padding: 10px; } diff --git a/core/http/react-ui/src/components/nodes/NodeInspector.jsx b/core/http/react-ui/src/components/nodes/NodeInspector.jsx index c14cc9fcb..51860bf5e 100644 --- a/core/http/react-ui/src/components/nodes/NodeInspector.jsx +++ b/core/http/react-ui/src/components/nodes/NodeInspector.jsx @@ -1,6 +1,6 @@ import { useEffect, useRef, useState } from 'react' import StatusPill from './StatusPill' -import { formatBytes, formatCapacity, timeAgo } from './nodeStatus' +import { formatBytes, formatCapacity, timeAgo, modelStateConfig } from './nodeStatus' import { nodesApi } from '../../utils/api' import { capacityReading, nodeLifecycleAction } from '../../utils/nodeFleet' import useInspectorDrawer from './useInspectorDrawer' @@ -24,6 +24,8 @@ function ResourceBar({ label, total, available, tone }) { export default function NodeInspector({ node, open, onClose, onApprove, onDrain, onResume, onBack, backLabel }) { const [backends, setBackends] = useState(null) const [backendError, setBackendError] = useState('') + const [models, setModels] = useState(null) + const [modelError, setModelError] = useState('') const nodeId = node?.id const backRef = useRef(null) const closeRef = useRef(null) @@ -48,6 +50,19 @@ export default function NodeInspector({ node, open, onClose, onApprove, onDrain, return () => { current = false } }, [open, nodeId]) + useEffect(() => { + if (!open || !nodeId) return undefined + let current = true + setModels(null) + setModelError('') + nodesApi.getModels(nodeId).then(data => { + if (current) setModels(Array.isArray(data) ? data : []) + }).catch(error => { + if (current) setModelError(error.message || 'Unable to load models') + }) + return () => { current = false } + }, [open, nodeId]) + if (!open || !node) return null const cpuKnown = node.cpu_logical_cores > 0 && Number.isFinite(node.cpu_usage_percent) && Number.isFinite(node.cpu_load_1) const disk = capacityReading(node.total_disk, node.available_disk) @@ -74,6 +89,7 @@ export default function NodeInspector({ node, open, onClose, onApprove, onDrain,

Node

{node.address || 'No address reported'} + {node.version || '—'} {timeAgo(node.last_heartbeat)}
{Object.keys(node.labels || {}).length ? Object.entries(node.labels).map(([key, value]) => {key}={value}) : No labels}
@@ -94,6 +110,26 @@ export default function NodeInspector({ node, open, onClose, onApprove, onDrain, {backendError ? {backendError} : backends === null ? 'Loading…' : `${backends.length} backend${backends.length === 1 ? '' : 's'}`} {node.in_flight_count ?? 0} +
+
Running models
+
+ {modelError ? {modelError} + : models === null ? Loading… + : models.length === 0 ? No models loaded + :
    + {models.map(model => { + const stCfg = modelStateConfig[model.state] || modelStateConfig.idle + return ( +
  • + {model.model_name} + {model.state} + {model.in_flight ?? 0} in flight +
  • + ) + })} +
} +
+