diff --git a/backend/cpp/audio-cpp/Makefile b/backend/cpp/audio-cpp/Makefile index e8d27eb52..64638b0c1 100644 --- a/backend/cpp/audio-cpp/Makefile +++ b/backend/cpp/audio-cpp/Makefile @@ -9,7 +9,7 @@ # recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean # rebuild and so the bump bot can see the pin. -AUDIO_CPP_VERSION?=94bd4656399180befc141b17bd6696bf84df0a9f +AUDIO_CPP_VERSION?=77491a33c589c53ff18add050095cf35647c8213 AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST)))) diff --git a/backend/cpp/ik-llama-cpp/Makefile b/backend/cpp/ik-llama-cpp/Makefile index d6bfd7490..d32465ec9 100644 --- a/backend/cpp/ik-llama-cpp/Makefile +++ b/backend/cpp/ik-llama-cpp/Makefile @@ -1,5 +1,5 @@ -IK_LLAMA_VERSION?=cdf232cc17e410e60c1bc3b85516c4a41199b662 +IK_LLAMA_VERSION?=ed27bf7ed25e637692e89cd341d802522a2cee8a LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp CMAKE_ARGS?= diff --git a/backend/cpp/llama-cpp/grpc-server.cpp b/backend/cpp/llama-cpp/grpc-server.cpp index 9a2240881..8b206612c 100644 --- a/backend/cpp/llama-cpp/grpc-server.cpp +++ b/backend/cpp/llama-cpp/grpc-server.cpp @@ -539,8 +539,10 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt // Initialize ctx_shift to false by default (can be overridden by options) params.ctx_shift = false; - // Initialize cache_ram_mib to -1 by default (no limit, can be overridden by options) - params.cache_ram_mib = -1; + // cache_ram_mib keeps llama.cpp's own default (8192 MiB) unless overridden by + // options. It used to be forced to -1 (no limit): since kv_unified and + // cache_idle_slots are on by default, every distinct prompt then leaves its + // slot state in host RAM and the backend grows without bound. // Initialize n_parallel to 1 by default (can be overridden by options) params.n_parallel = 1; // Initialize grpc_servers to empty (can be overridden by options) @@ -656,7 +658,7 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt try { params.cache_ram_mib = std::stoi(optval_str); } catch (const std::exception& e) { - // If conversion fails, keep default value (-1) + // If conversion fails, keep the default value } } } else if (!strcmp(optname, "parallel") || !strcmp(optname, "n_parallel")) { diff --git a/backend/go/crispasr/Makefile b/backend/go/crispasr/Makefile index f2155ffb9..bbdfef58c 100644 --- a/backend/go/crispasr/Makefile +++ b/backend/go/crispasr/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # CrispASR version (release tag) CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR -CRISPASR_VERSION?=013ae1624dc40ecf059065d577180722439f804e +CRISPASR_VERSION?=ec98831d0776ec8a16ccaf93955693eb7ecfbec3 SO_TARGET?=libgocrispasr.so CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF diff --git a/backend/go/omnivoice-cpp/Makefile b/backend/go/omnivoice-cpp/Makefile index 6d36ea9be..6c8539bd5 100644 --- a/backend/go/omnivoice-cpp/Makefile +++ b/backend/go/omnivoice-cpp/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # omnivoice.cpp version OMNIVOICE_REPO?=https://github.com/ServeurpersoCom/omnivoice.cpp -OMNIVOICE_VERSION?=8ab42195a05a9d48a3942b17568c1f3a876e133a +OMNIVOICE_VERSION?=ead199a2bc4c53a57cac90095ae049a111d9e98d SO_TARGET?=libgomnivoicecpp.so CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF diff --git a/backend/go/stablediffusion-ggml/Makefile b/backend/go/stablediffusion-ggml/Makefile index 6c5e70484..b2d9c9049 100644 --- a/backend/go/stablediffusion-ggml/Makefile +++ b/backend/go/stablediffusion-ggml/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # stablediffusion.cpp (ggml) STABLEDIFFUSION_GGML_REPO?=https://github.com/leejet/stable-diffusion.cpp -STABLEDIFFUSION_GGML_VERSION?=2f886889e6e8b78738d6b87f7191f6018557c551 +STABLEDIFFUSION_GGML_VERSION?=3f8527a46c54ecf4cb4ed6003da8e8982283c73c CMAKE_ARGS+=-DGGML_MAX_NAME=128 diff --git a/backend/go/stablediffusion-ggml/gosd.go b/backend/go/stablediffusion-ggml/gosd.go index e1567bd18..60f2efede 100644 --- a/backend/go/stablediffusion-ggml/gosd.go +++ b/backend/go/stablediffusion-ggml/gosd.go @@ -137,8 +137,8 @@ func (sd *SDGGML) Load(opts *pb.ModelOptions) error { // If it's an option path, we resolve absolute path from the model path if strings.Contains(op, ":") && strings.Contains(op, "path") { data := strings.Split(op, ":") - data[1] = filepath.Join(opts.ModelPath, data[1]) if err := utils.VerifyPath(data[1], opts.ModelPath); err == nil { + data[1] = filepath.Join(opts.ModelPath, data[1]) oo = append(oo, strings.Join(data, ":")) } } else { diff --git a/backend/go/trellis2cpp/trellis2.go b/backend/go/trellis2cpp/trellis2.go index addbf1bb0..c83480aa4 100644 --- a/backend/go/trellis2cpp/trellis2.go +++ b/backend/go/trellis2cpp/trellis2.go @@ -161,10 +161,10 @@ func resolveModels(modelFile, modelPath string, options []string) (modelSet, err continue } if !filepath.IsAbs(value) { - value = filepath.Join(modelPath, value) if err := utils.VerifyPath(value, modelPath); err != nil { return modelSet{}, fmt.Errorf("option %s: %w", key, err) } + value = filepath.Join(modelPath, value) } overrides[key] = value } diff --git a/backend/go/trellis2cpp/trellis2_test.go b/backend/go/trellis2cpp/trellis2_test.go index d492756e4..b221f00c5 100644 --- a/backend/go/trellis2cpp/trellis2_test.go +++ b/backend/go/trellis2cpp/trellis2_test.go @@ -134,9 +134,12 @@ var _ = Describe("resolveModels", func() { It("rejects option paths escaping the model directory", func() { touch(dir, fullSet...) + // The escaping file exists, so only the containment check can + // reject it; a missing file would fail for an unrelated reason. + touch(filepath.Dir(dir), "outside.gguf") _, err := resolveModels("ss_flow_f16.gguf", dir, []string{"dino_path:../outside.gguf"}) - Expect(err).To(HaveOccurred()) + Expect(err).To(MatchError(ContainSubstring("outside of trusted root"))) }) }) diff --git a/backend/python/sglang/backend.py b/backend/python/sglang/backend.py index 838d394bb..7936ddcf2 100644 --- a/backend/python/sglang/backend.py +++ b/backend/python/sglang/backend.py @@ -90,6 +90,19 @@ except Exception: _SEED_KEY = "sampling_seed" +# Engine.async_generate() only grew a require_reasoning keyword in sglang +# 0.5.13. The CPU build compiles v0.5.11 from source and the other profiles +# only set a >=0.5.11 floor, and async_generate() takes no **kwargs, so +# passing the keyword unconditionally fails every request with TypeError. +try: + import inspect as _inspect + _ASYNC_GENERATE_HAS_REQUIRE_REASONING = ( + "require_reasoning" in _inspect.signature(Engine.async_generate).parameters + ) +except Exception: + _ASYNC_GENERATE_HAS_REQUIRE_REASONING = False + + _ONE_DAY_IN_SECONDS = 60 * 60 * 24 # proto3 has no field presence, so an explicit 0 is indistinguishable from @@ -105,6 +118,12 @@ MAX_WORKERS = int(os.environ.get('PYTHON_GRPC_MAX_WORKERS', '1')) class BackendServicer(backend_pb2_grpc.BackendServicer): """gRPC servicer implementing the Backend service for sglang.""" + # Class-level default so a servicer used before LoadModel (e.g. in unit + # tests that construct it directly) doesn't AttributeError in + # _build_sampling_params. + thinking_budget: Optional[int] = None + reasoning_default: Optional[str] = None + def _parse_options(self, options_list) -> Dict[str, str]: opts: Dict[str, str] = {} for opt in options_list: @@ -114,6 +133,49 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): opts[key.strip()] = value.strip() return opts + @staticmethod + def _parse_thinking_budget(value) -> Optional[int]: + """Turn the `thinking_budget` model option into a positive int, or None. + + Options arrive as strings from the YAML `options:` list, but a value + like "5000.0" is a plausible thing to write, and a crash here would + take down LoadModel for the whole model. So: integral numbers are + accepted in any spelling ("512", "512.0"), anything else is ignored + with a warning instead of raising. Zero and negative budgets are + ignored too: sglang gives them no defined meaning, and turning + reasoning off is what `reasoning_default: off` is for. + """ + if value is None or str(value).strip() == "": + return None + raw = str(value).strip() + try: + number = float(raw) + except ValueError: + print(f"thinking_budget {raw!r} is not a number, ignoring it", file=sys.stderr) + return None + if not number.is_integer(): + print(f"thinking_budget {raw!r} is not a whole number of tokens, ignoring it", file=sys.stderr) + return None + if number <= 0: + print( + f"thinking_budget {raw!r} must be positive, ignoring it " + "(use reasoning_default:off to disable reasoning)", + file=sys.stderr, + ) + return None + return int(number) + + @staticmethod + def _strict_thinking_warning(thinking_budget: Optional[int], engine_kwargs: dict) -> Optional[str]: + """sglang only enforces the budget with enable_strict_thinking on; without + it the budget is silently ignored, so say so at load time.""" + if thinking_budget is not None and not engine_kwargs.get("enable_strict_thinking"): + return ( + f"thinking_budget={thinking_budget} is set but enable_strict_thinking is not " + "in engine_args; sglang will ignore the budget" + ) + return None + def _apply_engine_args(self, engine_kwargs: dict, engine_args_json: str) -> dict: """Merge user-supplied engine_args (JSON object) into the kwargs dict that will be forwarded to ``sglang.Engine`` (which constructs a @@ -230,6 +292,35 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): self.tool_parser_name: Optional[str] = opts.get("tool_parser") or None self.reasoning_parser_name: Optional[str] = opts.get("reasoning_parser") or None + # Fixed reasoning-length budget for every request on this model, in + # tokens. There is no protobuf field to carry a per-request + # custom_params blob, so this rides the same model-level `options:` + # mechanism as tool_parser/reasoning_parser above — mirroring how + # sglang's own `--preferred-sampling-params` is a server-wide + # default, not a per-request choice. Requires `enable_strict_thinking` + # in `engine_args:` (sglang >=0.5.12); without it sglang has no + # tokenizer-derived budget mechanism to enforce this against. + self.thinking_budget: Optional[int] = self._parse_thinking_budget( + opts.get("thinking_budget") + ) + + # Model-level default for whether the chat template opens a reasoning + # block, as "off" or "on". Rides the same `options:` mechanism as + # thinking_budget above. + # + # Why this is needed even though `reasoning_effort` exists: that one + # only reaches this backend when a *caller* sets it per request (the + # Go side turns it into Metadata["enable_thinking"]). As a model-level + # `parameters:` default it is silently dropped, so a config reading + # `reasoning_effort: none` still produces full reasoning on every + # request - the config says one thing and the model does another. + # + # A per-request value always wins; this only fills in the gap when the + # request says nothing. + self.reasoning_default: Optional[str] = ( + opts.get("reasoning_default") or "" + ).lower() or None + # Also hand the parser names to sglang's engine so its HTTP/OAI # paths work identically if someone hits the engine directly. if self.tool_parser_name: @@ -247,6 +338,10 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): print(f"engine_args error: {err}", file=sys.stderr) return backend_pb2.Result(success=False, message=str(err)) + warning = self._strict_thinking_warning(self.thinking_budget, engine_kwargs) + if warning: + print(warning, file=sys.stderr) + try: self.llm = Engine(**engine_kwargs) except Exception as err: @@ -362,8 +457,28 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): except json.JSONDecodeError: sampling_params["ebnf"] = grammar + if self.thinking_budget is not None: + sampling_params["custom_params"] = {"thinking_budget": self.thinking_budget} + return sampling_params + def _thinking_default(self, request) -> Optional[bool]: + """Whether this request should render with reasoning on, off, or unset. + + Per-request ``Metadata["enable_thinking"]`` wins; the model-level + ``reasoning_default`` option fills in when the request is silent. + Returns None when neither says anything, leaving template behaviour + untouched. + """ + wanted = request.Metadata.get("enable_thinking", "").lower() + if wanted in ("true", "false"): + return wanted == "true" + if self.reasoning_default == "off": + return False + if self.reasoning_default == "on": + return True + return None + def _build_prompt(self, request) -> str: prompt = request.Prompt if prompt or not request.UseTokenizerTemplate or not request.Messages: @@ -384,9 +499,9 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): template_kwargs["tools"] = json.loads(request.Tools) except json.JSONDecodeError: pass - _thinking = request.Metadata.get("enable_thinking", "").lower() - if _thinking in ("true", "false"): - template_kwargs["enable_thinking"] = (_thinking == "true") + _thinking = self._thinking_default(request) + if _thinking is not None: + template_kwargs["enable_thinking"] = _thinking # sglang locates the attached images/videos by scanning the rendered # prompt for the model's own media token, so the template has to be @@ -438,12 +553,19 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): there files the answer as reasoning and leaves content empty. sglang's own server keeps the two apart for the same reason — its grammar backend owns the reasoning prefix when a reasoning parser is set. + + Returns a ``(parser, forced)`` pair. ``forced`` is also the signal + ``_predict`` passes as ``Engine.async_generate(require_reasoning=...)``: + sglang's own OpenAI server derives that flag from per-template + config (``ChatServing._get_reasoning_from_request``); this backend + has no template manager, so the same prompt-suffix heuristic that + already decides parser forcing doubles as that signal. """ if grammar_constrained: prompt = "" if not (HAS_REASONING_PARSERS and self.reasoning_parser_name): - return None + return None, False kwargs = { "model_type": self.reasoning_parser_name, @@ -453,10 +575,12 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): parser = ReasoningParser(**kwargs) except Exception as e: print(f"ReasoningParser init failed: {e!r}", file=sys.stderr) - return None + return None, False + forced = False start = getattr(getattr(parser, "detector", None), "think_start_token", None) if start and prompt and prompt.rstrip().endswith(start): + forced = True try: parser = ReasoningParser(force_reasoning=True, **kwargs) except TypeError: @@ -469,10 +593,16 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): file=sys.stderr, ) - return parser + return parser, forced def _make_parsers(self, request, prompt: str = ""): - """Construct fresh per-request parser instances (stateful).""" + """Construct fresh per-request parser instances (stateful). + + Also returns ``require_reasoning`` (see ``_new_reasoning_parser``), + which ``_predict`` forwards to ``Engine.async_generate()`` so + sglang's ``--enable-strict-thinking`` grammar backend knows this + request is in a reasoning block. + """ tool_parser = None if HAS_TOOL_PARSERS and self.tool_parser_name and request.Tools: @@ -485,23 +615,27 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): except Exception as e: print(f"FunctionCallParser init failed: {e!r}", file=sys.stderr) - reasoning_parser = self._new_reasoning_parser( + reasoning_parser, require_reasoning = self._new_reasoning_parser( True, prompt, bool(getattr(request, "Grammar", "")), ) - return tool_parser, reasoning_parser + return tool_parser, reasoning_parser, require_reasoning async def _predict(self, request, context, streaming: bool = False): sampling_params = self._build_sampling_params(request) prompt = self._build_prompt(request) - tool_parser, reasoning_parser = self._make_parsers(request, prompt) + tool_parser, reasoning_parser, require_reasoning = self._make_parsers(request, prompt) image_data = list(request.Images) if request.Images else None video_data = list(request.Videos) if request.Videos else None # Kick off streaming generation. We always use stream=True so the # non-stream path still gets parser coverage on the final text. + generate_kwargs = {} + if _ASYNC_GENERATE_HAS_REQUIRE_REASONING: + generate_kwargs["require_reasoning"] = require_reasoning + try: iterator = await self.llm.async_generate( prompt=prompt, @@ -509,6 +643,7 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): image_data=image_data, video_data=video_data, stream=True, + **generate_kwargs, ) except Exception as e: print(f"sglang async_generate failed: {e!r}", file=sys.stderr) @@ -591,7 +726,7 @@ class BackendServicer(backend_pb2_grpc.BackendServicer): final_tool_calls: List[backend_pb2.ToolCallDelta] = [] if not streaming: - final_reasoning_parser = self._new_reasoning_parser( + final_reasoning_parser, _ = self._new_reasoning_parser( False, prompt, bool(getattr(request, "Grammar", "")), ) diff --git a/backend/python/sglang/test.py b/backend/python/sglang/test.py index 51158791c..1b1a10d23 100644 --- a/backend/python/sglang/test.py +++ b/backend/python/sglang/test.py @@ -9,6 +9,13 @@ because ``_apply_engine_args`` validates keys against ``ServerArgs`` import unittest + +def _request(metadata=None): + """Minimal stand-in for a PredictOptions request in reasoning tests.""" + from types import SimpleNamespace + + return SimpleNamespace(Metadata=metadata or {}) + class TestSglangHelpers(unittest.TestCase): """Tests for the pure helpers on BackendServicer (no gRPC, no engine).""" @@ -170,13 +177,17 @@ class TestSglangHelpers(unittest.TestCase): # What the model actually emits when the prompt ends in "". completion = "adding two and two4" - forced = servicer._new_reasoning_parser(False, prompt="user: hi\n\n") + forced, require_reasoning = servicer._new_reasoning_parser( + False, prompt="user: hi\n\n" + ) + self.assertTrue(require_reasoning) reasoning, content = forced.parse_non_stream(completion) self.assertEqual(reasoning, "adding two and two") self.assertEqual(content, "4") # No prefilled tag in the prompt: detector default, unchanged behaviour. - unforced = servicer._new_reasoning_parser(False, prompt="user: hi\n") + unforced, require_reasoning = servicer._new_reasoning_parser(False, prompt="user: hi\n") + self.assertFalse(require_reasoning) reasoning, content = unforced.parse_non_stream(completion) self.assertFalse(reasoning) self.assertEqual(content, completion) @@ -187,7 +198,8 @@ class TestSglangHelpers(unittest.TestCase): servicer = self._servicer() servicer.reasoning_parser_name = "qwen3" - parser = servicer._new_reasoning_parser(False, prompt="user: primes?\n") + parser, require_reasoning = servicer._new_reasoning_parser(False, prompt="user: primes?\n") + self.assertFalse(require_reasoning) reasoning, content = parser.parse_non_stream("2,3,5,7,11") self.assertFalse(reasoning) self.assertEqual(content, "2,3,5,7,11") @@ -200,9 +212,10 @@ class TestSglangHelpers(unittest.TestCase): servicer.reasoning_parser_name = "qwen3" schema_out = '{"findings": [{"line": 42, "issue": "off-by-one"}]}' - parser = servicer._new_reasoning_parser( + parser, require_reasoning = servicer._new_reasoning_parser( False, prompt="audit this\n\n", grammar_constrained=True, ) + self.assertFalse(require_reasoning) reasoning, content = parser.parse_non_stream(schema_out) self.assertFalse(reasoning) self.assertEqual(content, schema_out) @@ -210,7 +223,110 @@ class TestSglangHelpers(unittest.TestCase): def test_reasoning_parser_absent_without_configured_parser(self): servicer = self._servicer() servicer.reasoning_parser_name = None - self.assertIsNone(servicer._new_reasoning_parser(False, prompt="")) + parser, require_reasoning = servicer._new_reasoning_parser(False, prompt="") + self.assertIsNone(parser) + self.assertFalse(require_reasoning) + + def test_reasoning_default_off_applies_when_request_is_silent(self): + """A model configured with reasoning_default:off must render with + thinking disabled even when the request carries no enable_thinking - + that is the whole point: `parameters: reasoning_effort:` never + reaches this backend, so without this the config lies about the + default.""" + servicer = self._servicer() + servicer.reasoning_default = "off" + self.assertIs(servicer._thinking_default(_request(metadata={})), False) + + def test_request_metadata_overrides_reasoning_default(self): + """A per-request value always wins over the model-level default - + in both directions.""" + servicer = self._servicer() + servicer.reasoning_default = "off" + self.assertIs( + servicer._thinking_default(_request(metadata={"enable_thinking": "true"})), + True, + ) + servicer.reasoning_default = "on" + self.assertIs( + servicer._thinking_default(_request(metadata={"enable_thinking": "false"})), + False, + ) + + def test_no_reasoning_default_leaves_template_untouched(self): + """Unconfigured must stay unconfigured: returning None means the + backend adds no enable_thinking kwarg at all, so the template keeps + whatever default it ships with.""" + servicer = self._servicer() + self.assertIsNone(servicer._thinking_default(_request(metadata={}))) + + def test_thinking_budget_added_to_sampling_params_as_custom_params(self): + """The model-level thinking_budget option (set from LoadModel's + Options, mirroring tool_parser/reasoning_parser) must ride along as + sampling_params['custom_params']['thinking_budget'] on every + request — that's the only field sglang's --enable-strict-thinking + grammar backend reads to bound the reasoning length.""" + from types import SimpleNamespace + + servicer = self._servicer() + servicer.thinking_budget = 512 + request = SimpleNamespace( + Temperature=0.7, N=0, PresencePenalty=0, FrequencyPenalty=0, + RepetitionPenalty=0, TopP=0, TopK=0, MinP=0, Seed=0, + StopPrompts=[], StopTokenIds=[], IgnoreEOS=False, Tokens=0, + MinTokens=0, SkipSpecialTokens=False, Grammar="", + ) + params = servicer._build_sampling_params(request) + self.assertEqual(params["custom_params"], {"thinking_budget": 512}) + + def test_no_thinking_budget_means_no_custom_params_key(self): + """Unconfigured is unconfigured: no thinking_budget option must not + add an empty/None custom_params that could clobber a sglang-side + --preferred-sampling-params default (see sglang#40634).""" + from types import SimpleNamespace + + servicer = self._servicer() + request = SimpleNamespace( + Temperature=0.7, N=0, PresencePenalty=0, FrequencyPenalty=0, + RepetitionPenalty=0, TopP=0, TopK=0, MinP=0, Seed=0, + StopPrompts=[], StopTokenIds=[], IgnoreEOS=False, Tokens=0, + MinTokens=0, SkipSpecialTokens=False, Grammar="", + ) + params = servicer._build_sampling_params(request) + self.assertNotIn("custom_params", params) + + def test_thinking_budget_accepts_integral_spellings(self): + """YAML options arrive as strings; "512" and "512.0" both mean 512.""" + servicer = self._servicer() + self.assertEqual(servicer._parse_thinking_budget("512"), 512) + self.assertEqual(servicer._parse_thinking_budget("512.0"), 512) + self.assertEqual(servicer._parse_thinking_budget(" 64 "), 64) + self.assertEqual(servicer._parse_thinking_budget(256), 256) + + def test_thinking_budget_unset_is_none(self): + servicer = self._servicer() + self.assertIsNone(servicer._parse_thinking_budget(None)) + self.assertIsNone(servicer._parse_thinking_budget("")) + + def test_thinking_budget_zero_and_negative_are_ignored(self): + """No defined meaning in sglang -- ignored, not passed through.""" + servicer = self._servicer() + self.assertIsNone(servicer._parse_thinking_budget("0")) + self.assertIsNone(servicer._parse_thinking_budget("-100")) + + def test_thinking_budget_non_integer_does_not_raise(self): + """A bad value must not crash LoadModel for the whole model.""" + servicer = self._servicer() + self.assertIsNone(servicer._parse_thinking_budget("12.5")) + self.assertIsNone(servicer._parse_thinking_budget("lots")) + + def test_warns_when_budget_set_without_strict_thinking(self): + servicer = self._servicer() + self.assertIn( + "enable_strict_thinking", + servicer._strict_thinking_warning(512, {"model_path": "x"}), + ) + self.assertIsNone(servicer._strict_thinking_warning(512, {"enable_strict_thinking": True})) + self.assertIsNone(servicer._strict_thinking_warning(None, {})) def test_explicit_zero_temperature_and_seed_are_preserved(self): """Temperature=0 is greedy decoding and 0 is a valid seed — neither is diff --git a/core/application/distributed.go b/core/application/distributed.go index c3ec192b2..bff509648 100644 --- a/core/application/distributed.go +++ b/core/application/distributed.go @@ -623,9 +623,13 @@ func initDistributed(cfg *config.ApplicationConfig, authDB *gorm.DB, configLoade // All dependencies ready — build SmartRouter with all options at once var conflictResolver nodes.ConcurrencyConflictResolver var pinnedResolver nodes.PinnedModelResolver + var modelFiles func(string) []string if configLoader != nil { conflictResolver = configLoader pinnedResolver = configLoader + if cfg.SystemState != nil { + modelFiles = declaredModelFiles(configLoader, cfg.SystemState.Model.ModelsPath) + } } modelCleanup := nodes.NewModelCleanupService(registry, remoteUnloader) // Absence is stamped on by distributedSchedulerOptions rather than written @@ -645,6 +649,7 @@ func initDistributed(cfg *config.ApplicationConfig, authDB *gorm.DB, configLoade DataPath: cfg.DataPath, ConflictResolver: conflictResolver, PinnedResolver: pinnedResolver, + ModelFiles: modelFiles, PrefixProvider: prefixProvider, PrefixConfig: prefixCfg, Pressure: pressure, diff --git a/core/application/model_files.go b/core/application/model_files.go new file mode 100644 index 000000000..53bd069df --- /dev/null +++ b/core/application/model_files.go @@ -0,0 +1,29 @@ +package application + +import ( + "path/filepath" + + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/gallery" + "github.com/mudler/LocalAI/pkg/utils" +) + +// declaredModelFiles resolves the files a model needs on disk beyond the ones +// its config names: what its gallery install or import declared under +// `files:`, and what the config itself lists under download_files. The +// distributed router stages these to workers, which cannot see the frontend's +// models directory. +func declaredModelFiles(configLoader *config.ModelConfigLoader, modelsPath string) func(modelName string) []string { + return func(modelName string) []string { + files := gallery.InstalledModelFiles(modelsPath, modelName) + if cfg, ok := configLoader.GetModelConfig(modelName); ok { + for _, f := range cfg.DownloadFiles { + if utils.VerifyPath(f.Filename, modelsPath) != nil { + continue + } + files = append(files, filepath.Join(modelsPath, f.Filename)) + } + } + return files + } +} diff --git a/core/application/model_files_test.go b/core/application/model_files_test.go new file mode 100644 index 000000000..f0ab2bf5e --- /dev/null +++ b/core/application/model_files_test.go @@ -0,0 +1,41 @@ +package application + +import ( + "os" + "path/filepath" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/gallery" +) + +var _ = Describe("declaredModelFiles", func() { + It("combines the gallery install's files with the config's download_files", func() { + modelsPath := GinkgoT().TempDir() + Expect(os.WriteFile(filepath.Join(modelsPath, "big.yaml"), []byte(` +name: big +backend: llama-cpp +parameters: + model: big/Big-00001-of-00002.gguf +download_files: + - filename: big/extra.bin + uri: https://example.com/extra.bin +`), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("big")), []byte(` +files: + - filename: big/Big-00001-of-00002.gguf + - filename: big/Big-00002-of-00002.gguf +`), 0o644)).To(Succeed()) + + loader := config.NewModelConfigLoader(modelsPath) + Expect(loader.LoadModelConfigsFromPath(modelsPath)).To(Succeed()) + + Expect(declaredModelFiles(loader, modelsPath)("big")).To(ConsistOf( + filepath.Join(modelsPath, "big/Big-00001-of-00002.gguf"), + filepath.Join(modelsPath, "big/Big-00002-of-00002.gguf"), + filepath.Join(modelsPath, "big/extra.bin"), + )) + }) +}) diff --git a/core/backend/tts.go b/core/backend/tts.go index 3451eb471..1172fb8f2 100644 --- a/core/backend/tts.go +++ b/core/backend/tts.go @@ -88,7 +88,7 @@ func ModelTTS( // a FS path mp := filepath.Join(loader.ModelPath, modelConfig.Model) if _, err := os.Stat(mp); err == nil { - if err := utils.VerifyPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil { + if err := utils.VerifyResolvedPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil { return "", nil, err } modelPath = mp @@ -189,7 +189,7 @@ func ModelTTSStream( // a FS path mp := filepath.Join(loader.ModelPath, modelConfig.Model) if _, err := os.Stat(mp); err == nil { - if err := utils.VerifyPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil { + if err := utils.VerifyResolvedPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil { return err } modelPath = mp diff --git a/core/cli/agent_worker.go b/core/cli/agent_worker.go index acdfdf81c..fbf380508 100644 --- a/core/cli/agent_worker.go +++ b/core/cli/agent_worker.go @@ -20,6 +20,7 @@ import ( "github.com/mudler/LocalAI/core/services/jobs" mcpRemote "github.com/mudler/LocalAI/core/services/mcp" "github.com/mudler/LocalAI/core/services/messaging" + "github.com/mudler/LocalAI/internal" "github.com/mudler/cogito" "github.com/mudler/cogito/clients" "github.com/mudler/xlog" @@ -163,6 +164,8 @@ func (cmd *AgentWorkerCMD) Run(ctx *cliContext.Context) error { registrationBody := map[string]any{ "name": nodeName, "node_type": "agent", + "version": internal.Version, + "commit": internal.Commit, } if cmd.RegistrationToken != "" { registrationBody["token"] = cmd.RegistrationToken diff --git a/core/gallery/delete_model_paths_test.go b/core/gallery/delete_model_paths_test.go new file mode 100644 index 000000000..a61cfe636 --- /dev/null +++ b/core/gallery/delete_model_paths_test.go @@ -0,0 +1,72 @@ +package gallery_test + +import ( + "os" + "path/filepath" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + "github.com/mudler/LocalAI/core/gallery" + "github.com/mudler/LocalAI/pkg/system" +) + +// DeleteModelFromSystem removes files named by a model name and by the +// model's gallery file. Neither may reach outside the models directory: the +// name can come from an API caller or from an assistant tool call, and the +// gallery file is a YAML file on disk. +var _ = Describe("DeleteModelFromSystem path containment", func() { + var ( + root string + modelsPath string + outside string + state *system.SystemState + ) + + BeforeEach(func() { + root = GinkgoT().TempDir() + modelsPath = filepath.Join(root, "models") + outside = filepath.Join(root, "outside") + Expect(os.MkdirAll(modelsPath, 0o755)).To(Succeed()) + Expect(os.MkdirAll(outside, 0o755)).To(Succeed()) + var err error + state, err = system.GetSystemState(system.WithModelPath(modelsPath)) + Expect(err).ToNot(HaveOccurred()) + }) + + It("refuses a model name that escapes the models directory", func() { + victim := filepath.Join(outside, "victim.yaml") + Expect(os.WriteFile(victim, []byte("name: victim\n"), 0o644)).To(Succeed()) + + Expect(gallery.DeleteModelFromSystem(state, "../outside/victim")).ToNot(Succeed()) + Expect(victim).To(BeARegularFile()) + }) + + It("does not remove gallery-declared files outside the models directory", func() { + secret := filepath.Join(outside, "secret.bin") + Expect(os.WriteFile(secret, []byte("x"), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelsPath, "m.yaml"), []byte("name: m\n"), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("m")), []byte(` +files: + - filename: ../outside/secret.bin +`), 0o644)).To(Succeed()) + + _ = gallery.DeleteModelFromSystem(state, "m") + Expect(secret).To(BeARegularFile()) + }) + + It("still deletes a normal model and its declared files", func() { + weights := filepath.Join(modelsPath, "m", "w.gguf") + Expect(os.MkdirAll(filepath.Dir(weights), 0o755)).To(Succeed()) + Expect(os.WriteFile(weights, []byte("w"), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelsPath, "m.yaml"), []byte("name: m\nparameters:\n model: m/w.gguf\n"), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("m")), []byte(` +files: + - filename: m/w.gguf +`), 0o644)).To(Succeed()) + + Expect(gallery.DeleteModelFromSystem(state, "m")).To(Succeed()) + Expect(weights).ToNot(BeAnExistingFile()) + Expect(filepath.Join(modelsPath, "m.yaml")).ToNot(BeAnExistingFile()) + }) +}) diff --git a/core/gallery/installed_files.go b/core/gallery/installed_files.go new file mode 100644 index 000000000..c7f6773dc --- /dev/null +++ b/core/gallery/installed_files.go @@ -0,0 +1,46 @@ +package gallery + +import ( + "os" + "path/filepath" + "strings" + + "github.com/mudler/LocalAI/pkg/utils" + "github.com/mudler/xlog" +) + +// InstalledModelFiles returns the absolute paths of the files that the install +// of model name declared (the entry's `files:`), as recorded in its gallery +// file. A model config names only the file a backend opens first, while a +// backend can read more by itself (llama.cpp opens the other shards of a split +// GGUF by name), so this is the complete list of what the model needs on disk. +// It returns nil for a model that was not installed from a gallery or import. +func InstalledModelFiles(modelsPath, name string) []string { + // Model names can hold path separators; the gallery file flattens them + // the same way listModelFiles does. + rel := galleryFileName(strings.ReplaceAll(name, string(os.PathSeparator), "__")) + if err := utils.VerifyPath(rel, modelsPath); err != nil { + return nil + } + galleryFile := filepath.Join(modelsPath, rel) + if _, err := os.Stat(galleryFile); err != nil { + return nil + } + cfg, err := ReadConfigFile[ModelConfig](galleryFile) + if err != nil { + xlog.Warn("Failed to read gallery file for installed model files", "model", name, "file", galleryFile, "error", err) + return nil + } + + files := make([]string, 0, len(cfg.Files)) + for _, f := range cfg.Files { + // VerifyPath joins its argument onto modelsPath itself, so it must + // get the relative name; an absolute path would always pass. + if err := utils.VerifyPath(f.Filename, modelsPath); err != nil { + xlog.Warn("Ignoring declared model file outside the models path", "model", name, "file", f.Filename) + continue + } + files = append(files, filepath.Join(modelsPath, f.Filename)) + } + return files +} diff --git a/core/gallery/installed_files_test.go b/core/gallery/installed_files_test.go new file mode 100644 index 000000000..d376ef1e6 --- /dev/null +++ b/core/gallery/installed_files_test.go @@ -0,0 +1,53 @@ +package gallery_test + +import ( + "os" + "path/filepath" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + "github.com/mudler/LocalAI/core/gallery" +) + +var _ = Describe("InstalledModelFiles", func() { + var modelsPath string + + BeforeEach(func() { + modelsPath = GinkgoT().TempDir() + }) + + writeGalleryFile := func(name, body string) { + Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName(name)), []byte(body), 0o644)).To(Succeed()) + } + + It("returns the files the install declared, under the models path", func() { + writeGalleryFile("big", ` +name: big +files: + - filename: llama-cpp/models/big/Big-00001-of-00002.gguf + uri: huggingface://org/repo/Big-00001-of-00002.gguf + - filename: llama-cpp/models/big/Big-00002-of-00002.gguf + uri: huggingface://org/repo/Big-00002-of-00002.gguf +`) + Expect(gallery.InstalledModelFiles(modelsPath, "big")).To(Equal([]string{ + filepath.Join(modelsPath, "llama-cpp/models/big/Big-00001-of-00002.gguf"), + filepath.Join(modelsPath, "llama-cpp/models/big/Big-00002-of-00002.gguf"), + })) + }) + + It("drops entries that escape the models path", func() { + writeGalleryFile("evil", ` +files: + - filename: ../outside.gguf + - filename: ok.gguf +`) + Expect(gallery.InstalledModelFiles(modelsPath, "evil")).To(Equal([]string{ + filepath.Join(modelsPath, "ok.gguf"), + })) + }) + + It("returns nothing for a model that was not installed from a gallery", func() { + Expect(gallery.InstalledModelFiles(modelsPath, "handwritten")).To(BeEmpty()) + }) +}) diff --git a/core/gallery/models.go b/core/gallery/models.go index c787dcde4..6d4ceb346 100644 --- a/core/gallery/models.go +++ b/core/gallery/models.go @@ -808,8 +808,11 @@ func GetLocalModelConfiguration(basePath string, name string) (*ModelConfig, err func listModelFiles(systemState *system.SystemState, name string) ([]string, error) { + // VerifyPath joins its argument onto the models path itself, so every + // check below passes the relative name: an already-joined absolute path + // always lands inside the base and the check would pass anything. configFile := filepath.Join(systemState.Model.ModelsPath, fmt.Sprintf("%s.yaml", name)) - if err := utils.VerifyPath(configFile, systemState.Model.ModelsPath); err != nil { + if err := utils.VerifyPath(fmt.Sprintf("%s.yaml", name), systemState.Model.ModelsPath); err != nil { return nil, fmt.Errorf("failed to verify path %s: %w", configFile, err) } @@ -817,7 +820,7 @@ func listModelFiles(systemState *system.SystemState, name string) ([]string, err name = strings.ReplaceAll(name, string(os.PathSeparator), "__") galleryFile := filepath.Join(systemState.Model.ModelsPath, galleryFileName(name)) - if err := utils.VerifyPath(galleryFile, systemState.Model.ModelsPath); err != nil { + if err := utils.VerifyPath(galleryFileName(name), systemState.Model.ModelsPath); err != nil { return nil, fmt.Errorf("failed to verify path %s: %w", galleryFile, err) } @@ -847,7 +850,7 @@ func listModelFiles(systemState *system.SystemState, name string) ([]string, err if err == nil && galleryconfig != nil { for _, f := range galleryconfig.Files { fullPath := filepath.Join(systemState.Model.ModelsPath, f.Filename) - if err := utils.VerifyPath(fullPath, systemState.Model.ModelsPath); err != nil { + if err := utils.VerifyPath(f.Filename, systemState.Model.ModelsPath); err != nil { return allFiles, fmt.Errorf("failed to verify path %s: %w", fullPath, err) } allFiles = append(allFiles, fullPath) @@ -858,7 +861,7 @@ func listModelFiles(systemState *system.SystemState, name string) ([]string, err for _, f := range additionalFiles { fullPath := filepath.Join(filepath.Join(systemState.Model.ModelsPath, f)) - if err := utils.VerifyPath(fullPath, systemState.Model.ModelsPath); err != nil { + if err := utils.VerifyPath(f, systemState.Model.ModelsPath); err != nil { return allFiles, fmt.Errorf("failed to verify path %s: %w", fullPath, err) } allFiles = append(allFiles, fullPath) diff --git a/core/http/endpoints/localai/nodes.go b/core/http/endpoints/localai/nodes.go index bf284c847..4d307b3f3 100644 --- a/core/http/endpoints/localai/nodes.go +++ b/core/http/endpoints/localai/nodes.go @@ -114,6 +114,11 @@ type RegisterNodeRequest struct { // VRAMBudget is the worker's operator-set VRAM cap ("80%" or "12GB"). The // registry resolves and enforces it against the raw reported VRAM. VRAMBudget string `json:"vram_budget,omitempty"` + // Version is the LocalAI build version reported by the worker at + // registration. Empty for workers registered before this field existed. + Version string `json:"version,omitempty"` + // Commit is the git commit hash the worker binary was built from. + Commit string `json:"commit,omitempty"` } // RegisterNodeEndpoint registers a new backend node. @@ -191,6 +196,8 @@ func RegisterNodeEndpoint(registry *nodes.NodeRegistry, expectedToken string, au Capability: req.Capability, MaxReplicasPerModel: maxReplicasPerModel, VRAMBudget: req.VRAMBudget, + Version: req.Version, + Commit: req.Commit, } ctx := c.Request().Context() diff --git a/core/http/react-ui/src/App.css b/core/http/react-ui/src/App.css index c3d4c30b3..5cd69eca7 100644 --- a/core/http/react-ui/src/App.css +++ b/core/http/react-ui/src/App.css @@ -9966,6 +9966,13 @@ button.collapsible-header:focus-visible { .node-inspector__actions .btn { justify-content: center; min-width: 0; } .node-inspector__back { align-items: center; background: transparent; border: 0; color: var(--color-primary); cursor: pointer; display: flex; font: inherit; font-size: var(--text-xs); gap: 6px; max-width: 285px; overflow: hidden; padding: 3px 0; text-overflow: ellipsis; white-space: nowrap; } .node-inspector__back:focus-visible { border-radius: var(--radius-sm); outline: 2px solid var(--color-primary); outline-offset: 3px; } +.node-inspector__models { margin: 10px 0 0; } +.node-inspector__models > dd { margin: 0; } +.node-inspector__model-list { display: grid; gap: 4px; list-style: none; margin: 6px 0 0; padding: 0; } +.node-inspector__model-row { align-items: center; display: flex; flex-wrap: wrap; gap: 6px; font-size: var(--text-xs); } +.node-inspector__model-row .cell-mono { font-family: var(--font-mono); font-size: .625rem; overflow-wrap: anywhere; } +.node-inspector__model-row .state-pill { border-radius: var(--radius-full); font-size: .5625rem; font-weight: 600; padding: 1px 7px; text-transform: capitalize; } +.node-inspector__model-row .text-muted { font-size: .5625rem; } .model-inspector__backends { margin-top: 10px; } .model-inspector__nodes { display: grid; gap: 9px; } .model-inspector__node { background: var(--color-bg-tertiary); border: 1px solid var(--color-border-subtle); border-radius: var(--radius-md); padding: 10px; } diff --git a/core/http/react-ui/src/components/nodes/NodeInspector.jsx b/core/http/react-ui/src/components/nodes/NodeInspector.jsx index c14cc9fcb..51860bf5e 100644 --- a/core/http/react-ui/src/components/nodes/NodeInspector.jsx +++ b/core/http/react-ui/src/components/nodes/NodeInspector.jsx @@ -1,6 +1,6 @@ import { useEffect, useRef, useState } from 'react' import StatusPill from './StatusPill' -import { formatBytes, formatCapacity, timeAgo } from './nodeStatus' +import { formatBytes, formatCapacity, timeAgo, modelStateConfig } from './nodeStatus' import { nodesApi } from '../../utils/api' import { capacityReading, nodeLifecycleAction } from '../../utils/nodeFleet' import useInspectorDrawer from './useInspectorDrawer' @@ -24,6 +24,8 @@ function ResourceBar({ label, total, available, tone }) { export default function NodeInspector({ node, open, onClose, onApprove, onDrain, onResume, onBack, backLabel }) { const [backends, setBackends] = useState(null) const [backendError, setBackendError] = useState('') + const [models, setModels] = useState(null) + const [modelError, setModelError] = useState('') const nodeId = node?.id const backRef = useRef(null) const closeRef = useRef(null) @@ -48,6 +50,19 @@ export default function NodeInspector({ node, open, onClose, onApprove, onDrain, return () => { current = false } }, [open, nodeId]) + useEffect(() => { + if (!open || !nodeId) return undefined + let current = true + setModels(null) + setModelError('') + nodesApi.getModels(nodeId).then(data => { + if (current) setModels(Array.isArray(data) ? data : []) + }).catch(error => { + if (current) setModelError(error.message || 'Unable to load models') + }) + return () => { current = false } + }, [open, nodeId]) + if (!open || !node) return null const cpuKnown = node.cpu_logical_cores > 0 && Number.isFinite(node.cpu_usage_percent) && Number.isFinite(node.cpu_load_1) const disk = capacityReading(node.total_disk, node.available_disk) @@ -74,6 +89,7 @@ export default function NodeInspector({ node, open, onClose, onApprove, onDrain,

Node

{node.address || 'No address reported'} + {node.version || '—'} {timeAgo(node.last_heartbeat)}
{Object.keys(node.labels || {}).length ? Object.entries(node.labels).map(([key, value]) => {key}={value}) : No labels}
@@ -94,6 +110,26 @@ export default function NodeInspector({ node, open, onClose, onApprove, onDrain, {backendError ? {backendError} : backends === null ? 'Loading…' : `${backends.length} backend${backends.length === 1 ? '' : 's'}`} {node.in_flight_count ?? 0} +
+
Running models
+
+ {modelError ? {modelError} + : models === null ? Loading… + : models.length === 0 ? No models loaded + :
    + {models.map(model => { + const stCfg = modelStateConfig[model.state] || modelStateConfig.idle + return ( +
  • + {model.model_name} + {model.state} + {model.in_flight ?? 0} in flight +
  • + ) + })} +
} +
+