diff --git a/backend/python/sglang/backend.py b/backend/python/sglang/backend.py
index 838d394bb..7936ddcf2 100644
--- a/backend/python/sglang/backend.py
+++ b/backend/python/sglang/backend.py
@@ -90,6 +90,19 @@ except Exception:
_SEED_KEY = "sampling_seed"
+# Engine.async_generate() only grew a require_reasoning keyword in sglang
+# 0.5.13. The CPU build compiles v0.5.11 from source and the other profiles
+# only set a >=0.5.11 floor, and async_generate() takes no **kwargs, so
+# passing the keyword unconditionally fails every request with TypeError.
+try:
+ import inspect as _inspect
+ _ASYNC_GENERATE_HAS_REQUIRE_REASONING = (
+ "require_reasoning" in _inspect.signature(Engine.async_generate).parameters
+ )
+except Exception:
+ _ASYNC_GENERATE_HAS_REQUIRE_REASONING = False
+
+
_ONE_DAY_IN_SECONDS = 60 * 60 * 24
# proto3 has no field presence, so an explicit 0 is indistinguishable from
@@ -105,6 +118,12 @@ MAX_WORKERS = int(os.environ.get('PYTHON_GRPC_MAX_WORKERS', '1'))
class BackendServicer(backend_pb2_grpc.BackendServicer):
"""gRPC servicer implementing the Backend service for sglang."""
+ # Class-level default so a servicer used before LoadModel (e.g. in unit
+ # tests that construct it directly) doesn't AttributeError in
+ # _build_sampling_params.
+ thinking_budget: Optional[int] = None
+ reasoning_default: Optional[str] = None
+
def _parse_options(self, options_list) -> Dict[str, str]:
opts: Dict[str, str] = {}
for opt in options_list:
@@ -114,6 +133,49 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
opts[key.strip()] = value.strip()
return opts
+ @staticmethod
+ def _parse_thinking_budget(value) -> Optional[int]:
+ """Turn the `thinking_budget` model option into a positive int, or None.
+
+ Options arrive as strings from the YAML `options:` list, but a value
+ like "5000.0" is a plausible thing to write, and a crash here would
+ take down LoadModel for the whole model. So: integral numbers are
+ accepted in any spelling ("512", "512.0"), anything else is ignored
+ with a warning instead of raising. Zero and negative budgets are
+ ignored too: sglang gives them no defined meaning, and turning
+ reasoning off is what `reasoning_default: off` is for.
+ """
+ if value is None or str(value).strip() == "":
+ return None
+ raw = str(value).strip()
+ try:
+ number = float(raw)
+ except ValueError:
+ print(f"thinking_budget {raw!r} is not a number, ignoring it", file=sys.stderr)
+ return None
+ if not number.is_integer():
+ print(f"thinking_budget {raw!r} is not a whole number of tokens, ignoring it", file=sys.stderr)
+ return None
+ if number <= 0:
+ print(
+ f"thinking_budget {raw!r} must be positive, ignoring it "
+ "(use reasoning_default:off to disable reasoning)",
+ file=sys.stderr,
+ )
+ return None
+ return int(number)
+
+ @staticmethod
+ def _strict_thinking_warning(thinking_budget: Optional[int], engine_kwargs: dict) -> Optional[str]:
+ """sglang only enforces the budget with enable_strict_thinking on; without
+ it the budget is silently ignored, so say so at load time."""
+ if thinking_budget is not None and not engine_kwargs.get("enable_strict_thinking"):
+ return (
+ f"thinking_budget={thinking_budget} is set but enable_strict_thinking is not "
+ "in engine_args; sglang will ignore the budget"
+ )
+ return None
+
def _apply_engine_args(self, engine_kwargs: dict, engine_args_json: str) -> dict:
"""Merge user-supplied engine_args (JSON object) into the kwargs dict
that will be forwarded to ``sglang.Engine`` (which constructs a
@@ -230,6 +292,35 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
self.tool_parser_name: Optional[str] = opts.get("tool_parser") or None
self.reasoning_parser_name: Optional[str] = opts.get("reasoning_parser") or None
+ # Fixed reasoning-length budget for every request on this model, in
+ # tokens. There is no protobuf field to carry a per-request
+ # custom_params blob, so this rides the same model-level `options:`
+ # mechanism as tool_parser/reasoning_parser above — mirroring how
+ # sglang's own `--preferred-sampling-params` is a server-wide
+ # default, not a per-request choice. Requires `enable_strict_thinking`
+ # in `engine_args:` (sglang >=0.5.12); without it sglang has no
+ # tokenizer-derived budget mechanism to enforce this against.
+ self.thinking_budget: Optional[int] = self._parse_thinking_budget(
+ opts.get("thinking_budget")
+ )
+
+ # Model-level default for whether the chat template opens a reasoning
+ # block, as "off" or "on". Rides the same `options:` mechanism as
+ # thinking_budget above.
+ #
+ # Why this is needed even though `reasoning_effort` exists: that one
+ # only reaches this backend when a *caller* sets it per request (the
+ # Go side turns it into Metadata["enable_thinking"]). As a model-level
+ # `parameters:` default it is silently dropped, so a config reading
+ # `reasoning_effort: none` still produces full reasoning on every
+ # request - the config says one thing and the model does another.
+ #
+ # A per-request value always wins; this only fills in the gap when the
+ # request says nothing.
+ self.reasoning_default: Optional[str] = (
+ opts.get("reasoning_default") or ""
+ ).lower() or None
+
# Also hand the parser names to sglang's engine so its HTTP/OAI
# paths work identically if someone hits the engine directly.
if self.tool_parser_name:
@@ -247,6 +338,10 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
print(f"engine_args error: {err}", file=sys.stderr)
return backend_pb2.Result(success=False, message=str(err))
+ warning = self._strict_thinking_warning(self.thinking_budget, engine_kwargs)
+ if warning:
+ print(warning, file=sys.stderr)
+
try:
self.llm = Engine(**engine_kwargs)
except Exception as err:
@@ -362,8 +457,28 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
except json.JSONDecodeError:
sampling_params["ebnf"] = grammar
+ if self.thinking_budget is not None:
+ sampling_params["custom_params"] = {"thinking_budget": self.thinking_budget}
+
return sampling_params
+ def _thinking_default(self, request) -> Optional[bool]:
+ """Whether this request should render with reasoning on, off, or unset.
+
+ Per-request ``Metadata["enable_thinking"]`` wins; the model-level
+ ``reasoning_default`` option fills in when the request is silent.
+ Returns None when neither says anything, leaving template behaviour
+ untouched.
+ """
+ wanted = request.Metadata.get("enable_thinking", "").lower()
+ if wanted in ("true", "false"):
+ return wanted == "true"
+ if self.reasoning_default == "off":
+ return False
+ if self.reasoning_default == "on":
+ return True
+ return None
+
def _build_prompt(self, request) -> str:
prompt = request.Prompt
if prompt or not request.UseTokenizerTemplate or not request.Messages:
@@ -384,9 +499,9 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
template_kwargs["tools"] = json.loads(request.Tools)
except json.JSONDecodeError:
pass
- _thinking = request.Metadata.get("enable_thinking", "").lower()
- if _thinking in ("true", "false"):
- template_kwargs["enable_thinking"] = (_thinking == "true")
+ _thinking = self._thinking_default(request)
+ if _thinking is not None:
+ template_kwargs["enable_thinking"] = _thinking
# sglang locates the attached images/videos by scanning the rendered
# prompt for the model's own media token, so the template has to be
@@ -438,12 +553,19 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
there files the answer as reasoning and leaves content empty. sglang's
own server keeps the two apart for the same reason — its grammar
backend owns the reasoning prefix when a reasoning parser is set.
+
+ Returns a ``(parser, forced)`` pair. ``forced`` is also the signal
+ ``_predict`` passes as ``Engine.async_generate(require_reasoning=...)``:
+ sglang's own OpenAI server derives that flag from per-template
+ config (``ChatServing._get_reasoning_from_request``); this backend
+ has no template manager, so the same prompt-suffix heuristic that
+ already decides parser forcing doubles as that signal.
"""
if grammar_constrained:
prompt = ""
if not (HAS_REASONING_PARSERS and self.reasoning_parser_name):
- return None
+ return None, False
kwargs = {
"model_type": self.reasoning_parser_name,
@@ -453,10 +575,12 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
parser = ReasoningParser(**kwargs)
except Exception as e:
print(f"ReasoningParser init failed: {e!r}", file=sys.stderr)
- return None
+ return None, False
+ forced = False
start = getattr(getattr(parser, "detector", None), "think_start_token", None)
if start and prompt and prompt.rstrip().endswith(start):
+ forced = True
try:
parser = ReasoningParser(force_reasoning=True, **kwargs)
except TypeError:
@@ -469,10 +593,16 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
file=sys.stderr,
)
- return parser
+ return parser, forced
def _make_parsers(self, request, prompt: str = ""):
- """Construct fresh per-request parser instances (stateful)."""
+ """Construct fresh per-request parser instances (stateful).
+
+ Also returns ``require_reasoning`` (see ``_new_reasoning_parser``),
+ which ``_predict`` forwards to ``Engine.async_generate()`` so
+ sglang's ``--enable-strict-thinking`` grammar backend knows this
+ request is in a reasoning block.
+ """
tool_parser = None
if HAS_TOOL_PARSERS and self.tool_parser_name and request.Tools:
@@ -485,23 +615,27 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
except Exception as e:
print(f"FunctionCallParser init failed: {e!r}", file=sys.stderr)
- reasoning_parser = self._new_reasoning_parser(
+ reasoning_parser, require_reasoning = self._new_reasoning_parser(
True, prompt, bool(getattr(request, "Grammar", "")),
)
- return tool_parser, reasoning_parser
+ return tool_parser, reasoning_parser, require_reasoning
async def _predict(self, request, context, streaming: bool = False):
sampling_params = self._build_sampling_params(request)
prompt = self._build_prompt(request)
- tool_parser, reasoning_parser = self._make_parsers(request, prompt)
+ tool_parser, reasoning_parser, require_reasoning = self._make_parsers(request, prompt)
image_data = list(request.Images) if request.Images else None
video_data = list(request.Videos) if request.Videos else None
# Kick off streaming generation. We always use stream=True so the
# non-stream path still gets parser coverage on the final text.
+ generate_kwargs = {}
+ if _ASYNC_GENERATE_HAS_REQUIRE_REASONING:
+ generate_kwargs["require_reasoning"] = require_reasoning
+
try:
iterator = await self.llm.async_generate(
prompt=prompt,
@@ -509,6 +643,7 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
image_data=image_data,
video_data=video_data,
stream=True,
+ **generate_kwargs,
)
except Exception as e:
print(f"sglang async_generate failed: {e!r}", file=sys.stderr)
@@ -591,7 +726,7 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
final_tool_calls: List[backend_pb2.ToolCallDelta] = []
if not streaming:
- final_reasoning_parser = self._new_reasoning_parser(
+ final_reasoning_parser, _ = self._new_reasoning_parser(
False, prompt, bool(getattr(request, "Grammar", "")),
)
diff --git a/backend/python/sglang/test.py b/backend/python/sglang/test.py
index 51158791c..1b1a10d23 100644
--- a/backend/python/sglang/test.py
+++ b/backend/python/sglang/test.py
@@ -9,6 +9,13 @@ because ``_apply_engine_args`` validates keys against ``ServerArgs``
import unittest
+
+def _request(metadata=None):
+ """Minimal stand-in for a PredictOptions request in reasoning tests."""
+ from types import SimpleNamespace
+
+ return SimpleNamespace(Metadata=metadata or {})
+
class TestSglangHelpers(unittest.TestCase):
"""Tests for the pure helpers on BackendServicer (no gRPC, no engine)."""
@@ -170,13 +177,17 @@ class TestSglangHelpers(unittest.TestCase):
# What the model actually emits when the prompt ends in "".
completion = "adding two and two4"
- forced = servicer._new_reasoning_parser(False, prompt="user: hi\n\n")
+ forced, require_reasoning = servicer._new_reasoning_parser(
+ False, prompt="user: hi\n\n"
+ )
+ self.assertTrue(require_reasoning)
reasoning, content = forced.parse_non_stream(completion)
self.assertEqual(reasoning, "adding two and two")
self.assertEqual(content, "4")
# No prefilled tag in the prompt: detector default, unchanged behaviour.
- unforced = servicer._new_reasoning_parser(False, prompt="user: hi\n")
+ unforced, require_reasoning = servicer._new_reasoning_parser(False, prompt="user: hi\n")
+ self.assertFalse(require_reasoning)
reasoning, content = unforced.parse_non_stream(completion)
self.assertFalse(reasoning)
self.assertEqual(content, completion)
@@ -187,7 +198,8 @@ class TestSglangHelpers(unittest.TestCase):
servicer = self._servicer()
servicer.reasoning_parser_name = "qwen3"
- parser = servicer._new_reasoning_parser(False, prompt="user: primes?\n")
+ parser, require_reasoning = servicer._new_reasoning_parser(False, prompt="user: primes?\n")
+ self.assertFalse(require_reasoning)
reasoning, content = parser.parse_non_stream("2,3,5,7,11")
self.assertFalse(reasoning)
self.assertEqual(content, "2,3,5,7,11")
@@ -200,9 +212,10 @@ class TestSglangHelpers(unittest.TestCase):
servicer.reasoning_parser_name = "qwen3"
schema_out = '{"findings": [{"line": 42, "issue": "off-by-one"}]}'
- parser = servicer._new_reasoning_parser(
+ parser, require_reasoning = servicer._new_reasoning_parser(
False, prompt="audit this\n\n", grammar_constrained=True,
)
+ self.assertFalse(require_reasoning)
reasoning, content = parser.parse_non_stream(schema_out)
self.assertFalse(reasoning)
self.assertEqual(content, schema_out)
@@ -210,7 +223,110 @@ class TestSglangHelpers(unittest.TestCase):
def test_reasoning_parser_absent_without_configured_parser(self):
servicer = self._servicer()
servicer.reasoning_parser_name = None
- self.assertIsNone(servicer._new_reasoning_parser(False, prompt=""))
+ parser, require_reasoning = servicer._new_reasoning_parser(False, prompt="")
+ self.assertIsNone(parser)
+ self.assertFalse(require_reasoning)
+
+ def test_reasoning_default_off_applies_when_request_is_silent(self):
+ """A model configured with reasoning_default:off must render with
+ thinking disabled even when the request carries no enable_thinking -
+ that is the whole point: `parameters: reasoning_effort:` never
+ reaches this backend, so without this the config lies about the
+ default."""
+ servicer = self._servicer()
+ servicer.reasoning_default = "off"
+ self.assertIs(servicer._thinking_default(_request(metadata={})), False)
+
+ def test_request_metadata_overrides_reasoning_default(self):
+ """A per-request value always wins over the model-level default -
+ in both directions."""
+ servicer = self._servicer()
+ servicer.reasoning_default = "off"
+ self.assertIs(
+ servicer._thinking_default(_request(metadata={"enable_thinking": "true"})),
+ True,
+ )
+ servicer.reasoning_default = "on"
+ self.assertIs(
+ servicer._thinking_default(_request(metadata={"enable_thinking": "false"})),
+ False,
+ )
+
+ def test_no_reasoning_default_leaves_template_untouched(self):
+ """Unconfigured must stay unconfigured: returning None means the
+ backend adds no enable_thinking kwarg at all, so the template keeps
+ whatever default it ships with."""
+ servicer = self._servicer()
+ self.assertIsNone(servicer._thinking_default(_request(metadata={})))
+
+ def test_thinking_budget_added_to_sampling_params_as_custom_params(self):
+ """The model-level thinking_budget option (set from LoadModel's
+ Options, mirroring tool_parser/reasoning_parser) must ride along as
+ sampling_params['custom_params']['thinking_budget'] on every
+ request — that's the only field sglang's --enable-strict-thinking
+ grammar backend reads to bound the reasoning length."""
+ from types import SimpleNamespace
+
+ servicer = self._servicer()
+ servicer.thinking_budget = 512
+ request = SimpleNamespace(
+ Temperature=0.7, N=0, PresencePenalty=0, FrequencyPenalty=0,
+ RepetitionPenalty=0, TopP=0, TopK=0, MinP=0, Seed=0,
+ StopPrompts=[], StopTokenIds=[], IgnoreEOS=False, Tokens=0,
+ MinTokens=0, SkipSpecialTokens=False, Grammar="",
+ )
+ params = servicer._build_sampling_params(request)
+ self.assertEqual(params["custom_params"], {"thinking_budget": 512})
+
+ def test_no_thinking_budget_means_no_custom_params_key(self):
+ """Unconfigured is unconfigured: no thinking_budget option must not
+ add an empty/None custom_params that could clobber a sglang-side
+ --preferred-sampling-params default (see sglang#40634)."""
+ from types import SimpleNamespace
+
+ servicer = self._servicer()
+ request = SimpleNamespace(
+ Temperature=0.7, N=0, PresencePenalty=0, FrequencyPenalty=0,
+ RepetitionPenalty=0, TopP=0, TopK=0, MinP=0, Seed=0,
+ StopPrompts=[], StopTokenIds=[], IgnoreEOS=False, Tokens=0,
+ MinTokens=0, SkipSpecialTokens=False, Grammar="",
+ )
+ params = servicer._build_sampling_params(request)
+ self.assertNotIn("custom_params", params)
+
+ def test_thinking_budget_accepts_integral_spellings(self):
+ """YAML options arrive as strings; "512" and "512.0" both mean 512."""
+ servicer = self._servicer()
+ self.assertEqual(servicer._parse_thinking_budget("512"), 512)
+ self.assertEqual(servicer._parse_thinking_budget("512.0"), 512)
+ self.assertEqual(servicer._parse_thinking_budget(" 64 "), 64)
+ self.assertEqual(servicer._parse_thinking_budget(256), 256)
+
+ def test_thinking_budget_unset_is_none(self):
+ servicer = self._servicer()
+ self.assertIsNone(servicer._parse_thinking_budget(None))
+ self.assertIsNone(servicer._parse_thinking_budget(""))
+
+ def test_thinking_budget_zero_and_negative_are_ignored(self):
+ """No defined meaning in sglang -- ignored, not passed through."""
+ servicer = self._servicer()
+ self.assertIsNone(servicer._parse_thinking_budget("0"))
+ self.assertIsNone(servicer._parse_thinking_budget("-100"))
+
+ def test_thinking_budget_non_integer_does_not_raise(self):
+ """A bad value must not crash LoadModel for the whole model."""
+ servicer = self._servicer()
+ self.assertIsNone(servicer._parse_thinking_budget("12.5"))
+ self.assertIsNone(servicer._parse_thinking_budget("lots"))
+
+ def test_warns_when_budget_set_without_strict_thinking(self):
+ servicer = self._servicer()
+ self.assertIn(
+ "enable_strict_thinking",
+ servicer._strict_thinking_warning(512, {"model_path": "x"}),
+ )
+ self.assertIsNone(servicer._strict_thinking_warning(512, {"enable_strict_thinking": True}))
+ self.assertIsNone(servicer._strict_thinking_warning(None, {}))
def test_explicit_zero_temperature_and_seed_are_preserved(self):
"""Temperature=0 is greedy decoding and 0 is a valid seed — neither is
diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md
index dadb5e9af..c28bf9213 100644
--- a/docs/content/features/text-generation.md
+++ b/docs/content/features/text-generation.md
@@ -987,6 +987,41 @@ options:
The full list of registered parsers lives in `sglang.srt.function_call`
and `sglang.srt.parser.reasoning_parser`.
+#### Reasoning defaults and token budgets
+
+Set SGLang reasoning options in the model's `options:` list:
+
+```yaml
+options:
+ - reasoning_parser:qwen3
+ - thinking_budget:512
+ - reasoning_default:on
+engine_args:
+ enable_strict_thinking: true
+```
+
+`thinking_budget` sets a positive integer token budget for reasoning on each request.
+Invalid, zero, and negative values produce a warning and leave the budget unset.
+SGLang requires `engine_args.enable_strict_thinking: true` to enforce the budget.
+LocalAI warns if you configure a budget without that engine option.
+Keep the budget well below the `max_tokens` of your requests: if `max_tokens` is reached first,
+the budget never triggers and the whole reply can be spent on reasoning, leaving the answer empty.
+
+`reasoning_default:on` or `reasoning_default:off` sets the default for LocalAI's tokenizer chat template.
+Request metadata `enable_thinking` set to `"true"` or `"false"` overrides this default.
+An explicit prompt bypasses tokenizer template rendering.
+When no default or request override is set, the template keeps its own behavior.
+
+LocalAI signals required reasoning when the rendered prompt ends with the configured parser's opening reasoning token.
+An explicit output grammar disables this detection.
+Configure a reasoning parser that matches your model.
+
+The backend reads these options when it loads the model.
+`POST /models/reload` rereads model configuration files but does not update options in an already loaded backend.
+Restarting only the backend does not reread configuration files.
+Restart LocalAI after changing these options to reload both the configuration and the backend.
+
+
### vllm.cpp
[vllm.cpp](https://github.com/mudler/vllm.cpp) is the LocalAI team's C++ port of