mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-29 01:25:03 -04:00
chore: merge master into distributed transport PR
Keep the version-reporting import from master and omit the unused sanitize import after the transport changes. Assisted-by: Codex:GPT-6
This commit is contained in:
commit
4151ceda85
51 files changed
+1785
-72
No files matched your search
@@ -9,7 +9,7 @@
|
||||
# recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean
|
||||
# rebuild and so the bump bot can see the pin.
|
||||
|
||||
AUDIO_CPP_VERSION?=94bd4656399180befc141b17bd6696bf84df0a9f
|
||||
AUDIO_CPP_VERSION?=77491a33c589c53ff18add050095cf35647c8213
|
||||
AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp
|
||||
|
||||
CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST))))
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
|
||||
IK_LLAMA_VERSION?=cdf232cc17e410e60c1bc3b85516c4a41199b662
|
||||
IK_LLAMA_VERSION?=ed27bf7ed25e637692e89cd341d802522a2cee8a
|
||||
LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp
|
||||
|
||||
CMAKE_ARGS?=
|
||||
|
||||
@@ -539,8 +539,10 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
|
||||
|
||||
// Initialize ctx_shift to false by default (can be overridden by options)
|
||||
params.ctx_shift = false;
|
||||
// Initialize cache_ram_mib to -1 by default (no limit, can be overridden by options)
|
||||
params.cache_ram_mib = -1;
|
||||
// cache_ram_mib keeps llama.cpp's own default (8192 MiB) unless overridden by
|
||||
// options. It used to be forced to -1 (no limit): since kv_unified and
|
||||
// cache_idle_slots are on by default, every distinct prompt then leaves its
|
||||
// slot state in host RAM and the backend grows without bound.
|
||||
// Initialize n_parallel to 1 by default (can be overridden by options)
|
||||
params.n_parallel = 1;
|
||||
// Initialize grpc_servers to empty (can be overridden by options)
|
||||
@@ -656,7 +658,7 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt
|
||||
try {
|
||||
params.cache_ram_mib = std::stoi(optval_str);
|
||||
} catch (const std::exception& e) {
|
||||
// If conversion fails, keep default value (-1)
|
||||
// If conversion fails, keep the default value
|
||||
}
|
||||
}
|
||||
} else if (!strcmp(optname, "parallel") || !strcmp(optname, "n_parallel")) {
|
||||
|
||||
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
|
||||
|
||||
# CrispASR version (release tag)
|
||||
CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR
|
||||
CRISPASR_VERSION?=013ae1624dc40ecf059065d577180722439f804e
|
||||
CRISPASR_VERSION?=ec98831d0776ec8a16ccaf93955693eb7ecfbec3
|
||||
SO_TARGET?=libgocrispasr.so
|
||||
|
||||
CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
|
||||
|
||||
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
|
||||
|
||||
# omnivoice.cpp version
|
||||
OMNIVOICE_REPO?=https://github.com/ServeurpersoCom/omnivoice.cpp
|
||||
OMNIVOICE_VERSION?=8ab42195a05a9d48a3942b17568c1f3a876e133a
|
||||
OMNIVOICE_VERSION?=ead199a2bc4c53a57cac90095ae049a111d9e98d
|
||||
SO_TARGET?=libgomnivoicecpp.so
|
||||
|
||||
CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
|
||||
|
||||
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
|
||||
|
||||
# stablediffusion.cpp (ggml)
|
||||
STABLEDIFFUSION_GGML_REPO?=https://github.com/leejet/stable-diffusion.cpp
|
||||
STABLEDIFFUSION_GGML_VERSION?=2f886889e6e8b78738d6b87f7191f6018557c551
|
||||
STABLEDIFFUSION_GGML_VERSION?=3f8527a46c54ecf4cb4ed6003da8e8982283c73c
|
||||
|
||||
CMAKE_ARGS+=-DGGML_MAX_NAME=128
|
||||
|
||||
|
||||
@@ -137,8 +137,8 @@ func (sd *SDGGML) Load(opts *pb.ModelOptions) error {
|
||||
// If it's an option path, we resolve absolute path from the model path
|
||||
if strings.Contains(op, ":") && strings.Contains(op, "path") {
|
||||
data := strings.Split(op, ":")
|
||||
data[1] = filepath.Join(opts.ModelPath, data[1])
|
||||
if err := utils.VerifyPath(data[1], opts.ModelPath); err == nil {
|
||||
data[1] = filepath.Join(opts.ModelPath, data[1])
|
||||
oo = append(oo, strings.Join(data, ":"))
|
||||
}
|
||||
} else {
|
||||
|
||||
@@ -161,10 +161,10 @@ func resolveModels(modelFile, modelPath string, options []string) (modelSet, err
|
||||
continue
|
||||
}
|
||||
if !filepath.IsAbs(value) {
|
||||
value = filepath.Join(modelPath, value)
|
||||
if err := utils.VerifyPath(value, modelPath); err != nil {
|
||||
return modelSet{}, fmt.Errorf("option %s: %w", key, err)
|
||||
}
|
||||
value = filepath.Join(modelPath, value)
|
||||
}
|
||||
overrides[key] = value
|
||||
}
|
||||
|
||||
@@ -134,9 +134,12 @@ var _ = Describe("resolveModels", func() {
|
||||
|
||||
It("rejects option paths escaping the model directory", func() {
|
||||
touch(dir, fullSet...)
|
||||
// The escaping file exists, so only the containment check can
|
||||
// reject it; a missing file would fail for an unrelated reason.
|
||||
touch(filepath.Dir(dir), "outside.gguf")
|
||||
|
||||
_, err := resolveModels("ss_flow_f16.gguf", dir, []string{"dino_path:../outside.gguf"})
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(err).To(MatchError(ContainSubstring("outside of trusted root")))
|
||||
})
|
||||
})
|
||||
|
||||
|
||||
@@ -90,6 +90,19 @@ except Exception:
|
||||
_SEED_KEY = "sampling_seed"
|
||||
|
||||
|
||||
# Engine.async_generate() only grew a require_reasoning keyword in sglang
|
||||
# 0.5.13. The CPU build compiles v0.5.11 from source and the other profiles
|
||||
# only set a >=0.5.11 floor, and async_generate() takes no **kwargs, so
|
||||
# passing the keyword unconditionally fails every request with TypeError.
|
||||
try:
|
||||
import inspect as _inspect
|
||||
_ASYNC_GENERATE_HAS_REQUIRE_REASONING = (
|
||||
"require_reasoning" in _inspect.signature(Engine.async_generate).parameters
|
||||
)
|
||||
except Exception:
|
||||
_ASYNC_GENERATE_HAS_REQUIRE_REASONING = False
|
||||
|
||||
|
||||
_ONE_DAY_IN_SECONDS = 60 * 60 * 24
|
||||
|
||||
# proto3 has no field presence, so an explicit 0 is indistinguishable from
|
||||
@@ -105,6 +118,12 @@ MAX_WORKERS = int(os.environ.get('PYTHON_GRPC_MAX_WORKERS', '1'))
|
||||
class BackendServicer(backend_pb2_grpc.BackendServicer):
|
||||
"""gRPC servicer implementing the Backend service for sglang."""
|
||||
|
||||
# Class-level default so a servicer used before LoadModel (e.g. in unit
|
||||
# tests that construct it directly) doesn't AttributeError in
|
||||
# _build_sampling_params.
|
||||
thinking_budget: Optional[int] = None
|
||||
reasoning_default: Optional[str] = None
|
||||
|
||||
def _parse_options(self, options_list) -> Dict[str, str]:
|
||||
opts: Dict[str, str] = {}
|
||||
for opt in options_list:
|
||||
@@ -114,6 +133,49 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
|
||||
opts[key.strip()] = value.strip()
|
||||
return opts
|
||||
|
||||
@staticmethod
|
||||
def _parse_thinking_budget(value) -> Optional[int]:
|
||||
"""Turn the `thinking_budget` model option into a positive int, or None.
|
||||
|
||||
Options arrive as strings from the YAML `options:` list, but a value
|
||||
like "5000.0" is a plausible thing to write, and a crash here would
|
||||
take down LoadModel for the whole model. So: integral numbers are
|
||||
accepted in any spelling ("512", "512.0"), anything else is ignored
|
||||
with a warning instead of raising. Zero and negative budgets are
|
||||
ignored too: sglang gives them no defined meaning, and turning
|
||||
reasoning off is what `reasoning_default: off` is for.
|
||||
"""
|
||||
if value is None or str(value).strip() == "":
|
||||
return None
|
||||
raw = str(value).strip()
|
||||
try:
|
||||
number = float(raw)
|
||||
except ValueError:
|
||||
print(f"thinking_budget {raw!r} is not a number, ignoring it", file=sys.stderr)
|
||||
return None
|
||||
if not number.is_integer():
|
||||
print(f"thinking_budget {raw!r} is not a whole number of tokens, ignoring it", file=sys.stderr)
|
||||
return None
|
||||
if number <= 0:
|
||||
print(
|
||||
f"thinking_budget {raw!r} must be positive, ignoring it "
|
||||
"(use reasoning_default:off to disable reasoning)",
|
||||
file=sys.stderr,
|
||||
)
|
||||
return None
|
||||
return int(number)
|
||||
|
||||
@staticmethod
|
||||
def _strict_thinking_warning(thinking_budget: Optional[int], engine_kwargs: dict) -> Optional[str]:
|
||||
"""sglang only enforces the budget with enable_strict_thinking on; without
|
||||
it the budget is silently ignored, so say so at load time."""
|
||||
if thinking_budget is not None and not engine_kwargs.get("enable_strict_thinking"):
|
||||
return (
|
||||
f"thinking_budget={thinking_budget} is set but enable_strict_thinking is not "
|
||||
"in engine_args; sglang will ignore the budget"
|
||||
)
|
||||
return None
|
||||
|
||||
def _apply_engine_args(self, engine_kwargs: dict, engine_args_json: str) -> dict:
|
||||
"""Merge user-supplied engine_args (JSON object) into the kwargs dict
|
||||
that will be forwarded to ``sglang.Engine`` (which constructs a
|
||||
@@ -230,6 +292,35 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
|
||||
self.tool_parser_name: Optional[str] = opts.get("tool_parser") or None
|
||||
self.reasoning_parser_name: Optional[str] = opts.get("reasoning_parser") or None
|
||||
|
||||
# Fixed reasoning-length budget for every request on this model, in
|
||||
# tokens. There is no protobuf field to carry a per-request
|
||||
# custom_params blob, so this rides the same model-level `options:`
|
||||
# mechanism as tool_parser/reasoning_parser above — mirroring how
|
||||
# sglang's own `--preferred-sampling-params` is a server-wide
|
||||
# default, not a per-request choice. Requires `enable_strict_thinking`
|
||||
# in `engine_args:` (sglang >=0.5.12); without it sglang has no
|
||||
# tokenizer-derived budget mechanism to enforce this against.
|
||||
self.thinking_budget: Optional[int] = self._parse_thinking_budget(
|
||||
opts.get("thinking_budget")
|
||||
)
|
||||
|
||||
# Model-level default for whether the chat template opens a reasoning
|
||||
# block, as "off" or "on". Rides the same `options:` mechanism as
|
||||
# thinking_budget above.
|
||||
#
|
||||
# Why this is needed even though `reasoning_effort` exists: that one
|
||||
# only reaches this backend when a *caller* sets it per request (the
|
||||
# Go side turns it into Metadata["enable_thinking"]). As a model-level
|
||||
# `parameters:` default it is silently dropped, so a config reading
|
||||
# `reasoning_effort: none` still produces full reasoning on every
|
||||
# request - the config says one thing and the model does another.
|
||||
#
|
||||
# A per-request value always wins; this only fills in the gap when the
|
||||
# request says nothing.
|
||||
self.reasoning_default: Optional[str] = (
|
||||
opts.get("reasoning_default") or ""
|
||||
).lower() or None
|
||||
|
||||
# Also hand the parser names to sglang's engine so its HTTP/OAI
|
||||
# paths work identically if someone hits the engine directly.
|
||||
if self.tool_parser_name:
|
||||
@@ -247,6 +338,10 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
|
||||
print(f"engine_args error: {err}", file=sys.stderr)
|
||||
return backend_pb2.Result(success=False, message=str(err))
|
||||
|
||||
warning = self._strict_thinking_warning(self.thinking_budget, engine_kwargs)
|
||||
if warning:
|
||||
print(warning, file=sys.stderr)
|
||||
|
||||
try:
|
||||
self.llm = Engine(**engine_kwargs)
|
||||
except Exception as err:
|
||||
@@ -362,8 +457,28 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
|
||||
except json.JSONDecodeError:
|
||||
sampling_params["ebnf"] = grammar
|
||||
|
||||
if self.thinking_budget is not None:
|
||||
sampling_params["custom_params"] = {"thinking_budget": self.thinking_budget}
|
||||
|
||||
return sampling_params
|
||||
|
||||
def _thinking_default(self, request) -> Optional[bool]:
|
||||
"""Whether this request should render with reasoning on, off, or unset.
|
||||
|
||||
Per-request ``Metadata["enable_thinking"]`` wins; the model-level
|
||||
``reasoning_default`` option fills in when the request is silent.
|
||||
Returns None when neither says anything, leaving template behaviour
|
||||
untouched.
|
||||
"""
|
||||
wanted = request.Metadata.get("enable_thinking", "").lower()
|
||||
if wanted in ("true", "false"):
|
||||
return wanted == "true"
|
||||
if self.reasoning_default == "off":
|
||||
return False
|
||||
if self.reasoning_default == "on":
|
||||
return True
|
||||
return None
|
||||
|
||||
def _build_prompt(self, request) -> str:
|
||||
prompt = request.Prompt
|
||||
if prompt or not request.UseTokenizerTemplate or not request.Messages:
|
||||
@@ -384,9 +499,9 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
|
||||
template_kwargs["tools"] = json.loads(request.Tools)
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
_thinking = request.Metadata.get("enable_thinking", "").lower()
|
||||
if _thinking in ("true", "false"):
|
||||
template_kwargs["enable_thinking"] = (_thinking == "true")
|
||||
_thinking = self._thinking_default(request)
|
||||
if _thinking is not None:
|
||||
template_kwargs["enable_thinking"] = _thinking
|
||||
|
||||
# sglang locates the attached images/videos by scanning the rendered
|
||||
# prompt for the model's own media token, so the template has to be
|
||||
@@ -438,12 +553,19 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
|
||||
there files the answer as reasoning and leaves content empty. sglang's
|
||||
own server keeps the two apart for the same reason — its grammar
|
||||
backend owns the reasoning prefix when a reasoning parser is set.
|
||||
|
||||
Returns a ``(parser, forced)`` pair. ``forced`` is also the signal
|
||||
``_predict`` passes as ``Engine.async_generate(require_reasoning=...)``:
|
||||
sglang's own OpenAI server derives that flag from per-template
|
||||
config (``ChatServing._get_reasoning_from_request``); this backend
|
||||
has no template manager, so the same prompt-suffix heuristic that
|
||||
already decides parser forcing doubles as that signal.
|
||||
"""
|
||||
if grammar_constrained:
|
||||
prompt = ""
|
||||
|
||||
if not (HAS_REASONING_PARSERS and self.reasoning_parser_name):
|
||||
return None
|
||||
return None, False
|
||||
|
||||
kwargs = {
|
||||
"model_type": self.reasoning_parser_name,
|
||||
@@ -453,10 +575,12 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
|
||||
parser = ReasoningParser(**kwargs)
|
||||
except Exception as e:
|
||||
print(f"ReasoningParser init failed: {e!r}", file=sys.stderr)
|
||||
return None
|
||||
return None, False
|
||||
|
||||
forced = False
|
||||
start = getattr(getattr(parser, "detector", None), "think_start_token", None)
|
||||
if start and prompt and prompt.rstrip().endswith(start):
|
||||
forced = True
|
||||
try:
|
||||
parser = ReasoningParser(force_reasoning=True, **kwargs)
|
||||
except TypeError:
|
||||
@@ -469,10 +593,16 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
|
||||
file=sys.stderr,
|
||||
)
|
||||
|
||||
return parser
|
||||
return parser, forced
|
||||
|
||||
def _make_parsers(self, request, prompt: str = ""):
|
||||
"""Construct fresh per-request parser instances (stateful)."""
|
||||
"""Construct fresh per-request parser instances (stateful).
|
||||
|
||||
Also returns ``require_reasoning`` (see ``_new_reasoning_parser``),
|
||||
which ``_predict`` forwards to ``Engine.async_generate()`` so
|
||||
sglang's ``--enable-strict-thinking`` grammar backend knows this
|
||||
request is in a reasoning block.
|
||||
"""
|
||||
tool_parser = None
|
||||
|
||||
if HAS_TOOL_PARSERS and self.tool_parser_name and request.Tools:
|
||||
@@ -485,23 +615,27 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
|
||||
except Exception as e:
|
||||
print(f"FunctionCallParser init failed: {e!r}", file=sys.stderr)
|
||||
|
||||
reasoning_parser = self._new_reasoning_parser(
|
||||
reasoning_parser, require_reasoning = self._new_reasoning_parser(
|
||||
True, prompt, bool(getattr(request, "Grammar", "")),
|
||||
)
|
||||
|
||||
return tool_parser, reasoning_parser
|
||||
return tool_parser, reasoning_parser, require_reasoning
|
||||
|
||||
async def _predict(self, request, context, streaming: bool = False):
|
||||
sampling_params = self._build_sampling_params(request)
|
||||
prompt = self._build_prompt(request)
|
||||
|
||||
tool_parser, reasoning_parser = self._make_parsers(request, prompt)
|
||||
tool_parser, reasoning_parser, require_reasoning = self._make_parsers(request, prompt)
|
||||
|
||||
image_data = list(request.Images) if request.Images else None
|
||||
video_data = list(request.Videos) if request.Videos else None
|
||||
|
||||
# Kick off streaming generation. We always use stream=True so the
|
||||
# non-stream path still gets parser coverage on the final text.
|
||||
generate_kwargs = {}
|
||||
if _ASYNC_GENERATE_HAS_REQUIRE_REASONING:
|
||||
generate_kwargs["require_reasoning"] = require_reasoning
|
||||
|
||||
try:
|
||||
iterator = await self.llm.async_generate(
|
||||
prompt=prompt,
|
||||
@@ -509,6 +643,7 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
|
||||
image_data=image_data,
|
||||
video_data=video_data,
|
||||
stream=True,
|
||||
**generate_kwargs,
|
||||
)
|
||||
except Exception as e:
|
||||
print(f"sglang async_generate failed: {e!r}", file=sys.stderr)
|
||||
@@ -591,7 +726,7 @@ class BackendServicer(backend_pb2_grpc.BackendServicer):
|
||||
final_tool_calls: List[backend_pb2.ToolCallDelta] = []
|
||||
|
||||
if not streaming:
|
||||
final_reasoning_parser = self._new_reasoning_parser(
|
||||
final_reasoning_parser, _ = self._new_reasoning_parser(
|
||||
False, prompt, bool(getattr(request, "Grammar", "")),
|
||||
)
|
||||
|
||||
|
||||
@@ -9,6 +9,13 @@ because ``_apply_engine_args`` validates keys against ``ServerArgs``
|
||||
import unittest
|
||||
|
||||
|
||||
|
||||
def _request(metadata=None):
|
||||
"""Minimal stand-in for a PredictOptions request in reasoning tests."""
|
||||
from types import SimpleNamespace
|
||||
|
||||
return SimpleNamespace(Metadata=metadata or {})
|
||||
|
||||
class TestSglangHelpers(unittest.TestCase):
|
||||
"""Tests for the pure helpers on BackendServicer (no gRPC, no engine)."""
|
||||
|
||||
@@ -170,13 +177,17 @@ class TestSglangHelpers(unittest.TestCase):
|
||||
# What the model actually emits when the prompt ends in "<think>".
|
||||
completion = "adding two and two</think>4"
|
||||
|
||||
forced = servicer._new_reasoning_parser(False, prompt="user: hi\n<think>\n")
|
||||
forced, require_reasoning = servicer._new_reasoning_parser(
|
||||
False, prompt="user: hi\n<think>\n"
|
||||
)
|
||||
self.assertTrue(require_reasoning)
|
||||
reasoning, content = forced.parse_non_stream(completion)
|
||||
self.assertEqual(reasoning, "adding two and two")
|
||||
self.assertEqual(content, "4")
|
||||
|
||||
# No prefilled tag in the prompt: detector default, unchanged behaviour.
|
||||
unforced = servicer._new_reasoning_parser(False, prompt="user: hi\n")
|
||||
unforced, require_reasoning = servicer._new_reasoning_parser(False, prompt="user: hi\n")
|
||||
self.assertFalse(require_reasoning)
|
||||
reasoning, content = unforced.parse_non_stream(completion)
|
||||
self.assertFalse(reasoning)
|
||||
self.assertEqual(content, completion)
|
||||
@@ -187,7 +198,8 @@ class TestSglangHelpers(unittest.TestCase):
|
||||
servicer = self._servicer()
|
||||
servicer.reasoning_parser_name = "qwen3"
|
||||
|
||||
parser = servicer._new_reasoning_parser(False, prompt="user: primes?\n")
|
||||
parser, require_reasoning = servicer._new_reasoning_parser(False, prompt="user: primes?\n")
|
||||
self.assertFalse(require_reasoning)
|
||||
reasoning, content = parser.parse_non_stream("2,3,5,7,11")
|
||||
self.assertFalse(reasoning)
|
||||
self.assertEqual(content, "2,3,5,7,11")
|
||||
@@ -200,9 +212,10 @@ class TestSglangHelpers(unittest.TestCase):
|
||||
servicer.reasoning_parser_name = "qwen3"
|
||||
|
||||
schema_out = '{"findings": [{"line": 42, "issue": "off-by-one"}]}'
|
||||
parser = servicer._new_reasoning_parser(
|
||||
parser, require_reasoning = servicer._new_reasoning_parser(
|
||||
False, prompt="audit this\n<think>\n", grammar_constrained=True,
|
||||
)
|
||||
self.assertFalse(require_reasoning)
|
||||
reasoning, content = parser.parse_non_stream(schema_out)
|
||||
self.assertFalse(reasoning)
|
||||
self.assertEqual(content, schema_out)
|
||||
@@ -210,7 +223,110 @@ class TestSglangHelpers(unittest.TestCase):
|
||||
def test_reasoning_parser_absent_without_configured_parser(self):
|
||||
servicer = self._servicer()
|
||||
servicer.reasoning_parser_name = None
|
||||
self.assertIsNone(servicer._new_reasoning_parser(False, prompt="<think>"))
|
||||
parser, require_reasoning = servicer._new_reasoning_parser(False, prompt="<think>")
|
||||
self.assertIsNone(parser)
|
||||
self.assertFalse(require_reasoning)
|
||||
|
||||
def test_reasoning_default_off_applies_when_request_is_silent(self):
|
||||
"""A model configured with reasoning_default:off must render with
|
||||
thinking disabled even when the request carries no enable_thinking -
|
||||
that is the whole point: `parameters: reasoning_effort:` never
|
||||
reaches this backend, so without this the config lies about the
|
||||
default."""
|
||||
servicer = self._servicer()
|
||||
servicer.reasoning_default = "off"
|
||||
self.assertIs(servicer._thinking_default(_request(metadata={})), False)
|
||||
|
||||
def test_request_metadata_overrides_reasoning_default(self):
|
||||
"""A per-request value always wins over the model-level default -
|
||||
in both directions."""
|
||||
servicer = self._servicer()
|
||||
servicer.reasoning_default = "off"
|
||||
self.assertIs(
|
||||
servicer._thinking_default(_request(metadata={"enable_thinking": "true"})),
|
||||
True,
|
||||
)
|
||||
servicer.reasoning_default = "on"
|
||||
self.assertIs(
|
||||
servicer._thinking_default(_request(metadata={"enable_thinking": "false"})),
|
||||
False,
|
||||
)
|
||||
|
||||
def test_no_reasoning_default_leaves_template_untouched(self):
|
||||
"""Unconfigured must stay unconfigured: returning None means the
|
||||
backend adds no enable_thinking kwarg at all, so the template keeps
|
||||
whatever default it ships with."""
|
||||
servicer = self._servicer()
|
||||
self.assertIsNone(servicer._thinking_default(_request(metadata={})))
|
||||
|
||||
def test_thinking_budget_added_to_sampling_params_as_custom_params(self):
|
||||
"""The model-level thinking_budget option (set from LoadModel's
|
||||
Options, mirroring tool_parser/reasoning_parser) must ride along as
|
||||
sampling_params['custom_params']['thinking_budget'] on every
|
||||
request — that's the only field sglang's --enable-strict-thinking
|
||||
grammar backend reads to bound the reasoning length."""
|
||||
from types import SimpleNamespace
|
||||
|
||||
servicer = self._servicer()
|
||||
servicer.thinking_budget = 512
|
||||
request = SimpleNamespace(
|
||||
Temperature=0.7, N=0, PresencePenalty=0, FrequencyPenalty=0,
|
||||
RepetitionPenalty=0, TopP=0, TopK=0, MinP=0, Seed=0,
|
||||
StopPrompts=[], StopTokenIds=[], IgnoreEOS=False, Tokens=0,
|
||||
MinTokens=0, SkipSpecialTokens=False, Grammar="",
|
||||
)
|
||||
params = servicer._build_sampling_params(request)
|
||||
self.assertEqual(params["custom_params"], {"thinking_budget": 512})
|
||||
|
||||
def test_no_thinking_budget_means_no_custom_params_key(self):
|
||||
"""Unconfigured is unconfigured: no thinking_budget option must not
|
||||
add an empty/None custom_params that could clobber a sglang-side
|
||||
--preferred-sampling-params default (see sglang#40634)."""
|
||||
from types import SimpleNamespace
|
||||
|
||||
servicer = self._servicer()
|
||||
request = SimpleNamespace(
|
||||
Temperature=0.7, N=0, PresencePenalty=0, FrequencyPenalty=0,
|
||||
RepetitionPenalty=0, TopP=0, TopK=0, MinP=0, Seed=0,
|
||||
StopPrompts=[], StopTokenIds=[], IgnoreEOS=False, Tokens=0,
|
||||
MinTokens=0, SkipSpecialTokens=False, Grammar="",
|
||||
)
|
||||
params = servicer._build_sampling_params(request)
|
||||
self.assertNotIn("custom_params", params)
|
||||
|
||||
def test_thinking_budget_accepts_integral_spellings(self):
|
||||
"""YAML options arrive as strings; "512" and "512.0" both mean 512."""
|
||||
servicer = self._servicer()
|
||||
self.assertEqual(servicer._parse_thinking_budget("512"), 512)
|
||||
self.assertEqual(servicer._parse_thinking_budget("512.0"), 512)
|
||||
self.assertEqual(servicer._parse_thinking_budget(" 64 "), 64)
|
||||
self.assertEqual(servicer._parse_thinking_budget(256), 256)
|
||||
|
||||
def test_thinking_budget_unset_is_none(self):
|
||||
servicer = self._servicer()
|
||||
self.assertIsNone(servicer._parse_thinking_budget(None))
|
||||
self.assertIsNone(servicer._parse_thinking_budget(""))
|
||||
|
||||
def test_thinking_budget_zero_and_negative_are_ignored(self):
|
||||
"""No defined meaning in sglang -- ignored, not passed through."""
|
||||
servicer = self._servicer()
|
||||
self.assertIsNone(servicer._parse_thinking_budget("0"))
|
||||
self.assertIsNone(servicer._parse_thinking_budget("-100"))
|
||||
|
||||
def test_thinking_budget_non_integer_does_not_raise(self):
|
||||
"""A bad value must not crash LoadModel for the whole model."""
|
||||
servicer = self._servicer()
|
||||
self.assertIsNone(servicer._parse_thinking_budget("12.5"))
|
||||
self.assertIsNone(servicer._parse_thinking_budget("lots"))
|
||||
|
||||
def test_warns_when_budget_set_without_strict_thinking(self):
|
||||
servicer = self._servicer()
|
||||
self.assertIn(
|
||||
"enable_strict_thinking",
|
||||
servicer._strict_thinking_warning(512, {"model_path": "x"}),
|
||||
)
|
||||
self.assertIsNone(servicer._strict_thinking_warning(512, {"enable_strict_thinking": True}))
|
||||
self.assertIsNone(servicer._strict_thinking_warning(None, {}))
|
||||
|
||||
def test_explicit_zero_temperature_and_seed_are_preserved(self):
|
||||
"""Temperature=0 is greedy decoding and 0 is a valid seed — neither is
|
||||
|
||||
@@ -623,9 +623,13 @@ func initDistributed(cfg *config.ApplicationConfig, authDB *gorm.DB, configLoade
|
||||
// All dependencies ready — build SmartRouter with all options at once
|
||||
var conflictResolver nodes.ConcurrencyConflictResolver
|
||||
var pinnedResolver nodes.PinnedModelResolver
|
||||
var modelFiles func(string) []string
|
||||
if configLoader != nil {
|
||||
conflictResolver = configLoader
|
||||
pinnedResolver = configLoader
|
||||
if cfg.SystemState != nil {
|
||||
modelFiles = declaredModelFiles(configLoader, cfg.SystemState.Model.ModelsPath)
|
||||
}
|
||||
}
|
||||
modelCleanup := nodes.NewModelCleanupService(registry, remoteUnloader)
|
||||
// Absence is stamped on by distributedSchedulerOptions rather than written
|
||||
@@ -645,6 +649,7 @@ func initDistributed(cfg *config.ApplicationConfig, authDB *gorm.DB, configLoade
|
||||
DataPath: cfg.DataPath,
|
||||
ConflictResolver: conflictResolver,
|
||||
PinnedResolver: pinnedResolver,
|
||||
ModelFiles: modelFiles,
|
||||
PrefixProvider: prefixProvider,
|
||||
PrefixConfig: prefixCfg,
|
||||
Pressure: pressure,
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
package application
|
||||
|
||||
import (
|
||||
"path/filepath"
|
||||
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
"github.com/mudler/LocalAI/core/gallery"
|
||||
"github.com/mudler/LocalAI/pkg/utils"
|
||||
)
|
||||
|
||||
// declaredModelFiles resolves the files a model needs on disk beyond the ones
|
||||
// its config names: what its gallery install or import declared under
|
||||
// `files:`, and what the config itself lists under download_files. The
|
||||
// distributed router stages these to workers, which cannot see the frontend's
|
||||
// models directory.
|
||||
func declaredModelFiles(configLoader *config.ModelConfigLoader, modelsPath string) func(modelName string) []string {
|
||||
return func(modelName string) []string {
|
||||
files := gallery.InstalledModelFiles(modelsPath, modelName)
|
||||
if cfg, ok := configLoader.GetModelConfig(modelName); ok {
|
||||
for _, f := range cfg.DownloadFiles {
|
||||
if utils.VerifyPath(f.Filename, modelsPath) != nil {
|
||||
continue
|
||||
}
|
||||
files = append(files, filepath.Join(modelsPath, f.Filename))
|
||||
}
|
||||
}
|
||||
return files
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,41 @@
|
||||
package application
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
"github.com/mudler/LocalAI/core/gallery"
|
||||
)
|
||||
|
||||
var _ = Describe("declaredModelFiles", func() {
|
||||
It("combines the gallery install's files with the config's download_files", func() {
|
||||
modelsPath := GinkgoT().TempDir()
|
||||
Expect(os.WriteFile(filepath.Join(modelsPath, "big.yaml"), []byte(`
|
||||
name: big
|
||||
backend: llama-cpp
|
||||
parameters:
|
||||
model: big/Big-00001-of-00002.gguf
|
||||
download_files:
|
||||
- filename: big/extra.bin
|
||||
uri: https://example.com/extra.bin
|
||||
`), 0o644)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("big")), []byte(`
|
||||
files:
|
||||
- filename: big/Big-00001-of-00002.gguf
|
||||
- filename: big/Big-00002-of-00002.gguf
|
||||
`), 0o644)).To(Succeed())
|
||||
|
||||
loader := config.NewModelConfigLoader(modelsPath)
|
||||
Expect(loader.LoadModelConfigsFromPath(modelsPath)).To(Succeed())
|
||||
|
||||
Expect(declaredModelFiles(loader, modelsPath)("big")).To(ConsistOf(
|
||||
filepath.Join(modelsPath, "big/Big-00001-of-00002.gguf"),
|
||||
filepath.Join(modelsPath, "big/Big-00002-of-00002.gguf"),
|
||||
filepath.Join(modelsPath, "big/extra.bin"),
|
||||
))
|
||||
})
|
||||
})
|
||||
+2
-2
@@ -88,7 +88,7 @@ func ModelTTS(
|
||||
// a FS path
|
||||
mp := filepath.Join(loader.ModelPath, modelConfig.Model)
|
||||
if _, err := os.Stat(mp); err == nil {
|
||||
if err := utils.VerifyPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil {
|
||||
if err := utils.VerifyResolvedPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil {
|
||||
return "", nil, err
|
||||
}
|
||||
modelPath = mp
|
||||
@@ -189,7 +189,7 @@ func ModelTTSStream(
|
||||
// a FS path
|
||||
mp := filepath.Join(loader.ModelPath, modelConfig.Model)
|
||||
if _, err := os.Stat(mp); err == nil {
|
||||
if err := utils.VerifyPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil {
|
||||
if err := utils.VerifyResolvedPath(mp, appConfig.SystemState.Model.ModelsPath); err != nil {
|
||||
return err
|
||||
}
|
||||
modelPath = mp
|
||||
|
||||
@@ -20,6 +20,7 @@ import (
|
||||
"github.com/mudler/LocalAI/core/services/jobs"
|
||||
mcpRemote "github.com/mudler/LocalAI/core/services/mcp"
|
||||
"github.com/mudler/LocalAI/core/services/messaging"
|
||||
"github.com/mudler/LocalAI/internal"
|
||||
"github.com/mudler/cogito"
|
||||
"github.com/mudler/cogito/clients"
|
||||
"github.com/mudler/xlog"
|
||||
@@ -163,6 +164,8 @@ func (cmd *AgentWorkerCMD) Run(ctx *cliContext.Context) error {
|
||||
registrationBody := map[string]any{
|
||||
"name": nodeName,
|
||||
"node_type": "agent",
|
||||
"version": internal.Version,
|
||||
"commit": internal.Commit,
|
||||
}
|
||||
if cmd.RegistrationToken != "" {
|
||||
registrationBody["token"] = cmd.RegistrationToken
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
package gallery_test
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
|
||||
"github.com/mudler/LocalAI/core/gallery"
|
||||
"github.com/mudler/LocalAI/pkg/system"
|
||||
)
|
||||
|
||||
// DeleteModelFromSystem removes files named by a model name and by the
|
||||
// model's gallery file. Neither may reach outside the models directory: the
|
||||
// name can come from an API caller or from an assistant tool call, and the
|
||||
// gallery file is a YAML file on disk.
|
||||
var _ = Describe("DeleteModelFromSystem path containment", func() {
|
||||
var (
|
||||
root string
|
||||
modelsPath string
|
||||
outside string
|
||||
state *system.SystemState
|
||||
)
|
||||
|
||||
BeforeEach(func() {
|
||||
root = GinkgoT().TempDir()
|
||||
modelsPath = filepath.Join(root, "models")
|
||||
outside = filepath.Join(root, "outside")
|
||||
Expect(os.MkdirAll(modelsPath, 0o755)).To(Succeed())
|
||||
Expect(os.MkdirAll(outside, 0o755)).To(Succeed())
|
||||
var err error
|
||||
state, err = system.GetSystemState(system.WithModelPath(modelsPath))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
})
|
||||
|
||||
It("refuses a model name that escapes the models directory", func() {
|
||||
victim := filepath.Join(outside, "victim.yaml")
|
||||
Expect(os.WriteFile(victim, []byte("name: victim\n"), 0o644)).To(Succeed())
|
||||
|
||||
Expect(gallery.DeleteModelFromSystem(state, "../outside/victim")).ToNot(Succeed())
|
||||
Expect(victim).To(BeARegularFile())
|
||||
})
|
||||
|
||||
It("does not remove gallery-declared files outside the models directory", func() {
|
||||
secret := filepath.Join(outside, "secret.bin")
|
||||
Expect(os.WriteFile(secret, []byte("x"), 0o644)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(modelsPath, "m.yaml"), []byte("name: m\n"), 0o644)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("m")), []byte(`
|
||||
files:
|
||||
- filename: ../outside/secret.bin
|
||||
`), 0o644)).To(Succeed())
|
||||
|
||||
_ = gallery.DeleteModelFromSystem(state, "m")
|
||||
Expect(secret).To(BeARegularFile())
|
||||
})
|
||||
|
||||
It("still deletes a normal model and its declared files", func() {
|
||||
weights := filepath.Join(modelsPath, "m", "w.gguf")
|
||||
Expect(os.MkdirAll(filepath.Dir(weights), 0o755)).To(Succeed())
|
||||
Expect(os.WriteFile(weights, []byte("w"), 0o644)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(modelsPath, "m.yaml"), []byte("name: m\nparameters:\n model: m/w.gguf\n"), 0o644)).To(Succeed())
|
||||
Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName("m")), []byte(`
|
||||
files:
|
||||
- filename: m/w.gguf
|
||||
`), 0o644)).To(Succeed())
|
||||
|
||||
Expect(gallery.DeleteModelFromSystem(state, "m")).To(Succeed())
|
||||
Expect(weights).ToNot(BeAnExistingFile())
|
||||
Expect(filepath.Join(modelsPath, "m.yaml")).ToNot(BeAnExistingFile())
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,46 @@
|
||||
package gallery
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"github.com/mudler/LocalAI/pkg/utils"
|
||||
"github.com/mudler/xlog"
|
||||
)
|
||||
|
||||
// InstalledModelFiles returns the absolute paths of the files that the install
|
||||
// of model name declared (the entry's `files:`), as recorded in its gallery
|
||||
// file. A model config names only the file a backend opens first, while a
|
||||
// backend can read more by itself (llama.cpp opens the other shards of a split
|
||||
// GGUF by name), so this is the complete list of what the model needs on disk.
|
||||
// It returns nil for a model that was not installed from a gallery or import.
|
||||
func InstalledModelFiles(modelsPath, name string) []string {
|
||||
// Model names can hold path separators; the gallery file flattens them
|
||||
// the same way listModelFiles does.
|
||||
rel := galleryFileName(strings.ReplaceAll(name, string(os.PathSeparator), "__"))
|
||||
if err := utils.VerifyPath(rel, modelsPath); err != nil {
|
||||
return nil
|
||||
}
|
||||
galleryFile := filepath.Join(modelsPath, rel)
|
||||
if _, err := os.Stat(galleryFile); err != nil {
|
||||
return nil
|
||||
}
|
||||
cfg, err := ReadConfigFile[ModelConfig](galleryFile)
|
||||
if err != nil {
|
||||
xlog.Warn("Failed to read gallery file for installed model files", "model", name, "file", galleryFile, "error", err)
|
||||
return nil
|
||||
}
|
||||
|
||||
files := make([]string, 0, len(cfg.Files))
|
||||
for _, f := range cfg.Files {
|
||||
// VerifyPath joins its argument onto modelsPath itself, so it must
|
||||
// get the relative name; an absolute path would always pass.
|
||||
if err := utils.VerifyPath(f.Filename, modelsPath); err != nil {
|
||||
xlog.Warn("Ignoring declared model file outside the models path", "model", name, "file", f.Filename)
|
||||
continue
|
||||
}
|
||||
files = append(files, filepath.Join(modelsPath, f.Filename))
|
||||
}
|
||||
return files
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
package gallery_test
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
|
||||
"github.com/mudler/LocalAI/core/gallery"
|
||||
)
|
||||
|
||||
var _ = Describe("InstalledModelFiles", func() {
|
||||
var modelsPath string
|
||||
|
||||
BeforeEach(func() {
|
||||
modelsPath = GinkgoT().TempDir()
|
||||
})
|
||||
|
||||
writeGalleryFile := func(name, body string) {
|
||||
Expect(os.WriteFile(filepath.Join(modelsPath, gallery.GalleryFileName(name)), []byte(body), 0o644)).To(Succeed())
|
||||
}
|
||||
|
||||
It("returns the files the install declared, under the models path", func() {
|
||||
writeGalleryFile("big", `
|
||||
name: big
|
||||
files:
|
||||
- filename: llama-cpp/models/big/Big-00001-of-00002.gguf
|
||||
uri: huggingface://org/repo/Big-00001-of-00002.gguf
|
||||
- filename: llama-cpp/models/big/Big-00002-of-00002.gguf
|
||||
uri: huggingface://org/repo/Big-00002-of-00002.gguf
|
||||
`)
|
||||
Expect(gallery.InstalledModelFiles(modelsPath, "big")).To(Equal([]string{
|
||||
filepath.Join(modelsPath, "llama-cpp/models/big/Big-00001-of-00002.gguf"),
|
||||
filepath.Join(modelsPath, "llama-cpp/models/big/Big-00002-of-00002.gguf"),
|
||||
}))
|
||||
})
|
||||
|
||||
It("drops entries that escape the models path", func() {
|
||||
writeGalleryFile("evil", `
|
||||
files:
|
||||
- filename: ../outside.gguf
|
||||
- filename: ok.gguf
|
||||
`)
|
||||
Expect(gallery.InstalledModelFiles(modelsPath, "evil")).To(Equal([]string{
|
||||
filepath.Join(modelsPath, "ok.gguf"),
|
||||
}))
|
||||
})
|
||||
|
||||
It("returns nothing for a model that was not installed from a gallery", func() {
|
||||
Expect(gallery.InstalledModelFiles(modelsPath, "handwritten")).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
@@ -808,8 +808,11 @@ func GetLocalModelConfiguration(basePath string, name string) (*ModelConfig, err
|
||||
|
||||
func listModelFiles(systemState *system.SystemState, name string) ([]string, error) {
|
||||
|
||||
// VerifyPath joins its argument onto the models path itself, so every
|
||||
// check below passes the relative name: an already-joined absolute path
|
||||
// always lands inside the base and the check would pass anything.
|
||||
configFile := filepath.Join(systemState.Model.ModelsPath, fmt.Sprintf("%s.yaml", name))
|
||||
if err := utils.VerifyPath(configFile, systemState.Model.ModelsPath); err != nil {
|
||||
if err := utils.VerifyPath(fmt.Sprintf("%s.yaml", name), systemState.Model.ModelsPath); err != nil {
|
||||
return nil, fmt.Errorf("failed to verify path %s: %w", configFile, err)
|
||||
}
|
||||
|
||||
@@ -817,7 +820,7 @@ func listModelFiles(systemState *system.SystemState, name string) ([]string, err
|
||||
name = strings.ReplaceAll(name, string(os.PathSeparator), "__")
|
||||
|
||||
galleryFile := filepath.Join(systemState.Model.ModelsPath, galleryFileName(name))
|
||||
if err := utils.VerifyPath(galleryFile, systemState.Model.ModelsPath); err != nil {
|
||||
if err := utils.VerifyPath(galleryFileName(name), systemState.Model.ModelsPath); err != nil {
|
||||
return nil, fmt.Errorf("failed to verify path %s: %w", galleryFile, err)
|
||||
}
|
||||
|
||||
@@ -847,7 +850,7 @@ func listModelFiles(systemState *system.SystemState, name string) ([]string, err
|
||||
if err == nil && galleryconfig != nil {
|
||||
for _, f := range galleryconfig.Files {
|
||||
fullPath := filepath.Join(systemState.Model.ModelsPath, f.Filename)
|
||||
if err := utils.VerifyPath(fullPath, systemState.Model.ModelsPath); err != nil {
|
||||
if err := utils.VerifyPath(f.Filename, systemState.Model.ModelsPath); err != nil {
|
||||
return allFiles, fmt.Errorf("failed to verify path %s: %w", fullPath, err)
|
||||
}
|
||||
allFiles = append(allFiles, fullPath)
|
||||
@@ -858,7 +861,7 @@ func listModelFiles(systemState *system.SystemState, name string) ([]string, err
|
||||
|
||||
for _, f := range additionalFiles {
|
||||
fullPath := filepath.Join(filepath.Join(systemState.Model.ModelsPath, f))
|
||||
if err := utils.VerifyPath(fullPath, systemState.Model.ModelsPath); err != nil {
|
||||
if err := utils.VerifyPath(f, systemState.Model.ModelsPath); err != nil {
|
||||
return allFiles, fmt.Errorf("failed to verify path %s: %w", fullPath, err)
|
||||
}
|
||||
allFiles = append(allFiles, fullPath)
|
||||
|
||||
@@ -114,6 +114,11 @@ type RegisterNodeRequest struct {
|
||||
// VRAMBudget is the worker's operator-set VRAM cap ("80%" or "12GB"). The
|
||||
// registry resolves and enforces it against the raw reported VRAM.
|
||||
VRAMBudget string `json:"vram_budget,omitempty"`
|
||||
// Version is the LocalAI build version reported by the worker at
|
||||
// registration. Empty for workers registered before this field existed.
|
||||
Version string `json:"version,omitempty"`
|
||||
// Commit is the git commit hash the worker binary was built from.
|
||||
Commit string `json:"commit,omitempty"`
|
||||
}
|
||||
|
||||
// RegisterNodeEndpoint registers a new backend node.
|
||||
@@ -191,6 +196,8 @@ func RegisterNodeEndpoint(registry *nodes.NodeRegistry, expectedToken string, au
|
||||
Capability: req.Capability,
|
||||
MaxReplicasPerModel: maxReplicasPerModel,
|
||||
VRAMBudget: req.VRAMBudget,
|
||||
Version: req.Version,
|
||||
Commit: req.Commit,
|
||||
}
|
||||
|
||||
ctx := c.Request().Context()
|
||||
|
||||
@@ -9966,6 +9966,13 @@ button.collapsible-header:focus-visible {
|
||||
.node-inspector__actions .btn { justify-content: center; min-width: 0; }
|
||||
.node-inspector__back { align-items: center; background: transparent; border: 0; color: var(--color-primary); cursor: pointer; display: flex; font: inherit; font-size: var(--text-xs); gap: 6px; max-width: 285px; overflow: hidden; padding: 3px 0; text-overflow: ellipsis; white-space: nowrap; }
|
||||
.node-inspector__back:focus-visible { border-radius: var(--radius-sm); outline: 2px solid var(--color-primary); outline-offset: 3px; }
|
||||
.node-inspector__models { margin: 10px 0 0; }
|
||||
.node-inspector__models > dd { margin: 0; }
|
||||
.node-inspector__model-list { display: grid; gap: 4px; list-style: none; margin: 6px 0 0; padding: 0; }
|
||||
.node-inspector__model-row { align-items: center; display: flex; flex-wrap: wrap; gap: 6px; font-size: var(--text-xs); }
|
||||
.node-inspector__model-row .cell-mono { font-family: var(--font-mono); font-size: .625rem; overflow-wrap: anywhere; }
|
||||
.node-inspector__model-row .state-pill { border-radius: var(--radius-full); font-size: .5625rem; font-weight: 600; padding: 1px 7px; text-transform: capitalize; }
|
||||
.node-inspector__model-row .text-muted { font-size: .5625rem; }
|
||||
.model-inspector__backends { margin-top: 10px; }
|
||||
.model-inspector__nodes { display: grid; gap: 9px; }
|
||||
.model-inspector__node { background: var(--color-bg-tertiary); border: 1px solid var(--color-border-subtle); border-radius: var(--radius-md); padding: 10px; }
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { useEffect, useRef, useState } from 'react'
|
||||
import StatusPill from './StatusPill'
|
||||
import { formatBytes, formatCapacity, timeAgo } from './nodeStatus'
|
||||
import { formatBytes, formatCapacity, timeAgo, modelStateConfig } from './nodeStatus'
|
||||
import { nodesApi } from '../../utils/api'
|
||||
import { capacityReading, nodeLifecycleAction } from '../../utils/nodeFleet'
|
||||
import useInspectorDrawer from './useInspectorDrawer'
|
||||
@@ -24,6 +24,8 @@ function ResourceBar({ label, total, available, tone }) {
|
||||
export default function NodeInspector({ node, open, onClose, onApprove, onDrain, onResume, onBack, backLabel }) {
|
||||
const [backends, setBackends] = useState(null)
|
||||
const [backendError, setBackendError] = useState('')
|
||||
const [models, setModels] = useState(null)
|
||||
const [modelError, setModelError] = useState('')
|
||||
const nodeId = node?.id
|
||||
const backRef = useRef(null)
|
||||
const closeRef = useRef(null)
|
||||
@@ -48,6 +50,19 @@ export default function NodeInspector({ node, open, onClose, onApprove, onDrain,
|
||||
return () => { current = false }
|
||||
}, [open, nodeId])
|
||||
|
||||
useEffect(() => {
|
||||
if (!open || !nodeId) return undefined
|
||||
let current = true
|
||||
setModels(null)
|
||||
setModelError('')
|
||||
nodesApi.getModels(nodeId).then(data => {
|
||||
if (current) setModels(Array.isArray(data) ? data : [])
|
||||
}).catch(error => {
|
||||
if (current) setModelError(error.message || 'Unable to load models')
|
||||
})
|
||||
return () => { current = false }
|
||||
}, [open, nodeId])
|
||||
|
||||
if (!open || !node) return null
|
||||
const cpuKnown = node.cpu_logical_cores > 0 && Number.isFinite(node.cpu_usage_percent) && Number.isFinite(node.cpu_load_1)
|
||||
const disk = capacityReading(node.total_disk, node.available_disk)
|
||||
@@ -74,6 +89,7 @@ export default function NodeInspector({ node, open, onClose, onApprove, onDrain,
|
||||
<h3>Node</h3>
|
||||
<dl className="node-inspector__metrics">
|
||||
<InspectorMetric label="Address"><span className="node-inspector__address">{node.address || 'No address reported'}</span></InspectorMetric>
|
||||
<InspectorMetric label="Version">{node.version || '—'}</InspectorMetric>
|
||||
<InspectorMetric label="Heartbeat">{timeAgo(node.last_heartbeat)}</InspectorMetric>
|
||||
</dl>
|
||||
<div className="node-inspector__labels" aria-label="Node labels">{Object.keys(node.labels || {}).length ? Object.entries(node.labels).map(([key, value]) => <span key={key}>{key}={value}</span>) : <span className="text-muted">No labels</span>}</div>
|
||||
@@ -94,6 +110,26 @@ export default function NodeInspector({ node, open, onClose, onApprove, onDrain,
|
||||
<InspectorMetric label="Backends">{backendError ? <span className="text-error">{backendError}</span> : backends === null ? 'Loading…' : `${backends.length} backend${backends.length === 1 ? '' : 's'}`}</InspectorMetric>
|
||||
<InspectorMetric label="In-flight work">{node.in_flight_count ?? 0}</InspectorMetric>
|
||||
</dl>
|
||||
<div className="node-inspector__models">
|
||||
<dt className="drawer-eyebrow">Running models</dt>
|
||||
<dd>
|
||||
{modelError ? <span className="text-error">{modelError}</span>
|
||||
: models === null ? <span className="text-muted">Loading…</span>
|
||||
: models.length === 0 ? <span className="text-muted">No models loaded</span>
|
||||
: <ul className="node-inspector__model-list">
|
||||
{models.map(model => {
|
||||
const stCfg = modelStateConfig[model.state] || modelStateConfig.idle
|
||||
return (
|
||||
<li key={model.id || `${model.model_name}#${model.replica_index}`} className="node-inspector__model-row">
|
||||
<span className="cell-mono">{model.model_name}</span>
|
||||
<span className="state-pill" style={{ background: stCfg.bg, color: stCfg.color, border: `1px solid ${stCfg.border}` }}>{model.state}</span>
|
||||
<span className="text-muted">{model.in_flight ?? 0} in flight</span>
|
||||
</li>
|
||||
)
|
||||
})}
|
||||
</ul>}
|
||||
</dd>
|
||||
</div>
|
||||
</section>
|
||||
</div>
|
||||
<footer className="node-inspector__actions">
|
||||
|
||||
@@ -138,6 +138,10 @@ export default function NodeDetail() {
|
||||
<div className="drawer-eyebrow">In-flight</div>
|
||||
<span className="cell-mono">{node.in_flight_count || 0}</span>
|
||||
</div>
|
||||
<div>
|
||||
<div className="drawer-eyebrow">Version</div>
|
||||
<span className="cell-mono">{node.version || '—'}</span>
|
||||
</div>
|
||||
<div>
|
||||
<div className="drawer-eyebrow">Heartbeat</div>
|
||||
<span>{timeAgo(node.last_heartbeat)}</span>
|
||||
|
||||
@@ -93,7 +93,7 @@ func (s *ConfigService) GetConfig(_ context.Context, name string) (*ConfigView,
|
||||
if configPath == "" {
|
||||
return nil, ErrConfigFileMissing
|
||||
}
|
||||
if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil {
|
||||
if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil {
|
||||
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
|
||||
}
|
||||
data, err := os.ReadFile(configPath)
|
||||
@@ -137,7 +137,7 @@ func (s *ConfigService) patchConfig(ctx context.Context, name string, patch map[
|
||||
return nil, fmt.Errorf("%w: PATCH cannot rename model %q to %q; use the model edit endpoint", ErrInvalidConfig, name, patchedName)
|
||||
}
|
||||
configPath := cfg.GetModelConfigFile()
|
||||
if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil {
|
||||
if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil {
|
||||
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
|
||||
}
|
||||
diskYAML, err := os.ReadFile(configPath)
|
||||
@@ -289,7 +289,7 @@ func (s *ConfigService) editYAML(ctx context.Context, name string, body []byte)
|
||||
|
||||
configPath := existing.GetModelConfigFile()
|
||||
modelsPath := s.modelsPath()
|
||||
if err := utils.VerifyPath(configPath, modelsPath); err != nil {
|
||||
if err := utils.VerifyResolvedPath(configPath, modelsPath); err != nil {
|
||||
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
|
||||
}
|
||||
|
||||
@@ -304,7 +304,7 @@ func (s *ConfigService) editYAML(ctx context.Context, name string, body []byte)
|
||||
}
|
||||
newConfigPath := filepath.Join(modelsPath, req.Name+".yaml")
|
||||
paths = append(paths, newConfigPath, filepath.Join(modelsPath, gallery.GalleryFileName(name)), filepath.Join(modelsPath, gallery.GalleryFileName(req.Name)))
|
||||
if err := utils.VerifyPath(newConfigPath, modelsPath); err != nil {
|
||||
if err := utils.VerifyPath(req.Name+".yaml", modelsPath); err != nil {
|
||||
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
|
||||
}
|
||||
if _, err := os.Stat(newConfigPath); err == nil {
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
package modeladmin
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
// A model config can be loaded from outside the models directory (for
|
||||
// example with --config-file). The admin mutations write the config file
|
||||
// back, so they must refuse a file outside the models directory rather than
|
||||
// write wherever the loader found it.
|
||||
var _ = Describe("ConfigService config file containment", func() {
|
||||
var (
|
||||
svc *ConfigService
|
||||
ctx context.Context
|
||||
outside string
|
||||
orig []byte
|
||||
)
|
||||
|
||||
BeforeEach(func() {
|
||||
svc, _ = newTestService()
|
||||
ctx = context.Background()
|
||||
outside = filepath.Join(GinkgoT().TempDir(), "external.yaml")
|
||||
orig = []byte("name: external\nbackend: llama-cpp\n")
|
||||
Expect(os.WriteFile(outside, orig, 0o644)).To(Succeed())
|
||||
Expect(svc.Loader.ReadModelConfig(outside, svc.AppConfig.ToConfigLoaderOptions()...)).To(Succeed())
|
||||
cfg, ok := svc.Loader.GetModelConfig("external")
|
||||
Expect(ok).To(BeTrue())
|
||||
Expect(cfg.GetModelConfigFile()).To(Equal(outside))
|
||||
})
|
||||
|
||||
It("refuses to pin a model whose config file is outside the models directory", func() {
|
||||
_, err := svc.TogglePinned(ctx, "external", ActionPin, nil)
|
||||
Expect(err).To(MatchError(ErrPathNotTrusted))
|
||||
Expect(os.ReadFile(outside)).To(Equal(orig))
|
||||
})
|
||||
|
||||
It("refuses to toggle the state of such a model", func() {
|
||||
_, err := svc.ToggleState(ctx, "external", ActionDisable)
|
||||
Expect(err).To(MatchError(ErrPathNotTrusted))
|
||||
Expect(os.ReadFile(outside)).To(Equal(orig))
|
||||
})
|
||||
|
||||
It("refuses to patch such a model", func() {
|
||||
_, err := svc.PatchConfig(ctx, "external", map[string]any{"context_size": 4096})
|
||||
Expect(err).To(MatchError(ErrPathNotTrusted))
|
||||
Expect(os.ReadFile(outside)).To(Equal(orig))
|
||||
})
|
||||
|
||||
It("refuses to read such a model's config", func() {
|
||||
_, err := svc.GetConfig(ctx, "external")
|
||||
Expect(err).To(MatchError(ErrPathNotTrusted))
|
||||
})
|
||||
})
|
||||
@@ -29,7 +29,7 @@ func (s *ConfigService) TogglePinned(_ context.Context, name string, action Acti
|
||||
if configPath == "" {
|
||||
return nil, ErrConfigFileMissing
|
||||
}
|
||||
if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil {
|
||||
if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil {
|
||||
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
|
||||
}
|
||||
if err := mutateYAMLBoolFlag(configPath, "pinned", action == ActionPin); err != nil {
|
||||
|
||||
@@ -49,7 +49,7 @@ func (s *ConfigService) toggleState(ctx context.Context, name string, action Act
|
||||
if configPath == "" {
|
||||
return nil, ErrConfigFileMissing
|
||||
}
|
||||
if err := utils.VerifyPath(configPath, s.modelsPath()); err != nil {
|
||||
if err := utils.VerifyResolvedPath(configPath, s.modelsPath()); err != nil {
|
||||
return nil, fmt.Errorf("%w: %v", ErrPathNotTrusted, err)
|
||||
}
|
||||
var result *ToggleResult
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
package nodes
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
|
||||
"github.com/mudler/xlog"
|
||||
)
|
||||
|
||||
// declaredExtraFiles returns the files the model's install declared that the
|
||||
// path fields of opts do not already stage: neither named by a field nor
|
||||
// inside a directory a field names. It must run on the local paths, before
|
||||
// staging rewrites the fields to remote ones.
|
||||
func (r *SmartRouter) declaredExtraFiles(trackingKey string, opts *pb.ModelOptions) []string {
|
||||
if r.modelFiles == nil || opts == nil || trackingKey == "" {
|
||||
return nil
|
||||
}
|
||||
covered := append([]string{
|
||||
opts.ModelFile, opts.MMProj, opts.LoraAdapter, opts.DraftModel,
|
||||
opts.CLIPModel, opts.Tokenizer, opts.AudioPath, opts.LoraBase,
|
||||
}, opts.LoraAdapters...)
|
||||
|
||||
seen := map[string]struct{}{}
|
||||
var extra []string
|
||||
for _, p := range r.modelFiles(trackingKey) {
|
||||
p = filepath.Clean(p)
|
||||
if _, dup := seen[p]; dup || coveredByField(p, covered) {
|
||||
continue
|
||||
}
|
||||
seen[p] = struct{}{}
|
||||
extra = append(extra, p)
|
||||
}
|
||||
return extra
|
||||
}
|
||||
|
||||
func coveredByField(path string, fields []string) bool {
|
||||
for _, f := range fields {
|
||||
if f == "" {
|
||||
continue
|
||||
}
|
||||
f = filepath.Clean(f)
|
||||
if path == f || strings.HasPrefix(path, f+string(filepath.Separator)) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// existingFiles drops declared files that are not on the frontend. An install
|
||||
// can declare files that are gone by load time (an archive unpacked and then
|
||||
// removed, say), so a missing one is not a reason to refuse the load; the
|
||||
// backend reports it if it really needed it.
|
||||
func existingFiles(paths []string, nodeName, trackingKey string) []string {
|
||||
out := paths[:0:0]
|
||||
for _, p := range paths {
|
||||
if _, err := os.Stat(p); err != nil {
|
||||
xlog.Warn("Skipping staging for declared model file that is not on the frontend", "path", p, "node", nodeName, "model", trackingKey, "error", err)
|
||||
continue
|
||||
}
|
||||
out = append(out, p)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// stagingPayloadBytes totals the on-disk size of everything staging uploads
|
||||
// for a model: the path fields plus the declared files they do not cover. The
|
||||
// first shard of a split GGUF can be a few MB of metadata while the weights
|
||||
// sit in the others, so sizing the fields alone starves the load budget and
|
||||
// the disk-headroom check.
|
||||
func (r *SmartRouter) stagingPayloadBytes(trackingKey string, opts *pb.ModelOptions) int64 {
|
||||
total := modelPayloadBytes(opts)
|
||||
for _, p := range r.declaredExtraFiles(trackingKey, opts) {
|
||||
total += pathBytes(p)
|
||||
}
|
||||
return total
|
||||
}
|
||||
@@ -124,6 +124,11 @@ type BackendNode struct {
|
||||
// worker's re-registration value does not clobber it (mirrors
|
||||
// MaxReplicasPerModelManuallySet).
|
||||
VRAMBudgetManuallySet bool `gorm:"column:vram_budget_manually_set;default:false" json:"vram_budget_manually_set"`
|
||||
// Version is the LocalAI build version reported by the worker at
|
||||
// registration. Empty for workers registered before this field existed.
|
||||
Version string `gorm:"column:version;size:64" json:"version,omitempty"`
|
||||
// Commit is the git commit hash the worker binary was built from.
|
||||
Commit string `gorm:"column:commit;size:64" json:"commit,omitempty"`
|
||||
APIKeyID string `gorm:"size:36" json:"-"` // auto-provisioned API key ID (for cleanup)
|
||||
AuthUserID string `gorm:"size:36" json:"-"` // auto-provisioned user ID (for cleanup)
|
||||
LastHeartbeat time.Time `gorm:"column:last_heartbeat" json:"last_heartbeat"`
|
||||
|
||||
@@ -73,6 +73,12 @@ type SmartRouterOptions struct {
|
||||
// nil disables the exclusion. Deliberate teardown (UnloadModel, admin
|
||||
// endpoints, node drain) is unaffected.
|
||||
PinnedResolver PinnedModelResolver
|
||||
// ModelFiles, when set, returns the absolute local paths of every file a
|
||||
// model's install declared (gallery `files:`, config `download_files`).
|
||||
// The path fields of a load request name only what the backend opens
|
||||
// first; this is how staging learns about the rest, such as the other
|
||||
// shards of a split GGUF. nil stages the path fields alone.
|
||||
ModelFiles func(modelName string) []string
|
||||
// PrefixProvider, when set, enables prefix-cache-aware routing: requests
|
||||
// carrying a prompt prefix chain (distributedhdr.PrefixChain) are biased
|
||||
// toward the node that already holds the longest matching prefix, subject
|
||||
@@ -189,6 +195,9 @@ type SmartRouter struct {
|
||||
// pinnedResolver feeds the eviction paths the set of pinned model names
|
||||
// (see SmartRouterOptions.PinnedResolver). nil disables the exclusion.
|
||||
pinnedResolver PinnedModelResolver
|
||||
// modelFiles resolves a model's declared files (see
|
||||
// SmartRouterOptions.ModelFiles). nil stages the path fields alone.
|
||||
modelFiles func(modelName string) []string
|
||||
// prefixProvider is the prefix-cache routing seam (nil disables it; see
|
||||
// SmartRouterOptions.PrefixProvider). prefixConfig holds the global policy
|
||||
// and thresholds.
|
||||
@@ -283,6 +292,7 @@ func NewSmartRouter(registry ModelRouter, opts SmartRouterOptions) *SmartRouter
|
||||
stagingTracker: NewStagingTracker(),
|
||||
conflictResolver: opts.ConflictResolver,
|
||||
pinnedResolver: opts.PinnedResolver,
|
||||
modelFiles: opts.ModelFiles,
|
||||
probeCache: newProbeCache(probeCacheTTL),
|
||||
prefixProvider: opts.PrefixProvider,
|
||||
prefixConfig: opts.PrefixConfig,
|
||||
@@ -425,7 +435,7 @@ func (r *SmartRouter) scheduleAndLoad(ctx context.Context, backendType, tracking
|
||||
// Size the remote load budget BEFORE staging: stageModelFiles rewrites the
|
||||
// path fields to their remote equivalents on a clone, and only the local
|
||||
// paths can be stat'ed here.
|
||||
payloadBytes := modelPayloadBytes(modelOpts)
|
||||
payloadBytes := r.stagingPayloadBytes(trackingKey, modelOpts)
|
||||
loadTimeout := r.loadTimeoutFor(payloadBytes)
|
||||
|
||||
// Pre-stage model files via FileStager before loading
|
||||
@@ -1367,7 +1377,7 @@ func (r *SmartRouter) narrowByDiskHeadroom(ctx context.Context, modelID string,
|
||||
return candidateNodeIDs, nil
|
||||
}
|
||||
|
||||
requiredDisk := DiskRequirementFor(modelPayloadBytes(modelOpts))
|
||||
requiredDisk := DiskRequirementFor(r.stagingPayloadBytes(modelID, modelOpts))
|
||||
diskCandidates, diskErr := r.registry.NarrowByDiskHeadroom(ctx, candidateNodeIDs, requiredDisk)
|
||||
|
||||
// The check runs even when disabled. "Disabled" means do not BLOCK, not do
|
||||
@@ -1568,6 +1578,10 @@ func (r *SmartRouter) stageModelFiles(ctx context.Context, node *BackendNode, op
|
||||
localModelDir = filepath.Dir(opts.ModelFile)
|
||||
}
|
||||
|
||||
// Resolved before the path fields are rewritten to remote paths below,
|
||||
// since that is what tells which declared files the fields already cover.
|
||||
declared := existingFiles(r.declaredExtraFiles(trackingKey, opts), node.Name, trackingKey)
|
||||
|
||||
// keyMapper generates storage keys namespaced under trackingKey, preserving
|
||||
// subdirectory structure relative to frontendModelsDir. This ensures:
|
||||
// 1. All files for a model land in one directory on the worker for clean deletion
|
||||
@@ -1614,6 +1628,7 @@ func (r *SmartRouter) stageModelFiles(ctx context.Context, node *BackendNode, op
|
||||
totalFiles++
|
||||
}
|
||||
}
|
||||
totalFiles += len(declared)
|
||||
|
||||
// Start tracking staging progress
|
||||
r.stagingTracker.Start(trackingKey, node.Name, totalFiles)
|
||||
@@ -1757,6 +1772,21 @@ func (r *SmartRouter) stageModelFiles(ctx context.Context, node *BackendNode, op
|
||||
}
|
||||
}
|
||||
|
||||
for _, localPath := range declared {
|
||||
fileIdx++
|
||||
fileName := filepath.Base(localPath)
|
||||
stageCtx := r.withStagingCallback(ctx, trackingKey, fileName, fileIdx, totalFiles)
|
||||
|
||||
xlog.Info("Staging declared model file", "model", trackingKey, "node", node.Name, "file", fileName, "fileIndex", fileIdx, "totalFiles", totalFiles)
|
||||
if _, err := r.fileStager.EnsureRemote(stageCtx, node.ID, localPath, keyMapper.Key(localPath)); err != nil {
|
||||
// The install declared it, so the backend may read it: loading
|
||||
// without it fails later with a less useful error.
|
||||
xlog.Error("Failed to stage declared model file for remote node", "node", node.Name, "path", localPath, "error", err)
|
||||
return nil, fmt.Errorf("staging declared model file %s: %w", localPath, err)
|
||||
}
|
||||
r.stagingTracker.FileComplete(trackingKey, fileIdx, totalFiles)
|
||||
}
|
||||
|
||||
// Stage file paths referenced in generic Options (key:value pairs where values
|
||||
// are file paths). Options stay as relative paths — backends resolve them via ModelPath.
|
||||
for _, options := range [][]string{opts.Options, opts.Overrides} {
|
||||
|
||||
@@ -0,0 +1,126 @@
|
||||
package nodes
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
|
||||
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
|
||||
)
|
||||
|
||||
// A model's config names only the file the backend opens first, but its
|
||||
// install can declare more that the backend reads by itself: llama.cpp opens
|
||||
// the "-0000N-of-0000M" shards of a split GGUF from the directory of the first
|
||||
// one. The worker has no view of the frontend's models directory, so every
|
||||
// declared file must be staged, or the load fails with "failed to load GGUF
|
||||
// split".
|
||||
var _ = Describe("stageModelFiles declared model files", func() {
|
||||
var (
|
||||
stager *fakeFileStager
|
||||
router *SmartRouter
|
||||
node *BackendNode
|
||||
modelDir string
|
||||
shards []string
|
||||
mmproj string
|
||||
declared map[string][]string
|
||||
)
|
||||
|
||||
BeforeEach(func() {
|
||||
stager = &fakeFileStager{}
|
||||
declared = map[string][]string{}
|
||||
router = &SmartRouter{
|
||||
fileStager: stager,
|
||||
stagingTracker: NewStagingTracker(),
|
||||
modelFiles: func(name string) []string { return declared[name] },
|
||||
}
|
||||
node = &BackendNode{ID: "node-1", Name: "node-1", Address: "10.0.0.1:50051"}
|
||||
root := GinkgoT().TempDir()
|
||||
modelDir = filepath.Join(root, "llama-cpp", "models", "big")
|
||||
Expect(os.MkdirAll(modelDir, 0o755)).To(Succeed())
|
||||
|
||||
shards = nil
|
||||
for _, name := range []string{
|
||||
"Big-Q4_K_M-00001-of-00003.gguf",
|
||||
"Big-Q4_K_M-00002-of-00003.gguf",
|
||||
"Big-Q4_K_M-00003-of-00003.gguf",
|
||||
} {
|
||||
p := filepath.Join(modelDir, name)
|
||||
Expect(os.WriteFile(p, []byte("shard "+name), 0o644)).To(Succeed())
|
||||
shards = append(shards, p)
|
||||
}
|
||||
mmproj = filepath.Join(root, "llama-cpp", "mmproj", "big", "mmproj.gguf")
|
||||
Expect(os.MkdirAll(filepath.Dir(mmproj), 0o755)).To(Succeed())
|
||||
Expect(os.WriteFile(mmproj, []byte("mmproj"), 0o644)).To(Succeed())
|
||||
})
|
||||
|
||||
opts := func() *pb.ModelOptions {
|
||||
return &pb.ModelOptions{
|
||||
Model: "llama-cpp/models/big/Big-Q4_K_M-00001-of-00003.gguf",
|
||||
ModelFile: shards[0],
|
||||
MMProj: mmproj,
|
||||
}
|
||||
}
|
||||
|
||||
stagedPaths := func() []string {
|
||||
out := make([]string, 0, len(stager.ensureCalls))
|
||||
for _, c := range stager.ensureCalls {
|
||||
out = append(out, c.localPath)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
It("stages every declared file once, beside the ones the config names", func() {
|
||||
declared["big"] = append(append([]string{}, shards...), mmproj)
|
||||
|
||||
staged, err := router.stageModelFiles(context.Background(), node, opts(), "big")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(stagedPaths()).To(ConsistOf(shards[0], mmproj, shards[1], shards[2]))
|
||||
|
||||
// llama.cpp derives the other shards' paths from the first one, so
|
||||
// they must land in the same remote directory.
|
||||
for _, c := range stager.ensureCalls {
|
||||
if c.localPath != mmproj {
|
||||
Expect(filepath.Dir(c.key)).To(Equal(filepath.Dir(stager.ensureCalls[0].key)))
|
||||
}
|
||||
}
|
||||
Expect(staged.ModelFile).To(Equal("/remote/" + stager.ensureCalls[0].key))
|
||||
})
|
||||
|
||||
It("sizes declared files for the load budget and disk check", func() {
|
||||
declared["big"] = append(append([]string{}, shards...), mmproj)
|
||||
|
||||
var want int64
|
||||
for _, p := range append(append([]string{}, shards...), mmproj) {
|
||||
fi, err := os.Stat(p)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
want += fi.Size()
|
||||
}
|
||||
Expect(router.stagingPayloadBytes("big", opts())).To(Equal(want))
|
||||
})
|
||||
|
||||
It("skips a declared file that is missing locally instead of failing", func() {
|
||||
declared["big"] = append(append([]string{}, shards...), filepath.Join(modelDir, "gone.bin"))
|
||||
|
||||
_, err := router.stageModelFiles(context.Background(), node, opts(), "big")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(stagedPaths()).To(ConsistOf(shards[0], mmproj, shards[1], shards[2]))
|
||||
})
|
||||
|
||||
It("does not stage a declared file twice when a directory field covers it", func() {
|
||||
declared["dir"] = []string{shards[1]}
|
||||
|
||||
_, err := router.stageModelFiles(context.Background(), node,
|
||||
&pb.ModelOptions{Model: "llama-cpp/models/big", ModelFile: modelDir}, "dir")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(stagedPaths()).To(ConsistOf(shards[0], shards[1], shards[2]))
|
||||
})
|
||||
|
||||
It("stages only the named files for a model that declares none", func() {
|
||||
_, err := router.stageModelFiles(context.Background(), node, opts(), "handwritten")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(stagedPaths()).To(ConsistOf(shards[0], mmproj))
|
||||
})
|
||||
})
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"github.com/mudler/LocalAI/internal"
|
||||
"github.com/mudler/LocalAI/pkg/system"
|
||||
"github.com/mudler/LocalAI/pkg/xsysinfo"
|
||||
"github.com/mudler/xlog"
|
||||
@@ -154,6 +155,8 @@ func (cfg *Config) registrationBody() map[string]any {
|
||||
"gpu_compute_capability": gpuComputeCap,
|
||||
"capability": capability,
|
||||
"max_replicas_per_model": maxReplicas,
|
||||
"version": internal.Version,
|
||||
"commit": internal.Commit,
|
||||
}
|
||||
|
||||
// Report free space on the filesystem that backs the MODELS directory.
|
||||
|
||||
@@ -41,7 +41,7 @@ func InstallModelsWithOptions(ctx context.Context, galleryService *galleryop.Gal
|
||||
// Check if it's a model gallery, or print a warning
|
||||
e, found := installModel(ctx, galleries, backendGalleries, url, systemState, modelLoader, downloadStatus, enforceScan, autoloadBackendGalleries, requireBackendIntegrity, installOptions...)
|
||||
if e != nil && found {
|
||||
xlog.Error("[startup] failed installing model", "error", err, "model", url)
|
||||
xlog.Error("[startup] failed installing model", "error", e, "model", url)
|
||||
err = errors.Join(err, e)
|
||||
} else if !found {
|
||||
xlog.Debug("[startup] model not found in the gallery", "model", url)
|
||||
@@ -54,7 +54,7 @@ func InstallModelsWithOptions(ctx context.Context, galleryService *galleryop.Gal
|
||||
modelConfig, discoverErr := importers.DiscoverModelConfig(url, json.RawMessage{})
|
||||
if discoverErr != nil {
|
||||
xlog.Error("[startup] failed to discover model config", "error", discoverErr, "model", url)
|
||||
err = errors.Join(discoverErr, fmt.Errorf("failed to discover model config: %w", err))
|
||||
err = errors.Join(err, fmt.Errorf("failed to discover model config: %w", discoverErr))
|
||||
continue
|
||||
}
|
||||
|
||||
|
||||
+1
-1
@@ -41,7 +41,7 @@ services:
|
||||
# Here we can specify a list of models to run (see quickstart https://localai.io/basics/getting_started/#running-models )
|
||||
# or an URL pointing to a YAML configuration file, for example:
|
||||
# - https://gist.githubusercontent.com/mudler/ad601a0488b497b69ec549150d9edd18/raw/a8a8869ef1bb7e3830bf5c0bae29a0cce991ff8d/phi-2.yaml
|
||||
- phi-2
|
||||
- phi-2-chat
|
||||
# For NVIDIA GPU support with CDI (recommended for NVIDIA Container Toolkit 1.14+):
|
||||
# Uncomment the following deploy section and use driver: nvidia.com/gpu.
|
||||
# Include `utility` in capabilities so nvidia-smi / NVML are available —
|
||||
|
||||
@@ -74,6 +74,8 @@ When using `--models-config-file`, you can define multiple models as a list:
|
||||
backend: llama-cpp
|
||||
```
|
||||
|
||||
LocalAI changes only config files that are inside the models directory. If the file from `--models-config-file` is outside the models directory, you cannot view, edit, pin, enable or disable its models from the web UI or the model admin API. Edit the file directly, then restart LocalAI.
|
||||
|
||||
## Core Configuration Fields
|
||||
|
||||
### Basic Model Settings
|
||||
|
||||
@@ -243,7 +243,7 @@ The devices in the following list have been tested with `hipblas` images.
|
||||
|
||||
1. Check your GPU LLVM target is compatible with the version of ROCm. This can be found in the [LLVM Docs](https://llvm.org/docs/AMDGPUUsage.html).
|
||||
2. Check which ROCm version is compatible with your LLVM target and your chosen OS (pay special attention to supported kernel versions). See the [ROCm compatibility matrix](https://rocm.docs.amd.com/en/latest/compatibility/compatibility-matrix.html).
|
||||
3. Install your chosen version of the `dkms` and `rocm` (it is recommended that the native package manager be used for this process for any OS as version changes are executed more easily via this method if updates are required). Take care to restart after installing `amdgpu-dkms` and before installing `rocm`, for details regarding this see the [ROCm installation documentation](https://rocm.docs.amd.com/projects/install-on-linux/en/latest/how-to/native-install/index.html).
|
||||
3. Install your chosen version of the `dkms` and `rocm` (it is recommended that the native package manager be used for this process for any OS as version changes are executed more easily via this method if updates are required). Take care to restart after installing `amdgpu-dkms` and before installing `rocm`, for details regarding this see the [ROCm installation documentation](https://rocm.docs.amd.com/projects/install-on-linux/en/latest/install/install-methods/package-manager-index.html).
|
||||
4. Deploy. Yes it's that easy.
|
||||
|
||||
#### Setup Example (Docker/containerd)
|
||||
|
||||
@@ -740,6 +740,19 @@ Set `LOCALAI_DISTRIBUTED_SHARED_MODELS=true` (or `--distributed-shared-models`)
|
||||
|
||||
This flag is a contract you assert: all nodes must mount identical paths. Leave it off (the default) when workers have independent models directories - the frontend stages files to them over HTTP (or S3) as described above.
|
||||
|
||||
### Which files are staged
|
||||
|
||||
The frontend stages the files that the model config names (`parameters.model`, `mmproj`, draft model, LoRA adapters and similar fields). It also stages every other file that the model declares:
|
||||
|
||||
- The `files:` of the gallery entry or `/import-model` import that installed the model. LocalAI records these in `._gallery_<name>.yaml` next to the model config.
|
||||
- The `download_files:` of the model config.
|
||||
|
||||
A backend can read files that the config does not name. For example, llama.cpp opens all shards of a split GGUF (`<name>-00002-of-00004.gguf` and the rest) from the directory of the first shard. The worker cannot see the frontend's models directory, so it gets only the files that the frontend stages.
|
||||
|
||||
If you write a model config by hand and the model has files like these, list them under `download_files:`. If you do not, the worker gets only the first shard and the load fails with `failed to load GGUF split`.
|
||||
|
||||
The file sizes used for the load deadline and for the disk headroom check include all of these files.
|
||||
|
||||
### Model artifact staging
|
||||
|
||||
For managed Hugging Face artifacts, the controller resolves the repository and
|
||||
|
||||
@@ -98,7 +98,7 @@ LLAMACPP_GRPC_SERVERS="address1:port,address2:port" local-ai run
|
||||
```
|
||||
The workload on the LocalAI server will then be distributed across the specified nodes.
|
||||
|
||||
Alternatively, you can build the RPC workers/server following the llama.cpp [README](https://github.com/ggerganov/llama.cpp/blob/master/examples/rpc/README.md), which is compatible with LocalAI.
|
||||
Alternatively, you can build the RPC workers/server following the llama.cpp [README](https://github.com/ggml-org/llama.cpp/blob/master/tools/rpc/README.md), which is compatible with LocalAI.
|
||||
|
||||
## Manual example (worker)
|
||||
|
||||
|
||||
@@ -373,7 +373,7 @@ curl $LOCALAI/models/apply -H "Content-Type: application/json" -d '{
|
||||
where:
|
||||
- `localai` is the repository. It is optional and can be omitted. If the repository is omitted LocalAI will search the model by name in all the repositories. In the case the same model name is present in both galleries the first match wins.
|
||||
- `bert-embeddings` is the model name in the gallery
|
||||
(read its [config here](https://github.com/mudler/LocalAI/tree/master/gallery/blob/main/bert-embeddings.yaml)).
|
||||
(read its [config here](https://github.com/mudler/LocalAI/blob/master/gallery/index.yaml)).
|
||||
|
||||
### Model variants
|
||||
|
||||
|
||||
@@ -587,7 +587,7 @@ The `llama.cpp` backend supports additional configuration options that can be sp
|
||||
|--------|------|-------------|---------|
|
||||
| `use_jinja` or `jinja` | boolean | Enable Jinja2 template processing for chat templates. When enabled, the backend uses Jinja2-based chat templates from the model for formatting messages. | `use_jinja:true` |
|
||||
| `context_shift` | boolean | Enable context shifting, which allows the model to dynamically adjust context window usage. | `context_shift:true` |
|
||||
| `cache_ram` | integer | Size budget in MiB for the **server-side prompt cache** (a host-RAM store of idle slot KV states that's reloaded on a prompt-prefix hit, see [upstream PR #16391](https://github.com/ggml-org/llama.cpp/pull/16391)). Default: `-1` (no limit). `0` disables the prompt cache entirely. Together with `kv_unified` and `cache_idle_slots` this is what makes a repeated system prompt skip prefill on subsequent calls. | `cache_ram:4096` |
|
||||
| `cache_ram` | integer | Size budget in MiB for the **server-side prompt cache** (a host-RAM store of idle slot KV states that's reloaded on a prompt-prefix hit, see [upstream PR #16391](https://github.com/ggml-org/llama.cpp/pull/16391)). Default: `8192` MiB (llama.cpp default). `-1` removes the limit. `0` disables the prompt cache entirely. Together with `kv_unified` and `cache_idle_slots` this is what makes a repeated system prompt skip prefill on subsequent calls. | `cache_ram:4096` |
|
||||
| `parallel` or `n_parallel` | integer | Enable parallel request processing. When set to a value greater than 1, enables continuous batching for handling multiple requests concurrently. | `parallel:4` |
|
||||
| `grpc_servers` or `rpc_servers` | string | Comma-separated list of gRPC server addresses for distributed inference. Allows distributing workload across multiple llama.cpp workers. | `grpc_servers:localhost:50051,localhost:50052` |
|
||||
| `fit_params` or `fit` | boolean | Enable auto-adjustment of model/context parameters to fit available device memory. Default: `true`. | `fit_params:true` |
|
||||
@@ -643,7 +643,7 @@ Agents, coding assistants, and Anthropic/OpenAI-compatible CLIs typically resend
|
||||
|
||||
| Setting | Default | Role |
|
||||
|---|---|---|
|
||||
| `cache_ram:N` | `-1` (no limit) | Allocates the host-side prompt cache. `0` disables it. |
|
||||
| `cache_ram:N` | `8192` (llama.cpp default) | Allocates the host-side prompt cache. `0` disables it. |
|
||||
| `kv_unified:true` | `true` | Single unified KV buffer (**prerequisite** for idle-slot saving). |
|
||||
| `cache_idle_slots:true` | `true` | Persists the idle slot's KV into the prompt cache on task switch. |
|
||||
|
||||
@@ -658,6 +658,8 @@ options:
|
||||
|
||||
Set `cache_ram:0` to opt out of the prompt cache entirely (saves host RAM at the cost of re-prefilling repeated prompts).
|
||||
|
||||
`cache_ram:-1` removes the limit. With idle-slot saving on, every distinct prompt then leaves its slot state in host RAM, so a workload with many different prompts (classification, ingestion) grows the backend by roughly the KV size of each prompt until the host runs out of memory.
|
||||
|
||||
#### Reference
|
||||
|
||||
- [llama](https://github.com/ggerganov/llama.cpp)
|
||||
@@ -988,6 +990,41 @@ options:
|
||||
The full list of registered parsers lives in `sglang.srt.function_call`
|
||||
and `sglang.srt.parser.reasoning_parser`.
|
||||
|
||||
#### Reasoning defaults and token budgets
|
||||
|
||||
Set SGLang reasoning options in the model's `options:` list:
|
||||
|
||||
```yaml
|
||||
options:
|
||||
- reasoning_parser:qwen3
|
||||
- thinking_budget:512
|
||||
- reasoning_default:on
|
||||
engine_args:
|
||||
enable_strict_thinking: true
|
||||
```
|
||||
|
||||
`thinking_budget` sets a positive integer token budget for reasoning on each request.
|
||||
Invalid, zero, and negative values produce a warning and leave the budget unset.
|
||||
SGLang requires `engine_args.enable_strict_thinking: true` to enforce the budget.
|
||||
LocalAI warns if you configure a budget without that engine option.
|
||||
Keep the budget well below the `max_tokens` of your requests: if `max_tokens` is reached first,
|
||||
the budget never triggers and the whole reply can be spent on reasoning, leaving the answer empty.
|
||||
|
||||
`reasoning_default:on` or `reasoning_default:off` sets the default for LocalAI's tokenizer chat template.
|
||||
Request metadata `enable_thinking` set to `"true"` or `"false"` overrides this default.
|
||||
An explicit prompt bypasses tokenizer template rendering.
|
||||
When no default or request override is set, the template keeps its own behavior.
|
||||
|
||||
LocalAI signals required reasoning when the rendered prompt ends with the configured parser's opening reasoning token.
|
||||
An explicit output grammar disables this detection.
|
||||
Configure a reasoning parser that matches your model.
|
||||
|
||||
The backend reads these options when it loads the model.
|
||||
`POST /models/reload` rereads model configuration files but does not update options in an already loaded backend.
|
||||
Restarting only the backend does not reread configuration files.
|
||||
Restart LocalAI after changing these options to reload both the configuration and the backend.
|
||||
|
||||
|
||||
### vllm.cpp
|
||||
|
||||
[vllm.cpp](https://github.com/mudler/vllm.cpp) is the LocalAI team's C++ port of
|
||||
|
||||
@@ -108,6 +108,8 @@ docker run -ti --name local-ai -p 8080:8080 --runtime nvidia --gpus all localai/
|
||||
|
||||
## Using Compose
|
||||
|
||||
The repository's `docker-compose.yaml` installs `phi-2-chat` from the model gallery by default. Change its `command` list to select a different gallery model.
|
||||
|
||||
For a more manageable setup, especially with persistent volumes, use Docker Compose or Podman Compose:
|
||||
|
||||
### Using CDI (Container Device Interface) - Recommended for NVIDIA Container Toolkit 1.14+
|
||||
|
||||
@@ -25,17 +25,17 @@ Here's an example to initiate the **phi-2** model:
|
||||
docker run -p 8080:8080 localai/localai:{{< version >}} https://gist.githubusercontent.com/mudler/ad601a0488b497b69ec549150d9edd18/raw/a8a8869ef1bb7e3830bf5c0bae29a0cce991ff8d/phi-2.yaml
|
||||
```
|
||||
|
||||
You can also check all the embedded models configurations [here](https://github.com/mudler/LocalAI/tree/master/embedded/models).
|
||||
You can also check all the embedded models configurations [here](https://github.com/mudler/LocalAI/tree/master/gallery).
|
||||
|
||||
{{% notice tip %}}
|
||||
The model configurations used in the quickstart are accessible here: [https://github.com/mudler/LocalAI/tree/master/embedded/models](https://github.com/mudler/LocalAI/tree/master/embedded/models). Contributions are welcome; please feel free to submit a Pull Request.
|
||||
The model configurations used in the quickstart are accessible here: [https://github.com/mudler/LocalAI/tree/master/gallery](https://github.com/mudler/LocalAI/tree/master/gallery). Contributions are welcome; please feel free to submit a Pull Request.
|
||||
|
||||
The `phi-2` model configuration from the quickstart is expanded from [https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml](https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml).
|
||||
The `phi-2` model configuration from the quickstart is expanded from [https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml](https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml).
|
||||
{{% /notice %}}
|
||||
|
||||
## Example: Customizing the Prompt Template
|
||||
|
||||
To modify the prompt template, create a Github gist or a Pastebin file, and copy the content from [https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml](https://github.com/mudler/LocalAI/blob/master/examples/configurations/phi-2.yaml). Alter the fields as needed:
|
||||
To modify the prompt template, create a Github gist or a Pastebin file, and copy the content from [https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml](https://github.com/mudler/LocalAI-examples/blob/main/configurations/phi-2.yaml). Alter the fields as needed:
|
||||
|
||||
```yaml
|
||||
name: phi-2
|
||||
|
||||
@@ -98,7 +98,7 @@ availability may lag upstream releases.
|
||||
- [AnythingLLM](https://github.com/Mintplex-Labs/anything-llm)
|
||||
- [Logseq GPT3 OpenAI plugin](https://github.com/briansunter/logseq-plugin-gpt3-openai)
|
||||
- [CodeGPT (JetBrains)](https://plugins.jetbrains.com/plugin/21056-codegpt) - Custom OpenAI-compatible endpoints
|
||||
- [Wave Terminal](https://docs.waveterm.dev/features/supportedLLMs/localai) - Native LocalAI support
|
||||
- [Wave Terminal](https://docs.waveterm.dev/ai-presets) - Native LocalAI support
|
||||
- [Obsidian BMO Chatbot](https://github.com/longy2k/obsidian-bmo-chatbot)
|
||||
- [spark](https://github.com/cedriking/spark)
|
||||
- [openops (Mattermost)](https://github.com/mattermost/openops)
|
||||
|
||||
@@ -45,7 +45,7 @@ All backends listed here can be installed on demand from the [Backend Gallery]({
|
||||
| [moonshine](https://github.com/moonshine-ai/moonshine) | Ultra-fast transcription for low-end devices (ONNX) | CPU, CUDA 12/13, Metal |
|
||||
| [parakeet.cpp](https://github.com/mudler/parakeet.cpp) | C++/GGML port of NVIDIA NeMo Parakeet (tdt/ctc/rnnt/hybrid), with cache-aware streaming | CPU, CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, Jetson L4T |
|
||||
| [CrispASR](https://github.com/CrispStrobe/CrispASR) | Unified speech engine (whisper.cpp fork) supporting Parakeet, Canary, and many ASR architectures, plus TTS | CPU, CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, Jetson L4T |
|
||||
| [voxtral](https://github.com/mudler/voxtral.c) | Voxtral Realtime 4B speech-to-text in pure C | CPU, Metal |
|
||||
| [voxtral](https://github.com/antirez/voxtral.c) | Voxtral Realtime 4B speech-to-text in pure C | CPU, Metal |
|
||||
| [Qwen3-ASR](https://github.com/QwenLM/Qwen3-ASR) | Qwen3 automatic speech recognition | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal, Jetson L4T |
|
||||
| [NeMo](https://github.com/NVIDIA/NeMo) | NVIDIA NeMo ASR toolkit | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal |
|
||||
| [sherpa-onnx](https://k2-fsa.github.io/sherpa/onnx/) | Sherpa-ONNX ASR (Whisper, Paraformer, SenseVoice) and TTS | CPU, CUDA 12, Metal |
|
||||
@@ -70,10 +70,10 @@ All backends listed here can be installed on demand from the [Backend Gallery]({
|
||||
| [OmniVoice](https://github.com/ServeurpersoCom/omnivoice.cpp) | Native C++/GGML TTS with voice cloning, voice design, and streaming | CPU, CUDA 12/13, ROCm, Intel SYCL, Vulkan, Metal, Jetson L4T |
|
||||
| [fish-speech](https://github.com/fishaudio/fish-speech) | High-quality TTS with voice cloning | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal, Jetson L4T |
|
||||
| [Pocket TTS](https://github.com/kyutai-labs/pocket-tts) | Lightweight CPU-efficient TTS with voice cloning | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal, Jetson L4T |
|
||||
| [OuteTTS](https://github.com/OuteAI/outetts) | TTS with custom speaker voices | CPU, CUDA 12 |
|
||||
| [OuteTTS](https://github.com/edwko/OuteTTS) | TTS with custom speaker voices | CPU, CUDA 12 |
|
||||
| [faster-qwen3-tts](https://github.com/andimarafioti/faster-qwen3-tts) | Real-time Qwen3-TTS with CUDA graph capture | CPU, CUDA 12/13, Jetson L4T |
|
||||
| [NeuTTS Air](https://github.com/neuphonic/neutts-air) | Instant voice cloning, on-device TTS | CPU, CUDA 12, ROCm |
|
||||
| [VoxCPM](https://github.com/ModelBest/VoxCPM) | Expressive end-to-end TTS | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal |
|
||||
| [VoxCPM](https://github.com/OpenBMB/VoxCPM) | Expressive end-to-end TTS | CPU, CUDA 12/13, ROCm, Intel SYCL, Metal |
|
||||
| [Kitten TTS](https://github.com/KittenML/KittenTTS) | Kitten TTS model | CPU, Metal |
|
||||
| [Supertonic](https://github.com/supertone-inc/supertonic) | Lightning-fast on-device multilingual TTS via ONNX | CPU |
|
||||
| [MLX-Audio](https://github.com/Blaizzy/mlx-audio) | Audio models on Apple Silicon | CPU, CUDA 12/13, Metal, Jetson L4T |
|
||||
|
||||
+4
-7
@@ -297,7 +297,7 @@
|
||||
files:
|
||||
- filename: ds4flash.gguf
|
||||
uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF
|
||||
sha256: 9c46395af7320ec1d68afe81ec7fa1c7060a07117dceabfd977f12a95fa30cdf
|
||||
sha256: f33633d55f5379e8571db06674bf7a07a2ea7bb7b44287e1a9bdc3686d69a5c6
|
||||
- name: "qwopus3.8-27b-flash-v2"
|
||||
variants:
|
||||
- model: qwopus3.8-27b-flash-v2-q8
|
||||
@@ -6420,7 +6420,6 @@
|
||||
- filename: Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q4_K_XL.gguf
|
||||
sha256: 8e5601dbd18fbc2b731cf674a040dd32f3ec2d09a312f4e0f3c4d7bc92998837
|
||||
|
||||
- name: sharp-spark-x2.5-4b-q5
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
@@ -6457,7 +6456,6 @@
|
||||
- filename: Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q5_K_XL.gguf
|
||||
sha256: f445f1a57e58b70ea85078e1edcd29763843f71f154bac2efc57eea1b8333a26
|
||||
|
||||
- name: sharp-spark-x2.5-4b-q6
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
@@ -6494,7 +6492,6 @@
|
||||
- filename: Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
|
||||
uri: https://huggingface.co/peculiar-ragdoll/Sharp-Spark-X2.5-4B-GGUF/resolve/e797ddf6a57d9ecfddf68394438d2667ecb42dad/Sharp-Spark-X2.5-4B-Q6_K_XL.gguf
|
||||
sha256: 793e673f34d2dde9674d24d277c25dbf03b89290333835aa31b7ee1d62e20dfc
|
||||
|
||||
- &spark-x2-5-4b
|
||||
name: "spark-x2.5-4b-q4"
|
||||
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
|
||||
@@ -9309,7 +9306,7 @@
|
||||
files:
|
||||
- filename: llama-cpp/models/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-APEX/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-APEX.gguf
|
||||
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-APEX.gguf
|
||||
sha256: 35026b978de6ee6ff63870d3d68be90ab0797a9a6cd5c83332e1f9c510c6a695
|
||||
sha256: 97602082e8639b1e6660b36de0193861742e035711b865faf76abf54a5a2be09
|
||||
- filename: llama-cpp/mmproj/Hermes3.6-35B-A3B-Uncensored-Genesis-Final/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
|
||||
sha256: 5129bb5eb19e4346c0f2071f1ce8e1b0a076e0ab08d57a77d6033ce01235252c
|
||||
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
|
||||
@@ -9399,7 +9396,7 @@
|
||||
files:
|
||||
- filename: llama-cpp/models/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX.gguf
|
||||
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX.gguf
|
||||
sha256: 612561952f0698539a133a479e1cf18ffce85bb6b4311a5d05e50d6160970c75
|
||||
sha256: 3efbc83f38ffa48251ff4c072fbbefce63c995df3cba20fd6bf0d8d7ab965cd9
|
||||
- filename: llama-cpp/mmproj/Hermes3.6-35B-A3B-Uncensored-Genesis-Final/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
|
||||
sha256: 5129bb5eb19e4346c0f2071f1ce8e1b0a076e0ab08d57a77d6033ce01235252c
|
||||
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
|
||||
@@ -9451,7 +9448,7 @@
|
||||
files:
|
||||
- filename: llama-cpp/models/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX-Compact/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX-Compact.gguf
|
||||
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/Hermes3.6-35B-A3B-Uncensored-Genesis-Final-MTP-APEX-Compact.gguf
|
||||
sha256: 7a17aaff5ec81ba34e33d3a8b032a699abf4231d6cbe9930d1c9a159db3e87dc
|
||||
sha256: 27a2edec6f66e585fbf02eabe397f694378474786711362d9f78cd129a0e8313
|
||||
- filename: llama-cpp/mmproj/Hermes3.6-35B-A3B-Uncensored-Genesis-Final/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
|
||||
sha256: 5129bb5eb19e4346c0f2071f1ce8e1b0a076e0ab08d57a77d6033ce01235252c
|
||||
uri: https://huggingface.co/LuffyTheFox/Qwen3.6-35B-A3B-Uncensored-Genesis-Hermes-Final-GGUF/resolve/d0cf3294f07f2c422f0cf793a75fa48f61d48931/mmproj-Hermes3.6-35B-A3B-Uncensored-Genesis-Final-F16.gguf
|
||||
|
||||
+18
-4
@@ -13,21 +13,35 @@ func ExistsInPath(path string, s string) bool {
|
||||
}
|
||||
|
||||
func InTrustedRoot(path string, trustedRoot string) error {
|
||||
for path != "/" {
|
||||
path = filepath.Dir(path)
|
||||
for {
|
||||
parent := filepath.Dir(path)
|
||||
// Dir stops changing at "/" for an absolute path and at "." for a
|
||||
// relative one; waiting for "/" alone spins forever on the latter.
|
||||
if parent == path {
|
||||
return fmt.Errorf("path is outside of trusted root")
|
||||
}
|
||||
path = parent
|
||||
if path == trustedRoot {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
return fmt.Errorf("path is outside of trusted root")
|
||||
}
|
||||
|
||||
// VerifyPath verifies that path is based in basePath.
|
||||
// VerifyPath verifies that path, taken relative to basePath, is based in
|
||||
// basePath. It joins path onto basePath first, so an absolute path is read as
|
||||
// relative to the base as well: give it the untrusted relative name, never a
|
||||
// path that has already been joined. For a full path use VerifyResolvedPath.
|
||||
func VerifyPath(path, basePath string) error {
|
||||
c := filepath.Clean(filepath.Join(basePath, path))
|
||||
return InTrustedRoot(c, filepath.Clean(basePath))
|
||||
}
|
||||
|
||||
// VerifyResolvedPath verifies that path, a full path rather than one relative
|
||||
// to basePath, is based in basePath.
|
||||
func VerifyResolvedPath(path, basePath string) error {
|
||||
return InTrustedRoot(filepath.Clean(path), filepath.Clean(basePath))
|
||||
}
|
||||
|
||||
// SanitizeFileName sanitizes the given filename
|
||||
func SanitizeFileName(fileName string) string {
|
||||
// filepath.Clean to clean the path
|
||||
|
||||
@@ -3,6 +3,7 @@ package utils_test
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"time"
|
||||
|
||||
. "github.com/mudler/LocalAI/pkg/utils"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
@@ -71,6 +72,25 @@ var _ = Describe("utils/path tests", func() {
|
||||
})
|
||||
})
|
||||
|
||||
Describe("VerifyResolvedPath", func() {
|
||||
It("accepts a full path inside the base", func() {
|
||||
Expect(VerifyResolvedPath("/srv/models/a/model.yaml", "/srv/models")).To(Succeed())
|
||||
})
|
||||
|
||||
It("rejects a full path outside the base", func() {
|
||||
// VerifyPath would join this onto the base and accept it.
|
||||
Expect(VerifyResolvedPath("/etc/passwd", "/srv/models")).ToNot(Succeed())
|
||||
})
|
||||
|
||||
It("rejects a joined path that climbed out of the base", func() {
|
||||
Expect(VerifyResolvedPath(filepath.Join("/srv/models", "../other/x"), "/srv/models")).ToNot(Succeed())
|
||||
})
|
||||
|
||||
It("cleans both paths before comparing", func() {
|
||||
Expect(VerifyResolvedPath("/srv/models/./a/../b.yaml", "/srv/models/")).To(Succeed())
|
||||
})
|
||||
})
|
||||
|
||||
Describe("InTrustedRoot", func() {
|
||||
It("accepts a strict descendant of the trusted root", func() {
|
||||
Expect(InTrustedRoot("/srv/models/file", "/srv/models")).To(Succeed())
|
||||
@@ -93,6 +113,21 @@ var _ = Describe("utils/path tests", func() {
|
||||
It("rejects an unrelated absolute path", func() {
|
||||
Expect(InTrustedRoot("/etc/passwd", "/srv/models")).ToNot(Succeed())
|
||||
})
|
||||
|
||||
It("rejects a relative path outside a relative root instead of looping", func() {
|
||||
// Walking up a relative path ends at ".", never at "/", so the
|
||||
// walk must stop when it stops making progress.
|
||||
done := make(chan error, 1)
|
||||
go func() { done <- InTrustedRoot("x", "models") }()
|
||||
Eventually(done).WithTimeout(2 * time.Second).Should(Receive(HaveOccurred()))
|
||||
|
||||
go func() { done <- VerifyPath("../x", "models") }()
|
||||
Eventually(done).WithTimeout(2 * time.Second).Should(Receive(HaveOccurred()))
|
||||
})
|
||||
|
||||
It("accepts a relative descendant of a relative root", func() {
|
||||
Expect(InTrustedRoot("models/a/file", "models")).To(Succeed())
|
||||
})
|
||||
})
|
||||
|
||||
Describe("SanitizeFileName", func() {
|
||||
|
||||
+279
-1
@@ -3770,6 +3770,90 @@ const docTemplate = `{
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/systemone": {
|
||||
"post": {
|
||||
"description": "Runs zero-shot NER over the supplied state and answers each question. Question types: noul (binary entity presence), choice (pick one option), score (pick one level).",
|
||||
"tags": [
|
||||
"systemone"
|
||||
],
|
||||
"summary": "Answer structured-extraction questions over state text.",
|
||||
"parameters": [
|
||||
{
|
||||
"description": "state + questions",
|
||||
"name": "request",
|
||||
"in": "body",
|
||||
"required": true,
|
||||
"schema": {
|
||||
"$ref": "#/definitions/schema.SystemOneRequest"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "OK",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/schema.SystemOneResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/systemone/permute": {
|
||||
"post": {
|
||||
"description": "Re-runs one choice question under n_perm option orders. Reports per-order probabilities, argmax stability, and spread.",
|
||||
"tags": [
|
||||
"systemone"
|
||||
],
|
||||
"summary": "Re-run a choice question under multiple option orders.",
|
||||
"parameters": [
|
||||
{
|
||||
"description": "request + question + n_perm + seed",
|
||||
"name": "request",
|
||||
"in": "body",
|
||||
"required": true,
|
||||
"schema": {
|
||||
"$ref": "#/definitions/schema.SystemOnePermuteRequest"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "OK",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/schema.SystemOnePermuteResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/systemone/separate": {
|
||||
"post": {
|
||||
"description": "Runs N independent NER passes, one per question, against the same state. Response shape matches /v1/systemone.",
|
||||
"tags": [
|
||||
"systemone"
|
||||
],
|
||||
"summary": "Answer each question in a separate NER pass.",
|
||||
"parameters": [
|
||||
{
|
||||
"description": "state + questions",
|
||||
"name": "request",
|
||||
"in": "body",
|
||||
"required": true,
|
||||
"schema": {
|
||||
"$ref": "#/definitions/schema.SystemOneRequest"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "OK",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/schema.SystemOneResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/text-to-speech/{voice-id}": {
|
||||
"post": {
|
||||
"tags": [
|
||||
@@ -4093,8 +4177,16 @@ const docTemplate = `{
|
||||
"config.Gallery": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"artifact_verification": {
|
||||
"description": "ArtifactVerification overrides Verification only for the gallery OCI artifact.\nBackend images keep their separate Verification policy.",
|
||||
"allOf": [
|
||||
{
|
||||
"$ref": "#/definitions/config.GalleryVerification"
|
||||
}
|
||||
]
|
||||
},
|
||||
"mirrors": {
|
||||
"description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://).",
|
||||
"description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://, oci://).",
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "string"
|
||||
@@ -4129,6 +4221,10 @@ const docTemplate = `{
|
||||
"not_before": {
|
||||
"description": "NotBefore is an RFC3339 timestamp. Empty disables the time check.",
|
||||
"type": "string"
|
||||
},
|
||||
"source_repository": {
|
||||
"description": "SourceRepository is an https URL compared exactly against the\ncertificate's source-repository extension. Empty skips the check.",
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
},
|
||||
@@ -7811,12 +7907,42 @@ const docTemplate = `{
|
||||
"id": {
|
||||
"type": "string"
|
||||
},
|
||||
"process": {
|
||||
"description": "Process is the backend process serving the model on this host. Absent\nwhen the model has no local process (a distributed worker holds it) or\nthe process could not be read.",
|
||||
"allOf": [
|
||||
{
|
||||
"$ref": "#/definitions/schema.SysInfoProcess"
|
||||
}
|
||||
]
|
||||
},
|
||||
"size_vram": {
|
||||
"description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
|
||||
"type": "integer"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SysInfoProcess": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"cpu_percent": {
|
||||
"description": "CPUPercent is the share of the whole host's CPU used since the previous\nreading, 0-100. Absent on the first reading of a process.",
|
||||
"type": "number"
|
||||
},
|
||||
"memory_percent": {
|
||||
"type": "number"
|
||||
},
|
||||
"pid": {
|
||||
"type": "integer"
|
||||
},
|
||||
"rss_bytes": {
|
||||
"description": "RSSBytes is resident host memory. Weights offloaded to a GPU are not\nin it.",
|
||||
"type": "integer"
|
||||
},
|
||||
"started_at": {
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemInformationResponse": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
@@ -7836,6 +7962,158 @@ const docTemplate = `{
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOneAnswer": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"choice": {
|
||||
"type": "string"
|
||||
},
|
||||
"confidence": {
|
||||
"type": "number"
|
||||
},
|
||||
"entities": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"$ref": "#/definitions/schema.SystemOneEntity"
|
||||
}
|
||||
},
|
||||
"legend": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"noul": {
|
||||
"type": "number"
|
||||
},
|
||||
"probabilities": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "number",
|
||||
"format": "float64"
|
||||
}
|
||||
},
|
||||
"score": {
|
||||
"type": "number"
|
||||
},
|
||||
"type": {
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOneEntity": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"confidence": {
|
||||
"type": "number"
|
||||
},
|
||||
"end": {
|
||||
"type": "integer"
|
||||
},
|
||||
"start": {
|
||||
"type": "integer"
|
||||
},
|
||||
"text": {
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOnePermuteRequest": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"n_perm": {
|
||||
"type": "integer"
|
||||
},
|
||||
"question": {
|
||||
"type": "string"
|
||||
},
|
||||
"request": {
|
||||
"$ref": "#/definitions/schema.SystemOneRequest"
|
||||
},
|
||||
"seed": {
|
||||
"type": "integer"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOnePermuteResponse": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"argmax_stable": {
|
||||
"type": "boolean"
|
||||
},
|
||||
"runs": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"$ref": "#/definitions/schema.SystemOnePermuteRun"
|
||||
}
|
||||
},
|
||||
"spread": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "number",
|
||||
"format": "float64"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOnePermuteRun": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"choice": {
|
||||
"type": "string"
|
||||
},
|
||||
"latency_ms": {
|
||||
"type": "number"
|
||||
},
|
||||
"order": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"probabilities": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "number",
|
||||
"format": "float64"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOneRequest": {
|
||||
"type": "object"
|
||||
},
|
||||
"schema.SystemOneResponse": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"answers": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"$ref": "#/definitions/schema.SystemOneAnswer"
|
||||
}
|
||||
},
|
||||
"latency_ms": {
|
||||
"type": "number"
|
||||
},
|
||||
"model": {
|
||||
"type": "string"
|
||||
},
|
||||
"usage": {
|
||||
"$ref": "#/definitions/schema.SystemOneUsage"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOneUsage": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"input_tokens": {
|
||||
"type": "integer"
|
||||
},
|
||||
"output_tokens": {
|
||||
"type": "integer"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.TTSRequest": {
|
||||
"description": "TTS request body",
|
||||
"type": "object",
|
||||
|
||||
+279
-1
@@ -3767,6 +3767,90 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/systemone": {
|
||||
"post": {
|
||||
"description": "Runs zero-shot NER over the supplied state and answers each question. Question types: noul (binary entity presence), choice (pick one option), score (pick one level).",
|
||||
"tags": [
|
||||
"systemone"
|
||||
],
|
||||
"summary": "Answer structured-extraction questions over state text.",
|
||||
"parameters": [
|
||||
{
|
||||
"description": "state + questions",
|
||||
"name": "request",
|
||||
"in": "body",
|
||||
"required": true,
|
||||
"schema": {
|
||||
"$ref": "#/definitions/schema.SystemOneRequest"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "OK",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/schema.SystemOneResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/systemone/permute": {
|
||||
"post": {
|
||||
"description": "Re-runs one choice question under n_perm option orders. Reports per-order probabilities, argmax stability, and spread.",
|
||||
"tags": [
|
||||
"systemone"
|
||||
],
|
||||
"summary": "Re-run a choice question under multiple option orders.",
|
||||
"parameters": [
|
||||
{
|
||||
"description": "request + question + n_perm + seed",
|
||||
"name": "request",
|
||||
"in": "body",
|
||||
"required": true,
|
||||
"schema": {
|
||||
"$ref": "#/definitions/schema.SystemOnePermuteRequest"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "OK",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/schema.SystemOnePermuteResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/systemone/separate": {
|
||||
"post": {
|
||||
"description": "Runs N independent NER passes, one per question, against the same state. Response shape matches /v1/systemone.",
|
||||
"tags": [
|
||||
"systemone"
|
||||
],
|
||||
"summary": "Answer each question in a separate NER pass.",
|
||||
"parameters": [
|
||||
{
|
||||
"description": "state + questions",
|
||||
"name": "request",
|
||||
"in": "body",
|
||||
"required": true,
|
||||
"schema": {
|
||||
"$ref": "#/definitions/schema.SystemOneRequest"
|
||||
}
|
||||
}
|
||||
],
|
||||
"responses": {
|
||||
"200": {
|
||||
"description": "OK",
|
||||
"schema": {
|
||||
"$ref": "#/definitions/schema.SystemOneResponse"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"/v1/text-to-speech/{voice-id}": {
|
||||
"post": {
|
||||
"tags": [
|
||||
@@ -4090,8 +4174,16 @@
|
||||
"config.Gallery": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"artifact_verification": {
|
||||
"description": "ArtifactVerification overrides Verification only for the gallery OCI artifact.\nBackend images keep their separate Verification policy.",
|
||||
"allOf": [
|
||||
{
|
||||
"$ref": "#/definitions/config.GalleryVerification"
|
||||
}
|
||||
]
|
||||
},
|
||||
"mirrors": {
|
||||
"description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://).",
|
||||
"description": "Mirrors are tried in order when URL cannot be fetched. They are a\nfallback for availability, not a load-balancing pool: the primary is\nalways preferred, and a mirror is only consulted after the one before\nit fails. Any URI the gallery loader understands works here\n(https://, github:, file://, oci://).",
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "string"
|
||||
@@ -4126,6 +4218,10 @@
|
||||
"not_before": {
|
||||
"description": "NotBefore is an RFC3339 timestamp. Empty disables the time check.",
|
||||
"type": "string"
|
||||
},
|
||||
"source_repository": {
|
||||
"description": "SourceRepository is an https URL compared exactly against the\ncertificate's source-repository extension. Empty skips the check.",
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
},
|
||||
@@ -7808,12 +7904,42 @@
|
||||
"id": {
|
||||
"type": "string"
|
||||
},
|
||||
"process": {
|
||||
"description": "Process is the backend process serving the model on this host. Absent\nwhen the model has no local process (a distributed worker holds it) or\nthe process could not be read.",
|
||||
"allOf": [
|
||||
{
|
||||
"$ref": "#/definitions/schema.SysInfoProcess"
|
||||
}
|
||||
]
|
||||
},
|
||||
"size_vram": {
|
||||
"description": "SizeVRAM is DRM-accounted resident device memory in bytes. Nil means\nthe backend process tree has no complete supported reading.",
|
||||
"type": "integer"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SysInfoProcess": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"cpu_percent": {
|
||||
"description": "CPUPercent is the share of the whole host's CPU used since the previous\nreading, 0-100. Absent on the first reading of a process.",
|
||||
"type": "number"
|
||||
},
|
||||
"memory_percent": {
|
||||
"type": "number"
|
||||
},
|
||||
"pid": {
|
||||
"type": "integer"
|
||||
},
|
||||
"rss_bytes": {
|
||||
"description": "RSSBytes is resident host memory. Weights offloaded to a GPU are not\nin it.",
|
||||
"type": "integer"
|
||||
},
|
||||
"started_at": {
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemInformationResponse": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
@@ -7833,6 +7959,158 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOneAnswer": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"choice": {
|
||||
"type": "string"
|
||||
},
|
||||
"confidence": {
|
||||
"type": "number"
|
||||
},
|
||||
"entities": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"$ref": "#/definitions/schema.SystemOneEntity"
|
||||
}
|
||||
},
|
||||
"legend": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"noul": {
|
||||
"type": "number"
|
||||
},
|
||||
"probabilities": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "number",
|
||||
"format": "float64"
|
||||
}
|
||||
},
|
||||
"score": {
|
||||
"type": "number"
|
||||
},
|
||||
"type": {
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOneEntity": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"confidence": {
|
||||
"type": "number"
|
||||
},
|
||||
"end": {
|
||||
"type": "integer"
|
||||
},
|
||||
"start": {
|
||||
"type": "integer"
|
||||
},
|
||||
"text": {
|
||||
"type": "string"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOnePermuteRequest": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"n_perm": {
|
||||
"type": "integer"
|
||||
},
|
||||
"question": {
|
||||
"type": "string"
|
||||
},
|
||||
"request": {
|
||||
"$ref": "#/definitions/schema.SystemOneRequest"
|
||||
},
|
||||
"seed": {
|
||||
"type": "integer"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOnePermuteResponse": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"argmax_stable": {
|
||||
"type": "boolean"
|
||||
},
|
||||
"runs": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"$ref": "#/definitions/schema.SystemOnePermuteRun"
|
||||
}
|
||||
},
|
||||
"spread": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "number",
|
||||
"format": "float64"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOnePermuteRun": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"choice": {
|
||||
"type": "string"
|
||||
},
|
||||
"latency_ms": {
|
||||
"type": "number"
|
||||
},
|
||||
"order": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "string"
|
||||
}
|
||||
},
|
||||
"probabilities": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"type": "number",
|
||||
"format": "float64"
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOneRequest": {
|
||||
"type": "object"
|
||||
},
|
||||
"schema.SystemOneResponse": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"answers": {
|
||||
"type": "object",
|
||||
"additionalProperties": {
|
||||
"$ref": "#/definitions/schema.SystemOneAnswer"
|
||||
}
|
||||
},
|
||||
"latency_ms": {
|
||||
"type": "number"
|
||||
},
|
||||
"model": {
|
||||
"type": "string"
|
||||
},
|
||||
"usage": {
|
||||
"$ref": "#/definitions/schema.SystemOneUsage"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.SystemOneUsage": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"input_tokens": {
|
||||
"type": "integer"
|
||||
},
|
||||
"output_tokens": {
|
||||
"type": "integer"
|
||||
}
|
||||
}
|
||||
},
|
||||
"schema.TTSRequest": {
|
||||
"description": "TTS request body",
|
||||
"type": "object",
|
||||
|
||||
+196
-1
@@ -2,13 +2,19 @@ basePath: /
|
||||
definitions:
|
||||
config.Gallery:
|
||||
properties:
|
||||
artifact_verification:
|
||||
allOf:
|
||||
- $ref: '#/definitions/config.GalleryVerification'
|
||||
description: |-
|
||||
ArtifactVerification overrides Verification only for the gallery OCI artifact.
|
||||
Backend images keep their separate Verification policy.
|
||||
mirrors:
|
||||
description: |-
|
||||
Mirrors are tried in order when URL cannot be fetched. They are a
|
||||
fallback for availability, not a load-balancing pool: the primary is
|
||||
always preferred, and a mirror is only consulted after the one before
|
||||
it fails. Any URI the gallery loader understands works here
|
||||
(https://, github:, file://).
|
||||
(https://, github:, file://, oci://).
|
||||
items:
|
||||
type: string
|
||||
type: array
|
||||
@@ -32,6 +38,11 @@ definitions:
|
||||
not_before:
|
||||
description: NotBefore is an RFC3339 timestamp. Empty disables the time check.
|
||||
type: string
|
||||
source_repository:
|
||||
description: |-
|
||||
SourceRepository is an https URL compared exactly against the
|
||||
certificate's source-repository extension. Empty skips the check.
|
||||
type: string
|
||||
type: object
|
||||
config.TTSVoice:
|
||||
properties:
|
||||
@@ -2654,12 +2665,38 @@ definitions:
|
||||
type: string
|
||||
id:
|
||||
type: string
|
||||
process:
|
||||
allOf:
|
||||
- $ref: '#/definitions/schema.SysInfoProcess'
|
||||
description: |-
|
||||
Process is the backend process serving the model on this host. Absent
|
||||
when the model has no local process (a distributed worker holds it) or
|
||||
the process could not be read.
|
||||
size_vram:
|
||||
description: |-
|
||||
SizeVRAM is DRM-accounted resident device memory in bytes. Nil means
|
||||
the backend process tree has no complete supported reading.
|
||||
type: integer
|
||||
type: object
|
||||
schema.SysInfoProcess:
|
||||
properties:
|
||||
cpu_percent:
|
||||
description: |-
|
||||
CPUPercent is the share of the whole host's CPU used since the previous
|
||||
reading, 0-100. Absent on the first reading of a process.
|
||||
type: number
|
||||
memory_percent:
|
||||
type: number
|
||||
pid:
|
||||
type: integer
|
||||
rss_bytes:
|
||||
description: |-
|
||||
RSSBytes is resident host memory. Weights offloaded to a GPU are not
|
||||
in it.
|
||||
type: integer
|
||||
started_at:
|
||||
type: string
|
||||
type: object
|
||||
schema.SystemInformationResponse:
|
||||
properties:
|
||||
backends:
|
||||
@@ -2673,6 +2710,106 @@ definitions:
|
||||
$ref: '#/definitions/schema.SysInfoModel'
|
||||
type: array
|
||||
type: object
|
||||
schema.SystemOneAnswer:
|
||||
properties:
|
||||
choice:
|
||||
type: string
|
||||
confidence:
|
||||
type: number
|
||||
entities:
|
||||
items:
|
||||
$ref: '#/definitions/schema.SystemOneEntity'
|
||||
type: array
|
||||
legend:
|
||||
additionalProperties:
|
||||
type: string
|
||||
type: object
|
||||
noul:
|
||||
type: number
|
||||
probabilities:
|
||||
additionalProperties:
|
||||
format: float64
|
||||
type: number
|
||||
type: object
|
||||
score:
|
||||
type: number
|
||||
type:
|
||||
type: string
|
||||
type: object
|
||||
schema.SystemOneEntity:
|
||||
properties:
|
||||
confidence:
|
||||
type: number
|
||||
end:
|
||||
type: integer
|
||||
start:
|
||||
type: integer
|
||||
text:
|
||||
type: string
|
||||
type: object
|
||||
schema.SystemOnePermuteRequest:
|
||||
properties:
|
||||
n_perm:
|
||||
type: integer
|
||||
question:
|
||||
type: string
|
||||
request:
|
||||
$ref: '#/definitions/schema.SystemOneRequest'
|
||||
seed:
|
||||
type: integer
|
||||
type: object
|
||||
schema.SystemOnePermuteResponse:
|
||||
properties:
|
||||
argmax_stable:
|
||||
type: boolean
|
||||
runs:
|
||||
items:
|
||||
$ref: '#/definitions/schema.SystemOnePermuteRun'
|
||||
type: array
|
||||
spread:
|
||||
additionalProperties:
|
||||
format: float64
|
||||
type: number
|
||||
type: object
|
||||
type: object
|
||||
schema.SystemOnePermuteRun:
|
||||
properties:
|
||||
choice:
|
||||
type: string
|
||||
latency_ms:
|
||||
type: number
|
||||
order:
|
||||
items:
|
||||
type: string
|
||||
type: array
|
||||
probabilities:
|
||||
additionalProperties:
|
||||
format: float64
|
||||
type: number
|
||||
type: object
|
||||
type: object
|
||||
schema.SystemOneRequest:
|
||||
type: object
|
||||
schema.SystemOneResponse:
|
||||
properties:
|
||||
answers:
|
||||
additionalProperties:
|
||||
$ref: '#/definitions/schema.SystemOneAnswer'
|
||||
type: object
|
||||
latency_ms:
|
||||
type: number
|
||||
model:
|
||||
type: string
|
||||
usage:
|
||||
$ref: '#/definitions/schema.SystemOneUsage'
|
||||
type: object
|
||||
schema.SystemOneUsage:
|
||||
properties:
|
||||
input_tokens:
|
||||
type: integer
|
||||
output_tokens:
|
||||
type: integer
|
||||
type: object
|
||||
schema.TTSRequest:
|
||||
description: TTS request body
|
||||
properties:
|
||||
@@ -5672,6 +5809,64 @@ paths:
|
||||
summary: Generates audio from the input text.
|
||||
tags:
|
||||
- audio
|
||||
/v1/systemone:
|
||||
post:
|
||||
description: 'Runs zero-shot NER over the supplied state and answers each question.
|
||||
Question types: noul (binary entity presence), choice (pick one option), score
|
||||
(pick one level).'
|
||||
parameters:
|
||||
- description: state + questions
|
||||
in: body
|
||||
name: request
|
||||
required: true
|
||||
schema:
|
||||
$ref: '#/definitions/schema.SystemOneRequest'
|
||||
responses:
|
||||
"200":
|
||||
description: OK
|
||||
schema:
|
||||
$ref: '#/definitions/schema.SystemOneResponse'
|
||||
summary: Answer structured-extraction questions over state text.
|
||||
tags:
|
||||
- systemone
|
||||
/v1/systemone/permute:
|
||||
post:
|
||||
description: Re-runs one choice question under n_perm option orders. Reports
|
||||
per-order probabilities, argmax stability, and spread.
|
||||
parameters:
|
||||
- description: request + question + n_perm + seed
|
||||
in: body
|
||||
name: request
|
||||
required: true
|
||||
schema:
|
||||
$ref: '#/definitions/schema.SystemOnePermuteRequest'
|
||||
responses:
|
||||
"200":
|
||||
description: OK
|
||||
schema:
|
||||
$ref: '#/definitions/schema.SystemOnePermuteResponse'
|
||||
summary: Re-run a choice question under multiple option orders.
|
||||
tags:
|
||||
- systemone
|
||||
/v1/systemone/separate:
|
||||
post:
|
||||
description: Runs N independent NER passes, one per question, against the same
|
||||
state. Response shape matches /v1/systemone.
|
||||
parameters:
|
||||
- description: state + questions
|
||||
in: body
|
||||
name: request
|
||||
required: true
|
||||
schema:
|
||||
$ref: '#/definitions/schema.SystemOneRequest'
|
||||
responses:
|
||||
"200":
|
||||
description: OK
|
||||
schema:
|
||||
$ref: '#/definitions/schema.SystemOneResponse'
|
||||
summary: Answer each question in a separate NER pass.
|
||||
tags:
|
||||
- systemone
|
||||
/v1/text-to-speech/{voice-id}:
|
||||
post:
|
||||
parameters:
|
||||
|
||||
Reference in new issue
Block a user