Compare commits

..

1 Commits

Author SHA1 Message Date
mudler
da701f9948 chore: bump inference defaults from unsloth 2026-08-06 06:49:54 +00:00
42 changed files with 79 additions and 1995 deletions

View File

@@ -9,7 +9,7 @@
# recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean
# rebuild and so the bump bot can see the pin.
AUDIO_CPP_VERSION?=7efbb58def443722ea540d931dd3debee3e4d5e8
AUDIO_CPP_VERSION?=238ab6a9e321c17de8e120559f57efeedaeb1345
AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp
CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST))))

View File

@@ -1,10 +1,10 @@
# ds4 backend Makefile.
#
# Upstream pin lives below as DS4_VERSION?=b0309611041655f4e45671cfd9c9886aff161406
# Upstream pin lives below as DS4_VERSION?=6747e7718dd08f00b680d0c16231f2d59ec3747e
# (.github/bump_deps.sh) can find and update it - matches the
# llama-cpp / ik-llama-cpp / turboquant convention.
DS4_VERSION?=b0309611041655f4e45671cfd9c9886aff161406
DS4_VERSION?=6747e7718dd08f00b680d0c16231f2d59ec3747e
DS4_REPO?=https://github.com/antirez/ds4
CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST))))

View File

@@ -1,5 +1,5 @@
IK_LLAMA_VERSION?=cf1aa57e1a0fabfd015831718fc99d1aec01ada5
IK_LLAMA_VERSION?=6b55d2c7504f482e7c8ec6cbf22a19f3778c522b
LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp
CMAKE_ARGS?=

View File

@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
# CrispASR version (release tag)
CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR
CRISPASR_VERSION?=21901d3f7c23554f072964828363e49ddbc2dc68
CRISPASR_VERSION?=ec730908a418b6032f9e69ded6186d3f042a7747
SO_TARGET?=libgocrispasr.so
CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF

View File

@@ -67,16 +67,7 @@ const defaultTTSSampleRate = 24000
// resampling, so the WAV header must match it. Returns ok=false for non-piper
// models (key absent) or an unreadable file, letting the caller fall back to
// defaultTTSSampleRate.
func piperSampleRate(modelPath string) (rate int, ok bool) {
// A malformed metadata length can make gguf-parser-go panic before it can
// return an error. Keep a bad voice file from crash-looping the backend.
defer func() {
if recover() != nil {
rate = 0
ok = false
}
}()
func piperSampleRate(modelPath string) (int, bool) {
// Only scalar architecture keys are read, so skip the large array metadata
// (phoneme map) and mmap the header - same rationale as pkg/vram's reader.
f, err := gguf.ParseGGUFFile(modelPath, gguf.UseMMap(), gguf.SkipLargeMetadata())
@@ -87,7 +78,7 @@ func piperSampleRate(modelPath string) (rate int, ok bool) {
if !ok || kv.ValueType != gguf.GGUFMetadataValueTypeUint32 {
return 0, false
}
rate = int(kv.ValueUint32())
rate := int(kv.ValueUint32())
if rate <= 0 {
return 0, false
}

View File

@@ -3,7 +3,6 @@ package main
import (
"bytes"
"encoding/binary"
"math"
"os"
"path/filepath"
@@ -103,24 +102,6 @@ var _ = Describe("piper sample rate", func() {
_, ok := piperSampleRate(p)
Expect(ok).To(BeFalse())
})
It("returns ok=false instead of panicking on a malformed string length", func() {
p := filepath.Join(GinkgoT().TempDir(), "malformed.gguf")
var b bytes.Buffer
b.WriteString("GGUF")
Expect(binary.Write(&b, binary.LittleEndian, uint32(3))).To(Succeed())
Expect(binary.Write(&b, binary.LittleEndian, uint64(0))).To(Succeed())
Expect(binary.Write(&b, binary.LittleEndian, uint64(1))).To(Succeed())
key := "general.name"
Expect(binary.Write(&b, binary.LittleEndian, uint64(len(key)))).To(Succeed())
b.WriteString(key)
Expect(binary.Write(&b, binary.LittleEndian, ggufTypeString)).To(Succeed())
Expect(binary.Write(&b, binary.LittleEndian, uint64(math.MaxInt64))).To(Succeed())
Expect(os.WriteFile(p, b.Bytes(), 0o644)).To(Succeed())
_, ok := piperSampleRate(p)
Expect(ok).To(BeFalse())
})
})
// End-to-end through the built .so. Gated on CRISPASR_PIPER_MODEL_PATH (a

View File

@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
# stablediffusion.cpp (ggml)
STABLEDIFFUSION_GGML_REPO?=https://github.com/leejet/stable-diffusion.cpp
STABLEDIFFUSION_GGML_VERSION?=c6beeef35526c6dc94b74a7fb69f9d2e6a2a7a12
STABLEDIFFUSION_GGML_VERSION?=ea7f0c87cfe4c673263b4c201c596c7f1cbe2528
CMAKE_ARGS+=-DGGML_MAX_NAME=128

View File

@@ -96,12 +96,6 @@ endif
UNAME_S := $(shell uname -s)
ifeq ($(UNAME_S),Darwin)
LIB=libvllm.dylib
# Apple Clang diagnoses a pair of constant-folded array bounds in the Metal
# build as a GNU extension. Disable that diagnostic for both Objective-C and
# C++ because vllm.cpp appends target-local -Werror after these global flags.
CMAKE_ARGS+=-DCMAKE_CXX_FLAGS=-Wno-gnu-folding-constant
CMAKE_ARGS+=-DCMAKE_OBJC_FLAGS=-Wno-gnu-folding-constant
CMAKE_ARGS+=-DCMAKE_OBJCXX_FLAGS=-Wno-gnu-folding-constant
else
LIB=libvllm.so
endif
@@ -139,26 +133,7 @@ MLX_STAMP=
MLX_CMAKE_ARGS=
endif
# govllmcpp.go mirrors vllm.h by hand, and the only guard against the two
# drifting apart is the vllm_abi_version check inside registerLib - which fires
# at runtime, on the user's machine, taking down every model load (issue
# #11379). Compare the two here instead, so moving VLLM_CPP_VERSION past the
# mirrors turns the build red while the header is still around to diff.
abi-check: sources/vllm.cpp
@engine=$$(sed -n 's/^#define VLLM_ABI_VERSION \([0-9][0-9]*\).*/\1/p' sources/vllm.cpp/include/vllm.h); \
backend=$$(sed -n 's/^const abiVersion = \([0-9][0-9]*\).*/\1/p' govllmcpp.go); \
if [ -z "$$engine" ] || [ -z "$$backend" ]; then \
echo "vllm-cpp: cannot read the ABI version (engine='$$engine' backend='$$backend')" >&2; exit 1; \
fi; \
if [ "$$engine" != "$$backend" ]; then \
echo "vllm-cpp: ABI mismatch: vllm.cpp $(VLLM_CPP_VERSION) is v$$engine, govllmcpp.go mirrors v$$backend." >&2; \
echo " Update the struct mirrors and abiVersion in govllmcpp.go (and the offsets in vllmcpp_test.go) to v$$engine." >&2; \
exit 1; \
fi; \
echo "vllm-cpp: ABI v$$engine matches the pinned engine"
$(LIB): sources/vllm.cpp $(MLX_STAMP)
$(MAKE) abi-check
mkdir -p build && \
cd build && \
cmake ../sources/vllm.cpp $(CMAKE_ARGS) $(MLX_CMAKE_ARGS) && \
@@ -179,8 +154,6 @@ clean: purge
purge:
rm -rf build
.PHONY: abi-check
.NOTPARALLEL:
# The unit specs are pure Go (struct mirrors, option mapping, load

View File

@@ -6,7 +6,7 @@ safetensors + GGUF loading, CUDA / CPU / Metal / Vulkan) with no Python at
inference time.
The backend dlopens the engine's stable C ABI (`libvllm`, `include/vllm.h`,
ABI v10) through purego:
ABI v2) through purego:
- `Load` -> `vllm_engine_load`: accepts a `.gguf` file or a HF-style model
directory (`config.json` + safetensors). `context_size` maps to
@@ -29,12 +29,6 @@ ABI v10) through purego:
LocalAI's Go-side grammar-constrained tool calling; JSON-schema / regex /
choice constraints are also exposed by the ABI.
The struct mirrors in `govllmcpp.go` are hand-written against one ABI version,
and the engine refuses to load against any other. Moving `VLLM_CPP_VERSION` in
the Makefile therefore means updating `abiVersion` plus the mirrors (and their
offsets in `vllmcpp_test.go`) in the same change; `make abi-check` compares the
pinned header against the bindings and the library build runs it first.
Model config example:
```yaml

View File

@@ -109,16 +109,6 @@ func (v *VllmCpp) Load(opts *pb.ModelOptions) error {
v.opts = parseOptions(opts)
// A DFlash draft is a second checkpoint the engine opens by path, and the
// engine never downloads one. Resolve it against LocalAI's models directory
// now so a repo-id spelling works, and so a missing draft fails here with an
// actionable message rather than as an HF-cache miss inside the load.
resolvedSpec, err := resolveDraftModelPath(v.opts.speculativeConfig, opts.ModelPath)
if err != nil {
return err
}
v.opts.speculativeConfig = resolvedSpec
mp := defaultModelParams()
if v.opts.blockSize > 0 {
mp.BlockSize = v.opts.blockSize
@@ -126,62 +116,34 @@ func (v *VllmCpp) Load(opts *pb.ModelOptions) error {
if v.opts.numBlocks > 0 {
mp.NumBlocks = v.opts.numBlocks
}
// Sequence-length precedence, narrowest source last: context_size is the
// generic LocalAI knob every backend honours, max_model_len is the
// vLLM-specific one, and engine_args.max_model_len is the explicit
// vllm-cpp override.
if opts.ContextSize > 0 {
mp.MaxModelLen = opts.ContextSize
}
if opts.MaxModelLen > 0 {
mp.MaxModelLen = opts.MaxModelLen
}
if v.opts.maxModelLen > 0 {
mp.MaxModelLen = v.opts.maxModelLen
}
if v.opts.maxNumSeqs > 0 {
mp.MaxNumSeqs = v.opts.maxNumSeqs
}
if v.opts.maxNumBatchedTokens > 0 {
mp.MaxNumBatchedTokens = v.opts.maxNumBatchedTokens
}
mp.EnablePrefixCaching = v.opts.enablePrefixCaching
mp.EnableJumpForward = v.opts.enableJumpForward
// Every string below is borrowed by C for the duration of the load call
// only (the library copies what it keeps), so the backing slices just have
// to outlive vllmEngineLoad - hence the single KeepAlive after it.
modelC := cString(model)
mp.ModelPath = uintptr(unsafe.Pointer(&modelC[0])) // #nosec G103 -- borrowed by C for the load call only
keep := [][]byte{modelC}
setStr := func(dst *uintptr, s string) {
if s == "" {
return
}
b := cString(s)
keep = append(keep, b)
*dst = uintptr(unsafe.Pointer(&b[0])) // #nosec G103 -- borrowed by C for the load call only
var toolParserC, reasoningParserC []byte
if v.opts.toolParser != "" {
toolParserC = cString(v.opts.toolParser)
mp.ToolParser = uintptr(unsafe.Pointer(&toolParserC[0])) // #nosec G103 -- borrowed by C for the load call only
}
if v.opts.reasoningParser != "" {
reasoningParserC = cString(v.opts.reasoningParser)
mp.ReasoningParser = uintptr(unsafe.Pointer(&reasoningParserC[0])) // #nosec G103 -- borrowed by C for the load call only
}
setStr(&mp.ToolParser, v.opts.toolParser)
setStr(&mp.ReasoningParser, v.opts.reasoningParser)
setStr(&mp.SpeculativeConfig, v.opts.speculativeConfig)
setStr(&mp.KVTransferConfig, v.opts.kvTransferConfig)
setStr(&mp.SchedulingPolicy, v.opts.schedulingPolicy)
setStr(&mp.TokenizerConfigPath, v.opts.tokenizerConfigPath)
xlog.Info("[vllm-cpp] Load", "model", model, "engine", vllmVersion(),
"blockSize", mp.BlockSize, "numBlocks", mp.NumBlocks,
"maxModelLen", mp.MaxModelLen, "maxNumSeqs", mp.MaxNumSeqs,
"maxNumBatchedTokens", mp.MaxNumBatchedTokens,
"prefixCaching", triStateName(mp.EnablePrefixCaching),
"jumpForward", triStateName(mp.EnableJumpForward),
"schedulingPolicy", v.opts.schedulingPolicy,
"speculativeConfig", v.opts.speculativeConfig,
"kvTransferConfig", v.opts.kvTransferConfig)
"maxModelLen", mp.MaxModelLen, "maxNumSeqs", mp.MaxNumSeqs)
var engine uintptr
rc := vllmEngineLoad(unsafe.Pointer(&mp), unsafe.Pointer(&engine)) // #nosec G103 -- POD out-params
runtime.KeepAlive(keep)
runtime.KeepAlive(modelC)
runtime.KeepAlive(toolParserC)
runtime.KeepAlive(reasoningParserC)
if rc != vllmOK {
return fmt.Errorf("vllm-cpp: engine load failed: %s", vllmLastError())
}

View File

@@ -1,6 +1,6 @@
package main
// purego bindings for the vllm.cpp stable C ABI (include/vllm.h, ABI v10).
// purego bindings for the vllm.cpp stable C ABI (include/vllm.h, ABI v2).
//
// The structs below are hand-mirrored PODs of the C declarations, with
// explicit padding so the Go layout matches the C layout on linux/darwin
@@ -17,65 +17,29 @@ import (
"github.com/ebitengine/purego"
)
// abiVersion is the VLLM_ABI_VERSION this file mirrors (vllm.h). It must track
// the header of the VLLM_CPP_VERSION pinned in the Makefile: the build checks
// the two against each other, because a mismatch is only caught at runtime by
// registerLib, where it takes the backend down on every load (issue #11379).
const abiVersion = 10
// The ABI's tri-state toggles (enable_prefix_caching ABI v7,
// enable_jump_forward ABI v10) share one encoding: 0 is NOT "off", it is
// "defer" - to the model capability for prefix caching, to the environment for
// jump forward. Only 2 is an explicit off.
const (
triStateDefer int32 = 0
triStateOn int32 = 1
triStateOff int32 = 2
)
// triStateName renders a tri-state for the load log line, where "0" would
// otherwise read as "off" rather than "whatever the default resolves to".
func triStateName(state int32) string {
switch state {
case triStateOn:
return "on"
case triStateOff:
return "off"
default:
return "model-default"
}
}
// abiVersion is the VLLM_ABI_VERSION this file mirrors (vllm.h).
const abiVersion = 5
// vllm_status (vllm.h).
const (
vllmOK = 0
)
// cModelParams mirrors vllm_model_params. The int32 fields sit in pairs so the
// interior needs no padding on LP64, but the struct is 8-aligned (it holds
// pointers) and ends on a lone int32, so the trailing pad is explicit. Offsets
// and total size are asserted in vllmcpp_test.go.
// cModelParams mirrors vllm_model_params.
type cModelParams struct {
ModelPath uintptr // const char*
TokenizerConfigPath uintptr // const char*; NULL = <model_dir>/... (ABI v9)
TokenizerConfigPath uintptr // const char*
BlockSize int32
NumBlocks int32
MaxModelLen int32
MaxNumSeqs int32
ToolParser uintptr // const char*; NULL = auto-detect (ABI v4)
ReasoningParser uintptr // const char*; NULL = auto-detect (ABI v5)
SpeculativeConfig uintptr // const char* JSON; NULL = no speculation (ABI v6)
EnablePrefixCaching int32 // tri-state 0/1/2 (ABI v7)
MaxNumBatchedTokens int32 // <= 0 = per-arch default (ABI v9)
SchedulingPolicy uintptr // const char*; NULL = "fcfs" (ABI v9)
KVTransferConfig uintptr // const char* JSON; NULL = no connector (ABI v9)
EnableJumpForward int32 // tri-state 0/1/2 (ABI v10)
_ [4]byte // trailing pad to the struct's 8-byte alignment
}
// cSamplingParams mirrors vllm_sampling_params (structured fields included).
// Padding matches the C compiler's: the uint64 seed is 8-aligned, and each
// pointer following an int32 is 8-aligned.
// cSamplingParams mirrors vllm_sampling_params (ABI v2, structured fields
// included). Padding matches the C compiler's: the uint64 seed is 8-aligned,
// and each pointer following an int32 is 8-aligned.
type cSamplingParams struct {
Temperature float32
TopP float32
@@ -101,12 +65,6 @@ type cSamplingParams struct {
StructuredGrammar uintptr // const char*
StructuredJSONObject int32
_ [4]byte
// ABI v8 tail. LocalAI installs no custom logits processor, but the fields
// MUST be mirrored: the C side reads them off the pointer we hand it, so a
// Go struct that stopped at StructuredJSONObject would have the engine read
// 16 bytes past our allocation and call whatever garbage sat there.
LogitsProcessor uintptr // vllm_logits_processor; NULL = none
LogitsProcessorUserData uintptr // void*
}
// cCompletion mirrors vllm_completion.

View File

@@ -1,80 +1,30 @@
package main
// Load-time engine configuration, from two config surfaces:
//
// - `engine_args:` (ModelOptions.EngineArgs, a JSON object) is the canonical
// one. Keys are spelled exactly as vLLM's own CLI flags, so a config written
// against vLLM works verbatim here - `speculative_config` and
// `kv_transfer_config` in particular take the same JSON documents vLLM's
// --speculative-config / --kv-transfer-config accept, and are handed to the
// engine unparsed.
// - `options:` (the free-form "key:value" list) is the older surface this
// backend shipped with. It is still honoured so existing configs keep
// working; engine_args wins on any key set in both.
//
// Anything unrecognised is ignored rather than fatal: the engine validates the
// documents it is given and reports a precise error at load, and a config that
// also carries knobs for a different backend must not fail the load here.
// Engine-sizing knobs carried through the model config's free-form
// `options:` list ("key:value" entries), mirroring how the other in-house
// backends pass engine-specific settings that have no proto field.
import (
"encoding/json"
"fmt"
"os"
"path"
"path/filepath"
"strconv"
"strings"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
"github.com/mudler/xlog"
)
type loadOptions struct {
blockSize int32 // KV block size (tokens/block); engine default 32.
numBlocks int32 // KV blocks to allocate; engine default 256.
maxNumSeqs int32 // max concurrent sequences; engine default 8.
// Max sequence length. Also settable through the model config's
// context_size / max_model_len; see Load for the precedence.
maxModelLen int32
// Per-step chunked-prefill token budget (ABI v9). 0 = the engine's
// bounded per-arch default.
maxNumBatchedTokens int32
// Automatic prefix caching tri-state (ABI v7): 0 = the model-capability
// default, 1 = force on, 2 = force off.
enablePrefixCaching int32
// Jump-forward decoding tri-state (ABI v10), SGLang's grammar-speed subset:
// 0 = defer to the environment (VT_ENABLE_JUMP_FORWARD, default off),
// 1 = force on, 2 = force off.
enableJumpForward int32
// Scheduler admission policy (ABI v9): "" = fcfs, else fcfs|priority|lpm.
schedulingPolicy string
// Engine-side parser selection (ABI v4/v5). Empty = the engine
// auto-detects from the chat template; "none" disables the reasoning
// split; unknown names fail the first chat call.
toolParser string
reasoningParser string
// Speculative decoding (ABI v6), as vLLM's --speculative-config JSON:
// {"method":"mtp"|"dflash"|"ngram", ...}. Empty = no speculation.
speculativeConfig string
// External KV connector / LMCache (ABI v9), as vLLM's --kv-transfer-config
// JSON. Empty = no connector.
kvTransferConfig string
// Override for the tokenizer_config.json the chat template is read from
// (ABI v9). Empty = <model_dir>/tokenizer_config.json.
tokenizerConfigPath string
}
func parseOptions(opts *pb.ModelOptions) loadOptions {
lo := loadOptions{}
applyOptionsList(&lo, opts.GetOptions())
applyEngineArgs(&lo, opts.GetEngineArgs())
return lo
}
// applyOptionsList reads the legacy free-form "key:value" list. strings.Cut
// splits on the FIRST colon only, so a JSON object value survives intact.
func applyOptionsList(lo *loadOptions, options []string) {
for _, o := range options {
for _, o := range opts.GetOptions() {
k, v, found := strings.Cut(o, ":")
if !found {
continue
@@ -86,211 +36,13 @@ func applyOptionsList(lo *loadOptions, options []string) {
lo.numBlocks = parseInt32(v, lo.numBlocks)
case "max_num_seqs":
lo.maxNumSeqs = parseInt32(v, lo.maxNumSeqs)
case "max_num_batched_tokens":
lo.maxNumBatchedTokens = parseInt32(v, lo.maxNumBatchedTokens)
case "max_model_len":
lo.maxModelLen = parseInt32(v, lo.maxModelLen)
case "scheduling_policy", "schedule_policy":
lo.schedulingPolicy = strings.TrimSpace(v)
case "tool_parser", "tool_call_parser":
case "tool_parser":
lo.toolParser = strings.TrimSpace(v)
case "reasoning_parser":
lo.reasoningParser = strings.TrimSpace(v)
case "speculative_config":
lo.speculativeConfig = strings.TrimSpace(v)
case "kv_transfer_config":
lo.kvTransferConfig = strings.TrimSpace(v)
case "tokenizer_config", "tokenizer_config_path":
lo.tokenizerConfigPath = strings.TrimSpace(v)
case "enable_prefix_caching", "enable_radix_attention":
if b, err := strconv.ParseBool(strings.TrimSpace(v)); err == nil {
lo.enablePrefixCaching = boolTriState(b)
}
case "enable_jump_forward":
if b, err := strconv.ParseBool(strings.TrimSpace(v)); err == nil {
lo.enableJumpForward = boolTriState(b)
}
}
}
}
// applyEngineArgs overlays the `engine_args:` JSON object. A document that does
// not parse is logged and skipped: engine_args is shared with the other engines
// (the vLLM and SGLang backends read the same field), so a stray key must not
// take the model down.
func applyEngineArgs(lo *loadOptions, engineArgs string) {
if strings.TrimSpace(engineArgs) == "" {
return
}
var args map[string]any
if err := json.Unmarshal([]byte(engineArgs), &args); err != nil {
xlog.Warn("[vllm-cpp] ignoring unparseable engine_args", "error", err)
return
}
for k, v := range args {
switch k {
case "block_size":
lo.blockSize = jsonInt32(v, lo.blockSize)
case "num_blocks":
lo.numBlocks = jsonInt32(v, lo.numBlocks)
case "max_num_seqs":
lo.maxNumSeqs = jsonInt32(v, lo.maxNumSeqs)
case "max_num_batched_tokens":
lo.maxNumBatchedTokens = jsonInt32(v, lo.maxNumBatchedTokens)
case "max_model_len":
lo.maxModelLen = jsonInt32(v, lo.maxModelLen)
case "scheduling_policy", "schedule_policy":
lo.schedulingPolicy = jsonString(v, lo.schedulingPolicy)
case "tool_parser", "tool_call_parser":
lo.toolParser = jsonString(v, lo.toolParser)
case "reasoning_parser":
lo.reasoningParser = jsonString(v, lo.reasoningParser)
case "tokenizer_config", "tokenizer_config_path":
lo.tokenizerConfigPath = jsonString(v, lo.tokenizerConfigPath)
case "speculative_config":
lo.speculativeConfig = jsonDocument(v, lo.speculativeConfig, k)
case "kv_transfer_config":
lo.kvTransferConfig = jsonDocument(v, lo.kvTransferConfig, k)
case "enable_prefix_caching", "enable_radix_attention":
if b, ok := v.(bool); ok {
lo.enablePrefixCaching = boolTriState(b)
}
case "enable_jump_forward":
if b, ok := v.(bool); ok {
lo.enableJumpForward = boolTriState(b)
}
default:
xlog.Debug("[vllm-cpp] ignoring unknown engine_args key", "key", k)
}
}
}
// boolTriState maps a YAML/JSON boolean onto the ABI's tri-state encoding. An
// explicit `false` must reach the engine as force-OFF (2), NOT as the 0 that
// means "defer". The difference is real in both directions: prefix caching
// defaults ON for dense archs and OFF for hybrid ones, and jump forward defers
// to VT_ENABLE_JUMP_FORWARD.
func boolTriState(on bool) int32 {
if on {
return triStateOn
}
return triStateOff
}
// jsonDocument normalises an object-valued engine_args entry to a JSON string
// for the C ABI. YAML nesting arrives as a map (the natural spelling); a
// pre-encoded JSON string is accepted too, since a config round-tripped through
// a flat store may carry it that way.
func jsonDocument(v any, fallback string, key string) string {
switch t := v.(type) {
case string:
if strings.TrimSpace(t) == "" {
return fallback
}
return t
default:
buf, err := json.Marshal(t)
if err != nil {
xlog.Warn("[vllm-cpp] ignoring unencodable engine_args value", "key", key, "error", err)
return fallback
}
return string(buf)
}
}
func jsonString(v any, fallback string) string {
s, ok := v.(string)
if !ok {
return fallback
}
return strings.TrimSpace(s)
}
// jsonInt32 accepts the float64 a JSON number decodes to, plus the string
// spelling a YAML config may produce. Non-positive values keep the fallback:
// every knob this covers uses "<= 0 means the engine default".
func jsonInt32(v any, fallback int32) int32 {
switch t := v.(type) {
case float64:
if t <= 0 || t > 1<<31-1 {
return fallback
}
return int32(t)
case string:
return parseInt32(t, fallback)
default:
return fallback
}
}
// resolveDraftModelPath rewrites a DFlash draft reference into an absolute path
// the engine can actually open.
//
// The engine resolves `speculative_config.model` against a directory containing
// config.json, or against ~/.cache/huggingface/hub/models--<org>--<repo>/
// snapshots/* - and it NEVER downloads. LocalAI keeps models in its own
// directory, so a bare HF repo id (the spelling the vLLM docs teach) misses the
// HF cache and dies deep in the load with "draft checkpoint not found", which
// reads like a broken checkpoint rather than a missing download.
//
// So: try the reference as given, then the last path segment under the models
// dir (`z-lab/Qwen3.6-27B-DFlash` -> `<models>/Qwen3.6-27B-DFlash`, which is
// what LocalAI's own downloader produces), then the whole reference under the
// models dir. If none exist, fail HERE with a message naming both what was
// asked for and where we looked.
//
// mtp and ngram carry no separate draft checkpoint, so they pass through. A
// document that does not parse also passes through: the engine owns config
// validation and produces the better error.
func resolveDraftModelPath(speculativeConfig, modelsDir string) (string, error) {
if strings.TrimSpace(speculativeConfig) == "" {
return speculativeConfig, nil
}
var spec map[string]any
if err := json.Unmarshal([]byte(speculativeConfig), &spec); err != nil {
return speculativeConfig, nil
}
if method, _ := spec["method"].(string); !strings.EqualFold(method, "dflash") {
return speculativeConfig, nil
}
ref, _ := spec["model"].(string)
ref = strings.TrimSpace(ref)
if ref == "" {
return "", fmt.Errorf(
"vllm-cpp: speculative_config method %q requires a \"model\" key naming the draft checkpoint", "dflash")
}
candidates := []string{ref}
if modelsDir != "" {
if base := path.Base(filepath.ToSlash(ref)); base != "" && base != "." && base != "/" {
candidates = append(candidates, filepath.Join(modelsDir, base))
}
candidates = append(candidates, filepath.Join(modelsDir, filepath.FromSlash(ref)))
}
for _, c := range candidates {
if _, err := os.Stat(filepath.Join(c, "config.json")); err != nil {
continue
}
abs, err := filepath.Abs(c)
if err != nil {
abs = c
}
spec["model"] = abs
out, err := json.Marshal(spec)
if err != nil {
return "", fmt.Errorf("vllm-cpp: re-encoding speculative_config: %w", err)
}
xlog.Info("[vllm-cpp] resolved DFlash draft checkpoint", "reference", ref, "path", abs)
return string(out), nil
}
return "", fmt.Errorf(
"vllm-cpp: DFlash draft checkpoint %q not found (looked in: %s). "+
"The engine does not download drafts - install the draft model into LocalAI first, "+
"or set speculative_config.model to an absolute path to a directory containing config.json",
ref, strings.Join(candidates, ", "))
return lo
}
func parseInt32(s string, fallback int32) int32 {

View File

@@ -16,17 +16,10 @@ func TestVllmCpp(t *testing.T) {
RunSpecs(t, "vllm-cpp suite")
}
// The Go POD mirrors must match the C struct layout of vllm.h (ABI v10)
// The Go POD mirrors must match the C struct layout of vllm.h (ABI v2)
// byte-for-byte: these offsets are the C offsets on LP64 (linux/darwin
// amd64+arm64). A failure here means govllmcpp.go drifted from vllm.h.
var _ = Describe("C ABI struct mirrors", func() {
It("declares the ABI version the pinned engine reports", func() {
// VLLM_ABI_VERSION in the vllm.h of VLLM_CPP_VERSION (Makefile).
// Moving the pin past this without growing the mirrors below ships a
// backend that refuses every load at startup (issue #11379).
Expect(abiVersion).To(Equal(10))
})
It("cModelParams matches vllm_model_params", func() {
var p cModelParams
Expect(unsafe.Offsetof(p.ModelPath)).To(Equal(uintptr(0)))
@@ -37,18 +30,10 @@ var _ = Describe("C ABI struct mirrors", func() {
Expect(unsafe.Offsetof(p.MaxNumSeqs)).To(Equal(uintptr(28)))
Expect(unsafe.Offsetof(p.ToolParser)).To(Equal(uintptr(32)))
Expect(unsafe.Offsetof(p.ReasoningParser)).To(Equal(uintptr(40)))
Expect(unsafe.Offsetof(p.SpeculativeConfig)).To(Equal(uintptr(48)))
Expect(unsafe.Offsetof(p.EnablePrefixCaching)).To(Equal(uintptr(56)))
Expect(unsafe.Offsetof(p.MaxNumBatchedTokens)).To(Equal(uintptr(60)))
Expect(unsafe.Offsetof(p.SchedulingPolicy)).To(Equal(uintptr(64)))
Expect(unsafe.Offsetof(p.KVTransferConfig)).To(Equal(uintptr(72)))
Expect(unsafe.Offsetof(p.EnableJumpForward)).To(Equal(uintptr(80)))
// 88, not 84: the struct is 8-aligned (it holds pointers), so the
// trailing int32 is padded out. Go pads identically.
Expect(unsafe.Sizeof(p)).To(Equal(uintptr(88)))
Expect(unsafe.Sizeof(p)).To(Equal(uintptr(48)))
})
It("cSamplingParams matches vllm_sampling_params (ABI v8)", func() {
It("cSamplingParams matches vllm_sampling_params (ABI v2)", func() {
var p cSamplingParams
Expect(unsafe.Offsetof(p.Temperature)).To(Equal(uintptr(0)))
Expect(unsafe.Offsetof(p.TopP)).To(Equal(uintptr(4)))
@@ -70,9 +55,7 @@ var _ = Describe("C ABI struct mirrors", func() {
Expect(unsafe.Offsetof(p.NStructuredChoice)).To(Equal(uintptr(96)))
Expect(unsafe.Offsetof(p.StructuredGrammar)).To(Equal(uintptr(104)))
Expect(unsafe.Offsetof(p.StructuredJSONObject)).To(Equal(uintptr(112)))
Expect(unsafe.Offsetof(p.LogitsProcessor)).To(Equal(uintptr(120)))
Expect(unsafe.Offsetof(p.LogitsProcessorUserData)).To(Equal(uintptr(128)))
Expect(unsafe.Sizeof(p)).To(Equal(uintptr(136)))
Expect(unsafe.Sizeof(p)).To(Equal(uintptr(120)))
})
It("cCompletion matches vllm_completion", func() {
@@ -85,23 +68,6 @@ var _ = Describe("C ABI struct mirrors", func() {
})
})
// Pin/mirror skew is the failure mode this backend is most exposed to: the Go
// PODs above are hand-written against one VLLM_ABI_VERSION, and the Makefile
// pins the vllm.cpp commit that produces it. This spec catches drift without
// needing model weights - set VLLM_CPP_LIBRARY to a built libvllm and it binds
// every symbol and compares the library's reported ABI against the mirrors'.
var _ = Describe("real library ABI handshake", func() {
It("binds every symbol and reports the ABI the mirrors were written against", func() {
lib := os.Getenv("VLLM_CPP_LIBRARY")
if lib == "" {
Skip("VLLM_CPP_LIBRARY not set; skipping the real-library handshake")
}
Expect(registerLib(lib)).To(Succeed())
Expect(vllmABIVersion()).To(Equal(int32(abiVersion)))
Expect(vllmVersion()).NotTo(BeEmpty())
})
})
var _ = Describe("parseOptions", func() {
It("extracts the engine sizing knobs", func() {
lo := parseOptions(&pb.ModelOptions{Options: []string{
@@ -117,129 +83,6 @@ var _ = Describe("parseOptions", func() {
}})
Expect(lo).To(Equal(loadOptions{}))
})
It("carries a speculative_config JSON value through the legacy options list", func() {
// strings.Cut splits on the FIRST colon only, so a JSON object value
// survives the "key:value" spelling intact.
lo := parseOptions(&pb.ModelOptions{Options: []string{
`speculative_config:{"method":"mtp","num_speculative_tokens":1}`,
}})
Expect(lo.speculativeConfig).To(Equal(`{"method":"mtp","num_speculative_tokens":1}`))
})
})
var _ = Describe("engine_args", func() {
It("maps every load knob onto the C model params", func() {
lo := parseOptions(&pb.ModelOptions{EngineArgs: `{
"block_size": 64,
"num_blocks": 1024,
"max_model_len": 16384,
"max_num_seqs": 32,
"max_num_batched_tokens": 8192,
"enable_prefix_caching": true,
"scheduling_policy": "lpm",
"tool_parser": "qwen3",
"reasoning_parser": "deepseek_r1",
"tokenizer_config": "/models/tok/tokenizer_config.json"
}`})
Expect(lo.blockSize).To(Equal(int32(64)))
Expect(lo.numBlocks).To(Equal(int32(1024)))
Expect(lo.maxModelLen).To(Equal(int32(16384)))
Expect(lo.maxNumSeqs).To(Equal(int32(32)))
Expect(lo.maxNumBatchedTokens).To(Equal(int32(8192)))
Expect(lo.enablePrefixCaching).To(Equal(int32(1)))
Expect(lo.schedulingPolicy).To(Equal("lpm"))
Expect(lo.toolParser).To(Equal("qwen3"))
Expect(lo.reasoningParser).To(Equal("deepseek_r1"))
Expect(lo.tokenizerConfigPath).To(Equal("/models/tok/tokenizer_config.json"))
})
It("re-marshals a nested speculative_config object to JSON for the engine", func() {
lo := parseOptions(&pb.ModelOptions{EngineArgs: `{
"speculative_config": {"method": "mtp", "num_speculative_tokens": 1}
}`})
Expect(lo.speculativeConfig).To(MatchJSON(`{"method":"mtp","num_speculative_tokens":1}`))
})
It("re-marshals a nested kv_transfer_config object (LMCache) to JSON", func() {
lo := parseOptions(&pb.ModelOptions{EngineArgs: `{
"kv_transfer_config": {
"kv_connector": "LMCacheConnector",
"kv_role": "kv_both",
"kv_connector_extra_config": {"host": "127.0.0.1", "port": 65432}
}
}`})
Expect(lo.kvTransferConfig).To(MatchJSON(`{
"kv_connector":"LMCacheConnector",
"kv_role":"kv_both",
"kv_connector_extra_config":{"host":"127.0.0.1","port":65432}
}`))
})
It("accepts a pre-encoded JSON string for the object-valued knobs", func() {
// A config written by hand (or round-tripped through a flat store) may
// carry the object as a string; both spellings reach the engine the same.
lo := parseOptions(&pb.ModelOptions{EngineArgs: `{
"speculative_config": "{\"method\":\"ngram\",\"num_speculative_tokens\":4}"
}`})
Expect(lo.speculativeConfig).To(MatchJSON(`{"method":"ngram","num_speculative_tokens":4}`))
})
It("maps enable_prefix_caching false onto the force-OFF tri-state", func() {
// The C ABI tri-state is 0=model default, 1=on, 2=off, so an explicit
// `false` must NOT collapse to the 0 that means "let the model decide".
lo := parseOptions(&pb.ModelOptions{EngineArgs: `{"enable_prefix_caching": false}`})
Expect(lo.enablePrefixCaching).To(Equal(int32(2)))
})
It("leaves the prefix-caching tri-state at the model default when unset", func() {
lo := parseOptions(&pb.ModelOptions{EngineArgs: `{"max_num_seqs": 4}`})
Expect(lo.enablePrefixCaching).To(Equal(int32(0)))
})
It("accepts the radix-attention alias upstream documents for prefix caching", func() {
lo := parseOptions(&pb.ModelOptions{EngineArgs: `{"enable_radix_attention": true}`})
Expect(lo.enablePrefixCaching).To(Equal(int32(1)))
})
It("maps enable_jump_forward onto its own tri-state", func() {
// ABI v10. Same tri-state shape as prefix caching, and the same trap:
// an explicit false must be force-OFF (2), not the 0 that defers to the
// environment.
on := parseOptions(&pb.ModelOptions{EngineArgs: `{"enable_jump_forward": true}`})
Expect(on.enableJumpForward).To(Equal(int32(1)))
off := parseOptions(&pb.ModelOptions{EngineArgs: `{"enable_jump_forward": false}`})
Expect(off.enableJumpForward).To(Equal(int32(2)))
unset := parseOptions(&pb.ModelOptions{EngineArgs: `{"max_num_seqs": 4}`})
Expect(unset.enableJumpForward).To(Equal(int32(0)))
})
It("reads enable_jump_forward from the legacy options list too", func() {
lo := parseOptions(&pb.ModelOptions{Options: []string{"enable_jump_forward:true"}})
Expect(lo.enableJumpForward).To(Equal(int32(1)))
})
It("lets engine_args override the legacy options list", func() {
lo := parseOptions(&pb.ModelOptions{
Options: []string{"max_num_seqs:8", "block_size:16"},
EngineArgs: `{"max_num_seqs": 64}`,
})
Expect(lo.maxNumSeqs).To(Equal(int32(64))) // engine_args wins
Expect(lo.blockSize).To(Equal(int32(16))) // untouched keys survive
})
It("ignores malformed engine_args rather than failing the load", func() {
lo := parseOptions(&pb.ModelOptions{
Options: []string{"max_num_seqs:8"},
EngineArgs: `{not json`,
})
Expect(lo.maxNumSeqs).To(Equal(int32(8)))
})
It("ignores unknown keys", func() {
lo := parseOptions(&pb.ModelOptions{EngineArgs: `{"gpu_memory_utilization": 0.9}`})
Expect(lo).To(Equal(loadOptions{}))
})
})
var _ = Describe("samplingFromPredict", func() {
@@ -292,91 +135,6 @@ var _ = Describe("samplingFromPredict", func() {
})
})
// The engine resolves speculative_config.model against a local directory or
// ~/.cache/huggingface/hub ONLY - it never downloads. LocalAI keeps models in
// its own directory, so a bare repo id would miss the HF cache and fail deep in
// the load with a confusing "draft checkpoint not found". Resolve it here.
var _ = Describe("resolveDraftModelPath", func() {
var modelsDir string
BeforeEach(func() {
modelsDir = GinkgoT().TempDir()
})
// draftDir creates a plausible draft checkpoint under models/.
draftDir := func(name string) string {
d := filepath.Join(modelsDir, name)
Expect(os.MkdirAll(d, 0o750)).To(Succeed())
Expect(os.WriteFile(filepath.Join(d, "config.json"), []byte("{}"), 0o600)).To(Succeed())
return d
}
It("rewrites a repo id to the matching directory in the models dir", func() {
want := draftDir("Qwen3.6-27B-DFlash")
spec := `{"method":"dflash","model":"z-lab/Qwen3.6-27B-DFlash"}`
out, err := resolveDraftModelPath(spec, modelsDir)
Expect(err).ToNot(HaveOccurred())
Expect(out).To(MatchJSON(`{"method":"dflash","model":"` + want + `"}`))
})
It("rewrites a models-dir-relative path", func() {
want := draftDir("drafts__dflash")
spec := `{"method":"dflash","model":"drafts__dflash"}`
out, err := resolveDraftModelPath(spec, modelsDir)
Expect(err).ToNot(HaveOccurred())
Expect(out).To(ContainSubstring(want))
})
It("leaves an absolute path that already resolves alone", func() {
abs := draftDir("elsewhere")
spec := `{"method":"dflash","model":"` + abs + `"}`
out, err := resolveDraftModelPath(spec, modelsDir)
Expect(err).ToNot(HaveOccurred())
Expect(out).To(MatchJSON(spec))
})
It("fails with an actionable error when the draft is nowhere on disk", func() {
// Silently passing the repo id through would surface as an HF-cache
// miss inside the engine, which reads as "your model is broken".
spec := `{"method":"dflash","model":"z-lab/Not-Downloaded"}`
_, err := resolveDraftModelPath(spec, modelsDir)
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(ContainSubstring("z-lab/Not-Downloaded"))
Expect(err.Error()).To(ContainSubstring(modelsDir))
})
It("requires a model key for dflash", func() {
_, err := resolveDraftModelPath(`{"method":"dflash"}`, modelsDir)
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(ContainSubstring("model"))
})
It("leaves mtp and ngram configs untouched", func() {
// Neither has a separate draft checkpoint to resolve.
for _, spec := range []string{
`{"method":"mtp"}`,
`{"method":"ngram","num_speculative_tokens":4}`,
} {
out, err := resolveDraftModelPath(spec, modelsDir)
Expect(err).ToNot(HaveOccurred())
Expect(out).To(MatchJSON(spec))
}
})
It("passes a malformed document through for the engine to reject", func() {
// The engine owns config validation and produces the better message.
out, err := resolveDraftModelPath(`{not json`, modelsDir)
Expect(err).ToNot(HaveOccurred())
Expect(out).To(Equal(`{not json`))
})
It("is a no-op on an empty config", func() {
out, err := resolveDraftModelPath("", modelsDir)
Expect(err).ToNot(HaveOccurred())
Expect(out).To(BeEmpty())
})
})
var _ = Describe("validModelPath", func() {
It("accepts a .gguf file", func() {
dir := GinkgoT().TempDir()

View File

@@ -24,14 +24,6 @@ func systemdActivatedListeners() ([]net.Listener, error) {
}
}()
// A half-populated environment is not an activation attempt. Container runtimes
// started from a socket-activated system unit leak a bare LISTEN_PID into every
// container they spawn, and systemd's own sd_listen_fds() treats either variable
// being absent as "not activated" rather than as an error.
if listenPID == "" || listenFDs == "" {
return nil, nil
}
pid, err := strconv.Atoi(listenPID)
if err != nil {
return nil, fmt.Errorf("invalid LISTEN_PID %q: %w", listenPID, err)

View File

@@ -85,34 +85,6 @@ var _ = Describe("systemdActivatedListeners", func() {
Expect(os.Getenv("LISTEN_FDNAMES")).To(BeEmpty())
})
It("binds normally when the environment leaks LISTEN_PID without LISTEN_FDS", func() {
Expect(os.Setenv("LISTEN_PID", strconv.Itoa(os.Getpid()))).To(Succeed())
Expect(os.Unsetenv("LISTEN_FDS")).To(Succeed())
DeferCleanup(func() {
_ = os.Unsetenv("LISTEN_PID")
})
listeners, err := systemdActivatedListeners()
Expect(err).NotTo(HaveOccurred())
Expect(listeners).To(BeEmpty())
Expect(os.Getenv("LISTEN_PID")).To(BeEmpty())
})
It("binds normally when the environment leaks LISTEN_FDS without LISTEN_PID", func() {
Expect(os.Unsetenv("LISTEN_PID")).To(Succeed())
Expect(os.Setenv("LISTEN_FDS", "1")).To(Succeed())
DeferCleanup(func() {
_ = os.Unsetenv("LISTEN_FDS")
})
listeners, err := systemdActivatedListeners()
Expect(err).NotTo(HaveOccurred())
Expect(listeners).To(BeEmpty())
Expect(os.Getenv("LISTEN_FDS")).To(BeEmpty())
})
It("reports malformed activation metadata instead of silently binding another socket", func() {
Expect(os.Setenv("LISTEN_PID", strconv.Itoa(os.Getpid()))).To(Succeed())
Expect(os.Setenv("LISTEN_FDS", "not-a-number")).To(Succeed())

View File

@@ -41,12 +41,12 @@
"glm-5": {"min_p":0.01,"repeat_penalty":1,"temperature":1,"top_k":-1,"top_p":0.95},
"glm-4": {"min_p":0.01,"repeat_penalty":1,"temperature":1,"top_k":-1,"top_p":0.95},
"nemotron": {"min_p":0.01,"repeat_penalty":1,"temperature":1,"top_k":-1,"top_p":1},
"minimax-m3": {"min_p":0.01,"repeat_penalty":1,"temperature":1,"top_k":40,"top_p":0.95},
"minimax-m2.7": {"min_p":0.01,"repeat_penalty":1,"temperature":1,"top_k":40,"top_p":0.95},
"minimax-m2.5": {"min_p":0.01,"repeat_penalty":1,"temperature":1,"top_k":40,"top_p":0.95},
"minimax": {"min_p":0.01,"repeat_penalty":1,"temperature":1,"top_k":40,"top_p":0.95},
"gpt-oss": {"min_p":0.01,"repeat_penalty":1,"temperature":1,"top_k":0,"top_p":1},
"granite-4": {"min_p":0.01,"repeat_penalty":1,"temperature":0,"top_k":0,"top_p":1},
"kimi-k3": {"min_p":0,"repeat_penalty":1,"temperature":1,"top_k":-1,"top_p":0.95},
"kimi-k2": {"min_p":0.01,"repeat_penalty":1,"temperature":0.6,"top_k":-1,"top_p":0.95},
"kimi": {"min_p":0.01,"repeat_penalty":1,"temperature":0.6,"top_k":-1,"top_p":0.95},
"lfm2": {"min_p":0.15,"repeat_penalty":1.05,"temperature":0.1,"top_k":50,"top_p":0.1},
@@ -58,5 +58,5 @@
"grok": {"min_p":0.01,"repeat_penalty":1,"temperature":1,"top_k":-1,"top_p":0.95},
"mimo": {"min_p":0.01,"repeat_penalty":1,"temperature":0.7,"top_k":-1,"top_p":0.95}
},
"patterns": ["qwen3.6","qwen3.5","qwen3-coder","qwen3-next","qwen3-vl","qwen3","qwen2.5-coder","qwen2.5-vl","qwen2.5-omni","qwen2.5-math","qwen2.5","qwen2-vl","qwen2","qwq","gemma-4","gemma-3n","gemma-3","medgemma","gemma-2","llama-4","llama-3.3","llama-3.2","llama-3.1","llama-3","phi-4","phi-3","mistral-nemo","mistral-small","mistral-large","magistral","ministral","devstral","pixtral","deepseek-v4","deepseek-r1","deepseek-v3","deepseek-ocr","glm-5","glm-4","nemotron","minimax-m3","minimax-m2.7","minimax-m2.5","minimax","gpt-oss","granite-4","kimi-k2","kimi","lfm2","smollm","olmo","falcon","ernie","seed","grok","mimo"]
"patterns": ["qwen3.6","qwen3.5","qwen3-coder","qwen3-next","qwen3-vl","qwen3","qwen2.5-coder","qwen2.5-vl","qwen2.5-omni","qwen2.5-math","qwen2.5","qwen2-vl","qwen2","qwq","gemma-4","gemma-3n","gemma-3","medgemma","gemma-2","llama-4","llama-3.3","llama-3.2","llama-3.1","llama-3","phi-4","phi-3","mistral-nemo","mistral-small","mistral-large","magistral","ministral","devstral","pixtral","deepseek-v4","deepseek-r1","deepseek-v3","deepseek-ocr","glm-5","glm-4","nemotron","minimax-m2.7","minimax-m2.5","minimax","gpt-oss","granite-4","kimi-k3","kimi-k2","kimi","lfm2","smollm","olmo","falcon","ernie","seed","grok","mimo"]
}

View File

@@ -1,117 +0,0 @@
package config
// Speculative-decoding auto-defaults for the vllm-cpp backend, the safetensors
// counterpart of the GGUF/llama.cpp hook in mtp.go.
//
// The two engines detect and spell the same feature differently. llama.cpp
// reads `<arch>.nextn_predict_layers` out of the GGUF header and takes
// `spec_type:draft-mtp` in `options:`; vllm.cpp reads `mtp_num_hidden_layers`
// out of the checkpoint's config.json and takes vLLM's own
// `--speculative-config` JSON, which LocalAI carries in `engine_args`. The
// engine resolves the draft depth and the default k itself, so the config only
// has to name the method.
import (
"encoding/json"
"github.com/mudler/xlog"
)
// hfSpecConfig is the subset of a HuggingFace config.json that decides whether
// speculative decoding can be auto-enabled.
type hfSpecConfig struct {
ModelType string `json:"model_type"`
// MtpNumHiddenLayers is the MTP head depth (upstream speculative.py reads
// it as n_predict for the qwen3_5 / qwen3_5_moe families).
MtpNumHiddenLayers uint32 `json:"mtp_num_hidden_layers"`
// DFlashConfig marks a z-lab DFlash DRAFT checkpoint (mask_token_id +
// target_layer_ids). Its presence means this repo is a draft, not a
// servable target.
DFlashConfig json.RawMessage `json:"dflash_config"`
// TextConfig is where multimodal checkpoints nest the language-model
// config, and therefore the MTP depth.
TextConfig *hfSpecConfig `json:"text_config"`
}
// parseHFSpecConfig decodes the speculative-relevant subset of a config.json.
// A document that does not parse yields nothing rather than an error: detection
// is best-effort and must never break an import.
func parseHFSpecConfig(configJSON []byte) (hfSpecConfig, bool) {
if len(configJSON) == 0 {
return hfSpecConfig{}, false
}
var c hfSpecConfig
if err := json.Unmarshal(configJSON, &c); err != nil {
xlog.Debug("[vllm-spec] config.json did not parse; skipping detection", "error", err)
return hfSpecConfig{}, false
}
return c, true
}
// IsDFlashDraftConfig reports whether a HuggingFace config.json describes a
// DFlash DRAFT checkpoint. Unlike MTP - whose head ships inside the target
// checkpoint's `mtp.*` tensors - a DFlash draft is its own repo that can only
// run paired with a target it verifies against, so it must never be configured
// as a standalone model.
func IsDFlashDraftConfig(configJSON []byte) bool {
c, ok := parseHFSpecConfig(configJSON)
if !ok {
return false
}
return len(c.DFlashConfig) > 0 ||
(c.TextConfig != nil && len(c.TextConfig.DFlashConfig) > 0)
}
// HasSafetensorsMTPHead reports whether a HuggingFace config.json declares a
// self-speculating Multi-Token Prediction head, returning its depth. The depth
// is informational: vllm.cpp resolves n_predict and the default
// num_speculative_tokens from the checkpoint itself.
//
// DFlash drafts are excluded for the same reason `gemma4-assistant` GGUFs are
// excluded from the llama.cpp hook: they carry head metadata but cannot
// self-speculate.
//
// NOTE this is a safetensors-only signal. vllm.cpp rejects an MTP config over a
// GGUF source, because the `mtp.*` draft tensors only exist in the safetensors
// checkpoint - so the GGUF import path must not use this.
func HasSafetensorsMTPHead(configJSON []byte) (uint32, bool) {
c, ok := parseHFSpecConfig(configJSON)
if !ok {
return 0, false
}
if IsDFlashDraftConfig(configJSON) {
return 0, false
}
n := c.MtpNumHiddenLayers
if n == 0 && c.TextConfig != nil {
n = c.TextConfig.MtpNumHiddenLayers
}
return n, n > 0
}
// ApplyVLLMSpeculativeDefaults enables MTP speculative decoding in cfg's
// engine_args when nothing is configured there yet. It is a no-op when the user
// already set a speculative_config, so an explicit choice (a different method,
// an explicit k, a DFlash draft) is never clobbered.
//
// `layers` is the detected head depth and is only used for the diagnostic log
// line - the engine derives the real k from the checkpoint.
func ApplyVLLMSpeculativeDefaults(cfg *ModelConfig, layers uint32) {
if cfg == nil {
return
}
if _, set := cfg.EngineArgs["speculative_config"]; set {
xlog.Debug("[vllm-spec] MTP head detected but speculative_config already configured; leaving user choice intact",
"name", cfg.Name, "mtp_num_hidden_layers", layers)
return
}
if cfg.EngineArgs == nil {
cfg.EngineArgs = map[string]any{}
}
// Only the method: vllm.cpp defaults num_speculative_tokens to the
// checkpoint's own n_predict (speculative.py:865-875), which is the right
// value far more reliably than anything guessable here.
cfg.EngineArgs["speculative_config"] = map[string]any{"method": "mtp"}
xlog.Info("[vllm-spec] MTP head detected; enabling mtp speculative decoding",
"name", cfg.Name, "mtp_num_hidden_layers", layers)
}

View File

@@ -1,117 +0,0 @@
package config_test
import (
. "github.com/mudler/LocalAI/core/config"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("vllm-cpp speculative-decoding auto-defaults", func() {
Context("HasSafetensorsMTPHead", func() {
It("detects a top-level mtp_num_hidden_layers", func() {
n, ok := HasSafetensorsMTPHead([]byte(`{
"model_type": "qwen3_5_moe",
"mtp_num_hidden_layers": 1
}`))
Expect(ok).To(BeTrue())
Expect(n).To(Equal(uint32(1)))
})
It("detects the head nested under text_config", func() {
// Multimodal checkpoints nest the language-model config, which is
// where the MTP depth lives (mirrors the engine's own resolution
// off config.raw text_config).
n, ok := HasSafetensorsMTPHead([]byte(`{
"model_type": "qwen3_5_moe",
"text_config": {"mtp_num_hidden_layers": 2}
}`))
Expect(ok).To(BeTrue())
Expect(n).To(Equal(uint32(2)))
})
It("reports no head when the key is absent", func() {
n, ok := HasSafetensorsMTPHead([]byte(`{"model_type": "llama"}`))
Expect(ok).To(BeFalse())
Expect(n).To(BeZero())
})
It("reports no head for a zero depth", func() {
_, ok := HasSafetensorsMTPHead([]byte(`{"mtp_num_hidden_layers": 0}`))
Expect(ok).To(BeFalse())
})
It("ignores a DFlash draft checkpoint", func() {
// A DFlash draft is a SEPARATE checkpoint that cannot serve alone:
// it needs a target to verify against. Same exclusion the GGUF path
// makes for gemma4-assistant drafts.
_, ok := HasSafetensorsMTPHead([]byte(`{
"model_type": "qwen3_dflash",
"mtp_num_hidden_layers": 1,
"dflash_config": {"mask_token_id": 151666, "target_layer_ids": [0, 1]}
}`))
Expect(ok).To(BeFalse())
})
It("reports no head on unparseable JSON", func() {
_, ok := HasSafetensorsMTPHead([]byte(`{not json`))
Expect(ok).To(BeFalse())
})
It("reports no head on empty input", func() {
_, ok := HasSafetensorsMTPHead(nil)
Expect(ok).To(BeFalse())
})
})
Context("IsDFlashDraftConfig", func() {
It("recognises a draft by its dflash_config block", func() {
Expect(IsDFlashDraftConfig([]byte(`{
"dflash_config": {"mask_token_id": 151666, "target_layer_ids": [0]}
}`))).To(BeTrue())
})
It("does not flag an ordinary checkpoint", func() {
Expect(IsDFlashDraftConfig([]byte(`{"model_type": "qwen3_5_moe"}`))).To(BeFalse())
})
})
Context("ApplyVLLMSpeculativeDefaults", func() {
It("writes the mtp method into engine_args", func() {
cfg := &ModelConfig{Name: "qwen"}
ApplyVLLMSpeculativeDefaults(cfg, 1)
Expect(cfg.EngineArgs).To(HaveKey("speculative_config"))
spec, ok := cfg.EngineArgs["speculative_config"].(map[string]any)
Expect(ok).To(BeTrue())
Expect(spec["method"]).To(Equal("mtp"))
})
It("leaves an existing speculative_config alone", func() {
cfg := &ModelConfig{
Name: "qwen",
LLMConfig: LLMConfig{
EngineArgs: map[string]any{
"speculative_config": map[string]any{"method": "ngram", "num_speculative_tokens": 4},
},
},
}
ApplyVLLMSpeculativeDefaults(cfg, 1)
spec := cfg.EngineArgs["speculative_config"].(map[string]any)
Expect(spec["method"]).To(Equal("ngram"))
})
It("preserves unrelated engine_args keys", func() {
cfg := &ModelConfig{
Name: "qwen",
LLMConfig: LLMConfig{EngineArgs: map[string]any{"max_num_seqs": 32}},
}
ApplyVLLMSpeculativeDefaults(cfg, 1)
Expect(cfg.EngineArgs).To(HaveKeyWithValue("max_num_seqs", 32))
Expect(cfg.EngineArgs).To(HaveKey("speculative_config"))
})
It("tolerates a nil config", func() {
Expect(func() { ApplyVLLMSpeculativeDefaults(nil, 1) }).ToNot(Panic())
})
})
})

View File

@@ -9,7 +9,6 @@ import (
"time"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/pkg/concurrency"
"github.com/mudler/LocalAI/pkg/system"
"github.com/mudler/LocalAI/pkg/vram"
"github.com/mudler/xlog"
@@ -102,7 +101,7 @@ func WarmEstimateCache(ctx context.Context, galleries []config.Gallery, systemSt
return
}
concurrency.SafeGo(func() {
go func() {
started := time.Now()
models, err := AvailableGalleryModelsCached(galleries, systemState)
@@ -132,7 +131,7 @@ func WarmEstimateCache(ctx context.Context, galleries []config.Gallery, systemSt
for i := 0; i < cfg.Concurrency; i++ {
wg.Add(1)
concurrency.SafeGo(func() {
go func() {
defer wg.Done()
for m := range cursor {
// Per entry, not for the run: one unreachable weight file
@@ -165,7 +164,7 @@ func WarmEstimateCache(ctx context.Context, galleries []config.Gallery, systemSt
cancel()
}
})
}()
}
feed:
@@ -184,7 +183,7 @@ func WarmEstimateCache(ctx context.Context, galleries []config.Gallery, systemSt
return
}
xlog.Info("gallery caches warmed", "estimates", warmed, "variants", warmedVariants, "of", len(models), "took", time.Since(started).Round(time.Second))
})
}()
}
// EstimateWarmConfigFromEnv reads the warm-up bounds from the environment,

View File

@@ -1,20 +1,11 @@
package gallery_test
import (
"bytes"
"context"
"encoding/binary"
"math"
"net/http"
"net/http/httptest"
"os"
"path/filepath"
"time"
gguf "github.com/gpustack/gguf-parser-go"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"gopkg.in/yaml.v3"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/gallery"
@@ -66,46 +57,6 @@ var _ = Describe("VRAM estimate warm-up", func() {
Consistently(func() bool { return true }, "100ms").Should(BeTrue())
})
It("does not crash the server when remote GGUF metadata is malformed", func() {
payload := warmMalformedGGUF()
requested := make(chan struct{})
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
select {
case <-requested:
default:
close(requested)
}
http.ServeContent(w, r, "model.gguf", time.Time{}, bytes.NewReader(payload))
}))
DeferCleanup(server.Close)
galleryPath := filepath.Join(state.Model.ModelsPath, "malformed-gallery.yaml")
index, err := yaml.Marshal([]gallery.GalleryModel{{Metadata: gallery.Metadata{
Name: "malformed-gguf",
AdditionalFiles: []gallery.File{{
Filename: "model.gguf",
URI: server.URL + "/model.gguf",
}},
}}})
Expect(err).NotTo(HaveOccurred())
Expect(os.WriteFile(galleryPath, index, 0600)).To(Succeed())
cfg := gallery.DefaultEstimateWarmConfig
cfg.Limit = 1
cfg.Concurrency = 1
cfg.Contexts = []uint32{8192}
gallery.WarmEstimateCache(context.Background(), []config.Gallery{{
Name: "malformed",
URL: "file://" + galleryPath,
}}, state, cfg)
Eventually(requested, "2s").Should(BeClosed())
// The warm-up is detached. Give its parser time to consume the response;
// before the recovery boundary, that goroutine panicked and killed the
// entire test process (and the LocalAI server in production).
Consistently(func() bool { return true }, "300ms").Should(BeTrue())
})
Describe("configuration from the environment", func() {
AfterEach(func() {
os.Unsetenv("LOCALAI_VRAM_WARM_LIMIT")
@@ -162,19 +113,3 @@ var _ = Describe("VRAM estimate warm-up", func() {
})
})
func warmMalformedGGUF() []byte {
payload := make([]byte, 0, 128)
payload = binary.LittleEndian.AppendUint32(payload, uint32(gguf.GGUFMagicGGUFLe))
payload = binary.LittleEndian.AppendUint32(payload, uint32(gguf.GGUFVersionV3))
payload = binary.LittleEndian.AppendUint64(payload, 0)
payload = binary.LittleEndian.AppendUint64(payload, 1)
key := "tokenizer.ggml.tokens"
payload = binary.LittleEndian.AppendUint64(payload, uint64(len(key)))
payload = append(payload, key...)
payload = binary.LittleEndian.AppendUint32(payload, uint32(gguf.GGUFMetadataValueTypeArray))
payload = binary.LittleEndian.AppendUint32(payload, uint32(gguf.GGUFMetadataValueTypeString))
payload = binary.LittleEndian.AppendUint64(payload, 1)
payload = binary.LittleEndian.AppendUint64(payload, math.MaxUint64)
return payload
}

View File

@@ -298,15 +298,7 @@ func (i *LlamaCPPImporter) Import(details Details) (gallery.ModelConfig, error)
// imported configs already carry spec_type:draft-mtp before the model is
// ever loaded - users see it in the YAML preview rather than discovering
// it after the first start.
//
// vllm-cpp is excluded on both counts: `spec_type:*` are llama.cpp option
// keys it does not read, and vllm.cpp rejects an MTP config over a GGUF
// source outright (the `mtp.*` draft tensors exist only in the safetensors
// checkpoint). Its MTP auto-config runs in the vllm importer instead, over
// the safetensors config.json.
if backend != "vllm-cpp" {
maybeApplyMTPDefaults(&modelConfig, details, &cfg)
}
maybeApplyMTPDefaults(&modelConfig, details, &cfg)
data, err := yaml.Marshal(modelConfig)
if err != nil {
@@ -409,10 +401,7 @@ func maybeApplyMTPDefaults(modelConfig *config.ModelConfig, details Details, cfg
}
}()
// MTP markers are architecture scalars. Avoid allocating tokenizer and
// other large arrays from an untrusted remote header; panic recovery cannot
// contain a fatal out-of-memory condition.
f, err := gguf.ParseGGUFFileRemote(ctx, probeURL, gguf.SkipLargeMetadata())
f, err := gguf.ParseGGUFFileRemote(ctx, probeURL)
if err != nil {
xlog.Debug("[mtp-importer] failed to read remote GGUF header for MTP detection", "uri", probeURL, "error", err)
return

View File

@@ -1,21 +1,13 @@
package importers
import (
"context"
"encoding/json"
"fmt"
"io"
"net/http"
"path/filepath"
"strings"
"time"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/gallery"
"github.com/mudler/LocalAI/core/schema"
"github.com/mudler/LocalAI/pkg/downloader"
"github.com/mudler/LocalAI/pkg/httpclient"
"github.com/mudler/xlog"
"go.yaml.in/yaml/v2"
)
@@ -115,12 +107,6 @@ func (i *VLLMImporter) Import(details Details) (gallery.ModelConfig, error) {
// vllm python backend, so use_tokenizer_template carries over), but
// tool/reasoning parsing is the engine's own autoparser pipeline -
// the vllm-python tool_parser/reasoning_parser options don't apply.
//
// Auto-detect a Multi-Token Prediction head, the safetensors analogue
// of the llama-cpp importer's GGUF hook, so a freshly imported
// Qwen3.5 / Qwen3.6 config already carries speculative decoding in its
// engine_args instead of leaving the throughput on the table.
maybeApplyVLLMSpeculativeDefaults(&modelConfig, details)
} else {
// Auto-detect tool_parser and reasoning_parser for known model families.
// Surfacing them in the generated YAML lets users see and edit the choices.
@@ -146,89 +132,3 @@ func (i *VLLMImporter) Import(details Details) (gallery.ModelConfig, error) {
ConfigFile: string(data),
}, nil
}
// maxSpecConfigProbeBytes caps the config.json body we read. Real ones are a
// few KB; the cap keeps a hostile or mislabelled URL from streaming into the
// importer.
const maxSpecConfigProbeBytes = 1 << 20 // 1 MiB
// specConfigProbeTimeout bounds the config.json fetch. Detection is an
// optimisation, so it must never hold an import open for long.
const specConfigProbeTimeout = 30 * time.Second
// specConfigFetcher is the seam the config.json probe goes through, so tests can
// drive the whole import path without a network round trip.
var specConfigFetcher = fetchProbeBody
// maybeApplyVLLMSpeculativeDefaults fetches the repository's config.json and,
// when it declares a Multi-Token Prediction head, enables MTP speculative
// decoding in the emitted engine_args. This is the safetensors counterpart of
// the llama-cpp importer's GGUF header probe.
//
// Every failure is non-fatal and logged at debug: a network blip, a private
// repo, or a config.json this doesn't understand must leave the import working
// exactly as it did before, just without the speculative default.
func maybeApplyVLLMSpeculativeDefaults(modelConfig *config.ModelConfig, details Details) {
probeURL := vllmSpecProbeURL(details)
if probeURL == "" {
return
}
body, err := specConfigFetcher(probeURL)
if err != nil {
xlog.Debug("[vllm-spec-importer] could not read config.json for MTP detection", "uri", probeURL, "error", err)
return
}
applySpecFromConfigJSON(modelConfig, body, details.URI)
}
// applySpecFromConfigJSON is the decision half of the probe, split out so it can
// be exercised without a network round trip.
func applySpecFromConfigJSON(modelConfig *config.ModelConfig, body []byte, uri string) {
if config.IsDFlashDraftConfig(body) {
// A DFlash draft cannot serve on its own - it only proposes tokens for
// a target model to verify. Say so rather than emitting a config that
// would fail at load.
xlog.Warn("[vllm-spec-importer] this repository is a DFlash DRAFT checkpoint, not a servable model; "+
"import the TARGET model and point engine_args.speculative_config at this repo "+
`({"method":"dflash","model":"<this repo>"})`, "uri", uri)
return
}
n, ok := config.HasSafetensorsMTPHead(body)
if !ok {
return
}
config.ApplyVLLMSpeculativeDefaults(modelConfig, n)
}
// vllmSpecProbeURL returns the HTTP(S) URL of the repository's config.json, or
// "" when the import isn't backed by a HuggingFace repo we can fetch from (a
// local directory import, an OCI artifact, ...).
func vllmSpecProbeURL(details Details) string {
if details.HuggingFace == nil || details.HuggingFace.ModelID == "" {
return ""
}
return resolveHTTPProbe(downloader.HuggingFacePrefix + details.HuggingFace.ModelID + "/config.json")
}
// fetchProbeBody GETs a small remote JSON document under a short timeout.
func fetchProbeBody(url string) ([]byte, error) {
ctx, cancel := context.WithTimeout(context.Background(), specConfigProbeTimeout)
defer cancel()
req, err := http.NewRequestWithContext(ctx, http.MethodGet, url, nil)
if err != nil {
return nil, err
}
resp, err := httpclient.NewWithTimeout(specConfigProbeTimeout).Do(req)
if err != nil {
return nil, err
}
defer func() { _ = resp.Body.Close() }()
if resp.StatusCode != http.StatusOK {
return nil, fmt.Errorf("unexpected status %d", resp.StatusCode)
}
return io.ReadAll(io.LimitReader(resp.Body, maxSpecConfigProbeBytes))
}

View File

@@ -1,118 +0,0 @@
package importers
import (
"encoding/json"
"errors"
"github.com/mudler/LocalAI/core/config"
hfapi "github.com/mudler/LocalAI/pkg/huggingface-api"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("vllm-cpp speculative auto-config (importer)", func() {
Context("applySpecFromConfigJSON", func() {
It("enables mtp when the checkpoint declares an MTP head", func() {
cfg := &config.ModelConfig{Name: "qwen3.5"}
applySpecFromConfigJSON(cfg, []byte(`{
"model_type": "qwen3_5_moe",
"mtp_num_hidden_layers": 1
}`), "huggingface://Qwen/Qwen3.5-A3B")
Expect(cfg.EngineArgs).To(HaveKeyWithValue("speculative_config",
map[string]any{"method": "mtp"}))
})
It("leaves a plain checkpoint untouched", func() {
cfg := &config.ModelConfig{Name: "llama"}
applySpecFromConfigJSON(cfg, []byte(`{"model_type": "llama"}`), "huggingface://meta/llama")
Expect(cfg.EngineArgs).To(BeEmpty())
})
It("refuses to configure a DFlash draft as a servable model", func() {
// The draft only proposes tokens; configuring it standalone would
// produce a model that cannot load.
cfg := &config.ModelConfig{Name: "dflash-draft"}
applySpecFromConfigJSON(cfg, []byte(`{
"model_type": "qwen3_dflash",
"dflash_config": {"mask_token_id": 151666, "target_layer_ids": [0, 1]}
}`), "huggingface://z-lab/Qwen3.6-27B-DFlash")
Expect(cfg.EngineArgs).To(BeEmpty())
})
It("survives a config.json it cannot parse", func() {
cfg := &config.ModelConfig{Name: "weird"}
Expect(func() {
applySpecFromConfigJSON(cfg, []byte(`<html>404</html>`), "huggingface://a/b")
}).ToNot(Panic())
Expect(cfg.EngineArgs).To(BeEmpty())
})
})
Context("Import over a repository with an MTP head", func() {
var restore func()
BeforeEach(func() {
original := specConfigFetcher
restore = func() { specConfigFetcher = original }
})
AfterEach(func() { restore() })
importWith := func(backend, configJSON string) string {
specConfigFetcher = func(string) ([]byte, error) {
return []byte(configJSON), nil
}
importer := &VLLMImporter{}
out, err := importer.Import(Details{
URI: "huggingface://Qwen/Qwen3.5-A3B",
Preferences: json.RawMessage(`{"backend": "` + backend + `"}`),
HuggingFace: &hfapi.ModelDetails{ModelID: "Qwen/Qwen3.5-A3B"},
})
Expect(err).ToNot(HaveOccurred())
return out.ConfigFile
}
It("emits engine_args.speculative_config for vllm-cpp", func() {
yaml := importWith("vllm-cpp", `{"model_type":"qwen3_5_moe","mtp_num_hidden_layers":1}`)
Expect(yaml).To(ContainSubstring("engine_args:"))
Expect(yaml).To(ContainSubstring("speculative_config:"))
Expect(yaml).To(ContainSubstring("method: mtp"))
})
It("emits nothing speculative for the python vllm backend", func() {
// The python backend has its own speculative surface and its own
// version-dependent MTP support; this hook is vllm-cpp only.
yaml := importWith("vllm", `{"model_type":"qwen3_5_moe","mtp_num_hidden_layers":1}`)
Expect(yaml).NotTo(ContainSubstring("speculative_config"))
})
It("emits nothing speculative when the probe fails", func() {
specConfigFetcher = func(string) ([]byte, error) {
return nil, errors.New("network down")
}
importer := &VLLMImporter{}
out, err := importer.Import(Details{
URI: "huggingface://Qwen/Qwen3.5-A3B",
Preferences: json.RawMessage(`{"backend": "vllm-cpp"}`),
HuggingFace: &hfapi.ModelDetails{ModelID: "Qwen/Qwen3.5-A3B"},
})
Expect(err).ToNot(HaveOccurred())
Expect(out.ConfigFile).NotTo(ContainSubstring("speculative_config"))
})
})
Context("vllmSpecProbeURL", func() {
It("resolves the repository's config.json to an HTTPS URL", func() {
url := vllmSpecProbeURL(Details{
URI: "huggingface://Qwen/Qwen3.5-A3B",
HuggingFace: &hfapi.ModelDetails{ModelID: "Qwen/Qwen3.5-A3B"},
})
Expect(url).To(ContainSubstring("Qwen/Qwen3.5-A3B"))
Expect(url).To(HaveSuffix("config.json"))
Expect(url).To(HavePrefix("https://"))
})
It("skips the probe when there is no HuggingFace repo behind the import", func() {
Expect(vllmSpecProbeURL(Details{URI: "/models/local-dir"})).To(BeEmpty())
})
})
})

View File

@@ -125,22 +125,6 @@ async function generateOnce(page) {
await page.locator('button[type="submit"]').click()
}
async function pasteImage(page) {
await page.locator('.biometrics-mediainput').focus()
await page.evaluate((base64) => {
const bytes = Uint8Array.from(atob(base64), char => char.charCodeAt(0))
const transfer = new DataTransfer()
transfer.items.add(new File([bytes], 'clipboard.png', { type: 'image/png' }))
const target = document.querySelector('.biometrics-mediainput')
target.dispatchEvent(new ClipboardEvent('paste', {
bubbles: true,
cancelable: true,
clipboardData: transfer,
}))
}, TINY_PNG.toString('base64'))
await expect(page.locator('.biometrics-mediainput__source-pill')).toContainText('Pasted image')
}
test.describe('3D generation', () => {
test.beforeEach(async ({ page }) => {
await mockCapabilities(page)
@@ -170,60 +154,6 @@ test.describe('3D generation', () => {
expect(requestBody.response_format).toBe('url')
})
test('caps auto-rotate at 30 FPS and renders still models on demand', async ({ page }) => {
await page.addInitScript(() => {
window.__glDrawTimes = []
const proto = window.WebGL2RenderingContext?.prototype
if (!proto) return
const drawElements = proto.drawElements
proto.drawElements = function (...args) {
window.__glDrawTimes.push(performance.now())
return drawElements.apply(this, args)
}
})
await mockGeneration(page)
await generateOnce(page)
await expect(page.getByTestId('glb-stats')).toBeVisible({ timeout: 15_000 })
await page.waitForTimeout(100)
await page.evaluate(() => { window.__glDrawTimes = [] })
await page.waitForTimeout(600)
const drawTimes = await page.evaluate(() => window.__glDrawTimes)
test.skip(drawTimes.length < 3, 'WebGL2 drawing is unavailable in this browser')
expect(drawTimes.length).toBeLessThanOrEqual(22)
const gaps = drawTimes.slice(1).map((time, index) => time - drawTimes[index]).sort((a, b) => a - b)
expect(gaps[Math.floor(gaps.length / 2)]).toBeGreaterThan(25)
await page.getByRole('button', { name: 'Auto-rotate' }).click()
await page.waitForTimeout(100)
const stoppedAt = await page.evaluate(() => window.__glDrawTimes.length)
await page.waitForTimeout(250)
const idleAt = await page.evaluate(() => window.__glDrawTimes.length)
expect(idleAt - stoppedAt).toBeLessThanOrEqual(1)
await page.getByTestId('glb-canvas').dispatchEvent('wheel', { deltaY: 20 })
await expect.poll(() => page.evaluate(() => window.__glDrawTimes.length)).toBeGreaterThan(idleAt)
})
test('pastes a conditioning image without mounting its base64 in the request panel', async ({ page }) => {
let requestBody = null
await mockGeneration(page, (body) => { requestBody = body })
await page.goto('/app/studio/threed')
await expect(page.getByRole('button', { name: 'trellis-test-model' })).toBeVisible({ timeout: 10_000 })
await pasteImage(page)
await page.locator('button[type="submit"]').click()
await expect(page.getByTestId('glb-stats')).toBeVisible({ timeout: 15_000 })
await expect(page.getByTestId('media-history-item')).toHaveCount(1)
const panel = page.locator('.request-panel')
await expect(panel).toContainText('<base64 image/png omitted>')
const panelText = await panel.textContent()
expect(panelText.length).toBeLessThan(2000)
expect(panelText).not.toContain(requestBody.image)
expect(requestBody.image).toBeTruthy()
})
test('advanced settings map to step/texture_steps/cfg_scale/seed', async ({ page }) => {
let requestBody = null
await mockGeneration(page, (body) => { requestBody = body })
@@ -296,18 +226,6 @@ test.describe('3D generation', () => {
await expect(page.getByTestId('glb-download')).toHaveAttribute('href', /^blob:/)
})
test('new history is visible on the Studio overview without a reload', async ({ page }) => {
await mockGeneration(page)
await page.goto('/app/studio/threed')
await expect(page.getByRole('button', { name: 'trellis-test-model' })).toBeVisible({ timeout: 10_000 })
await page.locator('#threed-image-file').setInputFiles({ name: 'input.png', mimeType: 'image/png', buffer: TINY_PNG })
await page.locator('button[type="submit"]').click()
await expect(page.getByTestId('media-history-item')).toHaveCount(1, { timeout: 15_000 })
await page.locator('.studio-tab[data-tab="overview"]').click()
await expect(page.getByTestId('studio-recent')).toContainText('trellis-test-model')
})
test('deleting a history entry removes it', async ({ page }) => {
await mockGeneration(page)
await generateOnce(page)

View File

@@ -20,46 +20,3 @@ test('marks an API trace with no response status as in progress', async ({ page
await expect(row.locator('[title="In progress"]')).toBeVisible()
await expect(row.locator('.fa-check-circle')).toHaveCount(0)
})
// Regression for #11376: switching from Backend Traces back to API Traces
// used to crash the page. `traces` holds whichever list was fetched last, so
// right after `setActiveTab('api')` — before the refetch effect lands — the
// API table renders the previous tab's backend rows, which carry no
// `response` envelope. The status column must tolerate that instead of
// dereferencing `trace.response.status` and tearing down the React tree.
test('switching from backend to API traces with a response-less row does not crash', async ({ page }) => {
const pageErrors = []
page.on('pageerror', (e) => pageErrors.push(e.message))
await page.route('**/api/traces?*', route => route.fulfill({
json: [{
id: 'api-1',
timestamp: '2026-08-05T02:00:00Z',
request: { method: 'POST', path: '/v1/chat/completions' },
response: { status: 200 },
}],
headers: { 'X-Total-Count': '1' },
}))
await page.route('**/api/backend-traces?*', route => route.fulfill({
json: [{
id: 'backend-1',
type: 'llm',
timestamp: '2026-08-05T02:00:00Z',
model_name: 'mock-model',
summary: 'generated a reply',
}],
headers: { 'X-Total-Count': '1' },
}))
await page.goto('/app/traces')
await expect(page.locator('tbody tr').filter({ hasText: '/v1/chat/completions' })).toBeVisible()
await page.getByRole('button', { name: /Backend Traces/ }).click()
await expect(page.locator('tbody tr').filter({ hasText: 'generated a reply' })).toBeVisible()
await page.getByRole('button', { name: /API Traces/ }).click()
// The stale backend row renders in the API table for one frame; the status
// column falls back to a neutral placeholder rather than throwing.
await expect(page.locator('tbody tr').filter({ hasText: '/v1/chat/completions' })).toBeVisible()
expect(pageErrors).toEqual([])
})

View File

@@ -132,7 +132,6 @@ const Q = {
// GLBs are already Y-up (the baker swaps axes on export), so unlike the demo
// there is no Z-up correction here — just a gentle 3/4 default view.
const QBASE = Q.norm(Q.mul(Q.axisAngle(1, 0, 0, -0.30), Q.axisAngle(0, 1, 0, 0.55)))
const FRAME_INTERVAL_MS = 1000 / 30
/* minimal mat4 helpers (column-major) */
const M = {
@@ -334,12 +333,10 @@ export function createGlbViewer(canvas, { onContextLost } = {}) {
nIndices = 0
nWire = 0
dropTextures()
requestRender()
}
function resetView() {
rot = QBASE.slice(); dist = 1.8; panX = panY = 0
requestRender()
}
/* input */
@@ -356,7 +353,6 @@ export function createGlbViewer(canvas, { onContextLost } = {}) {
}
const stopSpin = () => {
spin = false
requestRender()
if (onSpinChange) onSpinChange(false)
}
const onPointerDown = (e) => {
@@ -404,7 +400,6 @@ export function createGlbViewer(canvas, { onContextLost } = {}) {
pinchDistance = nextDistance
pinchX = nextX
pinchY = nextY
requestRender()
return
}
@@ -420,14 +415,12 @@ export function createGlbViewer(canvas, { onContextLost } = {}) {
rot = Q.norm(Q.mul(Q.axisAngle(1, 0, 0, dy * k), Q.mul(Q.axisAngle(0, 1, 0, dx * k), rot)))
stopSpin()
}
requestRender()
}
const onContextMenu = (e) => e.preventDefault()
const onWheel = (e) => {
e.preventDefault()
dist *= Math.exp(e.deltaY * 0.001)
dist = Math.max(0.3, Math.min(8, dist))
requestRender()
}
const onDblClick = () => resetView()
let onSpinChange = null
@@ -446,26 +439,10 @@ export function createGlbViewer(canvas, { onContextLost } = {}) {
gl.clearColor(0.063, 0.078, 0.094, 1)
let rafId = 0
let lastDraw = 0
let dirty = true
function requestRender() {
dirty = true
if (!disposed && !rafId) rafId = requestAnimationFrame(frame)
}
let last = performance.now()
function frame(now) {
rafId = 0
if (disposed) return
// requestAnimationFrame follows the display refresh rate, which can be
// 120-240 Hz. Skip expensive mesh draws until the 30 FPS budget is due.
if (spin && lastDraw && now - lastDraw < FRAME_INTERVAL_MS) {
rafId = requestAnimationFrame(frame)
return
}
if (!spin && !dirty) return
const dt = lastDraw ? Math.min((now - lastDraw) / 1000, 0.1) : 0
lastDraw = now
dirty = false
const dt = (now - last) / 1000; last = now
// auto-rotate: a slow turn about the screen-vertical axis (turntable feel)
if (spin) rot = Q.norm(Q.mul(Q.axisAngle(0, 1, 0, dt * 0.4), rot))
@@ -526,20 +503,13 @@ export function createGlbViewer(canvas, { onContextLost } = {}) {
}
gl.bindVertexArray(null)
}
// A still model is complete until input, resize, or a control invalidates
// it. Spinning models keep scheduling frames, subject to the cap above.
if (spin) rafId = requestAnimationFrame(frame)
rafId = requestAnimationFrame(frame)
}
const resizeObserver = typeof ResizeObserver === 'undefined'
? null
: new ResizeObserver(requestRender)
resizeObserver?.observe(canvas)
requestRender()
rafId = requestAnimationFrame(frame)
function dispose() {
disposed = true
cancelAnimationFrame(rafId)
resizeObserver?.disconnect()
canvas.removeEventListener('pointerdown', onPointerDown)
canvas.removeEventListener('pointerup', onPointerUp)
canvas.removeEventListener('pointercancel', onPointerUp)
@@ -562,8 +532,8 @@ export function createGlbViewer(canvas, { onContextLost } = {}) {
clear,
dispose,
resetView,
setWire(v) { wire = v; requestRender() },
setSpin(v) { spin = v; requestRender() },
setWire(v) { wire = v },
setSpin(v) { spin = v },
onSpinChanged(fn) { onSpinChange = fn },
}
}

View File

@@ -51,7 +51,9 @@ export default function MediaInput({ mode, label, value, onChange, onError, maxB
if (tab !== 'live' && cap.active) cap.stop()
}, [tab]) // eslint-disable-line react-hooks/exhaustive-deps
const acceptFile = async (f, source = 'file') => {
const handleFile = async (e) => {
const f = e.target.files?.[0]
if (!f) { onChange(null); return }
if (maxBytes && f.size > maxBytes) {
const error = new Error(`Selected file exceeds the ${Math.round(maxBytes / (1024 * 1024))} MiB limit`)
if (fileRef.current) fileRef.current.value = ''
@@ -60,11 +62,8 @@ export default function MediaInput({ mode, label, value, onChange, onError, maxB
return
}
try {
const name = source === 'paste'
? `pasted-image.${(f.type.split('/')[1] || 'png').replace('+xml', '')}`
: f.name
if (preferBlob) {
onChange({ blob: f, mime: f.type, source, name })
onChange({ blob: f, mime: f.type, source: 'file', name: f.name })
return
}
const base64 = await fileToBase64(f)
@@ -74,30 +73,13 @@ export default function MediaInput({ mode, label, value, onChange, onError, maxB
reader.onload = () => resolve(reader.result)
reader.readAsDataURL(f)
})
onChange({ base64, blob: f, dataUrl, mime: f.type, source, name })
onChange({ base64, blob: f, dataUrl, mime: f.type, source: 'file', name: f.name })
} catch (error) {
onChange(null)
onError?.(error)
}
}
const handleFile = async (e) => {
const f = e.target.files?.[0]
if (!f) { onChange(null); return }
await acceptFile(f)
}
const handlePaste = async (e) => {
if (mode !== 'image') return
const item = Array.from(e.clipboardData?.items || []).find(entry => entry.type.startsWith('image/'))
const f = item?.getAsFile()
|| Array.from(e.clipboardData?.files || []).find(file => file.type.startsWith('image/'))
if (!f) return
e.preventDefault()
setTab('file')
await acceptFile(f, 'paste')
}
const handleSnap = () => {
const shot = cap.snap()
if (shot) onChange({ ...shot, source: 'live' })
@@ -124,13 +106,7 @@ export default function MediaInput({ mode, label, value, onChange, onError, maxB
const inputId = `${idPrefix}-${mode}-file`
return (
<div
className="biometrics-mediainput"
onPaste={handlePaste}
tabIndex={mode === 'image' ? 0 : undefined}
role={mode === 'image' ? 'group' : undefined}
aria-label={mode === 'image' ? `${label || 'Image'} upload or clipboard paste` : undefined}
>
<div className="biometrics-mediainput">
{label && <label className="form-label" htmlFor={inputId}>{label}</label>}
<div className="biometrics-mediainput__tabs" role="tablist" aria-label={`${label || 'Media'} source`}>
@@ -157,9 +133,6 @@ export default function MediaInput({ mode, label, value, onChange, onError, maxB
accept={mode === 'image' ? 'image/*' : 'audio/*'}
onChange={handleFile}
/>
{mode === 'image' && (
<p className="form-hint"><i className="fas fa-clipboard" aria-hidden="true" /> Paste an image from the clipboard</p>
)}
</div>
)}
@@ -211,8 +184,8 @@ export default function MediaInput({ mode, label, value, onChange, onError, maxB
: <audio controls src={value.dataUrl} />}
<div className="biometrics-mediainput__preview-meta">
<span className="biometrics-mediainput__source-pill">
<i className={`fas ${value.source === 'live' ? (mode === 'image' ? 'fa-camera' : 'fa-microphone') : value.source === 'paste' ? 'fa-clipboard' : 'fa-file'}`} aria-hidden="true" />
{value.source === 'live' ? ' Captured' : value.source === 'paste' ? ' Pasted image' : ` ${value.name || 'Uploaded'}`}
<i className={`fas ${value.source === 'live' ? (mode === 'image' ? 'fa-camera' : 'fa-microphone') : 'fa-file'}`} aria-hidden="true" />
{value.source === 'live' ? ' Captured' : ` ${value.name || 'Uploaded'}`}
</span>
<button type="button" className="biometrics-mediainput__clear" onClick={clear} aria-label="Remove sample">
<i className="fas fa-xmark" aria-hidden="true" />

View File

@@ -17,14 +17,6 @@ const DB_NAME = 'localai-3d-history'
const DB_VERSION = 1
const STORE = 'generations'
const MAX_ENTRIES = 20
const historyListeners = new Set()
let sessionEntries = []
async function refreshOtherHooks(source) {
await Promise.all([...historyListeners]
.filter(listener => listener !== source)
.map(listener => listener()))
}
function openDb() {
return new Promise((resolve, reject) => {
@@ -86,19 +78,14 @@ export function use3DHistory() {
const refresh = useCallback(async () => {
try {
sessionEntries = await idbGetAll()
setEntries(sessionEntries)
setEntries(await idbGetAll())
} catch {
// IndexedDB unavailable (private mode etc.) — degrade to session-only.
setEntries(sessionEntries)
setEntries((prev) => prev)
}
}, [])
useEffect(() => {
historyListeners.add(refresh)
refresh()
return () => { historyListeners.delete(refresh) }
}, [refresh])
useEffect(() => { refresh() }, [refresh])
const addEntry = useCallback(async ({ model, params, inputThumb, glb, name }) => {
const entry = { id: generateId(), createdAt: Date.now(), model, params, inputThumb, glb, name }
@@ -106,10 +93,8 @@ export function use3DHistory() {
await idbPutAndEvict(entry)
await refresh()
} catch {
sessionEntries = [entry, ...sessionEntries.filter(e => e.id !== entry.id)].slice(0, MAX_ENTRIES)
setEntries(sessionEntries)
setEntries((prev) => [entry, ...prev].slice(0, MAX_ENTRIES))
}
await refreshOtherHooks(refresh)
return entry
}, [refresh])
@@ -119,21 +104,19 @@ export function use3DHistory() {
await idbDelete(id)
await refresh()
} catch {
sessionEntries = sessionEntries.filter((e) => e.id !== id)
setEntries(sessionEntries)
setEntries((prev) => prev.filter((e) => e.id !== id))
}
await refreshOtherHooks(refresh)
}, [refresh])
const clearAll = useCallback(async () => {
setSelectedId(null)
try {
await idbClear()
} catch { /* session-only history is cleared below */ }
sessionEntries = []
} catch {
// fall through to the local reset below
}
setEntries([])
await refreshOtherHooks(refresh)
}, [refresh])
}, [])
// Toggles: clicking the selected entry deselects it (back to latest result).
const selectEntry = useCallback((id) => {

View File

@@ -100,9 +100,7 @@ export default function ThreeDGen() {
if (guidance) body.cfg_scale = parseFloat(guidance)
if (seed) body.seed = parseInt(seed)
// RequestPanel renders and copies its body. Keeping a multi-megabyte image
// there duplicates the upload in React and can starve the result render.
setLastRequest({ ...body, image: `<base64 ${image.mime || 'image'} omitted>` })
setLastRequest(body)
try {
const data = await threeDApi.generate(body)

View File

@@ -667,8 +667,6 @@ export default function Traces() {
<td>
{trace.response?.status === 0
? <span className="badge badge-info">Running</span>
: trace.response?.status == null
? <span className="badge badge--soft">-</span>
: <span className={`badge ${trace.response.status < 400 ? 'badge-success' : 'badge-error'}`}>{trace.response.status}</span>}
</td>
<td><LatencyCell ns={trace.duration} max={slowestTrace} /></td>

View File

@@ -74,9 +74,6 @@ services:
GODEBUG: "netdns=go"
# Paths
MODELS_PATH: /models
# Avoid probing remote gallery GGUF metadata during container startup.
# Remove this line or set a positive limit to opt back into cache warming.
LOCALAI_VRAM_WARM_LIMIT: "0"
volumes:
- frontend_models:/models
- frontend_data:/data

View File

@@ -18,9 +18,6 @@ services:
- .env
environment:
- MODELS_PATH=/models
# Avoid probing remote gallery GGUF metadata during container startup.
# Remove this line or set a positive limit to opt back into cache warming.
- LOCALAI_VRAM_WARM_LIMIT=0
# - DEBUG=true
## Agents (LocalAGI) - https://localai.io/features/agents/
# - LOCALAI_DISABLE_AGENTS=false

View File

@@ -477,11 +477,6 @@ then on.
| `LOCALAI_VRAM_WARM_LIMIT` | `300` | How many gallery entries to warm at startup, estimates and variants alike. Set to `0` to disable the warm-up entirely. |
| `LOCALAI_VRAM_WARM_CONCURRENCY` | `4` | How many estimates to run at once. |
The provided Docker Compose configurations set `LOCALAI_VRAM_WARM_LIMIT=0`
as a defensive default, so container startup does not probe remote GGUF files.
Remove that override or set it to a positive number to opt into background
warming.
```bash
# Air-gapped, or you would rather not make the requests at all
LOCALAI_VRAM_WARM_LIMIT=0 local-ai run

View File

@@ -113,7 +113,7 @@ curl http://localhost:8080/3d/generations \
## WebUI
The React UI includes a 3D tab in the Studio (and a `/3d` page) with an interactive PBR viewer: upload or paste an image from the clipboard, pick the quality, and preview the generated mesh with orbit/pan/zoom and a wireframe toggle. Past generations are kept in the browser (IndexedDB). After generation, a single Detail slider and **Apply remeshing** button replace the preview with the exact watertight model that the GLB download exports; **Show original** switches back without regenerating.
The React UI includes a 3D tab in the Studio (and a `/3d` page) with an interactive PBR viewer: upload an image, pick the quality, and preview the generated mesh with orbit/pan/zoom and a wireframe toggle. Past generations are kept in the browser (IndexedDB). After generation, a single Detail slider and **Apply remeshing** button replace the preview with the exact watertight model that the GLB download exports; **Show original** switches back without regenerating.
## Notes

View File

@@ -918,200 +918,6 @@ options:
The full list of registered parsers lives in `sglang.srt.function_call`
and `sglang.srt.parser.reasoning_parser`.
### vllm.cpp
[vllm.cpp](https://github.com/mudler/vllm.cpp) is the LocalAI team's C++ port of
vLLM: the same continuous-batching scheduler, paged KV cache and prefix caching,
with no Python at inference time. It consumes either a HuggingFace safetensors
model directory or a `.gguf` file, and applies the model's chat template,
tool-call parsing and reasoning split engine-side.
#### Setup
```yaml
name: vllm-cpp
backend: vllm-cpp
parameters:
model: "Qwen/Qwen3-4B"
context_size: 8192
template:
use_tokenizer_template: true
```
#### Configuring the engine with `engine_args`
The same `engine_args:` map the vLLM and SGLang backends accept is honoured
here, with keys spelled exactly as vLLM's own CLI flags - so a `speculative_config`
or `kv_transfer_config` block written for vLLM works verbatim. Unknown keys are
ignored rather than fatal; the engine validates the documents it is handed and
reports a precise error at load.
```yaml
name: qwen35-a3b
backend: vllm-cpp
parameters:
model: "Qwen/Qwen3.5-A3B"
context_size: 16384
template:
use_tokenizer_template: true
engine_args:
# KV cache sizing: num_blocks * block_size tokens of cache.
block_size: 32
num_blocks: 1024
# Concurrency and the per-step chunked-prefill token budget.
max_num_seqs: 32
max_num_batched_tokens: 8192
# Automatic prefix caching. Omit to keep the model's own default
# (on for dense models, off for hybrid / attention-free ones).
enable_prefix_caching: true
# Scheduler admission order: fcfs (default), priority, or lpm
# (cache-aware longest-prefix-match; needs prefix caching to have any effect).
scheduling_policy: lpm
```
| Key | Meaning | Default |
|-----|---------|---------|
| `block_size` | KV-cache block size, in tokens per block | 32 |
| `num_blocks` | KV-cache blocks to allocate | 256 |
| `max_model_len` | Max sequence length; also settable as `context_size` / `max_model_len` | model config |
| `max_num_seqs` | Max concurrent sequences the scheduler admits | 8 |
| `max_num_batched_tokens` | Per-step chunked-prefill token budget | per-arch (2048 dense, 4096/8192 MoE) |
| `enable_prefix_caching` | Automatic prefix caching; `enable_radix_attention` is an accepted alias | model default |
| `enable_jump_forward` | Jump-forward decoding, which emits grammar-forced tokens without a model step. Only affects constrained requests (`grammar`, JSON schema) | off |
| `scheduling_policy` | `fcfs`, `priority`, or `lpm` | `fcfs` |
| `tool_parser` / `reasoning_parser` | Force a parser instead of chat-template auto-detection | auto |
| `tokenizer_config` | Override the `tokenizer_config.json` the chat template is read from | `<model_dir>/tokenizer_config.json` |
| `speculative_config` | Speculative decoding (see below) | disabled |
| `kv_transfer_config` | External KV connector / LMCache (see below) | none |
Raising `max_num_batched_tokens` lets more prefill land in a single step, at the
cost of decode latency for requests queued behind it. The default deliberately
does not scale with `max_num_seqs`, which is what keeps a large concurrent
prefill from blowing up the per-step activation on the hybrid architectures.
`enable_prefix_caching` and `enable_jump_forward` are tri-state at the engine
boundary: omitting the key defers to a default (the model's own capability for
prefix caching, an environment variable for jump forward), while an explicit
`false` forces the feature off. Those are genuinely different - prefix caching
defaults *on* for dense models - so write the key only when you mean to override.
#### Speculative decoding
`speculative_config:` takes the same JSON object as vLLM's
`--speculative-config`. Three methods are supported.
> **Architecture limit.** At the current engine pin, `mtp` and `dflash` are
> **Qwen3.5 / Qwen3.6 only**. The engine builds a widened speculative KV cache
> directly for those families rather than through the model registry, so a
> speculative config on any other architecture (Llama, GLM, Gemma, Mistral, ...)
> will not work regardless of checkpoint format. `ngram` needs no draft weights
> and is not subject to this limit.
> **Format support.** `mtp` and `dflash` now work from a `.gguf` target as well
> as safetensors. An MTP head is read from the GGUF's `nextn.*` tensors when the
> file declares `<arch>.nextn_predict_layers`; a GGUF exported WITHOUT the head
> (converted with `--no-mtp`, or predating llama.cpp's Qwen3.5 MTP support) is
> refused at load naming that as the reason. A DFlash draft may itself be a
> `dflash`-arch GGUF, and the target may be a GGUF too. `ngram` needs no draft
> weights and works on any format.
**MTP** (Multi-Token Prediction) uses a draft head shipped inside the target
checkpoint's own `mtp.*` tensors, so there is no second model to download. It
requires a **safetensors** checkpoint - the `mtp.*` tensors do not survive GGUF
conversion, and an MTP config over a `.gguf` model is rejected at load.
```yaml
engine_args:
speculative_config:
method: mtp
# Optional; defaults to the checkpoint's own head depth, which is
# usually the right value. Must be a multiple of that depth.
num_speculative_tokens: 1
```
**DFlash** uses a separate block-diffusion drafter that proposes a whole block
of tokens in one non-autoregressive forward pass. Unlike MTP, the draft is its
own checkpoint, so `model:` is **required**:
```yaml
engine_args:
speculative_config:
method: dflash
model: z-lab/Qwen3.6-27B-DFlash
num_speculative_tokens: 4
```
The draft shares the *target's* `embed_tokens` and `lm_head`, so both must come
from the same model family and the target must be safetensors.
**The engine does not download the draft.** `model:` is resolved, in order,
as a path as given, then as the last path segment under LocalAI's models
directory (`z-lab/Qwen3.6-27B-DFlash``<models>/Qwen3.6-27B-DFlash`, which is
what LocalAI's own downloader produces), then as the whole reference under the
models directory. Install the draft into LocalAI first, or give an absolute path
to a directory containing `config.json`. If none of those resolve, the load
fails immediately naming every location that was tried, rather than reporting a
missing checkpoint from inside the engine.
**N-gram** needs no draft model at all - it proposes from the prompt's own
suffix history. `num_speculative_tokens` is required:
```yaml
engine_args:
speculative_config:
method: ngram
num_speculative_tokens: 4
prompt_lookup_min: 5
prompt_lookup_max: 5
```
> **Auto-configuration on import.** When you import a safetensors repository
> with `backend: vllm-cpp`, LocalAI reads the checkpoint's `config.json` and, if
> it declares an MTP head (`mtp_num_hidden_layers`), writes
> `speculative_config: {method: mtp}` into the generated `engine_args` for you.
> An explicit `speculative_config` in your own config is never overwritten.
> Importing a DFlash *draft* repository is refused with a warning: a drafter
> cannot serve on its own, so import the target model and point
> `speculative_config.model` at the draft.
#### External KV cache with LMCache
`kv_transfer_config:` takes vLLM's `--kv-transfer-config` JSON and selects an
external KV-cache connector. The `lm://` LMCache client lets prefill KV be
stored to and reloaded from a shared `lmcache.v1.server`, so a prefix computed
by one replica does not have to be recomputed by the next:
```yaml
engine_args:
kv_transfer_config:
kv_connector: LMCacheConnector
kv_role: kv_both # required whenever kv_connector is set
kv_connector_extra_config:
host: 127.0.0.1
port: 65432
```
`kv_role` is one of `kv_producer` (store only), `kv_consumer` (load only), or
`kv_both`. An unregistered connector name, a missing role, or a malformed
document fails the load with an explicit error rather than silently running
without the cache.
#### Legacy `options:` list
Earlier versions configured this backend through the flat `options:` list, and
those configs keep working. Every key in the table above is still read from
there in `key:value` form, and `engine_args` wins on any key set in both:
```yaml
options:
- max_num_seqs:32
- enable_prefix_caching:true
```
New configs should prefer `engine_args:`, which is the only place the nested
`speculative_config` / `kv_transfer_config` documents can be written naturally
rather than as a single-line JSON string.
### Transformers
[Transformers](https://huggingface.co/docs/transformers/index) is a State-of-the-art Machine Learning library for PyTorch, TensorFlow, and JAX.

View File

@@ -111,11 +111,6 @@ For a Podman-managed container, configure Podman to preserve and pass the
systemd socket file descriptor into the container. The LocalAI process inside
the container consumes the same activation protocol.
Activation needs both `LISTEN_PID` and `LISTEN_FDS`. If only one of them is set,
LocalAI ignores them and binds `--address` as usual. A container engine started
from a socket-activated system unit can leak a bare `LISTEN_PID` into every
container it spawns, and that is not an activation attempt.
## Next Steps
- [Try it out with examples](/basics/try/)

View File

@@ -1,3 +1,3 @@
{
"version": "v4.8.0"
"version": "v4.7.1"
}

View File

@@ -1,101 +1,4 @@
---
- &qwen3-5-9b-defiant-fable
name: "qwen3.5-9b-defiant-fable-mtp"
variants:
- model: qwen3.5-9b-defiant-fable
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/DavidAU/Qwen3.5-9B-The-Defiant-Fable-Uncensored-Heretic-NEO-IMATRIX-MAX-MTP-GGUF
description: |
Qwen3.5 9B Defiant Fable is an Apache-2.0 multimodal fine-tune for
reasoning, coding, creative writing, and roleplay. It retains the 256K
context window and vision support of Qwen3.5 while reducing refusals.
This default entry uses the NEO-imatrix Q4_K_M build with multi-token
prediction enabled for faster generation.
license: apache-2.0
icon: https://huggingface.co/DavidAU/Qwen3.5-9B-The-Defiant-Fable-Uncensored-Heretic-NEO-IMATRIX-MAX-MTP-GGUF/resolve/main/defiant-fable-9b.png
tags:
- llm
- gguf
- cpu
- gpu
- qwen3.5
- reasoning
- coding
- creative-writing
- uncensored
- vision
- multimodal
- mtp
last_checked: "2026-08-04"
overrides:
backend: llama-cpp
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
- vision
mmproj: llama-cpp/mmproj/qwen3.5-9b-defiant-fable/mmproj-BF16.gguf
options:
- use_jinja:true
- spec_type:draft-mtp
- spec_n_max:6
- spec_p_min:0.75
parameters:
model: llama-cpp/models/qwen3.5-9b-defiant-fable/Qwen3.5-9B-The-Defiant-Fable-Uncnr-Heretic-NEO-MAX-MTP-Q4_K_M.gguf
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/qwen3.5-9b-defiant-fable/Qwen3.5-9B-The-Defiant-Fable-Uncnr-Heretic-NEO-MAX-MTP-Q4_K_M.gguf
uri: huggingface://DavidAU/Qwen3.5-9B-The-Defiant-Fable-Uncensored-Heretic-NEO-IMATRIX-MAX-MTP-GGUF/Qwen3.5-9B-The-Defiant-Fable-Uncnr-Heretic-NEO-MAX-MTP-Q4_K_M.gguf
sha256: d7eb4fac9389d53fa576f64a6ff53e914a00bc7705dc354d1065887565147320
- filename: llama-cpp/mmproj/qwen3.5-9b-defiant-fable/mmproj-BF16.gguf
uri: huggingface://DavidAU/Qwen3.5-9B-The-Defiant-Fable-Uncensored-Heretic-NEO-IMATRIX-MAX-MTP-GGUF/mmproj-BF16.gguf
sha256: 853698ce7aa6c7ba732478bad280240969ddf7b0fcbf93900046f63903a83383
- !!merge <<: *qwen3-5-9b-defiant-fable
name: "qwen3.5-9b-defiant-fable"
variants: []
description: |
Qwen3.5 9B Defiant Fable in the plain NEO-imatrix Q4_K_M GGUF format.
This fallback offers the same multimodal reasoning, coding, and creative
capabilities without enabling multi-token prediction.
tags:
- llm
- gguf
- cpu
- gpu
- qwen3.5
- reasoning
- coding
- creative-writing
- uncensored
- vision
- multimodal
overrides:
backend: llama-cpp
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
- vision
mmproj: llama-cpp/mmproj/qwen3.5-9b-defiant-fable/mmproj-BF16.gguf
options:
- use_jinja:true
parameters:
model: llama-cpp/models/qwen3.5-9b-defiant-fable/Qwen3.5-9B-The-Defiant-Fable-Uncnr-Heretic-NEO-MAX-Q4_K_M.gguf
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/qwen3.5-9b-defiant-fable/Qwen3.5-9B-The-Defiant-Fable-Uncnr-Heretic-NEO-MAX-Q4_K_M.gguf
uri: huggingface://DavidAU/Qwen3.5-9B-The-Defiant-Fable-Uncensored-Heretic-NEO-IMATRIX-MAX-MTP-GGUF/Qwen3.5-9B-The-Defiant-Fable-Uncnr-Heretic-NEO-MAX-Q4_K_M.gguf
sha256: d33db5e583b9c9251402e876443791bc979f12af934bfb0630eadfb456279f84
- filename: llama-cpp/mmproj/qwen3.5-9b-defiant-fable/mmproj-BF16.gguf
uri: huggingface://DavidAU/Qwen3.5-9B-The-Defiant-Fable-Uncensored-Heretic-NEO-IMATRIX-MAX-MTP-GGUF/mmproj-BF16.gguf
sha256: 853698ce7aa6c7ba732478bad280240969ddf7b0fcbf93900046f63903a83383
- &nemotron-3-embed-1b
name: "nemotron-3-embed-1b-q4"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
@@ -2089,7 +1992,7 @@
files:
- filename: ds4flash.gguf
uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-GGUF
sha256: ea3dc48cb9797ea1bfaa8a74d8a819756b06b16e8fbaa30728ad2cd0a643c605
sha256: ba1d64ad8d77038124839956b614db2e889daa1a4ddc83060bb06ccb5a1d7461
- name: "qwopus3.6-35b-a3b-coder-mtp"
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
@@ -2904,86 +2807,6 @@
- filename: llama-cpp/mmproj/Qwopus3.6-27B-Coder-Compat-MTP-GGUF/mmproj-F32.gguf
sha256: 32f7ea0600c07272547da401d460f8abbd980f3a57b69d6df87be0e2505e0b9c
uri: https://huggingface.co/Jackrong/Qwopus3.6-27B-Coder-Compat-MTP-GGUF/resolve/main/mmproj-F32.gguf
- &qwen3-5-9b-hauhaucs-aggressive
name: "qwen3.5-9b-hauhaucs-aggressive"
variants:
- model: qwen3.5-9b-hauhaucs-aggressive-q8
url: "github:mudler/LocalAI/gallery/virtual.yaml@master"
urls:
- https://huggingface.co/Qwen/Qwen3.5-9B
- https://huggingface.co/HauhauCS/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive
description: |
Qwen3.5 9B Aggressive is HauhauCS's refusal-removed fine-tune of the
multimodal Qwen3.5 9B model. It retains the base model's reasoning, tool
use, image and video understanding, and 262K-token native context window.
This entry uses the balanced Q4_K_M GGUF quantization and includes the
matching BF16 multimodal projector. The Q8_0 variant offers higher fidelity.
license: "apache-2.0"
tags:
- llm
- gguf
- cpu
- gpu
- qwen
- multimodal
- uncensored
icon: https://qianwen-res.oss-cn-beijing.aliyuncs.com/logo_qwen.jpg
last_checked: "2026-08-04"
overrides:
backend: llama-cpp
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
mmproj: llama-cpp/mmproj/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q4_K_M/mmproj-Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-BF16.gguf
options:
- use_jinja:true
parameters:
model: llama-cpp/models/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q4_K_M/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q4_K_M.gguf
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q4_K_M/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q4_K_M.gguf
sha256: 2ca636d9e81d3d23ca9b60c234fe185d30ec082eeba69ce770fdb0c76559a4f5
uri: huggingface://HauhauCS/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q4_K_M.gguf
- filename: llama-cpp/mmproj/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q4_K_M/mmproj-Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-BF16.gguf
sha256: 05f662501f8bd45607b079723a3e238a4e888fd085a10a53f4057a0e250f6934
uri: huggingface://HauhauCS/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive/mmproj-Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-BF16.gguf
- !!merge <<: *qwen3-5-9b-hauhaucs-aggressive
name: "qwen3.5-9b-hauhaucs-aggressive-q8"
variants: []
description: |
Qwen3.5 9B Aggressive is HauhauCS's refusal-removed fine-tune of the
multimodal Qwen3.5 9B model. It retains the base model's reasoning, tool
use, image and video understanding, and 262K-token native context window.
This entry uses the higher-fidelity Q8_0 GGUF quantization and includes the
matching BF16 multimodal projector.
overrides:
backend: llama-cpp
function:
automatic_tool_parsing_fallback: true
grammar:
disable: true
known_usecases:
- chat
mmproj: llama-cpp/mmproj/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q8_0/mmproj-Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-BF16.gguf
options:
- use_jinja:true
parameters:
model: llama-cpp/models/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q8_0/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q8_0.gguf
template:
use_tokenizer_template: true
files:
- filename: llama-cpp/models/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q8_0/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q8_0.gguf
sha256: 99e7f2201c0046b05d2825e4d8be6a2efad2b87b071cd55d37bdd9fbe201a58b
uri: huggingface://HauhauCS/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q8_0.gguf
- filename: llama-cpp/mmproj/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-Q8_0/mmproj-Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-BF16.gguf
sha256: 05f662501f8bd45607b079723a3e238a4e888fd085a10a53f4057a0e250f6934
uri: huggingface://HauhauCS/Qwen3.5-9B-Uncensored-HauhauCS-Aggressive/mmproj-Qwen3.5-9B-Uncensored-HauhauCS-Aggressive-BF16.gguf
# DFlash speculative-decoding pairs (upstream llama.cpp `draft-dflash`).
# Each entry ships a full target model plus a small block-diffusion drafter
# (z-lab DFlash, converted with upstream convert_hf_to_gguf.py, GGUF arch

2
go.mod
View File

@@ -24,7 +24,7 @@ require (
github.com/gofrs/flock v0.13.0
github.com/google/go-containerregistry v0.21.6
github.com/google/uuid v1.6.0
github.com/gpustack/gguf-parser-go v0.25.0
github.com/gpustack/gguf-parser-go v0.24.0
github.com/hpcloud/tail v1.0.0
github.com/ipfs/go-log v1.0.5
github.com/jaypipes/ghw v0.24.0

4
go.sum
View File

@@ -666,8 +666,8 @@ github.com/gorilla/css v1.0.1/go.mod h1:BvnYkspnSzMmwRK+b8/xgNPLiIuNZr6vbZBTPQ2A
github.com/gorilla/websocket v1.4.2/go.mod h1:YR8l580nyteQvAITg2hZ9XVh4b55+EU/adAjf1fMHhE=
github.com/gorilla/websocket v1.5.4-0.20250319132907-e064f32e3674 h1:JeSE6pjso5THxAzdVpqr6/geYxZytqFMBCOtn/ujyeo=
github.com/gorilla/websocket v1.5.4-0.20250319132907-e064f32e3674/go.mod h1:r4w70xmWCQKmi1ONH4KIaBptdivuRPyosB9RmPlGEwA=
github.com/gpustack/gguf-parser-go v0.25.0 h1:1AMBhMKtI24nTtn588Bq53FqNiOvEw1x9Nb4HbRrThs=
github.com/gpustack/gguf-parser-go v0.25.0/go.mod h1:y4TwTtDqFWTK+xvprOjRUh+dowgU2TKCX37vRKvGiZ0=
github.com/gpustack/gguf-parser-go v0.24.0 h1:tdJceXYp9e5RhE9RwVYIuUpir72Jz2D68NEtDXkKCKc=
github.com/gpustack/gguf-parser-go v0.24.0/go.mod h1:y4TwTtDqFWTK+xvprOjRUh+dowgU2TKCX37vRKvGiZ0=
github.com/grpc-ecosystem/go-grpc-middleware v1.4.0 h1:UH//fgunKIs4JdUbpDl1VZCDaL56wXCB/5+wF6uHfaI=
github.com/grpc-ecosystem/go-grpc-middleware v1.4.0/go.mod h1:g5qyo/la0ALbONm6Vbp88Yd8NsDy6rZz+RcrMPxvld8=
github.com/grpc-ecosystem/grpc-gateway v1.16.0/go.mod h1:BDjrQk3hbvj6Nolgz8mAMFbcEtjT1g+wF4CSlocrBnw=

View File

@@ -2,7 +2,6 @@ package vram
import (
"context"
"fmt"
"strings"
gguf "github.com/gpustack/gguf-parser-go"
@@ -11,18 +10,7 @@ import (
type defaultGGUFReader struct{}
func (defaultGGUFReader) ReadMetadata(ctx context.Context, uri string) (meta *GGUFMeta, err error) {
// gguf-parser-go parses lengths supplied by the file and has historically
// panicked on values that cannot fit in a Go slice. Metadata can come from
// an untrusted remote host, and this reader is also used by a background
// gallery worker, where an escaped panic would terminate the whole server.
defer func() {
if recovered := recover(); recovered != nil {
meta = nil
err = fmt.Errorf("read GGUF metadata: parser panic: %v", recovered)
}
}()
func (defaultGGUFReader) ReadMetadata(ctx context.Context, uri string) (*GGUFMeta, error) {
u := downloader.URI(uri)
urlStr := u.ResolveURL()
@@ -40,10 +28,7 @@ func (defaultGGUFReader) ReadMetadata(ctx context.Context, uri string) (meta *GG
if !u.LooksLikeHTTPURL() {
return nil, nil
}
// The estimator only consumes architecture scalars. Tokenizer arrays can
// be very large and are unnecessary here, so avoid downloading or
// allocating them for remote files just as the local path does above.
f, err := gguf.ParseGGUFFileRemote(ctx, urlStr, gguf.SkipLargeMetadata())
f, err := gguf.ParseGGUFFileRemote(ctx, urlStr)
if err != nil {
return nil, err
}

View File

@@ -1,115 +0,0 @@
package vram_test
import (
"bytes"
"context"
"encoding/binary"
"math"
"net/http"
"net/http/httptest"
"time"
gguf "github.com/gpustack/gguf-parser-go"
"github.com/mudler/LocalAI/pkg/vram"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("DefaultGGUFReader", func() {
It("reads architecture scalars from a valid remote GGUF", func() {
server := serveGGUF(validRemoteGGUF())
meta, err := vram.DefaultGGUFReader().ReadMetadata(context.Background(), server.URL+"/model.gguf")
Expect(err).NotTo(HaveOccurred())
Expect(meta).To(Equal(&vram.GGUFMeta{
BlockCount: 32,
EmbeddingLength: 4096,
HeadCount: 32,
HeadCountKV: 8,
MaximumContextLength: 8192,
}))
})
It("rejects an overflowing tokenizer array without allocating it", func() {
server := serveGGUF(malformedGGUFArray(math.MaxUint64))
_, err := vram.DefaultGGUFReader().ReadMetadata(context.Background(), server.URL+"/model.gguf")
Expect(err).To(HaveOccurred())
Expect(err.Error()).NotTo(ContainSubstring("parser panic"),
"large tokenizer metadata should be skipped with a bounds error")
})
It("converts a parser panic from malformed string metadata to an error", func() {
server := serveGGUF(malformedGGUFString(uint64(math.MaxInt64)))
_, err := vram.DefaultGGUFReader().ReadMetadata(context.Background(), server.URL+"/model.gguf")
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(ContainSubstring("parser panic"))
})
})
func serveGGUF(payload []byte) *httptest.Server {
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
http.ServeContent(w, r, "model.gguf", time.Time{}, bytes.NewReader(payload))
}))
DeferCleanup(server.Close)
return server
}
func malformedGGUFString(length uint64) []byte {
payload := ggufHeader(1)
payload = appendGGUFString(payload, "general.name")
payload = binary.LittleEndian.AppendUint32(payload, uint32(gguf.GGUFMetadataValueTypeString))
payload = binary.LittleEndian.AppendUint64(payload, length)
return payload
}
func validRemoteGGUF() []byte {
payload := ggufHeader(6)
payload = appendGGUFStringValue(payload, "general.architecture", "llama")
payload = appendGGUFUint32(payload, "llama.block_count", 32)
payload = appendGGUFUint32(payload, "llama.embedding_length", 4096)
payload = appendGGUFUint32(payload, "llama.attention.head_count", 32)
payload = appendGGUFUint32(payload, "llama.attention.head_count_kv", 8)
payload = appendGGUFUint32(payload, "llama.context_length", 8192)
return payload
}
func malformedGGUFArray(itemLength uint64) []byte {
payload := ggufHeader(1)
payload = appendGGUFString(payload, "tokenizer.ggml.tokens")
payload = binary.LittleEndian.AppendUint32(payload, uint32(gguf.GGUFMetadataValueTypeArray))
payload = binary.LittleEndian.AppendUint32(payload, uint32(gguf.GGUFMetadataValueTypeString))
payload = binary.LittleEndian.AppendUint64(payload, 1)
payload = binary.LittleEndian.AppendUint64(payload, itemLength)
return payload
}
func ggufHeader(metadataCount uint64) []byte {
payload := make([]byte, 0, 128)
payload = binary.LittleEndian.AppendUint32(payload, uint32(gguf.GGUFMagicGGUFLe))
payload = binary.LittleEndian.AppendUint32(payload, uint32(gguf.GGUFVersionV3))
payload = binary.LittleEndian.AppendUint64(payload, 0)
payload = binary.LittleEndian.AppendUint64(payload, metadataCount)
return payload
}
func appendGGUFString(payload []byte, value string) []byte {
payload = binary.LittleEndian.AppendUint64(payload, uint64(len(value)))
return append(payload, value...)
}
func appendGGUFStringValue(payload []byte, key, value string) []byte {
payload = appendGGUFString(payload, key)
payload = binary.LittleEndian.AppendUint32(payload, uint32(gguf.GGUFMetadataValueTypeString))
return appendGGUFString(payload, value)
}
func appendGGUFUint32(payload []byte, key string, value uint32) []byte {
payload = appendGGUFString(payload, key)
payload = binary.LittleEndian.AppendUint32(payload, uint32(gguf.GGUFMetadataValueTypeUint32))
return binary.LittleEndian.AppendUint32(payload, value)
}