mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-12 22:33:54 -04:00
The backend could configure four of the engine's knobs (block size, KV block
count, max sequence length, max concurrent sequences) out of a config surface
that is considerably larger. Speculative decoding, prefix caching, the
chunked-prefill token budget, the scheduling policy and the external KV
connector were reachable from vllm.cpp's own HTTP server and from nothing
LocalAI could write in a model config.
Config now goes through `engine_args:`, the same map the vLLM and SGLang
backends take, with keys spelled as vLLM's own CLI flags so a speculative_config
or kv_transfer_config block written for vLLM works verbatim. The legacy
`options:` list keeps working and reads every key too; engine_args wins where
both set one. Unknown keys are logged and ignored rather than fatal: the field
is shared with the other engines, so a config carrying their knobs must not take
the model down.
Two details worth knowing:
`enable_prefix_caching: false` maps to the ABI tri-state force-OFF (2), not 0.
0 means "let the model capability decide" and dense architectures default the
cache on, so collapsing the two would silently enable it against an explicit
false. enable_jump_forward (ABI v10) shares the encoding, deferring to
VT_ENABLE_JUMP_FORWARD instead of to the model.
The importer probes config.json on a vllm-cpp import and writes
speculative_config: {method: mtp} when the checkpoint declares an MTP head, the
safetensors analogue of the llama-cpp importer's GGUF probe. DFlash draft repos
are refused with a warning instead, since a drafter cannot serve alone and the
pairing is not derivable from either repo. The draft path is resolved against
LocalAI's model directory, because the engine only looks in a directory holding
config.json or in the HF cache and never downloads: the repo-id spelling the
vLLM docs teach used to die deep in the load with "draft checkpoint not found".
docs/content/features/text-generation.md gains a vllm.cpp section covering the
engine_args table, all three speculative methods, LMCache and the legacy list.
The backend had no documentation page before.
This replaces a branch that had gone stale behind master and carried its own
route to ABI v10, which #11386 has since landed in minimal form. Rebased onto
that as a single commit rather than replaying the intermediate steps, whose
ABI v9 mirrors no longer make sense against master's pin. The Darwin build
fixes for Apple Clang's gnu-folding-constant diagnostic on C++, Objective-C and
Objective-C++, originally authored by localai-org-maint-bot, are folded in here.
Verified: `make abi-check` agrees at v10; unit specs, core/config and
core/gallery/importers green; and the full e2e passes in 1330s against a CPU
libvllm.so reporting ABI v10 with Qwen_Qwen3.5-0.8B-Q4_K_M.gguf (load, blocking
completion, streaming, chat and tool calls).
Assisted-by: Claude:claude-fable-5 golangci-lint
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
118 lines
3.6 KiB
Go
118 lines
3.6 KiB
Go
package config_test
|
|
|
|
import (
|
|
. "github.com/mudler/LocalAI/core/config"
|
|
|
|
. "github.com/onsi/ginkgo/v2"
|
|
. "github.com/onsi/gomega"
|
|
)
|
|
|
|
var _ = Describe("vllm-cpp speculative-decoding auto-defaults", func() {
|
|
Context("HasSafetensorsMTPHead", func() {
|
|
It("detects a top-level mtp_num_hidden_layers", func() {
|
|
n, ok := HasSafetensorsMTPHead([]byte(`{
|
|
"model_type": "qwen3_5_moe",
|
|
"mtp_num_hidden_layers": 1
|
|
}`))
|
|
Expect(ok).To(BeTrue())
|
|
Expect(n).To(Equal(uint32(1)))
|
|
})
|
|
|
|
It("detects the head nested under text_config", func() {
|
|
// Multimodal checkpoints nest the language-model config, which is
|
|
// where the MTP depth lives (mirrors the engine's own resolution
|
|
// off config.raw text_config).
|
|
n, ok := HasSafetensorsMTPHead([]byte(`{
|
|
"model_type": "qwen3_5_moe",
|
|
"text_config": {"mtp_num_hidden_layers": 2}
|
|
}`))
|
|
Expect(ok).To(BeTrue())
|
|
Expect(n).To(Equal(uint32(2)))
|
|
})
|
|
|
|
It("reports no head when the key is absent", func() {
|
|
n, ok := HasSafetensorsMTPHead([]byte(`{"model_type": "llama"}`))
|
|
Expect(ok).To(BeFalse())
|
|
Expect(n).To(BeZero())
|
|
})
|
|
|
|
It("reports no head for a zero depth", func() {
|
|
_, ok := HasSafetensorsMTPHead([]byte(`{"mtp_num_hidden_layers": 0}`))
|
|
Expect(ok).To(BeFalse())
|
|
})
|
|
|
|
It("ignores a DFlash draft checkpoint", func() {
|
|
// A DFlash draft is a SEPARATE checkpoint that cannot serve alone:
|
|
// it needs a target to verify against. Same exclusion the GGUF path
|
|
// makes for gemma4-assistant drafts.
|
|
_, ok := HasSafetensorsMTPHead([]byte(`{
|
|
"model_type": "qwen3_dflash",
|
|
"mtp_num_hidden_layers": 1,
|
|
"dflash_config": {"mask_token_id": 151666, "target_layer_ids": [0, 1]}
|
|
}`))
|
|
Expect(ok).To(BeFalse())
|
|
})
|
|
|
|
It("reports no head on unparseable JSON", func() {
|
|
_, ok := HasSafetensorsMTPHead([]byte(`{not json`))
|
|
Expect(ok).To(BeFalse())
|
|
})
|
|
|
|
It("reports no head on empty input", func() {
|
|
_, ok := HasSafetensorsMTPHead(nil)
|
|
Expect(ok).To(BeFalse())
|
|
})
|
|
})
|
|
|
|
Context("IsDFlashDraftConfig", func() {
|
|
It("recognises a draft by its dflash_config block", func() {
|
|
Expect(IsDFlashDraftConfig([]byte(`{
|
|
"dflash_config": {"mask_token_id": 151666, "target_layer_ids": [0]}
|
|
}`))).To(BeTrue())
|
|
})
|
|
|
|
It("does not flag an ordinary checkpoint", func() {
|
|
Expect(IsDFlashDraftConfig([]byte(`{"model_type": "qwen3_5_moe"}`))).To(BeFalse())
|
|
})
|
|
})
|
|
|
|
Context("ApplyVLLMSpeculativeDefaults", func() {
|
|
It("writes the mtp method into engine_args", func() {
|
|
cfg := &ModelConfig{Name: "qwen"}
|
|
ApplyVLLMSpeculativeDefaults(cfg, 1)
|
|
Expect(cfg.EngineArgs).To(HaveKey("speculative_config"))
|
|
spec, ok := cfg.EngineArgs["speculative_config"].(map[string]any)
|
|
Expect(ok).To(BeTrue())
|
|
Expect(spec["method"]).To(Equal("mtp"))
|
|
})
|
|
|
|
It("leaves an existing speculative_config alone", func() {
|
|
cfg := &ModelConfig{
|
|
Name: "qwen",
|
|
LLMConfig: LLMConfig{
|
|
EngineArgs: map[string]any{
|
|
"speculative_config": map[string]any{"method": "ngram", "num_speculative_tokens": 4},
|
|
},
|
|
},
|
|
}
|
|
ApplyVLLMSpeculativeDefaults(cfg, 1)
|
|
spec := cfg.EngineArgs["speculative_config"].(map[string]any)
|
|
Expect(spec["method"]).To(Equal("ngram"))
|
|
})
|
|
|
|
It("preserves unrelated engine_args keys", func() {
|
|
cfg := &ModelConfig{
|
|
Name: "qwen",
|
|
LLMConfig: LLMConfig{EngineArgs: map[string]any{"max_num_seqs": 32}},
|
|
}
|
|
ApplyVLLMSpeculativeDefaults(cfg, 1)
|
|
Expect(cfg.EngineArgs).To(HaveKeyWithValue("max_num_seqs", 32))
|
|
Expect(cfg.EngineArgs).To(HaveKey("speculative_config"))
|
|
})
|
|
|
|
It("tolerates a nil config", func() {
|
|
Expect(func() { ApplyVLLMSpeculativeDefaults(nil, 1) }).ToNot(Panic())
|
|
})
|
|
})
|
|
})
|