mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-14 23:28:23 -04:00
The backend could configure four of the engine's knobs (block size, KV block
count, max sequence length, max concurrent sequences) out of a config surface
that is considerably larger. Speculative decoding, prefix caching, the
chunked-prefill token budget, the scheduling policy and the external KV
connector were reachable from vllm.cpp's own HTTP server and from nothing
LocalAI could write in a model config.
Config now goes through `engine_args:`, the same map the vLLM and SGLang
backends take, with keys spelled as vLLM's own CLI flags so a speculative_config
or kv_transfer_config block written for vLLM works verbatim. The legacy
`options:` list keeps working and reads every key too; engine_args wins where
both set one. Unknown keys are logged and ignored rather than fatal: the field
is shared with the other engines, so a config carrying their knobs must not take
the model down.
Two details worth knowing:
`enable_prefix_caching: false` maps to the ABI tri-state force-OFF (2), not 0.
0 means "let the model capability decide" and dense architectures default the
cache on, so collapsing the two would silently enable it against an explicit
false. enable_jump_forward (ABI v10) shares the encoding, deferring to
VT_ENABLE_JUMP_FORWARD instead of to the model.
The importer probes config.json on a vllm-cpp import and writes
speculative_config: {method: mtp} when the checkpoint declares an MTP head, the
safetensors analogue of the llama-cpp importer's GGUF probe. DFlash draft repos
are refused with a warning instead, since a drafter cannot serve alone and the
pairing is not derivable from either repo. The draft path is resolved against
LocalAI's model directory, because the engine only looks in a directory holding
config.json or in the HF cache and never downloads: the repo-id spelling the
vLLM docs teach used to die deep in the load with "draft checkpoint not found".
docs/content/features/text-generation.md gains a vllm.cpp section covering the
engine_args table, all three speculative methods, LMCache and the legacy list.
The backend had no documentation page before.
This replaces a branch that had gone stale behind master and carried its own
route to ABI v10, which #11386 has since landed in minimal form. Rebased onto
that as a single commit rather than replaying the intermediate steps, whose
ABI v9 mirrors no longer make sense against master's pin. The Darwin build
fixes for Apple Clang's gnu-folding-constant diagnostic on C++, Objective-C and
Objective-C++, originally authored by localai-org-maint-bot, are folded in here.
Verified: `make abi-check` agrees at v10; unit specs, core/config and
core/gallery/importers green; and the full e2e passes in 1330s against a CPU
libvllm.so reporting ABI v10 with Qwen_Qwen3.5-0.8B-Q4_K_M.gguf (load, blocking
completion, streaming, chat and tool calls).
Assisted-by: Claude:claude-fable-5 golangci-lint
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
118 lines
4.7 KiB
Go
118 lines
4.7 KiB
Go
package config
|
|
|
|
// Speculative-decoding auto-defaults for the vllm-cpp backend, the safetensors
|
|
// counterpart of the GGUF/llama.cpp hook in mtp.go.
|
|
//
|
|
// The two engines detect and spell the same feature differently. llama.cpp
|
|
// reads `<arch>.nextn_predict_layers` out of the GGUF header and takes
|
|
// `spec_type:draft-mtp` in `options:`; vllm.cpp reads `mtp_num_hidden_layers`
|
|
// out of the checkpoint's config.json and takes vLLM's own
|
|
// `--speculative-config` JSON, which LocalAI carries in `engine_args`. The
|
|
// engine resolves the draft depth and the default k itself, so the config only
|
|
// has to name the method.
|
|
|
|
import (
|
|
"encoding/json"
|
|
|
|
"github.com/mudler/xlog"
|
|
)
|
|
|
|
// hfSpecConfig is the subset of a HuggingFace config.json that decides whether
|
|
// speculative decoding can be auto-enabled.
|
|
type hfSpecConfig struct {
|
|
ModelType string `json:"model_type"`
|
|
// MtpNumHiddenLayers is the MTP head depth (upstream speculative.py reads
|
|
// it as n_predict for the qwen3_5 / qwen3_5_moe families).
|
|
MtpNumHiddenLayers uint32 `json:"mtp_num_hidden_layers"`
|
|
// DFlashConfig marks a z-lab DFlash DRAFT checkpoint (mask_token_id +
|
|
// target_layer_ids). Its presence means this repo is a draft, not a
|
|
// servable target.
|
|
DFlashConfig json.RawMessage `json:"dflash_config"`
|
|
// TextConfig is where multimodal checkpoints nest the language-model
|
|
// config, and therefore the MTP depth.
|
|
TextConfig *hfSpecConfig `json:"text_config"`
|
|
}
|
|
|
|
// parseHFSpecConfig decodes the speculative-relevant subset of a config.json.
|
|
// A document that does not parse yields nothing rather than an error: detection
|
|
// is best-effort and must never break an import.
|
|
func parseHFSpecConfig(configJSON []byte) (hfSpecConfig, bool) {
|
|
if len(configJSON) == 0 {
|
|
return hfSpecConfig{}, false
|
|
}
|
|
var c hfSpecConfig
|
|
if err := json.Unmarshal(configJSON, &c); err != nil {
|
|
xlog.Debug("[vllm-spec] config.json did not parse; skipping detection", "error", err)
|
|
return hfSpecConfig{}, false
|
|
}
|
|
return c, true
|
|
}
|
|
|
|
// IsDFlashDraftConfig reports whether a HuggingFace config.json describes a
|
|
// DFlash DRAFT checkpoint. Unlike MTP - whose head ships inside the target
|
|
// checkpoint's `mtp.*` tensors - a DFlash draft is its own repo that can only
|
|
// run paired with a target it verifies against, so it must never be configured
|
|
// as a standalone model.
|
|
func IsDFlashDraftConfig(configJSON []byte) bool {
|
|
c, ok := parseHFSpecConfig(configJSON)
|
|
if !ok {
|
|
return false
|
|
}
|
|
return len(c.DFlashConfig) > 0 ||
|
|
(c.TextConfig != nil && len(c.TextConfig.DFlashConfig) > 0)
|
|
}
|
|
|
|
// HasSafetensorsMTPHead reports whether a HuggingFace config.json declares a
|
|
// self-speculating Multi-Token Prediction head, returning its depth. The depth
|
|
// is informational: vllm.cpp resolves n_predict and the default
|
|
// num_speculative_tokens from the checkpoint itself.
|
|
//
|
|
// DFlash drafts are excluded for the same reason `gemma4-assistant` GGUFs are
|
|
// excluded from the llama.cpp hook: they carry head metadata but cannot
|
|
// self-speculate.
|
|
//
|
|
// NOTE this is a safetensors-only signal. vllm.cpp rejects an MTP config over a
|
|
// GGUF source, because the `mtp.*` draft tensors only exist in the safetensors
|
|
// checkpoint - so the GGUF import path must not use this.
|
|
func HasSafetensorsMTPHead(configJSON []byte) (uint32, bool) {
|
|
c, ok := parseHFSpecConfig(configJSON)
|
|
if !ok {
|
|
return 0, false
|
|
}
|
|
if IsDFlashDraftConfig(configJSON) {
|
|
return 0, false
|
|
}
|
|
n := c.MtpNumHiddenLayers
|
|
if n == 0 && c.TextConfig != nil {
|
|
n = c.TextConfig.MtpNumHiddenLayers
|
|
}
|
|
return n, n > 0
|
|
}
|
|
|
|
// ApplyVLLMSpeculativeDefaults enables MTP speculative decoding in cfg's
|
|
// engine_args when nothing is configured there yet. It is a no-op when the user
|
|
// already set a speculative_config, so an explicit choice (a different method,
|
|
// an explicit k, a DFlash draft) is never clobbered.
|
|
//
|
|
// `layers` is the detected head depth and is only used for the diagnostic log
|
|
// line - the engine derives the real k from the checkpoint.
|
|
func ApplyVLLMSpeculativeDefaults(cfg *ModelConfig, layers uint32) {
|
|
if cfg == nil {
|
|
return
|
|
}
|
|
if _, set := cfg.EngineArgs["speculative_config"]; set {
|
|
xlog.Debug("[vllm-spec] MTP head detected but speculative_config already configured; leaving user choice intact",
|
|
"name", cfg.Name, "mtp_num_hidden_layers", layers)
|
|
return
|
|
}
|
|
if cfg.EngineArgs == nil {
|
|
cfg.EngineArgs = map[string]any{}
|
|
}
|
|
// Only the method: vllm.cpp defaults num_speculative_tokens to the
|
|
// checkpoint's own n_predict (speculative.py:865-875), which is the right
|
|
// value far more reliably than anything guessable here.
|
|
cfg.EngineArgs["speculative_config"] = map[string]any{"method": "mtp"}
|
|
xlog.Info("[vllm-spec] MTP head detected; enabling mtp speculative decoding",
|
|
"name", cfg.Name, "mtp_num_hidden_layers", layers)
|
|
}
|