diff --git a/backend/go/vllm-cpp/Makefile b/backend/go/vllm-cpp/Makefile index 56a32dbaf..32733f11d 100644 --- a/backend/go/vllm-cpp/Makefile +++ b/backend/go/vllm-cpp/Makefile @@ -11,7 +11,7 @@ JOBS?=$(shell nproc --ignore=1 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || e # vllm.cpp version VLLM_CPP_REPO?=https://github.com/mudler/vllm.cpp -VLLM_CPP_VERSION?=c3bebc357385990f721af66a3a6c69328dd4fc6c +VLLM_CPP_VERSION?=96788348627b6a079fcc3ef7fc6676a970965d7b # MLX GEMM provider (darwin/metal only; see the metal branch below for why). # Consumed as the prebuilt pip wheel: building MLX from source needs `xcrun @@ -40,6 +40,14 @@ MLX_ROOT=$(shell echo $(MLX_VENV)/lib/python*/site-packages/mlx) # server, examples and tests of the engine are never built here. CMAKE_ARGS+=-DVLLM_CPP_SERVER=OFF -DVLLM_CPP_BUILD_TESTS=OFF -DVLLM_CPP_BUILD_EXAMPLES=OFF CMAKE_ARGS+=-DCMAKE_BUILD_TYPE=Release +# Diarization (ABI v30) is ON upstream and FetchContents parakeet.cpp at +# configure time, building a second (static) ggml into libvllm. The pinned +# engine now pins that fetch to a commit, so ON configures, but it still adds a +# network fetch and a second ggml to every build for calls this backend never +# makes: it binds none of the vllm_diariz*/vllm_transcribe_and_* entry points, +# and with the option OFF they compile as stubs that refuse by name, so the ABI +# stays v30-complete without the extra dependency. +CMAKE_ARGS+=-DVLLM_CPP_WITH_DIARIZATION=OFF # vllm.cpp sets no global -march: SIMD tiers are per-file with runtime dispatch, # so ONE portable library serves every CPU of the target arch (unlike the diff --git a/backend/go/vllm-cpp/README.md b/backend/go/vllm-cpp/README.md index 49804ffb4..11427f926 100644 --- a/backend/go/vllm-cpp/README.md +++ b/backend/go/vllm-cpp/README.md @@ -9,7 +9,7 @@ It serves two things: text generation, and MiniMax-H3 joint video+audio generation. The backend dlopens the engine's stable C ABI (`libvllm`, `include/vllm.h`, -ABI v20) through purego: +ABI v30) through purego: - `Load` -> `vllm_engine_load`: accepts a `.gguf` file or a HF-style model directory (`config.json` + safetensors). `context_size` maps to @@ -44,6 +44,14 @@ the Makefile therefore means updating `abiVersion` plus the mirrors (and their offsets in `vllmcpp_test.go`) in the same change; `make abi-check` compares the pinned header against the bindings and the library build runs it first. +The Makefile builds libvllm with `-DVLLM_CPP_WITH_DIARIZATION=OFF`. vllm.cpp +turns that option ON by default since ABI v30, and ON fetches a pinned +parakeet.cpp (with its own ggml) at configure time. This backend does not bind +the diarization entry points, so OFF adds no dependency and changes nothing it +serves: those calls exist in libvllm but refuse with "not compiled in". If a +future change binds them, pin the parakeet.cpp source with +`-DVLLM_CPP_PARAKEET_CPP_DIR` and make `package.sh` bundle what it links. + Model config example: ```yaml @@ -56,6 +64,25 @@ options: - max_num_seqs:16 ``` +## hf_overrides + +`engine_args.hf_overrides` (vLLM parity) is a JSON object of top-level +`config.json` keys merged over the model directory's `config.json`. The C ABI +has no override input and the engine reads `config.json` from the directory it +is given, so `Load` builds an overlay (`hfoverrides.go`): a temp dir with the +merged `config.json` plus a symlink to every other entry of the model dir, and +passes the overlay as `model_path`. `validModelPath` and the DFlash draft +resolution still run against the real model dir. `Free` (and a failed load, or +the next `Load`) removes the overlay. A value that is not an object, a `.gguf` +model, or a dir without `config.json` fails the load instead of being ignored, +because loading the unmodified config would serve another architecture. + +```yaml +engine_args: + hf_overrides: + architectures: ["Tev1Model"] # opt a Qwen3.5-declared Tev1 snapshot into the Tev1 adapter +``` + ## MiniMax-H3 video+audio generation `GenerateVideo` -> `vllm_video_generate` (ABI v12). H3 renders picture and sound diff --git a/backend/go/vllm-cpp/backend.go b/backend/go/vllm-cpp/backend.go index 40e06fe8c..a2796b105 100644 --- a/backend/go/vllm-cpp/backend.go +++ b/backend/go/vllm-cpp/backend.go @@ -37,6 +37,10 @@ type VllmCpp struct { // other's checkpoints. Exactly one of the two is ever non-zero. videoEngine uintptr opts loadOptions + // overlayDir is the hf_overrides overlay handed to the engine in place of + // the model directory. It must outlive the engine handle (the engine may + // reopen files through it), so it is removed in Free, not after Load. + overlayDir string } // Stream registry: the per-request bridge between the C token callback and @@ -141,6 +145,23 @@ func (v *VllmCpp) Load(opts *pb.ModelOptions) error { } v.opts.speculativeConfig = resolvedSpec + // A reload reuses this struct: drop any overlay from the previous model + // before building a new one so it is not leaked. + if err := removeConfigOverlay(v.overlayDir); err != nil { + xlog.Warn("[vllm-cpp] stale overlay", "error", err) + } + v.overlayDir = "" + enginePath := model + if hasHFOverrides(v.opts.hfOverrides) { + overlay, err := newConfigOverlay(model, v.opts.hfOverrides) + if err != nil { + return err + } + v.overlayDir = overlay + enginePath = overlay + xlog.Info("[vllm-cpp] hf_overrides applied through overlay", "model", model, "overlay", overlay, "overrides", v.opts.hfOverrides) + } + mp := defaultModelParams() if v.opts.blockSize > 0 { mp.BlockSize = v.opts.blockSize @@ -173,7 +194,7 @@ func (v *VllmCpp) Load(opts *pb.ModelOptions) error { // Every string below is borrowed by C for the duration of the load call // only (the library copies what it keeps), so the backing slices just have // to outlive vllmEngineLoad - hence the single KeepAlive after it. - modelC := cString(model) + modelC := cString(enginePath) mp.ModelPath = uintptr(unsafe.Pointer(&modelC[0])) // #nosec G103 -- borrowed by C for the load call only keep := [][]byte{modelC} setStr := func(dst *uintptr, s string) { @@ -205,7 +226,12 @@ func (v *VllmCpp) Load(opts *pb.ModelOptions) error { rc := vllmEngineLoad(unsafe.Pointer(&mp), unsafe.Pointer(&engine)) // #nosec G103 -- POD out-params runtime.KeepAlive(keep) if rc != vllmOK { - return fmt.Errorf("vllm-cpp: engine load failed: %s", vllmLastError()) + loadErr := fmt.Errorf("vllm-cpp: engine load failed: %s", vllmLastError()) + if err := removeConfigOverlay(v.overlayDir); err != nil { + xlog.Warn("[vllm-cpp] overlay cleanup after failed load", "error", err) + } + v.overlayDir = "" + return loadErr } v.engine = engine return nil @@ -220,7 +246,10 @@ func (v *VllmCpp) Free() error { vllmVideoEngineFree(v.videoEngine) v.videoEngine = 0 } - return nil + // After the engine is gone, so nothing still reads through the links. + err := removeConfigOverlay(v.overlayDir) + v.overlayDir = "" + return err } // samplingFromPredict lowers PredictOptions into the C sampling POD plus the diff --git a/backend/go/vllm-cpp/govllmcpp.go b/backend/go/vllm-cpp/govllmcpp.go index f3723d096..71794ff31 100644 --- a/backend/go/vllm-cpp/govllmcpp.go +++ b/backend/go/vllm-cpp/govllmcpp.go @@ -1,6 +1,6 @@ package main -// purego bindings for the vllm.cpp stable C ABI (include/vllm.h, ABI v27). +// purego bindings for the vllm.cpp stable C ABI (include/vllm.h, ABI v30). // // The structs below are hand-mirrored PODs of the C declarations, with // explicit padding so the Go layout matches the C layout on linux/darwin @@ -21,7 +21,12 @@ import ( // the header of the VLLM_CPP_VERSION pinned in the Makefile: the build checks // the two against each other, because a mismatch is only caught at runtime by // registerLib, where it takes the backend down on every load (issue #11379). -const abiVersion = 29 +// +// v30 only ADDED the diarization and speaker-attributed-ASR entry points; every +// struct mirrored here is byte-identical to v29. They are not bound because the +// Makefile builds libvllm with VLLM_CPP_WITH_DIARIZATION=OFF, where they are +// stubs that refuse every call. +const abiVersion = 30 // The ABI's tri-state toggles (enable_prefix_caching ABI v7, // enable_jump_forward ABI v10) share one encoding: 0 is NOT "off", it is diff --git a/backend/go/vllm-cpp/hfoverrides.go b/backend/go/vllm-cpp/hfoverrides.go new file mode 100644 index 000000000..2d7e3abdf --- /dev/null +++ b/backend/go/vllm-cpp/hfoverrides.go @@ -0,0 +1,124 @@ +package main + +// hf_overrides, vLLM parity: a JSON object of config.json keys laid over the +// model directory's own config.json at load time. +// +// The engine reads config.json straight from the model directory and has no +// override input on the C ABI, so the only way to change what it sees without +// editing the snapshot is to hand it a different directory. The overlay built +// here is that directory: a private temp dir holding the merged config.json +// and a symlink for every other entry of the original. The snapshot itself is +// never written, which matters because it is a content-addressed download that +// a gallery reinstall or a hash check would otherwise flag or overwrite. +// +// The canonical use is opting a published checkpoint into an engine adapter +// its config does not name, e.g. {"architectures": ["Tev1Model"]} on a Tev1 +// snapshot that declares Qwen3_5ForConditionalGeneration. + +import ( + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "strings" +) + +// newConfigOverlay builds the overlay directory for modelDir with overrides +// merged over its config.json and returns its path. Keys are merged at the top +// level only: an override replaces the whole value of its key, nested objects +// included, which is what vLLM does for a plain (non sub-config) key. +// +// Bad input is refused rather than skipped, unlike an unknown engine_args key: +// hf_overrides exists to change which architecture loads, so silently loading +// the unmodified config would serve a different model than the one configured. +func newConfigOverlay(modelDir, overrides string) (dir string, err error) { + var patch map[string]any + if err := json.Unmarshal([]byte(overrides), &patch); err != nil || patch == nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides must be a JSON object of config.json keys, got %q", overrides) + } + + info, err := os.Stat(modelDir) + if err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: %w", err) + } + if !info.IsDir() { + return "", fmt.Errorf("vllm-cpp: hf_overrides needs a model directory with a config.json, %q is a file", modelDir) + } + absDir, err := filepath.Abs(modelDir) + if err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: %w", err) + } + + raw, err := os.ReadFile(filepath.Join(absDir, "config.json")) + if err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides needs %s: %w", filepath.Join(absDir, "config.json"), err) + } + var config map[string]any + if err := json.Unmarshal(raw, &config); err != nil || config == nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: %s is not a JSON object", filepath.Join(absDir, "config.json")) + } + for k, v := range patch { + config[k] = v + } + merged, err := json.MarshalIndent(config, "", " ") + if err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: encoding the merged config.json: %w", err) + } + + entries, err := os.ReadDir(absDir) + if err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: %w", err) + } + + dir, err = os.MkdirTemp("", "vllm-cpp-hf-overrides-*") + if err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: creating the overlay: %w", err) + } + defer func() { + if err != nil { + _ = os.RemoveAll(dir) + dir = "" + } + }() + + // Links point at the entry path, not at what it resolves to: an HF cache + // snapshot is itself a tree of links into blobs/, and the engine already + // follows those. + for _, e := range entries { + if e.Name() == "config.json" { + continue + } + if err := os.Symlink(filepath.Join(absDir, e.Name()), filepath.Join(dir, e.Name())); err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: linking %s into the overlay: %w", e.Name(), err) + } + } + if err := os.WriteFile(filepath.Join(dir, "config.json"), merged, 0o600); err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: writing the merged config.json: %w", err) + } + return dir, nil +} + +// removeConfigOverlay deletes an overlay built by newConfigOverlay. RemoveAll +// removes the symlinks themselves and never descends into their targets, so +// the original model directory is safe. +func removeConfigOverlay(dir string) error { + if dir == "" { + return nil + } + if err := os.RemoveAll(dir); err != nil && !errors.Is(err, os.ErrNotExist) { + return fmt.Errorf("vllm-cpp: removing the hf_overrides overlay %s: %w", dir, err) + } + return nil +} + +// hasHFOverrides reports whether an overlay is needed. An empty object is a +// no-op in vLLM too, so it loads the directory directly instead of paying for +// an overlay that changes nothing. +func hasHFOverrides(overrides string) bool { + switch strings.TrimSpace(overrides) { + case "", "{}", "null": + return false + } + return true +} diff --git a/backend/go/vllm-cpp/hfoverrides_test.go b/backend/go/vllm-cpp/hfoverrides_test.go new file mode 100644 index 000000000..cf3159f84 --- /dev/null +++ b/backend/go/vllm-cpp/hfoverrides_test.go @@ -0,0 +1,138 @@ +package main + +import ( + "encoding/json" + "os" + "path/filepath" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + pb "github.com/mudler/LocalAI/pkg/grpc/proto" +) + +var _ = Describe("hf_overrides", func() { + var modelDir string + const originalConfig = `{"architectures":["Qwen3_5ForConditionalGeneration"],"hidden_size":2560,"text_config":{"num_hidden_layers":36}}` + + BeforeEach(func() { + modelDir = GinkgoT().TempDir() + Expect(os.WriteFile(filepath.Join(modelDir, "config.json"), []byte(originalConfig), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelDir, "model.safetensors"), []byte("weights"), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelDir, "tokenizer.json"), []byte("{}"), 0o644)).To(Succeed()) + Expect(os.MkdirAll(filepath.Join(modelDir, "tokenizer"), 0o755)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelDir, "tokenizer", "vocab.json"), []byte("{}"), 0o644)).To(Succeed()) + }) + + Describe("parsing", func() { + It("reads a YAML-nested object from engine_args as a JSON document", func() { + lo := parseOptions(&pb.ModelOptions{ + EngineArgs: `{"hf_overrides":{"architectures":["Tev1Model"]},"max_num_seqs":2}`, + }) + Expect(lo.hfOverrides).To(MatchJSON(`{"architectures":["Tev1Model"]}`)) + Expect(lo.maxNumSeqs).To(Equal(int32(2))) + }) + + It("accepts a pre-encoded JSON string", func() { + lo := parseOptions(&pb.ModelOptions{ + EngineArgs: `{"hf_overrides":"{\"architectures\":[\"Tev1Model\"]}"}`, + }) + Expect(lo.hfOverrides).To(MatchJSON(`{"architectures":["Tev1Model"]}`)) + }) + }) + + Describe("config overlay", func() { + It("writes the merged config.json and symlinks every other entry to the original", func() { + overlay, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"],"new_key":1}`) + Expect(err).ToNot(HaveOccurred()) + DeferCleanup(os.RemoveAll, overlay) + Expect(overlay).ToNot(Equal(modelDir)) + + raw, err := os.ReadFile(filepath.Join(overlay, "config.json")) + Expect(err).ToNot(HaveOccurred()) + Expect(raw).To(MatchJSON(`{"architectures":["Tev1Model"],"hidden_size":2560,"text_config":{"num_hidden_layers":36},"new_key":1}`)) + info, err := os.Lstat(filepath.Join(overlay, "config.json")) + Expect(err).ToNot(HaveOccurred()) + Expect(info.Mode() & os.ModeSymlink).To(BeZero()) + + for _, name := range []string{"model.safetensors", "tokenizer.json", "tokenizer"} { + target, err := os.Readlink(filepath.Join(overlay, name)) + Expect(err).ToNot(HaveOccurred(), name) + Expect(target).To(Equal(filepath.Join(modelDir, name))) + } + // The subdir resolves through the link, so a tokenizer/ fallback + // in the engine still finds its files. + Expect(filepath.Join(overlay, "tokenizer", "vocab.json")).To(BeARegularFile()) + + entries, err := os.ReadDir(overlay) + Expect(err).ToNot(HaveOccurred()) + Expect(entries).To(HaveLen(4)) + }) + + It("leaves the original config.json untouched", func() { + overlay, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"]}`) + Expect(err).ToNot(HaveOccurred()) + DeferCleanup(os.RemoveAll, overlay) + + raw, err := os.ReadFile(filepath.Join(modelDir, "config.json")) + Expect(err).ToNot(HaveOccurred()) + Expect(string(raw)).To(Equal(originalConfig)) + }) + + It("is removed by Free", func() { + overlay, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"]}`) + Expect(err).ToNot(HaveOccurred()) + Expect(overlay).To(BeADirectory()) + + v := &VllmCpp{overlayDir: overlay} + Expect(v.Free()).To(Succeed()) + Expect(overlay).ToNot(BeAnExistingFile()) + Expect(v.overlayDir).To(BeEmpty()) + // The originals the links pointed at survive the cleanup. + Expect(filepath.Join(modelDir, "model.safetensors")).To(BeARegularFile()) + Expect(filepath.Join(modelDir, "tokenizer", "vocab.json")).To(BeARegularFile()) + }) + + DescribeTable("refuses bad input", + func(overrides string, useFile bool, substr string) { + target := modelDir + if useFile { + target = filepath.Join(modelDir, "model.safetensors") + } + overlay, err := newConfigOverlay(target, overrides) + Expect(err).To(MatchError(ContainSubstring(substr))) + Expect(overlay).To(BeEmpty()) + }, + Entry("a JSON array", `["Tev1Model"]`, false, "must be a JSON object"), + Entry("a JSON scalar", `5`, false, "must be a JSON object"), + Entry("unparseable JSON", `{"architectures":`, false, "must be a JSON object"), + Entry("a model that is not a directory", `{"architectures":["Tev1Model"]}`, true, "model directory"), + ) + + It("refuses a directory without config.json", func() { + Expect(os.Remove(filepath.Join(modelDir, "config.json"))).To(Succeed()) + _, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"]}`) + Expect(err).To(MatchError(ContainSubstring("config.json"))) + }) + + It("does not leave a half-built overlay behind on failure", func() { + Expect(os.WriteFile(filepath.Join(modelDir, "config.json"), []byte("not json"), 0o644)).To(Succeed()) + before, _ := filepath.Glob(filepath.Join(os.TempDir(), "vllm-cpp-hf-overrides-*")) + _, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"]}`) + Expect(err).To(HaveOccurred()) + after, _ := filepath.Glob(filepath.Join(os.TempDir(), "vllm-cpp-hf-overrides-*")) + Expect(after).To(ConsistOf(before)) + }) + }) + + It("keeps the merged document a valid object when overrides replace a nested key", func() { + overlay, err := newConfigOverlay(modelDir, `{"text_config":{"num_hidden_layers":2}}`) + Expect(err).ToNot(HaveOccurred()) + DeferCleanup(os.RemoveAll, overlay) + raw, err := os.ReadFile(filepath.Join(overlay, "config.json")) + Expect(err).ToNot(HaveOccurred()) + var doc map[string]any + Expect(json.Unmarshal(raw, &doc)).To(Succeed()) + Expect(doc["text_config"]).To(Equal(map[string]any{"num_hidden_layers": float64(2)})) + }) +}) diff --git a/backend/go/vllm-cpp/options.go b/backend/go/vllm-cpp/options.go index 1a36e62b3..0e95eb63a 100644 --- a/backend/go/vllm-cpp/options.go +++ b/backend/go/vllm-cpp/options.go @@ -74,6 +74,10 @@ type loadOptions struct { nerThreshold float32 // nerMaxWidth is the maximum span width in tokens (0 = engine default 12). nerMaxWidth int32 + // hf_overrides (vLLM parity): a JSON object of config.json keys merged + // over the model directory's config.json through a private overlay dir, + // see newConfigOverlay. Empty = load the directory as is. + hfOverrides string } // videoOptions is the MiniMax-H3 checkpoint SET plus its generation defaults. @@ -208,6 +212,8 @@ func applyOptionsList(lo *loadOptions, options []string) { lo.kvTransferConfig = strings.TrimSpace(v) case "tokenizer_config", "tokenizer_config_path": lo.tokenizerConfigPath = strings.TrimSpace(v) + case "hf_overrides": + lo.hfOverrides = strings.TrimSpace(v) case "enable_prefix_caching", "enable_radix_attention": if b, err := strconv.ParseBool(strings.TrimSpace(v)); err == nil { lo.enablePrefixCaching = boolTriState(b) @@ -343,6 +349,10 @@ func applyEngineArgs(lo *loadOptions, engineArgs string) { lo.speculativeConfig = jsonDocument(v, lo.speculativeConfig, k) case "kv_transfer_config": lo.kvTransferConfig = jsonDocument(v, lo.kvTransferConfig, k) + case "hf_overrides": + // Kept verbatim even when it is not an object: Load refuses a + // malformed value instead of loading the unmodified config. + lo.hfOverrides = jsonDocument(v, lo.hfOverrides, k) case "enable_prefix_caching", "enable_radix_attention": if b, ok := v.(bool); ok { lo.enablePrefixCaching = boolTriState(b) diff --git a/backend/go/vllm-cpp/vllmcpp_test.go b/backend/go/vllm-cpp/vllmcpp_test.go index 51ca81ace..1b6a3f6ea 100644 --- a/backend/go/vllm-cpp/vllmcpp_test.go +++ b/backend/go/vllm-cpp/vllmcpp_test.go @@ -16,7 +16,7 @@ func TestVllmCpp(t *testing.T) { RunSpecs(t, "vllm-cpp suite") } -// The Go POD mirrors must match the C struct layout of vllm.h (ABI v29) +// The Go POD mirrors must match the C struct layout of vllm.h (ABI v30) // byte-for-byte: these offsets are the C offsets on LP64 (linux/darwin // amd64+arm64). A failure here means govllmcpp.go drifted from vllm.h. var _ = Describe("C ABI struct mirrors", func() { @@ -24,7 +24,7 @@ var _ = Describe("C ABI struct mirrors", func() { // VLLM_ABI_VERSION in the vllm.h of VLLM_CPP_VERSION (Makefile). // Moving the pin past this without growing the mirrors below ships a // backend that refuses every load at startup (issue #11379). - Expect(abiVersion).To(Equal(29)) + Expect(abiVersion).To(Equal(30)) }) It("cModelParams matches vllm_model_params", func() { diff --git a/core/gallery/vllm_cpp_tags_test.go b/core/gallery/vllm_cpp_tags_test.go index b84f9a2b1..7c028d8a8 100644 --- a/core/gallery/vllm_cpp_tags_test.go +++ b/core/gallery/vllm_cpp_tags_test.go @@ -2,10 +2,13 @@ package gallery_test import ( "fmt" + "os" + "path/filepath" "slices" . "github.com/onsi/ginkgo/v2" . "github.com/onsi/gomega" + "gopkg.in/yaml.v3" "github.com/mudler/LocalAI/core/config" ) @@ -46,3 +49,25 @@ var _ = Describe("gallery/index.yaml vllm-cpp capability tags", func() { Expect(violations).To(BeEmpty()) }) }) + +// artifacts: is a model-config key, so the installer only sees it inside +// overrides:. At the top level of an entry it is silently dropped, the +// installed config keeps a bare HF repo id as its model, and a backend that +// does not infer artifacts (vllm-cpp among them) fails the first load with +// "model path not found" while the install itself reported success. +var _ = Describe("gallery/index.yaml artifacts placement", func() { + It("declares artifacts under overrides, never at the entry top level", func() { + data, err := os.ReadFile(filepath.Join("..", "..", "gallery", "index.yaml")) + Expect(err).ToNot(HaveOccurred()) + var raw []map[string]any + Expect(yaml.Unmarshal(data, &raw)).To(Succeed()) + + var misplaced []string + for _, e := range raw { + if _, ok := e["artifacts"]; ok { + misplaced = append(misplaced, fmt.Sprint(e["name"])) + } + } + Expect(misplaced).To(BeEmpty()) + }) +}) diff --git a/docs/content/features/decisions.md b/docs/content/features/decisions.md index 0118f110d..2893d47dc 100644 --- a/docs/content/features/decisions.md +++ b/docs/content/features/decisions.md @@ -105,13 +105,22 @@ Install one from the gallery and filter on the `decisions` tag: |---|---|---| | `laya-vllm-cpp` | Laya | ModernBERT-large, non-autoregressive, about 800 MB | | `gliner25-decide-vllm-cpp` | GLiNER2.5-Decide | DeBERTa-v3-large with a classification head, about 2 GB | +| `tev1-4b-vllm-cpp` | Tev1 4B | Autoregressive Qwen3.5-4B fine-tune that answers with an option letter, about 9.3 GB | +| `tev1-0.8b-vllm-cpp` | Tev1 0.8B | Autoregressive Qwen3.5-0.8B fine-tune that answers with an option letter, about 1.8 GB | The engine, [vllm.cpp]({{% relref "features/vllm-cpp" %}}), also supports the kev, CLM and xor decision models. Those checkpoints need a conversion step, so they are not gallery entries yet. -Tev1 is an autoregressive decision model. It answers through chat completions -and does not serve `/v1/systemone` yet. +Tev1 is an autoregressive decision model. The engine answers each question by +scoring the option letters, so its `confidence` is the entropy measure Ollama +uses. A Tev1 `choice` or `score` question accepts at most 24 options (Ollama +allows 26), because the model is trained on the letters A to X, and every +option needs a nonempty description. The published checkpoints name another +architecture in `config.json`, so the Tev1 gallery entries set +`engine_args.hf_overrides` to load them as `Tev1Model` (see +[Overriding config.json keys]({{% relref "features/vllm-cpp" %}}#overriding-configjson-keys-hf_overrides)). +The same model also answers `/v1/chat/completions` requests. ## Request limits @@ -127,8 +136,8 @@ A request is refused with `400` (or `413` for the body size) when: A `noul` question may carry `criteria` with a description for each outcome, for example `{"false": "No refund is requested", "true": "The customer requests a refund"}`. -Some models cap the number of options for a `choice` or `score` question (models -that answer with a letter accept at most 26). The engine refuses more options than +Some models cap the number of options for a `choice` or `score` question. Models +that answer with a letter accept at most 26, and Tev1 accepts at most 24. The engine refuses more options than the model supports and the error names the limit. ## Compatibility with Ollama diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md index 19dcf64b0..2b4fad656 100644 --- a/docs/content/features/text-generation.md +++ b/docs/content/features/text-generation.md @@ -1093,6 +1093,7 @@ engine_args: | `tokenizer_config` | Override the `tokenizer_config.json` the chat template is read from | `/tokenizer_config.json` | | `speculative_config` | Speculative decoding (see below) | disabled | | `kv_transfer_config` | External KV connector / LMCache (see below) | none | +| `hf_overrides` | JSON object of `config.json` keys merged over the model directory's own, as vLLM's `--hf-overrides` (see the [vllm.cpp backend page]({{% relref "features/vllm-cpp" %}}#overriding-configjson-keys-hf_overrides)) | none | Raising `max_num_batched_tokens` lets more prefill land in a single step, at the cost of decode latency for requests queued behind it. The default deliberately diff --git a/docs/content/features/vllm-cpp.md b/docs/content/features/vllm-cpp.md index bf7323e0d..1b7389f1f 100644 --- a/docs/content/features/vllm-cpp.md +++ b/docs/content/features/vllm-cpp.md @@ -133,6 +133,39 @@ engine_args: tool_parser: qwen3_coder ``` +## Overriding config.json keys (`hf_overrides`) + +`engine_args.hf_overrides` is a JSON object of top-level `config.json` keys that +the backend merges over the model directory's own `config.json` before the +engine loads it, like vLLM's `--hf-overrides`. The main use is to opt a published +checkpoint into an engine adapter that its config does not name. For example, +the Tev1 repositories declare `Qwen3_5ForConditionalGeneration`, and vllm.cpp +serves them as decision models only when the architecture is `Tev1Model`: + +```yaml +engine_args: + hf_overrides: + architectures: ["Tev1Model"] +``` + +The downloaded model files do not change. At load the backend creates a private +temporary directory. It writes the merged `config.json` there and adds a +symlink for each other entry of the model directory (weights, tokenizer files, +a `tokenizer/` subdirectory). Then it gives that directory to the engine. When +the model unloads, the backend removes the directory. + +Rules: + +- The merge is top-level only. An override key replaces the whole value of that + key, including a nested object such as `text_config`. +- The model must be a directory that contains a `config.json`. A `.gguf` file + or a directory without `config.json` fails the load. +- A value that is not a JSON object (an array, a scalar, or JSON that does not + parse) fails the load. The backend does not ignore it, because loading the + unchanged config would serve a different architecture than the one you + configured. +- An empty object (`{}`) does nothing. + ## Named entity recognition (GLiNER2.5) The `vllm-cpp` backend serves [GLiNER2.5](https://huggingface.co/fastino/gliner2.5-multi-v1), diff --git a/gallery/index.yaml b/gallery/index.yaml index 197b5f306..104585e01 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -63666,12 +63666,12 @@ - decisions parameters: model: convaiinnovations/laya - artifacts: - - name: model - target: model - source: - type: huggingface - repo: convaiinnovations/laya + artifacts: + - name: model + target: model + source: + type: huggingface + repo: convaiinnovations/laya - name: gliner25-decide-vllm-cpp url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: @@ -63703,13 +63703,124 @@ - decisions parameters: model: fastino/GLiNER2.5-Decide - artifacts: - - name: model - target: model - source: - type: huggingface - repo: fastino/GLiNER2.5-Decide - revision: 5a7adf72a23b4d311abae6ce050d7f0012bb3416 + artifacts: + - name: model + target: model + source: + type: huggingface + repo: fastino/GLiNER2.5-Decide + revision: 5a7adf72a23b4d311abae6ce050d7f0012bb3416 +- name: tev1-4b-vllm-cpp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/togethercomputer/Tev1-4B-experimental + - https://github.com/mudler/vllm.cpp + description: | + Tev1-4B-experimental is an experimental decision model from Together + AI: a supervised fine-tune of Qwen3.5-4B that picks one option letter + for a state, a question and 2 to 24 labeled options. It keeps the standard + next-token head, so it is autoregressive, unlike Laya or GLiNER2.5-Decide. + + In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine scores the + answer letters of each choice, noul and score question through the + vllm_decide C ABI and returns probabilities with an entropy confidence, as + Ollama does for tev1. A choice or score question accepts at most 24 options + (Ollama allows 26) and every option needs a nonempty description. The + published config.json names Qwen3_5ForConditionalGeneration, so this entry + sets hf_overrides to load it as Tev1Model without editing the download. + + Checked against transformers BF16 on CPU over seven + questions: 7/7 answers equal, largest probability difference 0.0004. The decision route is verified on CPU only; GPU serving has not + been measured. BF16 weights, about 9.3GB, pinned to a revision. The + fine-tune license is still being finalized by Together AI (base model + Apache-2.0). + tags: + - decisions + - systemone + - vllm-cpp + - cpu + - gpu + size: 9.3GB + last_checked: "2026-09-30" + overrides: + backend: vllm-cpp + known_usecases: + - decisions + template: + use_tokenizer_template: true + context_size: 2048 + engine_args: + hf_overrides: + architectures: + - Tev1Model + block_size: 32 + num_blocks: 256 + max_num_seqs: 4 + parameters: + model: togethercomputer/Tev1-4B-experimental + artifacts: + - name: model + target: model + source: + type: huggingface + repo: togethercomputer/Tev1-4B-experimental + revision: 0b7becf017daa0e5eb222f8ce7483c8c8259c52f +- name: tev1-0.8b-vllm-cpp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/togethercomputer/Tev1-0.8B-experimental + - https://github.com/mudler/vllm.cpp + description: | + Tev1-0.8B-experimental is an experimental decision model from Together + AI: a supervised fine-tune of Qwen3.5-0.8B that picks one option letter + for a state, a question and 2 to 24 labeled options. It keeps the standard + next-token head, so it is autoregressive, unlike Laya or GLiNER2.5-Decide. + + In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine scores the + answer letters of each choice, noul and score question through the + vllm_decide C ABI and returns probabilities with an entropy confidence, as + Ollama does for tev1. A choice or score question accepts at most 24 options + (Ollama allows 26) and every option needs a nonempty description. The + published config.json names Qwen3_5ForConditionalGeneration, so this entry + sets hf_overrides to load it as Tev1Model without editing the download. + + Checked against transformers BF16 on CPU over seven + questions: 6/7 answers equal, the miss a near tie (0.453 against 0.514 + in transformers, 0.4845 each here), largest probability difference 0.031. The decision route is verified on CPU only; GPU serving has not + been measured. BF16 weights, about 1.8GB, pinned to a revision. The + fine-tune license is still being finalized by Together AI (base model + Apache-2.0). + tags: + - decisions + - systemone + - vllm-cpp + - cpu + - gpu + size: 1.8GB + last_checked: "2026-09-30" + overrides: + backend: vllm-cpp + known_usecases: + - decisions + template: + use_tokenizer_template: true + context_size: 2048 + engine_args: + hf_overrides: + architectures: + - Tev1Model + block_size: 32 + num_blocks: 256 + max_num_seqs: 4 + parameters: + model: togethercomputer/Tev1-0.8B-experimental + artifacts: + - name: model + target: model + source: + type: huggingface + repo: togethercomputer/Tev1-0.8B-experimental + revision: 6bb2dff14b38fea90ddb14d870166ccaf77374e9 - name: qwen3-vl-4b-vllm-cpp url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: @@ -63749,13 +63860,13 @@ max_num_seqs: 4 parameters: model: Qwen/Qwen3-VL-4B-Instruct - artifacts: - - name: model - target: model - source: - type: huggingface - repo: Qwen/Qwen3-VL-4B-Instruct - revision: ebb281ec70b05090aa6165b016eac8ec08e71b17 + artifacts: + - name: model + target: model + source: + type: huggingface + repo: Qwen/Qwen3-VL-4B-Instruct + revision: ebb281ec70b05090aa6165b016eac8ec08e71b17 - name: cua-s1-forms-vllm-cpp url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: @@ -63784,12 +63895,12 @@ - score parameters: model: cua-ai/cua-s1-forms - artifacts: - - name: model - target: model - source: - type: huggingface - repo: cua-ai/cua-s1-forms + artifacts: + - name: model + target: model + source: + type: huggingface + repo: cua-ai/cua-s1-forms - name: gliner2.5-vllm-cpp url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: @@ -63821,12 +63932,12 @@ - token_classify parameters: model: fastino/gliner2.5-multi-v1 - artifacts: - - name: model - target: model - source: - type: huggingface - repo: fastino/gliner2.5-multi-v1 + artifacts: + - name: model + target: model + source: + type: huggingface + repo: fastino/gliner2.5-multi-v1 - name: nemo-speech-cpp-sortformer-diarization-v2 url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: