diff --git a/core/gallery/vllm_cpp_tags_test.go b/core/gallery/vllm_cpp_tags_test.go new file mode 100644 index 000000000..799dd471e --- /dev/null +++ b/core/gallery/vllm_cpp_tags_test.go @@ -0,0 +1,48 @@ +package gallery_test + +import ( + "fmt" + "slices" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + "github.com/mudler/LocalAI/core/config" +) + +// A gallery tag that names a capability is what users filter on, and +// known_usecases is what the server routes on. When they disagree, the entry +// is listed under a filter it cannot serve, or is hidden from one it can. +var _ = Describe("gallery/index.yaml vllm-cpp capability tags", func() { + It("keeps capability tags and known_usecases in agreement", func() { + entries, err := loadGalleryIndex() + Expect(err).ToNot(HaveOccurred()) + + tagToFlag := map[string]config.ModelConfigUsecase{ + "systemone": config.FLAG_SYSTEMONE, + "vision": config.FLAG_VISION, + "token-classify": config.FLAG_TOKEN_CLASSIFY, + "scoring": config.FLAG_SCORE, + } + + var violations []string + seen := 0 + for i := range entries { + e := &entries[i] + if backend, _ := e.Overrides["backend"].(string); backend != "vllm-cpp" { + continue + } + seen++ + declared := e.GetKnownUsecases() + for tag, flag := range tagToFlag { + tagged := slices.Contains(e.Tags, tag) + has := declared != nil && *declared&flag == flag + if tagged != has { + violations = append(violations, fmt.Sprintf("%s: tag %q present=%v but known_usecases declares it=%v", e.Name, tag, tagged, has)) + } + } + } + Expect(seen).To(BeNumerically(">", 0)) + Expect(violations).To(BeEmpty()) + }) +}) diff --git a/gallery/index.yaml b/gallery/index.yaml index 71c572e6b..10276000a 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -19360,6 +19360,7 @@ - qwen3.6 - nvfp4 - vllm-cpp + - vision - tool-calling - reasoning - gpu @@ -19372,6 +19373,7 @@ known_usecases: - chat - completion + - vision # Tool calls and the split are parsed by the engine's own streaming # parsers, so LocalAI's Go-side grammar path stays out of the way. function: @@ -19429,6 +19431,7 @@ - qwen3.6 - nvfp4 - vllm-cpp + - vision - speculative-decoding - mtp - tool-calling @@ -19442,6 +19445,7 @@ known_usecases: - chat - completion + - vision function: grammar: disable: true @@ -19497,6 +19501,7 @@ - qwen3.6 - nvfp4 - vllm-cpp + - vision - speculative-decoding - dflash - tool-calling @@ -19510,6 +19515,7 @@ known_usecases: - chat - completion + - vision function: grammar: disable: true @@ -19557,6 +19563,9 @@ with roughly 3B parameters active per token, so it reads like a much larger model while costing about as much per token as a small one. + Image input is implemented in the engine but is not token-gated against + vLLM yet, so the vision usecase on this entry is experimental. + This is the engine's gated MoE checkpoint: token-for-token identical to vLLM over the 315-prompt battery on both the synchronous and asynchronous paths, at 0.92x to 0.97x vLLM's throughput from concurrency 1 to 32. @@ -19575,6 +19584,8 @@ - moe - nvfp4 - vllm-cpp + - vision + - experimental - tool-calling - reasoning - gpu @@ -19587,6 +19598,7 @@ known_usecases: - chat - completion + - vision function: grammar: disable: true @@ -19615,6 +19627,9 @@ description: | Qwen3.6-35B-A3B NVFP4 on vllm.cpp with MTP speculative decoding enabled. + Image input is implemented in the engine but is not token-gated against + vLLM yet, so the vision usecase on this entry is experimental. + The draft head ships inside the checkpoint's own mtp.* tensors, so there is no second model to download. On this model the speculative path is token-exact against speculation-off on both the synchronous and asynchronous @@ -19632,6 +19647,8 @@ - moe - nvfp4 - vllm-cpp + - vision + - experimental - speculative-decoding - mtp - tool-calling @@ -19645,6 +19662,7 @@ known_usecases: - chat - completion + - vision function: grammar: disable: true @@ -63645,7 +63663,7 @@ overrides: backend: vllm-cpp known_usecases: - - chat + - systemone parameters: model: convaiinnovations/laya artifacts: @@ -63654,6 +63672,90 @@ source: type: huggingface repo: convaiinnovations/laya +- name: gliner25-decide-vllm-cpp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/fastino/GLiNER2.5-Decide + - https://github.com/mudler/vllm.cpp + description: | + GLiNER2.5-Decide is a DeBERTa-v3-large encoder with a classification head + that answers typed decision questions over a state text in one forward + pass. It never generates text, so there is nothing to parse. + + In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine runs the + decision pipeline (choice, noul and score question types) through the + vllm_decide C ABI. This is the decision model, not the zero-shot NER model: + use the gliner2.5 entry for entity extraction. F32 weights, about 2 GB. + The weights are pinned to a revision so the entry keeps serving the + checkpoint it was checked against. + license: apache-2.0 + tags: + - decision + - systemone + - vllm-cpp + - cpu + - gpu + size: 2GB + last_checked: "2026-09-30" + overrides: + backend: vllm-cpp + known_usecases: + - systemone + parameters: + model: fastino/GLiNER2.5-Decide + artifacts: + - name: model + target: model + source: + type: huggingface + repo: fastino/GLiNER2.5-Decide + revision: 5a7adf72a23b4d311abae6ce050d7f0012bb3416 +- name: qwen3-vl-4b-vllm-cpp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct + - https://github.com/mudler/vllm.cpp + description: | + Qwen3-VL-4B-Instruct on vllm.cpp, in bf16: a small vision-language model + that takes images alongside text. In the engine's correctness battery the + image path matches vLLM token for token, and video input is a near tie. + + Roughly 9 GB of weights plus KV cache at the context configured here. It + runs where the flagship NVFP4 checkpoints cannot, including plain CPU. + license: apache-2.0 + tags: + - llm + - vision + - multimodal + - qwen + - qwen3-vl + - vllm-cpp + - cpu + - gpu + size: 9GB + last_checked: "2026-09-30" + overrides: + backend: vllm-cpp + known_usecases: + - chat + - completion + - vision + template: + use_tokenizer_template: true + context_size: 8192 + engine_args: + block_size: 32 + num_blocks: 512 + max_num_seqs: 4 + parameters: + model: Qwen/Qwen3-VL-4B-Instruct + artifacts: + - name: model + target: model + source: + type: huggingface + repo: Qwen/Qwen3-VL-4B-Instruct + revision: ebb281ec70b05090aa6165b016eac8ec08e71b17 - name: cua-s1-forms-vllm-cpp url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: