mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-03 11:34:35 -04:00
feat(gallery): tag vllm-cpp entries by capability and add decision and vision models
laya declares the systemone usecase instead of chat. The gated Qwen3.6 27B NVFP4 entries gain vision; the 35B-A3B entries gain it as experimental because image input is not token-gated. Adds GLiNER2.5-Decide and Qwen3-VL-4B. A guard test keeps capability tags and known_usecases in agreement for every vllm-cpp entry. Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
This commit is contained in:
1 parent
b4852d62d2
commit
c2af8d56ea
2 files changed
+151
-1
No files matched your search
@@ -0,0 +1,48 @@
|
||||
package gallery_test
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"slices"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
)
|
||||
|
||||
// A gallery tag that names a capability is what users filter on, and
|
||||
// known_usecases is what the server routes on. When they disagree, the entry
|
||||
// is listed under a filter it cannot serve, or is hidden from one it can.
|
||||
var _ = Describe("gallery/index.yaml vllm-cpp capability tags", func() {
|
||||
It("keeps capability tags and known_usecases in agreement", func() {
|
||||
entries, err := loadGalleryIndex()
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
|
||||
tagToFlag := map[string]config.ModelConfigUsecase{
|
||||
"systemone": config.FLAG_SYSTEMONE,
|
||||
"vision": config.FLAG_VISION,
|
||||
"token-classify": config.FLAG_TOKEN_CLASSIFY,
|
||||
"scoring": config.FLAG_SCORE,
|
||||
}
|
||||
|
||||
var violations []string
|
||||
seen := 0
|
||||
for i := range entries {
|
||||
e := &entries[i]
|
||||
if backend, _ := e.Overrides["backend"].(string); backend != "vllm-cpp" {
|
||||
continue
|
||||
}
|
||||
seen++
|
||||
declared := e.GetKnownUsecases()
|
||||
for tag, flag := range tagToFlag {
|
||||
tagged := slices.Contains(e.Tags, tag)
|
||||
has := declared != nil && *declared&flag == flag
|
||||
if tagged != has {
|
||||
violations = append(violations, fmt.Sprintf("%s: tag %q present=%v but known_usecases declares it=%v", e.Name, tag, tagged, has))
|
||||
}
|
||||
}
|
||||
}
|
||||
Expect(seen).To(BeNumerically(">", 0))
|
||||
Expect(violations).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
+103
-1
@@ -19360,6 +19360,7 @@
|
||||
- qwen3.6
|
||||
- nvfp4
|
||||
- vllm-cpp
|
||||
- vision
|
||||
- tool-calling
|
||||
- reasoning
|
||||
- gpu
|
||||
@@ -19372,6 +19373,7 @@
|
||||
known_usecases:
|
||||
- chat
|
||||
- completion
|
||||
- vision
|
||||
# Tool calls and the <think> split are parsed by the engine's own streaming
|
||||
# parsers, so LocalAI's Go-side grammar path stays out of the way.
|
||||
function:
|
||||
@@ -19429,6 +19431,7 @@
|
||||
- qwen3.6
|
||||
- nvfp4
|
||||
- vllm-cpp
|
||||
- vision
|
||||
- speculative-decoding
|
||||
- mtp
|
||||
- tool-calling
|
||||
@@ -19442,6 +19445,7 @@
|
||||
known_usecases:
|
||||
- chat
|
||||
- completion
|
||||
- vision
|
||||
function:
|
||||
grammar:
|
||||
disable: true
|
||||
@@ -19497,6 +19501,7 @@
|
||||
- qwen3.6
|
||||
- nvfp4
|
||||
- vllm-cpp
|
||||
- vision
|
||||
- speculative-decoding
|
||||
- dflash
|
||||
- tool-calling
|
||||
@@ -19510,6 +19515,7 @@
|
||||
known_usecases:
|
||||
- chat
|
||||
- completion
|
||||
- vision
|
||||
function:
|
||||
grammar:
|
||||
disable: true
|
||||
@@ -19557,6 +19563,9 @@
|
||||
with roughly 3B parameters active per token, so it reads like a much larger
|
||||
model while costing about as much per token as a small one.
|
||||
|
||||
Image input is implemented in the engine but is not token-gated against
|
||||
vLLM yet, so the vision usecase on this entry is experimental.
|
||||
|
||||
This is the engine's gated MoE checkpoint: token-for-token identical to vLLM
|
||||
over the 315-prompt battery on both the synchronous and asynchronous paths,
|
||||
at 0.92x to 0.97x vLLM's throughput from concurrency 1 to 32.
|
||||
@@ -19575,6 +19584,8 @@
|
||||
- moe
|
||||
- nvfp4
|
||||
- vllm-cpp
|
||||
- vision
|
||||
- experimental
|
||||
- tool-calling
|
||||
- reasoning
|
||||
- gpu
|
||||
@@ -19587,6 +19598,7 @@
|
||||
known_usecases:
|
||||
- chat
|
||||
- completion
|
||||
- vision
|
||||
function:
|
||||
grammar:
|
||||
disable: true
|
||||
@@ -19615,6 +19627,9 @@
|
||||
description: |
|
||||
Qwen3.6-35B-A3B NVFP4 on vllm.cpp with MTP speculative decoding enabled.
|
||||
|
||||
Image input is implemented in the engine but is not token-gated against
|
||||
vLLM yet, so the vision usecase on this entry is experimental.
|
||||
|
||||
The draft head ships inside the checkpoint's own mtp.* tensors, so there is
|
||||
no second model to download. On this model the speculative path is
|
||||
token-exact against speculation-off on both the synchronous and asynchronous
|
||||
@@ -19632,6 +19647,8 @@
|
||||
- moe
|
||||
- nvfp4
|
||||
- vllm-cpp
|
||||
- vision
|
||||
- experimental
|
||||
- speculative-decoding
|
||||
- mtp
|
||||
- tool-calling
|
||||
@@ -19645,6 +19662,7 @@
|
||||
known_usecases:
|
||||
- chat
|
||||
- completion
|
||||
- vision
|
||||
function:
|
||||
grammar:
|
||||
disable: true
|
||||
@@ -63645,7 +63663,7 @@
|
||||
overrides:
|
||||
backend: vllm-cpp
|
||||
known_usecases:
|
||||
- chat
|
||||
- systemone
|
||||
parameters:
|
||||
model: convaiinnovations/laya
|
||||
artifacts:
|
||||
@@ -63654,6 +63672,90 @@
|
||||
source:
|
||||
type: huggingface
|
||||
repo: convaiinnovations/laya
|
||||
- name: gliner25-decide-vllm-cpp
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
- https://huggingface.co/fastino/GLiNER2.5-Decide
|
||||
- https://github.com/mudler/vllm.cpp
|
||||
description: |
|
||||
GLiNER2.5-Decide is a DeBERTa-v3-large encoder with a classification head
|
||||
that answers typed decision questions over a state text in one forward
|
||||
pass. It never generates text, so there is nothing to parse.
|
||||
|
||||
In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine runs the
|
||||
decision pipeline (choice, noul and score question types) through the
|
||||
vllm_decide C ABI. This is the decision model, not the zero-shot NER model:
|
||||
use the gliner2.5 entry for entity extraction. F32 weights, about 2 GB.
|
||||
The weights are pinned to a revision so the entry keeps serving the
|
||||
checkpoint it was checked against.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- decision
|
||||
- systemone
|
||||
- vllm-cpp
|
||||
- cpu
|
||||
- gpu
|
||||
size: 2GB
|
||||
last_checked: "2026-09-30"
|
||||
overrides:
|
||||
backend: vllm-cpp
|
||||
known_usecases:
|
||||
- systemone
|
||||
parameters:
|
||||
model: fastino/GLiNER2.5-Decide
|
||||
artifacts:
|
||||
- name: model
|
||||
target: model
|
||||
source:
|
||||
type: huggingface
|
||||
repo: fastino/GLiNER2.5-Decide
|
||||
revision: 5a7adf72a23b4d311abae6ce050d7f0012bb3416
|
||||
- name: qwen3-vl-4b-vllm-cpp
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
- https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct
|
||||
- https://github.com/mudler/vllm.cpp
|
||||
description: |
|
||||
Qwen3-VL-4B-Instruct on vllm.cpp, in bf16: a small vision-language model
|
||||
that takes images alongside text. In the engine's correctness battery the
|
||||
image path matches vLLM token for token, and video input is a near tie.
|
||||
|
||||
Roughly 9 GB of weights plus KV cache at the context configured here. It
|
||||
runs where the flagship NVFP4 checkpoints cannot, including plain CPU.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- llm
|
||||
- vision
|
||||
- multimodal
|
||||
- qwen
|
||||
- qwen3-vl
|
||||
- vllm-cpp
|
||||
- cpu
|
||||
- gpu
|
||||
size: 9GB
|
||||
last_checked: "2026-09-30"
|
||||
overrides:
|
||||
backend: vllm-cpp
|
||||
known_usecases:
|
||||
- chat
|
||||
- completion
|
||||
- vision
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
context_size: 8192
|
||||
engine_args:
|
||||
block_size: 32
|
||||
num_blocks: 512
|
||||
max_num_seqs: 4
|
||||
parameters:
|
||||
model: Qwen/Qwen3-VL-4B-Instruct
|
||||
artifacts:
|
||||
- name: model
|
||||
target: model
|
||||
source:
|
||||
type: huggingface
|
||||
repo: Qwen/Qwen3-VL-4B-Instruct
|
||||
revision: ebb281ec70b05090aa6165b016eac8ec08e71b17
|
||||
- name: cua-s1-forms-vllm-cpp
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
|
||||
Reference in new issue
Block a user