Merge pull request #12373 from mudler/feat/systemone-capability

feat: decisions usecase for decision models, with gallery tagging
This commit is contained in:
Ettore Di Giacinto committed 2026-09-30 16:29:58 +00:00
commit f378fe89d0
28 files changed
+846 -39

No files matched your search

+8 -2
View File
@@ -35,6 +35,7 @@ const (
UsecaseSpeakerRecognition = "speaker_recognition"
UsecaseTokenClassify = "token_classify"
UsecaseScore = "score"
UsecaseDecisions = "decisions"
)
// GRPCMethod identifies a Backend service RPC from backend.proto.
@@ -216,6 +217,11 @@ var UsecaseInfoMap = map[string]UsecaseInfo{
GRPCMethod: MethodScore,
Description: "Joint log-probability scoring of candidate continuations via the Score RPC. Declared explicitly via known_usecases and usable alongside generation usecases.",
},
UsecaseDecisions: {
Flag: FLAG_DECISIONS,
GRPCMethod: MethodScore,
Description: "Decision models (served by POST /v1/systemone): typed choice, noul and score questions over a state text, answered by a non-generative decision model through the Score RPC (question_type systemone). Declared explicitly via known_usecases.",
},
}
// BackendCapability describes which gRPC methods and usecases a backend supports.
@@ -349,10 +355,10 @@ var BackendCapabilities = map[string]BackendCapability{
// model returns an error rather than silent garbage.
"vllm-cpp": {
GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateVideo, MethodTokenClassify, MethodScore},
PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVideo, UsecaseTokenClassify, UsecaseScore},
PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVision, UsecaseVideo, UsecaseTokenClassify, UsecaseScore, UsecaseDecisions},
DefaultUsecases: []string{UsecaseChat},
AcceptsImages: true,
Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and kev/laya decision pipelines",
Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and decision models (kev, laya, CLM, GLiNER2.5-Decide, xor, nimble)",
},
"vllm-omni": {
GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateImage, MethodGenerateVideo, MethodTTS},
+2 -2
View File
@@ -16,14 +16,14 @@ import (
// reservedNonChatModel reports whether the operator reserved this model for an
// internal primitive — the router score classifier or the PII NER
// token_classify tier. Such a model has no chat template and must not be
// token_classify tier, or a decision head. Such a model has no chat template and must not be
// given the generative-chat defaults the GGUF importer otherwise applies
// (FLAG_CHAT, jinja templating): surfacing it in chat pickers defeats the
// reservation. Operators who do want a combined model declare both usecases
// explicitly — the combination is valid.
func reservedNonChatModel(cfg *ModelConfig) bool {
return cfg.KnownUsecases != nil &&
(*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY)) != 0
(*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY|FLAG_DECISIONS)) != 0
}
// genAudioEncoderKey is the mmproj metadata flag llama.cpp's mtmd writes for a
+20 -4
View File
@@ -2056,6 +2056,13 @@ const (
FLAG_3D ModelConfigUsecase = 0b100000000000000000000000
FLAG_3D_ANIMATION ModelConfigUsecase = 1 << 24
// Marks a model as a decision model: it answers typed choice / noul /
// score questions over a state (served by POST /v1/systemone).
// Explicit only, like FLAG_SCORE: a decision model never generates
// text, so guessing chat or embeddings for it would surface it in
// pickers it cannot serve.
FLAG_DECISIONS ModelConfigUsecase = 1 << 25
// Common Subsets
FLAG_LLM ModelConfigUsecase = FLAG_CHAT | FLAG_COMPLETION | FLAG_EDIT
)
@@ -2118,6 +2125,7 @@ func GetAllModelConfigUsecases() map[string]ModelConfigUsecase {
"FLAG_TOKEN_CLASSIFY": FLAG_TOKEN_CLASSIFY,
"FLAG_3D": FLAG_3D,
"FLAG_3D_ANIMATION": FLAG_3D_ANIMATION,
"FLAG_DECISIONS": FLAG_DECISIONS,
}
}
@@ -2146,9 +2154,9 @@ func GetUsecasesFromYAML(input []string) *ModelConfigUsecase {
//
// Declared known_usecases are normally additive — the guessing heuristic
// still adds whatever it can infer from backend/templates. The exceptions
// are FLAG_SCORE and FLAG_TOKEN_CLASSIFY: when the operator declared
// either, they reserved the model for an internal direct-decode primitive
// (the router classifier, or the PII NER tier). Letting GuessUsecases
// are FLAG_SCORE, FLAG_TOKEN_CLASSIFY and FLAG_DECISIONS: when the operator
// declared any of them, they reserved the model for a direct-decode primitive
// (the router classifier, the PII NER tier, or a decision head). Letting GuessUsecases
// paint chat/completion/embeddings on top would surface it in pickers it
// was deliberately kept out of. So a declared score or token_classify
// list is authoritative; declare the generation usecases explicitly
@@ -2158,7 +2166,7 @@ func (c *ModelConfig) HasUsecases(u ModelConfigUsecase) bool {
if (u & *c.KnownUsecases) == u {
return true
}
if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY)) != 0 {
if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY | FLAG_DECISIONS)) != 0 {
return false
}
}
@@ -2381,6 +2389,14 @@ func (c *ModelConfig) GuessUsecases(u ModelConfigUsecase) bool {
return false
}
if (u & FLAG_DECISIONS) == FLAG_DECISIONS {
// No heuristic: decisions intent is a deliberate operator choice
// (the model is a non-generative decision head), so
// HasUsecases(FLAG_DECISIONS) is true only when KnownUsecases
// declares it explicitly.
return false
}
return true
}
+33
View File
@@ -955,3 +955,36 @@ var _ = Describe("ModelConfig alias", func() {
Expect(err).To(MatchError(ContainSubstring("alias")))
})
})
var _ = Describe("decisions usecase", func() {
// A decision model never generates text, so a declared decisions list
// must stay authoritative and the heuristic must never guess the flag.
It("is authoritative when declared and never guessed", func() {
declared := GetUsecasesFromYAML([]string{"decisions"})
Expect(declared).NotTo(BeNil())
Expect(*declared).NotTo(Equal(FLAG_ANY))
cfg := ModelConfig{
Name: "laya",
Backend: "vllm-cpp",
KnownUsecases: declared,
TemplateConfig: TemplateConfig{
Chat: "inherited from chatml",
ChatMessage: "inherited from chatml",
Completion: "inherited from chatml",
},
}
Expect(cfg.HasUsecases(*declared)).To(BeTrue())
Expect(cfg.HasUsecases(FLAG_CHAT)).To(BeFalse())
Expect(cfg.HasUsecases(FLAG_COMPLETION)).To(BeFalse())
Expect(cfg.HasUsecases(FLAG_EMBEDDINGS)).To(BeFalse())
undeclared := ModelConfig{Name: "laya", Backend: "vllm-cpp"}
Expect(undeclared.HasUsecases(*declared)).To(BeFalse())
})
It("is a reserved usecase for the GGUF importer chat-default guard", func() {
declared := GetUsecasesFromYAML([]string{"decisions"})
Expect(reservedNonChatModel(&ModelConfig{Backend: "vllm-cpp", KnownUsecases: declared})).To(BeTrue())
})
})
+48
View File
@@ -0,0 +1,48 @@
package gallery_test
import (
"fmt"
"slices"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"github.com/mudler/LocalAI/core/config"
)
// A gallery tag that names a capability is what users filter on, and
// known_usecases is what the server routes on. When they disagree, the entry
// is listed under a filter it cannot serve, or is hidden from one it can.
var _ = Describe("gallery/index.yaml vllm-cpp capability tags", func() {
It("keeps capability tags and known_usecases in agreement", func() {
entries, err := loadGalleryIndex()
Expect(err).ToNot(HaveOccurred())
tagToFlag := map[string]config.ModelConfigUsecase{
"decisions": config.FLAG_DECISIONS,
"vision": config.FLAG_VISION,
"token-classify": config.FLAG_TOKEN_CLASSIFY,
"scoring": config.FLAG_SCORE,
}
var violations []string
seen := 0
for i := range entries {
e := &entries[i]
if backend, _ := e.Overrides["backend"].(string); backend != "vllm-cpp" {
continue
}
seen++
declared := e.GetKnownUsecases()
for tag, flag := range tagToFlag {
tagged := slices.Contains(e.Tags, tag)
has := declared != nil && *declared&flag == flag
if tagged != has {
violations = append(violations, fmt.Sprintf("%s: tag %q present=%v but known_usecases declares it=%v", e.Name, tag, tagged, has))
}
}
}
Expect(seen).To(BeNumerically(">", 0))
Expect(violations).To(BeEmpty())
})
})
+6
View File
@@ -71,6 +71,11 @@ var RouteFeatureRegistry = []RouteFeature{
// Detection
{"POST", "/v1/detection", FeatureDetection},
// Decisions API (SystemOne wire contract)
{"POST", "/v1/systemone", FeatureDecisions},
{"POST", "/v1/systemone/permute", FeatureDecisions},
{"POST", "/v1/systemone/separate", FeatureDecisions},
// Face recognition
{"POST", "/v1/face/verify", FeatureFaceRecognition},
{"POST", "/v1/face/analyze", FeatureFaceRecognition},
@@ -209,5 +214,6 @@ func APIFeatureMetas() []FeatureMeta {
{FeatureVoiceRecognition, "Voice Recognition", true},
{FeatureAudioTransform, "Audio Transform", true},
{FeaturePIIFilter, "PII Analyze / Redact", true},
{FeatureDecisions, "Decisions", true},
}
}
+24
View File
@@ -0,0 +1,24 @@
package auth_test
import (
. "github.com/mudler/LocalAI/core/http/auth"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("Decisions feature registration", func() {
It("gates the three decision routes behind one default-on API feature", func() {
Expect(APIFeatures).To(ContainElement(FeatureDecisions))
patterns := []string{}
for _, route := range RouteFeatureRegistry {
if route.Feature == FeatureDecisions {
Expect(route.Method).To(Equal("POST"))
patterns = append(patterns, route.Pattern)
}
}
Expect(patterns).To(ConsistOf("/v1/systemone", "/v1/systemone/permute", "/v1/systemone/separate"))
Expect(APIFeatureMetas()).To(ContainElement(FeatureMeta{Key: FeatureDecisions, Label: "Decisions", DefaultValue: true}))
})
})
+2 -1
View File
@@ -59,6 +59,7 @@ const (
FeatureFaceRecognition = "face_recognition"
FeatureVoiceRecognition = "voice_recognition"
FeatureAudioTransform = "audio_transform"
FeatureDecisions = "decisions"
// FeaturePIIFilter gates the synchronous PII analyze/redact service
// (POST /api/pii/{analyze,redact}). Default ON like the other API
// features; the admin-only events log is gated separately in-handler.
@@ -78,7 +79,7 @@ var APIFeatures = []string{
FeatureVAD, FeatureDetection, FeatureVideo, Feature3D, FeatureEmbeddings, FeatureSound,
FeatureRealtime, FeatureModeration, FeatureRerank, FeatureTokenize, FeatureMCP, FeatureStores,
FeatureFaceRecognition, FeatureVoiceRecognition, FeatureAudioTransform,
FeaturePIIFilter,
FeaturePIIFilter, FeatureDecisions,
}
// AllFeatures lists all known features (used by UI and validation).
@@ -105,6 +105,12 @@ var instructionDefs = []instructionDef{
Tags: []string{"voice-recognition"},
Intro: "Voice (speaker) recognition — the audio analog to /v1/face/*. Use /v1/voice/verify for 1:1 speaker comparison, /v1/voice/identify for 1:N match against the registered store, /v1/voice/{register,forget} to manage that store, /v1/voice/embed for a raw speaker-encoder vector, and /v1/voice/analyze for age / gender / emotion inferred from speech. Registrations are in-memory by default and lost on restart. Audio inputs accept URL, base64, or data-URI; /v1/embeddings remains text-only.",
},
{
Name: "decisions",
Description: "Typed decisions (choice, noul, score) over a state text with calibrated confidence",
Tags: []string{"systemone"},
Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { <id>: { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [decisions] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. Field names and question types follow Ollama's /v1/systemone, with differences in confidence, error shape and keep_alive (see the Decisions API docs). A request over 64 KiB, with more than 64 questions, or with a malformed question is refused.",
},
{
Name: "branding",
Description: "Whitelabel the instance: configure name, tagline, logo, and favicon",
@@ -39,7 +39,7 @@ var _ = Describe("API Instructions Endpoints", func() {
instructions, ok := resp["instructions"].([]any)
Expect(ok).To(BeTrue())
Expect(instructions).To(HaveLen(20))
Expect(instructions).To(HaveLen(21))
// Verify each instruction has required fields and correct URL format
for _, s := range instructions {
@@ -82,6 +82,7 @@ var _ = Describe("API Instructions Endpoints", func() {
"voice-library",
"3d",
"failover",
"decisions",
))
})
})
@@ -136,6 +137,17 @@ var _ = Describe("API Instructions Endpoints", func() {
Expect(string(body)).NotTo(ContainSubstring("/v1/3d/generations"))
})
It("should advertise the Decisions API", func() {
req := httptest.NewRequest(http.MethodGet, "/api/instructions/decisions", nil)
rec := httptest.NewRecorder()
app.ServeHTTP(rec, req)
Expect(rec.Code).To(Equal(http.StatusOK))
body, _ := io.ReadAll(rec.Body)
Expect(string(body)).To(ContainSubstring("POST /v1/systemone"))
Expect(string(body)).To(ContainSubstring("known_usecases: [decisions]"))
})
It("should return JSON fragment when format=json", func() {
req := httptest.NewRequest(http.MethodGet, "/api/instructions/chat-inference?format=json", nil)
rec := httptest.NewRecorder()
+217 -7
View File
@@ -2,6 +2,7 @@ package localai
import (
"encoding/json"
"errors"
"fmt"
"math"
"math/rand"
@@ -371,6 +372,191 @@ func systemOneError(c echo.Context, status int, msg string) error {
})
}
// systemOneModelAllowed keeps chat and embedding models out of the decision
// API with an actionable error instead of a backend failure. A config that
// declares no usecases predates the flag and stays allowed, and a
// token_classify model is allowed because the NER path serves it.
func systemOneModelAllowed(cfg config.ModelConfig) error {
if cfg.KnownUsecases == nil {
return nil
}
if *cfg.KnownUsecases&(config.FLAG_DECISIONS|config.FLAG_TOKEN_CLASSIFY) != 0 {
return nil
}
return fmt.Errorf("model %q does not declare the decisions usecase (known_usecases: [decisions])", cfg.Name)
}
// checkSystemOneModel applies systemOneModelAllowed to a model looked up by
// name. An unknown model passes here so the existing not-found handling
// downstream keeps its status code.
func checkSystemOneModel(app *application.Application, modelName string) error {
cl := app.ModelConfigLoader()
if cl == nil {
return nil
}
cfg, ok := cl.GetModelConfig(modelName)
if !ok {
return nil
}
return systemOneModelAllowed(cfg)
}
// systemOneUsesDecisionPipeline reports whether /v1/systemone forwards the
// request to the backend's Score RPC (the decision pipeline) for this model.
// A model that declares token_classify without systemone is a zero-shot NER
// model: the backend's decision entry point refuses those architectures, so it
// goes to the NER path instead. A config that declares nothing keeps the
// decision pipeline, which is what setups that predate the decisions usecase
// relied on.
func systemOneUsesDecisionPipeline(cfg config.ModelConfig) bool {
if !backendSupportsScore(cfg.Backend) {
return false
}
if cfg.KnownUsecases == nil {
return true
}
declared := *cfg.KnownUsecases
if declared&config.FLAG_DECISIONS != 0 {
return true
}
return declared&config.FLAG_TOKEN_CLASSIFY == 0
}
// systemOneNERAllowed guards /permute and /separate, which always run the NER
// path. A decision model cannot serve them: the backend's NER entry point
// refuses its architecture, and the caller would see a backend error.
func systemOneNERAllowed(cfg config.ModelConfig) error {
if cfg.KnownUsecases == nil {
return nil
}
declared := *cfg.KnownUsecases
if declared&config.FLAG_DECISIONS != 0 && declared&config.FLAG_TOKEN_CLASSIFY == 0 {
return fmt.Errorf("model %q is a decision model: /permute and /separate use the NER path, use POST /v1/systemone instead", cfg.Name)
}
return nil
}
// checkSystemOneNERModel applies systemOneNERAllowed to a model looked up by
// name; an unknown model passes so the not-found handling keeps its status.
func checkSystemOneNERModel(app *application.Application, modelName string) error {
cl := app.ModelConfigLoader()
if cl == nil {
return nil
}
cfg, ok := cl.GetModelConfig(modelName)
if !ok {
return nil
}
return systemOneNERAllowed(cfg)
}
// systemOneMaxBody and systemOneMaxQuestions bound one request. They keep a
// single call from pinning a decision model on an unbounded prompt, and match
// the limits Ollama documents for the same wire contract, so a client written
// for one server behaves the same on the other. The engine enforces any
// per-model option cap (letter-answer models refuse more than 26 options).
const (
systemOneMaxBody = 64 << 10
systemOneMaxQuestions = 64
)
// systemOneBind binds the JSON body with a size cap. Bind reads the whole body
// first, so the cap has to be on the reader.
func systemOneBind(c echo.Context, v any) error {
c.Request().Body = http.MaxBytesReader(c.Response(), c.Request().Body, systemOneMaxBody)
return c.Bind(v)
}
// systemOneBindStatus maps a bind failure to its status: 413 when the body
// exceeded the cap, 400 for anything else.
func systemOneBindStatus(err error) int {
var tooLarge *http.MaxBytesError
if errors.As(err, &tooLarge) {
return http.StatusRequestEntityTooLarge
}
return http.StatusBadRequest
}
func systemOneBindMessage(err error) string {
if systemOneBindStatus(err) == http.StatusRequestEntityTooLarge {
return fmt.Sprintf("request body exceeds %d KiB", systemOneMaxBody>>10)
}
return "invalid request body"
}
// validateSystemOneRequest checks the structure every path needs, before the
// request is forwarded to a decision model or run through the NER path. The
// forwarded path never sees parseSystemOneRequest, so without this a malformed
// question would surface as a backend error instead of a 400.
func validateSystemOneRequest(req *schema.SystemOneRequest) error {
if len(req.State) == 0 || string(req.State) == "null" {
return fmt.Errorf("state is required")
}
var state any
if err := json.Unmarshal(req.State, &state); err != nil {
return fmt.Errorf("state is not valid JSON: %w", err)
}
if s, ok := state.(string); ok && strings.TrimSpace(s) == "" {
return fmt.Errorf("state is required")
}
if len(req.Questions) == 0 {
return fmt.Errorf("questions is required and must contain at least one question")
}
if len(req.Questions) > systemOneMaxQuestions {
return fmt.Errorf("questions must contain at most %d questions", systemOneMaxQuestions)
}
qids := make([]string, 0, len(req.Questions))
for id := range req.Questions {
qids = append(qids, id)
}
sort.Strings(qids)
for _, id := range qids {
if strings.TrimSpace(id) == "" {
return fmt.Errorf("question ids must not be blank")
}
q := req.Questions[id]
switch q.Type {
case "choice":
var criteria map[string]json.RawMessage
if err := json.Unmarshal(q.Criteria, &criteria); err != nil {
return fmt.Errorf("question %q (choice) requires a criteria object", id)
}
if len(criteria) < 2 {
return fmt.Errorf("question %q (choice) requires at least 2 options", id)
}
for k := range criteria {
if strings.TrimSpace(k) == "" {
return fmt.Errorf("question %q (choice) has a blank option key", id)
}
}
case "score":
var criteria []json.RawMessage
if err := json.Unmarshal(q.Criteria, &criteria); err != nil {
return fmt.Errorf("question %q (score) requires a criteria array", id)
}
if len(criteria) < 2 {
return fmt.Errorf("question %q (score) requires at least 2 levels", id)
}
case "noul":
if len(q.Criteria) == 0 || string(q.Criteria) == "null" {
continue
}
var criteria map[string]json.RawMessage
if err := json.Unmarshal(q.Criteria, &criteria); err != nil {
return fmt.Errorf("question %q (noul) criteria must be an object with \"false\" and \"true\" descriptions", id)
}
for k := range criteria {
if k != "false" && k != "true" {
return fmt.Errorf("question %q (noul) criteria may only have \"false\" and \"true\" keys", id)
}
}
default:
return fmt.Errorf("question %q has unknown type: %s", id, q.Type)
}
}
return nil
}
// backendSupportsScore reports whether the named backend implements the
// Score gRPC RPC. vllm-cpp does (kev/laya decision pipeline and cua-s1-forms
// scoring via the unified vllm_decide C ABI); other backends fall through to
@@ -402,18 +588,24 @@ func backendSupportsScore(backendName string) bool {
func SystemOneEndpoint(app *application.Application) echo.HandlerFunc {
return func(c echo.Context) error {
var req schema.SystemOneRequest
if err := c.Bind(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, "invalid request body")
if err := systemOneBind(c, &req); err != nil {
return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err))
}
if req.Model == "" {
return systemOneError(c, http.StatusBadRequest, "model is required")
}
if err := checkSystemOneModel(app, req.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := validateSystemOneRequest(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
// vllm-cpp models (kev/laya) implement the decision pipeline natively
// via the vllm_decide C ABI. Forward the raw request JSON through the
// Score RPC and return the backend's response as-is.
cl := app.ModelConfigLoader()
if cl != nil {
if cfg, ok := cl.GetModelConfig(req.Model); ok && backendSupportsScore(cfg.Backend) {
if cfg, ok := cl.GetModelConfig(req.Model); ok && systemOneUsesDecisionPipeline(cfg) {
reqJSON, err := json.Marshal(req)
if err != nil {
return systemOneError(c, http.StatusInternalServerError, "failed to marshal request: "+err.Error())
@@ -468,12 +660,21 @@ func SystemOneEndpoint(app *application.Application) echo.HandlerFunc {
func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc {
return func(c echo.Context) error {
var req schema.SystemOnePermuteRequest
if err := c.Bind(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, "invalid request body")
if err := systemOneBind(c, &req); err != nil {
return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err))
}
if req.Request.Model == "" {
return systemOneError(c, http.StatusBadRequest, "model is required")
}
if err := checkSystemOneModel(app, req.Request.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := checkSystemOneNERModel(app, req.Request.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := validateSystemOneRequest(&req.Request); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if req.Question == "" {
return systemOneError(c, http.StatusBadRequest, "question is required")
}
@@ -604,12 +805,21 @@ func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc {
func SystemOneSeparateEndpoint(app *application.Application) echo.HandlerFunc {
return func(c echo.Context) error {
var req schema.SystemOneRequest
if err := c.Bind(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, "invalid request body")
if err := systemOneBind(c, &req); err != nil {
return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err))
}
if req.Model == "" {
return systemOneError(c, http.StatusBadRequest, "model is required")
}
if err := checkSystemOneModel(app, req.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := checkSystemOneNERModel(app, req.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := validateSystemOneRequest(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
parsed, err := parseSystemOneRequest(&req)
if err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
@@ -0,0 +1,74 @@
package localai
import (
"github.com/mudler/LocalAI/core/config"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("systemOneModelAllowed", func() {
mk := func(usecases ...string) config.ModelConfig {
return config.ModelConfig{
Name: "m",
Backend: "vllm-cpp",
KnownUsecases: config.GetUsecasesFromYAML(usecases),
}
}
It("accepts a declared decisions model", func() {
Expect(systemOneModelAllowed(mk("decisions"))).To(Succeed())
})
It("accepts a token_classify model, which the NER path serves", func() {
Expect(systemOneModelAllowed(mk("token_classify"))).To(Succeed())
})
It("keeps configs that declare no usecases working", func() {
Expect(systemOneModelAllowed(config.ModelConfig{Name: "laya", Backend: "vllm-cpp"})).To(Succeed())
})
It("refuses a chat-only model with an actionable message", func() {
Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [decisions]")))
})
})
var _ = Describe("systemone routing by model kind", func() {
mk := func(backend string, usecases ...string) config.ModelConfig {
c := config.ModelConfig{Name: "m", Backend: backend}
if len(usecases) > 0 {
c.KnownUsecases = config.GetUsecasesFromYAML(usecases)
}
return c
}
Describe("systemOneUsesDecisionPipeline", func() {
It("sends a declared decision model to the decision pipeline", func() {
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions"))).To(BeTrue())
})
It("sends a token_classify model to the NER path, since vllm_decide refuses NER architectures", func() {
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "token_classify"))).To(BeFalse())
})
It("keeps configs that declare nothing on the decision pipeline", func() {
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp"))).To(BeTrue())
})
It("prefers the decision pipeline when both usecases are declared", func() {
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions", "token_classify"))).To(BeTrue())
})
It("never uses it for a backend without the Score RPC", func() {
Expect(systemOneUsesDecisionPipeline(mk("no-such-backend", "decisions"))).To(BeFalse())
})
})
Describe("systemOneNERAllowed", func() {
It("refuses a decision model on the NER-only routes with an actionable message", func() {
Expect(systemOneNERAllowed(mk("vllm-cpp", "decisions"))).To(MatchError(ContainSubstring("/v1/systemone")))
})
It("accepts a token_classify model", func() {
Expect(systemOneNERAllowed(mk("vllm-cpp", "token_classify"))).To(Succeed())
})
It("accepts configs that declare nothing", func() {
Expect(systemOneNERAllowed(mk("vllm-cpp"))).To(Succeed())
})
})
})
@@ -0,0 +1,96 @@
package localai
import (
"encoding/json"
"net/http"
"net/http/httptest"
"strings"
"github.com/labstack/echo/v4"
"github.com/mudler/LocalAI/core/schema"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("validateSystemOneRequest", func() {
req := func(state string, questions string) *schema.SystemOneRequest {
r := &schema.SystemOneRequest{Model: "m", State: json.RawMessage(state)}
Expect(json.Unmarshal([]byte(questions), &r.Questions)).To(Succeed())
return r
}
It("accepts the three question types", func() {
r := req(`"ticket text"`, `{
"team": {"type":"choice","instructions":"which","criteria":{"a":"A","b":null}},
"refund": {"type":"noul","instructions":"refund?","criteria":{"false":"No refund","true":"Refund asked"}},
"urgency": {"type":"score","instructions":"how urgent","criteria":["low","high"]}
}`)
Expect(validateSystemOneRequest(r)).To(Succeed())
})
It("accepts a noul question with no criteria", func() {
Expect(validateSystemOneRequest(req(`"x"`, `{"q":{"type":"noul","instructions":"i"}}`))).To(Succeed())
})
DescribeTable("refuses a malformed request with a message that names the problem",
func(state, questions, want string) {
Expect(validateSystemOneRequest(req(state, questions))).To(MatchError(ContainSubstring(want)))
},
Entry("missing state", ``, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"),
Entry("null state", `null`, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"),
Entry("blank string state", `" "`, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"),
Entry("no questions", `"x"`, `{}`, "at least one question"),
Entry("blank question id", `"x"`, `{" ":{"type":"noul","instructions":"i"}}`, "blank"),
Entry("unknown type", `"x"`, `{"q":{"type":"rank","instructions":"i"}}`, "unknown type"),
Entry("choice with one option", `"x"`, `{"q":{"type":"choice","instructions":"i","criteria":{"a":"A"}}}`, "at least 2"),
Entry("choice with a blank option key", `"x"`, `{"q":{"type":"choice","instructions":"i","criteria":{"a":"A"," ":"B"}}}`, "blank"),
Entry("score with one level", `"x"`, `{"q":{"type":"score","instructions":"i","criteria":["only"]}}`, "at least 2"),
Entry("noul criteria with a stray key", `"x"`, `{"q":{"type":"noul","instructions":"i","criteria":{"maybe":"M"}}}`, `"false" and "true"`),
)
It("refuses more than 64 questions", func() {
var b strings.Builder
b.WriteString("{")
for i := 0; i < 65; i++ {
if i > 0 {
b.WriteString(",")
}
b.WriteString(`"q` + strings.Repeat("x", i) + `":{"type":"noul","instructions":"i"}`)
}
b.WriteString("}")
Expect(validateSystemOneRequest(req(`"x"`, b.String()))).To(MatchError(ContainSubstring("at most 64")))
})
})
var _ = Describe("systemOneBind", func() {
bind := func(body string) (int, error) {
e := echo.New()
r := httptest.NewRequest(http.MethodPost, "/v1/systemone", strings.NewReader(body))
r.Header.Set("Content-Type", "application/json")
c := e.NewContext(r, httptest.NewRecorder())
var out schema.SystemOneRequest
if err := systemOneBind(c, &out); err != nil {
return systemOneBindStatus(err), err
}
return http.StatusOK, nil
}
It("binds a normal body", func() {
status, err := bind(`{"model":"m","state":"x","questions":{}}`)
Expect(err).ToNot(HaveOccurred())
Expect(status).To(Equal(http.StatusOK))
})
It("answers 413 for a body over 64 KiB", func() {
status, err := bind(`{"model":"m","state":"` + strings.Repeat("a", 65*1024) + `"}`)
Expect(err).To(HaveOccurred())
Expect(status).To(Equal(http.StatusRequestEntityTooLarge))
})
It("answers 400 for malformed JSON", func() {
status, err := bind(`{not json`)
Expect(err).To(HaveOccurred())
Expect(status).To(Equal(http.StatusBadRequest))
})
})
@@ -172,6 +172,19 @@ test.describe('Models lifecycle', () => {
await expect(installedPane(page)).toContainText('Worker one')
})
test('shows the decisions use case on a decision model', async ({ page }) => {
await page.route('**/api/models/capabilities', route => route.fulfill({
contentType: 'application/json',
body: JSON.stringify({
data: [...installedModels, { id: 'decider', backend: 'vllm-cpp', capabilities: ['FLAG_DECISIONS'] }],
}),
}))
await page.goto('/app/models?view=installed&model=decider')
await expect(installedPane(page)).toContainText('decider')
await expect(installedPane(page)).toContainText('Decisions')
})
test('stops a running model with confirmation', async ({ page }) => {
await page.goto('/app/models?view=installed&model=alpha')
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -22,7 +22,7 @@ import {
CAP_CHAT, CAP_COMPLETION, CAP_IMAGE, CAP_VIDEO, CAP_TTS,
CAP_TRANSCRIPT, CAP_SOUND_GENERATION, CAP_FACE_RECOGNITION,
CAP_SPEAKER_RECOGNITION, CAP_EMBEDDINGS, CAP_RERANK,
CAP_VAD, CAP_SCORE,
CAP_VAD, CAP_SCORE, CAP_DECISIONS,
} from '../utils/capabilities'
const USE_CASES = [
@@ -39,6 +39,7 @@ const USE_CASES = [
{ cap: CAP_RERANK, labelKey: 'rerank' },
{ cap: CAP_VAD, labelKey: 'vad' },
{ cap: CAP_SCORE, labelKey: 'score' },
{ cap: CAP_DECISIONS, labelKey: 'decisions' },
]
export function modelUseCases(model) {
+1
View File
@@ -29,4 +29,5 @@ export const CAP_SPEAKER_RECOGNITION = 'FLAG_SPEAKER_RECOGNITION'
export const CAP_AUDIO_TRANSFORM = 'FLAG_AUDIO_TRANSFORM'
export const CAP_REALTIME_AUDIO = 'FLAG_REALTIME_AUDIO'
export const CAP_SCORE = 'FLAG_SCORE'
export const CAP_DECISIONS = 'FLAG_DECISIONS'
export const CAP_TOKEN_CLASSIFY = 'FLAG_TOKEN_CLASSIFY'
+3 -1
View File
@@ -1066,7 +1066,9 @@ known_usecases:
- embeddings
```
Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `llm` (combination of CHAT, COMPLETION, EDIT).
Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `score`, `token_classify`, `decisions`, `llm` (combination of CHAT, COMPLETION, EDIT).
`decisions` marks a model as a decision model for the [Decisions API]({{% relref "features/decisions" %}}) (`POST /v1/systemone`). It is never guessed, and a model that declares it is not listed as a chat, completion or embeddings model.
`token_classify` marks a model as a token-classification (NER) provider for the PII filter (e.g. an `openai-privacy-filter` GGUF). Declare it explicitly together with `embeddings: true` (the classifier loads via TOKEN_CLS pooling). It runs on the dedicated `privacy-filter` backend (`backend/cpp/privacy-filter`), a standalone GGML engine for the `openai-privacy-filter` family - separate from `llama-cpp`, which no longer carries the token-classification path.
+153
View File
@@ -0,0 +1,153 @@
+++
disableToc = false
title = "Decisions API"
weight = 66
url = "/features/decisions/"
+++
The Decisions API is a fast, typed decision layer. You send a piece of text (the
*state*) and a set of named questions. A decision model answers each question
with a value and a confidence, in one pass. The model does not generate text, so
there is nothing to parse and no free-form output to validate.
LocalAI serves it on the `/v1/systemone` routes. The request and response shapes
follow the [kev](https://github.com/jaredpalmer/kev) project, and the field names
and question types are the same ones Ollama serves on its `/v1/systemone`
endpoint (Ollama 0.35 and later). The wire contract is called SystemOne; the
capability a model declares is called `decisions`. See
[Compatibility with Ollama](#compatibility-with-ollama) for what differs.
OpenAI announced its own Decisions API in limited preview on 2026-09-29. It has no
public request or response schema yet, so LocalAI does not serve a `/v1/decisions`
route.
## Endpoints
| Endpoint | Method | Description |
|---|---|---|
| `/v1/systemone` | POST | Answer all questions in one pass |
| `/v1/systemone/permute` | POST | Re-run one choice question under `n_perm` option orders |
| `/v1/systemone/separate` | POST | Answer each question in its own pass |
Which route a model can serve depends on its kind:
| Model kind | `/v1/systemone` | `/permute` and `/separate` |
|---|---|---|
| Decision model (`decisions`), such as Laya or GLiNER2.5-Decide | Yes | No, returns `400` |
| Zero-shot NER model (`token_classify`), such as GLiNER2.5 | Yes, through the NER path | Yes |
## Question types
| Type | Answer | Fields in the answer |
|---|---|---|
| `choice` | One option out of a named set | `choice`, `probabilities`, `confidence` |
| `noul` | Yes, no or unknown for a statement | `noul` (0 to 1), `entities` |
| `score` | One level on a scale | `score`, `legend`, `probabilities`, `confidence` |
## Example
```bash
curl http://localhost:8080/v1/systemone -H "Content-Type: application/json" -d '{
"model": "laya-vllm-cpp",
"state": "My order arrived broken and I want my money back. This is the second time.",
"questions": {
"team": {
"type": "choice",
"instructions": "Which team should handle this ticket?",
"criteria": {
"billing": "Payments, invoices and refunds",
"shipping": "Delivery and damaged goods",
"product": "Questions about how the product works"
}
},
"refund_requested": {
"type": "noul",
"instructions": "The customer explicitly asks for a refund"
},
"urgency": {
"type": "score",
"instructions": "How urgent is this ticket?",
"criteria": ["not urgent", "somewhat urgent", "urgent", "critical"]
}
}
}'
```
Answers from a decision model carry a `confidence` value, and the response
reports token usage and `latency_ms`. The NER path does not report token usage.
## Choosing a model
A model can serve the Decisions API only if it is a decision model. Declare the usecase
in the model config:
```yaml
name: laya
backend: vllm-cpp
known_usecases:
- decisions
parameters:
model: convaiinnovations/laya
```
`decisions` is never guessed, and a model that declares it is not listed as a
chat, completion or embeddings model. A model that declares usecases without
`decisions` or `token_classify` gets a `400` from these endpoints that names the
missing usecase. A model that declares `token_classify` and not `decisions` is
served by the zero-shot NER path. A vllm-cpp config that declares no usecases is
treated as a decision model, so setups that predate the flag keep working, but a
config that declares only `chat` (as an older `laya` gallery entry did) now gets
the `400` and needs `known_usecases: [decisions]`.
Install one from the gallery and filter on the `decisions` tag:
| Gallery entry | Model | Notes |
|---|---|---|
| `laya-vllm-cpp` | Laya | ModernBERT-large, non-autoregressive, about 800 MB |
| `gliner25-decide-vllm-cpp` | GLiNER2.5-Decide | DeBERTa-v3-large with a classification head, about 2 GB |
The engine, [vllm.cpp]({{% relref "features/vllm-cpp" %}}), also supports the
kev, CLM and xor decision models. Those checkpoints need a conversion step, so
they are not gallery entries yet.
Tev1 is an autoregressive decision model. It answers through chat completions
and does not serve `/v1/systemone` yet.
## Request limits
A request is refused with `400` (or `413` for the body size) when:
- the body is larger than 64 KiB,
- `state` is missing or blank,
- there are no questions, or more than 64,
- a question id is blank,
- a `choice` question has fewer than 2 options or a blank option key,
- a `score` question has fewer than 2 levels,
- a `noul` question has `criteria` with keys other than `"false"` and `"true"`.
A `noul` question may carry `criteria` with a description for each outcome, for
example `{"false": "No refund is requested", "true": "The customer requests a refund"}`.
Some models cap the number of options for a `choice` or `score` question (models
that answer with a letter accept at most 26). The engine refuses more options than
the model supports and the error names the limit.
## Compatibility with Ollama
The field names, question types and answer fields are the same as Ollama's
`/v1/systemone`, so a client written for one works against the other for the
common case. These behaviors differ:
| | Ollama | LocalAI |
|---|---|---|
| `confidence` | `1 - H(p) / ln(N)`, an entropy measure | Computed by the model's pipeline. For kev and Laya it is a normalized margin, so the same probabilities give a different value |
| Errors | `{"error": "message"}` | `{"error": {"message": "...", "type": "invalid_request"}}` |
| `keep_alive` | Sets how long the model stays loaded | Accepted and ignored. Model lifetime follows the LocalAI idle and watchdog settings |
| `state` given as an object | Serialized as JSON text | Rendered as labeled lines, the way kev does it |
| `noul` answer on the NER path | `{type, noul}` | Also carries `entities` |
| Token `usage` | Full prompt lengths across all questions | Whatever the backend reports; the NER path reports 0 |
## Access control
When authentication is on, the three routes need the `decisions` feature. It is
on by default for every user, like the other API features, and an administrator
can turn it off per user.
+13 -10
View File
@@ -160,22 +160,25 @@ forward, which is the required contract for pooling models in vllm.cpp. A
device-resident forward is tracked as a performance optimization, not a
correctness gap.
### SystemOne structured-extraction API
### Decisions API
The `vllm-cpp` backend also exposes kev-compatible SystemOne endpoints that
turn zero-shot NER into structured question answering. These mirror the API
from the [kev](https://github.com/jaredpalmer/kev) project:
The `vllm-cpp` backend serves the kev-compatible SystemOne endpoints (the Decisions API): typed
`choice`, `noul` and `score` questions over a state text, answered by a
non-generative decision model in one pass. A decision model declares
`known_usecases: [decisions]`. See [Decisions API]({{% relref "features/decisions" %}})
for the request shape, the models you can install and the access rules.
| Endpoint | Method | Description |
|---|---|---|
| `/v1/systemone` | POST | Answer all questions in one NER pass |
| `/v1/systemone` | POST | Answer all questions in one pass |
| `/v1/systemone/permute` | POST | Re-run one choice question under n_perm option orders |
| `/v1/systemone/separate` | POST | Answer each question in its own NER pass (N passes) |
| `/v1/systemone/separate` | POST | Answer each question in its own pass (N passes) |
Each question has a `type` of `noul` (binary entity presence), `choice` (pick
one option), or `score` (pick one level). The `model` field in the request body
selects the NER model. Labels are derived from the question definition, so no
`ner_labels` configuration is needed for these endpoints.
The GLiNER2.5 zero-shot NER model (`token_classify`) also serves
`/v1/systemone`, through the NER path, and it is the model to use for
`/v1/systemone/permute` and `/v1/systemone/separate`, which decision models
refuse with a `400`. It derives its NER labels from the question definitions, so
no `ner_labels` configuration is needed.
## Beyond text generation
+104 -2
View File
@@ -19360,6 +19360,7 @@
- qwen3.6
- nvfp4
- vllm-cpp
- vision
- tool-calling
- reasoning
- gpu
@@ -19372,6 +19373,7 @@
known_usecases:
- chat
- completion
- vision
# Tool calls and the <think> split are parsed by the engine's own streaming
# parsers, so LocalAI's Go-side grammar path stays out of the way.
function:
@@ -19429,6 +19431,7 @@
- qwen3.6
- nvfp4
- vllm-cpp
- vision
- speculative-decoding
- mtp
- tool-calling
@@ -19442,6 +19445,7 @@
known_usecases:
- chat
- completion
- vision
function:
grammar:
disable: true
@@ -19497,6 +19501,7 @@
- qwen3.6
- nvfp4
- vllm-cpp
- vision
- speculative-decoding
- dflash
- tool-calling
@@ -19510,6 +19515,7 @@
known_usecases:
- chat
- completion
- vision
function:
grammar:
disable: true
@@ -19557,6 +19563,9 @@
with roughly 3B parameters active per token, so it reads like a much larger
model while costing about as much per token as a small one.
Image input is implemented in the engine but is not token-gated against
vLLM yet, so the vision usecase on this entry is experimental.
This is the engine's gated MoE checkpoint: token-for-token identical to vLLM
over the 315-prompt battery on both the synchronous and asynchronous paths,
at 0.92x to 0.97x vLLM's throughput from concurrency 1 to 32.
@@ -19575,6 +19584,8 @@
- moe
- nvfp4
- vllm-cpp
- vision
- experimental
- tool-calling
- reasoning
- gpu
@@ -19587,6 +19598,7 @@
known_usecases:
- chat
- completion
- vision
function:
grammar:
disable: true
@@ -19615,6 +19627,9 @@
description: |
Qwen3.6-35B-A3B NVFP4 on vllm.cpp with MTP speculative decoding enabled.
Image input is implemented in the engine but is not token-gated against
vLLM yet, so the vision usecase on this entry is experimental.
The draft head ships inside the checkpoint's own mtp.* tensors, so there is
no second model to download. On this model the speculative path is
token-exact against speculation-off on both the synchronous and asynchronous
@@ -19632,6 +19647,8 @@
- moe
- nvfp4
- vllm-cpp
- vision
- experimental
- speculative-decoding
- mtp
- tool-calling
@@ -19645,6 +19662,7 @@
known_usecases:
- chat
- completion
- vision
function:
grammar:
disable: true
@@ -63635,7 +63653,7 @@
512-token context. F16 weights, ~804 MB.
license: apache-2.0
tags:
- decision
- decisions
- systemone
- vllm-cpp
- cpu
@@ -63645,7 +63663,7 @@
overrides:
backend: vllm-cpp
known_usecases:
- chat
- decisions
parameters:
model: convaiinnovations/laya
artifacts:
@@ -63654,6 +63672,90 @@
source:
type: huggingface
repo: convaiinnovations/laya
- name: gliner25-decide-vllm-cpp
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/fastino/GLiNER2.5-Decide
- https://github.com/mudler/vllm.cpp
description: |
GLiNER2.5-Decide is a DeBERTa-v3-large encoder with a classification head
that answers typed decision questions over a state text in one forward
pass. It never generates text, so there is nothing to parse.
In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine runs the
decision pipeline (choice, noul and score question types) through the
vllm_decide C ABI. This is the decision model, not the zero-shot NER model:
use the gliner2.5 entry for entity extraction. F32 weights, about 2 GB.
The weights are pinned to a revision so the entry keeps serving the
checkpoint it was checked against.
license: apache-2.0
tags:
- decisions
- systemone
- vllm-cpp
- cpu
- gpu
size: 2GB
last_checked: "2026-09-30"
overrides:
backend: vllm-cpp
known_usecases:
- decisions
parameters:
model: fastino/GLiNER2.5-Decide
artifacts:
- name: model
target: model
source:
type: huggingface
repo: fastino/GLiNER2.5-Decide
revision: 5a7adf72a23b4d311abae6ce050d7f0012bb3416
- name: qwen3-vl-4b-vllm-cpp
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct
- https://github.com/mudler/vllm.cpp
description: |
Qwen3-VL-4B-Instruct on vllm.cpp, in bf16: a small vision-language model
that takes images alongside text. In the engine's correctness battery the
image path matches vLLM token for token, and video input is a near tie.
Roughly 9 GB of weights plus KV cache at the context configured here. It
runs where the flagship NVFP4 checkpoints cannot, including plain CPU.
license: apache-2.0
tags:
- llm
- vision
- multimodal
- qwen
- qwen3-vl
- vllm-cpp
- cpu
- gpu
size: 9GB
last_checked: "2026-09-30"
overrides:
backend: vllm-cpp
known_usecases:
- chat
- completion
- vision
template:
use_tokenizer_template: true
context_size: 8192
engine_args:
block_size: 32
num_blocks: 512
max_num_seqs: 4
parameters:
model: Qwen/Qwen3-VL-4B-Instruct
artifacts:
- name: model
target: model
source:
type: huggingface
repo: Qwen/Qwen3-VL-4B-Instruct
revision: ebb281ec70b05090aa6165b016eac8ec08e71b17
- name: cua-s1-forms-vllm-cpp
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls: