diff --git a/core/config/backend_capabilities.go b/core/config/backend_capabilities.go index b48266ead..650bc3f3c 100644 --- a/core/config/backend_capabilities.go +++ b/core/config/backend_capabilities.go @@ -35,6 +35,7 @@ const ( UsecaseSpeakerRecognition = "speaker_recognition" UsecaseTokenClassify = "token_classify" UsecaseScore = "score" + UsecaseDecisions = "decisions" ) // GRPCMethod identifies a Backend service RPC from backend.proto. @@ -216,6 +217,11 @@ var UsecaseInfoMap = map[string]UsecaseInfo{ GRPCMethod: MethodScore, Description: "Joint log-probability scoring of candidate continuations via the Score RPC. Declared explicitly via known_usecases and usable alongside generation usecases.", }, + UsecaseDecisions: { + Flag: FLAG_DECISIONS, + GRPCMethod: MethodScore, + Description: "Decision models (served by POST /v1/systemone): typed choice, noul and score questions over a state text, answered by a non-generative decision model through the Score RPC (question_type systemone). Declared explicitly via known_usecases.", + }, } // BackendCapability describes which gRPC methods and usecases a backend supports. @@ -349,10 +355,10 @@ var BackendCapabilities = map[string]BackendCapability{ // model returns an error rather than silent garbage. "vllm-cpp": { GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateVideo, MethodTokenClassify, MethodScore}, - PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVideo, UsecaseTokenClassify, UsecaseScore}, + PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVision, UsecaseVideo, UsecaseTokenClassify, UsecaseScore, UsecaseDecisions}, DefaultUsecases: []string{UsecaseChat}, AcceptsImages: true, - Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and kev/laya decision pipelines", + Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and decision models (kev, laya, CLM, GLiNER2.5-Decide, xor, nimble)", }, "vllm-omni": { GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateImage, MethodGenerateVideo, MethodTTS}, diff --git a/core/config/gguf.go b/core/config/gguf.go index e5f3bc5b4..fad00a6c7 100644 --- a/core/config/gguf.go +++ b/core/config/gguf.go @@ -16,14 +16,14 @@ import ( // reservedNonChatModel reports whether the operator reserved this model for an // internal primitive — the router score classifier or the PII NER -// token_classify tier. Such a model has no chat template and must not be +// token_classify tier, or a decision head. Such a model has no chat template and must not be // given the generative-chat defaults the GGUF importer otherwise applies // (FLAG_CHAT, jinja templating): surfacing it in chat pickers defeats the // reservation. Operators who do want a combined model declare both usecases // explicitly — the combination is valid. func reservedNonChatModel(cfg *ModelConfig) bool { return cfg.KnownUsecases != nil && - (*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY)) != 0 + (*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY|FLAG_DECISIONS)) != 0 } // genAudioEncoderKey is the mmproj metadata flag llama.cpp's mtmd writes for a diff --git a/core/config/model_config.go b/core/config/model_config.go index c8502fae5..bc084aa87 100644 --- a/core/config/model_config.go +++ b/core/config/model_config.go @@ -2056,6 +2056,13 @@ const ( FLAG_3D ModelConfigUsecase = 0b100000000000000000000000 FLAG_3D_ANIMATION ModelConfigUsecase = 1 << 24 + // Marks a model as a decision model: it answers typed choice / noul / + // score questions over a state (served by POST /v1/systemone). + // Explicit only, like FLAG_SCORE: a decision model never generates + // text, so guessing chat or embeddings for it would surface it in + // pickers it cannot serve. + FLAG_DECISIONS ModelConfigUsecase = 1 << 25 + // Common Subsets FLAG_LLM ModelConfigUsecase = FLAG_CHAT | FLAG_COMPLETION | FLAG_EDIT ) @@ -2118,6 +2125,7 @@ func GetAllModelConfigUsecases() map[string]ModelConfigUsecase { "FLAG_TOKEN_CLASSIFY": FLAG_TOKEN_CLASSIFY, "FLAG_3D": FLAG_3D, "FLAG_3D_ANIMATION": FLAG_3D_ANIMATION, + "FLAG_DECISIONS": FLAG_DECISIONS, } } @@ -2146,9 +2154,9 @@ func GetUsecasesFromYAML(input []string) *ModelConfigUsecase { // // Declared known_usecases are normally additive — the guessing heuristic // still adds whatever it can infer from backend/templates. The exceptions -// are FLAG_SCORE and FLAG_TOKEN_CLASSIFY: when the operator declared -// either, they reserved the model for an internal direct-decode primitive -// (the router classifier, or the PII NER tier). Letting GuessUsecases +// are FLAG_SCORE, FLAG_TOKEN_CLASSIFY and FLAG_DECISIONS: when the operator +// declared any of them, they reserved the model for a direct-decode primitive +// (the router classifier, the PII NER tier, or a decision head). Letting GuessUsecases // paint chat/completion/embeddings on top would surface it in pickers it // was deliberately kept out of. So a declared score or token_classify // list is authoritative; declare the generation usecases explicitly @@ -2158,7 +2166,7 @@ func (c *ModelConfig) HasUsecases(u ModelConfigUsecase) bool { if (u & *c.KnownUsecases) == u { return true } - if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY)) != 0 { + if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY | FLAG_DECISIONS)) != 0 { return false } } @@ -2381,6 +2389,14 @@ func (c *ModelConfig) GuessUsecases(u ModelConfigUsecase) bool { return false } + if (u & FLAG_DECISIONS) == FLAG_DECISIONS { + // No heuristic: decisions intent is a deliberate operator choice + // (the model is a non-generative decision head), so + // HasUsecases(FLAG_DECISIONS) is true only when KnownUsecases + // declares it explicitly. + return false + } + return true } diff --git a/core/config/model_config_test.go b/core/config/model_config_test.go index 828160fe1..b9d46f1c2 100644 --- a/core/config/model_config_test.go +++ b/core/config/model_config_test.go @@ -955,3 +955,36 @@ var _ = Describe("ModelConfig alias", func() { Expect(err).To(MatchError(ContainSubstring("alias"))) }) }) + +var _ = Describe("decisions usecase", func() { + // A decision model never generates text, so a declared decisions list + // must stay authoritative and the heuristic must never guess the flag. + It("is authoritative when declared and never guessed", func() { + declared := GetUsecasesFromYAML([]string{"decisions"}) + Expect(declared).NotTo(BeNil()) + Expect(*declared).NotTo(Equal(FLAG_ANY)) + + cfg := ModelConfig{ + Name: "laya", + Backend: "vllm-cpp", + KnownUsecases: declared, + TemplateConfig: TemplateConfig{ + Chat: "inherited from chatml", + ChatMessage: "inherited from chatml", + Completion: "inherited from chatml", + }, + } + Expect(cfg.HasUsecases(*declared)).To(BeTrue()) + Expect(cfg.HasUsecases(FLAG_CHAT)).To(BeFalse()) + Expect(cfg.HasUsecases(FLAG_COMPLETION)).To(BeFalse()) + Expect(cfg.HasUsecases(FLAG_EMBEDDINGS)).To(BeFalse()) + + undeclared := ModelConfig{Name: "laya", Backend: "vllm-cpp"} + Expect(undeclared.HasUsecases(*declared)).To(BeFalse()) + }) + + It("is a reserved usecase for the GGUF importer chat-default guard", func() { + declared := GetUsecasesFromYAML([]string{"decisions"}) + Expect(reservedNonChatModel(&ModelConfig{Backend: "vllm-cpp", KnownUsecases: declared})).To(BeTrue()) + }) +}) diff --git a/core/gallery/vllm_cpp_tags_test.go b/core/gallery/vllm_cpp_tags_test.go new file mode 100644 index 000000000..b84f9a2b1 --- /dev/null +++ b/core/gallery/vllm_cpp_tags_test.go @@ -0,0 +1,48 @@ +package gallery_test + +import ( + "fmt" + "slices" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + "github.com/mudler/LocalAI/core/config" +) + +// A gallery tag that names a capability is what users filter on, and +// known_usecases is what the server routes on. When they disagree, the entry +// is listed under a filter it cannot serve, or is hidden from one it can. +var _ = Describe("gallery/index.yaml vllm-cpp capability tags", func() { + It("keeps capability tags and known_usecases in agreement", func() { + entries, err := loadGalleryIndex() + Expect(err).ToNot(HaveOccurred()) + + tagToFlag := map[string]config.ModelConfigUsecase{ + "decisions": config.FLAG_DECISIONS, + "vision": config.FLAG_VISION, + "token-classify": config.FLAG_TOKEN_CLASSIFY, + "scoring": config.FLAG_SCORE, + } + + var violations []string + seen := 0 + for i := range entries { + e := &entries[i] + if backend, _ := e.Overrides["backend"].(string); backend != "vllm-cpp" { + continue + } + seen++ + declared := e.GetKnownUsecases() + for tag, flag := range tagToFlag { + tagged := slices.Contains(e.Tags, tag) + has := declared != nil && *declared&flag == flag + if tagged != has { + violations = append(violations, fmt.Sprintf("%s: tag %q present=%v but known_usecases declares it=%v", e.Name, tag, tagged, has)) + } + } + } + Expect(seen).To(BeNumerically(">", 0)) + Expect(violations).To(BeEmpty()) + }) +}) diff --git a/core/http/auth/features.go b/core/http/auth/features.go index 45411f824..4c1f53ec2 100644 --- a/core/http/auth/features.go +++ b/core/http/auth/features.go @@ -71,6 +71,11 @@ var RouteFeatureRegistry = []RouteFeature{ // Detection {"POST", "/v1/detection", FeatureDetection}, + // Decisions API (SystemOne wire contract) + {"POST", "/v1/systemone", FeatureDecisions}, + {"POST", "/v1/systemone/permute", FeatureDecisions}, + {"POST", "/v1/systemone/separate", FeatureDecisions}, + // Face recognition {"POST", "/v1/face/verify", FeatureFaceRecognition}, {"POST", "/v1/face/analyze", FeatureFaceRecognition}, @@ -209,5 +214,6 @@ func APIFeatureMetas() []FeatureMeta { {FeatureVoiceRecognition, "Voice Recognition", true}, {FeatureAudioTransform, "Audio Transform", true}, {FeaturePIIFilter, "PII Analyze / Redact", true}, + {FeatureDecisions, "Decisions", true}, } } diff --git a/core/http/auth/features_decisions_test.go b/core/http/auth/features_decisions_test.go new file mode 100644 index 000000000..4c5528618 --- /dev/null +++ b/core/http/auth/features_decisions_test.go @@ -0,0 +1,24 @@ +package auth_test + +import ( + . "github.com/mudler/LocalAI/core/http/auth" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("Decisions feature registration", func() { + It("gates the three decision routes behind one default-on API feature", func() { + Expect(APIFeatures).To(ContainElement(FeatureDecisions)) + + patterns := []string{} + for _, route := range RouteFeatureRegistry { + if route.Feature == FeatureDecisions { + Expect(route.Method).To(Equal("POST")) + patterns = append(patterns, route.Pattern) + } + } + Expect(patterns).To(ConsistOf("/v1/systemone", "/v1/systemone/permute", "/v1/systemone/separate")) + + Expect(APIFeatureMetas()).To(ContainElement(FeatureMeta{Key: FeatureDecisions, Label: "Decisions", DefaultValue: true})) + }) +}) diff --git a/core/http/auth/permissions.go b/core/http/auth/permissions.go index 95e76f572..3f8b6deff 100644 --- a/core/http/auth/permissions.go +++ b/core/http/auth/permissions.go @@ -59,6 +59,7 @@ const ( FeatureFaceRecognition = "face_recognition" FeatureVoiceRecognition = "voice_recognition" FeatureAudioTransform = "audio_transform" + FeatureDecisions = "decisions" // FeaturePIIFilter gates the synchronous PII analyze/redact service // (POST /api/pii/{analyze,redact}). Default ON like the other API // features; the admin-only events log is gated separately in-handler. @@ -78,7 +79,7 @@ var APIFeatures = []string{ FeatureVAD, FeatureDetection, FeatureVideo, Feature3D, FeatureEmbeddings, FeatureSound, FeatureRealtime, FeatureModeration, FeatureRerank, FeatureTokenize, FeatureMCP, FeatureStores, FeatureFaceRecognition, FeatureVoiceRecognition, FeatureAudioTransform, - FeaturePIIFilter, + FeaturePIIFilter, FeatureDecisions, } // AllFeatures lists all known features (used by UI and validation). diff --git a/core/http/endpoints/localai/api_instructions.go b/core/http/endpoints/localai/api_instructions.go index 8d0ea6d2f..e65e61c38 100644 --- a/core/http/endpoints/localai/api_instructions.go +++ b/core/http/endpoints/localai/api_instructions.go @@ -105,6 +105,12 @@ var instructionDefs = []instructionDef{ Tags: []string{"voice-recognition"}, Intro: "Voice (speaker) recognition — the audio analog to /v1/face/*. Use /v1/voice/verify for 1:1 speaker comparison, /v1/voice/identify for 1:N match against the registered store, /v1/voice/{register,forget} to manage that store, /v1/voice/embed for a raw speaker-encoder vector, and /v1/voice/analyze for age / gender / emotion inferred from speech. Registrations are in-memory by default and lost on restart. Audio inputs accept URL, base64, or data-URI; /v1/embeddings remains text-only.", }, + { + Name: "decisions", + Description: "Typed decisions (choice, noul, score) over a state text with calibrated confidence", + Tags: []string{"systemone"}, + Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { : { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [decisions] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. Field names and question types follow Ollama's /v1/systemone, with differences in confidence, error shape and keep_alive (see the Decisions API docs). A request over 64 KiB, with more than 64 questions, or with a malformed question is refused.", + }, { Name: "branding", Description: "Whitelabel the instance: configure name, tagline, logo, and favicon", diff --git a/core/http/endpoints/localai/api_instructions_test.go b/core/http/endpoints/localai/api_instructions_test.go index f42e1c92d..a3b504304 100644 --- a/core/http/endpoints/localai/api_instructions_test.go +++ b/core/http/endpoints/localai/api_instructions_test.go @@ -39,7 +39,7 @@ var _ = Describe("API Instructions Endpoints", func() { instructions, ok := resp["instructions"].([]any) Expect(ok).To(BeTrue()) - Expect(instructions).To(HaveLen(20)) + Expect(instructions).To(HaveLen(21)) // Verify each instruction has required fields and correct URL format for _, s := range instructions { @@ -82,6 +82,7 @@ var _ = Describe("API Instructions Endpoints", func() { "voice-library", "3d", "failover", + "decisions", )) }) }) @@ -136,6 +137,17 @@ var _ = Describe("API Instructions Endpoints", func() { Expect(string(body)).NotTo(ContainSubstring("/v1/3d/generations")) }) + It("should advertise the Decisions API", func() { + req := httptest.NewRequest(http.MethodGet, "/api/instructions/decisions", nil) + rec := httptest.NewRecorder() + app.ServeHTTP(rec, req) + + Expect(rec.Code).To(Equal(http.StatusOK)) + body, _ := io.ReadAll(rec.Body) + Expect(string(body)).To(ContainSubstring("POST /v1/systemone")) + Expect(string(body)).To(ContainSubstring("known_usecases: [decisions]")) + }) + It("should return JSON fragment when format=json", func() { req := httptest.NewRequest(http.MethodGet, "/api/instructions/chat-inference?format=json", nil) rec := httptest.NewRecorder() diff --git a/core/http/endpoints/localai/systemone.go b/core/http/endpoints/localai/systemone.go index e272e1b8b..7bb3e81b8 100644 --- a/core/http/endpoints/localai/systemone.go +++ b/core/http/endpoints/localai/systemone.go @@ -2,6 +2,7 @@ package localai import ( "encoding/json" + "errors" "fmt" "math" "math/rand" @@ -371,6 +372,191 @@ func systemOneError(c echo.Context, status int, msg string) error { }) } +// systemOneModelAllowed keeps chat and embedding models out of the decision +// API with an actionable error instead of a backend failure. A config that +// declares no usecases predates the flag and stays allowed, and a +// token_classify model is allowed because the NER path serves it. +func systemOneModelAllowed(cfg config.ModelConfig) error { + if cfg.KnownUsecases == nil { + return nil + } + if *cfg.KnownUsecases&(config.FLAG_DECISIONS|config.FLAG_TOKEN_CLASSIFY) != 0 { + return nil + } + return fmt.Errorf("model %q does not declare the decisions usecase (known_usecases: [decisions])", cfg.Name) +} + +// checkSystemOneModel applies systemOneModelAllowed to a model looked up by +// name. An unknown model passes here so the existing not-found handling +// downstream keeps its status code. +func checkSystemOneModel(app *application.Application, modelName string) error { + cl := app.ModelConfigLoader() + if cl == nil { + return nil + } + cfg, ok := cl.GetModelConfig(modelName) + if !ok { + return nil + } + return systemOneModelAllowed(cfg) +} + +// systemOneUsesDecisionPipeline reports whether /v1/systemone forwards the +// request to the backend's Score RPC (the decision pipeline) for this model. +// A model that declares token_classify without systemone is a zero-shot NER +// model: the backend's decision entry point refuses those architectures, so it +// goes to the NER path instead. A config that declares nothing keeps the +// decision pipeline, which is what setups that predate the decisions usecase +// relied on. +func systemOneUsesDecisionPipeline(cfg config.ModelConfig) bool { + if !backendSupportsScore(cfg.Backend) { + return false + } + if cfg.KnownUsecases == nil { + return true + } + declared := *cfg.KnownUsecases + if declared&config.FLAG_DECISIONS != 0 { + return true + } + return declared&config.FLAG_TOKEN_CLASSIFY == 0 +} + +// systemOneNERAllowed guards /permute and /separate, which always run the NER +// path. A decision model cannot serve them: the backend's NER entry point +// refuses its architecture, and the caller would see a backend error. +func systemOneNERAllowed(cfg config.ModelConfig) error { + if cfg.KnownUsecases == nil { + return nil + } + declared := *cfg.KnownUsecases + if declared&config.FLAG_DECISIONS != 0 && declared&config.FLAG_TOKEN_CLASSIFY == 0 { + return fmt.Errorf("model %q is a decision model: /permute and /separate use the NER path, use POST /v1/systemone instead", cfg.Name) + } + return nil +} + +// checkSystemOneNERModel applies systemOneNERAllowed to a model looked up by +// name; an unknown model passes so the not-found handling keeps its status. +func checkSystemOneNERModel(app *application.Application, modelName string) error { + cl := app.ModelConfigLoader() + if cl == nil { + return nil + } + cfg, ok := cl.GetModelConfig(modelName) + if !ok { + return nil + } + return systemOneNERAllowed(cfg) +} + +// systemOneMaxBody and systemOneMaxQuestions bound one request. They keep a +// single call from pinning a decision model on an unbounded prompt, and match +// the limits Ollama documents for the same wire contract, so a client written +// for one server behaves the same on the other. The engine enforces any +// per-model option cap (letter-answer models refuse more than 26 options). +const ( + systemOneMaxBody = 64 << 10 + systemOneMaxQuestions = 64 +) + +// systemOneBind binds the JSON body with a size cap. Bind reads the whole body +// first, so the cap has to be on the reader. +func systemOneBind(c echo.Context, v any) error { + c.Request().Body = http.MaxBytesReader(c.Response(), c.Request().Body, systemOneMaxBody) + return c.Bind(v) +} + +// systemOneBindStatus maps a bind failure to its status: 413 when the body +// exceeded the cap, 400 for anything else. +func systemOneBindStatus(err error) int { + var tooLarge *http.MaxBytesError + if errors.As(err, &tooLarge) { + return http.StatusRequestEntityTooLarge + } + return http.StatusBadRequest +} + +func systemOneBindMessage(err error) string { + if systemOneBindStatus(err) == http.StatusRequestEntityTooLarge { + return fmt.Sprintf("request body exceeds %d KiB", systemOneMaxBody>>10) + } + return "invalid request body" +} + +// validateSystemOneRequest checks the structure every path needs, before the +// request is forwarded to a decision model or run through the NER path. The +// forwarded path never sees parseSystemOneRequest, so without this a malformed +// question would surface as a backend error instead of a 400. +func validateSystemOneRequest(req *schema.SystemOneRequest) error { + if len(req.State) == 0 || string(req.State) == "null" { + return fmt.Errorf("state is required") + } + var state any + if err := json.Unmarshal(req.State, &state); err != nil { + return fmt.Errorf("state is not valid JSON: %w", err) + } + if s, ok := state.(string); ok && strings.TrimSpace(s) == "" { + return fmt.Errorf("state is required") + } + if len(req.Questions) == 0 { + return fmt.Errorf("questions is required and must contain at least one question") + } + if len(req.Questions) > systemOneMaxQuestions { + return fmt.Errorf("questions must contain at most %d questions", systemOneMaxQuestions) + } + qids := make([]string, 0, len(req.Questions)) + for id := range req.Questions { + qids = append(qids, id) + } + sort.Strings(qids) + for _, id := range qids { + if strings.TrimSpace(id) == "" { + return fmt.Errorf("question ids must not be blank") + } + q := req.Questions[id] + switch q.Type { + case "choice": + var criteria map[string]json.RawMessage + if err := json.Unmarshal(q.Criteria, &criteria); err != nil { + return fmt.Errorf("question %q (choice) requires a criteria object", id) + } + if len(criteria) < 2 { + return fmt.Errorf("question %q (choice) requires at least 2 options", id) + } + for k := range criteria { + if strings.TrimSpace(k) == "" { + return fmt.Errorf("question %q (choice) has a blank option key", id) + } + } + case "score": + var criteria []json.RawMessage + if err := json.Unmarshal(q.Criteria, &criteria); err != nil { + return fmt.Errorf("question %q (score) requires a criteria array", id) + } + if len(criteria) < 2 { + return fmt.Errorf("question %q (score) requires at least 2 levels", id) + } + case "noul": + if len(q.Criteria) == 0 || string(q.Criteria) == "null" { + continue + } + var criteria map[string]json.RawMessage + if err := json.Unmarshal(q.Criteria, &criteria); err != nil { + return fmt.Errorf("question %q (noul) criteria must be an object with \"false\" and \"true\" descriptions", id) + } + for k := range criteria { + if k != "false" && k != "true" { + return fmt.Errorf("question %q (noul) criteria may only have \"false\" and \"true\" keys", id) + } + } + default: + return fmt.Errorf("question %q has unknown type: %s", id, q.Type) + } + } + return nil +} + // backendSupportsScore reports whether the named backend implements the // Score gRPC RPC. vllm-cpp does (kev/laya decision pipeline and cua-s1-forms // scoring via the unified vllm_decide C ABI); other backends fall through to @@ -402,18 +588,24 @@ func backendSupportsScore(backendName string) bool { func SystemOneEndpoint(app *application.Application) echo.HandlerFunc { return func(c echo.Context) error { var req schema.SystemOneRequest - if err := c.Bind(&req); err != nil { - return systemOneError(c, http.StatusBadRequest, "invalid request body") + if err := systemOneBind(c, &req); err != nil { + return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err)) } if req.Model == "" { return systemOneError(c, http.StatusBadRequest, "model is required") } + if err := checkSystemOneModel(app, req.Model); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } + if err := validateSystemOneRequest(&req); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } // vllm-cpp models (kev/laya) implement the decision pipeline natively // via the vllm_decide C ABI. Forward the raw request JSON through the // Score RPC and return the backend's response as-is. cl := app.ModelConfigLoader() if cl != nil { - if cfg, ok := cl.GetModelConfig(req.Model); ok && backendSupportsScore(cfg.Backend) { + if cfg, ok := cl.GetModelConfig(req.Model); ok && systemOneUsesDecisionPipeline(cfg) { reqJSON, err := json.Marshal(req) if err != nil { return systemOneError(c, http.StatusInternalServerError, "failed to marshal request: "+err.Error()) @@ -468,12 +660,21 @@ func SystemOneEndpoint(app *application.Application) echo.HandlerFunc { func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc { return func(c echo.Context) error { var req schema.SystemOnePermuteRequest - if err := c.Bind(&req); err != nil { - return systemOneError(c, http.StatusBadRequest, "invalid request body") + if err := systemOneBind(c, &req); err != nil { + return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err)) } if req.Request.Model == "" { return systemOneError(c, http.StatusBadRequest, "model is required") } + if err := checkSystemOneModel(app, req.Request.Model); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } + if err := checkSystemOneNERModel(app, req.Request.Model); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } + if err := validateSystemOneRequest(&req.Request); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } if req.Question == "" { return systemOneError(c, http.StatusBadRequest, "question is required") } @@ -604,12 +805,21 @@ func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc { func SystemOneSeparateEndpoint(app *application.Application) echo.HandlerFunc { return func(c echo.Context) error { var req schema.SystemOneRequest - if err := c.Bind(&req); err != nil { - return systemOneError(c, http.StatusBadRequest, "invalid request body") + if err := systemOneBind(c, &req); err != nil { + return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err)) } if req.Model == "" { return systemOneError(c, http.StatusBadRequest, "model is required") } + if err := checkSystemOneModel(app, req.Model); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } + if err := checkSystemOneNERModel(app, req.Model); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } + if err := validateSystemOneRequest(&req); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } parsed, err := parseSystemOneRequest(&req) if err != nil { return systemOneError(c, http.StatusBadRequest, err.Error()) diff --git a/core/http/endpoints/localai/systemone_gate_test.go b/core/http/endpoints/localai/systemone_gate_test.go new file mode 100644 index 000000000..65978c463 --- /dev/null +++ b/core/http/endpoints/localai/systemone_gate_test.go @@ -0,0 +1,74 @@ +package localai + +import ( + "github.com/mudler/LocalAI/core/config" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("systemOneModelAllowed", func() { + mk := func(usecases ...string) config.ModelConfig { + return config.ModelConfig{ + Name: "m", + Backend: "vllm-cpp", + KnownUsecases: config.GetUsecasesFromYAML(usecases), + } + } + + It("accepts a declared decisions model", func() { + Expect(systemOneModelAllowed(mk("decisions"))).To(Succeed()) + }) + + It("accepts a token_classify model, which the NER path serves", func() { + Expect(systemOneModelAllowed(mk("token_classify"))).To(Succeed()) + }) + + It("keeps configs that declare no usecases working", func() { + Expect(systemOneModelAllowed(config.ModelConfig{Name: "laya", Backend: "vllm-cpp"})).To(Succeed()) + }) + + It("refuses a chat-only model with an actionable message", func() { + Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [decisions]"))) + }) +}) + +var _ = Describe("systemone routing by model kind", func() { + mk := func(backend string, usecases ...string) config.ModelConfig { + c := config.ModelConfig{Name: "m", Backend: backend} + if len(usecases) > 0 { + c.KnownUsecases = config.GetUsecasesFromYAML(usecases) + } + return c + } + + Describe("systemOneUsesDecisionPipeline", func() { + It("sends a declared decision model to the decision pipeline", func() { + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions"))).To(BeTrue()) + }) + It("sends a token_classify model to the NER path, since vllm_decide refuses NER architectures", func() { + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "token_classify"))).To(BeFalse()) + }) + It("keeps configs that declare nothing on the decision pipeline", func() { + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp"))).To(BeTrue()) + }) + It("prefers the decision pipeline when both usecases are declared", func() { + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions", "token_classify"))).To(BeTrue()) + }) + It("never uses it for a backend without the Score RPC", func() { + Expect(systemOneUsesDecisionPipeline(mk("no-such-backend", "decisions"))).To(BeFalse()) + }) + }) + + Describe("systemOneNERAllowed", func() { + It("refuses a decision model on the NER-only routes with an actionable message", func() { + Expect(systemOneNERAllowed(mk("vllm-cpp", "decisions"))).To(MatchError(ContainSubstring("/v1/systemone"))) + }) + It("accepts a token_classify model", func() { + Expect(systemOneNERAllowed(mk("vllm-cpp", "token_classify"))).To(Succeed()) + }) + It("accepts configs that declare nothing", func() { + Expect(systemOneNERAllowed(mk("vllm-cpp"))).To(Succeed()) + }) + }) +}) diff --git a/core/http/endpoints/localai/systemone_validate_test.go b/core/http/endpoints/localai/systemone_validate_test.go new file mode 100644 index 000000000..dfec4c27c --- /dev/null +++ b/core/http/endpoints/localai/systemone_validate_test.go @@ -0,0 +1,96 @@ +package localai + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + + "github.com/labstack/echo/v4" + "github.com/mudler/LocalAI/core/schema" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("validateSystemOneRequest", func() { + req := func(state string, questions string) *schema.SystemOneRequest { + r := &schema.SystemOneRequest{Model: "m", State: json.RawMessage(state)} + Expect(json.Unmarshal([]byte(questions), &r.Questions)).To(Succeed()) + return r + } + + It("accepts the three question types", func() { + r := req(`"ticket text"`, `{ + "team": {"type":"choice","instructions":"which","criteria":{"a":"A","b":null}}, + "refund": {"type":"noul","instructions":"refund?","criteria":{"false":"No refund","true":"Refund asked"}}, + "urgency": {"type":"score","instructions":"how urgent","criteria":["low","high"]} + }`) + Expect(validateSystemOneRequest(r)).To(Succeed()) + }) + + It("accepts a noul question with no criteria", func() { + Expect(validateSystemOneRequest(req(`"x"`, `{"q":{"type":"noul","instructions":"i"}}`))).To(Succeed()) + }) + + DescribeTable("refuses a malformed request with a message that names the problem", + func(state, questions, want string) { + Expect(validateSystemOneRequest(req(state, questions))).To(MatchError(ContainSubstring(want))) + }, + Entry("missing state", ``, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"), + Entry("null state", `null`, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"), + Entry("blank string state", `" "`, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"), + Entry("no questions", `"x"`, `{}`, "at least one question"), + Entry("blank question id", `"x"`, `{" ":{"type":"noul","instructions":"i"}}`, "blank"), + Entry("unknown type", `"x"`, `{"q":{"type":"rank","instructions":"i"}}`, "unknown type"), + Entry("choice with one option", `"x"`, `{"q":{"type":"choice","instructions":"i","criteria":{"a":"A"}}}`, "at least 2"), + Entry("choice with a blank option key", `"x"`, `{"q":{"type":"choice","instructions":"i","criteria":{"a":"A"," ":"B"}}}`, "blank"), + Entry("score with one level", `"x"`, `{"q":{"type":"score","instructions":"i","criteria":["only"]}}`, "at least 2"), + Entry("noul criteria with a stray key", `"x"`, `{"q":{"type":"noul","instructions":"i","criteria":{"maybe":"M"}}}`, `"false" and "true"`), + ) + + It("refuses more than 64 questions", func() { + var b strings.Builder + b.WriteString("{") + for i := 0; i < 65; i++ { + if i > 0 { + b.WriteString(",") + } + b.WriteString(`"q` + strings.Repeat("x", i) + `":{"type":"noul","instructions":"i"}`) + } + b.WriteString("}") + Expect(validateSystemOneRequest(req(`"x"`, b.String()))).To(MatchError(ContainSubstring("at most 64"))) + }) +}) + +var _ = Describe("systemOneBind", func() { + bind := func(body string) (int, error) { + e := echo.New() + r := httptest.NewRequest(http.MethodPost, "/v1/systemone", strings.NewReader(body)) + r.Header.Set("Content-Type", "application/json") + c := e.NewContext(r, httptest.NewRecorder()) + var out schema.SystemOneRequest + if err := systemOneBind(c, &out); err != nil { + return systemOneBindStatus(err), err + } + return http.StatusOK, nil + } + + It("binds a normal body", func() { + status, err := bind(`{"model":"m","state":"x","questions":{}}`) + Expect(err).ToNot(HaveOccurred()) + Expect(status).To(Equal(http.StatusOK)) + }) + + It("answers 413 for a body over 64 KiB", func() { + status, err := bind(`{"model":"m","state":"` + strings.Repeat("a", 65*1024) + `"}`) + Expect(err).To(HaveOccurred()) + Expect(status).To(Equal(http.StatusRequestEntityTooLarge)) + }) + + It("answers 400 for malformed JSON", func() { + status, err := bind(`{not json`) + Expect(err).To(HaveOccurred()) + Expect(status).To(Equal(http.StatusBadRequest)) + }) +}) diff --git a/core/http/react-ui/e2e/models-lifecycle.spec.js b/core/http/react-ui/e2e/models-lifecycle.spec.js index 3a8d125a5..0dff49857 100644 --- a/core/http/react-ui/e2e/models-lifecycle.spec.js +++ b/core/http/react-ui/e2e/models-lifecycle.spec.js @@ -172,6 +172,19 @@ test.describe('Models lifecycle', () => { await expect(installedPane(page)).toContainText('Worker one') }) + test('shows the decisions use case on a decision model', async ({ page }) => { + await page.route('**/api/models/capabilities', route => route.fulfill({ + contentType: 'application/json', + body: JSON.stringify({ + data: [...installedModels, { id: 'decider', backend: 'vllm-cpp', capabilities: ['FLAG_DECISIONS'] }], + }), + })) + await page.goto('/app/models?view=installed&model=decider') + + await expect(installedPane(page)).toContainText('decider') + await expect(installedPane(page)).toContainText('Decisions') + }) + test('stops a running model with confirmation', async ({ page }) => { await page.goto('/app/models?view=installed&model=alpha') diff --git a/core/http/react-ui/public/locales/de/models.json b/core/http/react-ui/public/locales/de/models.json index d88cb8c70..af487c86e 100644 --- a/core/http/react-ui/public/locales/de/models.json +++ b/core/http/react-ui/public/locales/de/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/en/models.json b/core/http/react-ui/public/locales/en/models.json index a2150e785..f60dc3051 100644 --- a/core/http/react-ui/public/locales/en/models.json +++ b/core/http/react-ui/public/locales/en/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/es/models.json b/core/http/react-ui/public/locales/es/models.json index 989189850..eddf6a0b4 100644 --- a/core/http/react-ui/public/locales/es/models.json +++ b/core/http/react-ui/public/locales/es/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/id/models.json b/core/http/react-ui/public/locales/id/models.json index 1dee74031..4ca0bcd2a 100644 --- a/core/http/react-ui/public/locales/id/models.json +++ b/core/http/react-ui/public/locales/id/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/it/models.json b/core/http/react-ui/public/locales/it/models.json index edcc1b587..b67d9da75 100644 --- a/core/http/react-ui/public/locales/it/models.json +++ b/core/http/react-ui/public/locales/it/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/ko/models.json b/core/http/react-ui/public/locales/ko/models.json index b2a20016e..e47a7da87 100644 --- a/core/http/react-ui/public/locales/ko/models.json +++ b/core/http/react-ui/public/locales/ko/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/pt-BR/models.json b/core/http/react-ui/public/locales/pt-BR/models.json index 26e567a44..3f0e5c97a 100644 --- a/core/http/react-ui/public/locales/pt-BR/models.json +++ b/core/http/react-ui/public/locales/pt-BR/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/zh-CN/models.json b/core/http/react-ui/public/locales/zh-CN/models.json index 40f78260e..524e49933 100644 --- a/core/http/react-ui/public/locales/zh-CN/models.json +++ b/core/http/react-ui/public/locales/zh-CN/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/src/pages/InstalledModels.jsx b/core/http/react-ui/src/pages/InstalledModels.jsx index 6013d984f..fce53a124 100644 --- a/core/http/react-ui/src/pages/InstalledModels.jsx +++ b/core/http/react-ui/src/pages/InstalledModels.jsx @@ -22,7 +22,7 @@ import { CAP_CHAT, CAP_COMPLETION, CAP_IMAGE, CAP_VIDEO, CAP_TTS, CAP_TRANSCRIPT, CAP_SOUND_GENERATION, CAP_FACE_RECOGNITION, CAP_SPEAKER_RECOGNITION, CAP_EMBEDDINGS, CAP_RERANK, - CAP_VAD, CAP_SCORE, + CAP_VAD, CAP_SCORE, CAP_DECISIONS, } from '../utils/capabilities' const USE_CASES = [ @@ -39,6 +39,7 @@ const USE_CASES = [ { cap: CAP_RERANK, labelKey: 'rerank' }, { cap: CAP_VAD, labelKey: 'vad' }, { cap: CAP_SCORE, labelKey: 'score' }, + { cap: CAP_DECISIONS, labelKey: 'decisions' }, ] export function modelUseCases(model) { diff --git a/core/http/react-ui/src/utils/capabilities.js b/core/http/react-ui/src/utils/capabilities.js index f01cc781c..722c85842 100644 --- a/core/http/react-ui/src/utils/capabilities.js +++ b/core/http/react-ui/src/utils/capabilities.js @@ -29,4 +29,5 @@ export const CAP_SPEAKER_RECOGNITION = 'FLAG_SPEAKER_RECOGNITION' export const CAP_AUDIO_TRANSFORM = 'FLAG_AUDIO_TRANSFORM' export const CAP_REALTIME_AUDIO = 'FLAG_REALTIME_AUDIO' export const CAP_SCORE = 'FLAG_SCORE' +export const CAP_DECISIONS = 'FLAG_DECISIONS' export const CAP_TOKEN_CLASSIFY = 'FLAG_TOKEN_CLASSIFY' diff --git a/docs/content/advanced/model-configuration.md b/docs/content/advanced/model-configuration.md index 5cfa74ccd..c1870e3ec 100644 --- a/docs/content/advanced/model-configuration.md +++ b/docs/content/advanced/model-configuration.md @@ -1066,7 +1066,9 @@ known_usecases: - embeddings ``` -Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `llm` (combination of CHAT, COMPLETION, EDIT). +Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `score`, `token_classify`, `decisions`, `llm` (combination of CHAT, COMPLETION, EDIT). + +`decisions` marks a model as a decision model for the [Decisions API]({{% relref "features/decisions" %}}) (`POST /v1/systemone`). It is never guessed, and a model that declares it is not listed as a chat, completion or embeddings model. `token_classify` marks a model as a token-classification (NER) provider for the PII filter (e.g. an `openai-privacy-filter` GGUF). Declare it explicitly together with `embeddings: true` (the classifier loads via TOKEN_CLS pooling). It runs on the dedicated `privacy-filter` backend (`backend/cpp/privacy-filter`), a standalone GGML engine for the `openai-privacy-filter` family - separate from `llama-cpp`, which no longer carries the token-classification path. diff --git a/docs/content/features/decisions.md b/docs/content/features/decisions.md new file mode 100644 index 000000000..0118f110d --- /dev/null +++ b/docs/content/features/decisions.md @@ -0,0 +1,153 @@ ++++ +disableToc = false +title = "Decisions API" +weight = 66 +url = "/features/decisions/" ++++ + +The Decisions API is a fast, typed decision layer. You send a piece of text (the +*state*) and a set of named questions. A decision model answers each question +with a value and a confidence, in one pass. The model does not generate text, so +there is nothing to parse and no free-form output to validate. + +LocalAI serves it on the `/v1/systemone` routes. The request and response shapes +follow the [kev](https://github.com/jaredpalmer/kev) project, and the field names +and question types are the same ones Ollama serves on its `/v1/systemone` +endpoint (Ollama 0.35 and later). The wire contract is called SystemOne; the +capability a model declares is called `decisions`. See +[Compatibility with Ollama](#compatibility-with-ollama) for what differs. + +OpenAI announced its own Decisions API in limited preview on 2026-09-29. It has no +public request or response schema yet, so LocalAI does not serve a `/v1/decisions` +route. + +## Endpoints + +| Endpoint | Method | Description | +|---|---|---| +| `/v1/systemone` | POST | Answer all questions in one pass | +| `/v1/systemone/permute` | POST | Re-run one choice question under `n_perm` option orders | +| `/v1/systemone/separate` | POST | Answer each question in its own pass | + +Which route a model can serve depends on its kind: + +| Model kind | `/v1/systemone` | `/permute` and `/separate` | +|---|---|---| +| Decision model (`decisions`), such as Laya or GLiNER2.5-Decide | Yes | No, returns `400` | +| Zero-shot NER model (`token_classify`), such as GLiNER2.5 | Yes, through the NER path | Yes | + +## Question types + +| Type | Answer | Fields in the answer | +|---|---|---| +| `choice` | One option out of a named set | `choice`, `probabilities`, `confidence` | +| `noul` | Yes, no or unknown for a statement | `noul` (0 to 1), `entities` | +| `score` | One level on a scale | `score`, `legend`, `probabilities`, `confidence` | + +## Example + +```bash +curl http://localhost:8080/v1/systemone -H "Content-Type: application/json" -d '{ + "model": "laya-vllm-cpp", + "state": "My order arrived broken and I want my money back. This is the second time.", + "questions": { + "team": { + "type": "choice", + "instructions": "Which team should handle this ticket?", + "criteria": { + "billing": "Payments, invoices and refunds", + "shipping": "Delivery and damaged goods", + "product": "Questions about how the product works" + } + }, + "refund_requested": { + "type": "noul", + "instructions": "The customer explicitly asks for a refund" + }, + "urgency": { + "type": "score", + "instructions": "How urgent is this ticket?", + "criteria": ["not urgent", "somewhat urgent", "urgent", "critical"] + } + } +}' +``` + +Answers from a decision model carry a `confidence` value, and the response +reports token usage and `latency_ms`. The NER path does not report token usage. + +## Choosing a model + +A model can serve the Decisions API only if it is a decision model. Declare the usecase +in the model config: + +```yaml +name: laya +backend: vllm-cpp +known_usecases: + - decisions +parameters: + model: convaiinnovations/laya +``` + +`decisions` is never guessed, and a model that declares it is not listed as a +chat, completion or embeddings model. A model that declares usecases without +`decisions` or `token_classify` gets a `400` from these endpoints that names the +missing usecase. A model that declares `token_classify` and not `decisions` is +served by the zero-shot NER path. A vllm-cpp config that declares no usecases is +treated as a decision model, so setups that predate the flag keep working, but a +config that declares only `chat` (as an older `laya` gallery entry did) now gets +the `400` and needs `known_usecases: [decisions]`. + +Install one from the gallery and filter on the `decisions` tag: + +| Gallery entry | Model | Notes | +|---|---|---| +| `laya-vllm-cpp` | Laya | ModernBERT-large, non-autoregressive, about 800 MB | +| `gliner25-decide-vllm-cpp` | GLiNER2.5-Decide | DeBERTa-v3-large with a classification head, about 2 GB | + +The engine, [vllm.cpp]({{% relref "features/vllm-cpp" %}}), also supports the +kev, CLM and xor decision models. Those checkpoints need a conversion step, so +they are not gallery entries yet. + +Tev1 is an autoregressive decision model. It answers through chat completions +and does not serve `/v1/systemone` yet. + +## Request limits + +A request is refused with `400` (or `413` for the body size) when: + +- the body is larger than 64 KiB, +- `state` is missing or blank, +- there are no questions, or more than 64, +- a question id is blank, +- a `choice` question has fewer than 2 options or a blank option key, +- a `score` question has fewer than 2 levels, +- a `noul` question has `criteria` with keys other than `"false"` and `"true"`. + +A `noul` question may carry `criteria` with a description for each outcome, for +example `{"false": "No refund is requested", "true": "The customer requests a refund"}`. +Some models cap the number of options for a `choice` or `score` question (models +that answer with a letter accept at most 26). The engine refuses more options than +the model supports and the error names the limit. + +## Compatibility with Ollama + +The field names, question types and answer fields are the same as Ollama's +`/v1/systemone`, so a client written for one works against the other for the +common case. These behaviors differ: + +| | Ollama | LocalAI | +|---|---|---| +| `confidence` | `1 - H(p) / ln(N)`, an entropy measure | Computed by the model's pipeline. For kev and Laya it is a normalized margin, so the same probabilities give a different value | +| Errors | `{"error": "message"}` | `{"error": {"message": "...", "type": "invalid_request"}}` | +| `keep_alive` | Sets how long the model stays loaded | Accepted and ignored. Model lifetime follows the LocalAI idle and watchdog settings | +| `state` given as an object | Serialized as JSON text | Rendered as labeled lines, the way kev does it | +| `noul` answer on the NER path | `{type, noul}` | Also carries `entities` | +| Token `usage` | Full prompt lengths across all questions | Whatever the backend reports; the NER path reports 0 | + +## Access control + +When authentication is on, the three routes need the `decisions` feature. It is +on by default for every user, like the other API features, and an administrator +can turn it off per user. diff --git a/docs/content/features/vllm-cpp.md b/docs/content/features/vllm-cpp.md index 74e3c0d38..bf7323e0d 100644 --- a/docs/content/features/vllm-cpp.md +++ b/docs/content/features/vllm-cpp.md @@ -160,22 +160,25 @@ forward, which is the required contract for pooling models in vllm.cpp. A device-resident forward is tracked as a performance optimization, not a correctness gap. -### SystemOne structured-extraction API +### Decisions API -The `vllm-cpp` backend also exposes kev-compatible SystemOne endpoints that -turn zero-shot NER into structured question answering. These mirror the API -from the [kev](https://github.com/jaredpalmer/kev) project: +The `vllm-cpp` backend serves the kev-compatible SystemOne endpoints (the Decisions API): typed +`choice`, `noul` and `score` questions over a state text, answered by a +non-generative decision model in one pass. A decision model declares +`known_usecases: [decisions]`. See [Decisions API]({{% relref "features/decisions" %}}) +for the request shape, the models you can install and the access rules. | Endpoint | Method | Description | |---|---|---| -| `/v1/systemone` | POST | Answer all questions in one NER pass | +| `/v1/systemone` | POST | Answer all questions in one pass | | `/v1/systemone/permute` | POST | Re-run one choice question under n_perm option orders | -| `/v1/systemone/separate` | POST | Answer each question in its own NER pass (N passes) | +| `/v1/systemone/separate` | POST | Answer each question in its own pass (N passes) | -Each question has a `type` of `noul` (binary entity presence), `choice` (pick -one option), or `score` (pick one level). The `model` field in the request body -selects the NER model. Labels are derived from the question definition, so no -`ner_labels` configuration is needed for these endpoints. +The GLiNER2.5 zero-shot NER model (`token_classify`) also serves +`/v1/systemone`, through the NER path, and it is the model to use for +`/v1/systemone/permute` and `/v1/systemone/separate`, which decision models +refuse with a `400`. It derives its NER labels from the question definitions, so +no `ner_labels` configuration is needed. ## Beyond text generation diff --git a/gallery/index.yaml b/gallery/index.yaml index 58c1daba9..197b5f306 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -19360,6 +19360,7 @@ - qwen3.6 - nvfp4 - vllm-cpp + - vision - tool-calling - reasoning - gpu @@ -19372,6 +19373,7 @@ known_usecases: - chat - completion + - vision # Tool calls and the split are parsed by the engine's own streaming # parsers, so LocalAI's Go-side grammar path stays out of the way. function: @@ -19429,6 +19431,7 @@ - qwen3.6 - nvfp4 - vllm-cpp + - vision - speculative-decoding - mtp - tool-calling @@ -19442,6 +19445,7 @@ known_usecases: - chat - completion + - vision function: grammar: disable: true @@ -19497,6 +19501,7 @@ - qwen3.6 - nvfp4 - vllm-cpp + - vision - speculative-decoding - dflash - tool-calling @@ -19510,6 +19515,7 @@ known_usecases: - chat - completion + - vision function: grammar: disable: true @@ -19557,6 +19563,9 @@ with roughly 3B parameters active per token, so it reads like a much larger model while costing about as much per token as a small one. + Image input is implemented in the engine but is not token-gated against + vLLM yet, so the vision usecase on this entry is experimental. + This is the engine's gated MoE checkpoint: token-for-token identical to vLLM over the 315-prompt battery on both the synchronous and asynchronous paths, at 0.92x to 0.97x vLLM's throughput from concurrency 1 to 32. @@ -19575,6 +19584,8 @@ - moe - nvfp4 - vllm-cpp + - vision + - experimental - tool-calling - reasoning - gpu @@ -19587,6 +19598,7 @@ known_usecases: - chat - completion + - vision function: grammar: disable: true @@ -19615,6 +19627,9 @@ description: | Qwen3.6-35B-A3B NVFP4 on vllm.cpp with MTP speculative decoding enabled. + Image input is implemented in the engine but is not token-gated against + vLLM yet, so the vision usecase on this entry is experimental. + The draft head ships inside the checkpoint's own mtp.* tensors, so there is no second model to download. On this model the speculative path is token-exact against speculation-off on both the synchronous and asynchronous @@ -19632,6 +19647,8 @@ - moe - nvfp4 - vllm-cpp + - vision + - experimental - speculative-decoding - mtp - tool-calling @@ -19645,6 +19662,7 @@ known_usecases: - chat - completion + - vision function: grammar: disable: true @@ -63635,7 +63653,7 @@ 512-token context. F16 weights, ~804 MB. license: apache-2.0 tags: - - decision + - decisions - systemone - vllm-cpp - cpu @@ -63645,7 +63663,7 @@ overrides: backend: vllm-cpp known_usecases: - - chat + - decisions parameters: model: convaiinnovations/laya artifacts: @@ -63654,6 +63672,90 @@ source: type: huggingface repo: convaiinnovations/laya +- name: gliner25-decide-vllm-cpp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/fastino/GLiNER2.5-Decide + - https://github.com/mudler/vllm.cpp + description: | + GLiNER2.5-Decide is a DeBERTa-v3-large encoder with a classification head + that answers typed decision questions over a state text in one forward + pass. It never generates text, so there is nothing to parse. + + In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine runs the + decision pipeline (choice, noul and score question types) through the + vllm_decide C ABI. This is the decision model, not the zero-shot NER model: + use the gliner2.5 entry for entity extraction. F32 weights, about 2 GB. + The weights are pinned to a revision so the entry keeps serving the + checkpoint it was checked against. + license: apache-2.0 + tags: + - decisions + - systemone + - vllm-cpp + - cpu + - gpu + size: 2GB + last_checked: "2026-09-30" + overrides: + backend: vllm-cpp + known_usecases: + - decisions + parameters: + model: fastino/GLiNER2.5-Decide + artifacts: + - name: model + target: model + source: + type: huggingface + repo: fastino/GLiNER2.5-Decide + revision: 5a7adf72a23b4d311abae6ce050d7f0012bb3416 +- name: qwen3-vl-4b-vllm-cpp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct + - https://github.com/mudler/vllm.cpp + description: | + Qwen3-VL-4B-Instruct on vllm.cpp, in bf16: a small vision-language model + that takes images alongside text. In the engine's correctness battery the + image path matches vLLM token for token, and video input is a near tie. + + Roughly 9 GB of weights plus KV cache at the context configured here. It + runs where the flagship NVFP4 checkpoints cannot, including plain CPU. + license: apache-2.0 + tags: + - llm + - vision + - multimodal + - qwen + - qwen3-vl + - vllm-cpp + - cpu + - gpu + size: 9GB + last_checked: "2026-09-30" + overrides: + backend: vllm-cpp + known_usecases: + - chat + - completion + - vision + template: + use_tokenizer_template: true + context_size: 8192 + engine_args: + block_size: 32 + num_blocks: 512 + max_num_seqs: 4 + parameters: + model: Qwen/Qwen3-VL-4B-Instruct + artifacts: + - name: model + target: model + source: + type: huggingface + repo: Qwen/Qwen3-VL-4B-Instruct + revision: ebb281ec70b05090aa6165b016eac8ec08e71b17 - name: cua-s1-forms-vllm-cpp url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: