mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-02 19:14:38 -04:00
Merge pull request #12373 from mudler/feat/systemone-capability
feat: decisions usecase for decision models, with gallery tagging
This commit is contained in:
commit
f378fe89d0
28 files changed
+846
-39
No files matched your search
@@ -35,6 +35,7 @@ const (
|
||||
UsecaseSpeakerRecognition = "speaker_recognition"
|
||||
UsecaseTokenClassify = "token_classify"
|
||||
UsecaseScore = "score"
|
||||
UsecaseDecisions = "decisions"
|
||||
)
|
||||
|
||||
// GRPCMethod identifies a Backend service RPC from backend.proto.
|
||||
@@ -216,6 +217,11 @@ var UsecaseInfoMap = map[string]UsecaseInfo{
|
||||
GRPCMethod: MethodScore,
|
||||
Description: "Joint log-probability scoring of candidate continuations via the Score RPC. Declared explicitly via known_usecases and usable alongside generation usecases.",
|
||||
},
|
||||
UsecaseDecisions: {
|
||||
Flag: FLAG_DECISIONS,
|
||||
GRPCMethod: MethodScore,
|
||||
Description: "Decision models (served by POST /v1/systemone): typed choice, noul and score questions over a state text, answered by a non-generative decision model through the Score RPC (question_type systemone). Declared explicitly via known_usecases.",
|
||||
},
|
||||
}
|
||||
|
||||
// BackendCapability describes which gRPC methods and usecases a backend supports.
|
||||
@@ -349,10 +355,10 @@ var BackendCapabilities = map[string]BackendCapability{
|
||||
// model returns an error rather than silent garbage.
|
||||
"vllm-cpp": {
|
||||
GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateVideo, MethodTokenClassify, MethodScore},
|
||||
PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVideo, UsecaseTokenClassify, UsecaseScore},
|
||||
PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVision, UsecaseVideo, UsecaseTokenClassify, UsecaseScore, UsecaseDecisions},
|
||||
DefaultUsecases: []string{UsecaseChat},
|
||||
AcceptsImages: true,
|
||||
Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and kev/laya decision pipelines",
|
||||
Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and decision models (kev, laya, CLM, GLiNER2.5-Decide, xor, nimble)",
|
||||
},
|
||||
"vllm-omni": {
|
||||
GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateImage, MethodGenerateVideo, MethodTTS},
|
||||
|
||||
+2
-2
@@ -16,14 +16,14 @@ import (
|
||||
|
||||
// reservedNonChatModel reports whether the operator reserved this model for an
|
||||
// internal primitive — the router score classifier or the PII NER
|
||||
// token_classify tier. Such a model has no chat template and must not be
|
||||
// token_classify tier, or a decision head. Such a model has no chat template and must not be
|
||||
// given the generative-chat defaults the GGUF importer otherwise applies
|
||||
// (FLAG_CHAT, jinja templating): surfacing it in chat pickers defeats the
|
||||
// reservation. Operators who do want a combined model declare both usecases
|
||||
// explicitly — the combination is valid.
|
||||
func reservedNonChatModel(cfg *ModelConfig) bool {
|
||||
return cfg.KnownUsecases != nil &&
|
||||
(*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY)) != 0
|
||||
(*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY|FLAG_DECISIONS)) != 0
|
||||
}
|
||||
|
||||
// genAudioEncoderKey is the mmproj metadata flag llama.cpp's mtmd writes for a
|
||||
|
||||
@@ -2056,6 +2056,13 @@ const (
|
||||
FLAG_3D ModelConfigUsecase = 0b100000000000000000000000
|
||||
FLAG_3D_ANIMATION ModelConfigUsecase = 1 << 24
|
||||
|
||||
// Marks a model as a decision model: it answers typed choice / noul /
|
||||
// score questions over a state (served by POST /v1/systemone).
|
||||
// Explicit only, like FLAG_SCORE: a decision model never generates
|
||||
// text, so guessing chat or embeddings for it would surface it in
|
||||
// pickers it cannot serve.
|
||||
FLAG_DECISIONS ModelConfigUsecase = 1 << 25
|
||||
|
||||
// Common Subsets
|
||||
FLAG_LLM ModelConfigUsecase = FLAG_CHAT | FLAG_COMPLETION | FLAG_EDIT
|
||||
)
|
||||
@@ -2118,6 +2125,7 @@ func GetAllModelConfigUsecases() map[string]ModelConfigUsecase {
|
||||
"FLAG_TOKEN_CLASSIFY": FLAG_TOKEN_CLASSIFY,
|
||||
"FLAG_3D": FLAG_3D,
|
||||
"FLAG_3D_ANIMATION": FLAG_3D_ANIMATION,
|
||||
"FLAG_DECISIONS": FLAG_DECISIONS,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2146,9 +2154,9 @@ func GetUsecasesFromYAML(input []string) *ModelConfigUsecase {
|
||||
//
|
||||
// Declared known_usecases are normally additive — the guessing heuristic
|
||||
// still adds whatever it can infer from backend/templates. The exceptions
|
||||
// are FLAG_SCORE and FLAG_TOKEN_CLASSIFY: when the operator declared
|
||||
// either, they reserved the model for an internal direct-decode primitive
|
||||
// (the router classifier, or the PII NER tier). Letting GuessUsecases
|
||||
// are FLAG_SCORE, FLAG_TOKEN_CLASSIFY and FLAG_DECISIONS: when the operator
|
||||
// declared any of them, they reserved the model for a direct-decode primitive
|
||||
// (the router classifier, the PII NER tier, or a decision head). Letting GuessUsecases
|
||||
// paint chat/completion/embeddings on top would surface it in pickers it
|
||||
// was deliberately kept out of. So a declared score or token_classify
|
||||
// list is authoritative; declare the generation usecases explicitly
|
||||
@@ -2158,7 +2166,7 @@ func (c *ModelConfig) HasUsecases(u ModelConfigUsecase) bool {
|
||||
if (u & *c.KnownUsecases) == u {
|
||||
return true
|
||||
}
|
||||
if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY)) != 0 {
|
||||
if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY | FLAG_DECISIONS)) != 0 {
|
||||
return false
|
||||
}
|
||||
}
|
||||
@@ -2381,6 +2389,14 @@ func (c *ModelConfig) GuessUsecases(u ModelConfigUsecase) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
if (u & FLAG_DECISIONS) == FLAG_DECISIONS {
|
||||
// No heuristic: decisions intent is a deliberate operator choice
|
||||
// (the model is a non-generative decision head), so
|
||||
// HasUsecases(FLAG_DECISIONS) is true only when KnownUsecases
|
||||
// declares it explicitly.
|
||||
return false
|
||||
}
|
||||
|
||||
return true
|
||||
}
|
||||
|
||||
|
||||
@@ -955,3 +955,36 @@ var _ = Describe("ModelConfig alias", func() {
|
||||
Expect(err).To(MatchError(ContainSubstring("alias")))
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("decisions usecase", func() {
|
||||
// A decision model never generates text, so a declared decisions list
|
||||
// must stay authoritative and the heuristic must never guess the flag.
|
||||
It("is authoritative when declared and never guessed", func() {
|
||||
declared := GetUsecasesFromYAML([]string{"decisions"})
|
||||
Expect(declared).NotTo(BeNil())
|
||||
Expect(*declared).NotTo(Equal(FLAG_ANY))
|
||||
|
||||
cfg := ModelConfig{
|
||||
Name: "laya",
|
||||
Backend: "vllm-cpp",
|
||||
KnownUsecases: declared,
|
||||
TemplateConfig: TemplateConfig{
|
||||
Chat: "inherited from chatml",
|
||||
ChatMessage: "inherited from chatml",
|
||||
Completion: "inherited from chatml",
|
||||
},
|
||||
}
|
||||
Expect(cfg.HasUsecases(*declared)).To(BeTrue())
|
||||
Expect(cfg.HasUsecases(FLAG_CHAT)).To(BeFalse())
|
||||
Expect(cfg.HasUsecases(FLAG_COMPLETION)).To(BeFalse())
|
||||
Expect(cfg.HasUsecases(FLAG_EMBEDDINGS)).To(BeFalse())
|
||||
|
||||
undeclared := ModelConfig{Name: "laya", Backend: "vllm-cpp"}
|
||||
Expect(undeclared.HasUsecases(*declared)).To(BeFalse())
|
||||
})
|
||||
|
||||
It("is a reserved usecase for the GGUF importer chat-default guard", func() {
|
||||
declared := GetUsecasesFromYAML([]string{"decisions"})
|
||||
Expect(reservedNonChatModel(&ModelConfig{Backend: "vllm-cpp", KnownUsecases: declared})).To(BeTrue())
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,48 @@
|
||||
package gallery_test
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"slices"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
)
|
||||
|
||||
// A gallery tag that names a capability is what users filter on, and
|
||||
// known_usecases is what the server routes on. When they disagree, the entry
|
||||
// is listed under a filter it cannot serve, or is hidden from one it can.
|
||||
var _ = Describe("gallery/index.yaml vllm-cpp capability tags", func() {
|
||||
It("keeps capability tags and known_usecases in agreement", func() {
|
||||
entries, err := loadGalleryIndex()
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
|
||||
tagToFlag := map[string]config.ModelConfigUsecase{
|
||||
"decisions": config.FLAG_DECISIONS,
|
||||
"vision": config.FLAG_VISION,
|
||||
"token-classify": config.FLAG_TOKEN_CLASSIFY,
|
||||
"scoring": config.FLAG_SCORE,
|
||||
}
|
||||
|
||||
var violations []string
|
||||
seen := 0
|
||||
for i := range entries {
|
||||
e := &entries[i]
|
||||
if backend, _ := e.Overrides["backend"].(string); backend != "vllm-cpp" {
|
||||
continue
|
||||
}
|
||||
seen++
|
||||
declared := e.GetKnownUsecases()
|
||||
for tag, flag := range tagToFlag {
|
||||
tagged := slices.Contains(e.Tags, tag)
|
||||
has := declared != nil && *declared&flag == flag
|
||||
if tagged != has {
|
||||
violations = append(violations, fmt.Sprintf("%s: tag %q present=%v but known_usecases declares it=%v", e.Name, tag, tagged, has))
|
||||
}
|
||||
}
|
||||
}
|
||||
Expect(seen).To(BeNumerically(">", 0))
|
||||
Expect(violations).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
@@ -71,6 +71,11 @@ var RouteFeatureRegistry = []RouteFeature{
|
||||
// Detection
|
||||
{"POST", "/v1/detection", FeatureDetection},
|
||||
|
||||
// Decisions API (SystemOne wire contract)
|
||||
{"POST", "/v1/systemone", FeatureDecisions},
|
||||
{"POST", "/v1/systemone/permute", FeatureDecisions},
|
||||
{"POST", "/v1/systemone/separate", FeatureDecisions},
|
||||
|
||||
// Face recognition
|
||||
{"POST", "/v1/face/verify", FeatureFaceRecognition},
|
||||
{"POST", "/v1/face/analyze", FeatureFaceRecognition},
|
||||
@@ -209,5 +214,6 @@ func APIFeatureMetas() []FeatureMeta {
|
||||
{FeatureVoiceRecognition, "Voice Recognition", true},
|
||||
{FeatureAudioTransform, "Audio Transform", true},
|
||||
{FeaturePIIFilter, "PII Analyze / Redact", true},
|
||||
{FeatureDecisions, "Decisions", true},
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
package auth_test
|
||||
|
||||
import (
|
||||
. "github.com/mudler/LocalAI/core/http/auth"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("Decisions feature registration", func() {
|
||||
It("gates the three decision routes behind one default-on API feature", func() {
|
||||
Expect(APIFeatures).To(ContainElement(FeatureDecisions))
|
||||
|
||||
patterns := []string{}
|
||||
for _, route := range RouteFeatureRegistry {
|
||||
if route.Feature == FeatureDecisions {
|
||||
Expect(route.Method).To(Equal("POST"))
|
||||
patterns = append(patterns, route.Pattern)
|
||||
}
|
||||
}
|
||||
Expect(patterns).To(ConsistOf("/v1/systemone", "/v1/systemone/permute", "/v1/systemone/separate"))
|
||||
|
||||
Expect(APIFeatureMetas()).To(ContainElement(FeatureMeta{Key: FeatureDecisions, Label: "Decisions", DefaultValue: true}))
|
||||
})
|
||||
})
|
||||
@@ -59,6 +59,7 @@ const (
|
||||
FeatureFaceRecognition = "face_recognition"
|
||||
FeatureVoiceRecognition = "voice_recognition"
|
||||
FeatureAudioTransform = "audio_transform"
|
||||
FeatureDecisions = "decisions"
|
||||
// FeaturePIIFilter gates the synchronous PII analyze/redact service
|
||||
// (POST /api/pii/{analyze,redact}). Default ON like the other API
|
||||
// features; the admin-only events log is gated separately in-handler.
|
||||
@@ -78,7 +79,7 @@ var APIFeatures = []string{
|
||||
FeatureVAD, FeatureDetection, FeatureVideo, Feature3D, FeatureEmbeddings, FeatureSound,
|
||||
FeatureRealtime, FeatureModeration, FeatureRerank, FeatureTokenize, FeatureMCP, FeatureStores,
|
||||
FeatureFaceRecognition, FeatureVoiceRecognition, FeatureAudioTransform,
|
||||
FeaturePIIFilter,
|
||||
FeaturePIIFilter, FeatureDecisions,
|
||||
}
|
||||
|
||||
// AllFeatures lists all known features (used by UI and validation).
|
||||
|
||||
@@ -105,6 +105,12 @@ var instructionDefs = []instructionDef{
|
||||
Tags: []string{"voice-recognition"},
|
||||
Intro: "Voice (speaker) recognition — the audio analog to /v1/face/*. Use /v1/voice/verify for 1:1 speaker comparison, /v1/voice/identify for 1:N match against the registered store, /v1/voice/{register,forget} to manage that store, /v1/voice/embed for a raw speaker-encoder vector, and /v1/voice/analyze for age / gender / emotion inferred from speech. Registrations are in-memory by default and lost on restart. Audio inputs accept URL, base64, or data-URI; /v1/embeddings remains text-only.",
|
||||
},
|
||||
{
|
||||
Name: "decisions",
|
||||
Description: "Typed decisions (choice, noul, score) over a state text with calibrated confidence",
|
||||
Tags: []string{"systemone"},
|
||||
Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { <id>: { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [decisions] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. Field names and question types follow Ollama's /v1/systemone, with differences in confidence, error shape and keep_alive (see the Decisions API docs). A request over 64 KiB, with more than 64 questions, or with a malformed question is refused.",
|
||||
},
|
||||
{
|
||||
Name: "branding",
|
||||
Description: "Whitelabel the instance: configure name, tagline, logo, and favicon",
|
||||
|
||||
@@ -39,7 +39,7 @@ var _ = Describe("API Instructions Endpoints", func() {
|
||||
|
||||
instructions, ok := resp["instructions"].([]any)
|
||||
Expect(ok).To(BeTrue())
|
||||
Expect(instructions).To(HaveLen(20))
|
||||
Expect(instructions).To(HaveLen(21))
|
||||
|
||||
// Verify each instruction has required fields and correct URL format
|
||||
for _, s := range instructions {
|
||||
@@ -82,6 +82,7 @@ var _ = Describe("API Instructions Endpoints", func() {
|
||||
"voice-library",
|
||||
"3d",
|
||||
"failover",
|
||||
"decisions",
|
||||
))
|
||||
})
|
||||
})
|
||||
@@ -136,6 +137,17 @@ var _ = Describe("API Instructions Endpoints", func() {
|
||||
Expect(string(body)).NotTo(ContainSubstring("/v1/3d/generations"))
|
||||
})
|
||||
|
||||
It("should advertise the Decisions API", func() {
|
||||
req := httptest.NewRequest(http.MethodGet, "/api/instructions/decisions", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
app.ServeHTTP(rec, req)
|
||||
|
||||
Expect(rec.Code).To(Equal(http.StatusOK))
|
||||
body, _ := io.ReadAll(rec.Body)
|
||||
Expect(string(body)).To(ContainSubstring("POST /v1/systemone"))
|
||||
Expect(string(body)).To(ContainSubstring("known_usecases: [decisions]"))
|
||||
})
|
||||
|
||||
It("should return JSON fragment when format=json", func() {
|
||||
req := httptest.NewRequest(http.MethodGet, "/api/instructions/chat-inference?format=json", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
|
||||
@@ -2,6 +2,7 @@ package localai
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
"math/rand"
|
||||
@@ -371,6 +372,191 @@ func systemOneError(c echo.Context, status int, msg string) error {
|
||||
})
|
||||
}
|
||||
|
||||
// systemOneModelAllowed keeps chat and embedding models out of the decision
|
||||
// API with an actionable error instead of a backend failure. A config that
|
||||
// declares no usecases predates the flag and stays allowed, and a
|
||||
// token_classify model is allowed because the NER path serves it.
|
||||
func systemOneModelAllowed(cfg config.ModelConfig) error {
|
||||
if cfg.KnownUsecases == nil {
|
||||
return nil
|
||||
}
|
||||
if *cfg.KnownUsecases&(config.FLAG_DECISIONS|config.FLAG_TOKEN_CLASSIFY) != 0 {
|
||||
return nil
|
||||
}
|
||||
return fmt.Errorf("model %q does not declare the decisions usecase (known_usecases: [decisions])", cfg.Name)
|
||||
}
|
||||
|
||||
// checkSystemOneModel applies systemOneModelAllowed to a model looked up by
|
||||
// name. An unknown model passes here so the existing not-found handling
|
||||
// downstream keeps its status code.
|
||||
func checkSystemOneModel(app *application.Application, modelName string) error {
|
||||
cl := app.ModelConfigLoader()
|
||||
if cl == nil {
|
||||
return nil
|
||||
}
|
||||
cfg, ok := cl.GetModelConfig(modelName)
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
return systemOneModelAllowed(cfg)
|
||||
}
|
||||
|
||||
// systemOneUsesDecisionPipeline reports whether /v1/systemone forwards the
|
||||
// request to the backend's Score RPC (the decision pipeline) for this model.
|
||||
// A model that declares token_classify without systemone is a zero-shot NER
|
||||
// model: the backend's decision entry point refuses those architectures, so it
|
||||
// goes to the NER path instead. A config that declares nothing keeps the
|
||||
// decision pipeline, which is what setups that predate the decisions usecase
|
||||
// relied on.
|
||||
func systemOneUsesDecisionPipeline(cfg config.ModelConfig) bool {
|
||||
if !backendSupportsScore(cfg.Backend) {
|
||||
return false
|
||||
}
|
||||
if cfg.KnownUsecases == nil {
|
||||
return true
|
||||
}
|
||||
declared := *cfg.KnownUsecases
|
||||
if declared&config.FLAG_DECISIONS != 0 {
|
||||
return true
|
||||
}
|
||||
return declared&config.FLAG_TOKEN_CLASSIFY == 0
|
||||
}
|
||||
|
||||
// systemOneNERAllowed guards /permute and /separate, which always run the NER
|
||||
// path. A decision model cannot serve them: the backend's NER entry point
|
||||
// refuses its architecture, and the caller would see a backend error.
|
||||
func systemOneNERAllowed(cfg config.ModelConfig) error {
|
||||
if cfg.KnownUsecases == nil {
|
||||
return nil
|
||||
}
|
||||
declared := *cfg.KnownUsecases
|
||||
if declared&config.FLAG_DECISIONS != 0 && declared&config.FLAG_TOKEN_CLASSIFY == 0 {
|
||||
return fmt.Errorf("model %q is a decision model: /permute and /separate use the NER path, use POST /v1/systemone instead", cfg.Name)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// checkSystemOneNERModel applies systemOneNERAllowed to a model looked up by
|
||||
// name; an unknown model passes so the not-found handling keeps its status.
|
||||
func checkSystemOneNERModel(app *application.Application, modelName string) error {
|
||||
cl := app.ModelConfigLoader()
|
||||
if cl == nil {
|
||||
return nil
|
||||
}
|
||||
cfg, ok := cl.GetModelConfig(modelName)
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
return systemOneNERAllowed(cfg)
|
||||
}
|
||||
|
||||
// systemOneMaxBody and systemOneMaxQuestions bound one request. They keep a
|
||||
// single call from pinning a decision model on an unbounded prompt, and match
|
||||
// the limits Ollama documents for the same wire contract, so a client written
|
||||
// for one server behaves the same on the other. The engine enforces any
|
||||
// per-model option cap (letter-answer models refuse more than 26 options).
|
||||
const (
|
||||
systemOneMaxBody = 64 << 10
|
||||
systemOneMaxQuestions = 64
|
||||
)
|
||||
|
||||
// systemOneBind binds the JSON body with a size cap. Bind reads the whole body
|
||||
// first, so the cap has to be on the reader.
|
||||
func systemOneBind(c echo.Context, v any) error {
|
||||
c.Request().Body = http.MaxBytesReader(c.Response(), c.Request().Body, systemOneMaxBody)
|
||||
return c.Bind(v)
|
||||
}
|
||||
|
||||
// systemOneBindStatus maps a bind failure to its status: 413 when the body
|
||||
// exceeded the cap, 400 for anything else.
|
||||
func systemOneBindStatus(err error) int {
|
||||
var tooLarge *http.MaxBytesError
|
||||
if errors.As(err, &tooLarge) {
|
||||
return http.StatusRequestEntityTooLarge
|
||||
}
|
||||
return http.StatusBadRequest
|
||||
}
|
||||
|
||||
func systemOneBindMessage(err error) string {
|
||||
if systemOneBindStatus(err) == http.StatusRequestEntityTooLarge {
|
||||
return fmt.Sprintf("request body exceeds %d KiB", systemOneMaxBody>>10)
|
||||
}
|
||||
return "invalid request body"
|
||||
}
|
||||
|
||||
// validateSystemOneRequest checks the structure every path needs, before the
|
||||
// request is forwarded to a decision model or run through the NER path. The
|
||||
// forwarded path never sees parseSystemOneRequest, so without this a malformed
|
||||
// question would surface as a backend error instead of a 400.
|
||||
func validateSystemOneRequest(req *schema.SystemOneRequest) error {
|
||||
if len(req.State) == 0 || string(req.State) == "null" {
|
||||
return fmt.Errorf("state is required")
|
||||
}
|
||||
var state any
|
||||
if err := json.Unmarshal(req.State, &state); err != nil {
|
||||
return fmt.Errorf("state is not valid JSON: %w", err)
|
||||
}
|
||||
if s, ok := state.(string); ok && strings.TrimSpace(s) == "" {
|
||||
return fmt.Errorf("state is required")
|
||||
}
|
||||
if len(req.Questions) == 0 {
|
||||
return fmt.Errorf("questions is required and must contain at least one question")
|
||||
}
|
||||
if len(req.Questions) > systemOneMaxQuestions {
|
||||
return fmt.Errorf("questions must contain at most %d questions", systemOneMaxQuestions)
|
||||
}
|
||||
qids := make([]string, 0, len(req.Questions))
|
||||
for id := range req.Questions {
|
||||
qids = append(qids, id)
|
||||
}
|
||||
sort.Strings(qids)
|
||||
for _, id := range qids {
|
||||
if strings.TrimSpace(id) == "" {
|
||||
return fmt.Errorf("question ids must not be blank")
|
||||
}
|
||||
q := req.Questions[id]
|
||||
switch q.Type {
|
||||
case "choice":
|
||||
var criteria map[string]json.RawMessage
|
||||
if err := json.Unmarshal(q.Criteria, &criteria); err != nil {
|
||||
return fmt.Errorf("question %q (choice) requires a criteria object", id)
|
||||
}
|
||||
if len(criteria) < 2 {
|
||||
return fmt.Errorf("question %q (choice) requires at least 2 options", id)
|
||||
}
|
||||
for k := range criteria {
|
||||
if strings.TrimSpace(k) == "" {
|
||||
return fmt.Errorf("question %q (choice) has a blank option key", id)
|
||||
}
|
||||
}
|
||||
case "score":
|
||||
var criteria []json.RawMessage
|
||||
if err := json.Unmarshal(q.Criteria, &criteria); err != nil {
|
||||
return fmt.Errorf("question %q (score) requires a criteria array", id)
|
||||
}
|
||||
if len(criteria) < 2 {
|
||||
return fmt.Errorf("question %q (score) requires at least 2 levels", id)
|
||||
}
|
||||
case "noul":
|
||||
if len(q.Criteria) == 0 || string(q.Criteria) == "null" {
|
||||
continue
|
||||
}
|
||||
var criteria map[string]json.RawMessage
|
||||
if err := json.Unmarshal(q.Criteria, &criteria); err != nil {
|
||||
return fmt.Errorf("question %q (noul) criteria must be an object with \"false\" and \"true\" descriptions", id)
|
||||
}
|
||||
for k := range criteria {
|
||||
if k != "false" && k != "true" {
|
||||
return fmt.Errorf("question %q (noul) criteria may only have \"false\" and \"true\" keys", id)
|
||||
}
|
||||
}
|
||||
default:
|
||||
return fmt.Errorf("question %q has unknown type: %s", id, q.Type)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// backendSupportsScore reports whether the named backend implements the
|
||||
// Score gRPC RPC. vllm-cpp does (kev/laya decision pipeline and cua-s1-forms
|
||||
// scoring via the unified vllm_decide C ABI); other backends fall through to
|
||||
@@ -402,18 +588,24 @@ func backendSupportsScore(backendName string) bool {
|
||||
func SystemOneEndpoint(app *application.Application) echo.HandlerFunc {
|
||||
return func(c echo.Context) error {
|
||||
var req schema.SystemOneRequest
|
||||
if err := c.Bind(&req); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, "invalid request body")
|
||||
if err := systemOneBind(c, &req); err != nil {
|
||||
return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err))
|
||||
}
|
||||
if req.Model == "" {
|
||||
return systemOneError(c, http.StatusBadRequest, "model is required")
|
||||
}
|
||||
if err := checkSystemOneModel(app, req.Model); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
}
|
||||
if err := validateSystemOneRequest(&req); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
}
|
||||
// vllm-cpp models (kev/laya) implement the decision pipeline natively
|
||||
// via the vllm_decide C ABI. Forward the raw request JSON through the
|
||||
// Score RPC and return the backend's response as-is.
|
||||
cl := app.ModelConfigLoader()
|
||||
if cl != nil {
|
||||
if cfg, ok := cl.GetModelConfig(req.Model); ok && backendSupportsScore(cfg.Backend) {
|
||||
if cfg, ok := cl.GetModelConfig(req.Model); ok && systemOneUsesDecisionPipeline(cfg) {
|
||||
reqJSON, err := json.Marshal(req)
|
||||
if err != nil {
|
||||
return systemOneError(c, http.StatusInternalServerError, "failed to marshal request: "+err.Error())
|
||||
@@ -468,12 +660,21 @@ func SystemOneEndpoint(app *application.Application) echo.HandlerFunc {
|
||||
func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc {
|
||||
return func(c echo.Context) error {
|
||||
var req schema.SystemOnePermuteRequest
|
||||
if err := c.Bind(&req); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, "invalid request body")
|
||||
if err := systemOneBind(c, &req); err != nil {
|
||||
return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err))
|
||||
}
|
||||
if req.Request.Model == "" {
|
||||
return systemOneError(c, http.StatusBadRequest, "model is required")
|
||||
}
|
||||
if err := checkSystemOneModel(app, req.Request.Model); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
}
|
||||
if err := checkSystemOneNERModel(app, req.Request.Model); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
}
|
||||
if err := validateSystemOneRequest(&req.Request); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
}
|
||||
if req.Question == "" {
|
||||
return systemOneError(c, http.StatusBadRequest, "question is required")
|
||||
}
|
||||
@@ -604,12 +805,21 @@ func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc {
|
||||
func SystemOneSeparateEndpoint(app *application.Application) echo.HandlerFunc {
|
||||
return func(c echo.Context) error {
|
||||
var req schema.SystemOneRequest
|
||||
if err := c.Bind(&req); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, "invalid request body")
|
||||
if err := systemOneBind(c, &req); err != nil {
|
||||
return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err))
|
||||
}
|
||||
if req.Model == "" {
|
||||
return systemOneError(c, http.StatusBadRequest, "model is required")
|
||||
}
|
||||
if err := checkSystemOneModel(app, req.Model); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
}
|
||||
if err := checkSystemOneNERModel(app, req.Model); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
}
|
||||
if err := validateSystemOneRequest(&req); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
}
|
||||
parsed, err := parseSystemOneRequest(&req)
|
||||
if err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
package localai
|
||||
|
||||
import (
|
||||
"github.com/mudler/LocalAI/core/config"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("systemOneModelAllowed", func() {
|
||||
mk := func(usecases ...string) config.ModelConfig {
|
||||
return config.ModelConfig{
|
||||
Name: "m",
|
||||
Backend: "vllm-cpp",
|
||||
KnownUsecases: config.GetUsecasesFromYAML(usecases),
|
||||
}
|
||||
}
|
||||
|
||||
It("accepts a declared decisions model", func() {
|
||||
Expect(systemOneModelAllowed(mk("decisions"))).To(Succeed())
|
||||
})
|
||||
|
||||
It("accepts a token_classify model, which the NER path serves", func() {
|
||||
Expect(systemOneModelAllowed(mk("token_classify"))).To(Succeed())
|
||||
})
|
||||
|
||||
It("keeps configs that declare no usecases working", func() {
|
||||
Expect(systemOneModelAllowed(config.ModelConfig{Name: "laya", Backend: "vllm-cpp"})).To(Succeed())
|
||||
})
|
||||
|
||||
It("refuses a chat-only model with an actionable message", func() {
|
||||
Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [decisions]")))
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("systemone routing by model kind", func() {
|
||||
mk := func(backend string, usecases ...string) config.ModelConfig {
|
||||
c := config.ModelConfig{Name: "m", Backend: backend}
|
||||
if len(usecases) > 0 {
|
||||
c.KnownUsecases = config.GetUsecasesFromYAML(usecases)
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
Describe("systemOneUsesDecisionPipeline", func() {
|
||||
It("sends a declared decision model to the decision pipeline", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions"))).To(BeTrue())
|
||||
})
|
||||
It("sends a token_classify model to the NER path, since vllm_decide refuses NER architectures", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "token_classify"))).To(BeFalse())
|
||||
})
|
||||
It("keeps configs that declare nothing on the decision pipeline", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp"))).To(BeTrue())
|
||||
})
|
||||
It("prefers the decision pipeline when both usecases are declared", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions", "token_classify"))).To(BeTrue())
|
||||
})
|
||||
It("never uses it for a backend without the Score RPC", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("no-such-backend", "decisions"))).To(BeFalse())
|
||||
})
|
||||
})
|
||||
|
||||
Describe("systemOneNERAllowed", func() {
|
||||
It("refuses a decision model on the NER-only routes with an actionable message", func() {
|
||||
Expect(systemOneNERAllowed(mk("vllm-cpp", "decisions"))).To(MatchError(ContainSubstring("/v1/systemone")))
|
||||
})
|
||||
It("accepts a token_classify model", func() {
|
||||
Expect(systemOneNERAllowed(mk("vllm-cpp", "token_classify"))).To(Succeed())
|
||||
})
|
||||
It("accepts configs that declare nothing", func() {
|
||||
Expect(systemOneNERAllowed(mk("vllm-cpp"))).To(Succeed())
|
||||
})
|
||||
})
|
||||
})
|
||||
@@ -0,0 +1,96 @@
|
||||
package localai
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
|
||||
"github.com/labstack/echo/v4"
|
||||
"github.com/mudler/LocalAI/core/schema"
|
||||
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("validateSystemOneRequest", func() {
|
||||
req := func(state string, questions string) *schema.SystemOneRequest {
|
||||
r := &schema.SystemOneRequest{Model: "m", State: json.RawMessage(state)}
|
||||
Expect(json.Unmarshal([]byte(questions), &r.Questions)).To(Succeed())
|
||||
return r
|
||||
}
|
||||
|
||||
It("accepts the three question types", func() {
|
||||
r := req(`"ticket text"`, `{
|
||||
"team": {"type":"choice","instructions":"which","criteria":{"a":"A","b":null}},
|
||||
"refund": {"type":"noul","instructions":"refund?","criteria":{"false":"No refund","true":"Refund asked"}},
|
||||
"urgency": {"type":"score","instructions":"how urgent","criteria":["low","high"]}
|
||||
}`)
|
||||
Expect(validateSystemOneRequest(r)).To(Succeed())
|
||||
})
|
||||
|
||||
It("accepts a noul question with no criteria", func() {
|
||||
Expect(validateSystemOneRequest(req(`"x"`, `{"q":{"type":"noul","instructions":"i"}}`))).To(Succeed())
|
||||
})
|
||||
|
||||
DescribeTable("refuses a malformed request with a message that names the problem",
|
||||
func(state, questions, want string) {
|
||||
Expect(validateSystemOneRequest(req(state, questions))).To(MatchError(ContainSubstring(want)))
|
||||
},
|
||||
Entry("missing state", ``, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"),
|
||||
Entry("null state", `null`, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"),
|
||||
Entry("blank string state", `" "`, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"),
|
||||
Entry("no questions", `"x"`, `{}`, "at least one question"),
|
||||
Entry("blank question id", `"x"`, `{" ":{"type":"noul","instructions":"i"}}`, "blank"),
|
||||
Entry("unknown type", `"x"`, `{"q":{"type":"rank","instructions":"i"}}`, "unknown type"),
|
||||
Entry("choice with one option", `"x"`, `{"q":{"type":"choice","instructions":"i","criteria":{"a":"A"}}}`, "at least 2"),
|
||||
Entry("choice with a blank option key", `"x"`, `{"q":{"type":"choice","instructions":"i","criteria":{"a":"A"," ":"B"}}}`, "blank"),
|
||||
Entry("score with one level", `"x"`, `{"q":{"type":"score","instructions":"i","criteria":["only"]}}`, "at least 2"),
|
||||
Entry("noul criteria with a stray key", `"x"`, `{"q":{"type":"noul","instructions":"i","criteria":{"maybe":"M"}}}`, `"false" and "true"`),
|
||||
)
|
||||
|
||||
It("refuses more than 64 questions", func() {
|
||||
var b strings.Builder
|
||||
b.WriteString("{")
|
||||
for i := 0; i < 65; i++ {
|
||||
if i > 0 {
|
||||
b.WriteString(",")
|
||||
}
|
||||
b.WriteString(`"q` + strings.Repeat("x", i) + `":{"type":"noul","instructions":"i"}`)
|
||||
}
|
||||
b.WriteString("}")
|
||||
Expect(validateSystemOneRequest(req(`"x"`, b.String()))).To(MatchError(ContainSubstring("at most 64")))
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("systemOneBind", func() {
|
||||
bind := func(body string) (int, error) {
|
||||
e := echo.New()
|
||||
r := httptest.NewRequest(http.MethodPost, "/v1/systemone", strings.NewReader(body))
|
||||
r.Header.Set("Content-Type", "application/json")
|
||||
c := e.NewContext(r, httptest.NewRecorder())
|
||||
var out schema.SystemOneRequest
|
||||
if err := systemOneBind(c, &out); err != nil {
|
||||
return systemOneBindStatus(err), err
|
||||
}
|
||||
return http.StatusOK, nil
|
||||
}
|
||||
|
||||
It("binds a normal body", func() {
|
||||
status, err := bind(`{"model":"m","state":"x","questions":{}}`)
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(status).To(Equal(http.StatusOK))
|
||||
})
|
||||
|
||||
It("answers 413 for a body over 64 KiB", func() {
|
||||
status, err := bind(`{"model":"m","state":"` + strings.Repeat("a", 65*1024) + `"}`)
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(status).To(Equal(http.StatusRequestEntityTooLarge))
|
||||
})
|
||||
|
||||
It("answers 400 for malformed JSON", func() {
|
||||
status, err := bind(`{not json`)
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(status).To(Equal(http.StatusBadRequest))
|
||||
})
|
||||
})
|
||||
@@ -172,6 +172,19 @@ test.describe('Models lifecycle', () => {
|
||||
await expect(installedPane(page)).toContainText('Worker one')
|
||||
})
|
||||
|
||||
test('shows the decisions use case on a decision model', async ({ page }) => {
|
||||
await page.route('**/api/models/capabilities', route => route.fulfill({
|
||||
contentType: 'application/json',
|
||||
body: JSON.stringify({
|
||||
data: [...installedModels, { id: 'decider', backend: 'vllm-cpp', capabilities: ['FLAG_DECISIONS'] }],
|
||||
}),
|
||||
}))
|
||||
await page.goto('/app/models?view=installed&model=decider')
|
||||
|
||||
await expect(installedPane(page)).toContainText('decider')
|
||||
await expect(installedPane(page)).toContainText('Decisions')
|
||||
})
|
||||
|
||||
test('stops a running model with confirmation', async ({ page }) => {
|
||||
await page.goto('/app/models?view=installed&model=alpha')
|
||||
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -22,7 +22,7 @@ import {
|
||||
CAP_CHAT, CAP_COMPLETION, CAP_IMAGE, CAP_VIDEO, CAP_TTS,
|
||||
CAP_TRANSCRIPT, CAP_SOUND_GENERATION, CAP_FACE_RECOGNITION,
|
||||
CAP_SPEAKER_RECOGNITION, CAP_EMBEDDINGS, CAP_RERANK,
|
||||
CAP_VAD, CAP_SCORE,
|
||||
CAP_VAD, CAP_SCORE, CAP_DECISIONS,
|
||||
} from '../utils/capabilities'
|
||||
|
||||
const USE_CASES = [
|
||||
@@ -39,6 +39,7 @@ const USE_CASES = [
|
||||
{ cap: CAP_RERANK, labelKey: 'rerank' },
|
||||
{ cap: CAP_VAD, labelKey: 'vad' },
|
||||
{ cap: CAP_SCORE, labelKey: 'score' },
|
||||
{ cap: CAP_DECISIONS, labelKey: 'decisions' },
|
||||
]
|
||||
|
||||
export function modelUseCases(model) {
|
||||
|
||||
@@ -29,4 +29,5 @@ export const CAP_SPEAKER_RECOGNITION = 'FLAG_SPEAKER_RECOGNITION'
|
||||
export const CAP_AUDIO_TRANSFORM = 'FLAG_AUDIO_TRANSFORM'
|
||||
export const CAP_REALTIME_AUDIO = 'FLAG_REALTIME_AUDIO'
|
||||
export const CAP_SCORE = 'FLAG_SCORE'
|
||||
export const CAP_DECISIONS = 'FLAG_DECISIONS'
|
||||
export const CAP_TOKEN_CLASSIFY = 'FLAG_TOKEN_CLASSIFY'
|
||||
@@ -1066,7 +1066,9 @@ known_usecases:
|
||||
- embeddings
|
||||
```
|
||||
|
||||
Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `llm` (combination of CHAT, COMPLETION, EDIT).
|
||||
Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `score`, `token_classify`, `decisions`, `llm` (combination of CHAT, COMPLETION, EDIT).
|
||||
|
||||
`decisions` marks a model as a decision model for the [Decisions API]({{% relref "features/decisions" %}}) (`POST /v1/systemone`). It is never guessed, and a model that declares it is not listed as a chat, completion or embeddings model.
|
||||
|
||||
`token_classify` marks a model as a token-classification (NER) provider for the PII filter (e.g. an `openai-privacy-filter` GGUF). Declare it explicitly together with `embeddings: true` (the classifier loads via TOKEN_CLS pooling). It runs on the dedicated `privacy-filter` backend (`backend/cpp/privacy-filter`), a standalone GGML engine for the `openai-privacy-filter` family - separate from `llama-cpp`, which no longer carries the token-classification path.
|
||||
|
||||
|
||||
@@ -0,0 +1,153 @@
|
||||
+++
|
||||
disableToc = false
|
||||
title = "Decisions API"
|
||||
weight = 66
|
||||
url = "/features/decisions/"
|
||||
+++
|
||||
|
||||
The Decisions API is a fast, typed decision layer. You send a piece of text (the
|
||||
*state*) and a set of named questions. A decision model answers each question
|
||||
with a value and a confidence, in one pass. The model does not generate text, so
|
||||
there is nothing to parse and no free-form output to validate.
|
||||
|
||||
LocalAI serves it on the `/v1/systemone` routes. The request and response shapes
|
||||
follow the [kev](https://github.com/jaredpalmer/kev) project, and the field names
|
||||
and question types are the same ones Ollama serves on its `/v1/systemone`
|
||||
endpoint (Ollama 0.35 and later). The wire contract is called SystemOne; the
|
||||
capability a model declares is called `decisions`. See
|
||||
[Compatibility with Ollama](#compatibility-with-ollama) for what differs.
|
||||
|
||||
OpenAI announced its own Decisions API in limited preview on 2026-09-29. It has no
|
||||
public request or response schema yet, so LocalAI does not serve a `/v1/decisions`
|
||||
route.
|
||||
|
||||
## Endpoints
|
||||
|
||||
| Endpoint | Method | Description |
|
||||
|---|---|---|
|
||||
| `/v1/systemone` | POST | Answer all questions in one pass |
|
||||
| `/v1/systemone/permute` | POST | Re-run one choice question under `n_perm` option orders |
|
||||
| `/v1/systemone/separate` | POST | Answer each question in its own pass |
|
||||
|
||||
Which route a model can serve depends on its kind:
|
||||
|
||||
| Model kind | `/v1/systemone` | `/permute` and `/separate` |
|
||||
|---|---|---|
|
||||
| Decision model (`decisions`), such as Laya or GLiNER2.5-Decide | Yes | No, returns `400` |
|
||||
| Zero-shot NER model (`token_classify`), such as GLiNER2.5 | Yes, through the NER path | Yes |
|
||||
|
||||
## Question types
|
||||
|
||||
| Type | Answer | Fields in the answer |
|
||||
|---|---|---|
|
||||
| `choice` | One option out of a named set | `choice`, `probabilities`, `confidence` |
|
||||
| `noul` | Yes, no or unknown for a statement | `noul` (0 to 1), `entities` |
|
||||
| `score` | One level on a scale | `score`, `legend`, `probabilities`, `confidence` |
|
||||
|
||||
## Example
|
||||
|
||||
```bash
|
||||
curl http://localhost:8080/v1/systemone -H "Content-Type: application/json" -d '{
|
||||
"model": "laya-vllm-cpp",
|
||||
"state": "My order arrived broken and I want my money back. This is the second time.",
|
||||
"questions": {
|
||||
"team": {
|
||||
"type": "choice",
|
||||
"instructions": "Which team should handle this ticket?",
|
||||
"criteria": {
|
||||
"billing": "Payments, invoices and refunds",
|
||||
"shipping": "Delivery and damaged goods",
|
||||
"product": "Questions about how the product works"
|
||||
}
|
||||
},
|
||||
"refund_requested": {
|
||||
"type": "noul",
|
||||
"instructions": "The customer explicitly asks for a refund"
|
||||
},
|
||||
"urgency": {
|
||||
"type": "score",
|
||||
"instructions": "How urgent is this ticket?",
|
||||
"criteria": ["not urgent", "somewhat urgent", "urgent", "critical"]
|
||||
}
|
||||
}
|
||||
}'
|
||||
```
|
||||
|
||||
Answers from a decision model carry a `confidence` value, and the response
|
||||
reports token usage and `latency_ms`. The NER path does not report token usage.
|
||||
|
||||
## Choosing a model
|
||||
|
||||
A model can serve the Decisions API only if it is a decision model. Declare the usecase
|
||||
in the model config:
|
||||
|
||||
```yaml
|
||||
name: laya
|
||||
backend: vllm-cpp
|
||||
known_usecases:
|
||||
- decisions
|
||||
parameters:
|
||||
model: convaiinnovations/laya
|
||||
```
|
||||
|
||||
`decisions` is never guessed, and a model that declares it is not listed as a
|
||||
chat, completion or embeddings model. A model that declares usecases without
|
||||
`decisions` or `token_classify` gets a `400` from these endpoints that names the
|
||||
missing usecase. A model that declares `token_classify` and not `decisions` is
|
||||
served by the zero-shot NER path. A vllm-cpp config that declares no usecases is
|
||||
treated as a decision model, so setups that predate the flag keep working, but a
|
||||
config that declares only `chat` (as an older `laya` gallery entry did) now gets
|
||||
the `400` and needs `known_usecases: [decisions]`.
|
||||
|
||||
Install one from the gallery and filter on the `decisions` tag:
|
||||
|
||||
| Gallery entry | Model | Notes |
|
||||
|---|---|---|
|
||||
| `laya-vllm-cpp` | Laya | ModernBERT-large, non-autoregressive, about 800 MB |
|
||||
| `gliner25-decide-vllm-cpp` | GLiNER2.5-Decide | DeBERTa-v3-large with a classification head, about 2 GB |
|
||||
|
||||
The engine, [vllm.cpp]({{% relref "features/vllm-cpp" %}}), also supports the
|
||||
kev, CLM and xor decision models. Those checkpoints need a conversion step, so
|
||||
they are not gallery entries yet.
|
||||
|
||||
Tev1 is an autoregressive decision model. It answers through chat completions
|
||||
and does not serve `/v1/systemone` yet.
|
||||
|
||||
## Request limits
|
||||
|
||||
A request is refused with `400` (or `413` for the body size) when:
|
||||
|
||||
- the body is larger than 64 KiB,
|
||||
- `state` is missing or blank,
|
||||
- there are no questions, or more than 64,
|
||||
- a question id is blank,
|
||||
- a `choice` question has fewer than 2 options or a blank option key,
|
||||
- a `score` question has fewer than 2 levels,
|
||||
- a `noul` question has `criteria` with keys other than `"false"` and `"true"`.
|
||||
|
||||
A `noul` question may carry `criteria` with a description for each outcome, for
|
||||
example `{"false": "No refund is requested", "true": "The customer requests a refund"}`.
|
||||
Some models cap the number of options for a `choice` or `score` question (models
|
||||
that answer with a letter accept at most 26). The engine refuses more options than
|
||||
the model supports and the error names the limit.
|
||||
|
||||
## Compatibility with Ollama
|
||||
|
||||
The field names, question types and answer fields are the same as Ollama's
|
||||
`/v1/systemone`, so a client written for one works against the other for the
|
||||
common case. These behaviors differ:
|
||||
|
||||
| | Ollama | LocalAI |
|
||||
|---|---|---|
|
||||
| `confidence` | `1 - H(p) / ln(N)`, an entropy measure | Computed by the model's pipeline. For kev and Laya it is a normalized margin, so the same probabilities give a different value |
|
||||
| Errors | `{"error": "message"}` | `{"error": {"message": "...", "type": "invalid_request"}}` |
|
||||
| `keep_alive` | Sets how long the model stays loaded | Accepted and ignored. Model lifetime follows the LocalAI idle and watchdog settings |
|
||||
| `state` given as an object | Serialized as JSON text | Rendered as labeled lines, the way kev does it |
|
||||
| `noul` answer on the NER path | `{type, noul}` | Also carries `entities` |
|
||||
| Token `usage` | Full prompt lengths across all questions | Whatever the backend reports; the NER path reports 0 |
|
||||
|
||||
## Access control
|
||||
|
||||
When authentication is on, the three routes need the `decisions` feature. It is
|
||||
on by default for every user, like the other API features, and an administrator
|
||||
can turn it off per user.
|
||||
@@ -160,22 +160,25 @@ forward, which is the required contract for pooling models in vllm.cpp. A
|
||||
device-resident forward is tracked as a performance optimization, not a
|
||||
correctness gap.
|
||||
|
||||
### SystemOne structured-extraction API
|
||||
### Decisions API
|
||||
|
||||
The `vllm-cpp` backend also exposes kev-compatible SystemOne endpoints that
|
||||
turn zero-shot NER into structured question answering. These mirror the API
|
||||
from the [kev](https://github.com/jaredpalmer/kev) project:
|
||||
The `vllm-cpp` backend serves the kev-compatible SystemOne endpoints (the Decisions API): typed
|
||||
`choice`, `noul` and `score` questions over a state text, answered by a
|
||||
non-generative decision model in one pass. A decision model declares
|
||||
`known_usecases: [decisions]`. See [Decisions API]({{% relref "features/decisions" %}})
|
||||
for the request shape, the models you can install and the access rules.
|
||||
|
||||
| Endpoint | Method | Description |
|
||||
|---|---|---|
|
||||
| `/v1/systemone` | POST | Answer all questions in one NER pass |
|
||||
| `/v1/systemone` | POST | Answer all questions in one pass |
|
||||
| `/v1/systemone/permute` | POST | Re-run one choice question under n_perm option orders |
|
||||
| `/v1/systemone/separate` | POST | Answer each question in its own NER pass (N passes) |
|
||||
| `/v1/systemone/separate` | POST | Answer each question in its own pass (N passes) |
|
||||
|
||||
Each question has a `type` of `noul` (binary entity presence), `choice` (pick
|
||||
one option), or `score` (pick one level). The `model` field in the request body
|
||||
selects the NER model. Labels are derived from the question definition, so no
|
||||
`ner_labels` configuration is needed for these endpoints.
|
||||
The GLiNER2.5 zero-shot NER model (`token_classify`) also serves
|
||||
`/v1/systemone`, through the NER path, and it is the model to use for
|
||||
`/v1/systemone/permute` and `/v1/systemone/separate`, which decision models
|
||||
refuse with a `400`. It derives its NER labels from the question definitions, so
|
||||
no `ner_labels` configuration is needed.
|
||||
|
||||
## Beyond text generation
|
||||
|
||||
|
||||
+104
-2
@@ -19360,6 +19360,7 @@
|
||||
- qwen3.6
|
||||
- nvfp4
|
||||
- vllm-cpp
|
||||
- vision
|
||||
- tool-calling
|
||||
- reasoning
|
||||
- gpu
|
||||
@@ -19372,6 +19373,7 @@
|
||||
known_usecases:
|
||||
- chat
|
||||
- completion
|
||||
- vision
|
||||
# Tool calls and the <think> split are parsed by the engine's own streaming
|
||||
# parsers, so LocalAI's Go-side grammar path stays out of the way.
|
||||
function:
|
||||
@@ -19429,6 +19431,7 @@
|
||||
- qwen3.6
|
||||
- nvfp4
|
||||
- vllm-cpp
|
||||
- vision
|
||||
- speculative-decoding
|
||||
- mtp
|
||||
- tool-calling
|
||||
@@ -19442,6 +19445,7 @@
|
||||
known_usecases:
|
||||
- chat
|
||||
- completion
|
||||
- vision
|
||||
function:
|
||||
grammar:
|
||||
disable: true
|
||||
@@ -19497,6 +19501,7 @@
|
||||
- qwen3.6
|
||||
- nvfp4
|
||||
- vllm-cpp
|
||||
- vision
|
||||
- speculative-decoding
|
||||
- dflash
|
||||
- tool-calling
|
||||
@@ -19510,6 +19515,7 @@
|
||||
known_usecases:
|
||||
- chat
|
||||
- completion
|
||||
- vision
|
||||
function:
|
||||
grammar:
|
||||
disable: true
|
||||
@@ -19557,6 +19563,9 @@
|
||||
with roughly 3B parameters active per token, so it reads like a much larger
|
||||
model while costing about as much per token as a small one.
|
||||
|
||||
Image input is implemented in the engine but is not token-gated against
|
||||
vLLM yet, so the vision usecase on this entry is experimental.
|
||||
|
||||
This is the engine's gated MoE checkpoint: token-for-token identical to vLLM
|
||||
over the 315-prompt battery on both the synchronous and asynchronous paths,
|
||||
at 0.92x to 0.97x vLLM's throughput from concurrency 1 to 32.
|
||||
@@ -19575,6 +19584,8 @@
|
||||
- moe
|
||||
- nvfp4
|
||||
- vllm-cpp
|
||||
- vision
|
||||
- experimental
|
||||
- tool-calling
|
||||
- reasoning
|
||||
- gpu
|
||||
@@ -19587,6 +19598,7 @@
|
||||
known_usecases:
|
||||
- chat
|
||||
- completion
|
||||
- vision
|
||||
function:
|
||||
grammar:
|
||||
disable: true
|
||||
@@ -19615,6 +19627,9 @@
|
||||
description: |
|
||||
Qwen3.6-35B-A3B NVFP4 on vllm.cpp with MTP speculative decoding enabled.
|
||||
|
||||
Image input is implemented in the engine but is not token-gated against
|
||||
vLLM yet, so the vision usecase on this entry is experimental.
|
||||
|
||||
The draft head ships inside the checkpoint's own mtp.* tensors, so there is
|
||||
no second model to download. On this model the speculative path is
|
||||
token-exact against speculation-off on both the synchronous and asynchronous
|
||||
@@ -19632,6 +19647,8 @@
|
||||
- moe
|
||||
- nvfp4
|
||||
- vllm-cpp
|
||||
- vision
|
||||
- experimental
|
||||
- speculative-decoding
|
||||
- mtp
|
||||
- tool-calling
|
||||
@@ -19645,6 +19662,7 @@
|
||||
known_usecases:
|
||||
- chat
|
||||
- completion
|
||||
- vision
|
||||
function:
|
||||
grammar:
|
||||
disable: true
|
||||
@@ -63635,7 +63653,7 @@
|
||||
512-token context. F16 weights, ~804 MB.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- decision
|
||||
- decisions
|
||||
- systemone
|
||||
- vllm-cpp
|
||||
- cpu
|
||||
@@ -63645,7 +63663,7 @@
|
||||
overrides:
|
||||
backend: vllm-cpp
|
||||
known_usecases:
|
||||
- chat
|
||||
- decisions
|
||||
parameters:
|
||||
model: convaiinnovations/laya
|
||||
artifacts:
|
||||
@@ -63654,6 +63672,90 @@
|
||||
source:
|
||||
type: huggingface
|
||||
repo: convaiinnovations/laya
|
||||
- name: gliner25-decide-vllm-cpp
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
- https://huggingface.co/fastino/GLiNER2.5-Decide
|
||||
- https://github.com/mudler/vllm.cpp
|
||||
description: |
|
||||
GLiNER2.5-Decide is a DeBERTa-v3-large encoder with a classification head
|
||||
that answers typed decision questions over a state text in one forward
|
||||
pass. It never generates text, so there is nothing to parse.
|
||||
|
||||
In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine runs the
|
||||
decision pipeline (choice, noul and score question types) through the
|
||||
vllm_decide C ABI. This is the decision model, not the zero-shot NER model:
|
||||
use the gliner2.5 entry for entity extraction. F32 weights, about 2 GB.
|
||||
The weights are pinned to a revision so the entry keeps serving the
|
||||
checkpoint it was checked against.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- decisions
|
||||
- systemone
|
||||
- vllm-cpp
|
||||
- cpu
|
||||
- gpu
|
||||
size: 2GB
|
||||
last_checked: "2026-09-30"
|
||||
overrides:
|
||||
backend: vllm-cpp
|
||||
known_usecases:
|
||||
- decisions
|
||||
parameters:
|
||||
model: fastino/GLiNER2.5-Decide
|
||||
artifacts:
|
||||
- name: model
|
||||
target: model
|
||||
source:
|
||||
type: huggingface
|
||||
repo: fastino/GLiNER2.5-Decide
|
||||
revision: 5a7adf72a23b4d311abae6ce050d7f0012bb3416
|
||||
- name: qwen3-vl-4b-vllm-cpp
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
- https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct
|
||||
- https://github.com/mudler/vllm.cpp
|
||||
description: |
|
||||
Qwen3-VL-4B-Instruct on vllm.cpp, in bf16: a small vision-language model
|
||||
that takes images alongside text. In the engine's correctness battery the
|
||||
image path matches vLLM token for token, and video input is a near tie.
|
||||
|
||||
Roughly 9 GB of weights plus KV cache at the context configured here. It
|
||||
runs where the flagship NVFP4 checkpoints cannot, including plain CPU.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- llm
|
||||
- vision
|
||||
- multimodal
|
||||
- qwen
|
||||
- qwen3-vl
|
||||
- vllm-cpp
|
||||
- cpu
|
||||
- gpu
|
||||
size: 9GB
|
||||
last_checked: "2026-09-30"
|
||||
overrides:
|
||||
backend: vllm-cpp
|
||||
known_usecases:
|
||||
- chat
|
||||
- completion
|
||||
- vision
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
context_size: 8192
|
||||
engine_args:
|
||||
block_size: 32
|
||||
num_blocks: 512
|
||||
max_num_seqs: 4
|
||||
parameters:
|
||||
model: Qwen/Qwen3-VL-4B-Instruct
|
||||
artifacts:
|
||||
- name: model
|
||||
target: model
|
||||
source:
|
||||
type: huggingface
|
||||
repo: Qwen/Qwen3-VL-4B-Instruct
|
||||
revision: ebb281ec70b05090aa6165b016eac8ec08e71b17
|
||||
- name: cua-s1-forms-vllm-cpp
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
|
||||
Reference in new issue
Block a user