mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-03 03:24:34 -04:00
refactor: name the capability decisions instead of systemone
The usecase describes what a model can do, and the category is the Decisions API. SystemOne stays as the wire contract: the /v1/systemone routes, the Score RPC question_type and the swagger tag are unchanged. The usecase, flag, auth feature, UI label, gallery tags and docs page are now decisions. Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
This commit is contained in:
1 parent
b3d65fd538
commit
70ce62901f
27 files changed
+91
-89
No files matched your search
@@ -35,7 +35,7 @@ const (
|
||||
UsecaseSpeakerRecognition = "speaker_recognition"
|
||||
UsecaseTokenClassify = "token_classify"
|
||||
UsecaseScore = "score"
|
||||
UsecaseSystemOne = "systemone"
|
||||
UsecaseDecisions = "decisions"
|
||||
)
|
||||
|
||||
// GRPCMethod identifies a Backend service RPC from backend.proto.
|
||||
@@ -217,10 +217,10 @@ var UsecaseInfoMap = map[string]UsecaseInfo{
|
||||
GRPCMethod: MethodScore,
|
||||
Description: "Joint log-probability scoring of candidate continuations via the Score RPC. Declared explicitly via known_usecases and usable alongside generation usecases.",
|
||||
},
|
||||
UsecaseSystemOne: {
|
||||
Flag: FLAG_SYSTEMONE,
|
||||
UsecaseDecisions: {
|
||||
Flag: FLAG_DECISIONS,
|
||||
GRPCMethod: MethodScore,
|
||||
Description: "SystemOne decision API (POST /v1/systemone): typed choice, noul and score questions over a state text, answered by a non-generative decision model through the Score RPC (question_type systemone). Declared explicitly via known_usecases.",
|
||||
Description: "Decision models (served by POST /v1/systemone): typed choice, noul and score questions over a state text, answered by a non-generative decision model through the Score RPC (question_type systemone). Declared explicitly via known_usecases.",
|
||||
},
|
||||
}
|
||||
|
||||
@@ -355,10 +355,10 @@ var BackendCapabilities = map[string]BackendCapability{
|
||||
// model returns an error rather than silent garbage.
|
||||
"vllm-cpp": {
|
||||
GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateVideo, MethodTokenClassify, MethodScore},
|
||||
PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVision, UsecaseVideo, UsecaseTokenClassify, UsecaseScore, UsecaseSystemOne},
|
||||
PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVision, UsecaseVideo, UsecaseTokenClassify, UsecaseScore, UsecaseDecisions},
|
||||
DefaultUsecases: []string{UsecaseChat},
|
||||
AcceptsImages: true,
|
||||
Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and SystemOne decision models (kev, laya, CLM, GLiNER2.5-Decide, xor, nimble)",
|
||||
Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and decision models (kev, laya, CLM, GLiNER2.5-Decide, xor, nimble)",
|
||||
},
|
||||
"vllm-omni": {
|
||||
GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateImage, MethodGenerateVideo, MethodTTS},
|
||||
|
||||
+2
-2
@@ -16,14 +16,14 @@ import (
|
||||
|
||||
// reservedNonChatModel reports whether the operator reserved this model for an
|
||||
// internal primitive — the router score classifier or the PII NER
|
||||
// token_classify tier, or a SystemOne decision head. Such a model has no chat template and must not be
|
||||
// token_classify tier, or a decision head. Such a model has no chat template and must not be
|
||||
// given the generative-chat defaults the GGUF importer otherwise applies
|
||||
// (FLAG_CHAT, jinja templating): surfacing it in chat pickers defeats the
|
||||
// reservation. Operators who do want a combined model declare both usecases
|
||||
// explicitly — the combination is valid.
|
||||
func reservedNonChatModel(cfg *ModelConfig) bool {
|
||||
return cfg.KnownUsecases != nil &&
|
||||
(*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY|FLAG_SYSTEMONE)) != 0
|
||||
(*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY|FLAG_DECISIONS)) != 0
|
||||
}
|
||||
|
||||
// genAudioEncoderKey is the mmproj metadata flag llama.cpp's mtmd writes for a
|
||||
|
||||
+10
-10
@@ -2056,12 +2056,12 @@ const (
|
||||
FLAG_3D ModelConfigUsecase = 0b100000000000000000000000
|
||||
FLAG_3D_ANIMATION ModelConfigUsecase = 1 << 24
|
||||
|
||||
// Marks a model as wired for the SystemOne decision API (POST
|
||||
// /v1/systemone: typed choice / noul / score questions over a state).
|
||||
// Marks a model as a decision model: it answers typed choice / noul /
|
||||
// score questions over a state (served by POST /v1/systemone).
|
||||
// Explicit only, like FLAG_SCORE: a decision model never generates
|
||||
// text, so guessing chat or embeddings for it would surface it in
|
||||
// pickers it cannot serve.
|
||||
FLAG_SYSTEMONE ModelConfigUsecase = 1 << 25
|
||||
FLAG_DECISIONS ModelConfigUsecase = 1 << 25
|
||||
|
||||
// Common Subsets
|
||||
FLAG_LLM ModelConfigUsecase = FLAG_CHAT | FLAG_COMPLETION | FLAG_EDIT
|
||||
@@ -2125,7 +2125,7 @@ func GetAllModelConfigUsecases() map[string]ModelConfigUsecase {
|
||||
"FLAG_TOKEN_CLASSIFY": FLAG_TOKEN_CLASSIFY,
|
||||
"FLAG_3D": FLAG_3D,
|
||||
"FLAG_3D_ANIMATION": FLAG_3D_ANIMATION,
|
||||
"FLAG_SYSTEMONE": FLAG_SYSTEMONE,
|
||||
"FLAG_DECISIONS": FLAG_DECISIONS,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2154,9 +2154,9 @@ func GetUsecasesFromYAML(input []string) *ModelConfigUsecase {
|
||||
//
|
||||
// Declared known_usecases are normally additive — the guessing heuristic
|
||||
// still adds whatever it can infer from backend/templates. The exceptions
|
||||
// are FLAG_SCORE, FLAG_TOKEN_CLASSIFY and FLAG_SYSTEMONE: when the operator
|
||||
// are FLAG_SCORE, FLAG_TOKEN_CLASSIFY and FLAG_DECISIONS: when the operator
|
||||
// declared any of them, they reserved the model for a direct-decode primitive
|
||||
// (the router classifier, the PII NER tier, or a SystemOne decision head). Letting GuessUsecases
|
||||
// (the router classifier, the PII NER tier, or a decision head). Letting GuessUsecases
|
||||
// paint chat/completion/embeddings on top would surface it in pickers it
|
||||
// was deliberately kept out of. So a declared score or token_classify
|
||||
// list is authoritative; declare the generation usecases explicitly
|
||||
@@ -2166,7 +2166,7 @@ func (c *ModelConfig) HasUsecases(u ModelConfigUsecase) bool {
|
||||
if (u & *c.KnownUsecases) == u {
|
||||
return true
|
||||
}
|
||||
if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY | FLAG_SYSTEMONE)) != 0 {
|
||||
if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY | FLAG_DECISIONS)) != 0 {
|
||||
return false
|
||||
}
|
||||
}
|
||||
@@ -2389,10 +2389,10 @@ func (c *ModelConfig) GuessUsecases(u ModelConfigUsecase) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
if (u & FLAG_SYSTEMONE) == FLAG_SYSTEMONE {
|
||||
// No heuristic: SystemOne intent is a deliberate operator choice
|
||||
if (u & FLAG_DECISIONS) == FLAG_DECISIONS {
|
||||
// No heuristic: decisions intent is a deliberate operator choice
|
||||
// (the model is a non-generative decision head), so
|
||||
// HasUsecases(FLAG_SYSTEMONE) is true only when KnownUsecases
|
||||
// HasUsecases(FLAG_DECISIONS) is true only when KnownUsecases
|
||||
// declares it explicitly.
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -956,11 +956,11 @@ var _ = Describe("ModelConfig alias", func() {
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("systemone usecase", func() {
|
||||
// A decision model never generates text, so a declared systemone list
|
||||
var _ = Describe("decisions usecase", func() {
|
||||
// A decision model never generates text, so a declared decisions list
|
||||
// must stay authoritative and the heuristic must never guess the flag.
|
||||
It("is authoritative when declared and never guessed", func() {
|
||||
declared := GetUsecasesFromYAML([]string{"systemone"})
|
||||
declared := GetUsecasesFromYAML([]string{"decisions"})
|
||||
Expect(declared).NotTo(BeNil())
|
||||
Expect(*declared).NotTo(Equal(FLAG_ANY))
|
||||
|
||||
@@ -984,7 +984,7 @@ var _ = Describe("systemone usecase", func() {
|
||||
})
|
||||
|
||||
It("is a reserved usecase for the GGUF importer chat-default guard", func() {
|
||||
declared := GetUsecasesFromYAML([]string{"systemone"})
|
||||
declared := GetUsecasesFromYAML([]string{"decisions"})
|
||||
Expect(reservedNonChatModel(&ModelConfig{Backend: "vllm-cpp", KnownUsecases: declared})).To(BeTrue())
|
||||
})
|
||||
})
|
||||
@@ -19,7 +19,7 @@ var _ = Describe("gallery/index.yaml vllm-cpp capability tags", func() {
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
|
||||
tagToFlag := map[string]config.ModelConfigUsecase{
|
||||
"systemone": config.FLAG_SYSTEMONE,
|
||||
"decisions": config.FLAG_DECISIONS,
|
||||
"vision": config.FLAG_VISION,
|
||||
"token-classify": config.FLAG_TOKEN_CLASSIFY,
|
||||
"scoring": config.FLAG_SCORE,
|
||||
|
||||
@@ -71,10 +71,10 @@ var RouteFeatureRegistry = []RouteFeature{
|
||||
// Detection
|
||||
{"POST", "/v1/detection", FeatureDetection},
|
||||
|
||||
// SystemOne decision API
|
||||
{"POST", "/v1/systemone", FeatureSystemOne},
|
||||
{"POST", "/v1/systemone/permute", FeatureSystemOne},
|
||||
{"POST", "/v1/systemone/separate", FeatureSystemOne},
|
||||
// Decisions API (SystemOne wire contract)
|
||||
{"POST", "/v1/systemone", FeatureDecisions},
|
||||
{"POST", "/v1/systemone/permute", FeatureDecisions},
|
||||
{"POST", "/v1/systemone/separate", FeatureDecisions},
|
||||
|
||||
// Face recognition
|
||||
{"POST", "/v1/face/verify", FeatureFaceRecognition},
|
||||
@@ -214,6 +214,6 @@ func APIFeatureMetas() []FeatureMeta {
|
||||
{FeatureVoiceRecognition, "Voice Recognition", true},
|
||||
{FeatureAudioTransform, "Audio Transform", true},
|
||||
{FeaturePIIFilter, "PII Analyze / Redact", true},
|
||||
{FeatureSystemOne, "SystemOne Decisions", true},
|
||||
{FeatureDecisions, "Decisions", true},
|
||||
}
|
||||
}
|
||||
+4
-4
@@ -6,19 +6,19 @@ import (
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("SystemOne feature registration", func() {
|
||||
var _ = Describe("Decisions feature registration", func() {
|
||||
It("gates the three decision routes behind one default-on API feature", func() {
|
||||
Expect(APIFeatures).To(ContainElement(FeatureSystemOne))
|
||||
Expect(APIFeatures).To(ContainElement(FeatureDecisions))
|
||||
|
||||
patterns := []string{}
|
||||
for _, route := range RouteFeatureRegistry {
|
||||
if route.Feature == FeatureSystemOne {
|
||||
if route.Feature == FeatureDecisions {
|
||||
Expect(route.Method).To(Equal("POST"))
|
||||
patterns = append(patterns, route.Pattern)
|
||||
}
|
||||
}
|
||||
Expect(patterns).To(ConsistOf("/v1/systemone", "/v1/systemone/permute", "/v1/systemone/separate"))
|
||||
|
||||
Expect(APIFeatureMetas()).To(ContainElement(FeatureMeta{Key: FeatureSystemOne, Label: "SystemOne Decisions", DefaultValue: true}))
|
||||
Expect(APIFeatureMetas()).To(ContainElement(FeatureMeta{Key: FeatureDecisions, Label: "Decisions", DefaultValue: true}))
|
||||
})
|
||||
})
|
||||
@@ -59,7 +59,7 @@ const (
|
||||
FeatureFaceRecognition = "face_recognition"
|
||||
FeatureVoiceRecognition = "voice_recognition"
|
||||
FeatureAudioTransform = "audio_transform"
|
||||
FeatureSystemOne = "systemone"
|
||||
FeatureDecisions = "decisions"
|
||||
// FeaturePIIFilter gates the synchronous PII analyze/redact service
|
||||
// (POST /api/pii/{analyze,redact}). Default ON like the other API
|
||||
// features; the admin-only events log is gated separately in-handler.
|
||||
@@ -79,7 +79,7 @@ var APIFeatures = []string{
|
||||
FeatureVAD, FeatureDetection, FeatureVideo, Feature3D, FeatureEmbeddings, FeatureSound,
|
||||
FeatureRealtime, FeatureModeration, FeatureRerank, FeatureTokenize, FeatureMCP, FeatureStores,
|
||||
FeatureFaceRecognition, FeatureVoiceRecognition, FeatureAudioTransform,
|
||||
FeaturePIIFilter, FeatureSystemOne,
|
||||
FeaturePIIFilter, FeatureDecisions,
|
||||
}
|
||||
|
||||
// AllFeatures lists all known features (used by UI and validation).
|
||||
|
||||
@@ -106,10 +106,10 @@ var instructionDefs = []instructionDef{
|
||||
Intro: "Voice (speaker) recognition — the audio analog to /v1/face/*. Use /v1/voice/verify for 1:1 speaker comparison, /v1/voice/identify for 1:N match against the registered store, /v1/voice/{register,forget} to manage that store, /v1/voice/embed for a raw speaker-encoder vector, and /v1/voice/analyze for age / gender / emotion inferred from speech. Registrations are in-memory by default and lost on restart. Audio inputs accept URL, base64, or data-URI; /v1/embeddings remains text-only.",
|
||||
},
|
||||
{
|
||||
Name: "systemone",
|
||||
Name: "decisions",
|
||||
Description: "Typed decisions (choice, noul, score) over a state text with calibrated confidence",
|
||||
Tags: []string{"systemone"},
|
||||
Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { <id>: { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [systemone] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. The wire shape matches Ollama's /v1/systemone.",
|
||||
Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { <id>: { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [decisions] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. The wire shape matches Ollama's /v1/systemone.",
|
||||
},
|
||||
{
|
||||
Name: "branding",
|
||||
|
||||
@@ -82,7 +82,7 @@ var _ = Describe("API Instructions Endpoints", func() {
|
||||
"voice-library",
|
||||
"3d",
|
||||
"failover",
|
||||
"systemone",
|
||||
"decisions",
|
||||
))
|
||||
})
|
||||
})
|
||||
@@ -137,15 +137,15 @@ var _ = Describe("API Instructions Endpoints", func() {
|
||||
Expect(string(body)).NotTo(ContainSubstring("/v1/3d/generations"))
|
||||
})
|
||||
|
||||
It("should advertise the SystemOne decisions API", func() {
|
||||
req := httptest.NewRequest(http.MethodGet, "/api/instructions/systemone", nil)
|
||||
It("should advertise the Decisions API", func() {
|
||||
req := httptest.NewRequest(http.MethodGet, "/api/instructions/decisions", nil)
|
||||
rec := httptest.NewRecorder()
|
||||
app.ServeHTTP(rec, req)
|
||||
|
||||
Expect(rec.Code).To(Equal(http.StatusOK))
|
||||
body, _ := io.ReadAll(rec.Body)
|
||||
Expect(string(body)).To(ContainSubstring("POST /v1/systemone"))
|
||||
Expect(string(body)).To(ContainSubstring("known_usecases: [systemone]"))
|
||||
Expect(string(body)).To(ContainSubstring("known_usecases: [decisions]"))
|
||||
})
|
||||
|
||||
It("should return JSON fragment when format=json", func() {
|
||||
|
||||
@@ -379,10 +379,10 @@ func systemOneModelAllowed(cfg config.ModelConfig) error {
|
||||
if cfg.KnownUsecases == nil {
|
||||
return nil
|
||||
}
|
||||
if *cfg.KnownUsecases&(config.FLAG_SYSTEMONE|config.FLAG_TOKEN_CLASSIFY) != 0 {
|
||||
if *cfg.KnownUsecases&(config.FLAG_DECISIONS|config.FLAG_TOKEN_CLASSIFY) != 0 {
|
||||
return nil
|
||||
}
|
||||
return fmt.Errorf("model %q does not declare the systemone usecase (known_usecases: [systemone])", cfg.Name)
|
||||
return fmt.Errorf("model %q does not declare the decisions usecase (known_usecases: [decisions])", cfg.Name)
|
||||
}
|
||||
|
||||
// checkSystemOneModel applies systemOneModelAllowed to a model looked up by
|
||||
@@ -405,7 +405,7 @@ func checkSystemOneModel(app *application.Application, modelName string) error {
|
||||
// A model that declares token_classify without systemone is a zero-shot NER
|
||||
// model: the backend's decision entry point refuses those architectures, so it
|
||||
// goes to the NER path instead. A config that declares nothing keeps the
|
||||
// decision pipeline, which is what setups that predate the systemone usecase
|
||||
// decision pipeline, which is what setups that predate the decisions usecase
|
||||
// relied on.
|
||||
func systemOneUsesDecisionPipeline(cfg config.ModelConfig) bool {
|
||||
if !backendSupportsScore(cfg.Backend) {
|
||||
@@ -415,7 +415,7 @@ func systemOneUsesDecisionPipeline(cfg config.ModelConfig) bool {
|
||||
return true
|
||||
}
|
||||
declared := *cfg.KnownUsecases
|
||||
if declared&config.FLAG_SYSTEMONE != 0 {
|
||||
if declared&config.FLAG_DECISIONS != 0 {
|
||||
return true
|
||||
}
|
||||
return declared&config.FLAG_TOKEN_CLASSIFY == 0
|
||||
@@ -429,7 +429,7 @@ func systemOneNERAllowed(cfg config.ModelConfig) error {
|
||||
return nil
|
||||
}
|
||||
declared := *cfg.KnownUsecases
|
||||
if declared&config.FLAG_SYSTEMONE != 0 && declared&config.FLAG_TOKEN_CLASSIFY == 0 {
|
||||
if declared&config.FLAG_DECISIONS != 0 && declared&config.FLAG_TOKEN_CLASSIFY == 0 {
|
||||
return fmt.Errorf("model %q is a decision model: /permute and /separate use the NER path, use POST /v1/systemone instead", cfg.Name)
|
||||
}
|
||||
return nil
|
||||
|
||||
@@ -16,8 +16,8 @@ var _ = Describe("systemOneModelAllowed", func() {
|
||||
}
|
||||
}
|
||||
|
||||
It("accepts a declared systemone model", func() {
|
||||
Expect(systemOneModelAllowed(mk("systemone"))).To(Succeed())
|
||||
It("accepts a declared decisions model", func() {
|
||||
Expect(systemOneModelAllowed(mk("decisions"))).To(Succeed())
|
||||
})
|
||||
|
||||
It("accepts a token_classify model, which the NER path serves", func() {
|
||||
@@ -29,7 +29,7 @@ var _ = Describe("systemOneModelAllowed", func() {
|
||||
})
|
||||
|
||||
It("refuses a chat-only model with an actionable message", func() {
|
||||
Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [systemone]")))
|
||||
Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [decisions]")))
|
||||
})
|
||||
})
|
||||
|
||||
@@ -44,7 +44,7 @@ var _ = Describe("systemone routing by model kind", func() {
|
||||
|
||||
Describe("systemOneUsesDecisionPipeline", func() {
|
||||
It("sends a declared decision model to the decision pipeline", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "systemone"))).To(BeTrue())
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions"))).To(BeTrue())
|
||||
})
|
||||
It("sends a token_classify model to the NER path, since vllm_decide refuses NER architectures", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "token_classify"))).To(BeFalse())
|
||||
@@ -53,16 +53,16 @@ var _ = Describe("systemone routing by model kind", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp"))).To(BeTrue())
|
||||
})
|
||||
It("prefers the decision pipeline when both usecases are declared", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "systemone", "token_classify"))).To(BeTrue())
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions", "token_classify"))).To(BeTrue())
|
||||
})
|
||||
It("never uses it for a backend without the Score RPC", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("no-such-backend", "systemone"))).To(BeFalse())
|
||||
Expect(systemOneUsesDecisionPipeline(mk("no-such-backend", "decisions"))).To(BeFalse())
|
||||
})
|
||||
})
|
||||
|
||||
Describe("systemOneNERAllowed", func() {
|
||||
It("refuses a decision model on the NER-only routes with an actionable message", func() {
|
||||
Expect(systemOneNERAllowed(mk("vllm-cpp", "systemone"))).To(MatchError(ContainSubstring("/v1/systemone")))
|
||||
Expect(systemOneNERAllowed(mk("vllm-cpp", "decisions"))).To(MatchError(ContainSubstring("/v1/systemone")))
|
||||
})
|
||||
It("accepts a token_classify model", func() {
|
||||
Expect(systemOneNERAllowed(mk("vllm-cpp", "token_classify"))).To(Succeed())
|
||||
|
||||
@@ -172,17 +172,17 @@ test.describe('Models lifecycle', () => {
|
||||
await expect(installedPane(page)).toContainText('Worker one')
|
||||
})
|
||||
|
||||
test('shows the systemone use case on a decision model', async ({ page }) => {
|
||||
test('shows the decisions use case on a decision model', async ({ page }) => {
|
||||
await page.route('**/api/models/capabilities', route => route.fulfill({
|
||||
contentType: 'application/json',
|
||||
body: JSON.stringify({
|
||||
data: [...installedModels, { id: 'decider', backend: 'vllm-cpp', capabilities: ['FLAG_SYSTEMONE'] }],
|
||||
data: [...installedModels, { id: 'decider', backend: 'vllm-cpp', capabilities: ['FLAG_DECISIONS'] }],
|
||||
}),
|
||||
}))
|
||||
await page.goto('/app/models?view=installed&model=decider')
|
||||
|
||||
await expect(installedPane(page)).toContainText('decider')
|
||||
await expect(installedPane(page)).toContainText('SystemOne')
|
||||
await expect(installedPane(page)).toContainText('Decisions')
|
||||
})
|
||||
|
||||
test('stops a running model with confirmation', async ({ page }) => {
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -46,7 +46,7 @@
|
||||
"open": {
|
||||
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
|
||||
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne"
|
||||
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
|
||||
},
|
||||
"empty": {
|
||||
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
|
||||
|
||||
@@ -22,7 +22,7 @@ import {
|
||||
CAP_CHAT, CAP_COMPLETION, CAP_IMAGE, CAP_VIDEO, CAP_TTS,
|
||||
CAP_TRANSCRIPT, CAP_SOUND_GENERATION, CAP_FACE_RECOGNITION,
|
||||
CAP_SPEAKER_RECOGNITION, CAP_EMBEDDINGS, CAP_RERANK,
|
||||
CAP_VAD, CAP_SCORE, CAP_SYSTEMONE,
|
||||
CAP_VAD, CAP_SCORE, CAP_DECISIONS,
|
||||
} from '../utils/capabilities'
|
||||
|
||||
const USE_CASES = [
|
||||
@@ -39,7 +39,7 @@ const USE_CASES = [
|
||||
{ cap: CAP_RERANK, labelKey: 'rerank' },
|
||||
{ cap: CAP_VAD, labelKey: 'vad' },
|
||||
{ cap: CAP_SCORE, labelKey: 'score' },
|
||||
{ cap: CAP_SYSTEMONE, labelKey: 'systemone' },
|
||||
{ cap: CAP_DECISIONS, labelKey: 'decisions' },
|
||||
]
|
||||
|
||||
export function modelUseCases(model) {
|
||||
|
||||
+1
-1
@@ -29,5 +29,5 @@ export const CAP_SPEAKER_RECOGNITION = 'FLAG_SPEAKER_RECOGNITION'
|
||||
export const CAP_AUDIO_TRANSFORM = 'FLAG_AUDIO_TRANSFORM'
|
||||
export const CAP_REALTIME_AUDIO = 'FLAG_REALTIME_AUDIO'
|
||||
export const CAP_SCORE = 'FLAG_SCORE'
|
||||
export const CAP_SYSTEMONE = 'FLAG_SYSTEMONE'
|
||||
export const CAP_DECISIONS = 'FLAG_DECISIONS'
|
||||
export const CAP_TOKEN_CLASSIFY = 'FLAG_TOKEN_CLASSIFY'
|
||||
@@ -1066,9 +1066,9 @@ known_usecases:
|
||||
- embeddings
|
||||
```
|
||||
|
||||
Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `score`, `token_classify`, `systemone`, `llm` (combination of CHAT, COMPLETION, EDIT).
|
||||
Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `score`, `token_classify`, `decisions`, `llm` (combination of CHAT, COMPLETION, EDIT).
|
||||
|
||||
`systemone` marks a model as a decision model for the [SystemOne API]({{% relref "features/systemone" %}}) (`POST /v1/systemone`). It is never guessed, and a model that declares it is not listed as a chat, completion or embeddings model.
|
||||
`decisions` marks a model as a decision model for the [Decisions API]({{% relref "features/decisions" %}}) (`POST /v1/systemone`). It is never guessed, and a model that declares it is not listed as a chat, completion or embeddings model.
|
||||
|
||||
`token_classify` marks a model as a token-classification (NER) provider for the PII filter (e.g. an `openai-privacy-filter` GGUF). Declare it explicitly together with `embeddings: true` (the classifier loads via TOKEN_CLS pooling). It runs on the dedicated `privacy-filter` backend (`backend/cpp/privacy-filter`), a standalone GGML engine for the `openai-privacy-filter` family - separate from `llama-cpp`, which no longer carries the token-classification path.
|
||||
|
||||
|
||||
@@ -1,17 +1,19 @@
|
||||
+++
|
||||
disableToc = false
|
||||
title = "SystemOne decisions"
|
||||
title = "Decisions API"
|
||||
weight = 66
|
||||
url = "/features/systemone/"
|
||||
url = "/features/decisions/"
|
||||
+++
|
||||
|
||||
SystemOne is an API for fast, typed decisions. You send a piece of text (the
|
||||
The Decisions API is a fast, typed decision layer. You send a piece of text (the
|
||||
*state*) and a set of named questions. A decision model answers each question
|
||||
with a value and a confidence, in one pass. The model does not generate text, so
|
||||
there is nothing to parse and no free-form output to validate.
|
||||
|
||||
The request and response shapes follow the [kev](https://github.com/jaredpalmer/kev)
|
||||
project and match the `/v1/systemone` endpoint that Ollama added in 0.35.
|
||||
LocalAI serves it on the `/v1/systemone` routes. The request and response shapes
|
||||
follow the [kev](https://github.com/jaredpalmer/kev) project and match the
|
||||
`/v1/systemone` endpoint that Ollama added in 0.35. The wire contract is called
|
||||
SystemOne; the capability a model declares is called `decisions`.
|
||||
|
||||
## Endpoints
|
||||
|
||||
@@ -25,7 +27,7 @@ Which route a model can serve depends on its kind:
|
||||
|
||||
| Model kind | `/v1/systemone` | `/permute` and `/separate` |
|
||||
|---|---|---|
|
||||
| Decision model (`systemone`), such as Laya or GLiNER2.5-Decide | Yes | No, returns `400` |
|
||||
| Decision model (`decisions`), such as Laya or GLiNER2.5-Decide | Yes | No, returns `400` |
|
||||
| Zero-shot NER model (`token_classify`), such as GLiNER2.5 | Yes, through the NER path | Yes |
|
||||
|
||||
## Question types
|
||||
@@ -70,28 +72,28 @@ reports token usage and `latency_ms`. The NER path does not report token usage.
|
||||
|
||||
## Choosing a model
|
||||
|
||||
A model can serve SystemOne only if it is a decision model. Declare the usecase
|
||||
A model can serve the Decisions API only if it is a decision model. Declare the usecase
|
||||
in the model config:
|
||||
|
||||
```yaml
|
||||
name: laya
|
||||
backend: vllm-cpp
|
||||
known_usecases:
|
||||
- systemone
|
||||
- decisions
|
||||
parameters:
|
||||
model: convaiinnovations/laya
|
||||
```
|
||||
|
||||
`systemone` is never guessed, and a model that declares it is not listed as a
|
||||
`decisions` is never guessed, and a model that declares it is not listed as a
|
||||
chat, completion or embeddings model. A model that declares usecases without
|
||||
`systemone` or `token_classify` gets a `400` from these endpoints that names the
|
||||
missing usecase. A model that declares `token_classify` and not `systemone` is
|
||||
`decisions` or `token_classify` gets a `400` from these endpoints that names the
|
||||
missing usecase. A model that declares `token_classify` and not `decisions` is
|
||||
served by the zero-shot NER path. A vllm-cpp config that declares no usecases is
|
||||
treated as a decision model, so setups that predate the flag keep working, but a
|
||||
config that declares only `chat` (as an older `laya` gallery entry did) now gets
|
||||
the `400` and needs `known_usecases: [systemone]`.
|
||||
the `400` and needs `known_usecases: [decisions]`.
|
||||
|
||||
Install one from the gallery and filter on the `systemone` tag:
|
||||
Install one from the gallery and filter on the `decisions` tag:
|
||||
|
||||
| Gallery entry | Model | Notes |
|
||||
|---|---|---|
|
||||
@@ -107,6 +109,6 @@ and does not serve `/v1/systemone` yet.
|
||||
|
||||
## Access control
|
||||
|
||||
When authentication is on, the three routes need the `systemone` feature. It is
|
||||
When authentication is on, the three routes need the `decisions` feature. It is
|
||||
on by default for every user, like the other API features, and an administrator
|
||||
can turn it off per user.
|
||||
@@ -160,12 +160,12 @@ forward, which is the required contract for pooling models in vllm.cpp. A
|
||||
device-resident forward is tracked as a performance optimization, not a
|
||||
correctness gap.
|
||||
|
||||
### SystemOne decision API
|
||||
### Decisions API
|
||||
|
||||
The `vllm-cpp` backend serves the kev-compatible SystemOne endpoints: typed
|
||||
The `vllm-cpp` backend serves the kev-compatible SystemOne endpoints (the Decisions API): typed
|
||||
`choice`, `noul` and `score` questions over a state text, answered by a
|
||||
non-generative decision model in one pass. A decision model declares
|
||||
`known_usecases: [systemone]`. See [SystemOne decisions]({{% relref "features/systemone" %}})
|
||||
`known_usecases: [decisions]`. See [Decisions API]({{% relref "features/decisions" %}})
|
||||
for the request shape, the models you can install and the access rules.
|
||||
|
||||
| Endpoint | Method | Description |
|
||||
|
||||
+4
-4
@@ -63653,7 +63653,7 @@
|
||||
512-token context. F16 weights, ~804 MB.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- decision
|
||||
- decisions
|
||||
- systemone
|
||||
- vllm-cpp
|
||||
- cpu
|
||||
@@ -63663,7 +63663,7 @@
|
||||
overrides:
|
||||
backend: vllm-cpp
|
||||
known_usecases:
|
||||
- systemone
|
||||
- decisions
|
||||
parameters:
|
||||
model: convaiinnovations/laya
|
||||
artifacts:
|
||||
@@ -63690,7 +63690,7 @@
|
||||
checkpoint it was checked against.
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- decision
|
||||
- decisions
|
||||
- systemone
|
||||
- vllm-cpp
|
||||
- cpu
|
||||
@@ -63700,7 +63700,7 @@
|
||||
overrides:
|
||||
backend: vllm-cpp
|
||||
known_usecases:
|
||||
- systemone
|
||||
- decisions
|
||||
parameters:
|
||||
model: fastino/GLiNER2.5-Decide
|
||||
artifacts:
|
||||
|
||||
Reference in new issue
Block a user