feat(config): add systemone usecase for decision models

Explicit-only, reserving usecase like score and token_classify: a declared
list is authoritative and the heuristic never guesses it. vllm-cpp now
lists systemone and vision as possible usecases.

Assisted-by: Claude Code:claude-sonnet-5-5
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
This commit is contained in:
Ettore Di Giacinto committed 2026-09-30 11:26:22 +00:00
1 parent 2fa36e7147
commit 84c83a70dd
4 files changed
+63 -8

No files matched your search

+8 -2
View File
@@ -35,6 +35,7 @@ const (
UsecaseSpeakerRecognition = "speaker_recognition"
UsecaseTokenClassify = "token_classify"
UsecaseScore = "score"
UsecaseSystemOne = "systemone"
)
// GRPCMethod identifies a Backend service RPC from backend.proto.
@@ -216,6 +217,11 @@ var UsecaseInfoMap = map[string]UsecaseInfo{
GRPCMethod: MethodScore,
Description: "Joint log-probability scoring of candidate continuations via the Score RPC. Declared explicitly via known_usecases and usable alongside generation usecases.",
},
UsecaseSystemOne: {
Flag: FLAG_SYSTEMONE,
GRPCMethod: MethodScore,
Description: "SystemOne decision API (POST /v1/systemone): typed choice, noul and score questions over a state text, answered by a non-generative decision model through the Score RPC (question_type systemone). Declared explicitly via known_usecases.",
},
}
// BackendCapability describes which gRPC methods and usecases a backend supports.
@@ -349,10 +355,10 @@ var BackendCapabilities = map[string]BackendCapability{
// model returns an error rather than silent garbage.
"vllm-cpp": {
GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateVideo, MethodTokenClassify, MethodScore},
PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVideo, UsecaseTokenClassify, UsecaseScore},
PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVision, UsecaseVideo, UsecaseTokenClassify, UsecaseScore, UsecaseSystemOne},
DefaultUsecases: []string{UsecaseChat},
AcceptsImages: true,
Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and kev/laya decision pipelines",
Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and SystemOne decision models (kev, laya, CLM, GLiNER2.5-Decide, xor, nimble)",
},
"vllm-omni": {
GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateImage, MethodGenerateVideo, MethodTTS},
+2 -2
View File
@@ -16,14 +16,14 @@ import (
// reservedNonChatModel reports whether the operator reserved this model for an
// internal primitive — the router score classifier or the PII NER
// token_classify tier. Such a model has no chat template and must not be
// token_classify tier, or a SystemOne decision head. Such a model has no chat template and must not be
// given the generative-chat defaults the GGUF importer otherwise applies
// (FLAG_CHAT, jinja templating): surfacing it in chat pickers defeats the
// reservation. Operators who do want a combined model declare both usecases
// explicitly — the combination is valid.
func reservedNonChatModel(cfg *ModelConfig) bool {
return cfg.KnownUsecases != nil &&
(*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY)) != 0
(*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY|FLAG_SYSTEMONE)) != 0
}
// genAudioEncoderKey is the mmproj metadata flag llama.cpp's mtmd writes for a
+20 -4
View File
@@ -2056,6 +2056,13 @@ const (
FLAG_3D ModelConfigUsecase = 0b100000000000000000000000
FLAG_3D_ANIMATION ModelConfigUsecase = 1 << 24
// Marks a model as wired for the SystemOne decision API (POST
// /v1/systemone: typed choice / noul / score questions over a state).
// Explicit only, like FLAG_SCORE: a decision model never generates
// text, so guessing chat or embeddings for it would surface it in
// pickers it cannot serve.
FLAG_SYSTEMONE ModelConfigUsecase = 1 << 25
// Common Subsets
FLAG_LLM ModelConfigUsecase = FLAG_CHAT | FLAG_COMPLETION | FLAG_EDIT
)
@@ -2118,6 +2125,7 @@ func GetAllModelConfigUsecases() map[string]ModelConfigUsecase {
"FLAG_TOKEN_CLASSIFY": FLAG_TOKEN_CLASSIFY,
"FLAG_3D": FLAG_3D,
"FLAG_3D_ANIMATION": FLAG_3D_ANIMATION,
"FLAG_SYSTEMONE": FLAG_SYSTEMONE,
}
}
@@ -2146,9 +2154,9 @@ func GetUsecasesFromYAML(input []string) *ModelConfigUsecase {
//
// Declared known_usecases are normally additive — the guessing heuristic
// still adds whatever it can infer from backend/templates. The exceptions
// are FLAG_SCORE and FLAG_TOKEN_CLASSIFY: when the operator declared
// either, they reserved the model for an internal direct-decode primitive
// (the router classifier, or the PII NER tier). Letting GuessUsecases
// are FLAG_SCORE, FLAG_TOKEN_CLASSIFY and FLAG_SYSTEMONE: when the operator
// declared any of them, they reserved the model for a direct-decode primitive
// (the router classifier, the PII NER tier, or a SystemOne decision head). Letting GuessUsecases
// paint chat/completion/embeddings on top would surface it in pickers it
// was deliberately kept out of. So a declared score or token_classify
// list is authoritative; declare the generation usecases explicitly
@@ -2158,7 +2166,7 @@ func (c *ModelConfig) HasUsecases(u ModelConfigUsecase) bool {
if (u & *c.KnownUsecases) == u {
return true
}
if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY)) != 0 {
if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY | FLAG_SYSTEMONE)) != 0 {
return false
}
}
@@ -2381,6 +2389,14 @@ func (c *ModelConfig) GuessUsecases(u ModelConfigUsecase) bool {
return false
}
if (u & FLAG_SYSTEMONE) == FLAG_SYSTEMONE {
// No heuristic: SystemOne intent is a deliberate operator choice
// (the model is a non-generative decision head), so
// HasUsecases(FLAG_SYSTEMONE) is true only when KnownUsecases
// declares it explicitly.
return false
}
return true
}
+33
View File
@@ -955,3 +955,36 @@ var _ = Describe("ModelConfig alias", func() {
Expect(err).To(MatchError(ContainSubstring("alias")))
})
})
var _ = Describe("systemone usecase", func() {
// A decision model never generates text, so a declared systemone list
// must stay authoritative and the heuristic must never guess the flag.
It("is authoritative when declared and never guessed", func() {
declared := GetUsecasesFromYAML([]string{"systemone"})
Expect(declared).NotTo(BeNil())
Expect(*declared).NotTo(Equal(FLAG_ANY))
cfg := ModelConfig{
Name: "laya",
Backend: "vllm-cpp",
KnownUsecases: declared,
TemplateConfig: TemplateConfig{
Chat: "inherited from chatml",
ChatMessage: "inherited from chatml",
Completion: "inherited from chatml",
},
}
Expect(cfg.HasUsecases(*declared)).To(BeTrue())
Expect(cfg.HasUsecases(FLAG_CHAT)).To(BeFalse())
Expect(cfg.HasUsecases(FLAG_COMPLETION)).To(BeFalse())
Expect(cfg.HasUsecases(FLAG_EMBEDDINGS)).To(BeFalse())
undeclared := ModelConfig{Name: "laya", Backend: "vllm-cpp"}
Expect(undeclared.HasUsecases(*declared)).To(BeFalse())
})
It("is a reserved usecase for the GGUF importer chat-default guard", func() {
declared := GetUsecasesFromYAML([]string{"systemone"})
Expect(reservedNonChatModel(&ModelConfig{Backend: "vllm-cpp", KnownUsecases: declared})).To(BeTrue())
})
})