mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-02 19:14:38 -04:00
fix(systemone): route NER models to the NER path and refuse decision models on permute and separate
vllm_decide refuses NER architectures and the NER entry point refuses decision architectures, so each model kind 500ed on half of the routes. A token_classify model now goes to the NER path on /v1/systemone, and /permute and /separate return 400 for decision models. Docs and instructions state which kind serves which route. Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
This commit is contained in:
1 parent
c7f278dd0d
commit
b3d65fd538
5 files changed
+116
-10
No files matched your search
@@ -109,7 +109,7 @@ var instructionDefs = []instructionDef{
|
||||
Name: "systemone",
|
||||
Description: "Typed decisions (choice, noul, score) over a state text with calibrated confidence",
|
||||
Tags: []string{"systemone"},
|
||||
Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { <id>: { type: choice|noul|score, instructions, criteria } } }. The model must declare known_usecases: [systemone] (or token_classify for the zero-shot NER path); a config that declares no usecases keeps working. Responses carry per-question answers with confidence and probabilities plus token usage. The wire shape matches Ollama's /v1/systemone.",
|
||||
Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { <id>: { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [systemone] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. The wire shape matches Ollama's /v1/systemone.",
|
||||
},
|
||||
{
|
||||
Name: "branding",
|
||||
|
||||
@@ -400,6 +400,55 @@ func checkSystemOneModel(app *application.Application, modelName string) error {
|
||||
return systemOneModelAllowed(cfg)
|
||||
}
|
||||
|
||||
// systemOneUsesDecisionPipeline reports whether /v1/systemone forwards the
|
||||
// request to the backend's Score RPC (the decision pipeline) for this model.
|
||||
// A model that declares token_classify without systemone is a zero-shot NER
|
||||
// model: the backend's decision entry point refuses those architectures, so it
|
||||
// goes to the NER path instead. A config that declares nothing keeps the
|
||||
// decision pipeline, which is what setups that predate the systemone usecase
|
||||
// relied on.
|
||||
func systemOneUsesDecisionPipeline(cfg config.ModelConfig) bool {
|
||||
if !backendSupportsScore(cfg.Backend) {
|
||||
return false
|
||||
}
|
||||
if cfg.KnownUsecases == nil {
|
||||
return true
|
||||
}
|
||||
declared := *cfg.KnownUsecases
|
||||
if declared&config.FLAG_SYSTEMONE != 0 {
|
||||
return true
|
||||
}
|
||||
return declared&config.FLAG_TOKEN_CLASSIFY == 0
|
||||
}
|
||||
|
||||
// systemOneNERAllowed guards /permute and /separate, which always run the NER
|
||||
// path. A decision model cannot serve them: the backend's NER entry point
|
||||
// refuses its architecture, and the caller would see a backend error.
|
||||
func systemOneNERAllowed(cfg config.ModelConfig) error {
|
||||
if cfg.KnownUsecases == nil {
|
||||
return nil
|
||||
}
|
||||
declared := *cfg.KnownUsecases
|
||||
if declared&config.FLAG_SYSTEMONE != 0 && declared&config.FLAG_TOKEN_CLASSIFY == 0 {
|
||||
return fmt.Errorf("model %q is a decision model: /permute and /separate use the NER path, use POST /v1/systemone instead", cfg.Name)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// checkSystemOneNERModel applies systemOneNERAllowed to a model looked up by
|
||||
// name; an unknown model passes so the not-found handling keeps its status.
|
||||
func checkSystemOneNERModel(app *application.Application, modelName string) error {
|
||||
cl := app.ModelConfigLoader()
|
||||
if cl == nil {
|
||||
return nil
|
||||
}
|
||||
cfg, ok := cl.GetModelConfig(modelName)
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
return systemOneNERAllowed(cfg)
|
||||
}
|
||||
|
||||
// backendSupportsScore reports whether the named backend implements the
|
||||
// Score gRPC RPC. vllm-cpp does (kev/laya decision pipeline and cua-s1-forms
|
||||
// scoring via the unified vllm_decide C ABI); other backends fall through to
|
||||
@@ -445,7 +494,7 @@ func SystemOneEndpoint(app *application.Application) echo.HandlerFunc {
|
||||
// Score RPC and return the backend's response as-is.
|
||||
cl := app.ModelConfigLoader()
|
||||
if cl != nil {
|
||||
if cfg, ok := cl.GetModelConfig(req.Model); ok && backendSupportsScore(cfg.Backend) {
|
||||
if cfg, ok := cl.GetModelConfig(req.Model); ok && systemOneUsesDecisionPipeline(cfg) {
|
||||
reqJSON, err := json.Marshal(req)
|
||||
if err != nil {
|
||||
return systemOneError(c, http.StatusInternalServerError, "failed to marshal request: "+err.Error())
|
||||
@@ -509,6 +558,9 @@ func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc {
|
||||
if err := checkSystemOneModel(app, req.Request.Model); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
}
|
||||
if err := checkSystemOneNERModel(app, req.Request.Model); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
}
|
||||
if req.Question == "" {
|
||||
return systemOneError(c, http.StatusBadRequest, "question is required")
|
||||
}
|
||||
@@ -648,6 +700,9 @@ func SystemOneSeparateEndpoint(app *application.Application) echo.HandlerFunc {
|
||||
if err := checkSystemOneModel(app, req.Model); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
}
|
||||
if err := checkSystemOneNERModel(app, req.Model); err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
}
|
||||
parsed, err := parseSystemOneRequest(&req)
|
||||
if err != nil {
|
||||
return systemOneError(c, http.StatusBadRequest, err.Error())
|
||||
|
||||
@@ -32,3 +32,43 @@ var _ = Describe("systemOneModelAllowed", func() {
|
||||
Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [systemone]")))
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("systemone routing by model kind", func() {
|
||||
mk := func(backend string, usecases ...string) config.ModelConfig {
|
||||
c := config.ModelConfig{Name: "m", Backend: backend}
|
||||
if len(usecases) > 0 {
|
||||
c.KnownUsecases = config.GetUsecasesFromYAML(usecases)
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
Describe("systemOneUsesDecisionPipeline", func() {
|
||||
It("sends a declared decision model to the decision pipeline", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "systemone"))).To(BeTrue())
|
||||
})
|
||||
It("sends a token_classify model to the NER path, since vllm_decide refuses NER architectures", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "token_classify"))).To(BeFalse())
|
||||
})
|
||||
It("keeps configs that declare nothing on the decision pipeline", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp"))).To(BeTrue())
|
||||
})
|
||||
It("prefers the decision pipeline when both usecases are declared", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "systemone", "token_classify"))).To(BeTrue())
|
||||
})
|
||||
It("never uses it for a backend without the Score RPC", func() {
|
||||
Expect(systemOneUsesDecisionPipeline(mk("no-such-backend", "systemone"))).To(BeFalse())
|
||||
})
|
||||
})
|
||||
|
||||
Describe("systemOneNERAllowed", func() {
|
||||
It("refuses a decision model on the NER-only routes with an actionable message", func() {
|
||||
Expect(systemOneNERAllowed(mk("vllm-cpp", "systemone"))).To(MatchError(ContainSubstring("/v1/systemone")))
|
||||
})
|
||||
It("accepts a token_classify model", func() {
|
||||
Expect(systemOneNERAllowed(mk("vllm-cpp", "token_classify"))).To(Succeed())
|
||||
})
|
||||
It("accepts configs that declare nothing", func() {
|
||||
Expect(systemOneNERAllowed(mk("vllm-cpp"))).To(Succeed())
|
||||
})
|
||||
})
|
||||
})
|
||||
@@ -21,6 +21,13 @@ project and match the `/v1/systemone` endpoint that Ollama added in 0.35.
|
||||
| `/v1/systemone/permute` | POST | Re-run one choice question under `n_perm` option orders |
|
||||
| `/v1/systemone/separate` | POST | Answer each question in its own pass |
|
||||
|
||||
Which route a model can serve depends on its kind:
|
||||
|
||||
| Model kind | `/v1/systemone` | `/permute` and `/separate` |
|
||||
|---|---|---|
|
||||
| Decision model (`systemone`), such as Laya or GLiNER2.5-Decide | Yes | No, returns `400` |
|
||||
| Zero-shot NER model (`token_classify`), such as GLiNER2.5 | Yes, through the NER path | Yes |
|
||||
|
||||
## Question types
|
||||
|
||||
| Type | Answer | Fields in the answer |
|
||||
@@ -58,8 +65,8 @@ curl http://localhost:8080/v1/systemone -H "Content-Type: application/json" -d '
|
||||
}'
|
||||
```
|
||||
|
||||
Every answer carries a `confidence` value, and the response reports token usage
|
||||
and `latency_ms`.
|
||||
Answers from a decision model carry a `confidence` value, and the response
|
||||
reports token usage and `latency_ms`. The NER path does not report token usage.
|
||||
|
||||
## Choosing a model
|
||||
|
||||
@@ -78,9 +85,11 @@ parameters:
|
||||
`systemone` is never guessed, and a model that declares it is not listed as a
|
||||
chat, completion or embeddings model. A model that declares usecases without
|
||||
`systemone` or `token_classify` gets a `400` from these endpoints that names the
|
||||
missing usecase. A config that declares no usecases at all keeps working, so
|
||||
setups that predate the flag are not broken. Models that declare `token_classify`
|
||||
are served by the zero-shot NER path.
|
||||
missing usecase. A model that declares `token_classify` and not `systemone` is
|
||||
served by the zero-shot NER path. A vllm-cpp config that declares no usecases is
|
||||
treated as a decision model, so setups that predate the flag keep working, but a
|
||||
config that declares only `chat` (as an older `laya` gallery entry did) now gets
|
||||
the `400` and needs `known_usecases: [systemone]`.
|
||||
|
||||
Install one from the gallery and filter on the `systemone` tag:
|
||||
|
||||
|
||||
@@ -174,9 +174,11 @@ for the request shape, the models you can install and the access rules.
|
||||
| `/v1/systemone/permute` | POST | Re-run one choice question under n_perm option orders |
|
||||
| `/v1/systemone/separate` | POST | Answer each question in its own pass (N passes) |
|
||||
|
||||
The GLiNER2.5 zero-shot NER model (`token_classify`) also serves these
|
||||
endpoints. It derives its NER labels from the question definitions, so no
|
||||
`ner_labels` configuration is needed.
|
||||
The GLiNER2.5 zero-shot NER model (`token_classify`) also serves
|
||||
`/v1/systemone`, through the NER path, and it is the model to use for
|
||||
`/v1/systemone/permute` and `/v1/systemone/separate`, which decision models
|
||||
refuse with a `400`. It derives its NER labels from the question definitions, so
|
||||
no `ner_labels` configuration is needed.
|
||||
|
||||
## Beyond text generation
|
||||
|
||||
|
||||
Reference in new issue
Block a user