diff --git a/core/http/endpoints/localai/api_instructions.go b/core/http/endpoints/localai/api_instructions.go index dc60a3c21..702c771ff 100644 --- a/core/http/endpoints/localai/api_instructions.go +++ b/core/http/endpoints/localai/api_instructions.go @@ -109,7 +109,7 @@ var instructionDefs = []instructionDef{ Name: "systemone", Description: "Typed decisions (choice, noul, score) over a state text with calibrated confidence", Tags: []string{"systemone"}, - Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { : { type: choice|noul|score, instructions, criteria } } }. The model must declare known_usecases: [systemone] (or token_classify for the zero-shot NER path); a config that declares no usecases keeps working. Responses carry per-question answers with confidence and probabilities plus token usage. The wire shape matches Ollama's /v1/systemone.", + Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { : { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [systemone] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. The wire shape matches Ollama's /v1/systemone.", }, { Name: "branding", diff --git a/core/http/endpoints/localai/systemone.go b/core/http/endpoints/localai/systemone.go index 435414d39..17f68a71a 100644 --- a/core/http/endpoints/localai/systemone.go +++ b/core/http/endpoints/localai/systemone.go @@ -400,6 +400,55 @@ func checkSystemOneModel(app *application.Application, modelName string) error { return systemOneModelAllowed(cfg) } +// systemOneUsesDecisionPipeline reports whether /v1/systemone forwards the +// request to the backend's Score RPC (the decision pipeline) for this model. +// A model that declares token_classify without systemone is a zero-shot NER +// model: the backend's decision entry point refuses those architectures, so it +// goes to the NER path instead. A config that declares nothing keeps the +// decision pipeline, which is what setups that predate the systemone usecase +// relied on. +func systemOneUsesDecisionPipeline(cfg config.ModelConfig) bool { + if !backendSupportsScore(cfg.Backend) { + return false + } + if cfg.KnownUsecases == nil { + return true + } + declared := *cfg.KnownUsecases + if declared&config.FLAG_SYSTEMONE != 0 { + return true + } + return declared&config.FLAG_TOKEN_CLASSIFY == 0 +} + +// systemOneNERAllowed guards /permute and /separate, which always run the NER +// path. A decision model cannot serve them: the backend's NER entry point +// refuses its architecture, and the caller would see a backend error. +func systemOneNERAllowed(cfg config.ModelConfig) error { + if cfg.KnownUsecases == nil { + return nil + } + declared := *cfg.KnownUsecases + if declared&config.FLAG_SYSTEMONE != 0 && declared&config.FLAG_TOKEN_CLASSIFY == 0 { + return fmt.Errorf("model %q is a decision model: /permute and /separate use the NER path, use POST /v1/systemone instead", cfg.Name) + } + return nil +} + +// checkSystemOneNERModel applies systemOneNERAllowed to a model looked up by +// name; an unknown model passes so the not-found handling keeps its status. +func checkSystemOneNERModel(app *application.Application, modelName string) error { + cl := app.ModelConfigLoader() + if cl == nil { + return nil + } + cfg, ok := cl.GetModelConfig(modelName) + if !ok { + return nil + } + return systemOneNERAllowed(cfg) +} + // backendSupportsScore reports whether the named backend implements the // Score gRPC RPC. vllm-cpp does (kev/laya decision pipeline and cua-s1-forms // scoring via the unified vllm_decide C ABI); other backends fall through to @@ -445,7 +494,7 @@ func SystemOneEndpoint(app *application.Application) echo.HandlerFunc { // Score RPC and return the backend's response as-is. cl := app.ModelConfigLoader() if cl != nil { - if cfg, ok := cl.GetModelConfig(req.Model); ok && backendSupportsScore(cfg.Backend) { + if cfg, ok := cl.GetModelConfig(req.Model); ok && systemOneUsesDecisionPipeline(cfg) { reqJSON, err := json.Marshal(req) if err != nil { return systemOneError(c, http.StatusInternalServerError, "failed to marshal request: "+err.Error()) @@ -509,6 +558,9 @@ func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc { if err := checkSystemOneModel(app, req.Request.Model); err != nil { return systemOneError(c, http.StatusBadRequest, err.Error()) } + if err := checkSystemOneNERModel(app, req.Request.Model); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } if req.Question == "" { return systemOneError(c, http.StatusBadRequest, "question is required") } @@ -648,6 +700,9 @@ func SystemOneSeparateEndpoint(app *application.Application) echo.HandlerFunc { if err := checkSystemOneModel(app, req.Model); err != nil { return systemOneError(c, http.StatusBadRequest, err.Error()) } + if err := checkSystemOneNERModel(app, req.Model); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } parsed, err := parseSystemOneRequest(&req) if err != nil { return systemOneError(c, http.StatusBadRequest, err.Error()) diff --git a/core/http/endpoints/localai/systemone_gate_test.go b/core/http/endpoints/localai/systemone_gate_test.go index b7857da56..b790c8df2 100644 --- a/core/http/endpoints/localai/systemone_gate_test.go +++ b/core/http/endpoints/localai/systemone_gate_test.go @@ -32,3 +32,43 @@ var _ = Describe("systemOneModelAllowed", func() { Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [systemone]"))) }) }) + +var _ = Describe("systemone routing by model kind", func() { + mk := func(backend string, usecases ...string) config.ModelConfig { + c := config.ModelConfig{Name: "m", Backend: backend} + if len(usecases) > 0 { + c.KnownUsecases = config.GetUsecasesFromYAML(usecases) + } + return c + } + + Describe("systemOneUsesDecisionPipeline", func() { + It("sends a declared decision model to the decision pipeline", func() { + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "systemone"))).To(BeTrue()) + }) + It("sends a token_classify model to the NER path, since vllm_decide refuses NER architectures", func() { + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "token_classify"))).To(BeFalse()) + }) + It("keeps configs that declare nothing on the decision pipeline", func() { + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp"))).To(BeTrue()) + }) + It("prefers the decision pipeline when both usecases are declared", func() { + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "systemone", "token_classify"))).To(BeTrue()) + }) + It("never uses it for a backend without the Score RPC", func() { + Expect(systemOneUsesDecisionPipeline(mk("no-such-backend", "systemone"))).To(BeFalse()) + }) + }) + + Describe("systemOneNERAllowed", func() { + It("refuses a decision model on the NER-only routes with an actionable message", func() { + Expect(systemOneNERAllowed(mk("vllm-cpp", "systemone"))).To(MatchError(ContainSubstring("/v1/systemone"))) + }) + It("accepts a token_classify model", func() { + Expect(systemOneNERAllowed(mk("vllm-cpp", "token_classify"))).To(Succeed()) + }) + It("accepts configs that declare nothing", func() { + Expect(systemOneNERAllowed(mk("vllm-cpp"))).To(Succeed()) + }) + }) +}) diff --git a/docs/content/features/systemone.md b/docs/content/features/systemone.md index d58a3d436..84c65a665 100644 --- a/docs/content/features/systemone.md +++ b/docs/content/features/systemone.md @@ -21,6 +21,13 @@ project and match the `/v1/systemone` endpoint that Ollama added in 0.35. | `/v1/systemone/permute` | POST | Re-run one choice question under `n_perm` option orders | | `/v1/systemone/separate` | POST | Answer each question in its own pass | +Which route a model can serve depends on its kind: + +| Model kind | `/v1/systemone` | `/permute` and `/separate` | +|---|---|---| +| Decision model (`systemone`), such as Laya or GLiNER2.5-Decide | Yes | No, returns `400` | +| Zero-shot NER model (`token_classify`), such as GLiNER2.5 | Yes, through the NER path | Yes | + ## Question types | Type | Answer | Fields in the answer | @@ -58,8 +65,8 @@ curl http://localhost:8080/v1/systemone -H "Content-Type: application/json" -d ' }' ``` -Every answer carries a `confidence` value, and the response reports token usage -and `latency_ms`. +Answers from a decision model carry a `confidence` value, and the response +reports token usage and `latency_ms`. The NER path does not report token usage. ## Choosing a model @@ -78,9 +85,11 @@ parameters: `systemone` is never guessed, and a model that declares it is not listed as a chat, completion or embeddings model. A model that declares usecases without `systemone` or `token_classify` gets a `400` from these endpoints that names the -missing usecase. A config that declares no usecases at all keeps working, so -setups that predate the flag are not broken. Models that declare `token_classify` -are served by the zero-shot NER path. +missing usecase. A model that declares `token_classify` and not `systemone` is +served by the zero-shot NER path. A vllm-cpp config that declares no usecases is +treated as a decision model, so setups that predate the flag keep working, but a +config that declares only `chat` (as an older `laya` gallery entry did) now gets +the `400` and needs `known_usecases: [systemone]`. Install one from the gallery and filter on the `systemone` tag: diff --git a/docs/content/features/vllm-cpp.md b/docs/content/features/vllm-cpp.md index 6dc67c0b9..ba9840302 100644 --- a/docs/content/features/vllm-cpp.md +++ b/docs/content/features/vllm-cpp.md @@ -174,9 +174,11 @@ for the request shape, the models you can install and the access rules. | `/v1/systemone/permute` | POST | Re-run one choice question under n_perm option orders | | `/v1/systemone/separate` | POST | Answer each question in its own pass (N passes) | -The GLiNER2.5 zero-shot NER model (`token_classify`) also serves these -endpoints. It derives its NER labels from the question definitions, so no -`ner_labels` configuration is needed. +The GLiNER2.5 zero-shot NER model (`token_classify`) also serves +`/v1/systemone`, through the NER path, and it is the model to use for +`/v1/systemone/permute` and `/v1/systemone/separate`, which decision models +refuse with a `400`. It derives its NER labels from the question definitions, so +no `ner_labels` configuration is needed. ## Beyond text generation