From 21c5495a99e61f8e69b57d6b309bc6d64db146d1 Mon Sep 17 00:00:00 2001 From: mudler-agent Date: Wed, 23 Sep 2026 12:31:28 +0200 Subject: [PATCH] feat(vllm-cpp): add GLiNER2.5 NER via TokenClassify (#12140) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * feat(vllm-cpp): add GLiNER2.5 NER via TokenClassify Wire the vllm-cpp backend to the C ABI NER surface (vllm_gliner_ner, ABI v27) so LocalAI can serve zero-shot named entity recognition through the existing TokenClassify gRPC method. backend.go: TokenClassify method on *VllmCpp calls vllm_gliner_ner with the text and labels, copies the C-owned entity array into protobuf TokenClassifyEntity messages, and frees the result. govllmcpp.go: cNerEntity and cNerResult Go POD mirrors matching the C structs; vllmGlinerNer and vllmNerResultFree purego bindings; abiVersion bumped 26 -> 27. options.go: ner_labels, ner_threshold, ner_max_width parsed from engine_args. pkg/grpc: ClassifyModel interface and TokenClassify server handler (follows the Embedding locking pattern). core/config: vllm-cpp backend declares MethodTokenClassify and UsecaseTokenClassify. docs/content/features/vllm-cpp.md: NER section documenting the engine_args keys and the host-forward contract. Assisted-by: MAKI:regolo/glm5.2 [maki] Signed-off-by: Ettore Di Giacinto * fix(vllm-cpp): correct NER pointer lint directive Use the govet directive for the C-owned NER array, matching the other purego pointer conversions. The array remains valid until its deferred free; the misspelled directive caused CI to flag this conversion. Assisted-by: Codex:gpt-6 golangci-lint Signed-off-by: Ettore Di Giacinto * feat(vllm-cpp): add kev-compatible SystemOne API endpoints Add POST /v1/systemone, /v1/systemone/permute, and /v1/systemone/separate to LocalAI, mirroring the kev project's structured-extraction API. Each endpoint runs zero-shot NER over the rendered state text and builds kev-compatible answers for three question types: noul (binary entity presence), choice (pick one option), and score (pick one level). The TokenClassifyRequest proto gains a `repeated string labels` field so each question can supply its own labels at inference time, and TokenClassifier gains TokenClassifyWithLabels for per-call label selection. The vllm-cpp backend uses request labels when non-empty, falling back to configured ner_labels then the built-in defaults. Helpers (renderState, softmax, choiceConfidence, scoreConfidence, r2) are ported from kev/api.py and mirrored in vllm.cpp's api_server.cpp so both servers produce the same answer shape. Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:regolo/glm5.2 [maki] Signed-off-by: Ettore Di Giacinto * fix(vllm-cpp): suppress gosec G404 on seeded permutation RNG The SystemOne permute endpoint uses math/rand with a caller-supplied seed for reproducible option permutations, matching kev's random.seed. gosec flags this as G404 (weak RNG). Add #nosec with a comment naming the intent: this is reproducibility, not cryptography. Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:regolo/glm5.2 [maki] Signed-off-by: Ettore Di Giacinto * chore(vllm-cpp): bump vllm.cpp pin to GLiNER2.5 merge commit Advance VLLM_CPP_VERSION from f3cd97e to 5058268d, the commit that landed GLiNER2.5 zero-shot NER support (PR #3224) in vllm.cpp. This brings the DeBERTa v2 encoder, GLiNER2 boundary head, C ABI NER functions, and server endpoints into the LocalAI vllm-cpp backend. The ABI version (27) and Go struct mirrors already match. Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:regolo/glm5.2 [maki] Signed-off-by: Ettore Di Giacinto * fix(vllm-cpp): use instruction text as NER label in SystemOne handler The SystemOne handler was passing question IDs as NER labels for noul questions and bare key names for choice questions, so the model never matched any entities. Port the label mapping from vllm.cpp's ParseSystemOneBody: - noul: use the rendered instructions field (with instr alias) as the NER label, not the question ID - choice: use optionText(name, desc) — "name: description" or "name" when the description is null/empty — not the bare key - score: already correct (rendered criteria text) - permute: shuffle indices and build parallel key/label arrays so the NER call uses the optionText labels while the response is keyed by the original option names Also add the instructions field to the SystemOneQuestion schema struct (accepted alongside the instr backward-compat alias). Verified end-to-end against the real GLiNER2.5 model: noul questions now find "Apple Inc. is" (organization, 0.999) and "Tim Cook is" (person, 0.852) where they previously returned zero entities. Following-Agents-Protocol: true AI-Assisted: true Assisted-by: AGENT:regolo/glm5.2 [maki] Signed-off-by: Ettore Di Giacinto --------- Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- backend/backend.proto | 6 + backend/go/vllm-cpp/Makefile | 2 +- backend/go/vllm-cpp/backend.go | 62 +++ backend/go/vllm-cpp/govllmcpp.go | 26 +- backend/go/vllm-cpp/options.go | 25 + backend/go/vllm-cpp/vllmcpp_test.go | 23 +- core/backend/token_classify.go | 22 + core/config/backend_capabilities.go | 11 +- core/http/app.go | 1 + core/http/endpoints/localai/systemone.go | 599 +++++++++++++++++++++++ core/http/routes/systemone.go | 20 + core/schema/systemone.go | 95 ++++ docs/content/features/vllm-cpp.md | 44 ++ pkg/grpc/interface.go | 10 + pkg/grpc/server.go | 15 + 15 files changed, 954 insertions(+), 7 deletions(-) create mode 100644 core/http/endpoints/localai/systemone.go create mode 100644 core/http/routes/systemone.go create mode 100644 core/schema/systemone.go diff --git a/backend/backend.proto b/backend/backend.proto index a27b153ca..4f7c60fe9 100644 --- a/backend/backend.proto +++ b/backend/backend.proto @@ -144,6 +144,12 @@ message TokenClassifyRequest { // PredictOptions.ModelIdentity for the full rationale. Empty means "no // identity supplied" and backends MUST skip the check. string ModelIdentity = 3; + // Labels overrides the backend's configured entity labels for this + // request. Empty means "use the model's configured labels" (the + // default for PII detection, where labels are fixed at load time). + // Non-empty enables zero-shot per-request label selection (kev / + // SystemOne: each question type supplies its own labels). + repeated string labels = 4; } // TokenClassifyEntity is one detected entity span. Byte offsets are diff --git a/backend/go/vllm-cpp/Makefile b/backend/go/vllm-cpp/Makefile index 256763c1c..c19aeaa8e 100644 --- a/backend/go/vllm-cpp/Makefile +++ b/backend/go/vllm-cpp/Makefile @@ -11,7 +11,7 @@ JOBS?=$(shell nproc --ignore=1 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || e # vllm.cpp version VLLM_CPP_REPO?=https://github.com/mudler/vllm.cpp -VLLM_CPP_VERSION?=ea8c83d75f461a520e41328c44bde6c949453fa6 +VLLM_CPP_VERSION?=5058268d7c6308d4b3bca731b79fce0399c1672e # MLX GEMM provider (darwin/metal only; see the metal branch below for why). # Consumed as the prebuilt pip wheel: building MLX from source needs `xcrun diff --git a/backend/go/vllm-cpp/backend.go b/backend/go/vllm-cpp/backend.go index 5292c42ea..c99399a13 100644 --- a/backend/go/vllm-cpp/backend.go +++ b/backend/go/vllm-cpp/backend.go @@ -10,6 +10,7 @@ package main // backend embeds base.Base and not base.SingleThread). import ( + "context" "fmt" "os" "path/filepath" @@ -275,6 +276,67 @@ func (v *VllmCpp) Predict(opts *pb.PredictOptions) (string, error) { return text, nil } +// defaultNerLabels is the general-purpose entity type set used when the model +// config does not supply ner_labels. These cover the most common NER use cases +// and match the categories the GLiNER2.5 model card demonstrates. +var defaultNerLabels = []string{ + "person", "organization", "location", + "date", "time", "money", "quantity", +} + +// TokenClassify runs zero-shot NER on the loaded GLiNER2.5 engine via the +// vllm_gliner_ner C ABI (ABI v27). The engine refuses non-BoundaryExtractor +// architectures, so a model loaded for chat or embeddings returns an error +// here rather than silent garbage. +func (v *VllmCpp) TokenClassify(_ context.Context, in *pb.TokenClassifyRequest) (*pb.TokenClassifyResponse, error) { + if v.engine == 0 { + return nil, fmt.Errorf("vllm-cpp: model not loaded") + } + labels := v.opts.nerLabels + if len(in.Labels) > 0 { + labels = in.Labels + } + if len(labels) == 0 { + labels = defaultNerLabels + } + threshold := v.opts.nerThreshold + if in.Threshold > 0 { + threshold = in.Threshold + } + maxWidth := v.opts.nerMaxWidth + + labelPtrs, labelBacking := cStringArray(labels) + if len(labelPtrs) == 0 { + return nil, fmt.Errorf("vllm-cpp: no NER labels configured") + } + labelsPtr := uintptr(unsafe.Pointer(&labelPtrs[0])) // #nosec G103 -- borrowed by C for the call only + + var out cNerResult + rc := vllmGlinerNer(v.engine, in.Text, labelsPtr, int32(len(labelPtrs)), threshold, maxWidth, unsafe.Pointer(&out)) // #nosec G103 -- POD in/out params + runtime.KeepAlive(labelBacking) + if rc != vllmOK { + return nil, fmt.Errorf("vllm-cpp: NER failed: %s", vllmLastError()) + } + defer vllmNerResultFree(unsafe.Pointer(&out)) // #nosec G103 -- frees C-owned members + + entities := make([]*pb.TokenClassifyEntity, 0, out.nEntities) + if out.nEntities > 0 && out.entities != 0 { + //nolint:govet // C-owned array, valid for this call before vllmNerResultFree + cents := unsafe.Slice((*cNerEntity)(unsafe.Pointer(out.entities)), int(out.nEntities)) // #nosec G103 -- C-owned, copied out immediately + for i := range cents { + e := ¢s[i] + entities = append(entities, &pb.TokenClassifyEntity{ + EntityGroup: goString(e.label), + Start: e.charStart, + End: e.charEnd, + Score: e.confidence, + Text: goString(e.text), + }) + } + } + return &pb.TokenClassifyResponse{Entities: entities}, nil +} + func (v *VllmCpp) PredictStream(opts *pb.PredictOptions, results chan string) error { if v.engine == 0 { close(results) diff --git a/backend/go/vllm-cpp/govllmcpp.go b/backend/go/vllm-cpp/govllmcpp.go index db7555cd2..a6f1935d3 100644 --- a/backend/go/vllm-cpp/govllmcpp.go +++ b/backend/go/vllm-cpp/govllmcpp.go @@ -21,7 +21,7 @@ import ( // the header of the VLLM_CPP_VERSION pinned in the Makefile: the build checks // the two against each other, because a mismatch is only caught at runtime by // registerLib, where it takes the backend down on every load (issue #11379). -const abiVersion = 26 +const abiVersion = 27 // The ABI's tri-state toggles (enable_prefix_caching ABI v7, // enable_jump_forward ABI v10) share one encoding: 0 is NOT "off", it is @@ -252,8 +252,30 @@ var ( vllmVideoResultFree func(out unsafe.Pointer) vllmVideoMuxArgv func(params, outArgv, outArgc unsafe.Pointer) int32 vllmVideoMuxArgvFre func(argv uintptr, argc int32) + + // Zero-shot NER (ABI v27, GLiNER2.5). + vllmGlinerNer func(engine uintptr, text string, labels uintptr, nLabels int32, threshold float32, maxWidth int32, out unsafe.Pointer) int32 + vllmNerResultFree func(out unsafe.Pointer) ) +// cNerEntity mirrors vllm_ner_entity. Layout matches the C struct on LP64: +// two pointer-width fields, four int32, one float, padded to 40 bytes. +type cNerEntity struct { + label uintptr // char* + text uintptr // char* + charStart int32 + charEnd int32 + tokenStart int32 + tokenEnd int32 + confidence float32 +} + +// cNerResult mirrors vllm_ner_result. +type cNerResult struct { + entities uintptr // vllm_ner_entity* + nEntities int32 +} + type libFunc struct { ptr any name string @@ -285,6 +307,8 @@ func registerLib(libName string) error { {&vllmVideoResultFree, "vllm_video_result_free"}, {&vllmVideoMuxArgv, "vllm_video_mux_argv"}, {&vllmVideoMuxArgvFre, "vllm_video_mux_argv_free"}, + {&vllmGlinerNer, "vllm_gliner_ner"}, + {&vllmNerResultFree, "vllm_ner_result_free"}, } { purego.RegisterLibFunc(lf.ptr, lib, lf.name) } diff --git a/backend/go/vllm-cpp/options.go b/backend/go/vllm-cpp/options.go index 7e3edf4d3..1a36e62b3 100644 --- a/backend/go/vllm-cpp/options.go +++ b/backend/go/vllm-cpp/options.go @@ -65,6 +65,15 @@ type loadOptions struct { // MiniMax-H3 video+audio generation (ABI v12). Present only when the config // carries at least one of its keys; see videoOptions.engaged. video videoOptions + // Zero-shot NER labels (ABI v27, GLiNER2.5). GLiNER2.5 is truly zero-shot: + // the model ships no default labels, so the entity types to extract are + // supplied here from engine_args.ner_labels. When empty, a general-purpose + // default set is used. + nerLabels []string + // nerThreshold is the default sigmoid floor (0 = model default 0.5). + nerThreshold float32 + // nerMaxWidth is the maximum span width in tokens (0 = engine default 12). + nerMaxWidth int32 } // videoOptions is the MiniMax-H3 checkpoint SET plus its generation defaults. @@ -342,6 +351,22 @@ func applyEngineArgs(lo *loadOptions, engineArgs string) { if b, ok := v.(bool); ok { lo.enableJumpForward = boolTriState(b) } + case "ner_labels": + if arr, ok := v.([]any); ok { + for _, e := range arr { + if s, ok := e.(string); ok && s != "" { + lo.nerLabels = append(lo.nerLabels, s) + } + } + } + case "ner_threshold": + if f, ok := v.(float64); ok { + lo.nerThreshold = float32(f) + } + case "ner_max_width": + if f, ok := v.(float64); ok { + lo.nerMaxWidth = int32(f) + } default: if s, ok := videoScalarString(v); ok && applyVideoOption(&lo.video, k, s) { continue diff --git a/backend/go/vllm-cpp/vllmcpp_test.go b/backend/go/vllm-cpp/vllmcpp_test.go index 201fd1692..a4de363d0 100644 --- a/backend/go/vllm-cpp/vllmcpp_test.go +++ b/backend/go/vllm-cpp/vllmcpp_test.go @@ -16,7 +16,7 @@ func TestVllmCpp(t *testing.T) { RunSpecs(t, "vllm-cpp suite") } -// The Go POD mirrors must match the C struct layout of vllm.h (ABI v26) +// The Go POD mirrors must match the C struct layout of vllm.h (ABI v27) // byte-for-byte: these offsets are the C offsets on LP64 (linux/darwin // amd64+arm64). A failure here means govllmcpp.go drifted from vllm.h. var _ = Describe("C ABI struct mirrors", func() { @@ -24,7 +24,7 @@ var _ = Describe("C ABI struct mirrors", func() { // VLLM_ABI_VERSION in the vllm.h of VLLM_CPP_VERSION (Makefile). // Moving the pin past this without growing the mirrors below ships a // backend that refuses every load at startup (issue #11379). - Expect(abiVersion).To(Equal(26)) + Expect(abiVersion).To(Equal(27)) }) It("cModelParams matches vllm_model_params", func() { @@ -92,6 +92,25 @@ var _ = Describe("C ABI struct mirrors", func() { Expect(unsafe.Offsetof(c.CompletionTokens)).To(Equal(uintptr(20))) Expect(unsafe.Sizeof(c)).To(Equal(uintptr(24))) }) + + It("cNerEntity matches vllm_ner_entity (ABI v27)", func() { + var e cNerEntity + Expect(unsafe.Offsetof(e.label)).To(Equal(uintptr(0))) + Expect(unsafe.Offsetof(e.text)).To(Equal(uintptr(8))) + Expect(unsafe.Offsetof(e.charStart)).To(Equal(uintptr(16))) + Expect(unsafe.Offsetof(e.charEnd)).To(Equal(uintptr(20))) + Expect(unsafe.Offsetof(e.tokenStart)).To(Equal(uintptr(24))) + Expect(unsafe.Offsetof(e.tokenEnd)).To(Equal(uintptr(28))) + Expect(unsafe.Offsetof(e.confidence)).To(Equal(uintptr(32))) + Expect(unsafe.Sizeof(e)).To(Equal(uintptr(40))) + }) + + It("cNerResult matches vllm_ner_result (ABI v27)", func() { + var r cNerResult + Expect(unsafe.Offsetof(r.entities)).To(Equal(uintptr(0))) + Expect(unsafe.Offsetof(r.nEntities)).To(Equal(uintptr(8))) + Expect(unsafe.Sizeof(r)).To(Equal(uintptr(16))) + }) }) // Pin/mirror skew is the failure mode this backend is most exposed to: the Go diff --git a/core/backend/token_classify.go b/core/backend/token_classify.go index e316f2253..588d4696e 100644 --- a/core/backend/token_classify.go +++ b/core/backend/token_classify.go @@ -30,6 +30,11 @@ type TokenClassifyOptions struct { // callers (e.g. the PII redactor's MinScore) can still filter // further once they know the per-request policy. Threshold float32 + // Labels overrides the backend's configured entity labels for this + // request. Empty means "use the model's configured labels" (the PII + // default). Non-empty enables zero-shot per-request label selection + // (kev / SystemOne questions). + Labels []string } // TokenClassifier runs a token-classification model over text and @@ -39,6 +44,9 @@ type TokenClassifyOptions struct { // core/services/routing/piidetector). type TokenClassifier interface { TokenClassify(ctx context.Context, text string) ([]TokenEntity, error) + // TokenClassifyWithLabels runs NER with the given labels, overriding + // the model's configured labels for this call. + TokenClassifyWithLabels(ctx context.Context, text string, labels []string) ([]TokenEntity, error) } // NewTokenClassifier binds (loader, modelConfig, appConfig) into a @@ -63,6 +71,19 @@ func (m *modelTokenClassifier) TokenClassify(ctx context.Context, text string) ( return fn(ctx) } +// TokenClassifyWithLabels runs NER with the given labels, overriding the +// model's configured labels for this call. Used by the SystemOne endpoints +// where each question supplies its own labels. +func (m *modelTokenClassifier) TokenClassifyWithLabels(ctx context.Context, text string, labels []string) ([]TokenEntity, error) { + opts := m.opts + opts.Labels = labels + fn, err := ModelTokenClassify(text, opts, m.loader, m.modelConfig, m.appConfig) + if err != nil { + return nil, err + } + return fn(ctx) +} + // ModelTokenClassify loads the backend for modelConfig and returns a // closure that classifies `text`. Mirrors ModelScore: the closure is // bound to the loaded model so a caller can reuse it within a request @@ -98,6 +119,7 @@ func ModelTokenClassify(text string, opts TokenClassifyOptions, loader *model.Mo ModelIdentity: modelConfig.Model, Text: text, Threshold: opts.Threshold, + Labels: opts.Labels, }) entities := tokenClassifyResponseToEntities(resp) if appConfig.EnableTracing { diff --git a/core/config/backend_capabilities.go b/core/config/backend_capabilities.go index e97390645..19c628535 100644 --- a/core/config/backend_capabilities.go +++ b/core/config/backend_capabilities.go @@ -342,12 +342,17 @@ var BackendCapabilities = map[string]BackendCapability{ // // AcceptsImages is the fl2va keyframe (start_image/end_image), the same // reason longcat-video declares it; the text path takes no image input. + // + // TokenClassify is possible (GLiNER2.5 zero-shot NER via vllm_gliner_ner, + // ABI v27), declared explicitly via known_usecases: [token_classify]. The + // engine refuses non-BoundaryExtractor architectures, so a chat or embedding + // model returns an error rather than silent garbage. "vllm-cpp": { - GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateVideo}, - PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVideo}, + GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateVideo, MethodTokenClassify}, + PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVideo, UsecaseTokenClassify}, DefaultUsecases: []string{UsecaseChat}, AcceptsImages: true, - Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation plus MiniMax-H3 video+audio generation", + Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, and GLiNER2.5 zero-shot NER", }, "vllm-omni": { GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateImage, MethodGenerateVideo, MethodTTS}, diff --git a/core/http/app.go b/core/http/app.go index d7c8af3c4..f03522a45 100644 --- a/core/http/app.go +++ b/core/http/app.go @@ -464,6 +464,7 @@ func API(application *application.Application) (*echo.Echo, error) { // mode by attributing requests to the synthetic "local" user. routes.RegisterUsageRoutes(e, application) routes.RegisterPIIRoutes(e, application) + routes.RegisterSystemOneRoutes(e, application) routes.RegisterMiddlewareRoutes(e, application) routes.RegisterElevenLabsRoutes(e, requestExtractor, application.ModelConfigLoader(), application.ModelLoader(), application.ApplicationConfig()) diff --git a/core/http/endpoints/localai/systemone.go b/core/http/endpoints/localai/systemone.go new file mode 100644 index 000000000..9331dbf73 --- /dev/null +++ b/core/http/endpoints/localai/systemone.go @@ -0,0 +1,599 @@ +package localai + +import ( + "encoding/json" + "fmt" + "math" + "math/rand" + "net/http" + "sort" + "strconv" + "strings" + "time" + + "github.com/labstack/echo/v4" + "github.com/mudler/LocalAI/core/application" + "github.com/mudler/LocalAI/core/backend" + "github.com/mudler/LocalAI/core/schema" +) + +// --------------------------------------------------------------------------- +// Helpers — ported from kev/api.py (render, r2, choice_confidence, +// score_confidence, softmax) and mirrored in vllm.cpp api_server.cpp. +// --------------------------------------------------------------------------- + +// renderState flattens a JSON value into text, mirroring kev's render(). +// Field names are kept as labels; arrays become "- item" bullets; objects +// become "key: value" lines. +func renderState(v interface{}, indent int) string { + switch val := v.(type) { + case nil: + return "" + case string: + return val + case bool: + if val { + return "true" + } + return "false" + case float64: + b, _ := json.Marshal(val) + return string(b) + case []interface{}: + pad := strings.Repeat(" ", indent) + var parts []string + for _, item := range val { + rendered := renderState(item, indent+1) + rendered = strings.TrimLeft(rendered, " \t\n") + parts = append(parts, pad+"- "+rendered) + } + return strings.Join(parts, "\n") + case map[string]interface{}: + pad := strings.Repeat(" ", indent) + keys := make([]string, 0, len(val)) + for k := range val { + keys = append(keys, k) + } + sort.Strings(keys) + var parts []string + for i, k := range keys { + if i > 0 { + parts = append(parts, "\n") + } + switch vv := val[k].(type) { + case map[string]interface{}, []interface{}: + parts = append(parts, pad+k+":\n"+renderState(vv, indent+1)) + default: + parts = append(parts, pad+k+": "+renderState(vv, indent)) + } + } + return strings.Join(parts, "") + default: + b, _ := json.Marshal(v) + return string(b) + } +} + +// r2 rounds to 2 decimal places. +func r2(x float64) float64 { + return math.Round(x*100) / 100 +} + +// choiceConfidence is the normalized margin (kev/api.py:choice_confidence). +func choiceConfidence(p []float64) float64 { + k := len(p) + if k <= 1 { + return 1.0 + } + mx := p[0] + for _, v := range p[1:] { + if v > mx { + mx = v + } + } + return (mx - 1.0/float64(k)) / (1.0 - 1.0/float64(k)) +} + +// scoreConfidence is 1 - E|level - mode| / (L - 1) +// (kev/api.py:score_confidence). +func scoreConfidence(p []float64) float64 { + l := len(p) + if l <= 1 { + return 1.0 + } + mode := 0 + maxP := p[0] + for i := 1; i < l; i++ { + if p[i] > maxP { + maxP = p[i] + mode = i + } + } + s := 0.0 + for i := 0; i < l; i++ { + s += p[i] * math.Abs(float64(i)-float64(mode)) + } + return 1.0 - s/float64(l-1) +} + +// softmax is a numerically stable softmax. +func softmax(scores []float64) []float64 { + if len(scores) == 0 { + return nil + } + mx := scores[0] + for _, s := range scores[1:] { + if s > mx { + mx = s + } + } + exps := make([]float64, len(scores)) + sum := 0.0 + for i, s := range scores { + exps[i] = math.Exp(s - mx) + sum += exps[i] + } + if sum <= 0 { + inv := 1.0 / float64(len(scores)) + for i := range exps { + exps[i] = inv + } + return exps + } + for i := range exps { + exps[i] /= sum + } + return exps +} + +// optionText mirrors kev/api.py:option_text. "name" if desc is null/empty, +// else "name: rendered_desc". +func optionText(name string, desc json.RawMessage) string { + if len(desc) == 0 || string(desc) == "null" { + return name + } + var v interface{} + if err := json.Unmarshal(desc, &v); err != nil { + return name + } + rendered := renderState(v, 0) + if rendered == "" { + return name + } + return name + ": " + rendered +} + +// getQuestionInstructions checks instructions first, then instr (alias). +// The value is rendered to text, matching vllm.cpp GetInstructions. +func getQuestionInstructions(q schema.SystemOneQuestion) string { + if len(q.Instructions) > 0 { + var v interface{} + if err := json.Unmarshal(q.Instructions, &v); err == nil { + return renderState(v, 0) + } + } + return q.Instr +} + +// --------------------------------------------------------------------------- +// Parsed question (internal). +// --------------------------------------------------------------------------- + +type parsedQuestion struct { + id string + qtype string // "noul", "choice", "score" + keys []string + labels []string +} + +type parsedSystemOne struct { + text string + model string + threshold float32 + questions []parsedQuestion + allLabels []string +} + +func parseSystemOneRequest(req *schema.SystemOneRequest) (*parsedSystemOne, error) { + p := &parsedSystemOne{ + model: req.Model, + threshold: 0.5, + } + if req.Threshold != nil { + p.threshold = *req.Threshold + } + + var stateVal interface{} + if err := json.Unmarshal(req.State, &stateVal); err != nil { + return nil, fmt.Errorf("state is not valid JSON: %w", err) + } + p.text = renderState(stateVal, 0) + + if len(req.Questions) == 0 { + return nil, fmt.Errorf("questions is required and must contain at least one question") + } + + qids := make([]string, 0, len(req.Questions)) + for k := range req.Questions { + qids = append(qids, k) + } + sort.Strings(qids) + + for _, qid := range qids { + q := req.Questions[qid] + pq := parsedQuestion{id: qid, qtype: q.Type} + switch q.Type { + case "noul": + instr := getQuestionInstructions(q) + pq.labels = []string{instr} + pq.keys = []string{"no", "yes"} + case "choice": + var criteria map[string]json.RawMessage + if err := json.Unmarshal(q.Criteria, &criteria); err != nil || len(criteria) == 0 { + return nil, fmt.Errorf("question %q (choice) requires a non-empty criteria object", qid) + } + ckeys := make([]string, 0, len(criteria)) + for k := range criteria { + ckeys = append(ckeys, k) + } + sort.Strings(ckeys) + for _, ck := range ckeys { + pq.keys = append(pq.keys, ck) + pq.labels = append(pq.labels, optionText(ck, criteria[ck])) + } + case "score": + var criteria []json.RawMessage + if err := json.Unmarshal(q.Criteria, &criteria); err != nil || len(criteria) < 2 { + return nil, fmt.Errorf("question %q (score) requires a criteria array with >= 2 levels", qid) + } + for _, level := range criteria { + var lv interface{} + _ = json.Unmarshal(level, &lv) + rendered := renderState(lv, 0) + pq.keys = append(pq.keys, rendered) + pq.labels = append(pq.labels, rendered) + } + default: + return nil, fmt.Errorf("question %q has unknown type: %s", qid, q.Type) + } + p.questions = append(p.questions, pq) + p.allLabels = append(p.allLabels, pq.labels...) + } + return p, nil +} + +// buildSystemOneAnswer produces one kev answer from NER entities. +func buildSystemOneAnswer(q *parsedQuestion, entities []backend.TokenEntity) schema.SystemOneAnswer { + scores := make([]float64, len(q.labels)) + for i, label := range q.labels { + var maxConf float32 + for _, e := range entities { + if e.Group == label && e.Score > maxConf { + maxConf = e.Score + } + } + scores[i] = float64(maxConf) + } + + switch q.qtype { + case "noul": + probs := []float64{1.0 - scores[0], scores[0]} + var ents []schema.SystemOneEntity + for _, e := range entities { + if e.Group == q.labels[0] { + ents = append(ents, schema.SystemOneEntity{ + Text: e.Text, + Start: e.Start, + End: e.End, + Confidence: e.Score, + }) + } + } + noul := r2(probs[1]) + return schema.SystemOneAnswer{ + Type: "noul", + Noul: &noul, + Entities: ents, + } + + case "choice": + probs := softmax(scores) + argmax := 0 + for i := 1; i < len(probs); i++ { + if probs[i] > probs[argmax] { + argmax = i + } + } + dist := make(map[string]float64, len(q.keys)) + for i, k := range q.keys { + dist[k] = r2(probs[i]) + } + choice := q.keys[argmax] + conf := r2(choiceConfidence(probs)) + return schema.SystemOneAnswer{ + Type: "choice", + Choice: &choice, + Confidence: &conf, + Probabilities: dist, + } + + default: // score + probs := softmax(scores) + var score float64 + for i, pr := range probs { + score += float64(i) * pr + } + legend := make(map[string]string, len(q.keys)) + dist := make(map[string]float64, len(q.keys)) + for i, k := range q.keys { + legend[strconv.Itoa(i)] = k + dist[strconv.Itoa(i)] = r2(probs[i]) + } + sc := r2(score) + conf := r2(scoreConfidence(probs)) + return schema.SystemOneAnswer{ + Type: "score", + Score: &sc, + Legend: legend, + Probabilities: dist, + Confidence: &conf, + } + } +} + +// --------------------------------------------------------------------------- +// Model resolution. +// --------------------------------------------------------------------------- + +func resolveClassifier(app *application.Application, modelName string, threshold float32) (backend.TokenClassifier, error) { + cl := app.ModelConfigLoader() + if cl == nil { + return nil, fmt.Errorf("model config loader unavailable") + } + cfg, ok := cl.GetModelConfig(modelName) + if !ok { + return nil, fmt.Errorf("model %q not found", modelName) + } + opts := backend.TokenClassifyOptions{ + Threshold: threshold, + } + return backend.NewTokenClassifier(app.ModelLoader(), cfg, app.ApplicationConfig(), opts), nil +} + +func systemOneError(c echo.Context, status int, msg string) error { + return c.JSON(status, map[string]any{ + "error": map[string]string{ + "message": msg, + "type": "invalid_request", + }, + }) +} + +// --------------------------------------------------------------------------- +// Endpoints. +// --------------------------------------------------------------------------- + +// SystemOneEndpoint handles POST /v1/systemone. +// Runs one NER pass over the rendered state with all question labels, then +// builds a kev-compatible answer for each question. +// @Summary Answer structured-extraction questions over state text. +// @Description Runs zero-shot NER over the supplied state and answers each question. Question types: noul (binary entity presence), choice (pick one option), score (pick one level). +// @Tags systemone +// @Param request body schema.SystemOneRequest true "state + questions" +// @Success 200 {object} schema.SystemOneResponse +// @Router /v1/systemone [post] +func SystemOneEndpoint(app *application.Application) echo.HandlerFunc { + return func(c echo.Context) error { + var req schema.SystemOneRequest + if err := c.Bind(&req); err != nil { + return systemOneError(c, http.StatusBadRequest, "invalid request body") + } + if req.Model == "" { + return systemOneError(c, http.StatusBadRequest, "model is required") + } + parsed, err := parseSystemOneRequest(&req) + if err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } + classifier, err := resolveClassifier(app, req.Model, parsed.threshold) + if err != nil { + return systemOneError(c, http.StatusNotFound, err.Error()) + } + start := time.Now() + entities, err := classifier.TokenClassifyWithLabels(c.Request().Context(), parsed.text, parsed.allLabels) + if err != nil { + return systemOneError(c, http.StatusInternalServerError, err.Error()) + } + latencyMs := float64(time.Since(start).Microseconds()) / 1000.0 + answers := make(map[string]schema.SystemOneAnswer, len(parsed.questions)) + for i := range parsed.questions { + answers[parsed.questions[i].id] = buildSystemOneAnswer(&parsed.questions[i], entities) + } + return c.JSON(http.StatusOK, schema.SystemOneResponse{ + Model: req.Model, + Answers: answers, + Usage: schema.SystemOneUsage{InputTokens: 0, OutputTokens: 0}, + LatencyMs: r2(latencyMs), + }) + } +} + +// SystemOnePermuteEndpoint handles POST /v1/systemone/permute. +// Re-runs one choice question under n_perm option orders with a seeded RNG. +// @Summary Re-run a choice question under multiple option orders. +// @Description Re-runs one choice question under n_perm option orders. Reports per-order probabilities, argmax stability, and spread. +// @Tags systemone +// @Param request body schema.SystemOnePermuteRequest true "request + question + n_perm + seed" +// @Success 200 {object} schema.SystemOnePermuteResponse +// @Router /v1/systemone/permute [post] +func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc { + return func(c echo.Context) error { + var req schema.SystemOnePermuteRequest + if err := c.Bind(&req); err != nil { + return systemOneError(c, http.StatusBadRequest, "invalid request body") + } + if req.Request.Model == "" { + return systemOneError(c, http.StatusBadRequest, "model is required") + } + if req.Question == "" { + return systemOneError(c, http.StatusBadRequest, "question is required") + } + parsed, err := parseSystemOneRequest(&req.Request) + if err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } + var target *parsedQuestion + for i := range parsed.questions { + if parsed.questions[i].id == req.Question { + target = &parsed.questions[i] + break + } + } + if target == nil { + return systemOneError(c, http.StatusBadRequest, fmt.Sprintf("question %q not found", req.Question)) + } + if target.qtype != "choice" { + return systemOneError(c, http.StatusBadRequest, "question must be a choice question") + } + classifier, err := resolveClassifier(app, req.Request.Model, parsed.threshold) + if err != nil { + return systemOneError(c, http.StatusNotFound, err.Error()) + } + nPerm := req.NPerm + if nPerm <= 0 { + nPerm = 6 + } + rng := rand.New(rand.NewSource(req.Seed)) // #nosec G404 -- seeded RNG for reproducible permutations, not crypto + runs := make([]schema.SystemOnePermuteRun, 0, nPerm) + minProb := make([]float64, len(target.keys)) + maxProb := make([]float64, len(target.keys)) + for i := range minProb { + minProb[i] = 1.0 + maxProb[i] = 0.0 + } + firstChoice := "" + argmaxStable := true + + for i := 0; i < nPerm; i++ { + idx := make([]int, len(target.keys)) + for j := range idx { + idx[j] = j + } + if i > 0 { + rng.Shuffle(len(idx), func(a, b int) { idx[a], idx[b] = idx[b], idx[a] }) + } + orderKeys := make([]string, len(idx)) + orderLabels := make([]string, len(idx)) + for j, k := range idx { + orderKeys[j] = target.keys[k] + orderLabels[j] = target.labels[k] + } + start := time.Now() + entities, err := classifier.TokenClassifyWithLabels(c.Request().Context(), parsed.text, orderLabels) + if err != nil { + return systemOneError(c, http.StatusInternalServerError, err.Error()) + } + latencyMs := float64(time.Since(start).Microseconds()) / 1000.0 + + scores := make([]float64, len(orderLabels)) + for j, label := range orderLabels { + var maxConf float32 + for _, e := range entities { + if e.Group == label && e.Score > maxConf { + maxConf = e.Score + } + } + scores[j] = float64(maxConf) + } + probs := softmax(scores) + argmax := 0 + for j := 1; j < len(probs); j++ { + if probs[j] > probs[argmax] { + argmax = j + } + } + probDist := make(map[string]float64, len(orderKeys)) + for j, key := range orderKeys { + probDist[key] = r2(probs[j]) + for k, tk := range target.keys { + if key == tk { + if probs[j] < minProb[k] { + minProb[k] = probs[j] + } + if probs[j] > maxProb[k] { + maxProb[k] = probs[j] + } + break + } + } + } + choice := orderKeys[argmax] + if i == 0 { + firstChoice = choice + } else if choice != firstChoice { + argmaxStable = false + } + runs = append(runs, schema.SystemOnePermuteRun{ + Order: orderKeys, + Probabilities: probDist, + Choice: choice, + LatencyMs: r2(latencyMs), + }) + } + + spread := make(map[string]float64, len(target.keys)) + for k, key := range target.keys { + spread[key] = r2(maxProb[k] - minProb[k]) + } + return c.JSON(http.StatusOK, schema.SystemOnePermuteResponse{ + Runs: runs, + ArgmaxStable: argmaxStable, + Spread: spread, + }) + } +} + +// SystemOneSeparateEndpoint handles POST /v1/systemone/separate. +// Answers each question in its own NER call (N passes). Response shape +// matches /v1/systemone. +// @Summary Answer each question in a separate NER pass. +// @Description Runs N independent NER passes, one per question, against the same state. Response shape matches /v1/systemone. +// @Tags systemone +// @Param request body schema.SystemOneRequest true "state + questions" +// @Success 200 {object} schema.SystemOneResponse +// @Router /v1/systemone/separate [post] +func SystemOneSeparateEndpoint(app *application.Application) echo.HandlerFunc { + return func(c echo.Context) error { + var req schema.SystemOneRequest + if err := c.Bind(&req); err != nil { + return systemOneError(c, http.StatusBadRequest, "invalid request body") + } + if req.Model == "" { + return systemOneError(c, http.StatusBadRequest, "model is required") + } + parsed, err := parseSystemOneRequest(&req) + if err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } + classifier, err := resolveClassifier(app, req.Model, parsed.threshold) + if err != nil { + return systemOneError(c, http.StatusNotFound, err.Error()) + } + start := time.Now() + answers := make(map[string]schema.SystemOneAnswer, len(parsed.questions)) + for i := range parsed.questions { + entities, err := classifier.TokenClassifyWithLabels(c.Request().Context(), parsed.text, parsed.questions[i].labels) + if err != nil { + return systemOneError(c, http.StatusInternalServerError, err.Error()) + } + answers[parsed.questions[i].id] = buildSystemOneAnswer(&parsed.questions[i], entities) + } + latencyMs := float64(time.Since(start).Microseconds()) / 1000.0 + return c.JSON(http.StatusOK, schema.SystemOneResponse{ + Model: req.Model, + Answers: answers, + Usage: schema.SystemOneUsage{InputTokens: 0, OutputTokens: 0}, + LatencyMs: r2(latencyMs), + }) + } +} diff --git a/core/http/routes/systemone.go b/core/http/routes/systemone.go new file mode 100644 index 000000000..d6a70e5fc --- /dev/null +++ b/core/http/routes/systemone.go @@ -0,0 +1,20 @@ +package routes + +import ( + "github.com/labstack/echo/v4" + "github.com/mudler/LocalAI/core/application" + "github.com/mudler/LocalAI/core/http/endpoints/localai" +) + +// RegisterSystemOneRoutes wires the kev-compatible SystemOne endpoints. +// These provide zero-shot structured extraction over arbitrary state text +// using a GLiNER2-backed NER model. The API mirrors the kev project +// (jaredpalmer/kev serve.py): POST /v1/systemone answers all questions in +// one NER pass; POST /v1/systemone/permute re-runs one choice question +// under n_perm option orders; POST /v1/systemone/separate answers each +// question in its own NER pass. +func RegisterSystemOneRoutes(e *echo.Echo, app *application.Application) { + e.POST("/v1/systemone", localai.SystemOneEndpoint(app)) + e.POST("/v1/systemone/permute", localai.SystemOnePermuteEndpoint(app)) + e.POST("/v1/systemone/separate", localai.SystemOneSeparateEndpoint(app)) +} diff --git a/core/schema/systemone.go b/core/schema/systemone.go new file mode 100644 index 000000000..8fa954108 --- /dev/null +++ b/core/schema/systemone.go @@ -0,0 +1,95 @@ +package schema + +import "encoding/json" + +// SystemOneRequest is the body for POST /v1/systemone, +// /v1/systemone/separate, and the inner `request` of /v1/systemone/permute. +// Mirrors the kev project's SystemOneRequest (jaredpalmer/kev serve.py). +type SystemOneRequest struct { + // State is the text (or any JSON value) to extract from. A non-string + // value is rendered to its JSON representation before NER. + State json.RawMessage `json:"state"` + // Questions maps question IDs to their definitions. + Questions map[string]SystemOneQuestion `json:"questions"` + // Model names the NER model to use. Optional. + Model string `json:"model,omitempty"` + // Threshold is the minimum entity confidence (0–1). Default 0.5. + Threshold *float32 `json:"threshold,omitempty"` + // MaxWidth is the maximum span width in tokens. Default 12. + MaxWidth *int `json:"max_width,omitempty"` +} + +// SystemOneQuestion defines one question. Type is "noul", "choice", or +// "score". Instructions (or the instr alias) is human-readable instruction +// text. For noul questions the rendered instruction is the NER label. +// Criteria is: +// - noul: omitted +// - choice: a map of option_name → description (NER label is +// "name: description", or "name" when the description is null/empty) +// - score: an array of level descriptions (each rendered text is a NER label) +type SystemOneQuestion struct { + Type string `json:"type"` + Instructions json.RawMessage `json:"instructions,omitempty"` + Instr string `json:"instr,omitempty"` + Criteria json.RawMessage `json:"criteria,omitempty"` +} + +// SystemOneResponse is the shared response shape for /v1/systemone and +// /v1/systemone/separate. +type SystemOneResponse struct { + Model string `json:"model"` + Answers map[string]SystemOneAnswer `json:"answers"` + Usage SystemOneUsage `json:"usage"` + LatencyMs float64 `json:"latency_ms"` +} + +type SystemOneUsage struct { + InputTokens int `json:"input_tokens"` + OutputTokens int `json:"output_tokens"` +} + +// SystemOneAnswer is one question's answer. The fields populated depend on +// the question type: +// - noul: Noul (float 0–1), Entities +// - choice: Choice (string), Confidence, Probabilities (map) +// - score: Score (float), Legend (map), Probabilities (map), Confidence +type SystemOneAnswer struct { + Type string `json:"type"` + Noul *float64 `json:"noul,omitempty"` + Entities []SystemOneEntity `json:"entities,omitempty"` + Choice *string `json:"choice,omitempty"` + Confidence *float64 `json:"confidence,omitempty"` + Probabilities map[string]float64 `json:"probabilities,omitempty"` + Score *float64 `json:"score,omitempty"` + Legend map[string]string `json:"legend,omitempty"` +} + +type SystemOneEntity struct { + Text string `json:"text"` + Start int `json:"start"` + End int `json:"end"` + Confidence float32 `json:"confidence"` +} + +// SystemOnePermuteRequest is the body for POST /v1/systemone/permute. +type SystemOnePermuteRequest struct { + Request SystemOneRequest `json:"request"` + Question string `json:"question"` + NPerm int `json:"n_perm,omitempty"` + Seed int64 `json:"seed,omitempty"` +} + +// SystemOnePermuteRun is one permutation's result. +type SystemOnePermuteRun struct { + Order []string `json:"order"` + Probabilities map[string]float64 `json:"probabilities"` + Choice string `json:"choice"` + LatencyMs float64 `json:"latency_ms"` +} + +// SystemOnePermuteResponse is the response for POST /v1/systemone/permute. +type SystemOnePermuteResponse struct { + Runs []SystemOnePermuteRun `json:"runs"` + ArgmaxStable bool `json:"argmax_stable"` + Spread map[string]float64 `json:"spread"` +} diff --git a/docs/content/features/vllm-cpp.md b/docs/content/features/vllm-cpp.md index d4bd2561d..74e3c0d38 100644 --- a/docs/content/features/vllm-cpp.md +++ b/docs/content/features/vllm-cpp.md @@ -133,6 +133,50 @@ engine_args: tool_parser: qwen3_coder ``` +## Named entity recognition (GLiNER2.5) + +The `vllm-cpp` backend serves [GLiNER2.5](https://huggingface.co/fastino/gliner2.5-multi-v1), +a zero-shot NER and structured-extraction model. Point the backend at the +safetensors directory and the backend exposes the `TokenClassify` gRPC method, +which LocalAI maps to its standard NER API surface. + +Labels are supplied at inference time, not baked into the model config. Set +them in `engine_args`: + +```yaml +engine_args: + ner_labels: "person,organization,location,date,time,money,quantity" + ner_threshold: 0.5 + ner_max_width: 12 +``` + +`ner_labels` is a comma-separated list. When omitted, the backend falls back to +a built-in default set (`person`, `organization`, `location`, `date`, `time`, +`money`, `quantity`). `ner_threshold` is the sigmoid cutoff (default 0.5); +`ner_max_width` is the maximum span length in tokens (default 12). + +The model runs the DeBERTa v2 encoder with disentangled attention on the host +forward, which is the required contract for pooling models in vllm.cpp. A +device-resident forward is tracked as a performance optimization, not a +correctness gap. + +### SystemOne structured-extraction API + +The `vllm-cpp` backend also exposes kev-compatible SystemOne endpoints that +turn zero-shot NER into structured question answering. These mirror the API +from the [kev](https://github.com/jaredpalmer/kev) project: + +| Endpoint | Method | Description | +|---|---|---| +| `/v1/systemone` | POST | Answer all questions in one NER pass | +| `/v1/systemone/permute` | POST | Re-run one choice question under n_perm option orders | +| `/v1/systemone/separate` | POST | Answer each question in its own NER pass (N passes) | + +Each question has a `type` of `noul` (binary entity presence), `choice` (pick +one option), or `score` (pick one level). The `model` field in the request body +selects the NER model. Labels are derived from the question definition, so no +`ner_labels` configuration is needed for these endpoints. + ## Beyond text generation The `vllm-cpp` backend also serves MiniMax-H3, which generates video and audio diff --git a/pkg/grpc/interface.go b/pkg/grpc/interface.go index 715a0acf5..998eb0745 100644 --- a/pkg/grpc/interface.go +++ b/pkg/grpc/interface.go @@ -103,3 +103,13 @@ type AIModelRich interface { PredictRich(*pb.PredictOptions) (*pb.Reply, error) PredictStreamRich(*pb.PredictOptions, chan<- *pb.Reply) error } + +// ClassifyModel is an optional extension to AIModel for backends that +// implement the TokenClassify RPC (zero-shot NER). The gRPC server +// type-asserts to this interface; backends that do not implement it +// fall through to the UnimplementedBackendServer default. This mirrors +// the AIModelRich pattern: adding a method to AIModel itself would +// break every backend, so the capability is opt-in. +type ClassifyModel interface { + TokenClassify(context.Context, *pb.TokenClassifyRequest) (*pb.TokenClassifyResponse, error) +} diff --git a/pkg/grpc/server.go b/pkg/grpc/server.go index 67af9d60a..7841d1c40 100644 --- a/pkg/grpc/server.go +++ b/pkg/grpc/server.go @@ -102,6 +102,21 @@ func (s *server) Embedding(ctx context.Context, in *pb.PredictOptions) (*pb.Embe }, nil } +func (s *server) TokenClassify(ctx context.Context, in *pb.TokenClassifyRequest) (*pb.TokenClassifyResponse, error) { + if err := s.checkModelIdentity(in); err != nil { + return nil, err + } + cm, ok := s.llm.(ClassifyModel) + if !ok { + return nil, status.Errorf(codes.Unimplemented, "method TokenClassify not implemented") + } + if s.llm.Locking() { + s.llm.Lock() + defer s.llm.Unlock() + } + return cm.TokenClassify(ctx, in) +} + func (s *server) LoadModel(ctx context.Context, in *pb.ModelOptions) (*pb.Result, error) { if s.llm.Locking() { s.llm.Lock()