From 84a2b209a0522a1811a280295dd788371b1bc310 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Mon, 28 Sep 2026 21:41:46 +0200 Subject: [PATCH 01/33] chore: :arrow_up: Update ggml-org/llama.cpp to `4da6337767f973e2b4d0797e5b323d77d8565e4a` (#12318) :arrow_up: Update ggml-org/llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/llama-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/llama-cpp/Makefile b/backend/cpp/llama-cpp/Makefile index 8bf0ff5c0..af5529d48 100644 --- a/backend/cpp/llama-cpp/Makefile +++ b/backend/cpp/llama-cpp/Makefile @@ -1,5 +1,5 @@ -LLAMA_VERSION?=95887577ab5fead779581a7030a83c7752ff3234 +LLAMA_VERSION?=4da6337767f973e2b4d0797e5b323d77d8565e4a LLAMA_REPO?=https://github.com/ggerganov/llama.cpp CMAKE_ARGS?= From 2fe459ca5f10341054609e3e20361ac9ef0d4293 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Tue, 29 Sep 2026 08:16:50 +0200 Subject: [PATCH 02/33] chore(model-gallery): :arrow_up: update checksum (#12342) :arrow_up: Checksum updates in gallery/index.yaml Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- gallery/index.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/gallery/index.yaml b/gallery/index.yaml index c2298226b..4e37b0e78 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -297,7 +297,7 @@ files: - filename: ds4flash.gguf uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF - sha256: f33633d55f5379e8571db06674bf7a07a2ea7bb7b44287e1a9bdc3686d69a5c6 + sha256: 0ee532bce971b1464cce75ce74b7e5205dfd5cc4a844d27ce8f4b4de82a7136f - name: "qwopus3.8-27b-flash-v2" variants: - model: qwopus3.8-27b-flash-v2-q8 @@ -54137,7 +54137,7 @@ files: - filename: cohere-transcribe-q4_k.gguf uri: huggingface://cstr/cohere-transcribe-03-2026-GGUF/cohere-transcribe-q4_k.gguf - sha256: 237261c543dc9124a3f08f95b48c9c672896ef0d79dc8cadce3fb4ddc09a2ef8 + sha256: 116f4c4f7ff1b03997100d3a097e13fecd84350555b9979867b241cd6803e4f5 - name: wav2vec2-crispasr url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: From b700da3eb39c94aefe74c204c2ad8f66070796d6 Mon Sep 17 00:00:00 2001 From: mudler-agent Date: Tue, 29 Sep 2026 10:37:31 +0200 Subject: [PATCH 03/33] fix(huggingface): list repos nested more than one directory deep (#12355) The HuggingFace tree API returns each entry's path relative to the repo root ("assets/plots", not "plots"). The recursive listing prefixed the parent directory again, so it requested "assets/assets/plots", got a 404, and failed the whole listing. The importer then treated the URI as a non-HF repo and no importer matched. This broke the import of GGUF repos that keep per-quant subfolders next to a nested assets tree, such as ISTA-DASLab/Qwen3.8-Flash-Next-GSQ-RCO-GGUF. The test mock now returns root-relative directory paths like the real API and routes on the exact tree path, so a doubled path 404s. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto --- pkg/huggingface-api/client.go | 16 +++++----------- pkg/huggingface-api/client_test.go | 20 ++++++++++++++------ 2 files changed, 19 insertions(+), 17 deletions(-) diff --git a/pkg/huggingface-api/client.go b/pkg/huggingface-api/client.go index 1d1c7ae3c..524fe57c5 100644 --- a/pkg/huggingface-api/client.go +++ b/pkg/huggingface-api/client.go @@ -304,14 +304,12 @@ func (c *Client) listFilesInPath(repoID, path string) ([]FileInfo, error) { switch item.Type { // If it's a directory/folder, recursively list its contents case "directory", "folder": - // Build the subfolder path + // The tree API returns every entry's path relative to the repo + // root ("assets/plots", not "plots"), so it is already the path + // to recurse into. Prefixing the parent again requested + // "assets/assets/plots", which 404s and failed the whole listing + // for any repo nested more than one directory deep. subPath := item.Path - if path != "" { - subPath = fmt.Sprintf("%s/%s", path, item.Path) - } - - // Recursively get files from subfolder - // The recursive call will already prepend the subPath to each file's path subFiles, err := c.listFilesInPath(repoID, subPath) if err != nil { return nil, fmt.Errorf("failed to list files in subfolder %s: %w", subPath, err) @@ -319,10 +317,6 @@ func (c *Client) listFilesInPath(repoID, path string) ([]FileInfo, error) { allFiles = append(allFiles, subFiles...) case "file": - // It's a file, prepend the current path to make it relative to root - // if path != "" { - // item.Path = fmt.Sprintf("%s/%s", path, item.Path) - // } allFiles = append(allFiles, item) } } diff --git a/pkg/huggingface-api/client_test.go b/pkg/huggingface-api/client_test.go index feac4dba3..6f60d3ef8 100644 --- a/pkg/huggingface-api/client_test.go +++ b/pkg/huggingface-api/client_test.go @@ -459,7 +459,7 @@ var _ = Describe("HuggingFace API Client", func() { }, { "type": "directory", - "path": "nested", + "path": "subfolder/nested", "size": 0, "oid": "nesteddir123" } @@ -483,15 +483,23 @@ var _ = Describe("HuggingFace API Client", func() { server = httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { urlPath := r.URL.Path w.Header().Set("Content-Type", "application/json") - w.WriteHeader(http.StatusOK) - if strings.Contains(urlPath, "/tree/main/subfolder/nested") { + // Route on the exact tree path: the real API 404s on a wrong + // path, and nested entries carry root-relative paths + // ("subfolder/nested"), so a prefix match would hide a + // doubled "subfolder/subfolder/nested" request. + _, treePath, _ := strings.Cut(urlPath, "/tree/main") + switch treePath { + case "/subfolder/nested": + w.WriteHeader(http.StatusOK) w.Write([]byte(mockNestedResponse)) - } else if strings.Contains(urlPath, "/tree/main/subfolder") { + case "/subfolder": + w.WriteHeader(http.StatusOK) w.Write([]byte(mockSubfolderResponse)) - } else if strings.Contains(urlPath, "/tree/main") { + case "": + w.WriteHeader(http.StatusOK) w.Write([]byte(mockRootResponse)) - } else { + default: w.WriteHeader(http.StatusNotFound) } })) From f9d51ed42137890d99d9f5439cba4fe0e5b604ad Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Tue, 29 Sep 2026 12:29:08 +0200 Subject: [PATCH 04/33] chore: :arrow_up: Update 0xShug0/audio.cpp to `f825d1d1b92af309585aeb656b2a59c44fc603eb` (#12343) :arrow_up: Update 0xShug0/audio.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/audio-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/audio-cpp/Makefile b/backend/cpp/audio-cpp/Makefile index 64638b0c1..16ec08f5a 100644 --- a/backend/cpp/audio-cpp/Makefile +++ b/backend/cpp/audio-cpp/Makefile @@ -9,7 +9,7 @@ # recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean # rebuild and so the bump bot can see the pin. -AUDIO_CPP_VERSION?=77491a33c589c53ff18add050095cf35647c8213 +AUDIO_CPP_VERSION?=f825d1d1b92af309585aeb656b2a59c44fc603eb AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST)))) From bbaa545e272bdf56c5a4b604a7a8511ae5777722 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Tue, 29 Sep 2026 18:16:35 +0200 Subject: [PATCH 05/33] chore: :arrow_up: Update ikawrakow/ik_llama.cpp to `d741de5074cd424dd3ba7cfc4d9b7649f1eb0463` (#12351) :arrow_up: Update ikawrakow/ik_llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/ik-llama-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/ik-llama-cpp/Makefile b/backend/cpp/ik-llama-cpp/Makefile index d32465ec9..569df41fb 100644 --- a/backend/cpp/ik-llama-cpp/Makefile +++ b/backend/cpp/ik-llama-cpp/Makefile @@ -1,5 +1,5 @@ -IK_LLAMA_VERSION?=ed27bf7ed25e637692e89cd341d802522a2cee8a +IK_LLAMA_VERSION?=d741de5074cd424dd3ba7cfc4d9b7649f1eb0463 LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp CMAKE_ARGS?= From e0ad36a7cf0d59203e92d865e4caacb786ba3f5b Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Tue, 29 Sep 2026 18:16:47 +0200 Subject: [PATCH 06/33] chore: :arrow_up: Update ServeurpersoCom/omnivoice.cpp to `53e6c2066150802ad3cd4b655b31c696e78e0019` (#12350) :arrow_up: Update ServeurpersoCom/omnivoice.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/omnivoice-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/omnivoice-cpp/Makefile b/backend/go/omnivoice-cpp/Makefile index 6c8539bd5..82d48b5a9 100644 --- a/backend/go/omnivoice-cpp/Makefile +++ b/backend/go/omnivoice-cpp/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # omnivoice.cpp version OMNIVOICE_REPO?=https://github.com/ServeurpersoCom/omnivoice.cpp -OMNIVOICE_VERSION?=ead199a2bc4c53a57cac90095ae049a111d9e98d +OMNIVOICE_VERSION?=53e6c2066150802ad3cd4b655b31c696e78e0019 SO_TARGET?=libgomnivoicecpp.so CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF From 73cb1c3fdf4795689f55e4b84e4e7bcfe4cf436f Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Tue, 29 Sep 2026 18:16:56 +0200 Subject: [PATCH 07/33] chore: :arrow_up: Update ggml-org/whisper.cpp to `6e4ab854f67f743900934a703d5603419384c961` (#12349) :arrow_up: Update ggml-org/whisper.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/whisper/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/whisper/Makefile b/backend/go/whisper/Makefile index 966e9fb72..20fec260e 100644 --- a/backend/go/whisper/Makefile +++ b/backend/go/whisper/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # whisper.cpp version WHISPER_REPO?=https://github.com/ggml-org/whisper.cpp -WHISPER_CPP_VERSION?=d09f61a708f3487afa956ff578e60eae5e7a233c +WHISPER_CPP_VERSION?=6e4ab854f67f743900934a703d5603419384c961 SO_TARGET?=libgowhisper.so CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF From 01db8f33c15cbf312eb39c964ea9b665573ebb23 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Tue, 29 Sep 2026 18:17:09 +0200 Subject: [PATCH 08/33] chore: :arrow_up: Update CrispStrobe/CrispASR to `2cd383a926e3c37334e75eb5d8b8a85bd82ed22d` (#12348) :arrow_up: Update CrispStrobe/CrispASR Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/crispasr/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/crispasr/Makefile b/backend/go/crispasr/Makefile index bbdfef58c..35f915924 100644 --- a/backend/go/crispasr/Makefile +++ b/backend/go/crispasr/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # CrispASR version (release tag) CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR -CRISPASR_VERSION?=ec98831d0776ec8a16ccaf93955693eb7ecfbec3 +CRISPASR_VERSION?=2cd383a926e3c37334e75eb5d8b8a85bd82ed22d SO_TARGET?=libgocrispasr.so CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF From 7cadb5748a3312d31561f5cb9cb6d50c8e761371 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Tue, 29 Sep 2026 18:17:24 +0200 Subject: [PATCH 09/33] chore: :arrow_up: Update localai-org/ced.cpp to `b10237678d1c3b30c77d19f2e63f6c198c7f8d09` (#12346) :arrow_up: Update localai-org/ced.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/ced/Makefile | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/backend/go/ced/Makefile b/backend/go/ced/Makefile index 6274b8cf6..a2676f8f2 100644 --- a/backend/go/ced/Makefile +++ b/backend/go/ced/Makefile @@ -1,6 +1,6 @@ # ced sound-classification backend Makefile. # -# Upstream pin lives below as CED_VERSION?=db5aae02973a745722d6fbd2157cab1999106777 +# Upstream pin lives below as CED_VERSION?=b10237678d1c3b30c77d19f2e63f6c198c7f8d09 # and update it (matches the parakeet-cpp / whisper.cpp convention). # # Local dev shortcut: symlink an out-of-tree ced.cpp shared build + header and @@ -9,7 +9,7 @@ # ln -sf /path/to/ced.cpp/include/ced_capi.h . # go build -o ced-grpc . -CED_VERSION?=db5aae02973a745722d6fbd2157cab1999106777 +CED_VERSION?=b10237678d1c3b30c77d19f2e63f6c198c7f8d09 CED_REPO?=https://github.com/localai-org/ced.cpp GOCMD?=go From b93d111d9b7ac01c5ecde1020e2884baba788f0d Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Tue, 29 Sep 2026 23:56:46 +0200 Subject: [PATCH 10/33] chore: :arrow_up: Update NVIDIA/NeMo-Speech.cpp to `0f706e43cf1fbc031bad1423e05460d3acaeaa1c` (#12361) :arrow_up: Update NVIDIA/NeMo-Speech.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/nemo-speech-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/nemo-speech-cpp/Makefile b/backend/go/nemo-speech-cpp/Makefile index 13e2d26b2..81a713178 100644 --- a/backend/go/nemo-speech-cpp/Makefile +++ b/backend/go/nemo-speech-cpp/Makefile @@ -12,7 +12,7 @@ # runs 'make -C backend/go/$(BACKEND) build' and then copies package/), so it # has to produce the binary and the package, not just the shared libraries. -NEMO_SPEECH_VERSION?=97a15afa5caa9bce5baaa86c1184103877af4101 +NEMO_SPEECH_VERSION?=0f706e43cf1fbc031bad1423e05460d3acaeaa1c NEMO_SPEECH_REPO?=https://github.com/NVIDIA/NeMo-Speech.cpp GOCMD?=go From 2fa36e714789dc519ce64089f47c88ebffc18a51 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Tue, 29 Sep 2026 23:58:17 +0200 Subject: [PATCH 11/33] feat(parakeet-cpp): speaker diarization, sound detection and live scene events (#12335) * feat(parakeet-cpp): load diarization and CED models and companions Repin PARAKEET_VERSION to parakeet.cpp PR #75's head, which adds parakeet_capi_model_kind (ABI v8). Bind the new diarization, sound event and combined scene stream C symbols through the same purego.Dlsym probe pattern already used for the batched JSON entry point, so the backend still loads against an older libparakeet.so. Load now classifies the loaded GGUF by role (ASR, diarization or sound) via parakeet_capi_model_kind and can load up to two companion models from Options[] (asr_model:, diarization_model:, sound_model:, paths resolved against opts.ModelPath), verifying each companion's kind and freeing every context opened so far on any failure. Free releases the primary and every companion. AudioTranscription now names the loaded role when it is not ASR instead of a generic model not loaded error. The dynamic batcher starts only when an ASR context ends up loaded, primary or companion. This is groundwork only: the Diarize and SoundDetection RPCs and the live scene stream that actually use these new roles land in later commits. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(parakeet-cpp): reset role fields on a failed companion load loadRoles' freeLoaded only released the C contexts it had opened; it left ctxPtr/diarCtx/tagCtx and companions pointing at those now-freed contexts, so a later Free() on the same instance would double-free. Zero all four alongside the CppFree calls. Also route AudioTranscriptionStream and AudioTranscriptionLive through notASRError when ctxPtr is unset but a diarization or sound model is loaded, matching AudioTranscription: both used to return the generic model-not-loaded error instead of naming the loaded role. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * feat(parakeet-cpp): add speaker diarization Implement the Diarize RPC for the parakeet-cpp Go backend, wired to Nemotron-3-Diarization through libparakeet.so's diarization C-API. Plain diarization uses parakeet_capi_diarize_pcm; when include_text is set and an ASR companion is loaded, parakeet_capi_transcribe_and_ diarize_json fills each segment's text instead. Speaker labels are the decimal index, or "unknown" for -1 (no diarized speaker overlaps). min_duration_off merges same-speaker segments across a short gap before min_duration_on drops the segments still too short, then ids are renumbered. num_speakers/min_speakers/max_speakers/clustering_ threshold have no Sortformer equivalent and are logged at debug instead of rejected. Verified against the real Nemotron-3-Diarization + parakeet-tdt_ctc- 110m checkpoints on the two_speakers.wav fixture: correct A-B-A-B speaker segmentation and matching speaker-attributed transcripts. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * feat(parakeet-cpp): add sound event detection Wire the SoundDetection RPC to the CED tagger context (p.tagCtx) loaded by Task 1's role classification. It runs the whole clip through a one-shot parakeet_capi_sound_stream_* session (window 10s, hop 10s, top_k set to the tagger's class count so every drained window carries a full score list), averages each class's score across the drained windows, sorts descending, then applies the request's threshold and top_k (0 keeps every class). No tagCtx returns FailedPrecondition; a libparakeet.so missing the sound_stream symbols returns Unimplemented. Every C call runs under engineMu, and the stream is always freed, even when a feed or drain call fails partway through. Verified against a real ced-tiny-q8_0.gguf on the rooster.wav demo clip: "Chicken, rooster" tops the list at score 0.91. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(parakeet-cpp): cancel sound detection mid-feed, shrink the lock SoundDetection now checks ctx before each 10 s feed slice (mirroring driver.go's feedSlices) and returns Canceled if the caller gave up, so a long clip can be interrupted instead of feeding to completion regardless. The stream is still freed on every path, cancellation included. Also narrow engineMu to the C calls: the drained JSON document is now decoded after the lock is released, splitting soundStreamScores into a locked soundStreamDrain (opts, begin, feed, drain, free) and an unlocked json.Unmarshal. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * feat(parakeet-cpp): stream speaker and sound events during live transcription Add two additive proto fields, LiveSpeakerSegment and LiveSoundEvent, repeated on TranscriptLiveResponse. When a diarization or sound companion model is loaded, AudioTranscriptionLive now runs a no-ASR scene stream (parakeet_capi_scene_stream_begin) beside the ASR streaming session, feeding it the same PCM slices and forwarding any closed speaker or sound events alongside the matching ASR delta, or on their own when a slice has no ASR output. The scene stream is freed and reopened on a mid-stream Config reset, flushed with is_last before the closing FinalResult, and degrades gracefully (a warning, not an error) when begin or a later feed call fails, so live transcription keeps working ASR-only. Existing live behavior is unchanged when no companion is configured, and no scene C call is made in that case. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(parakeet-cpp): keep scene events off the ASR critical path in live Emit each slice's ASR result right after the ASR feed, before the scene feed for that slice runs, so a companion diarization/sound model never adds scene compute latency in front of the delta or that drives realtime turn detection. Closed speakers/sounds go out afterward as their own response, so a slice with both now produces two responses, ASR first. The live feed log line now reports ASR and scene wall time separately. Re-check the diarization/sound contexts a scene stream was begun with against the live contexts before every feed, under the same lock: Free() can race between an ASR feed and the matching scene feed and free the model the stream borrows. A mismatch now returns without touching the C side. Freeing the stream itself stays unconditional; the scene stream's destructor only releases its own buffers and never touches the borrowed contexts. Also recover a panicking stub inside the live test goroutine instead of crashing the test binary, and reset the live decode-lag tracker on a mid-stream config reset, matching what its own comment already promised. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * feat(realtime): surface live speaker and sound events Carry the backend's closed speaker segments and sound events (TranscriptLiveResponse fields 7/8) through LiveTranscriptionEvent as LiveSpeakerSegment/LiveSoundEvent (nanoseconds mapped to seconds), and forward them from the semantic_vad live path. Each speaker segment emits conversation.item.input_audio_transcription.segment with speaker, start, end and empty text under the turn's item id. Each sound event emits conversation.item.sound_detection with one tag (label, score = peak, index) and the event's new optional start/end seconds fields, omitted when unset so the existing unary/windowed sound-detection path is unaffected. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(realtime): keep start/end on a zero-second transcription segment ConversationItemInputAudioTranscriptionSegmentEvent.Start/End used omitempty, so a speaker segment starting at 0.0s dropped its "start" key. Nothing emitted this event before the live scene-event path, so drop omitempty: the segment always carries real times. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * chore(gallery): add parakeet-cpp diarization, CED and realtime scene models Add gallery entries for the new parakeet-cpp capabilities: standalone Nemotron-3-Diarization, the same paired with the Parakeet TDT+CTC 110M ASR model for speaker-attributed text, CED-Tiny and CED-Base sound classifiers, and a realtime scene bundle combining the streaming EOU ASR model with diarization and sound companions. SHA256 taken from the Hub API; licenses from each model card (openmdw-1.1 for Nemotron-3-Diarization, apache-2.0 for CED, cc-by-4.0 for the Parakeet ASR models). Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * docs: document parakeet-cpp diarization, sound detection and live scene events Cover the new parakeet-cpp capabilities across the feature pages: Nemotron-3-Diarization as a diarization backend (with and without speaker text, the ignored speaker-count hints, the Sortformer voice-like-sound quirk), CED as a sound classification backend, the asr_model/diarization_model/sound_model/diarization_latency companion options, and the realtime live speaker/sound events (event shapes, the speech-turn-only limitation, and using this or pipeline.sound_detection but not both). Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(gallery): correct the realtime-scene license and wording nits parakeet-cpp-realtime-scene mistakenly copied cc-by-4.0 from the existing realtime_eou_120m-v1 entry; the model card lists the NVIDIA open model license instead. Switch to the gallery's usual spelling for that license and keep the diarization/CED licenses called out in the description. Also: audio-diarization.md now says getting per-segment text needs both an asr_model companion and include_text=true on the request, and audio-to-text.md's option table reads "Use on" (a pairing the loader does not enforce) instead of "Allowed on". Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(parakeet-cpp): reject a companion role that duplicates the primary's loadRoles let a companion option (asr_model:/diarization_model:/ sound_model:) assign into a role field the primary already occupied, for example asr_model: on an already-ASR primary. The companion's context silently overwrote ctxPtr/diarCtx/tagCtx, and Free() only walks those three fields, so the original primary context was never freed again. Reject a companion whose role the primary already holds before its GGUF is even loaded, freeing everything loadRoles opened so far, the same way a wrong-kind companion is already rejected. Also warn, rather than silently fall through, when parakeet_capi_model_kind reports PARAKEET_MODEL_KIND_NONE for a successfully loaded primary; the primary is still treated as ASR, matching today's behavior. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(parakeet-cpp): cap live scene sound score retention sceneBegin started the live diarization/sound companion stream with the C API's default sound options, whose top_k keeps 5 scores per window forever until drained. The live scene path never drains sound scores (only the offline SoundDetection RPC does, with its own fresh stream), so this window queue on the C side grew for the whole session's lifetime. Set opts.Sound.TopK = 0 before starting the scene stream: this disables score retention while leaving sound event detection (onset/ offset), which the live path actually consumes, unaffected. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(parakeet-cpp): merge diarization segments per speaker, harden Diarize mergeCloseSegments only compared neighbors in the single start-sorted segment list, so two same-speaker segments never merged once another speaker's turn fell between them (A, B, A): the short B segment broke the adjacency the merge relied on. Group segments by speaker first, merge within each speaker's own start-ordered run, then re-sort the result by start so interleaved speakers come back out in timeline order. Also harden Diarize's entry points the same way streamFeedDoc/ sceneFeed already are: diarizeCall re-checks p.diarCtx (and, on the include_text path, p.ctxPtr) under engineMu right before the C call, so a Free() racing between Diarize's own checks and the lock can no longer reach the C side with a freed context. When the include_text call returns NULL, last_error is now read from both contexts and whichever came back non-empty is reported, since either side of the pairing can be the one that failed. A WAV decode failure is reported as InvalidArgument instead of an unwrapped/untyped error. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(parakeet-cpp): harden SoundDetection's engine checks soundStreamDrain ran every C call under engineMu but never re-checked p.tagCtx there, so a Free() racing between SoundDetection's own tagCtx==0 check and this lock could still reach the C side with a freed context. Re-check p.tagCtx under the lock and return ModelNotLoaded when it was cleared, mirroring diarizeCall's own re-check. A WAV decode failure is now reported as InvalidArgument instead of an unwrapped/untyped error. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * test(parakeet-cpp): cover a mid-session scene feed failure feedSlicesScene already degrades gracefully when a scene feed call fails mid-session: it frees the broken stream and carries the ASR-only session forward. Add a spec covering that path end to end: the scene stream is freed exactly once, later audio slices still produce ASR responses, and no speaker/sound events appear before or after the failure. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * docs: fix the parakeet-cpp companion role table and realtime scene docs audio-to-text.md's companion option table read "Use on" with a note that the loader did not enforce the pairing; it now rejects a companion whose role duplicates the primary's, so restore the "Allowed on" wording and describe the real enforcement. openai-realtime.md's live speaker/sound section claimed a mid-stream session.update resets the companion stream and that it flushes on session close; neither happens, since the realtime core opens one live stream (and so one scene stream) per speech turn and closes it at that turn's commit, with no mid-stream Config in between. Document that lifecycle instead, state precisely that start/end are seconds from the start of the turn's own audio, and note that the diarization model starts a fresh session every turn, so a speaker index is only meaningful within one turn. The example sound tag ("Rooster", index 17) did not match any real CED label; index 17 in ced-tiny-q8_0.gguf is "Baby laughter". Replaced with "Chicken, rooster" at its real index, 99. Assisted-by: Claude:claude-sonnet-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(parakeet-cpp): use CED's real index for Chicken, rooster The scene feed comment and the live test's canned document gave "Chicken, rooster" index 365. In CED's AudioSet label list it is 99, which is also what the realtime docs show. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(realtime): call the test event accessor The scene-event tests range over a method instead of its returned slice. Call the synchronized accessor so the OpenAI test package compiles. Assisted-by: Codex:gpt-6 Signed-off-by: Ettore Di Giacinto * chore(parakeet-cpp): pin parakeet.cpp master with sound events mudler/parakeet.cpp#75 (sound events, scene stream, model kinds) and #74 (the missing include that broke the image builds) are on master now. Pin 6dea76a instead of the #75 PR head, and update the header comment the bump bot reads. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * chore(parakeet-cpp): pin parakeet.cpp with ced.cpp on main parakeet.cpp #76 moved its ced.cpp submodule from the head of localai-org/ced.cpp#3 (a branch-only commit) to ced.cpp main, where #3 landed with an identical tree. Pin 623a968 so the image builds no longer depend on that branch. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * feat(transcription): carry speaker labels on words and streamed segments A diarizing backend could label transcript segments, but two paths dropped the label: TranscriptWord had no speaker field, so live transcription words and word-level timestamps could not carry one, and the stream=true transcript.text.done event left the speaker out of its segments. TranscriptWord gains an optional speaker (proto field 4, additive). It flows through the live event and result mapping, the JSON word output of the endpoint and the CLI, and transcript.text.done now includes a segment's speaker when there is one. Empty labels are omitted, so responses without diarization are unchanged. Assisted-by: Claude:claude-opus-5-5 [Claude Code] (cherry picked from commit 2f0049f979496c3428c08c9b1d5bf4315403dccf) Signed-off-by: Ettore Di Giacinto * feat(importers): detect the parakeet.cpp diarization GGUF The Nemotron-3-Diarization GGUFs are published in mudler/parakeet-cpp-gguf as nemotron-3-diarization-.gguf. The parakeet-cpp importer did not recognise that name, so a direct `local-ai models import` of the file fell through to another importer. A direct URL to the file now imports with the diarization usecase. A repo import still picks ASR weights when the repo also ships the diarization model, and falls back to the diarization weights only when there are no others. Ported from #12323. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(config): advertise diarization and sound detection for parakeet-cpp The capability table listed parakeet-cpp as transcription only, though the backend now answers Diarize (Nemotron-3-Diarization) and SoundDetection (CED) depending on the model kind it loads. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * feat(parakeet-cpp): label transcript segments with the diarization companion A diarization_model companion only fed live speaker events and Diarize; /v1/audio/transcriptions ignored it. With the companion attached and diarize=true (the OpenAI endpoint's default), unary transcription now labels each segment with its speaker and splits segments at speaker turns; with word timestamps each word carries its speaker. The stream=true final result labels each utterance with the speaker who said most of it. Both use the checkpoint's own diarization over the whole clip, as NeMo's diarize() does. Words take the speaker whose segments overlap them most, or the nearest segment within 0.5 s, the same rule as parakeet.cpp's speaker-attributed ASR. Docs: the diarization_model row and a paragraph on transcript speakers; Nemotron-3-Diarization handles up to 8 speakers. Ported from #12323. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * feat(realtime): speaker segments from committed-turn transcription Speaker events reached a realtime session only from the live semantic_vad path, which needs a cache-aware streaming transcription model. Committed-turn transcription (server_vad, or any offline model) always asked the backend for diarize=false and dropped the segments' speakers. pipeline.diarization (off by default) asks the transcription model for speaker labels on each committed turn and emits every labelled segment as a conversation.item.input_audio_transcription.segment event, with its text, before the turn's completed event. It is opt-in because some backends fail a diarization request they cannot serve. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * feat(gallery): add parakeet-cpp-realtime-scene-tdt parakeet-cpp-realtime-scene pairs the streaming EOU model with the diarization and CED companions; its speaker and sound events need a cache-aware streaming model. This entry does the same with Parakeet TDT 0.6B v3 (multilingual, offline) for realtime under server_vad: set it as both transcription and sound_detection and turn on pipeline.diarization, and each committed turn gets speaker segments and sound tags from one parakeet-cpp backend. Files and sha256 match the Hub and are shared with the existing TDT v3, diarization and CED-Tiny entries. A real-model spec checks the combination on a clip with two speakers and a rooster: A-B-A-B speaker turns, and "Chicken, rooster" among the sound tags. The test loader now binds the sound entry points like main.go. Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * feat(gallery): add CED-Base variants of the parakeet-cpp scene models parakeet-cpp-realtime-scene and parakeet-cpp-realtime-scene-tdt ship with CED-Tiny. The -base variants use CED-Base (86M), which tags sounds more confidently (on the rooster clip "Crowing" 0.65 against 0.49 for Tiny). Measured on CPU over a 37 s clip: the live diarization + sound stream runs at 0.125 of real time with CED-Base against 0.103 with CED-Tiny, because diarization dominates; sound detection per committed turn costs 0.031 against 0.005. The realtime docs list both and note that any CED size works as sound_model. Files and sha256 match the Hub and are shared with the existing parakeet-cpp-ced-base entry. The TDT variant passes the real-model scene spec with CED-Base (A-B-A-B speakers, "Chicken, rooster" found). Assisted-by: Claude:claude-opus-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto * fix(config): register pipeline.diarization in the config metadata TestAllFieldsHaveRegistryEntries fails on the branch because the new pipeline.diarization field has no registry entry. Add one so the model editor shows it as a toggle next to the sound detection options. Assisted-by: Claude:claude-sonnet-5-5 [Claude Code] Signed-off-by: Ettore Di Giacinto --------- Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- backend/backend.proto | 17 + backend/go/parakeet-cpp/Makefile | 4 +- backend/go/parakeet-cpp/diarize.go | 328 ++++++++++++++ backend/go/parakeet-cpp/diarize_test.go | 275 ++++++++++++ backend/go/parakeet-cpp/goparakeetcpp.go | 182 ++++++-- backend/go/parakeet-cpp/goparakeetcpp_test.go | 35 ++ backend/go/parakeet-cpp/live.go | 86 +++- backend/go/parakeet-cpp/live_test.go | 306 ++++++++++++- backend/go/parakeet-cpp/main.go | 28 ++ backend/go/parakeet-cpp/roles.go | 245 ++++++++++ backend/go/parakeet-cpp/roles_test.go | 417 ++++++++++++++++++ backend/go/parakeet-cpp/scene.go | 258 +++++++++++ backend/go/parakeet-cpp/scene_test.go | 42 ++ backend/go/parakeet-cpp/sound.go | 261 +++++++++++ backend/go/parakeet-cpp/sound_test.go | 383 ++++++++++++++++ backend/go/parakeet-cpp/speakers.go | 121 +++++ backend/go/parakeet-cpp/speakers_test.go | 166 +++++++ core/backend/transcript.go | 7 +- core/backend/transcript_live.go | 55 ++- core/backend/transcript_live_internal_test.go | 42 ++ core/cli/transcript.go | 14 +- core/config/backend_capabilities.go | 10 +- core/config/meta/registry.go | 7 + core/config/model_config.go | 8 + core/gallery/importers/parakeet-cpp.go | 31 +- core/gallery/importers/parakeet-cpp_test.go | 42 ++ .../endpoints/openai/realtime_doubles_test.go | 7 +- .../endpoints/openai/realtime_semantic_vad.go | 29 ++ .../openai/realtime_semantic_vad_test.go | 61 +++ .../openai/realtime_sound_detection_test.go | 64 +++ .../openai/realtime_transcription.go | 42 +- .../openai/realtime_transcription_test.go | 83 ++++ core/http/endpoints/openai/transcription.go | 22 +- .../endpoints/openai/types/server_events.go | 19 +- core/schema/transcription.go | 14 +- docs/content/features/audio-classification.md | 21 + docs/content/features/audio-diarization.md | 33 +- docs/content/features/audio-to-text.md | 17 +- docs/content/features/openai-realtime.md | 107 +++++ gallery/index.yaml | 383 ++++++++++++++++ 40 files changed, 4178 insertions(+), 94 deletions(-) create mode 100644 backend/go/parakeet-cpp/diarize.go create mode 100644 backend/go/parakeet-cpp/diarize_test.go create mode 100644 backend/go/parakeet-cpp/roles.go create mode 100644 backend/go/parakeet-cpp/roles_test.go create mode 100644 backend/go/parakeet-cpp/scene.go create mode 100644 backend/go/parakeet-cpp/scene_test.go create mode 100644 backend/go/parakeet-cpp/sound.go create mode 100644 backend/go/parakeet-cpp/sound_test.go create mode 100644 backend/go/parakeet-cpp/speakers.go create mode 100644 backend/go/parakeet-cpp/speakers_test.go diff --git a/backend/backend.proto b/backend/backend.proto index 54255528e..6a09b98eb 100644 --- a/backend/backend.proto +++ b/backend/backend.proto @@ -641,12 +641,29 @@ message TranscriptLiveResponse { repeated TranscriptWord words = 4; // words finalized by this feed (stream-relative ns) TranscriptResult final_result = 5; // terminal message only, after the send side closes bool eob = 6; // fired: a backchannel ("uh-huh") ended — NOT a turn boundary + repeated LiveSpeakerSegment speakers = 7; // closed speaker segments from a companion diarization/scene stream + repeated LiveSoundEvent sounds = 8; // closed sound events from a companion sound/scene stream +} + +message LiveSpeakerSegment { + string speaker = 1; // decimal speaker index + int64 start = 2; // stream-relative nanoseconds + int64 end = 3; +} + +message LiveSoundEvent { + string label = 1; + int32 index = 2; + float peak = 3; + int64 start = 4; // stream-relative nanoseconds + int64 end = 5; } message TranscriptWord { int64 start = 1; int64 end = 2; string text = 3; + string speaker = 4; // backend speaker label when diarizing; empty otherwise } message TranscriptSegment { diff --git a/backend/go/parakeet-cpp/Makefile b/backend/go/parakeet-cpp/Makefile index e288f6fcc..59fe56717 100644 --- a/backend/go/parakeet-cpp/Makefile +++ b/backend/go/parakeet-cpp/Makefile @@ -1,6 +1,6 @@ # parakeet-cpp backend Makefile. # -# Upstream pin lives below as PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6 +# Upstream pin lives below as PARAKEET_VERSION?=623a968bccbd2214588df398fcce687cd4218dea # (.github/bump_deps.sh) can find and update it - matches the # whisper.cpp / ds4 / vibevoice-cpp convention. # @@ -15,7 +15,7 @@ # That's what the L0 smoke test uses. The default target below does the # proper clone-at-pin + cmake build so CI doesn't need a side-checkout. -PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6 +PARAKEET_VERSION?=623a968bccbd2214588df398fcce687cd4218dea PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp GOCMD?=go diff --git a/backend/go/parakeet-cpp/diarize.go b/backend/go/parakeet-cpp/diarize.go new file mode 100644 index 000000000..1853aa666 --- /dev/null +++ b/backend/go/parakeet-cpp/diarize.go @@ -0,0 +1,328 @@ +package main + +import ( + "encoding/json" + "fmt" + "sort" + "strconv" + "strings" + + "github.com/mudler/LocalAI/pkg/grpc/grpcerrors" + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + "github.com/mudler/xlog" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" +) + +// diarizeSegmentJSON mirrors one element of parakeet_capi_diarize_pcm's +// "segments" array: {"speaker":0,"start":0.50,"end":5.52}. +type diarizeSegmentJSON struct { + Speaker int `json:"speaker"` + Start float64 `json:"start"` + End float64 `json:"end"` +} + +// diarizePCMDoc mirrors the document parakeet_capi_diarize_pcm returns. +// "speakers" is the model's CAPACITY (e.g. 8 for Nemotron-3-Diarization), +// not the count of speakers actually present, so it is not read here; the +// response's num_speakers is computed from distinct segment labels instead. +type diarizePCMDoc struct { + Segments []diarizeSegmentJSON `json:"segments"` +} + +// diarizeUtteranceJSON mirrors one element of +// parakeet_capi_transcribe_and_diarize_json's "utterances" array. Speaker is +// -1 when no diarized speaker overlaps the utterance. +type diarizeUtteranceJSON struct { + Speaker int `json:"speaker"` + Text string `json:"text"` + Start float64 `json:"start"` + End float64 `json:"end"` +} + +// transcribeAndDiarizeDoc mirrors the document +// parakeet_capi_transcribe_and_diarize_json returns. Only "utterances" is +// consumed here; the per-word "words" detail belongs to a speaker-attributed +// transcript RPC, not Diarize. +type transcribeAndDiarizeDoc struct { + Utterances []diarizeUtteranceJSON `json:"utterances"` +} + +// speakerLabel renders a 0-based speaker index as the decimal string +// DiarizeSegment.speaker documents, or "unknown" for -1 (no diarized speaker +// overlaps this utterance; only transcribe_and_diarize_json can report this). +func speakerLabel(speaker int) string { + if speaker < 0 { + return "unknown" + } + return strconv.Itoa(speaker) +} + +// unsupportedDiarizeFields names the DiarizeRequest fields Sortformer has no +// equivalent for: it is an end-to-end model with a fixed speaker capacity and +// no clustering stage, so there is no config knob to target a speaker count +// or a clustering distance. Logged rather than rejected, so a request naming +// one of these still gets the diarization it can have. +func unsupportedDiarizeFields(req *pb.DiarizeRequest) []string { + var out []string + if req.GetNumSpeakers() != 0 { + out = append(out, "num_speakers") + } + if req.GetMinSpeakers() != 0 { + out = append(out, "min_speakers") + } + if req.GetMaxSpeakers() != 0 { + out = append(out, "max_speakers") + } + if req.GetClusteringThreshold() != 0 { + out = append(out, "clustering_threshold") + } + return out +} + +// Diarize labels who spoke when in the audio at req.Dst, using the loaded +// diarization model (p.diarCtx). When req.IncludeText is set and an ASR +// companion (p.ctxPtr) is loaded, each segment also carries its transcript +// (parakeet_capi_transcribe_and_diarize_json, one utterance per speaker +// turn); otherwise, or when no ASR companion is loaded, segments carry no +// text (parakeet_capi_diarize_pcm) and no error is raised. +func (p *ParakeetCpp) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, error) { + if p.diarCtx == 0 { + return pb.DiarizeResponse{}, status.Error(codes.FailedPrecondition, + "parakeet-cpp: model is not a diarization model") + } + if CppDiarizePCM == nil { + return pb.DiarizeResponse{}, status.Error(codes.Unimplemented, + "parakeet-cpp: loaded libparakeet.so has no diarization support (parakeet_capi_diarize_pcm missing)") + } + if req.GetDst() == "" { + return pb.DiarizeResponse{}, status.Error(codes.InvalidArgument, + "parakeet-cpp: DiarizeRequest.dst (audio path) is required") + } + + if dropped := unsupportedDiarizeFields(req); len(dropped) > 0 { + xlog.Debug("parakeet-cpp: ignoring diarization request fields Sortformer has no equivalent for", + "fields", dropped) + } + + pcm, duration, err := decodeWavMono16k(req.GetDst()) + if err != nil { + return pb.DiarizeResponse{}, status.Errorf(codes.InvalidArgument, "parakeet-cpp: decode audio: %s", err) + } + if len(pcm) == 0 { + return pb.DiarizeResponse{}, status.Error(codes.InvalidArgument, "parakeet-cpp: empty audio") + } + + wantText := req.GetIncludeText() && p.ctxPtr != 0 && CppTranscribeAndDiarizeJSON != nil + + raw, err := p.diarizeCall(pcm, wantText) + if err != nil { + return pb.DiarizeResponse{}, err + } + segments, err := parseDiarizeDoc(raw, wantText) + if err != nil { + return pb.DiarizeResponse{}, err + } + + segments = applyDurationFilters(segments, req.GetMinDurationOn(), req.GetMinDurationOff()) + renumberDiarizeSegments(segments) + + return pb.DiarizeResponse{ + Segments: segments, + NumSpeakers: distinctDiarizeSpeakers(segments), + Duration: duration, + }, nil +} + +// diarizeCall runs the single C call Diarize needs (transcribe_and_diarize_json +// when wantText, else diarize_pcm) under engineMu, and returns the raw JSON +// document. p.diarCtx (and, on the include_text path, p.ctxPtr) is re-checked +// under the lock before the C call: Diarize's own p.diarCtx==0/wantText checks +// run before this lock is taken, so a Free() racing in between (which zeroes +// those fields under the same engineMu) would otherwise reach the C side with +// a freed context. last_error is ctx-shared, so it is read under the same +// lock as the failing call. +func (p *ParakeetCpp) diarizeCall(pcm []float32, wantText bool) (string, error) { + p.engineMu.Lock() + defer p.engineMu.Unlock() + + if p.diarCtx == 0 || (wantText && p.ctxPtr == 0) { + return "", grpcerrors.ModelNotLoaded("parakeet-cpp") + } + + var cstr uintptr + if wantText { + cstr = CppTranscribeAndDiarizeJSON(p.ctxPtr, p.diarCtx, &pcm[0], int32(len(pcm)), 16000) + } else { + cstr = CppDiarizePCM(p.diarCtx, &pcm[0], int32(len(pcm)), 16000) + } + if cstr == 0 { + return "", fmt.Errorf("parakeet-cpp: diarize failed: %s", diarizeLastError(p, wantText)) + } + raw := goStringFromCPtr(cstr) + CppFreeString(cstr) + return raw, nil +} + +// diarizeLastError reads last_error off p.diarCtx and, on the include_text +// path, p.ctxPtr too — the failing call is CppTranscribeAndDiarizeJSON there, +// and either side of the pairing may be the one that set it — then joins +// whichever came back non-empty. Called under the same engineMu as the +// failing call (last_error is ctx-shared state). +func diarizeLastError(p *ParakeetCpp, wantText bool) string { + var msgs []string + if m := CppLastError(p.diarCtx); m != "" { + msgs = append(msgs, m) + } + if wantText { + if m := CppLastError(p.ctxPtr); m != "" { + msgs = append(msgs, m) + } + } + if len(msgs) == 0 { + return "unknown error" + } + return strings.Join(msgs, "; ") +} + +// parseDiarizeDoc decodes the raw JSON diarizeCall returned into +// DiarizeSegments (without ids: renumberDiarizeSegments assigns those after +// filtering). +func parseDiarizeDoc(raw string, wantText bool) ([]*pb.DiarizeSegment, error) { + if wantText { + var doc transcribeAndDiarizeDoc + if err := json.Unmarshal([]byte(raw), &doc); err != nil { + return nil, fmt.Errorf("parakeet-cpp: decode diarize json: %w", err) + } + segs := make([]*pb.DiarizeSegment, 0, len(doc.Utterances)) + for _, u := range doc.Utterances { + segs = append(segs, &pb.DiarizeSegment{ + Start: float32(u.Start), + End: float32(u.End), + Speaker: speakerLabel(u.Speaker), + Text: u.Text, + }) + } + return segs, nil + } + + var doc diarizePCMDoc + if err := json.Unmarshal([]byte(raw), &doc); err != nil { + return nil, fmt.Errorf("parakeet-cpp: decode diarize json: %w", err) + } + segs := make([]*pb.DiarizeSegment, 0, len(doc.Segments)) + for _, s := range doc.Segments { + segs = append(segs, &pb.DiarizeSegment{ + Start: float32(s.Start), + End: float32(s.End), + Speaker: speakerLabel(s.Speaker), + }) + } + return segs, nil +} + +// applyDurationFilters applies the request's postprocessing knobs, in the +// order NeMo's diarization postprocessing does: merge first +// (min_duration_off), then drop short segments (min_duration_on) — dropping +// first would leave short gaps unmerged that the drop step just created. +// Segments are assumed sorted by start time, as parakeet_capi_diarize_pcm and +// parakeet_capi_transcribe_and_diarize_json document. A non-positive value +// disables that filter (the proto's "0 = backend default" reads here as "no +// filtering"). +func applyDurationFilters(segs []*pb.DiarizeSegment, minOn, minOff float32) []*pb.DiarizeSegment { + segs = mergeCloseSegments(segs, minOff) + segs = dropShortSegments(segs, minOn) + return segs +} + +// mergeCloseSegments merges SAME-SPEAKER segments separated by a gap shorter +// than minOff into one segment spanning both (and concatenating any text). +// Segments from different speakers are never merged, regardless of gap: the +// gap only ever means "the same speaker paused", never "two speakers are +// actually one". +// +// Merging runs per speaker rather than on the single start-sorted list: two +// segments of the same speaker are not necessarily adjacent in that list once +// another speaker's turn falls between them (A, B, A), and a start-sorted +// walk would then never compare the two A's at all. Grouping by speaker first +// keeps each group's own start order (segs is assumed start-sorted, as +// parakeet_capi_diarize_pcm and parakeet_capi_transcribe_and_diarize_json +// document), merges within the group, then the merged segments are re-sorted +// by start so interleaved speakers come back out in timeline order. +func mergeCloseSegments(segs []*pb.DiarizeSegment, minOff float32) []*pb.DiarizeSegment { + if minOff <= 0 || len(segs) < 2 { + return segs + } + + bySpeaker := make(map[string][]*pb.DiarizeSegment) + var order []string // first-seen speaker order, for a deterministic group walk + for _, s := range segs { + if _, ok := bySpeaker[s.GetSpeaker()]; !ok { + order = append(order, s.GetSpeaker()) + } + bySpeaker[s.GetSpeaker()] = append(bySpeaker[s.GetSpeaker()], s) + } + + out := make([]*pb.DiarizeSegment, 0, len(segs)) + for _, speaker := range order { + group := bySpeaker[speaker] + merged := make([]*pb.DiarizeSegment, 0, len(group)) + merged = append(merged, group[0]) + for _, s := range group[1:] { + prev := merged[len(merged)-1] + if s.GetStart()-prev.GetEnd() < minOff { + if s.GetEnd() > prev.GetEnd() { + prev.End = s.End + } + if s.GetText() != "" { + if prev.GetText() != "" { + prev.Text = prev.GetText() + " " + s.GetText() + } else { + prev.Text = s.GetText() + } + } + continue + } + merged = append(merged, s) + } + out = append(out, merged...) + } + + sort.Slice(out, func(i, j int) bool { return out[i].GetStart() < out[j].GetStart() }) + return out +} + +// dropShortSegments discards segments shorter than minOn. +func dropShortSegments(segs []*pb.DiarizeSegment, minOn float32) []*pb.DiarizeSegment { + if minOn <= 0 { + return segs + } + out := make([]*pb.DiarizeSegment, 0, len(segs)) + for _, s := range segs { + if s.GetEnd()-s.GetStart() < minOn { + continue + } + out = append(out, s) + } + return out +} + +// renumberDiarizeSegments assigns sequential ids (0..) to the final segment +// list, after filtering may have dropped or merged entries. +func renumberDiarizeSegments(segs []*pb.DiarizeSegment) { + for i, s := range segs { + s.Id = int32(i) + } +} + +// distinctDiarizeSpeakers counts the distinct speaker labels present in segs. +// This is what DiarizeResponse.num_speakers documents — the count of speakers +// actually present in the result — and is NOT the diarize_pcm JSON's +// top-level "speakers" field, which reports the model's fixed capacity. +func distinctDiarizeSpeakers(segs []*pb.DiarizeSegment) int32 { + seen := make(map[string]struct{}, len(segs)) + for _, s := range segs { + seen[s.GetSpeaker()] = struct{}{} + } + return int32(len(seen)) +} diff --git a/backend/go/parakeet-cpp/diarize_test.go b/backend/go/parakeet-cpp/diarize_test.go new file mode 100644 index 000000000..1db076771 --- /dev/null +++ b/backend/go/parakeet-cpp/diarize_test.go @@ -0,0 +1,275 @@ +package main + +import ( + "path/filepath" + "sync" + "unsafe" + + "github.com/mudler/LocalAI/pkg/grpc/grpcerrors" + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" +) + +// The Diarize specs drive it entirely against stubbed CppDiarizePCM / +// CppTranscribeAndDiarizeJSON / CppFreeString / CppLastError (the same seam +// live_test.go and roles_test.go use), so they run without libparakeet.so. + +// diarizeCstrPool hands out NUL-terminated C-style strings backed by Go +// memory and keeps them alive for the duration of a spec (goStringFromCPtr +// reads through the raw pointer; mirrors live_test.go's liveCstrPool). +type diarizeCstrPool struct { + mu sync.Mutex + bufs [][]byte +} + +func (p *diarizeCstrPool) cstr(s string) uintptr { + p.mu.Lock() + defer p.mu.Unlock() + b := append([]byte(s), 0) + p.bufs = append(p.bufs, b) + return uintptr(unsafe.Pointer(&b[0])) +} + +// diarizeStubs swaps every C entry point Diarize touches and returns a +// restore func for AfterEach (mirrors live_test.go's liveStubs). +func diarizeStubs() (restore func()) { + savedDiarize := CppDiarizePCM + savedTranscribeAndDiarize := CppTranscribeAndDiarizeJSON + savedFreeString := CppFreeString + savedLastError := CppLastError + return func() { + CppDiarizePCM = savedDiarize + CppTranscribeAndDiarizeJSON = savedTranscribeAndDiarize + CppFreeString = savedFreeString + CppLastError = savedLastError + } +} + +// diarizeWav writes a silent 16 kHz mono WAV of the given duration (seconds) +// to a fresh temp file and returns its path. decodeWavMono16k reads real +// audio bytes off disk, so Diarize needs a file on disk even though the +// stubbed C calls never look at its samples. +func diarizeWav(seconds float64) string { + GinkgoHelper() + path := filepath.Join(GinkgoT().TempDir(), "diarize.wav") + writeMono16kWav(path, int(seconds*16000)) + return path +} + +var _ = Describe("ParakeetCpp.Diarize", func() { + var restore func() + var pool *diarizeCstrPool + + BeforeEach(func() { + restore = diarizeStubs() + pool = &diarizeCstrPool{} + }) + AfterEach(func() { restore() }) + + It("fails with FailedPrecondition when no diarization model is loaded", func() { + p := &ParakeetCpp{} + _, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1)}) + Expect(err).To(HaveOccurred()) + Expect(status.Code(err)).To(Equal(codes.FailedPrecondition)) + Expect(err.Error()).To(ContainSubstring("model is not a diarization model")) + }) + + It("fails with Unimplemented when the loaded libparakeet.so has no diarize_pcm symbol", func() { + CppDiarizePCM = nil + p := &ParakeetCpp{diarCtx: 42} + _, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1)}) + Expect(err).To(HaveOccurred()) + Expect(status.Code(err)).To(Equal(codes.Unimplemented)) + }) + + It("maps plain segments with sequential ids, decimal speaker labels and distinct speaker count", func() { + CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr { + return pool.cstr(`{"speakers":8,"segments":[` + + `{"speaker":0,"start":0.00,"end":3.00},` + + `{"speaker":1,"start":3.00,"end":6.00}]}`) + } + CppFreeString = func(uintptr) {} + + p := &ParakeetCpp{diarCtx: 42} + resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(6)}) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.Segments).To(HaveLen(2)) + Expect(resp.Segments[0].Id).To(Equal(int32(0))) + Expect(resp.Segments[0].Speaker).To(Equal("0")) + Expect(resp.Segments[1].Id).To(Equal(int32(1))) + Expect(resp.Segments[1].Speaker).To(Equal("1")) + Expect(resp.Segments[0].Text).To(BeEmpty()) + Expect(resp.NumSpeakers).To(Equal(int32(2))) + Expect(resp.Duration).To(BeNumerically("~", 6.0, 0.01)) + }) + + It("fills text from utterances when include_text is set with an ASR companion, and maps speaker -1 to unknown", func() { + CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr { + Fail("diarize_pcm must not be called when include_text has an ASR companion to pair with") + return 0 + } + CppTranscribeAndDiarizeJSON = func(asr, diar uintptr, samples *float32, n int32, sampleRate int32) uintptr { + return pool.cstr(`{"speakers":8,"utterances":[` + + `{"speaker":0,"text":"hello there","start":0.00,"end":1.00,"conf":0.9},` + + `{"speaker":-1,"text":"mumble","start":1.00,"end":1.50,"conf":0.4}],"words":[]}`) + } + CppFreeString = func(uintptr) {} + + p := &ParakeetCpp{diarCtx: 42, ctxPtr: 7} + resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(2), IncludeText: true}) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.Segments).To(HaveLen(2)) + Expect(resp.Segments[0].Speaker).To(Equal("0")) + Expect(resp.Segments[0].Text).To(Equal("hello there")) + Expect(resp.Segments[1].Speaker).To(Equal("unknown")) + Expect(resp.Segments[1].Text).To(Equal("mumble")) + }) + + It("falls back to plain segments without error when include_text is set but no ASR companion is loaded", func() { + diarizeCalled := false + CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr { + diarizeCalled = true + return pool.cstr(`{"speakers":8,"segments":[{"speaker":0,"start":0.00,"end":1.00}]}`) + } + CppFreeString = func(uintptr) {} + CppTranscribeAndDiarizeJSON = func(asr, diar uintptr, samples *float32, n int32, sampleRate int32) uintptr { + Fail("transcribe_and_diarize_json must not be called without an ASR companion") + return 0 + } + + p := &ParakeetCpp{diarCtx: 42} // no ctxPtr companion + resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1), IncludeText: true}) + Expect(err).ToNot(HaveOccurred()) + Expect(diarizeCalled).To(BeTrue()) + Expect(resp.Segments).To(HaveLen(1)) + Expect(resp.Segments[0].Text).To(BeEmpty()) + }) + + It("drops a segment shorter than min_duration_on", func() { + CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr { + return pool.cstr(`{"speakers":8,"segments":[` + + `{"speaker":0,"start":0.30,"end":0.50},` + // 0.2s, at 0.3 + `{"speaker":0,"start":1.00,"end":2.00}]}`) // 1.0s, kept + } + CppFreeString = func(uintptr) {} + + p := &ParakeetCpp{diarCtx: 42} + resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(3), MinDurationOn: 0.3}) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.Segments).To(HaveLen(1)) + Expect(resp.Segments[0].Id).To(Equal(int32(0))) + Expect(resp.Segments[0].Start).To(BeNumerically("~", 1.0, 0.001)) + }) + + It("merges same-speaker segments across a short gap but not across a speaker change", func() { + CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr { + return pool.cstr(`{"speakers":8,"segments":[` + + `{"speaker":0,"start":0.00,"end":1.00},` + + `{"speaker":0,"start":1.30,"end":2.00},` + // 0.3s gap, same speaker: merges + `{"speaker":1,"start":2.10,"end":3.00}]}`) // 0.1s gap, different speaker: stays separate + } + CppFreeString = func(uintptr) {} + + p := &ParakeetCpp{diarCtx: 42} + resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(3), MinDurationOff: 0.5}) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.Segments).To(HaveLen(2)) + Expect(resp.Segments[0].Id).To(Equal(int32(0))) + Expect(resp.Segments[0].Speaker).To(Equal("0")) + Expect(resp.Segments[0].Start).To(BeNumerically("~", 0.0, 0.001)) + Expect(resp.Segments[0].End).To(BeNumerically("~", 2.0, 0.001)) + Expect(resp.Segments[1].Id).To(Equal(int32(1))) + Expect(resp.Segments[1].Speaker).To(Equal("1")) + }) + + It("surfaces last_error when the C call returns NULL", func() { + CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr { + return 0 + } + CppLastError = func(ctx uintptr) string { return "boom" } + + p := &ParakeetCpp{diarCtx: 42} + _, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1)}) + Expect(err).To(HaveOccurred()) + Expect(err.Error()).To(ContainSubstring("boom")) + }) + + It("reports last_error from both contexts when the include_text C call returns NULL", func() { + // Diarize's Unimplemented gate checks CppDiarizePCM regardless of + // wantText, so it needs a non-nil (never called) stub here too. + CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr { + Fail("diarize_pcm must not be called when include_text has an ASR companion to pair with") + return 0 + } + CppTranscribeAndDiarizeJSON = func(asr, diar uintptr, samples *float32, n int32, sampleRate int32) uintptr { + return 0 + } + CppLastError = func(ctx uintptr) string { + if ctx == 7 { + return "asr side broke" + } + return "diar side broke" + } + + p := &ParakeetCpp{diarCtx: 42, ctxPtr: 7} + _, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1), IncludeText: true}) + Expect(err).To(HaveOccurred()) + Expect(err.Error()).To(ContainSubstring("asr side broke")) + Expect(err.Error()).To(ContainSubstring("diar side broke")) + }) + + It("wraps a decode failure as InvalidArgument", func() { + CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr { + Fail("decode must fail before any C call is made") + return 0 + } + + p := &ParakeetCpp{diarCtx: 42} + _, err := p.Diarize(&pb.DiarizeRequest{Dst: filepath.Join(GinkgoT().TempDir(), "missing.wav")}) + Expect(err).To(HaveOccurred()) + Expect(status.Code(err)).To(Equal(codes.InvalidArgument)) + }) + + It("returns ModelNotLoaded without a C call when diarCtx is zeroed between the entry check and the call", func() { + called := false + CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr { + called = true + return pool.cstr(`{"speakers":8,"segments":[]}`) + } + CppFreeString = func(uintptr) {} + + p := &ParakeetCpp{diarCtx: 42} + // Simulate a Free() racing between Diarize's own diarCtx==0 check and + // diarizeCall's lock, exactly as it zeroes diarCtx under engineMu. + p.diarCtx = 0 + _, err := p.diarizeCall(make([]float32, 10), false) + Expect(grpcerrors.IsModelNotLoaded(err)).To(BeTrue()) + Expect(called).To(BeFalse(), "no C call once diarCtx was cleared") + }) + + It("merges same-speaker segments across an intervening different speaker (A, B, A)", func() { + CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr { + return pool.cstr(`{"speakers":8,"segments":[` + + `{"speaker":0,"start":0.00,"end":1.00},` + + `{"speaker":1,"start":1.05,"end":1.20},` + // short B segment sits between the two A's + `{"speaker":0,"start":1.30,"end":2.00}]}`) // 0.1s gap from the first A: same speaker, merges + } + CppFreeString = func(uintptr) {} + + p := &ParakeetCpp{diarCtx: 42} + resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(3), MinDurationOff: 0.5}) + Expect(err).ToNot(HaveOccurred()) + // The two speaker-0 segments merge into one spanning 0.00-2.00, and + // the timeline re-sort puts speaker 1's untouched segment in between. + Expect(resp.Segments).To(HaveLen(2)) + Expect(resp.Segments[0].Speaker).To(Equal("0")) + Expect(resp.Segments[0].Start).To(BeNumerically("~", 0.0, 0.001)) + Expect(resp.Segments[0].End).To(BeNumerically("~", 2.0, 0.001)) + Expect(resp.Segments[1].Speaker).To(Equal("1")) + Expect(resp.Segments[1].Start).To(BeNumerically("~", 1.05, 0.001)) + Expect(resp.Segments[1].End).To(BeNumerically("~", 1.20, 0.001)) + }) +}) diff --git a/backend/go/parakeet-cpp/goparakeetcpp.go b/backend/go/parakeet-cpp/goparakeetcpp.go index 23e5c548c..e8ed1c525 100644 --- a/backend/go/parakeet-cpp/goparakeetcpp.go +++ b/backend/go/parakeet-cpp/goparakeetcpp.go @@ -74,8 +74,55 @@ var ( // libparakeet.so; nil falls back to the text-only CppStreamFeed/Finalize path. CppStreamFeedJSON func(s uintptr, pcm []float32, nSamples int32) uintptr CppStreamFinalizeJSON func(s uintptr) uintptr + + // CppModelKind reports which kind of model a loaded context holds + // (parakeet_capi_model_kind, ABI v8): see the modelKind* constants in + // roles.go. nil on an older libparakeet.so; Load then treats the primary + // as ASR (pre-v8 behavior) and rejects companion model options. + CppModelKind func(ctx uintptr) int32 + + // Speaker diarization (ABI v7). CppDiarizePCM runs offline diarization + // over in-memory mono float PCM; CppTranscribeAndDiarizeJSON pairs it with + // an ASR context for speaker-attributed text. Both return a malloc'd char* + // JSON document (uintptr, freed via CppFreeString). + CppDiarizePCM func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr + CppTranscribeAndDiarizeJSON func(asr, diar uintptr, samples *float32, n int32, sampleRate int32) uintptr + + // Sound-event detection (CED) and the combined scene stream (ABI v8). + // CppNumClasses/CppSoundOptsDefault/CppSoundStreamBegin.../ + // CppSceneOptsDefault/CppSceneStreamBegin... are only registered when + // CppModelKind is present (see main.go); nil otherwise. + CppNumClasses func(ctx uintptr) int32 + CppSoundOptsDefault func(o *cSoundOpts) + CppSoundStreamBegin func(tagger uintptr, o *cSoundOpts) uintptr + CppSoundStreamFeed func(s uintptr, pcm *float32, n int32, isLast int32, out *uintptr, nOut *int32) int32 + CppSoundStreamDrainScoresJSON func(s uintptr) uintptr + CppFreeSoundSegments func(segs uintptr) + CppSoundStreamFree func(s uintptr) + CppSceneOptsDefault func(o *cSceneOpts) + CppSceneStreamBegin func(asr, diar, tagger uintptr, o *cSceneOpts) uintptr + CppSceneStreamFeedJSON func(s uintptr, pcm *float32, n int32, isLast int32) uintptr + CppSceneStreamLastError func(s uintptr) string + CppSceneStreamFree func(s uintptr) ) +// cSoundOpts and cSceneOpts mirror parakeet_sound_opts / parakeet_scene_opts +// in parakeet_capi.h field-for-field (int -> int32, float -> float32); the +// C side sizes/versions them via the leading `size` field, set by the +// matching *_opts_default call. +type cSoundOpts struct { + Size int32 + WindowSec, HopSec, OnThreshold, OffThreshold, MinDurationSec float32 + TopK int32 +} + +type cSceneOpts struct { + Size int32 + DiarLatency int32 + Sound cSoundOpts + Flags int32 +} + // streamChunkSamples is how much 16 kHz mono PCM we hand to stream_feed per // call (1 s). The session buffers internally and decodes once a full // cache-aware encoder chunk is available, so this only bounds how often we @@ -140,10 +187,23 @@ type transcriptToken struct { // touch it concurrently. type ParakeetCpp struct { base.Base - ctxPtr uintptr - engineMu sync.Mutex // sole guard of the one C engine (dispatcher + streaming) - bat *batcher - batStop chan struct{} + ctxPtr uintptr // ASR context: the primary when it is an ASR model, or the asr_model companion + // diarCtx / tagCtx are the diarization and sound (CED) model contexts: + // the primary when it is that kind, or the diarization_model/sound_model + // companion. See roles.go. + diarCtx uintptr + tagCtx uintptr + // diarLatency is the PARAKEET_DIAR_LATENCY_* mode for diarization + // streaming (diarization_latency: option, default "low"). Unused until + // the diarization/scene streaming paths land. + diarLatency int32 + // companions holds every context this backend loaded itself beyond the + // primary (asr_model:/diarization_model:/sound_model: options), so Free + // can release them after the primary. + companions []uintptr + engineMu sync.Mutex // sole guard of the one C engine (dispatcher + streaming) + bat *batcher + batStop chan struct{} // segmentGapFrames is NeMo's segment_gap_threshold in ENCODER FRAMES (model // YAML option, default 0=off). When >0 it adds NeMo's silence-gap split on // top of the punctuation split; converted to seconds via the JSON frame_sec. @@ -151,21 +211,17 @@ type ParakeetCpp struct { } // Load is the LocalAI gRPC entry point for LoadModel: it calls -// parakeet_capi_load with the GGUF path and stashes the resulting -// opaque context pointer for AudioTranscription. +// parakeet_capi_load with the GGUF path, classifies it and any companion +// models named in Options[] by role (see roles.go), and starts the dynamic +// batcher when an ASR context (primary or companion) ends up loaded. func (p *ParakeetCpp) Load(opts *pb.ModelOptions) error { if opts.ModelFile == "" { return errors.New("parakeet-cpp: ModelFile is required") } - ctx := CppLoad(opts.ModelFile) - if ctx == 0 { - // No ctx to ask for last_error (the C-API's last-error buffer - // lives on the ctx that was never returned). Surface the path - // so the operator at least knows which load failed. - return fmt.Errorf("parakeet-cpp: parakeet_capi_load failed for %q", opts.ModelFile) + if err := p.loadRoles(opts); err != nil { + return err } - p.ctxPtr = ctx // Dynamic batching knobs (model YAML options:, key:value form). Batching is // OFF by default (batch_max_size:1): each request runs on its own. On GPU, @@ -182,6 +238,12 @@ func (p *ParakeetCpp) Load(opts *pb.ModelOptions) error { // default matches NeMo's default (punctuation-only segments); when set it // additionally splits segments on inter-word silence (see transcriptResultFromDoc). p.segmentGapFrames = optInt(opts, "segment_gap_threshold", 0) + + // The batcher only ever drives the ASR context; a diarization/sound + // primary with no asr_model companion has no ctxPtr and needs none. + if p.ctxPtr == 0 { + return nil + } if CppTranscribePcmBatchJSON != nil { p.batStop = make(chan struct{}) p.bat = newBatcher(maxSize, time.Duration(maxWaitMs)*time.Millisecond, p.runBatch) @@ -287,12 +349,16 @@ func (p *ParakeetCpp) runBatch(reqs []*batchRequest) { // OpenAI API, whose default is segment-level); token ids always populate // Segment.Tokens. // -// translate/diarize/prompt/temperature/threads are not applicable to parakeet -// and are ignored; language is honored on the batched + streaming paths (see -// opts.GetLanguage() below); streaming is handled by AudioTranscriptionStream -// (L2). +// With a diarization_model companion, diarize=true labels segments with their +// speaker (speakers.go). translate/prompt/temperature/threads are not +// applicable to parakeet and are ignored; language is honored on the batched + +// streaming paths (see opts.GetLanguage() below); streaming is handled by +// AudioTranscriptionStream (L2). func (p *ParakeetCpp) AudioTranscription(ctx context.Context, opts *pb.TranscriptRequest) (pb.TranscriptResult, error) { if p.ctxPtr == 0 { + if err := p.notASRError(); err != nil { + return pb.TranscriptResult{}, err + } return pb.TranscriptResult{}, grpcerrors.ModelNotLoaded("parakeet-cpp") } if opts.Dst == "" { @@ -350,7 +416,17 @@ func (p *ParakeetCpp) AudioTranscription(ctx context.Context, opts *pb.Transcrip if err := json.Unmarshal([]byte(res.json), &doc); err != nil { return pb.TranscriptResult{}, fmt.Errorf("parakeet-cpp: decode transcript json: %w", err) } - return transcriptResultFromDoc(doc, opts, p.segmentGapFrames), nil + + // With a diarization_model companion, label each segment with its speaker. + var speakers []int + if p.wantSpeakers(opts.GetDiarize()) && len(doc.Words) > 0 { + segs, err := p.diarizeSegmentsPCM(pcm) + if err != nil { + return pb.TranscriptResult{}, err + } + speakers = assignSpeakers(doc.Words, segs) + } + return transcriptResultWithSpeakers(doc, opts, p.segmentGapFrames, speakers), nil } // segmentSeparators is NeMo's default segment_seperators (sentence-ending @@ -365,6 +441,14 @@ var segmentSeparators = []rune{'.', '?', '!'} // the caller requested word granularity; token ids populate each segment's // Tokens by time-window membership. Shared by the batched and direct paths. func transcriptResultFromDoc(doc transcriptJSON, opts *pb.TranscriptRequest, gapFrames int) pb.TranscriptResult { + return transcriptResultWithSpeakers(doc, opts, gapFrames, nil) +} + +// transcriptResultWithSpeakers is transcriptResultFromDoc plus optional +// per-word speakers (indexed like doc.Words, -1 = none; see speakers.go): +// segments additionally split wherever the speaker changes, and segments and +// words carry the speaker's label. +func transcriptResultWithSpeakers(doc transcriptJSON, opts *pb.TranscriptRequest, gapFrames int, speakers []int) pb.TranscriptResult { text, eou := stripEouMarker(strings.TrimSpace(doc.Text)) // Frame-unit gap threshold -> seconds (NeMo segment_gap_threshold). 0 = off. @@ -388,6 +472,11 @@ func transcriptResultFromDoc(doc transcriptJSON, opts *pb.TranscriptRequest, gap } } + var groupSpeakers []int + if speakers != nil && len(speakers) == len(doc.Words) { + groups, groupSpeakers = splitAtSpeakerChanges(groups, speakers) + } + wantWords := wordsRequested(opts.TimestampGranularities) segments := make([]*pb.TranscriptSegment, 0, len(groups)) for id, group := range groups { @@ -402,10 +491,14 @@ func transcriptResultFromDoc(doc transcriptJSON, opts *pb.TranscriptRequest, gap Text: strings.TrimSpace(strings.Join(parts, " ")), Tokens: tokensInWindow(doc.Tokens, group[0].Start, group[len(group)-1].End), } + if groupSpeakers != nil { + seg.Speaker = transcriptSpeaker(groupSpeakers[id]) + } if wantWords { ws := make([]*pb.TranscriptWord, len(group)) for i, gw := range group { - ws[i] = &pb.TranscriptWord{Start: secondsToNanos(gw.Start), End: secondsToNanos(gw.End), Text: gw.W} + ws[i] = &pb.TranscriptWord{Start: secondsToNanos(gw.Start), End: secondsToNanos(gw.End), Text: gw.W, + Speaker: seg.Speaker} } seg.Words = ws } @@ -503,10 +596,11 @@ func tokensInWindow(tokens []transcriptToken, start, end float64) []int32 { // text-only library (no words) it falls back to segmenting the delta text, so // the same assembler serves both paths. type streamSegmenter struct { - segs []*pb.TranscriptSegment - cur []transcriptWord // words for the open segment (ABI v4 JSON path) - curText []string // delta text for the open segment (text-only path) - nextID int32 + segs []*pb.TranscriptSegment + segWords [][]transcriptWord // words of each segment (nil for text-only ones) + cur []transcriptWord // words for the open segment (ABI v4 JSON path) + curText []string // delta text for the open segment (text-only path) + nextID int32 } func (s *streamSegmenter) add(r streamFeedResult) { @@ -534,12 +628,14 @@ func (s *streamSegmenter) flush() { End: secondsToNanos(s.cur[len(s.cur)-1].End), Text: strings.TrimSpace(strings.Join(parts, " ")), }) + s.segWords = append(s.segWords, s.cur) s.nextID++ case len(s.curText) > 0: // No words this segment: emit a text-only segment (no timestamps), // skipping a purely-whitespace one as the legacy text path did. if t := strings.TrimSpace(strings.Join(s.curText, "")); t != "" { s.segs = append(s.segs, &pb.TranscriptSegment{Id: s.nextID, Text: t}) + s.segWords = append(s.segWords, nil) s.nextID++ } } @@ -686,6 +782,9 @@ func (p *ParakeetCpp) AudioTranscriptionStream(ctx context.Context, opts *pb.Tra defer close(results) if p.ctxPtr == 0 { + if err := p.notASRError(); err != nil { + return err + } return grpcerrors.ModelNotLoaded("parakeet-cpp") } if opts.Dst == "" { @@ -753,6 +852,28 @@ func (p *ParakeetCpp) AudioTranscriptionStream(ctx context.Context, opts *pb.Tra // The single-segment fallback stays trimmed. fullText := full.String() segments := seg.segments() + + // With a diarization_model companion, label each utterance with the + // speaker who said most of it. The whole file is available, so this runs + // the same diarization as the unary path. + if p.wantSpeakers(opts.GetDiarize()) && len(seg.segWords) == len(segments) { + var all []transcriptWord + for _, ws := range seg.segWords { + all = append(all, ws...) + } + if len(all) > 0 { + segs, err := p.diarizeSegmentsPCM(data) + if err != nil { + return err + } + speakers := assignSpeakers(all, segs) + k := 0 + for i, ws := range seg.segWords { + segments[i].Speaker = transcriptSpeaker(majoritySpeaker(ws, speakers[k:k+len(ws)])) + k += len(ws) + } + } + } if trimmed := strings.TrimSpace(fullText); len(segments) == 0 && trimmed != "" { segments = append(segments, &pb.TranscriptSegment{Id: 0, Text: trimmed}) } @@ -817,8 +938,10 @@ func decodeWavMono16k(path string) ([]float32, float32, error) { return data, duration, nil } -// Free releases the underlying parakeet_ctx. Called by LocalAI when the -// model is unloaded. +// Free releases every parakeet_ctx this backend holds (the primary and any +// asr_model:/diarization_model:/sound_model: companions loaded in Load) and +// is idempotent: fields are zeroed as they are freed, so a second call frees +// nothing. Called by LocalAI when the model is unloaded. func (p *ParakeetCpp) Free() error { // Stop the dispatcher before releasing the engine so no in-flight runBatch // can touch a freed ctx (close leak / use-after-free on reload). @@ -830,10 +953,13 @@ func (p *ParakeetCpp) Free() error { // re-checks ctxPtr under the lock) can never feed into a freed ctx. p.engineMu.Lock() defer p.engineMu.Unlock() - if p.ctxPtr != 0 { - CppFree(p.ctxPtr) - p.ctxPtr = 0 + for _, ctxField := range [...]*uintptr{&p.ctxPtr, &p.diarCtx, &p.tagCtx} { + if *ctxField != 0 { + CppFree(*ctxField) + *ctxField = 0 + } } + p.companions = nil return nil } diff --git a/backend/go/parakeet-cpp/goparakeetcpp_test.go b/backend/go/parakeet-cpp/goparakeetcpp_test.go index a6f6af1f0..eda5eaa36 100644 --- a/backend/go/parakeet-cpp/goparakeetcpp_test.go +++ b/backend/go/parakeet-cpp/goparakeetcpp_test.go @@ -59,6 +59,23 @@ func ensureLibLoaded() { purego.RegisterLibFunc(&CppStreamFeedJSON, lib, "parakeet_capi_stream_feed_json") purego.RegisterLibFunc(&CppStreamFinalizeJSON, lib, "parakeet_capi_stream_finalize_json") } + // Diarization and model roles, probed like main.go (speakers_test.go). + if sym, err := purego.Dlsym(lib, "parakeet_capi_diarize_pcm"); err == nil && sym != 0 { + purego.RegisterLibFunc(&CppDiarizePCM, lib, "parakeet_capi_diarize_pcm") + } + if sym, err := purego.Dlsym(lib, "parakeet_capi_transcribe_and_diarize_json"); err == nil && sym != 0 { + purego.RegisterLibFunc(&CppTranscribeAndDiarizeJSON, lib, "parakeet_capi_transcribe_and_diarize_json") + } + if sym, err := purego.Dlsym(lib, "parakeet_capi_model_kind"); err == nil && sym != 0 { + purego.RegisterLibFunc(&CppModelKind, lib, "parakeet_capi_model_kind") + purego.RegisterLibFunc(&CppNumClasses, lib, "parakeet_capi_num_classes") + purego.RegisterLibFunc(&CppSoundOptsDefault, lib, "parakeet_capi_sound_opts_default") + purego.RegisterLibFunc(&CppSoundStreamBegin, lib, "parakeet_capi_sound_stream_begin") + purego.RegisterLibFunc(&CppSoundStreamFeed, lib, "parakeet_capi_sound_stream_feed") + purego.RegisterLibFunc(&CppSoundStreamDrainScoresJSON, lib, "parakeet_capi_sound_stream_drain_scores_json") + purego.RegisterLibFunc(&CppFreeSoundSegments, lib, "parakeet_capi_free_sound_segments") + purego.RegisterLibFunc(&CppSoundStreamFree, lib, "parakeet_capi_sound_stream_free") + } purego.RegisterLibFunc(&CppFreeString, lib, "parakeet_capi_free_string") purego.RegisterLibFunc(&CppLastError, lib, "parakeet_capi_last_error") }) @@ -203,6 +220,24 @@ var _ = Describe("ParakeetCpp", func() { }) Context("AudioTranscriptionStream", func() { + It("names the loaded role instead of a generic model-not-loaded error for a diarization primary", func() { + // CppStreamBegin/CppStreamBeginLang are left nil (zero value): if + // AudioTranscriptionStream tried to call either, this would panic + // instead of returning cleanly, so a clean typed error here also + // proves no C call was made. + p := &ParakeetCpp{diarCtx: 1} + results := make(chan *pb.TranscriptStreamResponse, 8) + err := p.AudioTranscriptionStream(context.Background(), + &pb.TranscriptRequest{Dst: "ignored.wav"}, results) + Expect(err).To(MatchError(ContainSubstring("diarization model"))) + + var emitted []*pb.TranscriptStreamResponse + for r := range results { + emitted = append(emitted, r) + } + Expect(emitted).To(BeEmpty()) + }) + It("returns the typed Unimplemented signal for non-streaming models (no offline fallback)", func() { // stream_begin == 0 means the loaded model is not a cache-aware // streaming model. The backend must surface that, not silently diff --git a/backend/go/parakeet-cpp/live.go b/backend/go/parakeet-cpp/live.go index 3d68a2914..6497779d5 100644 --- a/backend/go/parakeet-cpp/live.go +++ b/backend/go/parakeet-cpp/live.go @@ -41,6 +41,9 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest defer close(out) if p.ctxPtr == 0 { + if err := p.notASRError(); err != nil { + return err + } return grpcerrors.ModelNotLoaded("parakeet-cpp") } @@ -68,6 +71,23 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest // current when the RPC unwinds. defer func() { p.streamFree(stream) }() + // scene runs a no-ASR scene stream (diarization/sound only) beside the + // ASR session when a diarization_model:/sound_model: companion is loaded + // (see scene.go). A zero handle means scene events are disabled: no + // companions, or the begin/a later feed call failed (logged below / in + // feedSlicesScene), in which case live transcription continues ASR-only. + // Reassigned on a mid-stream Config reset alongside stream, which also + // brings back a scene stream the session had disabled after an earlier + // scene error. + var scene sceneStreamHandle + if p.sceneWanted() { + scene = p.sceneBegin() + if scene.s == 0 { + xlog.Warn("parakeet-cpp: scene stream begin failed; live continues without speaker/sound events") + } + } + defer func() { p.sceneFree(scene) }() + out <- &pb.TranscriptLiveResponse{Ready: true} var ( @@ -83,22 +103,31 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest behindWarned bool ) - // emit forwards one decode increment: it streams the per-feed tokens the - // realtime turn detector consumes (delta/eou/eob/words) and accumulates the - // running transcript for the closing FinalResult. No segmentation or - // boundary latch here — the live consumer reads only the streamed tokens - // and the final Text; per-utterance segments and the terminal flag - // are an offline-path concern (see AudioTranscriptionStream / boundary.go). - emit := func(r streamFeedResult) error { + // emit sends one decode increment as its own response when it carries + // anything: either the ASR side (delta/eou/eob/words, accumulated into + // the running transcript for the closing FinalResult) or the scene + // side (closed speaker/sound events), never both at once — the live + // audio loop below calls it once for the ASR result right after the ASR + // feed and, separately, once more for the scene document after the + // scene feed (see feedSlicesScene), so a slice with both produces two + // responses, ASR first. No segmentation or boundary latch here — the + // live consumer reads only the streamed tokens and the final Text; + // per-utterance segments and the terminal flag are an + // offline-path concern (see AudioTranscriptionStream / boundary.go). + emit := func(r streamFeedResult, sceneDoc sceneFeedJSON) error { if r.Delta != "" { full.WriteString(r.Delta) } - if r.Delta != "" || r.Eou || r.Eob || len(r.Words) > 0 { + speakers := liveSpeakersToProto(sceneDoc.Speakers) + sounds := liveSoundsToProto(sceneDoc.Sounds) + if r.Delta != "" || r.Eou || r.Eob || len(r.Words) > 0 || len(speakers) > 0 || len(sounds) > 0 { out <- &pb.TranscriptLiveResponse{ - Delta: r.Delta, - Eou: r.Eou, - Eob: r.Eob, - Words: liveWordsToProto(r.Words), + Delta: r.Delta, + Eou: r.Eou, + Eob: r.Eob, + Words: liveWordsToProto(r.Words), + Speakers: speakers, + Sounds: sounds, } } return nil @@ -120,8 +149,20 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest return grpcerrors.LiveTranscriptionUnsupported("parakeet-cpp", "loaded model is not a cache-aware streaming model") } + // The scene stream is freed and begun again alongside the ASR + // session, mirroring the reset above. + p.sceneFree(scene) + scene = sceneStreamHandle{} + if p.sceneWanted() { + scene = p.sceneBegin() + if scene.s == 0 { + xlog.Warn("parakeet-cpp: scene stream begin failed; live continues without speaker/sound events") + } + } full.Reset() fedSecs = 0 + behindSec = 0 + behindWarned = false case *pb.TranscriptLiveRequest_Audio: pcm := payload.Audio.GetPcm() audioSec := float64(len(pcm)) / liveSampleRate @@ -129,7 +170,9 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest start := time.Now() // nil ctx: a live session is bounded by this request channel, not a // context — cancellation is the caller closing the stream. - if err := p.feedSlices(nil, stream, pcm, emit); err != nil { + var asrWall, sceneWall time.Duration + scene, asrWall, sceneWall, err = p.feedSlicesScene(nil, stream, scene, pcm, emit) + if err != nil { return err } wallSec := time.Since(start).Seconds() @@ -139,6 +182,7 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest } xlog.Debug("parakeet-cpp: live feed", "audio_ms", int(audioSec*1000), "wall_ms", int(wallSec*1000), + "asr_wall_ms", int(asrWall.Seconds()*1000), "scene_wall_ms", int(sceneWall.Seconds()*1000), "behind_ms", int(behindSec*1000), "fed_s", fedSecs) if behindSec > 1 && !behindWarned { behindWarned = true @@ -153,9 +197,23 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest // The live FinalResult carries only Text — the authoritative full-turn // transcript the realtime core commits. Per-utterance segments, duration, // and the terminal flag are not produced on the live path. - if err := p.flushTail(stream, emit); err != nil { + if err := p.flushTail(stream, func(r streamFeedResult) error { + return emit(r, sceneFeedJSON{}) + }); err != nil { return err } + // The scene stream gets its own is_last flush (it consumes no new audio + // here, so it is not part of flushTail above); its remaining events go + // out before the terminal FinalResult, then the stream is released by the + // deferred sceneFree above. + if scene.s != 0 { + doc, err := p.sceneFeed(scene, nil, true) + if err != nil { + xlog.Warn("parakeet-cpp: live scene finalize failed", "err", err) + } else if err := emit(streamFeedResult{}, doc); err != nil { + return err + } + } out <- &pb.TranscriptLiveResponse{ FinalResult: &pb.TranscriptResult{Text: strings.TrimSpace(full.String())}, } diff --git a/backend/go/parakeet-cpp/live_test.go b/backend/go/parakeet-cpp/live_test.go index 0462ee521..11dc496b6 100644 --- a/backend/go/parakeet-cpp/live_test.go +++ b/backend/go/parakeet-cpp/live_test.go @@ -42,22 +42,61 @@ func liveStubs() (restore func()) { savedFinalize, savedFinalizeJSON := CppStreamFinalize, CppStreamFinalizeJSON savedFree, savedLastError := CppStreamFree, CppLastError savedFreeString := CppFreeString + savedSceneOptsDefault := CppSceneOptsDefault + savedSceneBegin := CppSceneStreamBegin + savedSceneFeedJSON := CppSceneStreamFeedJSON + savedSceneLastError := CppSceneStreamLastError + savedSceneFree := CppSceneStreamFree return func() { CppStreamBegin, CppStreamBeginLang = savedBegin, savedBeginLang CppStreamFeed, CppStreamFeedJSON = savedFeed, savedFeedJSON CppStreamFinalize, CppStreamFinalizeJSON = savedFinalize, savedFinalizeJSON CppStreamFree, CppLastError = savedFree, savedLastError CppFreeString = savedFreeString + CppSceneOptsDefault = savedSceneOptsDefault + CppSceneStreamBegin = savedSceneBegin + CppSceneStreamFeedJSON = savedSceneFeedJSON + CppSceneStreamLastError = savedSceneLastError + CppSceneStreamFree = savedSceneFree } } +// liveSceneStubs wires a minimal scene stream stub set onto p (a diarization +// and/or sound companion context so sceneWanted() is true) and returns the +// call-count trackers the specs assert on. feedJSON is called once per scene +// feed (including the is_last flush) with the stream handle the C side would +// have received (so a reset spec can tell a pre-reset feed from a post-reset +// one) and must return the canned document for that call. +func liveSceneStubs(feedJSON func(calls int, s uintptr, isLast int32) uintptr) (begun, freed *int) { + begun, freed = new(int), new(int) + CppSceneOptsDefault = func(o *cSceneOpts) { *o = cSceneOpts{} } + CppSceneStreamBegin = func(asr, diar, tagger uintptr, o *cSceneOpts) uintptr { + *begun++ + return uintptr(100 + *begun) + } + calls := 0 + CppSceneStreamFeedJSON = func(s uintptr, pcm *float32, n int32, isLast int32) uintptr { + calls++ + return feedJSON(calls, s, isLast) + } + CppSceneStreamLastError = func(s uintptr) string { return "scene stub error" } + CppSceneStreamFree = func(s uintptr) { *freed++ } + return begun, freed +} + // runLive starts the RPC on its own goroutine and returns the request -// channel plus a collector for everything the backend emitted. +// channel plus a collector for everything the backend emitted. GinkgoRecover +// turns an Expect/Fail failure inside a stub called from this goroutine into +// a normal spec failure instead of a panic that would crash the whole test +// binary (Ginkgo's failure handling is goroutine-local). func runLive(p *ParakeetCpp) (chan *pb.TranscriptLiveRequest, chan *pb.TranscriptLiveResponse, chan error) { in := make(chan *pb.TranscriptLiveRequest) out := make(chan *pb.TranscriptLiveResponse, 32) errCh := make(chan error, 1) - go func() { errCh <- p.AudioTranscriptionLive(in, out) }() + go func() { + defer GinkgoRecover() + errCh <- p.AudioTranscriptionLive(in, out) + }() return in, out, errCh } @@ -106,6 +145,19 @@ var _ = Describe("AudioTranscriptionLive (stubbed C API)", func() { AfterEach(func() { restore() }) + It("names the loaded role instead of a generic model-not-loaded error for a sound primary", func() { + // The ctxPtr==0 check returns before AudioTranscriptionLive ever reads + // from `in`, so nothing may be sent on it (unbuffered: a send would + // block forever waiting for a read that never happens). + p2 := &ParakeetCpp{tagCtx: 1} + in, out, errCh := runLive(p2) + close(in) + + err := <-errCh + Expect(err).To(MatchError(ContainSubstring("sound model"))) + Expect(collectLive(out)).To(BeEmpty()) + }) + It("rejects a stream whose first message is not a config", func() { in, out, errCh := runLive(p) in <- liveAudio([]float32{0.1}) @@ -368,6 +420,256 @@ var _ = Describe("AudioTranscriptionLive (stubbed C API)", func() { Expect(got).To(HaveLen(1)) // just the ready ack close(in) }) + + It("makes no scene C call and behaves unchanged when no companion is loaded", func() { + // p has ctxPtr only (no diarCtx/tagCtx): sceneWanted() must be false, + // and none of the scene entry points may be touched. + CppSceneOptsDefault = func(o *cSceneOpts) { Fail("scene_opts_default called with no companions loaded") } + CppSceneStreamBegin = func(asr, diar, tagger uintptr, o *cSceneOpts) uintptr { + Fail("scene_stream_begin called with no companions loaded") + return 0 + } + CppSceneStreamFeedJSON = func(s uintptr, pcm *float32, n int32, isLast int32) uintptr { + Fail("scene_stream_feed_json called with no companions loaded") + return 0 + } + CppSceneStreamFree = func(s uintptr) { Fail("scene_stream_free called with no companions loaded") } + + CppStreamFeedJSON = func(s uintptr, pcm []float32, n int32) uintptr { + return pool.cstr(`{"text":"hi","eou":0,"frame_sec":0.08,"words":[]}`) + } + CppStreamFinalizeJSON = func(s uintptr) uintptr { + return pool.cstr(`{"text":"","eou":0,"frame_sec":0.08,"words":[]}`) + } + + in, out, errCh := runLive(p) + in <- liveConfig("") + in <- liveAudio(make([]float32, 10)) + close(in) + Expect(<-errCh).NotTo(HaveOccurred()) + + got := collectLive(out) + Expect(got).To(HaveLen(3)) // ready, delta, final + Expect(got[1].Speakers).To(BeEmpty()) + Expect(got[1].Sounds).To(BeEmpty()) + }) +}) + +var _ = Describe("AudioTranscriptionLive scene events (stubbed C API)", func() { + var ( + pool *liveCstrPool + restore func() + p *ParakeetCpp + ) + + BeforeEach(func() { + pool = &liveCstrPool{} + restore = liveStubs() + p = &ParakeetCpp{ctxPtr: 1, diarCtx: 2} + + CppStreamBeginLang = nil + CppStreamBegin = func(ctx uintptr) uintptr { return 7 } + CppStreamFree = func(s uintptr) {} + CppFreeString = func(s uintptr) {} + CppLastError = func(ctx uintptr) string { return "stub error" } + CppStreamFeed = nil + CppStreamFeedJSON = func(s uintptr, pcm []float32, n int32) uintptr { + return pool.cstr(`{"text":"","eou":0,"frame_sec":0.08,"words":[]}`) + } + CppStreamFinalize = nil + CppStreamFinalizeJSON = func(s uintptr) uintptr { + return pool.cstr(`{"text":"","eou":0,"frame_sec":0.08,"words":[]}`) + } + }) + + AfterEach(func() { restore() }) + + It("emits a closed speaker segment as its own response", func() { + liveSceneStubs(func(calls int, s uintptr, isLast int32) uintptr { + if calls == 1 { + return pool.cstr(`{"speakers":[{"speaker":0,"start":0.1,"end":0.6}],"sounds":[]}`) + } + return pool.cstr(`{"speakers":[],"sounds":[]}`) + }) + + in, out, errCh := runLive(p) + in <- liveConfig("") + in <- liveAudio(make([]float32, 10)) + close(in) + Expect(<-errCh).NotTo(HaveOccurred()) + + got := collectLive(out) + Expect(got).To(HaveLen(3)) // ready, speaker-only response, final + Expect(got[1].Delta).To(BeEmpty()) + Expect(got[1].Speakers).To(HaveLen(1)) + Expect(got[1].Speakers[0].Speaker).To(Equal("0")) + Expect(got[1].Speakers[0].Start).To(Equal(int64(0.1 * 1e9))) + Expect(got[1].Speakers[0].End).To(Equal(int64(0.6 * 1e9))) + }) + + It("sends the ASR delta and a scene sound event as two responses, ASR first", func() { + CppStreamFeedJSON = func(s uintptr, pcm []float32, n int32) uintptr { + return pool.cstr(`{"text":"hello ","eou":0,"frame_sec":0.08,` + + `"words":[{"w":"hello","start":0.1,"end":0.4,"conf":0.9}]}`) + } + liveSceneStubs(func(calls int, s uintptr, isLast int32) uintptr { + if calls == 1 { + return pool.cstr(`{"speakers":[],"sounds":[{"index":99,` + + `"label":"Chicken, rooster","start":24.0,"end":30.0,"peak":0.86}]}`) + } + return pool.cstr(`{"speakers":[],"sounds":[]}`) + }) + + in, out, errCh := runLive(p) + in <- liveConfig("") + in <- liveAudio(make([]float32, 10)) + close(in) + Expect(<-errCh).NotTo(HaveOccurred()) + + got := collectLive(out) + Expect(got).To(HaveLen(4)) // ready, ASR delta, sound-only, final + Expect(got[1].Delta).To(Equal("hello ")) + Expect(got[1].Sounds).To(BeEmpty(), "the ASR response must not wait on the scene feed") + Expect(got[2].Delta).To(BeEmpty()) + Expect(got[2].Sounds).To(HaveLen(1)) + Expect(got[2].Sounds[0].Label).To(Equal("Chicken, rooster")) + Expect(got[2].Sounds[0].Index).To(Equal(int32(99))) + Expect(got[2].Sounds[0].Peak).To(BeNumerically("~", 0.86, 1e-6)) + Expect(got[2].Sounds[0].Start).To(Equal(int64(24.0 * 1e9))) + Expect(got[2].Sounds[0].End).To(Equal(int64(30.0 * 1e9))) + }) + + It("flushes the scene stream is_last before the final result, then frees it", func() { + begun, freed := liveSceneStubs(func(calls int, s uintptr, isLast int32) uintptr { + if calls == 2 { + Expect(isLast).To(Equal(int32(1))) + return pool.cstr(`{"speakers":[{"speaker":1,"start":1.0,"end":2.0}],"sounds":[]}`) + } + Expect(isLast).To(Equal(int32(0))) + return pool.cstr(`{"speakers":[],"sounds":[]}`) + }) + + in, out, errCh := runLive(p) + in <- liveConfig("") + in <- liveAudio(make([]float32, 10)) + close(in) + Expect(<-errCh).NotTo(HaveOccurred()) + + got := collectLive(out) + Expect(got).To(HaveLen(3)) // ready, speaker from the is_last flush, final + Expect(got[1].Speakers).To(HaveLen(1)) + Expect(got[1].Speakers[0].Speaker).To(Equal("1")) + Expect(got[2].FinalResult).NotTo(BeNil()) + Expect(*begun).To(Equal(1)) + Expect(*freed).To(Equal(1)) + }) + + It("frees and begins the scene stream again on a mid-stream config reset", func() { + streamBegun := 0 + CppStreamBegin = func(ctx uintptr) uintptr { streamBegun++; return uintptr(10 + streamBegun) } + var seenStreams []uintptr + begun, freed := liveSceneStubs(func(calls int, s uintptr, isLast int32) uintptr { + seenStreams = append(seenStreams, s) + return pool.cstr(`{"speakers":[],"sounds":[]}`) + }) + + in, out, errCh := runLive(p) + in <- liveConfig("") + in <- liveAudio(make([]float32, 10)) + in <- liveConfig("") // reset + in <- liveAudio(make([]float32, 10)) + close(in) + Expect(<-errCh).NotTo(HaveOccurred()) + collectLive(out) + + Expect(*begun).To(Equal(2), "scene stream begun again on reset") + Expect(*freed).To(Equal(2), "old scene stream freed on reset, new one on unwind") + // One scene feed per audio message (pre-reset, post-reset) plus the + // close is_last flush, which runs on the post-reset stream. + Expect(seenStreams).To(HaveLen(3)) + Expect(seenStreams[0]).NotTo(Equal(seenStreams[1]), "post-reset audio must go to the new scene stream handle") + Expect(seenStreams[2]).To(Equal(seenStreams[1]), "the close flush uses the post-reset stream too") + }) + + It("returns without a C call when Free() ran between begin and a scene feed", func() { + feedCalls := 0 + CppSceneOptsDefault = func(o *cSceneOpts) { *o = cSceneOpts{} } + CppSceneStreamBegin = func(asr, diar, tagger uintptr, o *cSceneOpts) uintptr { return 999 } + CppSceneStreamFeedJSON = func(s uintptr, pcm *float32, n int32, isLast int32) uintptr { + feedCalls++ + return pool.cstr(`{"speakers":[],"sounds":[]}`) + } + + h := p.sceneBegin() + Expect(h.s).NotTo(BeZero()) + + // Simulate a Free() racing in between the begin and the next feed: it + // zeroes the companion context under engineMu, exactly as the real + // Free() does. + p.diarCtx = 0 + + _, err := p.sceneFeed(h, make([]float32, 10), false) + Expect(grpcerrors.IsModelNotLoaded(err)).To(BeTrue()) + Expect(feedCalls).To(Equal(0), "no C call once the scene stream's contexts were freed") + }) + + It("degrades to ASR-only after a mid-session scene feed failure: freed once, no more scene events, ASR keeps working", func() { + CppStreamFeedJSON = func(s uintptr, pcm []float32, n int32) uintptr { + return pool.cstr(`{"text":"hi ","eou":0,"frame_sec":0.08,` + + `"words":[{"w":"hi","start":0.1,"end":0.3,"conf":0.9}]}`) + } + sceneFeedCalls := 0 + begun, freed := liveSceneStubs(func(calls int, s uintptr, isLast int32) uintptr { + sceneFeedCalls++ + return 0 // fails every call; only the first should ever be reached + }) + + in, out, errCh := runLive(p) + in <- liveConfig("") + in <- liveAudio(make([]float32, 10)) // scene feed fails here: warn, free, zero the handle + in <- liveAudio(make([]float32, 10)) // ASR-only: no scene C call at all + close(in) + Expect(<-errCh).NotTo(HaveOccurred()) + + got := collectLive(out) + // ready, ASR delta (msg 1), ASR delta (msg 2), final: no scene-only + // response ever appears, before or after the failure. + Expect(got).To(HaveLen(4)) + Expect(got[0].Ready).To(BeTrue()) + Expect(got[1].Delta).To(Equal("hi ")) + Expect(got[2].Delta).To(Equal("hi ")) + for _, r := range got { + Expect(r.Speakers).To(BeEmpty(), "no speaker events once the scene stream has failed") + Expect(r.Sounds).To(BeEmpty(), "no sound events once the scene stream has failed") + } + Expect(got[3].FinalResult).NotTo(BeNil()) + Expect(got[3].FinalResult.Text).To(Equal("hi hi")) + + Expect(*begun).To(Equal(1)) + Expect(sceneFeedCalls).To(Equal(1), "the second audio message must not retry the broken scene stream") + Expect(*freed).To(Equal(1), "the broken scene stream is freed exactly once, not again at RPC unwind") + }) + + It("continues without scene events when scene begin fails", func() { + CppSceneOptsDefault = func(o *cSceneOpts) { *o = cSceneOpts{} } + CppSceneStreamBegin = func(asr, diar, tagger uintptr, o *cSceneOpts) uintptr { return 0 } + sceneFeedCalled := false + CppSceneStreamFeedJSON = func(s uintptr, pcm *float32, n int32, isLast int32) uintptr { + sceneFeedCalled = true + return 0 + } + CppSceneStreamFree = func(s uintptr) {} + + in, out, errCh := runLive(p) + in <- liveConfig("") + in <- liveAudio(make([]float32, 10)) + close(in) + Expect(<-errCh).NotTo(HaveOccurred()) + + got := collectLive(out) + Expect(got).To(HaveLen(2)) // ready, final only: no scene events, no ASR delta this stub sends + Expect(sceneFeedCalled).To(BeFalse(), "no feed call once begin failed") + }) }) var _ = Describe("stripEouMarker", func() { diff --git a/backend/go/parakeet-cpp/main.go b/backend/go/parakeet-cpp/main.go index 9c6466b13..865d5b6f1 100644 --- a/backend/go/parakeet-cpp/main.go +++ b/backend/go/parakeet-cpp/main.go @@ -90,6 +90,34 @@ func main() { purego.RegisterLibFunc(&CppStreamFinalizeJSON, lib, "parakeet_capi_stream_finalize_json") } + // Model roles + diarization/sound (ABI v7-v8): parakeet_capi_model_kind is + // what lets Load tell an ASR/diarization/sound context apart, so it gates + // every other new symbol below (an older libparakeet.so gets none of + // them, and companion model options are rejected in roles.go). Diarization + // itself (diarize_pcm, transcribe_and_diarize_json) predates model_kind + // (ABI v7), so it is probed on its own. + if sym, err := purego.Dlsym(lib, "parakeet_capi_diarize_pcm"); err == nil && sym != 0 { + purego.RegisterLibFunc(&CppDiarizePCM, lib, "parakeet_capi_diarize_pcm") + } + if sym, err := purego.Dlsym(lib, "parakeet_capi_transcribe_and_diarize_json"); err == nil && sym != 0 { + purego.RegisterLibFunc(&CppTranscribeAndDiarizeJSON, lib, "parakeet_capi_transcribe_and_diarize_json") + } + if sym, err := purego.Dlsym(lib, "parakeet_capi_model_kind"); err == nil && sym != 0 { + purego.RegisterLibFunc(&CppModelKind, lib, "parakeet_capi_model_kind") + purego.RegisterLibFunc(&CppNumClasses, lib, "parakeet_capi_num_classes") + purego.RegisterLibFunc(&CppSoundOptsDefault, lib, "parakeet_capi_sound_opts_default") + purego.RegisterLibFunc(&CppSoundStreamBegin, lib, "parakeet_capi_sound_stream_begin") + purego.RegisterLibFunc(&CppSoundStreamFeed, lib, "parakeet_capi_sound_stream_feed") + purego.RegisterLibFunc(&CppSoundStreamDrainScoresJSON, lib, "parakeet_capi_sound_stream_drain_scores_json") + purego.RegisterLibFunc(&CppFreeSoundSegments, lib, "parakeet_capi_free_sound_segments") + purego.RegisterLibFunc(&CppSoundStreamFree, lib, "parakeet_capi_sound_stream_free") + purego.RegisterLibFunc(&CppSceneOptsDefault, lib, "parakeet_capi_scene_opts_default") + purego.RegisterLibFunc(&CppSceneStreamBegin, lib, "parakeet_capi_scene_stream_begin") + purego.RegisterLibFunc(&CppSceneStreamFeedJSON, lib, "parakeet_capi_scene_stream_feed_json") + purego.RegisterLibFunc(&CppSceneStreamLastError, lib, "parakeet_capi_scene_stream_last_error") + purego.RegisterLibFunc(&CppSceneStreamFree, lib, "parakeet_capi_scene_stream_free") + } + fmt.Fprintf(os.Stderr, "[parakeet-cpp] ABI=%d\n", CppAbiVersion()) flag.Parse() diff --git a/backend/go/parakeet-cpp/roles.go b/backend/go/parakeet-cpp/roles.go new file mode 100644 index 000000000..c6a1e912e --- /dev/null +++ b/backend/go/parakeet-cpp/roles.go @@ -0,0 +1,245 @@ +package main + +import ( + "errors" + "fmt" + "path/filepath" + "strings" + + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + "github.com/mudler/xlog" +) + +// Model kinds returned by parakeet_capi_model_kind (ABI v8; mirrors the +// PARAKEET_MODEL_KIND_* defines in parakeet_capi.h). +const ( + modelKindNone = 0 + modelKindASR = 1 + modelKindDiarization = 2 + modelKindSound = 3 +) + +// Diarization streaming latency modes (mirrors PARAKEET_DIAR_LATENCY_* in +// parakeet_capi.h). diarLatencyLow is the spec's default when +// diarization_latency: is unset. +const ( + diarLatencyModel int32 = 0 + diarLatencyLow int32 = 1 + diarLatencyVeryLow int32 = 2 + diarLatencyUltraLow int32 = 3 +) + +// modelKindName renders a model kind for error messages. +func modelKindName(kind int32) string { + switch kind { + case modelKindASR: + return "ASR" + case modelKindDiarization: + return "diarization" + case modelKindSound: + return "sound" + default: + return "unknown" + } +} + +// optString reads a string model option (key:value form) from ModelOptions, +// returning "" when the key is absent. Same strings.Cut parsing as optInt. +func optString(opts *pb.ModelOptions, key string) string { + for _, o := range opts.GetOptions() { + k, v, ok := strings.Cut(o, ":") + if ok && strings.TrimSpace(k) == key { + return strings.TrimSpace(v) + } + } + return "" +} + +// resolveModelPath resolves a companion model option's path against +// modelPath (opts.ModelPath, the LocalAI models root): an absolute p, or an +// empty modelPath, passes through unchanged; anything else is joined onto +// modelPath. Mirrors vibevoice-cpp's resolvePath for tokenizer=/voice=/etc. +func resolveModelPath(modelPath, p string) string { + if p == "" || filepath.IsAbs(p) || modelPath == "" { + return p + } + return filepath.Join(modelPath, p) +} + +// parseDiarLatency maps the diarization_latency option value to a +// PARAKEET_DIAR_LATENCY_* mode. "" defaults to "low" (the spec's default); +// any other unrecognized value is a Load error. +func parseDiarLatency(s string) (int32, error) { + switch strings.ToLower(strings.TrimSpace(s)) { + case "": + return diarLatencyLow, nil + case "model": + return diarLatencyModel, nil + case "low": + return diarLatencyLow, nil + case "very_low": + return diarLatencyVeryLow, nil + case "ultra_low": + return diarLatencyUltraLow, nil + default: + return 0, fmt.Errorf("parakeet-cpp: unknown diarization_latency %q (want model|low|very_low|ultra_low)", s) + } +} + +// companionSpec is one asr_model:/diarization_model:/sound_model: option: its +// name (for error messages and path resolution), the raw option value, the +// model kind the loaded companion must report, the ParakeetCpp field it is +// assigned to on success, and a getter for that same field's current value +// (used to reject a companion whose role the primary already occupies). +type companionSpec struct { + optName string + value string + wantKind int32 + assign func(*ParakeetCpp, uintptr) + current func(*ParakeetCpp) uintptr +} + +// indefiniteArticle returns "an" for a word starting with a vowel sound and +// "a" otherwise, for grammatical error messages built from modelKindName. +func indefiniteArticle(word string) string { + if len(word) == 0 { + return "a" + } + switch word[0] { + case 'A', 'E', 'I', 'O', 'U', 'a', 'e', 'i', 'o', 'u': + return "an" + default: + return "a" + } +} + +// loadRoles loads opts.ModelFile as the primary parakeet_ctx, classifies it +// with parakeet_capi_model_kind (ABI v8) into ctxPtr/diarCtx/tagCtx, and +// loads any companion models named in Options[] (asr_model:, +// diarization_model:, sound_model:; paths resolved against opts.ModelPath). +// It also parses diarization_latency: into p.diarLatency. +// +// Against an older libparakeet.so (CppModelKind == nil) the primary is +// treated as ASR — the pre-v8 behavior — and companion model options are +// rejected outright, since there is no way to verify what they loaded. +// +// On any failure every context this call opened (primary and any companions +// loaded before the failure) is freed before the error is returned. +func (p *ParakeetCpp) loadRoles(opts *pb.ModelOptions) error { + diarModelOpt := optString(opts, "diarization_model") + asrModelOpt := optString(opts, "asr_model") + soundModelOpt := optString(opts, "sound_model") + hasCompanionOpts := diarModelOpt != "" || asrModelOpt != "" || soundModelOpt != "" + + if hasCompanionOpts && CppModelKind == nil { + return errors.New("parakeet-cpp: asr_model/diarization_model/sound_model options need " + + "parakeet_capi_model_kind (ABI v8) to verify what they load; the loaded libparakeet.so " + + "is too old to report companion model roles") + } + + latency, err := parseDiarLatency(optString(opts, "diarization_latency")) + if err != nil { + return err + } + + primary := CppLoad(opts.ModelFile) + if primary == 0 { + // No ctx to ask for last_error (the C-API's last-error buffer lives on + // the ctx that was never returned). Surface the path so the operator + // at least knows which load failed. + return fmt.Errorf("parakeet-cpp: parakeet_capi_load failed for %q", opts.ModelFile) + } + loaded := []uintptr{primary} + // freeLoaded undoes everything loadRoles opened this call: every context + // it freed AND every ParakeetCpp field it may have assigned (the primary + // lands in one of ctxPtr/diarCtx/tagCtx before the companion loop runs, + // and an earlier companion's spec.assign runs before a later one fails). + // Leaving a role field pointing at a freed ctx would double-free it on a + // later Free() call. + freeLoaded := func() { + for _, c := range loaded { + CppFree(c) + } + p.ctxPtr, p.diarCtx, p.tagCtx = 0, 0, 0 + p.companions = nil + } + + primaryKind := int32(modelKindASR) // old-library default: today's behavior + if CppModelKind != nil { + primaryKind = CppModelKind(primary) + if primaryKind == modelKindNone { + xlog.Warn("parakeet-cpp: parakeet_capi_model_kind reported PARAKEET_MODEL_KIND_NONE " + + "for a successfully loaded primary; treating it as an ASR model") + } + } + switch primaryKind { + case modelKindDiarization: + p.diarCtx = primary + case modelKindSound: + p.tagCtx = primary + default: + p.ctxPtr = primary + } + + specs := []companionSpec{ + {"diarization_model", diarModelOpt, modelKindDiarization, + func(pp *ParakeetCpp, c uintptr) { pp.diarCtx = c }, + func(pp *ParakeetCpp) uintptr { return pp.diarCtx }}, + {"asr_model", asrModelOpt, modelKindASR, + func(pp *ParakeetCpp, c uintptr) { pp.ctxPtr = c }, + func(pp *ParakeetCpp) uintptr { return pp.ctxPtr }}, + {"sound_model", soundModelOpt, modelKindSound, + func(pp *ParakeetCpp, c uintptr) { pp.tagCtx = c }, + func(pp *ParakeetCpp) uintptr { return pp.tagCtx }}, + } + for _, spec := range specs { + if spec.value == "" { + continue + } + // A companion whose role the primary already occupies (e.g. asr_model: + // on an already-ASR primary) would overwrite that role field below, + // leaking the primary ctx: Free() only walks ctxPtr/diarCtx/tagCtx, so + // the overwritten pointer is never freed. Reject it before loading. + if spec.current(p) != 0 { + freeLoaded() + return fmt.Errorf("parakeet-cpp: %s is not allowed on %s %s model", + spec.optName, indefiniteArticle(modelKindName(spec.wantKind)), modelKindName(spec.wantKind)) + } + resolved := resolveModelPath(opts.ModelPath, spec.value) + cctx := CppLoad(resolved) + if cctx == 0 { + freeLoaded() + return fmt.Errorf("parakeet-cpp: failed to load %s %q", spec.optName, resolved) + } + loaded = append(loaded, cctx) + if gotKind := CppModelKind(cctx); gotKind != spec.wantKind { + freeLoaded() + return fmt.Errorf("parakeet-cpp: %s %q is a %s model, expected a %s model", + spec.optName, resolved, modelKindName(gotKind), modelKindName(spec.wantKind)) + } + spec.assign(p, cctx) + p.companions = append(p.companions, cctx) + } + + p.diarLatency = latency + return nil +} + +// notASRError reports why AudioTranscription (and the streaming/live RPCs) +// cannot run when p.ctxPtr == 0: a loaded diarization or sound primary with +// no asr_model companion, named explicitly so the caller knows to use the +// right RPC instead of a generic "model not loaded". Returns nil when +// neither role is loaded (genuinely no model), leaving the caller to report +// the ordinary ModelNotLoaded error. +func (p *ParakeetCpp) notASRError() error { + switch { + case p.diarCtx != 0: + return errors.New("parakeet-cpp: loaded model is a diarization model, not ASR " + + "(use Diarize, or load with an asr_model: companion)") + case p.tagCtx != 0: + return errors.New("parakeet-cpp: loaded model is a sound model, not ASR " + + "(use SoundDetection)") + default: + return nil + } +} diff --git a/backend/go/parakeet-cpp/roles_test.go b/backend/go/parakeet-cpp/roles_test.go new file mode 100644 index 000000000..7b2a892b3 --- /dev/null +++ b/backend/go/parakeet-cpp/roles_test.go @@ -0,0 +1,417 @@ +package main + +import ( + "context" + + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +// The role-loading specs drive Load/Free entirely against stubbed +// CppLoad/CppModelKind/CppFree (the same seam live_test.go and +// batcher_test.go use), so they run without libparakeet.so. + +// fakeLib is a tiny in-memory stand-in for libparakeet.so: paths registered +// via withModel resolve to a fresh ctx handle of the given model kind on +// CppLoad, any other path fails the load, CppModelKind reads the kind back +// by ctx, and CppFree records every ctx it was asked to free, in order. +type fakeLib struct { + kinds map[string]int32 + ctxKind map[uintptr]int32 + next uintptr + loadedPaths []string + freed []uintptr +} + +func newFakeLib() *fakeLib { + return &fakeLib{kinds: map[string]int32{}, ctxKind: map[uintptr]int32{}, next: 1} +} + +func (f *fakeLib) withModel(path string, kind int32) *fakeLib { + f.kinds[path] = kind + return f +} + +// install swaps CppLoad/CppModelKind/CppFree for fakes backed by f and +// returns a restore func for AfterEach (mirrors live_test.go's liveStubs). +func (f *fakeLib) install() (restore func()) { + savedLoad, savedKind, savedFree := CppLoad, CppModelKind, CppFree + CppLoad = func(path string) uintptr { + f.loadedPaths = append(f.loadedPaths, path) + kind, ok := f.kinds[path] + if !ok { + return 0 + } + ctx := f.next + f.next++ + f.ctxKind[ctx] = kind + return ctx + } + CppModelKind = func(ctx uintptr) int32 { return f.ctxKind[ctx] } + CppFree = func(ctx uintptr) { f.freed = append(f.freed, ctx) } + return func() { + CppLoad, CppModelKind, CppFree = savedLoad, savedKind, savedFree + } +} + +var _ = Describe("model roles (stubbed C API)", func() { + var restore func() + + AfterEach(func() { + if restore != nil { + restore() + restore = nil + } + }) + + It("loads an ASR primary with no options", func() { + f := newFakeLib().withModel("asr.gguf", modelKindASR) + restore = f.install() + + p := &ParakeetCpp{} + Expect(p.Load(&pb.ModelOptions{ModelFile: "asr.gguf"})).To(Succeed()) + + Expect(p.ctxPtr).ToNot(BeZero()) + Expect(p.diarCtx).To(BeZero()) + Expect(p.tagCtx).To(BeZero()) + }) + + It("loads a diarization primary and rejects AudioTranscription without any further C call", func() { + f := newFakeLib().withModel("diar.gguf", modelKindDiarization) + restore = f.install() + + p := &ParakeetCpp{} + Expect(p.Load(&pb.ModelOptions{ModelFile: "diar.gguf"})).To(Succeed()) + Expect(p.diarCtx).ToNot(BeZero()) + Expect(p.ctxPtr).To(BeZero()) + + // CppTranscribePathJSON/CppTranscribePcmBatchJSON are left nil (the + // zero value): if AudioTranscription tried to call either, this would + // panic instead of returning cleanly, so a clean typed error here also + // proves no C call was made. + _, err := p.AudioTranscription(context.Background(), &pb.TranscriptRequest{Dst: "x.wav"}) + Expect(err).To(MatchError(ContainSubstring("diarization model"))) + }) + + It("loads a sound primary", func() { + f := newFakeLib().withModel("sound.gguf", modelKindSound) + restore = f.install() + + p := &ParakeetCpp{} + Expect(p.Load(&pb.ModelOptions{ModelFile: "sound.gguf"})).To(Succeed()) + + Expect(p.tagCtx).ToNot(BeZero()) + Expect(p.ctxPtr).To(BeZero()) + Expect(p.diarCtx).To(BeZero()) + }) + + It("loads an ASR primary plus diarization and sound companions, and Free releases all three", func() { + f := newFakeLib(). + withModel("asr.gguf", modelKindASR). + withModel("/models/x.gguf", modelKindDiarization). + withModel("/abs/y.gguf", modelKindSound) + restore = f.install() + + p := &ParakeetCpp{} + err := p.Load(&pb.ModelOptions{ + ModelFile: "asr.gguf", + ModelPath: "/models", + Options: []string{"diarization_model:x.gguf", "sound_model:/abs/y.gguf"}, + }) + Expect(err).ToNot(HaveOccurred()) + + Expect(f.loadedPaths).To(Equal([]string{"asr.gguf", "/models/x.gguf", "/abs/y.gguf"}), + "diarization_model resolves against ModelPath, sound_model's absolute path passes through") + Expect(p.ctxPtr).ToNot(BeZero()) + Expect(p.diarCtx).ToNot(BeZero()) + Expect(p.tagCtx).ToNot(BeZero()) + Expect(p.companions).To(HaveLen(2)) + + asrCtx, diarCtx, tagCtx := p.ctxPtr, p.diarCtx, p.tagCtx + Expect(p.Free()).To(Succeed()) + Expect(f.freed).To(ConsistOf(asrCtx, diarCtx, tagCtx)) + Expect(p.ctxPtr).To(BeZero()) + Expect(p.diarCtx).To(BeZero()) + Expect(p.tagCtx).To(BeZero()) + }) + + It("loads a diarization primary plus an asr_model companion into ctxPtr", func() { + f := newFakeLib(). + withModel("diar.gguf", modelKindDiarization). + withModel("/models/z.gguf", modelKindASR) + restore = f.install() + + p := &ParakeetCpp{} + err := p.Load(&pb.ModelOptions{ + ModelFile: "diar.gguf", + ModelPath: "/models", + Options: []string{"asr_model:z.gguf"}, + }) + Expect(err).ToNot(HaveOccurred()) + + Expect(p.diarCtx).ToNot(BeZero()) + Expect(p.ctxPtr).ToNot(BeZero()) + Expect(p.ctxPtr).ToNot(Equal(p.diarCtx)) + }) + + It("fails a companion of the wrong kind and frees every ctx it opened", func() { + f := newFakeLib(). + withModel("asr.gguf", modelKindASR). + withModel("/models/wrong.gguf", modelKindDiarization) + restore = f.install() + + p := &ParakeetCpp{} + err := p.Load(&pb.ModelOptions{ + ModelFile: "asr.gguf", + ModelPath: "/models", + Options: []string{"sound_model:wrong.gguf"}, + }) + Expect(err).To(HaveOccurred()) + Expect(err.Error()).To(ContainSubstring("sound_model")) + Expect(err.Error()).To(ContainSubstring("diarization")) + Expect(f.freed).To(HaveLen(2), "the primary and the wrong-kind companion must both be freed") + + // Every role field the failed load may have assigned (the primary + // lands in ctxPtr before the companion loop runs) must be reset, or a + // later Free() would double-free an already-freed context. + Expect(p.ctxPtr).To(BeZero()) + Expect(p.diarCtx).To(BeZero()) + Expect(p.tagCtx).To(BeZero()) + Expect(p.companions).To(BeEmpty()) + + freedBeforeFree := len(f.freed) + Expect(p.Free()).To(Succeed()) + Expect(f.freed).To(HaveLen(freedBeforeFree), "Free after a failed Load must not free anything again") + }) + + It("rejects an asr_model companion on an already-ASR primary and frees everything it opened", func() { + f := newFakeLib(). + withModel("asr.gguf", modelKindASR). + withModel("/models/other.gguf", modelKindASR) + restore = f.install() + + p := &ParakeetCpp{} + err := p.Load(&pb.ModelOptions{ + ModelFile: "asr.gguf", + ModelPath: "/models", + Options: []string{"asr_model:other.gguf"}, + }) + Expect(err).To(HaveOccurred()) + Expect(err.Error()).To(Equal(`parakeet-cpp: asr_model is not allowed on an ASR model`)) + Expect(f.loadedPaths).To(Equal([]string{"asr.gguf"})) + Expect(f.freed).To(HaveLen(1), "the primary must be freed too, or it leaks") + + Expect(p.ctxPtr).To(BeZero()) + Expect(p.diarCtx).To(BeZero()) + Expect(p.tagCtx).To(BeZero()) + Expect(p.companions).To(BeEmpty()) + }) + + It("rejects a diarization_model companion on an already-diarization primary", func() { + f := newFakeLib(). + withModel("diar.gguf", modelKindDiarization). + withModel("/models/other.gguf", modelKindDiarization) + restore = f.install() + + p := &ParakeetCpp{} + err := p.Load(&pb.ModelOptions{ + ModelFile: "diar.gguf", + ModelPath: "/models", + Options: []string{"diarization_model:other.gguf"}, + }) + Expect(err).To(HaveOccurred()) + Expect(err.Error()).To(Equal(`parakeet-cpp: diarization_model is not allowed on a diarization model`)) + Expect(f.loadedPaths).To(Equal([]string{"diar.gguf"})) + Expect(p.diarCtx).To(BeZero()) + }) + + It("rejects a sound_model companion on an already-sound (CED) primary", func() { + f := newFakeLib(). + withModel("sound.gguf", modelKindSound). + withModel("/models/other.gguf", modelKindSound) + restore = f.install() + + p := &ParakeetCpp{} + err := p.Load(&pb.ModelOptions{ + ModelFile: "sound.gguf", + ModelPath: "/models", + Options: []string{"sound_model:other.gguf"}, + }) + Expect(err).To(HaveOccurred()) + Expect(err.Error()).To(Equal(`parakeet-cpp: sound_model is not allowed on a sound model`)) + Expect(f.loadedPaths).To(Equal([]string{"sound.gguf"})) + Expect(p.tagCtx).To(BeZero()) + }) + + It("does not overwrite ctxPtr (and so does not leak the primary) when an asr_model companion "+ + "duplicates an ASR primary loaded alongside other companions", func() { + // Regression for the leak this whole check exists to close: before + // the fix, spec.assign(p, cctx) overwrote p.ctxPtr with the + // companion's ctx, and Free() (which only walks + // ctxPtr/diarCtx/tagCtx) never saw the original primary again. + f := newFakeLib(). + withModel("asr.gguf", modelKindASR). + withModel("/models/diar.gguf", modelKindDiarization). + withModel("/models/dup.gguf", modelKindASR) + restore = f.install() + + p := &ParakeetCpp{} + err := p.Load(&pb.ModelOptions{ + ModelFile: "asr.gguf", + ModelPath: "/models", + // diarization_model loads first (declared first in loadRoles' + // specs) and succeeds; asr_model then collides with the primary. + Options: []string{"diarization_model:diar.gguf", "asr_model:dup.gguf"}, + }) + Expect(err).To(HaveOccurred()) + Expect(f.freed).To(HaveLen(2), "the primary and the already-loaded diarization companion") + Expect(f.loadedPaths).To(Equal([]string{"asr.gguf", "/models/diar.gguf"}), + "dup.gguf must never be loaded: the role check runs before CppLoad") + + Expect(p.ctxPtr).To(BeZero()) + Expect(p.diarCtx).To(BeZero()) + Expect(p.companions).To(BeEmpty()) + }) + + It("treats a primary reporting PARAKEET_MODEL_KIND_NONE as ASR", func() { + f := newFakeLib().withModel("model.gguf", modelKindNone) + restore = f.install() + + p := &ParakeetCpp{} + Expect(p.Load(&pb.ModelOptions{ModelFile: "model.gguf"})).To(Succeed()) + + Expect(p.ctxPtr).ToNot(BeZero()) + Expect(p.diarCtx).To(BeZero()) + Expect(p.tagCtx).To(BeZero()) + }) + + It("parses diarization_latency:very_low", func() { + f := newFakeLib().withModel("diar.gguf", modelKindDiarization) + restore = f.install() + + p := &ParakeetCpp{} + err := p.Load(&pb.ModelOptions{ + ModelFile: "diar.gguf", + Options: []string{"diarization_latency:very_low"}, + }) + Expect(err).ToNot(HaveOccurred()) + Expect(p.diarLatency).To(Equal(int32(2))) + }) + + It("defaults diarization_latency to low (1) when unset", func() { + f := newFakeLib().withModel("diar.gguf", modelKindDiarization) + restore = f.install() + + p := &ParakeetCpp{} + Expect(p.Load(&pb.ModelOptions{ModelFile: "diar.gguf"})).To(Succeed()) + Expect(p.diarLatency).To(Equal(int32(1))) + }) + + It("rejects an invalid diarization_latency before any C call", func() { + f := newFakeLib().withModel("diar.gguf", modelKindDiarization) + restore = f.install() + + p := &ParakeetCpp{} + err := p.Load(&pb.ModelOptions{ + ModelFile: "diar.gguf", + Options: []string{"diarization_latency:bogus"}, + }) + Expect(err).To(HaveOccurred()) + Expect(f.loadedPaths).To(BeEmpty()) + }) + + Context("old library (no parakeet_capi_model_kind)", func() { + It("treats the primary as ASR, matching pre-v8 behavior", func() { + f := newFakeLib().withModel("model.gguf", modelKindDiarization) // kind is never consulted + restore = f.install() + CppModelKind = nil + + p := &ParakeetCpp{} + Expect(p.Load(&pb.ModelOptions{ModelFile: "model.gguf"})).To(Succeed()) + + Expect(p.ctxPtr).ToNot(BeZero()) + Expect(p.diarCtx).To(BeZero()) + Expect(p.tagCtx).To(BeZero()) + }) + + It("rejects companion model options with an error naming the library as too old", func() { + f := newFakeLib().withModel("asr.gguf", modelKindASR) + restore = f.install() + CppModelKind = nil + + p := &ParakeetCpp{} + err := p.Load(&pb.ModelOptions{ + ModelFile: "asr.gguf", + Options: []string{"diarization_model:diar.gguf"}, + }) + Expect(err).To(HaveOccurred()) + Expect(err.Error()).To(ContainSubstring("too old")) + Expect(f.loadedPaths).To(BeEmpty(), "rejected before any C call") + }) + }) + + Context("batcher gating", func() { + It("does not start the batcher for a non-ASR primary with no asr_model companion", func() { + f := newFakeLib().withModel("diar.gguf", modelKindDiarization) + restore = f.install() + savedBatch := CppTranscribePcmBatchJSON + defer func() { CppTranscribePcmBatchJSON = savedBatch }() + CppTranscribePcmBatchJSON = func(uintptr, []float32, []int32, int32, int32, int32) uintptr { return 0 } + + p := &ParakeetCpp{} + Expect(p.Load(&pb.ModelOptions{ModelFile: "diar.gguf"})).To(Succeed()) + Expect(p.bat).To(BeNil()) + }) + + It("starts the batcher for an ASR ctxPtr", func() { + f := newFakeLib().withModel("asr.gguf", modelKindASR) + restore = f.install() + savedBatch := CppTranscribePcmBatchJSON + defer func() { CppTranscribePcmBatchJSON = savedBatch }() + CppTranscribePcmBatchJSON = func(uintptr, []float32, []int32, int32, int32, int32) uintptr { return 0 } + + p := &ParakeetCpp{} + Expect(p.Load(&pb.ModelOptions{ModelFile: "asr.gguf"})).To(Succeed()) + Expect(p.bat).ToNot(BeNil()) + Expect(p.Free()).To(Succeed()) // stop the dispatcher goroutine + }) + }) +}) + +var _ = Describe("resolveModelPath", func() { + It("keeps an absolute path unchanged", func() { + Expect(resolveModelPath("/models", "/abs/y.gguf")).To(Equal("/abs/y.gguf")) + }) + + It("joins a relative path onto modelPath", func() { + Expect(resolveModelPath("/models", "x.gguf")).To(Equal("/models/x.gguf")) + }) + + It("passes a relative path through when modelPath is empty", func() { + Expect(resolveModelPath("", "x.gguf")).To(Equal("x.gguf")) + }) +}) + +var _ = Describe("parseDiarLatency", func() { + It("defaults to low (1) for an empty value", func() { + v, err := parseDiarLatency("") + Expect(err).ToNot(HaveOccurred()) + Expect(v).To(Equal(int32(1))) + }) + + It("maps every named mode", func() { + for s, want := range map[string]int32{ + "model": 0, "low": 1, "very_low": 2, "ultra_low": 3, + } { + v, err := parseDiarLatency(s) + Expect(err).ToNot(HaveOccurred()) + Expect(v).To(Equal(want), "mode %q", s) + } + }) + + It("rejects an unknown value", func() { + _, err := parseDiarLatency("bogus") + Expect(err).To(HaveOccurred()) + }) +}) diff --git a/backend/go/parakeet-cpp/scene.go b/backend/go/parakeet-cpp/scene.go new file mode 100644 index 000000000..590740f13 --- /dev/null +++ b/backend/go/parakeet-cpp/scene.go @@ -0,0 +1,258 @@ +package main + +import ( + "context" + "encoding/json" + "fmt" + "time" + + "github.com/mudler/LocalAI/pkg/grpc/grpcerrors" + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + "github.com/mudler/xlog" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" +) + +// sceneSpeakerJSON mirrors one element of a scene feed document's "speakers" +// array: {"speaker":0,"start":0.0,"end":0.6}. +type sceneSpeakerJSON struct { + Speaker int `json:"speaker"` + Start float64 `json:"start"` + End float64 `json:"end"` +} + +// sceneSoundJSON mirrors one element of a scene feed document's "sounds" +// array: {"index":99,"label":"Chicken, rooster","start":24.0,"end":30.0,"peak":0.86}. +type sceneSoundJSON struct { + Index int `json:"index"` + Label string `json:"label"` + Start float64 `json:"start"` + End float64 `json:"end"` + Peak float32 `json:"peak"` +} + +// sceneFeedJSON mirrors the subset of the document +// parakeet_capi_scene_stream_feed_json returns (docs/sound.md) that the live +// path consumes: the closed "speakers" and "sounds" arrays. "t", +// "utterances", "words" and "active" belong to an offline scene/SAS +// consumer, not the live path, and are not decoded here. +type sceneFeedJSON struct { + Speakers []sceneSpeakerJSON `json:"speakers"` + Sounds []sceneSoundJSON `json:"sounds"` +} + +// sceneWanted reports whether AudioTranscriptionLive should run a companion +// scene stream beside the ASR streaming session: at least one of the +// diarization/sound companions must be loaded, and the scene C-API symbols +// must be present. In practice the nil checks are defensive rather than +// live: loadRoles only ever sets diarCtx/tagCtx when parakeet_capi_model_kind +// (ABI v8) is present, and main.go registers every scene symbol in the same +// Dlsym-gated block as model_kind, so a companion being loaded already +// guarantees the scene symbols exist. +func (p *ParakeetCpp) sceneWanted() bool { + return (p.diarCtx != 0 || p.tagCtx != 0) && + CppSceneOptsDefault != nil && CppSceneStreamBegin != nil && + CppSceneStreamFeedJSON != nil && CppSceneStreamFree != nil +} + +// sceneStreamHandle bundles the C scene_stream pointer with the diar/tag +// contexts it was begun with. sceneFeed re-checks those against p.diarCtx/ +// p.tagCtx under engineMu before every call, so a Free() racing between the +// begin and a later feed (freeing the very contexts the stream borrows) is +// caught instead of handed to the C side — mirroring streamFeedDoc's re-check +// of p.ctxPtr (see the "Per-C-call engine serialization" comment in +// goparakeetcpp.go). The zero value (s == 0) means "no scene stream". +type sceneStreamHandle struct { + s uintptr + diar uintptr + tag uintptr +} + +// sceneBegin opens a no-ASR scene stream (diarization and/or sound events +// only; the live path's own ASR session already covers transcription) under +// engineMu. Call only when sceneWanted() is true. Refuses to begin with both +// contexts 0 (defensive: sceneWanted() already guards this). A zero handle +// means the C call itself failed; the caller logs a warning and continues +// the live session without speaker/sound events. +func (p *ParakeetCpp) sceneBegin() sceneStreamHandle { + p.engineMu.Lock() + defer p.engineMu.Unlock() + diar, tag := p.diarCtx, p.tagCtx + if diar == 0 && tag == 0 { + return sceneStreamHandle{} + } + var opts cSceneOpts + CppSceneOptsDefault(&opts) + opts.DiarLatency = p.diarLatency + // The live scene path never drains sound scores (unlike the offline + // SoundDetection RPC, see sound.go), so the default top_k of 5 would + // leave the C side's per-window score queue growing for the session's + // whole lifetime. 0 disables per-class score retention; sound EVENTS + // (onset/offset, what the live path actually consumes) are unaffected. + opts.Sound.TopK = 0 + s := CppSceneStreamBegin(0, diar, tag, &opts) + if s == 0 { + return sceneStreamHandle{} + } + return sceneStreamHandle{s: s, diar: diar, tag: tag} +} + +// sceneFree releases a scene stream opened by sceneBegin. A zero handle +// (scene events disabled or never began) is a no-op. Safe to call even after +// the contexts the stream borrowed have been freed: parakeet_scene_stream's +// destructor only releases its own buffers and never dereferences the +// borrowed asr/diar/tagger pointers (verified against +// parakeet.cpp's parakeet_capi_scene_stream_free / SceneStream::~SceneStream +// / DiarPcmStream::~DiarPcmStream, all `= default`), unlike a feed call. +func (p *ParakeetCpp) sceneFree(h sceneStreamHandle) { + if h.s == 0 { + return + } + p.engineMu.Lock() + defer p.engineMu.Unlock() + CppSceneStreamFree(h.s) +} + +// sceneFeed runs one scene-stream feed (or the is_last flush) under +// engineMu and returns the parsed document. Before touching the C side it +// re-checks that p.diarCtx/p.tagCtx still match what the stream was begun +// with: Free() can run between the caller's ASR feed and this call (both +// take engineMu individually, never for a session's lifetime, so nothing +// blocks a concurrent Free()) and free the very model the stream borrows. +// A mismatch returns ModelNotLoaded without making the C call; last_error is +// otherwise stream-scoped (parakeet_capi_scene_stream_last_error), read +// under the same lock as the failing call. +func (p *ParakeetCpp) sceneFeed(h sceneStreamHandle, pcm []float32, isLast bool) (sceneFeedJSON, error) { + p.engineMu.Lock() + defer p.engineMu.Unlock() + + if p.diarCtx != h.diar || p.tagCtx != h.tag { + return sceneFeedJSON{}, grpcerrors.ModelNotLoaded("parakeet-cpp") + } + + var last int32 + if isLast { + last = 1 + } + var ptr *float32 + if len(pcm) > 0 { + ptr = &pcm[0] + } + ret := CppSceneStreamFeedJSON(h.s, ptr, int32(len(pcm)), last) + if ret == 0 { + msg := "" + if CppSceneStreamLastError != nil { + msg = CppSceneStreamLastError(h.s) + } + if msg == "" { + msg = "unknown error" + } + return sceneFeedJSON{}, fmt.Errorf("parakeet-cpp: scene stream feed failed: %s", msg) + } + raw := goStringFromCPtr(ret) + CppFreeString(ret) + var doc sceneFeedJSON + if err := json.Unmarshal([]byte(raw), &doc); err != nil { + return sceneFeedJSON{}, fmt.Errorf("parakeet-cpp: decode scene json: %w", err) + } + return doc, nil +} + +// feedSlicesScene mirrors driver.go's feedSlices but also feeds the same pcm +// slice to an optional companion scene stream right after each ASR slice, so +// the live path's speaker/sound events stay time-aligned with the ASR decode +// increments. scene.s == 0 disables scene feeding for this call (no +// companions, or a previous scene feed already disabled it this session). +// +// The ASR result is emitted immediately after the ASR feed — the same +// response contents/timing a no-companion session would produce — before the +// scene feed for that slice runs, so a companion model never adds scene +// compute latency in front of the ASR delta/ that drives realtime turn +// detection. Any closed speakers/sounds from the scene feed are emitted +// afterward as their own response, so a slice with both produces two +// responses, ASR first. +// +// A scene feed failure degrades gracefully rather than aborting live +// transcription over a secondary feature: it frees the broken stream, warns +// once, and zeroes the handle so the caller carries the ASR-only session +// forward. Returns the (possibly now-zeroed) scene handle plus the +// cumulative ASR and scene wall time this call spent in feedChunk/sceneFeed, +// for the caller's lag log line. +func (p *ParakeetCpp) feedSlicesScene(ctx context.Context, stream uintptr, scene sceneStreamHandle, pcm []float32, onFeed func(streamFeedResult, sceneFeedJSON) error) (sceneStreamHandle, time.Duration, time.Duration, error) { + var asrWall, sceneWall time.Duration + for off := 0; off < len(pcm); off += streamChunkSamples { + if ctx != nil { + if err := ctx.Err(); err != nil { + return scene, asrWall, sceneWall, status.Error(codes.Canceled, "transcription cancelled") + } + } + end := min(off+streamChunkSamples, len(pcm)) + chunk := pcm[off:end] + + asrStart := time.Now() + res, err := p.feedChunk(stream, chunk, false) + asrWall += time.Since(asrStart) + if err != nil { + return scene, asrWall, sceneWall, err + } + if err := onFeed(res, sceneFeedJSON{}); err != nil { + return scene, asrWall, sceneWall, err + } + + if scene.s == 0 { + continue + } + sceneStart := time.Now() + sceneDoc, serr := p.sceneFeed(scene, chunk, false) + sceneWall += time.Since(sceneStart) + if serr != nil { + xlog.Warn("parakeet-cpp: live scene feed failed; disabling speaker/sound events for this session", + "err", serr) + p.sceneFree(scene) + scene = sceneStreamHandle{} + continue + } + if err := onFeed(streamFeedResult{}, sceneDoc); err != nil { + return scene, asrWall, sceneWall, err + } + } + return scene, asrWall, sceneWall, nil +} + +// liveSpeakersToProto maps a scene feed document's closed "speakers" into +// TranscriptLiveResponse.speakers (stream-relative nanoseconds). Reuses +// diarize.go's speakerLabel so the live path renders speaker indices the +// same way the offline Diarize RPC does. +func liveSpeakersToProto(speakers []sceneSpeakerJSON) []*pb.LiveSpeakerSegment { + if len(speakers) == 0 { + return nil + } + out := make([]*pb.LiveSpeakerSegment, len(speakers)) + for i, s := range speakers { + out[i] = &pb.LiveSpeakerSegment{ + Speaker: speakerLabel(s.Speaker), + Start: secondsToNanos(s.Start), + End: secondsToNanos(s.End), + } + } + return out +} + +// liveSoundsToProto maps a scene feed document's closed "sounds" into +// TranscriptLiveResponse.sounds (stream-relative nanoseconds). +func liveSoundsToProto(sounds []sceneSoundJSON) []*pb.LiveSoundEvent { + if len(sounds) == 0 { + return nil + } + out := make([]*pb.LiveSoundEvent, len(sounds)) + for i, s := range sounds { + out[i] = &pb.LiveSoundEvent{ + Label: s.Label, + Index: int32(s.Index), + Peak: s.Peak, + Start: secondsToNanos(s.Start), + End: secondsToNanos(s.End), + } + } + return out +} diff --git a/backend/go/parakeet-cpp/scene_test.go b/backend/go/parakeet-cpp/scene_test.go new file mode 100644 index 000000000..9dbfa2205 --- /dev/null +++ b/backend/go/parakeet-cpp/scene_test.go @@ -0,0 +1,42 @@ +package main + +import ( + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +// The sceneBegin spec drives it entirely against stubbed +// CppSceneOptsDefault/CppSceneStreamBegin (the same seam live_test.go uses +// for the full live-session scene specs), so it runs without libparakeet.so. + +var _ = Describe("ParakeetCpp.sceneBegin", func() { + It("forces Sound.TopK to 0 and carries p.diarLatency into the begin opts", func() { + savedOptsDefault := CppSceneOptsDefault + savedBegin := CppSceneStreamBegin + defer func() { + CppSceneOptsDefault = savedOptsDefault + CppSceneStreamBegin = savedBegin + }() + + // parakeet_capi_scene_opts_default's real default is top_k = 5 (a + // sound-window score history); simulate that here so the test proves + // sceneBegin overrides it rather than merely never setting it. + CppSceneOptsDefault = func(o *cSceneOpts) { + *o = cSceneOpts{Sound: cSoundOpts{TopK: 5}} + } + var gotOpts cSceneOpts + CppSceneStreamBegin = func(asr, diar, tagger uintptr, o *cSceneOpts) uintptr { + gotOpts = *o + return 1 + } + + p := &ParakeetCpp{diarCtx: 42, diarLatency: diarLatencyVeryLow} + h := p.sceneBegin() + Expect(h.s).ToNot(BeZero()) + Expect(gotOpts.Sound.TopK).To(Equal(int32(0)), + "the live scene path never drains sound scores (see sound.go's SoundDetection, "+ + "which does); a nonzero top_k leaves the C side's per-window score queue "+ + "growing for the session's lifetime") + Expect(gotOpts.DiarLatency).To(Equal(diarLatencyVeryLow)) + }) +}) diff --git a/backend/go/parakeet-cpp/sound.go b/backend/go/parakeet-cpp/sound.go new file mode 100644 index 000000000..1a3a358ad --- /dev/null +++ b/backend/go/parakeet-cpp/sound.go @@ -0,0 +1,261 @@ +package main + +import ( + "context" + "encoding/json" + "fmt" + "sort" + + "github.com/mudler/LocalAI/pkg/grpc/grpcerrors" + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" +) + +// soundFeedChunkSamples is how much 16 kHz mono PCM soundStreamScores hands +// to sound_stream_feed per call (10 s), matching the window/hop it asks for +// below. The clip is fed in these pieces with is_last set on the final one, +// mirroring the streaming ASR path's chunked feed. +const soundFeedChunkSamples = 10 * 16000 + +// soundTagJSON mirrors one element of a soundWindowJSON's "tags" array. +type soundTagJSON struct { + Index int `json:"index"` + Label string `json:"label"` + Score float32 `json:"score"` +} + +// soundWindowJSON mirrors one element of the array +// parakeet_capi_sound_stream_drain_scores_json returns: +// +// [{"start":0.0,"end":10.0,"tags":[{"index":0,"label":"Speech","score":0.93}, ...]}] +type soundWindowJSON struct { + Start float64 `json:"start"` + End float64 `json:"end"` + Tags []soundTagJSON `json:"tags"` +} + +// classAvg is one class's score averaged across the drained windows, plus +// the label the tagger reported for it. +type classAvg struct { + Index int + Label string + Score float32 +} + +// SoundDetection runs the loaded CED model (p.tagCtx) over the clip at +// req.Src through a one-shot sound stream (window 10 s, hop 10 s, top_k set +// to the tagger's full class count so every window's drain carries a score +// for every class), averages each class's score across the drained windows, +// sorts descending, applies req.Threshold, then req.TopK (0 = all classes). +func (p *ParakeetCpp) SoundDetection(ctx context.Context, req *pb.SoundDetectionRequest) (*pb.SoundDetectionResponse, error) { + if p.tagCtx == 0 { + return nil, status.Error(codes.FailedPrecondition, + "parakeet-cpp: model is not a sound (CED) model") + } + if CppSoundStreamBegin == nil || CppSoundStreamFeed == nil || CppSoundStreamDrainScoresJSON == nil || + CppSoundStreamFree == nil || CppSoundOptsDefault == nil || CppNumClasses == nil { + return nil, status.Error(codes.Unimplemented, + "parakeet-cpp: loaded libparakeet.so has no sound-event detection support "+ + "(parakeet_capi_sound_stream_* missing)") + } + if req.GetSrc() == "" { + return nil, status.Error(codes.InvalidArgument, + "parakeet-cpp: SoundDetectionRequest.src (audio path) is required") + } + + pcm, _, err := decodeWavMono16k(req.GetSrc()) + if err != nil { + return nil, status.Errorf(codes.InvalidArgument, "parakeet-cpp: decode audio: %s", err) + } + + windows, nClasses, err := p.soundStreamScores(ctx, pcm) + if err != nil { + return nil, err + } + + avgs := averageWindowScores(windows, nClasses) + sortSoundDetectionsDesc(avgs) + avgs = filterSoundDetections(avgs, req.GetThreshold(), req.GetTopK()) + + resp := &pb.SoundDetectionResponse{Detections: make([]*pb.SoundClass, 0, len(avgs))} + for _, a := range avgs { + resp.Detections = append(resp.Detections, &pb.SoundClass{ + Label: a.Label, + Score: a.Score, + Index: int32(a.Index), + }) + } + return resp, nil +} + +// soundStreamScores runs pcm through a fresh sound stream and returns the +// drained per-window scores plus the tagger's class count. The C calls run +// under engineMu (see soundStreamDrain); JSON decoding happens after the +// lock is released. +func (p *ParakeetCpp) soundStreamScores(ctx context.Context, pcm []float32) ([]soundWindowJSON, int, error) { + doc, nClasses, err := p.soundStreamDrain(ctx, pcm) + if err != nil { + return nil, nClasses, err + } + + var windows []soundWindowJSON + if err := json.Unmarshal([]byte(doc), &windows); err != nil { + return nil, nClasses, fmt.Errorf("parakeet-cpp: decode sound scores json: %w", err) + } + return windows, nClasses, nil +} + +// soundStreamDrain runs pcm through a fresh sound stream and returns the +// raw JSON document parakeet_capi_sound_stream_drain_scores_json drained, +// plus the tagger's class count. Every C call (opts default, begin, feed, +// free, drain) runs under engineMu; the stream is freed (deferred right +// after a successful begin) even when a later feed or drain call fails, or +// ctx is cancelled mid-feed. Each feed's returned segments array is freed +// with parakeet_capi_free_sound_segments even though SoundDetection has no +// use for the segments themselves (it only reads the drained window +// scores). ctx.Err() is checked before each feed slice, mirroring +// driver.go's feedSlices, so a long clip can be cancelled mid-feed; the +// caller decodes the returned JSON outside the lock. +func (p *ParakeetCpp) soundStreamDrain(ctx context.Context, pcm []float32) (string, int, error) { + p.engineMu.Lock() + defer p.engineMu.Unlock() + + // SoundDetection's own p.tagCtx==0 check runs before this lock is taken; + // re-check here so a Free() racing in between (which zeroes p.tagCtx + // under this same engineMu) is caught instead of handed to the C side, + // mirroring streamFeedDoc's/sceneFeed's re-check. + if p.tagCtx == 0 { + return "", 0, grpcerrors.ModelNotLoaded("parakeet-cpp") + } + + nClasses := int(CppNumClasses(p.tagCtx)) + + var opts cSoundOpts + CppSoundOptsDefault(&opts) + opts.WindowSec = 10 + opts.HopSec = 10 + opts.TopK = int32(nClasses) + + stream := CppSoundStreamBegin(p.tagCtx, &opts) + if stream == 0 { + return "", nClasses, fmt.Errorf("parakeet-cpp: sound_stream_begin failed: %s", soundLastError(p.tagCtx)) + } + defer CppSoundStreamFree(stream) + + offset := 0 + for { + if ctx != nil { + if err := ctx.Err(); err != nil { + return "", nClasses, status.Error(codes.Canceled, "parakeet-cpp: sound detection cancelled") + } + } + + end := offset + soundFeedChunkSamples + isLast := int32(0) + if end >= len(pcm) { + end = len(pcm) + isLast = 1 + } + var samplePtr *float32 + if end > offset { + samplePtr = &pcm[offset] + } + + var segsOut uintptr + var nOut int32 + rc := CppSoundStreamFeed(stream, samplePtr, int32(end-offset), isLast, &segsOut, &nOut) + if segsOut != 0 && CppFreeSoundSegments != nil { + CppFreeSoundSegments(segsOut) + } + if rc != 0 { + return "", nClasses, fmt.Errorf("parakeet-cpp: sound_stream_feed failed: %s", soundLastError(p.tagCtx)) + } + + offset = end + if isLast == 1 { + break + } + } + + raw := CppSoundStreamDrainScoresJSON(stream) + if raw == 0 { + return "", nClasses, fmt.Errorf("parakeet-cpp: sound_stream_drain_scores_json failed: %s", soundLastError(p.tagCtx)) + } + doc := goStringFromCPtr(raw) + CppFreeString(raw) + return doc, nClasses, nil +} + +// soundLastError reads ctx's last_error, substituting a fallback message +// when the C side left it empty. +func soundLastError(ctx uintptr) string { + msg := CppLastError(ctx) + if msg == "" { + msg = "unknown error" + } + return msg +} + +// averageWindowScores averages each class's score across the drained +// per-window scores: CED's own long-clip method, summing a class's score +// over every window and dividing by the window count (a class absent from a +// window's tags counts as 0 in that window). Only classes that appeared in +// at least one window are returned, in no particular order; callers sort and +// filter afterward. A pure function so it is easy to unit test in isolation +// from the C stream. +func averageWindowScores(windows []soundWindowJSON, nClasses int) []classAvg { + if len(windows) == 0 { + return nil + } + + sums := make(map[int]float32) + labels := make(map[int]string) + for _, w := range windows { + for _, t := range w.Tags { + if t.Index < 0 || (nClasses > 0 && t.Index >= nClasses) { + continue + } + sums[t.Index] += t.Score + if _, ok := labels[t.Index]; !ok { + labels[t.Index] = t.Label + } + } + } + + n := float32(len(windows)) + out := make([]classAvg, 0, len(sums)) + for idx, sum := range sums { + out = append(out, classAvg{Index: idx, Label: labels[idx], Score: sum / n}) + } + return out +} + +// sortSoundDetectionsDesc sorts avgs by score descending, breaking ties by +// class index for a deterministic order (map iteration in +// averageWindowScores is otherwise unordered). +func sortSoundDetectionsDesc(avgs []classAvg) { + sort.Slice(avgs, func(i, j int) bool { + if avgs[i].Score != avgs[j].Score { + return avgs[i].Score > avgs[j].Score + } + return avgs[i].Index < avgs[j].Index + }) +} + +// filterSoundDetections drops entries scoring below threshold, then keeps +// only the first topK entries (0 = keep all). avgs is assumed already sorted +// descending by score. +func filterSoundDetections(avgs []classAvg, threshold float32, topK int32) []classAvg { + out := avgs[:0:0] + for _, a := range avgs { + if a.Score < threshold { + continue + } + out = append(out, a) + } + if topK > 0 && int32(len(out)) > topK { + out = out[:topK] + } + return out +} diff --git a/backend/go/parakeet-cpp/sound_test.go b/backend/go/parakeet-cpp/sound_test.go new file mode 100644 index 000000000..1162a84ac --- /dev/null +++ b/backend/go/parakeet-cpp/sound_test.go @@ -0,0 +1,383 @@ +package main + +import ( + "context" + "path/filepath" + "sync" + "unsafe" + + "github.com/mudler/LocalAI/pkg/grpc/grpcerrors" + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" +) + +// The SoundDetection specs drive it entirely against stubbed +// CppSoundStreamBegin / CppSoundStreamFeed / CppSoundStreamDrainScoresJSON / +// CppSoundStreamFree / CppFreeSoundSegments / CppSoundOptsDefault / +// CppNumClasses / CppFreeString / CppLastError (the same seam diarize_test.go +// and live_test.go use), so they run without libparakeet.so. + +// soundCstrPool hands out NUL-terminated C-style strings backed by Go memory +// and keeps them alive for the duration of a spec (goStringFromCPtr reads +// through the raw pointer; mirrors diarize_test.go's diarizeCstrPool). +type soundCstrPool struct { + mu sync.Mutex + bufs [][]byte +} + +func (p *soundCstrPool) cstr(s string) uintptr { + p.mu.Lock() + defer p.mu.Unlock() + b := append([]byte(s), 0) + p.bufs = append(p.bufs, b) + return uintptr(unsafe.Pointer(&b[0])) +} + +// soundStubs swaps every C entry point SoundDetection touches and returns a +// restore func for AfterEach (mirrors diarize_test.go's diarizeStubs). +func soundStubs() (restore func()) { + savedBegin := CppSoundStreamBegin + savedFeed := CppSoundStreamFeed + savedDrain := CppSoundStreamDrainScoresJSON + savedFree := CppSoundStreamFree + savedFreeSegs := CppFreeSoundSegments + savedOptsDefault := CppSoundOptsDefault + savedNumClasses := CppNumClasses + savedFreeString := CppFreeString + savedLastError := CppLastError + return func() { + CppSoundStreamBegin = savedBegin + CppSoundStreamFeed = savedFeed + CppSoundStreamDrainScoresJSON = savedDrain + CppSoundStreamFree = savedFree + CppFreeSoundSegments = savedFreeSegs + CppSoundOptsDefault = savedOptsDefault + CppNumClasses = savedNumClasses + CppFreeString = savedFreeString + CppLastError = savedLastError + } +} + +// soundWav writes a silent 16 kHz mono WAV of the given duration (seconds) to +// a fresh temp file and returns its path. decodeWavMono16k reads real audio +// bytes off disk, so SoundDetection needs a file on disk even though the +// stubbed C calls never look at its samples. +func soundWav(seconds float64) string { + GinkgoHelper() + path := filepath.Join(GinkgoT().TempDir(), "sound.wav") + writeMono16kWav(path, int(seconds*16000)) + return path +} + +// noopFeed is a CppSoundStreamFeed stub that always succeeds and returns no +// segments, for specs that only care about the drained window scores. +func noopFeed(s uintptr, pcm *float32, n int32, isLast int32, out *uintptr, nOut *int32) int32 { + *out = 0 + *nOut = 0 + return 0 +} + +var _ = Describe("ParakeetCpp.SoundDetection", func() { + var restore func() + var pool *soundCstrPool + + BeforeEach(func() { + restore = soundStubs() + pool = &soundCstrPool{} + CppFreeString = func(uintptr) {} + CppFreeSoundSegments = func(uintptr) {} + CppSoundOptsDefault = func(o *cSoundOpts) { *o = cSoundOpts{} } + }) + AfterEach(func() { restore() }) + + It("fails with FailedPrecondition when no sound model is loaded", func() { + p := &ParakeetCpp{} + _, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: soundWav(1)}) + Expect(err).To(HaveOccurred()) + Expect(status.Code(err)).To(Equal(codes.FailedPrecondition)) + Expect(err.Error()).To(ContainSubstring("model is not a sound")) + }) + + It("fails with Unimplemented when the loaded libparakeet.so has no sound_stream_begin symbol", func() { + CppSoundStreamBegin = nil + p := &ParakeetCpp{tagCtx: 42} + _, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: soundWav(1)}) + Expect(err).To(HaveOccurred()) + Expect(status.Code(err)).To(Equal(codes.Unimplemented)) + }) + + It("averages two windows per class and sorts descending", func() { + CppNumClasses = func(uintptr) int32 { return 2 } + CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 } + CppSoundStreamFeed = noopFeed + CppSoundStreamFree = func(uintptr) {} + CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { + return pool.cstr(`[` + + `{"start":0,"end":10,"tags":[{"index":0,"label":"Speech","score":0.8},{"index":1,"label":"Music","score":0.2}]},` + + `{"start":10,"end":20,"tags":[{"index":0,"label":"Speech","score":0.4},{"index":1,"label":"Music","score":0.6}]}` + + `]`) + } + + p := &ParakeetCpp{tagCtx: 42} + resp, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: soundWav(1)}) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.Detections).To(HaveLen(2)) + Expect(resp.Detections[0].Label).To(Equal("Speech")) + Expect(resp.Detections[0].Score).To(BeNumerically("~", 0.6, 1e-6)) + Expect(resp.Detections[1].Label).To(Equal("Music")) + Expect(resp.Detections[1].Score).To(BeNumerically("~", 0.4, 1e-6)) + }) + + It("drops classes scoring below threshold", func() { + CppNumClasses = func(uintptr) int32 { return 2 } + CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 } + CppSoundStreamFeed = noopFeed + CppSoundStreamFree = func(uintptr) {} + CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { + return pool.cstr(`[{"start":0,"end":10,"tags":[` + + `{"index":0,"label":"Speech","score":0.8},` + + `{"index":1,"label":"Music","score":0.2}]}]`) + } + + p := &ParakeetCpp{tagCtx: 42} + resp, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{ + Src: soundWav(1), Threshold: 0.5, + }) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.Detections).To(HaveLen(1)) + Expect(resp.Detections[0].Label).To(Equal("Speech")) + }) + + It("keeps only the top_k entries", func() { + CppNumClasses = func(uintptr) int32 { return 4 } + CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 } + CppSoundStreamFeed = noopFeed + CppSoundStreamFree = func(uintptr) {} + CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { + return pool.cstr(`[{"start":0,"end":10,"tags":[` + + `{"index":0,"label":"A","score":0.9},` + + `{"index":1,"label":"B","score":0.7},` + + `{"index":2,"label":"C","score":0.5},` + + `{"index":3,"label":"D","score":0.3}]}]`) + } + + p := &ParakeetCpp{tagCtx: 42} + resp, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{ + Src: soundWav(1), TopK: 3, + }) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.Detections).To(HaveLen(3)) + Expect(resp.Detections[0].Label).To(Equal("A")) + Expect(resp.Detections[1].Label).To(Equal("B")) + Expect(resp.Detections[2].Label).To(Equal("C")) + }) + + It("keeps all classes when top_k is 0", func() { + CppNumClasses = func(uintptr) int32 { return 4 } + CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 } + CppSoundStreamFeed = noopFeed + CppSoundStreamFree = func(uintptr) {} + CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { + return pool.cstr(`[{"start":0,"end":10,"tags":[` + + `{"index":0,"label":"A","score":0.9},` + + `{"index":1,"label":"B","score":0.7},` + + `{"index":2,"label":"C","score":0.5},` + + `{"index":3,"label":"D","score":0.3}]}]`) + } + + p := &ParakeetCpp{tagCtx: 42} + resp, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{ + Src: soundWav(1), TopK: 0, + }) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.Detections).To(HaveLen(4)) + }) + + It("passes window 10s, hop 10s and top_k = the tagger's class count to sound_stream_begin", func() { + CppNumClasses = func(uintptr) int32 { return 527 } + var gotOpts cSoundOpts + CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { + gotOpts = *o + return 1 + } + CppSoundStreamFeed = noopFeed + CppSoundStreamFree = func(uintptr) {} + CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { return pool.cstr(`[]`) } + + p := &ParakeetCpp{tagCtx: 42} + _, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: soundWav(1)}) + Expect(err).ToNot(HaveOccurred()) + Expect(gotOpts.WindowSec).To(BeNumerically("==", 10)) + Expect(gotOpts.HopSec).To(BeNumerically("==", 10)) + Expect(gotOpts.TopK).To(Equal(int32(527))) + }) + + It("returns no detections without error for a short clip whose drain is empty", func() { + CppNumClasses = func(uintptr) int32 { return 2 } + CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 } + CppSoundStreamFeed = noopFeed + CppSoundStreamFree = func(uintptr) {} + CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { return pool.cstr(`[]`) } + + p := &ParakeetCpp{tagCtx: 42} + resp, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: soundWav(0.1)}) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.Detections).To(BeEmpty()) + }) + + It("surfaces last_error and still frees the stream when feed fails", func() { + freed := false + CppNumClasses = func(uintptr) int32 { return 2 } + CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 } + CppSoundStreamFeed = func(s uintptr, pcm *float32, n int32, isLast int32, out *uintptr, nOut *int32) int32 { + *out = 0 + *nOut = 0 + return 1 + } + CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { + Fail("drain_scores_json must not be called when feed failed") + return 0 + } + CppSoundStreamFree = func(uintptr) { freed = true } + CppLastError = func(uintptr) string { return "boom" } + + p := &ParakeetCpp{tagCtx: 42} + _, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: soundWav(1)}) + Expect(err).To(HaveOccurred()) + Expect(err.Error()).To(ContainSubstring("boom")) + Expect(freed).To(BeTrue()) + }) + + It("returns Canceled without feeding when ctx is already cancelled, and frees the stream", func() { + freed := false + ctx, cancel := context.WithCancel(context.Background()) + cancel() + + CppNumClasses = func(uintptr) int32 { return 2 } + CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 } + CppSoundStreamFeed = func(s uintptr, pcm *float32, n int32, isLast int32, out *uintptr, nOut *int32) int32 { + Fail("sound_stream_feed must not be called when ctx is already cancelled") + return 0 + } + CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { + Fail("drain_scores_json must not be called when ctx is already cancelled") + return 0 + } + CppSoundStreamFree = func(uintptr) { freed = true } + + p := &ParakeetCpp{tagCtx: 42} + _, err := p.SoundDetection(ctx, &pb.SoundDetectionRequest{Src: soundWav(15)}) + Expect(err).To(HaveOccurred()) + Expect(status.Code(err)).To(Equal(codes.Canceled)) + Expect(freed).To(BeTrue()) + }) + + It("stops feeding and frees the stream when ctx is cancelled mid-feed", func() { + freed := false + feedCount := 0 + ctx, cancel := context.WithCancel(context.Background()) + + CppNumClasses = func(uintptr) int32 { return 2 } + CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 } + CppSoundStreamFeed = func(s uintptr, pcm *float32, n int32, isLast int32, out *uintptr, nOut *int32) int32 { + feedCount++ + cancel() // cancel after the first feed so a second chunk would exist if not stopped + *out = 0 + *nOut = 0 + return 0 + } + CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { + Fail("drain_scores_json must not be called when the feed loop was cancelled") + return 0 + } + CppSoundStreamFree = func(uintptr) { freed = true } + + // Two 10 s chunks, so a second feed call would happen without the cancel. + p := &ParakeetCpp{tagCtx: 42} + _, err := p.SoundDetection(ctx, &pb.SoundDetectionRequest{Src: soundWav(15)}) + Expect(err).To(HaveOccurred()) + Expect(status.Code(err)).To(Equal(codes.Canceled)) + Expect(feedCount).To(Equal(1)) + Expect(freed).To(BeTrue()) + }) + + It("wraps a decode failure as InvalidArgument", func() { + // Every required symbol must be non-nil to clear SoundDetection's own + // Unimplemented gate and reach the decode step this spec targets; + // none of them may actually be called. + fail := func(string) { Fail("no C call once the decode itself has failed") } + CppNumClasses = func(uintptr) int32 { fail("num_classes"); return 0 } + CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { fail("begin"); return 0 } + CppSoundStreamFeed = func(s uintptr, pcm *float32, n int32, isLast int32, out *uintptr, nOut *int32) int32 { + fail("feed") + return 0 + } + CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { fail("drain"); return 0 } + CppSoundStreamFree = func(uintptr) { fail("free") } + + p := &ParakeetCpp{tagCtx: 42} + _, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{ + Src: filepath.Join(GinkgoT().TempDir(), "missing.wav"), + }) + Expect(err).To(HaveOccurred()) + Expect(status.Code(err)).To(Equal(codes.InvalidArgument)) + }) + + It("returns ModelNotLoaded without a C call when tagCtx is zeroed between the entry check and the call", func() { + called := false + CppNumClasses = func(uintptr) int32 { called = true; return 2 } + + p := &ParakeetCpp{tagCtx: 42} + // Simulate a Free() racing between SoundDetection's own tagCtx==0 + // check and soundStreamDrain's lock, exactly as it zeroes tagCtx + // under engineMu. + p.tagCtx = 0 + _, _, err := p.soundStreamDrain(context.Background(), make([]float32, 10)) + Expect(grpcerrors.IsModelNotLoaded(err)).To(BeTrue()) + Expect(called).To(BeFalse(), "no C call once tagCtx was cleared") + }) +}) + +var _ = Describe("averageWindowScores", func() { + It("returns nil for no windows", func() { + Expect(averageWindowScores(nil, 2)).To(BeNil()) + }) + + It("treats a class absent from a window as 0 in that window's contribution", func() { + windows := []soundWindowJSON{ + {Tags: []soundTagJSON{{Index: 0, Label: "Speech", Score: 1.0}}}, + {Tags: []soundTagJSON{}}, // Speech absent this window + } + out := averageWindowScores(windows, 2) + Expect(out).To(HaveLen(1)) + Expect(out[0].Index).To(Equal(0)) + Expect(out[0].Score).To(BeNumerically("~", 0.5, 1e-6)) + }) + + It("ignores an out-of-range class index", func() { + windows := []soundWindowJSON{ + {Tags: []soundTagJSON{{Index: 5, Label: "Bogus", Score: 1.0}}}, + } + Expect(averageWindowScores(windows, 2)).To(BeEmpty()) + }) +}) + +var _ = Describe("filterSoundDetections", func() { + It("keeps everything when top_k is 0 and threshold is 0", func() { + in := []classAvg{{Index: 0, Score: 0.1}, {Index: 1, Score: 0.9}} + Expect(filterSoundDetections(in, 0, 0)).To(HaveLen(2)) + }) + + It("drops entries below threshold before applying top_k", func() { + in := []classAvg{ + {Index: 0, Score: 0.9}, + {Index: 1, Score: 0.4}, + {Index: 2, Score: 0.1}, + } + out := filterSoundDetections(in, 0.3, 5) + Expect(out).To(HaveLen(2)) + }) +}) diff --git a/backend/go/parakeet-cpp/speakers.go b/backend/go/parakeet-cpp/speakers.go new file mode 100644 index 000000000..ddd2c4292 --- /dev/null +++ b/backend/go/parakeet-cpp/speakers.go @@ -0,0 +1,121 @@ +package main + +import ( + "encoding/json" + "fmt" + "strconv" +) + +// Speaker labels on transcripts. With a diarization_model companion attached +// to an ASR model, unary transcription tags each segment (and, with word +// timestamps, each word) with its speaker, and the stream=true final result +// tags each utterance. Live transcription carries speakers through the scene +// stream instead (scene.go). + +// speakerSnapSeconds mirrors parakeet.cpp's merge_asr_diarization: a word that +// overlaps no speaker segment takes the nearest segment's speaker when that +// segment is this close. ASR word boundaries and diarization boundaries can +// disagree by a frame or two; a word farther than this has no speaker. +const speakerSnapSeconds = 0.5 + +// transcriptSpeaker renders a 0-based speaker for a transcript segment or +// word; -1 (no speaker) is left empty so the field is omitted. +func transcriptSpeaker(spk int) string { + if spk < 0 { + return "" + } + return strconv.Itoa(spk) +} + +// wantSpeakers reports whether a transcription should carry speaker labels: +// a diarization companion is attached, the library can diarize, and the +// request did not turn it off (the OpenAI endpoint sends diarize=true unless +// the client passes diarize=false). +func (p *ParakeetCpp) wantSpeakers(diarize bool) bool { + return diarize && p.ctxPtr != 0 && p.diarCtx != 0 && CppDiarizePCM != nil +} + +// diarizeSegmentsPCM runs the diarization companion over 16 kHz PCM +// (parakeet_capi_diarize_pcm, the checkpoint's own mode, as NeMo's +// diarize()) and returns its segments. +func (p *ParakeetCpp) diarizeSegmentsPCM(pcm []float32) ([]diarizeSegmentJSON, error) { + if len(pcm) == 0 { + return nil, nil + } + raw, err := p.diarizeCall(pcm, false) + if err != nil { + return nil, err + } + var doc diarizePCMDoc + if err := json.Unmarshal([]byte(raw), &doc); err != nil { + return nil, fmt.Errorf("parakeet-cpp: decode diarization json: %w", err) + } + return doc.Segments, nil +} + +// assignSpeakers gives each word the speaker whose segments overlap it most, +// falling back to the nearest segment within speakerSnapSeconds; -1 when none. +// Same rule as parakeet.cpp's merge_asr_diarization, so the labels match the +// library's own speaker-attributed ASR. +func assignSpeakers(words []transcriptWord, segs []diarizeSegmentJSON) []int { + out := make([]int, len(words)) + for i, w := range words { + best, bestOverlap := -1, 0.0 + for _, s := range segs { + if ov := min(w.End, s.End) - max(w.Start, s.Start); ov > bestOverlap { + best, bestOverlap = s.Speaker, ov + } + } + if best < 0 { + bestDist := speakerSnapSeconds + for _, s := range segs { + dist := s.Start - w.End + if s.End <= w.Start { + dist = w.Start - s.End + } + if dist >= 0 && dist <= bestDist { + best, bestDist = s.Speaker, dist + } + } + } + out[i] = best + } + return out +} + +// splitAtSpeakerChanges splits each word group wherever the speaker changes, +// so every segment has one speaker. speakers is indexed like the +// concatenation of groups. Returns the new groups and each group's speaker. +func splitAtSpeakerChanges(groups [][]transcriptWord, speakers []int) ([][]transcriptWord, []int) { + var outGroups [][]transcriptWord + var outSpk []int + k := 0 + for _, g := range groups { + start := 0 + for i := 1; i <= len(g); i++ { + if i == len(g) || speakers[k+i] != speakers[k+start] { + outGroups = append(outGroups, g[start:i]) + outSpk = append(outSpk, speakers[k+start]) + start = i + } + } + k += len(g) + } + return outGroups, outSpk +} + +// majoritySpeaker is the speaker covering most of the words' duration, or -1. +func majoritySpeaker(words []transcriptWord, speakers []int) int { + dur := map[int]float64{} + best, bestDur := -1, 0.0 + for i, w := range words { + if speakers[i] < 0 { + continue + } + dur[speakers[i]] += max(w.End-w.Start, 1e-3) + if dur[speakers[i]] > bestDur { + best, bestDur = speakers[i], dur[speakers[i]] + } + } + return best +} diff --git a/backend/go/parakeet-cpp/speakers_test.go b/backend/go/parakeet-cpp/speakers_test.go new file mode 100644 index 000000000..b302bb6e9 --- /dev/null +++ b/backend/go/parakeet-cpp/speakers_test.go @@ -0,0 +1,166 @@ +package main + +import ( + "context" + "os" + + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +func sseg(spk int, start, end float64) diarizeSegmentJSON { + return diarizeSegmentJSON{Speaker: spk, Start: start, End: end} +} + +// speakerTurnsOf collapses consecutive repeats: [0 0 1 0] -> [0 1 0]. +func speakerTurnsOf(labels []string) []string { + var out []string + for _, l := range labels { + if len(out) == 0 || out[len(out)-1] != l { + out = append(out, l) + } + } + return out +} + +var _ = Describe("transcript speaker labels", func() { + Context("assignSpeakers", func() { + It("picks the speaker with the largest overlap", func() { + words := []transcriptWord{tw("a", 0.5, 0.8)} + Expect(assignSpeakers(words, []diarizeSegmentJSON{sseg(0, 0.0, 0.6), sseg(1, 0.55, 2.0)})). + To(Equal([]int{1})) + }) + + It("snaps a word just outside a segment to it, but not a far one", func() { + words := []transcriptWord{tw("well", 19.92, 20.00), tw("far", 25.0, 25.2)} + segs := []diarizeSegmentJSON{sseg(0, 14.78, 18.75), sseg(1, 20.10, 23.60)} + Expect(assignSpeakers(words, segs)).To(Equal([]int{1, -1})) + }) + }) + + It("splits segments at speaker turns and labels segments and words", func() { + doc := transcriptJSON{ + Text: "hi there. hello", + Words: []transcriptWord{tw("hi", 0.0, 0.3), tw("there.", 0.3, 0.6), tw("hello", 1.0, 1.4)}, + } + // Punctuation gives [hi there.] [hello]; the turn after "hi" splits the first. + opts := &pb.TranscriptRequest{TimestampGranularities: []string{"word"}} + res := transcriptResultWithSpeakers(doc, opts, 0, []int{0, 1, 1}) + Expect(res.Segments).To(HaveLen(3)) + Expect([]string{res.Segments[0].Speaker, res.Segments[1].Speaker, res.Segments[2].Speaker}). + To(Equal([]string{"0", "1", "1"})) + Expect(res.Segments[1].Text).To(Equal("there.")) + Expect(res.Segments[2].Words[0].Speaker).To(Equal("1")) + for i, seg := range res.Segments { + Expect(seg.Id).To(Equal(int32(i))) + } + }) + + It("leaves speakers empty without diarization and for unknown words", func() { + doc := transcriptJSON{Text: "hi.", Words: []transcriptWord{tw("hi.", 0, 0.3)}} + Expect(transcriptResultFromDoc(doc, &pb.TranscriptRequest{}, 0).Segments[0].Speaker).To(BeEmpty()) + Expect(transcriptResultWithSpeakers(doc, &pb.TranscriptRequest{}, 0, []int{-1}).Segments[0].Speaker). + To(BeEmpty()) + }) + + It("picks the speaker covering most of an utterance", func() { + words := []transcriptWord{tw("a", 0, 0.2), tw("b", 0.2, 1.5), tw("c", 1.5, 1.6)} + Expect(majoritySpeaker(words, []int{0, 1, 0})).To(Equal(1)) + Expect(majoritySpeaker(words, []int{-1, -1, -1})).To(Equal(-1)) + }) + + It("only diarizes with a companion, a capable library and diarize=true", func() { + saved := CppDiarizePCM + defer func() { CppDiarizePCM = saved }() + CppDiarizePCM = func(uintptr, *float32, int32, int32) uintptr { return 0 } + p := &ParakeetCpp{ctxPtr: 1, diarCtx: 2} + Expect(p.wantSpeakers(true)).To(BeTrue()) + Expect(p.wantSpeakers(false)).To(BeFalse()) + Expect((&ParakeetCpp{ctxPtr: 1}).wantSpeakers(true)).To(BeFalse()) + CppDiarizePCM = nil + Expect(p.wantSpeakers(true)).To(BeFalse()) + }) +}) + +var _ = Describe("ParakeetCpp transcript speakers (real models)", func() { + It("labels unary and stream=true transcripts with a diarization_model companion", func() { + diarModel := os.Getenv("PARAKEET_BACKEND_TEST_DIAR_MODEL") + asrModel := os.Getenv("PARAKEET_BACKEND_TEST_MODEL") + wavPath := os.Getenv("PARAKEET_BACKEND_TEST_DIAR_WAV") + if diarModel == "" || asrModel == "" || wavPath == "" { + Skip("set PARAKEET_BACKEND_TEST_DIAR_MODEL, PARAKEET_BACKEND_TEST_MODEL and " + + "PARAKEET_BACKEND_TEST_DIAR_WAV (a multi-speaker 16 kHz WAV)") + } + ensureLibLoaded() + if CppDiarizePCM == nil || CppModelKind == nil { + Skip("libparakeet.so has no diarization / model-kind C-API") + } + p := &ParakeetCpp{} + Expect(p.Load(&pb.ModelOptions{ + ModelFile: asrModel, + Options: []string{"diarization_model:" + diarModel}, + })).To(Succeed()) + defer func() { _ = p.Free() }() + + res, err := p.AudioTranscription(context.Background(), &pb.TranscriptRequest{Dst: wavPath, Diarize: true}) + Expect(err).ToNot(HaveOccurred()) + labels := make([]string, len(res.Segments)) + for i, s := range res.Segments { + Expect(s.Speaker).ToNot(BeEmpty(), "segment %d %q", i, s.Text) + labels[i] = s.Speaker + } + // parakeet.cpp's tests/fixtures/two_speakers.wav alternates A-B-A-B. + Expect(speakerTurnsOf(labels)).To(Equal([]string{"0", "1", "0", "1"})) + + plain, err := p.AudioTranscription(context.Background(), &pb.TranscriptRequest{Dst: wavPath}) + Expect(err).ToNot(HaveOccurred()) + for _, s := range plain.Segments { + Expect(s.Speaker).To(BeEmpty()) + } + Expect(plain.Text).To(Equal(res.Text)) + }) +}) + +var _ = Describe("ParakeetCpp scene companions on an offline ASR model (real models)", func() { + It("labels speakers and detects sounds from one model with both companions", func() { + asrModel := os.Getenv("PARAKEET_BACKEND_TEST_SCENE_ASR_MODEL") // e.g. tdt-0.6b-v3 + diarModel := os.Getenv("PARAKEET_BACKEND_TEST_DIAR_MODEL") + soundModel := os.Getenv("PARAKEET_BACKEND_TEST_SOUND_MODEL") + wavPath := os.Getenv("PARAKEET_BACKEND_TEST_SCENE_WAV") // speech + a non-speech sound + wantLabel := os.Getenv("PARAKEET_BACKEND_TEST_SCENE_LABEL") // e.g. "Chicken, rooster" + if asrModel == "" || diarModel == "" || soundModel == "" || wavPath == "" || wantLabel == "" { + Skip("set PARAKEET_BACKEND_TEST_SCENE_ASR_MODEL, _DIAR_MODEL, _SOUND_MODEL, " + + "PARAKEET_BACKEND_TEST_SCENE_WAV and PARAKEET_BACKEND_TEST_SCENE_LABEL") + } + ensureLibLoaded() + if CppDiarizePCM == nil || CppModelKind == nil { + Skip("libparakeet.so has no diarization / model-kind C-API") + } + p := &ParakeetCpp{} + Expect(p.Load(&pb.ModelOptions{ + ModelFile: asrModel, + Options: []string{"diarization_model:" + diarModel, "sound_model:" + soundModel}, + })).To(Succeed()) + defer func() { _ = p.Free() }() + + res, err := p.AudioTranscription(context.Background(), &pb.TranscriptRequest{Dst: wavPath, Diarize: true}) + Expect(err).ToNot(HaveOccurred()) + labels := make([]string, 0, len(res.Segments)) + for _, s := range res.Segments { + GinkgoWriter.Printf("[%5.1f-%5.1f] spk %q: %s\n", float64(s.Start)/1e9, float64(s.End)/1e9, s.Speaker, s.Text) + labels = append(labels, s.Speaker) + } + Expect(len(speakerTurnsOf(labels))).To(BeNumerically(">=", 3), "speaker turns: %v", labels) + Expect(labels).To(ContainElements("0", "1")) + + sd, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: wavPath, TopK: 5}) + Expect(err).ToNot(HaveOccurred()) + var got []string + for _, d := range sd.GetDetections() { + GinkgoWriter.Printf("sound %q %.2f\n", d.GetLabel(), d.GetScore()) + got = append(got, d.GetLabel()) + } + Expect(got).To(ContainElement(wantLabel)) + }) +}) diff --git a/core/backend/transcript.go b/core/backend/transcript.go index 0ddee3add..27b13bc55 100644 --- a/core/backend/transcript.go +++ b/core/backend/transcript.go @@ -213,9 +213,10 @@ func transcriptResultFromProto(r *proto.TranscriptResult) *schema.TranscriptionR var words []schema.TranscriptionWord for _, w := range s.Words { var word = schema.TranscriptionWord{ - Start: time.Duration(w.Start), - End: time.Duration(w.End), - Text: w.Text, + Start: time.Duration(w.Start), + End: time.Duration(w.End), + Text: w.Text, + Speaker: w.Speaker, } words = append(words, word) tr.Words = append(tr.Words, word) diff --git a/core/backend/transcript_live.go b/core/backend/transcript_live.go index 0ad6d72e1..e64e54a57 100644 --- a/core/backend/transcript_live.go +++ b/core/backend/transcript_live.go @@ -26,11 +26,33 @@ import ( // backchannel ("uh-huh") ended — callers must NOT treat Eob as a turn // boundary. type LiveTranscriptionEvent struct { - Delta string - Eou bool - Eob bool - Words []schema.TranscriptionWord - Final *schema.TranscriptionResult + Delta string + Eou bool + Eob bool + Words []schema.TranscriptionWord + Speakers []LiveSpeakerSegment + Sounds []LiveSoundEvent + Final *schema.TranscriptionResult +} + +// LiveSpeakerSegment is one closed speaker segment from a companion +// diarization/scene stream running alongside live transcription. Start/End +// are stream-relative seconds (mapped from the backend's nanoseconds). +type LiveSpeakerSegment struct { + Speaker string + Start float64 + End float64 +} + +// LiveSoundEvent is one closed sound event from a companion sound/scene +// stream running alongside live transcription. Start/End are stream-relative +// seconds (mapped from the backend's nanoseconds). +type LiveSoundEvent struct { + Label string + Index int + Peak float32 + Start float64 + End float64 } // LiveTranscriptionSession is a handle on an open live transcription stream. @@ -298,9 +320,26 @@ func liveEventFromProto(r *proto.TranscriptLiveResponse) LiveTranscriptionEvent } for _, w := range r.GetWords() { ev.Words = append(ev.Words, schema.TranscriptionWord{ - Start: time.Duration(w.Start), - End: time.Duration(w.End), - Text: w.Text, + Start: time.Duration(w.Start), + End: time.Duration(w.End), + Text: w.Text, + Speaker: w.Speaker, + }) + } + for _, s := range r.GetSpeakers() { + ev.Speakers = append(ev.Speakers, LiveSpeakerSegment{ + Speaker: s.GetSpeaker(), + Start: time.Duration(s.GetStart()).Seconds(), + End: time.Duration(s.GetEnd()).Seconds(), + }) + } + for _, s := range r.GetSounds() { + ev.Sounds = append(ev.Sounds, LiveSoundEvent{ + Label: s.GetLabel(), + Index: int(s.GetIndex()), + Peak: s.GetPeak(), + Start: time.Duration(s.GetStart()).Seconds(), + End: time.Duration(s.GetEnd()).Seconds(), }) } if r.GetFinalResult() != nil { diff --git a/core/backend/transcript_live_internal_test.go b/core/backend/transcript_live_internal_test.go index 6f6bed6a4..8b2aee91e 100644 --- a/core/backend/transcript_live_internal_test.go +++ b/core/backend/transcript_live_internal_test.go @@ -54,11 +54,53 @@ var _ = Describe("liveEventFromProto", func() { Expect(ev.Final).To(BeNil()) }) + It("carries word speakers and final segment speakers from a diarizing backend", func() { + ev := liveEventFromProto(&proto.TranscriptLiveResponse{ + Words: []*proto.TranscriptWord{{Text: "hi", Speaker: "1"}}, + }) + Expect(ev.Words[0].Speaker).To(Equal("1")) + ev = liveEventFromProto(&proto.TranscriptLiveResponse{ + FinalResult: &proto.TranscriptResult{ + Text: "hi there", + Segments: []*proto.TranscriptSegment{{Text: "hi", Speaker: "0"}, {Text: "there", Speaker: "1"}}, + }, + }) + Expect(ev.Final.Segments[1].Speaker).To(Equal("1")) + }) + It("maps the eob backchannel flag separately from eou", func() { ev := liveEventFromProto(&proto.TranscriptLiveResponse{Delta: "uh-huh", Eob: true}) Expect(ev.Eob).To(BeTrue()) Expect(ev.Eou).To(BeFalse()) }) + + It("maps speaker segments and sound events (ns -> seconds)", func() { + ev := liveEventFromProto(&proto.TranscriptLiveResponse{ + Speakers: []*proto.LiveSpeakerSegment{ + {Speaker: "1", Start: int64(1500 * time.Millisecond), End: int64(3200 * time.Millisecond)}, + }, + Sounds: []*proto.LiveSoundEvent{ + {Label: "Dog bark", Index: 5, Peak: 0.8, Start: int64(500 * time.Millisecond), End: int64(900 * time.Millisecond)}, + }, + }) + Expect(ev.Speakers).To(HaveLen(1)) + Expect(ev.Speakers[0].Speaker).To(Equal("1")) + Expect(ev.Speakers[0].Start).To(BeNumerically("~", 1.5, 1e-9)) + Expect(ev.Speakers[0].End).To(BeNumerically("~", 3.2, 1e-9)) + + Expect(ev.Sounds).To(HaveLen(1)) + Expect(ev.Sounds[0].Label).To(Equal("Dog bark")) + Expect(ev.Sounds[0].Index).To(Equal(5)) + Expect(ev.Sounds[0].Peak).To(BeNumerically("~", 0.8, 1e-6)) + Expect(ev.Sounds[0].Start).To(BeNumerically("~", 0.5, 1e-9)) + Expect(ev.Sounds[0].End).To(BeNumerically("~", 0.9, 1e-9)) + }) + + It("leaves speakers and sounds nil when the proto carries none", func() { + ev := liveEventFromProto(&proto.TranscriptLiveResponse{Delta: "hi"}) + Expect(ev.Speakers).To(BeNil()) + Expect(ev.Sounds).To(BeNil()) + }) }) // liveTraceState is what makes streaming-only pipelines visible on the diff --git a/core/cli/transcript.go b/core/cli/transcript.go index 06764f4dd..e187453a6 100644 --- a/core/cli/transcript.go +++ b/core/cli/transcript.go @@ -93,18 +93,20 @@ func (t *TranscriptCMD) Run(ctx *cliContext.Context) error { } for _, word := range(tr.Words) { trs.Words = append(trs.Words, schema.TranscriptionWordSeconds{ - Start: word.Start.Seconds(), - End: word.End.Seconds(), - Text: word.Text, + Start: word.Start.Seconds(), + End: word.End.Seconds(), + Text: word.Text, + Speaker: word.Speaker, }) } for _, seg := range(tr.Segments) { segWords := []schema.TranscriptionWordSeconds{} for _, word := range(seg.Words) { segWords = append(segWords, schema.TranscriptionWordSeconds{ - Start: word.Start.Seconds(), - End: word.End.Seconds(), - Text: word.Text, + Start: word.Start.Seconds(), + End: word.End.Seconds(), + Text: word.Text, + Speaker: word.Speaker, }) } trs.Segments = append(trs.Segments, schema.TranscriptionSegmentSeconds{ diff --git a/core/config/backend_capabilities.go b/core/config/backend_capabilities.go index 886ddd22a..b48266ead 100644 --- a/core/config/backend_capabilities.go +++ b/core/config/backend_capabilities.go @@ -477,11 +477,15 @@ var BackendCapabilities = map[string]BackendCapability{ DefaultUsecases: []string{UsecaseTranscript}, Description: "NVIDIA NeMo speech recognition", }, + // parakeet-cpp loads three model kinds, picked from the GGUF: an ASR model + // transcribes (and labels speakers when a diarization_model companion is + // attached), a Nemotron-3-Diarization model answers Diarize, and a CED model + // answers SoundDetection. PossibleUsecases is their union. "parakeet-cpp": { - GRPCMethods: []GRPCMethod{MethodAudioTranscription}, - PossibleUsecases: []string{UsecaseTranscript}, + GRPCMethods: []GRPCMethod{MethodAudioTranscription, MethodDiarize, MethodSoundDetection}, + PossibleUsecases: []string{UsecaseTranscript, UsecaseDiarization, UsecaseSoundClassification}, DefaultUsecases: []string{UsecaseTranscript}, - Description: "NVIDIA NeMo Parakeet ASR (parakeet.cpp)", + Description: "NVIDIA NeMo Parakeet ASR, Nemotron-3-Diarization speaker diarization and CED sound-event detection (parakeet.cpp)", }, // nemo-speech-cpp is one gRPC server in front of four NeMo-Speech.cpp model // families, picked at load time from the GGUF general.architecture key, so diff --git a/core/config/meta/registry.go b/core/config/meta/registry.go index c98514a78..dfc2dc61a 100644 --- a/core/config/meta/registry.go +++ b/core/config/meta/registry.go @@ -509,6 +509,13 @@ func DefaultRegistry() map[string]FieldMetaOverride { Min: f64(0), Order: 66, }, + "pipeline.diarization": { + Section: "pipeline", + Label: "Speaker Diarization", + Description: "Label speakers on each committed utterance and emit every labelled segment as a conversation.item.input_audio_transcription.segment event. Needs a transcription model that diarizes (e.g. parakeet-cpp with a diarization_model companion). Speaker labels are per turn.", + Component: "toggle", + Order: 67, + }, "pipeline.reasoning_effort": { Section: "pipeline", Label: "Reasoning Effort", diff --git a/core/config/model_config.go b/core/config/model_config.go index 3c8987920..c8502fae5 100644 --- a/core/config/model_config.go +++ b/core/config/model_config.go @@ -833,6 +833,14 @@ type Pipeline struct { SoundDetectionWindowMs int `yaml:"sound_detection_window_ms,omitempty" json:"sound_detection_window_ms,omitempty"` SoundDetectionHopMs int `yaml:"sound_detection_hop_ms,omitempty" json:"sound_detection_hop_ms,omitempty"` + // Diarization asks the transcription model for speaker labels on each + // VAD-committed utterance and emits every labelled segment as a + // conversation.item.input_audio_transcription.segment event. It needs a + // transcription model that diarizes (e.g. parakeet-cpp with a + // diarization_model companion); off by default because some backends fail + // a diarize request they cannot serve. Speaker labels are per turn. + Diarization bool `yaml:"diarization,omitempty" json:"diarization,omitempty"` + // ReasoningEffort sets the reasoning effort (none|minimal|low|medium|high) for // the pipeline's LLM without editing the LLM model config. Overrides the LLM's // own reasoning_effort. Unset leaves the LLM model config in charge. diff --git a/core/gallery/importers/parakeet-cpp.go b/core/gallery/importers/parakeet-cpp.go index aa732aa3e..5b758167b 100644 --- a/core/gallery/importers/parakeet-cpp.go +++ b/core/gallery/importers/parakeet-cpp.go @@ -108,6 +108,11 @@ func (i *ParakeetCppImporter) Import(details Details) (gallery.ModelConfig, erro uri := downloader.URI(details.URI) directGGUF := isParakeetGGUF(filepath.Base(details.URI)) + // A speaker diarization GGUF is served by the same backend but answers + // /v1/audio/diarization, not transcription. + if directGGUF && isParakeetDiarGGUF(filepath.Base(details.URI)) { + modelConfig.KnownUsecaseStrings = []string{"diarization"} + } switch { case uri.LooksLikeURL() && directGGUF: // Direct file URL (e.g. .../resolve/main/tdt_ctc-110m-f16.gguf). The @@ -128,12 +133,23 @@ func (i *ParakeetCppImporter) Import(details Details) (gallery.ModelConfig, erro // HF repo: collect every parakeet GGUF, pick the preferred quant, and // nest under parakeet-cpp/models// so a multi-quant repo doesn't // collide on disk. - var ggufFiles []hfapi.ModelFile + // Prefer ASR weights: a repo that also ships the diarization model + // (mudler/parakeet-cpp-gguf) imports as a transcription model, and the + // diarization GGUF is imported by its direct URL. A repo with only + // diarization weights imports as a diarization model. + var ggufFiles, diarFiles []hfapi.ModelFile for _, f := range details.HuggingFace.Files { - if isParakeetGGUF(filepath.Base(f.Path)) { + switch base := filepath.Base(f.Path); { + case isParakeetDiarGGUF(base): + diarFiles = append(diarFiles, f) + case isParakeetGGUF(base): ggufFiles = append(ggufFiles, f) } } + if len(ggufFiles) == 0 && len(diarFiles) > 0 { + ggufFiles = diarFiles + modelConfig.KnownUsecaseStrings = []string{"diarization"} + } if chosen, ok := pickPreferredGGMLFile(ggufFiles, quants); ok { target := filepath.Join("parakeet-cpp", "models", name, filepath.Base(chosen.Path)) cfg.Files = append(cfg.Files, gallery.File{ @@ -176,5 +192,14 @@ func isParakeetGGUF(name string) bool { return true } } - return false + return isParakeetDiarGGUF(name) +} + +// isParakeetDiarGGUF reports whether name is the parakeet.cpp speaker +// diarization GGUF (nemotron-3-diarization-.gguf). Matched by its +// published name only, so diarization weights for other backends are not +// claimed. +func isParakeetDiarGGUF(name string) bool { + lower := strings.ToLower(name) + return strings.HasSuffix(lower, ".gguf") && strings.Contains(lower, "nemotron-3-diarization") } diff --git a/core/gallery/importers/parakeet-cpp_test.go b/core/gallery/importers/parakeet-cpp_test.go index 4aa87c411..8636c929d 100644 --- a/core/gallery/importers/parakeet-cpp_test.go +++ b/core/gallery/importers/parakeet-cpp_test.go @@ -50,6 +50,11 @@ var _ = Describe("ParakeetCppImporter", func() { Expect(imp.Match(d)).To(BeTrue()) }) + It("matches a direct URL to the diarization GGUF", func() { + d := parakeetDetails("https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/nemotron-3-diarization-q8_0.gguf", `{}`) + Expect(imp.Match(d)).To(BeTrue()) + }) + It("does NOT claim a generic llama-style GGUF", func() { d := parakeetDetails("huggingface://someorg/some-llm-gguf", `{}`, hfapi.ModelFile{Path: "llama-3-8b-instruct-q4_k_m.gguf"}, @@ -66,6 +71,43 @@ var _ = Describe("ParakeetCppImporter", func() { }) Context("import (Import)", func() { + It("imports the diarization GGUF as a diarization model", func() { + d := parakeetDetails("https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/nemotron-3-diarization-q8_0.gguf", + `{"name":"nemotron-diarization"}`) + cfg, err := imp.Import(d) + Expect(err).ToNot(HaveOccurred()) + Expect(cfg.ConfigFile).To(ContainSubstring("backend: parakeet-cpp")) + Expect(cfg.ConfigFile).To(ContainSubstring("diarization")) + Expect(cfg.ConfigFile).ToNot(ContainSubstring("transcript")) + Expect(cfg.Files).To(HaveLen(1)) + Expect(cfg.Files[0].Filename).To(HaveSuffix("nemotron-3-diarization-q8_0.gguf")) + }) + + It("keeps picking ASR weights from a repo that also ships the diarization model", func() { + d := parakeetDetails("huggingface://mudler/parakeet-cpp-gguf", `{"name":"parakeet-110m"}`, + hfapi.ModelFile{Path: "nemotron-3-diarization-f16.gguf", URL: "https://hf/diar-f16", SHA256: "ddd"}, + hfapi.ModelFile{Path: "tdt_ctc-110m-f16.gguf", URL: "https://hf/f16", SHA256: "aaa"}, + hfapi.ModelFile{Path: "nemotron-3-diarization-q8_0.gguf", URL: "https://hf/diar-q8", SHA256: "eee"}, + ) + cfg, err := imp.Import(d) + Expect(err).ToNot(HaveOccurred()) + Expect(cfg.Files).To(HaveLen(1)) + Expect(cfg.Files[0].URI).To(Equal("https://hf/f16")) + Expect(cfg.ConfigFile).To(ContainSubstring("transcript")) + }) + + It("imports a diarization-only repo as a diarization model", func() { + d := parakeetDetails("huggingface://someone/diar-gguf", `{"name":"diar"}`, + hfapi.ModelFile{Path: "nemotron-3-diarization-f16.gguf", URL: "https://hf/diar-f16", SHA256: "ddd"}, + hfapi.ModelFile{Path: "nemotron-3-diarization-q8_0.gguf", URL: "https://hf/diar-q8", SHA256: "eee"}, + ) + cfg, err := imp.Import(d) + Expect(err).ToNot(HaveOccurred()) + Expect(cfg.Files).To(HaveLen(1)) + Expect(cfg.Files[0].URI).To(Equal("https://hf/diar-q8"), "default quant ladder picks q8_0 before f16") + Expect(cfg.ConfigFile).To(ContainSubstring("diarization")) + }) + It("picks the default quant (q4_k) from a multi-quant HF repo", func() { d := parakeetDetails("huggingface://mudler/parakeet-cpp-gguf", `{"name":"parakeet-110m"}`, hfapi.ModelFile{Path: "tdt_ctc-110m-f16.gguf", URL: "https://hf/f16", SHA256: "aaa"}, diff --git a/core/http/endpoints/openai/realtime_doubles_test.go b/core/http/endpoints/openai/realtime_doubles_test.go index e82a4a130..9ddc1299f 100644 --- a/core/http/endpoints/openai/realtime_doubles_test.go +++ b/core/http/endpoints/openai/realtime_doubles_test.go @@ -88,6 +88,7 @@ type fakeModel struct { transcribeDeltas []string transcribeFinal *schema.TranscriptionResult transcribeErr error + lastDiarize bool // diarize flag of the last Transcribe/TranscribeStream call // TranscribeLive scripting: liveErr makes the open fail (degrade path); // liveEvents are delivered to onEvent synchronously at open; @@ -189,7 +190,8 @@ func (m *fakeModel) VAD(_ context.Context, req *schema.VADRequest) (*schema.VADR return &schema.VADResponse{Segments: m.vadSegments}, nil } -func (m *fakeModel) Transcribe(context.Context, string, string, bool, bool, string) (*schema.TranscriptionResult, error) { +func (m *fakeModel) Transcribe(_ context.Context, _, _ string, _, diarize bool, _ string) (*schema.TranscriptionResult, error) { + m.lastDiarize = diarize return m.transcribeFinal, m.transcribeErr } @@ -236,7 +238,8 @@ func (m *fakeModel) TTSStream(_ context.Context, _, _, _ string, onAudio func(pc return nil } -func (m *fakeModel) TranscribeStream(_ context.Context, _, _ string, _, _ bool, _ string, onDelta func(text string)) (*schema.TranscriptionResult, error) { +func (m *fakeModel) TranscribeStream(_ context.Context, _, _ string, _, diarize bool, _ string, onDelta func(text string)) (*schema.TranscriptionResult, error) { + m.lastDiarize = diarize for _, d := range m.transcribeDeltas { onDelta(d) } diff --git a/core/http/endpoints/openai/realtime_semantic_vad.go b/core/http/endpoints/openai/realtime_semantic_vad.go index 75a71ba25..df56abf19 100644 --- a/core/http/endpoints/openai/realtime_semantic_vad.go +++ b/core/http/endpoints/openai/realtime_semantic_vad.go @@ -209,6 +209,35 @@ func (l *liveTurnState) drainEvents(audioSec float64) { if ev.Final != nil && strings.TrimSpace(ev.Final.Text) != "" { l.finalText = ev.Final.Text } + // Speaker and sound events from a companion diarization/scene + // stream: forward each as its own event under the turn's item + // id, same as caption deltas. Text is empty — the event exists + // to carry the speaker/segment boundary, not transcript text. + if l.transport != nil && l.itemID != "" { + for _, seg := range ev.Speakers { + sendEvent(l.transport, types.ConversationItemInputAudioTranscriptionSegmentEvent{ + ServerEventBase: types.ServerEventBase{EventID: "event_TODO"}, + ItemID: l.itemID, + ContentIndex: 0, + Speaker: seg.Speaker, + Start: seg.Start, + End: seg.End, + }) + } + for _, sound := range ev.Sounds { + start, end := sound.Start, sound.End + sendEvent(l.transport, types.ConversationItemSoundDetectionEvent{ + ServerEventBase: types.ServerEventBase{EventID: "event_TODO"}, + ItemID: l.itemID, + ContentIndex: 0, + Detections: []types.SoundDetectionTag{ + {Label: sound.Label, Score: sound.Peak, Index: sound.Index}, + }, + Start: &start, + End: &end, + }) + } + } default: return } diff --git a/core/http/endpoints/openai/realtime_semantic_vad_test.go b/core/http/endpoints/openai/realtime_semantic_vad_test.go index b1107c1f6..5a92e3b15 100644 --- a/core/http/endpoints/openai/realtime_semantic_vad_test.go +++ b/core/http/endpoints/openai/realtime_semantic_vad_test.go @@ -291,6 +291,67 @@ var _ = Describe("liveTurnState", func() { Expect(ftr.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionFailed)).To(Equal(0)) }) }) + + Describe("scene events (speakers and sounds)", func() { + It("emits a segment event per speaker with empty text under the turn's item id", func() { + Expect(lts.openTurn(context.Background(), "item1")).To(BeTrue()) + turnID := lts.itemID + + m.liveSession.onEvent(backend.LiveTranscriptionEvent{ + Speakers: []backend.LiveSpeakerSegment{{Speaker: "1", Start: 1.2, End: 3.4}}, + }) + lts.drainEvents(3.4) + + var got []types.ConversationItemInputAudioTranscriptionSegmentEvent + for _, e := range ftr.events() { + if seg, ok := e.(types.ConversationItemInputAudioTranscriptionSegmentEvent); ok { + got = append(got, seg) + } + } + Expect(got).To(HaveLen(1)) + Expect(got[0].ItemID).To(Equal(turnID)) + Expect(got[0].Speaker).To(Equal("1")) + Expect(got[0].Start).To(BeNumerically("~", 1.2, 1e-9)) + Expect(got[0].End).To(BeNumerically("~", 3.4, 1e-9)) + Expect(got[0].Text).To(BeEmpty()) + }) + + It("emits a sound_detection event per sound with one tag and start/end", func() { + Expect(lts.openTurn(context.Background(), "item1")).To(BeTrue()) + turnID := lts.itemID + + m.liveSession.onEvent(backend.LiveTranscriptionEvent{ + Sounds: []backend.LiveSoundEvent{{Label: "Dog bark", Index: 5, Peak: 0.8, Start: 0.5, End: 0.9}}, + }) + lts.drainEvents(1.0) + + var got []types.ConversationItemSoundDetectionEvent + for _, e := range ftr.events() { + if sd, ok := e.(types.ConversationItemSoundDetectionEvent); ok { + got = append(got, sd) + } + } + Expect(got).To(HaveLen(1)) + Expect(got[0].ItemID).To(Equal(turnID)) + Expect(got[0].Detections).To(HaveLen(1)) + Expect(got[0].Detections[0].Label).To(Equal("Dog bark")) + Expect(got[0].Detections[0].Score).To(BeNumerically("~", 0.8, 1e-6)) + Expect(got[0].Detections[0].Index).To(Equal(5)) + Expect(got[0].Start).NotTo(BeNil()) + Expect(*got[0].Start).To(BeNumerically("~", 0.5, 1e-9)) + Expect(got[0].End).NotTo(BeNil()) + Expect(*got[0].End).To(BeNumerically("~", 0.9, 1e-9)) + }) + + It("sends neither event when a live event carries no speakers or sounds", func() { + Expect(lts.openTurn(context.Background(), "item1")).To(BeTrue()) + m.liveSession.onEvent(backend.LiveTranscriptionEvent{Delta: "hi"}) + lts.drainEvents(1.0) + + Expect(ftr.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionSegment)).To(Equal(0)) + Expect(ftr.countEvents(types.ServerEventTypeConversationItemSoundDetection)).To(Equal(0)) + }) + }) }) // commitUtteranceWithTranscript routes the three transcript sources: the diff --git a/core/http/endpoints/openai/realtime_sound_detection_test.go b/core/http/endpoints/openai/realtime_sound_detection_test.go index 058c74076..94406e7cc 100644 --- a/core/http/endpoints/openai/realtime_sound_detection_test.go +++ b/core/http/endpoints/openai/realtime_sound_detection_test.go @@ -3,6 +3,7 @@ package openai import ( "context" "encoding/binary" + "encoding/json" "errors" "os" @@ -14,6 +15,69 @@ import ( "github.com/mudler/LocalAI/core/schema" ) +// ConversationItemSoundDetectionEvent gained optional Start/End (seconds) +// for the live scene-event path; the unary/windowed paths never set them, +// so existing consumers must see no start/end keys at all. +var _ = Describe("ConversationItemSoundDetectionEvent JSON", func() { + It("omits start and end when nil", func() { + ev := types.ConversationItemSoundDetectionEvent{ + ItemID: "item1", + Detections: []types.SoundDetectionTag{{Label: "Speech", Score: 0.5, Index: 7}}, + } + b, err := json.Marshal(ev) + Expect(err).ToNot(HaveOccurred()) + + var got map[string]any + Expect(json.Unmarshal(b, &got)).To(Succeed()) + _, hasStart := got["start"] + _, hasEnd := got["end"] + Expect(hasStart).To(BeFalse()) + Expect(hasEnd).To(BeFalse()) + }) + + It("includes start and end when set", func() { + start, end := 0.5, 0.9 + ev := types.ConversationItemSoundDetectionEvent{ + ItemID: "item1", + Start: &start, + End: &end, + } + b, err := json.Marshal(ev) + Expect(err).ToNot(HaveOccurred()) + + var got map[string]any + Expect(json.Unmarshal(b, &got)).To(Succeed()) + Expect(got["start"]).To(BeNumerically("~", 0.5, 1e-9)) + Expect(got["end"]).To(BeNumerically("~", 0.9, 1e-9)) + }) +}) + +// ConversationItemInputAudioTranscriptionSegmentEvent.Start/End are plain +// float64 (no omitempty): a speaker segment starting at 0.0s must still +// carry "start" in the JSON, unlike the sound-detection event's optional +// pointer fields above. +var _ = Describe("ConversationItemInputAudioTranscriptionSegmentEvent JSON", func() { + It("marshals start:0 and end:1.5 even when start is the zero value", func() { + ev := types.ConversationItemInputAudioTranscriptionSegmentEvent{ + ItemID: "item1", + Speaker: "1", + Start: 0, + End: 1.5, + } + b, err := json.Marshal(ev) + Expect(err).ToNot(HaveOccurred()) + + var got map[string]any + Expect(json.Unmarshal(b, &got)).To(Succeed()) + _, hasStart := got["start"] + _, hasEnd := got["end"] + Expect(hasStart).To(BeTrue()) + Expect(hasEnd).To(BeTrue()) + Expect(got["start"]).To(BeNumerically("~", 0.0, 1e-9)) + Expect(got["end"]).To(BeNumerically("~", 1.5, 1e-9)) + }) +}) + // emitSoundDetection classifies a committed utterance and emits a single // conversation.item.sound_detection event carrying the scored AudioSet tags. var _ = Describe("emitSoundDetection", func() { diff --git a/core/http/endpoints/openai/realtime_transcription.go b/core/http/endpoints/openai/realtime_transcription.go index 28a5147c1..c10535f18 100644 --- a/core/http/endpoints/openai/realtime_transcription.go +++ b/core/http/endpoints/openai/realtime_transcription.go @@ -5,6 +5,7 @@ import ( "fmt" "github.com/mudler/LocalAI/core/http/endpoints/openai/types" + "github.com/mudler/LocalAI/core/schema" ) // emitPrecomputedTranscription emits the transcription events for a turn @@ -42,9 +43,10 @@ func emitPrecomputedTranscription(t Transport, itemID string, deltas []string, t // a single completed event. delta and completed events share itemID. func emitTranscription(ctx context.Context, t Transport, session *Session, itemID, audioPath string) (string, error) { cfg := session.InputAudioTranscription + diarize := session.ModelConfig != nil && session.ModelConfig.Pipeline.Diarization if session.ModelConfig != nil && session.ModelConfig.Pipeline.StreamTranscription() { - final, err := session.ModelInterface.TranscribeStream(ctx, audioPath, cfg.Language, false, false, cfg.Prompt, func(delta string) { + final, err := session.ModelInterface.TranscribeStream(ctx, audioPath, cfg.Language, false, diarize, cfg.Prompt, func(delta string) { _ = t.SendEvent(types.ConversationItemInputAudioTranscriptionDeltaEvent{ ServerEventBase: types.ServerEventBase{EventID: "event_TODO"}, ItemID: itemID, @@ -58,6 +60,11 @@ func emitTranscription(ctx context.Context, t Transport, session *Session, itemI transcript := "" if final != nil { transcript = final.Text + if diarize { + if err := emitSpeakerSegments(t, itemID, final); err != nil { + return "", err + } + } } if err := t.SendEvent(types.ConversationItemInputAudioTranscriptionCompletedEvent{ ServerEventBase: types.ServerEventBase{EventID: "event_TODO"}, @@ -71,13 +78,18 @@ func emitTranscription(ctx context.Context, t Transport, session *Session, itemI } // Unary fallback: transcribe the whole utterance, emit one completed event. - tr, err := session.ModelInterface.Transcribe(ctx, audioPath, cfg.Language, false, false, cfg.Prompt) + tr, err := session.ModelInterface.Transcribe(ctx, audioPath, cfg.Language, false, diarize, cfg.Prompt) if err != nil { return "", err } if tr == nil { return "", fmt.Errorf("transcribe result is nil") } + if diarize { + if err := emitSpeakerSegments(t, itemID, tr); err != nil { + return "", err + } + } if err := t.SendEvent(types.ConversationItemInputAudioTranscriptionCompletedEvent{ ServerEventBase: types.ServerEventBase{EventID: "event_TODO"}, ItemID: itemID, @@ -88,3 +100,29 @@ func emitTranscription(ctx context.Context, t Transport, session *Session, itemI } return tr.Text, nil } + +// emitSpeakerSegments forwards each speaker-labelled segment of a committed +// turn's transcript as a conversation.item.input_audio_transcription.segment +// event (pipeline.diarization), before the turn's completed event. Times are +// relative to the turn's audio and speaker labels are only consistent within +// the turn, as on the live path. Segments without a speaker are skipped. +func emitSpeakerSegments(t Transport, itemID string, tr *schema.TranscriptionResult) error { + for _, seg := range tr.Segments { + if seg.Speaker == "" { + continue + } + if err := t.SendEvent(types.ConversationItemInputAudioTranscriptionSegmentEvent{ + ServerEventBase: types.ServerEventBase{EventID: "event_TODO"}, + ItemID: itemID, + ContentIndex: 0, + ID: fmt.Sprintf("seg_%d", seg.Id), + Speaker: seg.Speaker, + Start: seg.Start.Seconds(), + End: seg.End.Seconds(), + Text: seg.Text, + }); err != nil { + return err + } + } + return nil +} diff --git a/core/http/endpoints/openai/realtime_transcription_test.go b/core/http/endpoints/openai/realtime_transcription_test.go index f3f760fd8..e8ab399fd 100644 --- a/core/http/endpoints/openai/realtime_transcription_test.go +++ b/core/http/endpoints/openai/realtime_transcription_test.go @@ -2,6 +2,7 @@ package openai import ( "context" + "time" . "github.com/onsi/ginkgo/v2" . "github.com/onsi/gomega" @@ -51,4 +52,86 @@ var _ = Describe("emitTranscription", func() { Expect(t.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionDelta)).To(Equal(0)) Expect(t.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionCompleted)).To(Equal(1)) }) + + Context("pipeline.diarization", func() { + labelled := &schema.TranscriptionResult{ + Text: "hi there. hello", + Segments: []schema.TranscriptionSegment{ + {Id: 0, Text: "hi there.", Start: 0, End: 600 * time.Millisecond, Speaker: "0"}, + {Id: 1, Text: "hello", Start: time.Second, End: 1400 * time.Millisecond, Speaker: "1"}, + {Id: 2, Text: "unlabelled"}, + }, + } + + segmentEvents := func(t *fakeTransport) []types.ConversationItemInputAudioTranscriptionSegmentEvent { + var out []types.ConversationItemInputAudioTranscriptionSegmentEvent + for _, e := range t.sent { + if seg, ok := e.(types.ConversationItemInputAudioTranscriptionSegmentEvent); ok { + out = append(out, seg) + } + } + return out + } + + It("requests speakers and emits one segment event per labelled segment", func() { + m := &fakeModel{transcribeFinal: labelled} + session := &Session{ + InputAudioTranscription: &types.AudioTranscription{}, + ModelConfig: &config.ModelConfig{Pipeline: config.Pipeline{Diarization: true}}, + ModelInterface: m, + } + t := &fakeTransport{} + + transcript, err := emitTranscription(context.Background(), t, session, "item1", "/tmp/x.wav") + + Expect(err).ToNot(HaveOccurred()) + Expect(transcript).To(Equal("hi there. hello")) + Expect(m.lastDiarize).To(BeTrue()) + segs := segmentEvents(t) + Expect(segs).To(HaveLen(2)) + Expect(segs[0].ItemID).To(Equal("item1")) + Expect(segs[0].Speaker).To(Equal("0")) + Expect(segs[0].Text).To(Equal("hi there.")) + Expect(segs[1].Speaker).To(Equal("1")) + Expect(segs[1].Start).To(BeNumerically("~", 1.0, 1e-9)) + Expect(segs[1].End).To(BeNumerically("~", 1.4, 1e-9)) + Expect(t.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionCompleted)).To(Equal(1)) + }) + + It("also emits segment events on the streaming transcription path", func() { + on := true + m := &fakeModel{transcribeDeltas: []string{"hi"}, transcribeFinal: labelled} + session := &Session{ + InputAudioTranscription: &types.AudioTranscription{}, + ModelConfig: &config.ModelConfig{Pipeline: config.Pipeline{ + Diarization: true, + Streaming: config.PipelineStreaming{Transcription: &on}, + }}, + ModelInterface: m, + } + t := &fakeTransport{} + + _, err := emitTranscription(context.Background(), t, session, "item1", "/tmp/x.wav") + + Expect(err).ToNot(HaveOccurred()) + Expect(m.lastDiarize).To(BeTrue()) + Expect(segmentEvents(t)).To(HaveLen(2)) + }) + + It("neither asks for speakers nor emits segments when off", func() { + m := &fakeModel{transcribeFinal: labelled} + session := &Session{ + InputAudioTranscription: &types.AudioTranscription{}, + ModelConfig: &config.ModelConfig{}, + ModelInterface: m, + } + t := &fakeTransport{} + + _, err := emitTranscription(context.Background(), t, session, "item1", "/tmp/x.wav") + + Expect(err).ToNot(HaveOccurred()) + Expect(m.lastDiarize).To(BeFalse()) + Expect(segmentEvents(t)).To(BeEmpty()) + }) + }) }) diff --git a/core/http/endpoints/openai/transcription.go b/core/http/endpoints/openai/transcription.go index 6c99d7afc..49825c3a1 100644 --- a/core/http/endpoints/openai/transcription.go +++ b/core/http/endpoints/openai/transcription.go @@ -210,18 +210,20 @@ func TranscriptEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, app } for _, word := range tr.Words { trs.Words = append(trs.Words, schema.TranscriptionWordSeconds{ - Start: word.Start.Seconds(), - End: word.End.Seconds(), - Text: word.Text, + Start: word.Start.Seconds(), + End: word.End.Seconds(), + Text: word.Text, + Speaker: word.Speaker, }) } for _, seg := range tr.Segments { segWords := []schema.TranscriptionWordSeconds{} for _, word := range seg.Words { segWords = append(segWords, schema.TranscriptionWordSeconds{ - Start: word.Start.Seconds(), - End: word.End.Seconds(), - Text: word.Text, + Start: word.Start.Seconds(), + End: word.End.Seconds(), + Text: word.Text, + Speaker: word.Speaker, }) } trs.Segments = append(trs.Segments, schema.TranscriptionSegmentSeconds{ @@ -338,12 +340,16 @@ func streamTranscription(c echo.Context, req backend.TranscriptionRequest, ml *m if len(finalResult.Segments) > 0 { segs := make([]map[string]any, 0, len(finalResult.Segments)) for _, seg := range finalResult.Segments { - segs = append(segs, map[string]any{ + entry := map[string]any{ "id": seg.Id, "start": seg.Start.Seconds(), "end": seg.End.Seconds(), "text": seg.Text, - }) + } + if seg.Speaker != "" { + entry["speaker"] = seg.Speaker + } + segs = append(segs, entry) } doneEvent["segments"] = segs } diff --git a/core/http/endpoints/openai/types/server_events.go b/core/http/endpoints/openai/types/server_events.go index 114a7065a..4cab30a61 100644 --- a/core/http/endpoints/openai/types/server_events.go +++ b/core/http/endpoints/openai/types/server_events.go @@ -512,6 +512,15 @@ type ConversationItemSoundDetectionEvent struct { // The scored sound-event tags, in score-descending order. Detections []SoundDetectionTag `json:"detections"` + + // The start time of the detection window in seconds, when known. Set by + // the live scene-event path (a companion sound stream alongside live + // transcription); omitted by the unary/windowed sound-detection paths, + // which have no per-event timing. + Start *float64 `json:"start,omitempty"` + + // The end time of the detection window in seconds, when known. + End *float64 `json:"end,omitempty"` } func (m ConversationItemSoundDetectionEvent) ServerEventType() ServerEventType { @@ -586,11 +595,13 @@ type ConversationItemInputAudioTranscriptionSegmentEvent struct { // The speaker label for the segment, if available. Speaker string `json:"speaker,omitempty"` - // The start time of the segment in seconds. - Start float64 `json:"start,omitempty"` + // The start time of the segment in seconds. Always present (not + // omitempty: a segment starting at 0.0s must still carry "start"). + Start float64 `json:"start"` - // The end time of the segment in seconds. - End float64 `json:"end,omitempty"` + // The end time of the segment in seconds. Always present (not + // omitempty: see Start). + End float64 `json:"end"` // The text content of the segment. Text string `json:"text,omitempty"` diff --git a/core/schema/transcription.go b/core/schema/transcription.go index 8414fd0ba..aab914f70 100644 --- a/core/schema/transcription.go +++ b/core/schema/transcription.go @@ -13,9 +13,10 @@ type TranscriptionSegment struct { } type TranscriptionWord struct { - Start time.Duration `json:"start"` - End time.Duration `json:"end"` - Text string `json:"text"` + Start time.Duration `json:"start"` + End time.Duration `json:"end"` + Text string `json:"text"` + Speaker string `json:"speaker,omitempty"` } type TranscriptionResult struct { @@ -42,9 +43,10 @@ type TranscriptionSegmentSeconds struct { } type TranscriptionWordSeconds struct { - Start float64 `json:"start"` - End float64 `json:"end"` - Text string `json:"text"` + Start float64 `json:"start"` + End float64 `json:"end"` + Text string `json:"text"` + Speaker string `json:"speaker,omitempty"` } type TranscriptionResultSeconds struct { diff --git a/docs/content/features/audio-classification.md b/docs/content/features/audio-classification.md index 09759ab72..4f9c58fb9 100644 --- a/docs/content/features/audio-classification.md +++ b/docs/content/features/audio-classification.md @@ -9,6 +9,8 @@ Sound-event classification (audio tagging) answers the question **"what am I hea LocalAI exposes this through the `/v1/audio/classification` endpoint, modelled after `/v1/audio/transcriptions`. The reference backend is **[ced.cpp](https://github.com/localai-org/ced.cpp)** (CED, a 527-class AudioSet tagger), a small ViT over a log-mel spectrogram ported to ggml with full PyTorch parity. Apache-2.0 weights are redistributable as GGUF. +**[parakeet.cpp](https://github.com/mudler/parakeet.cpp)** can also load a CED model (through `third_party/ced.cpp`) and serve `/v1/audio/classification` from the same backend used for ASR and diarization. It scores the clip in 10 s windows and averages each class's score across the windows before sorting and applying `top_k`/`threshold` - CED's own method for clips longer than one window. Install `parakeet-cpp-ced-tiny` or `parakeet-cpp-ced-base` from the gallery, or point `parameters.model` at a CED GGUF under `backend: parakeet-cpp`. A parakeet-cpp ASR model can also point `sound_model` at a CED GGUF to add live sound events during realtime transcription - see [Realtime API]({{% relref "openai-realtime" %}}). + Because classification is exposed as a regular OpenAI-style endpoint, any HTTP client works - there is no Python dependency on the consumer side. In distributed mode, LocalAI stages uploaded audio and realtime sound-detection @@ -59,6 +61,25 @@ curl http://localhost:8080/v1/audio/classification \ -F top_k=10 ``` +The same request works unchanged against a parakeet-cpp CED model: + +```yaml +name: parakeet-ced-tiny +backend: parakeet-cpp +parameters: + model: ced-tiny-q8_0.gguf +known_usecases: + - sound_classification +``` + +```bash +curl http://localhost:8080/v1/audio/classification \ + -H "Content-Type: multipart/form-data" \ + -F file="@/path/to/clip.wav" \ + -F model="parakeet-ced-tiny" \ + -F top_k=10 +``` + ## See also - [Audio to Text]({{% relref "audio-to-text" %}}) - speech transcription diff --git a/docs/content/features/audio-diarization.md b/docs/content/features/audio-diarization.md index 83a51343e..37f8c8159 100644 --- a/docs/content/features/audio-diarization.md +++ b/docs/content/features/audio-diarization.md @@ -9,12 +9,13 @@ url = "/features/audio-diarization/" Speaker diarization answers the question **"who spoke when?"** - given an audio clip with multiple speakers, it returns time-stamped segments labelled with a stable speaker ID (`SPEAKER_00`, `SPEAKER_01`, …). -LocalAI exposes this through the `/v1/audio/diarization` endpoint, modelled after `/v1/audio/transcriptions`. Four backends are supported today: +LocalAI exposes this through the `/v1/audio/diarization` endpoint, modelled after `/v1/audio/transcriptions`. Five backends are supported today: - **[sherpa-onnx](https://github.com/k2-fsa/sherpa-onnx)** - pyannote-3.0 segmentation + a speaker-embedding extractor (3D-Speaker, NeMo, WeSpeaker) + fast clustering. Pure diarization - no transcription cost. Recommended when you only need speaker turns. - **[vibevoice.cpp](https://github.com/microsoft/VibeVoice)** - produces speaker-labelled segments as a by-product of its long-form ASR pass, so you can optionally get a transcript per segment for free. - **[NeMo-Speech.cpp](https://github.com/NVIDIA/NeMo-Speech.cpp)** - NVIDIA Sortformer, served standalone by the [NeMo-Speech.cpp backend]({{%relref "features/nemo-speech-cpp" %}}). It is end to end, so the speaker capacity is fixed by the checkpoint and the count hints are ignored. The same backend can instead put speaker tags on a transcript, by attaching a Sortformer model to an ASR one. - **[audio.cpp](https://github.com/0xShug0/audio.cpp)** - the `sortformer_diar` family, served by the multi-modality [audio.cpp backend]({{%relref "features/audio-cpp" %}}). +- **[parakeet.cpp](https://github.com/mudler/parakeet.cpp)** - NVIDIA Nemotron-3-Diarization (Sortformer, up to 8 speakers), served standalone or paired with a Parakeet ASR model for per-segment text. See the [Audio to Text]({{% relref "audio-to-text" %}}) page for the parakeet-cpp option reference. Because diarization is exposed as a regular OpenAI-compatible endpoint, any HTTP client works. There is no Python dependency on pyannote or NeMo on the consumer side. @@ -157,10 +158,38 @@ curl http://localhost:8080/v1/audio/diarization \ -F response_format=verbose_json ``` +## Backend setup - parakeet-cpp (Nemotron-3-Diarization) + +Nemotron-3-Diarization is Sortformer, served standalone or paired with a Parakeet ASR model. Install `parakeet-cpp-nemotron-3-diarization` from the gallery for diarization only, or `parakeet-cpp-nemotron-3-diarization-asr` for the same model paired with `parakeet-cpp-tdt_ctc-110m` through the `asr_model` option: + +```yaml +name: parakeet-diarize +backend: parakeet-cpp +parameters: + model: nemotron-3-diarization-q8_0.gguf +options: + - asr_model:tdt_ctc-110m-f16.gguf +known_usecases: + - diarization +``` + +Getting text on each segment needs both: an `asr_model` companion loaded on the model, and `include_text=true` on the request. With only one of the two, segments carry no text and no error is raised. Sortformer has a fixed speaker capacity and no clustering stage, so `num_speakers`, `min_speakers`, `max_speakers` and `clustering_threshold` are ignored (logged at debug); `min_duration_on` and `min_duration_off` are honored. Speaker labels are the decimal index the model assigned (`"0"`, `"1"`, …), or `"unknown"` when a segment has no diarized speaker. + +```bash +curl http://localhost:8080/v1/audio/diarization \ + -H "Content-Type: multipart/form-data" \ + -F file="@meeting.wav" \ + -F model="parakeet-diarize" \ + -F include_text=true \ + -F response_format=verbose_json +``` + +Sortformer clusters on voice-like characteristics, not on "is this a human". A loud non-speech sound with voice-like pitch and rhythm (a rooster crow, in one test clip) can come back as its own speaker segment alongside the real speakers. This is model behavior, not a bug in the LocalAI integration: treat an unexpected extra speaker as a hint the clip may contain a non-speech sound, and use [Sound Classification]({{% relref "audio-classification" %}}) to confirm what it is. + ## Notes - **Speaker identity across files**: speaker IDs (`SPEAKER_00`, `SPEAKER_01`, …) are local to each request. To track the same person across multiple recordings, combine `/v1/audio/diarization` with `/v1/voice/embed` (speaker embedding) and maintain your own embedding store. -- **Hints vs. forces**: `num_speakers` overrides clustering when set; `min_speakers` / `max_speakers` are advisory and only honored by backends that expose a range hint. vibevoice.cpp ignores them - its model picks the count itself. +- **Hints vs. forces**: `num_speakers` overrides clustering when set; `min_speakers` / `max_speakers` are advisory and only honored by backends that expose a range hint. vibevoice.cpp and parakeet-cpp (Sortformer) ignore them - the model picks the count itself. - **Sample rate**: input is automatically converted to 16 kHz mono via ffmpeg before the backend sees it; sherpa-onnx pyannote-3.0 requires 16 kHz. ## See also diff --git a/docs/content/features/audio-to-text.md b/docs/content/features/audio-to-text.md index 5a5e833cf..d01edec4e 100644 --- a/docs/content/features/audio-to-text.md +++ b/docs/content/features/audio-to-text.md @@ -12,7 +12,7 @@ The transcription endpoint allows to convert audio files to text. The endpoint s - **moonshine**: Ultra-fast transcription engine optimized for low-end devices - **faster-whisper**: Fast Whisper implementation with CTranslate2 - **WhisperX**: Whisper transcription with word alignment and optional speaker diarization. Set `HF_TOKEN` and pass `diarize=true` to load WhisperX's gated pyannote diarization pipeline. -- **[parakeet-cpp](https://github.com/mudler/parakeet.cpp)**: A C++/ggml port of NVIDIA NeMo Parakeet (FastConformer TDT/CTC/RNNT/hybrid). Runs quantized GGUFs on CPU or GPU, emits word-level timestamps, and supports cache-aware streaming (the `realtime_eou` model surfaces end-of-utterance events). +- **[parakeet-cpp](https://github.com/mudler/parakeet.cpp)**: A C++/ggml port of NVIDIA NeMo Parakeet (FastConformer TDT/CTC/RNNT/hybrid). Runs quantized GGUFs on CPU or GPU, emits word-level timestamps, and supports cache-aware streaming (the `realtime_eou` model surfaces end-of-utterance events). The same backend also loads Nemotron-3-Diarization (`/v1/audio/diarization`) and CED sound models (`/v1/audio/classification`), and can attach either as a companion to a transcription model. - **llama-cpp**: Route transcription to any multimodal-audio GGUF model served by the `llama-cpp` backend (e.g. [Qwen3-ASR](https://huggingface.co/ggml-org/Qwen3-ASR-0.6B-GGUF), Voxtral, Qwen2-Audio). Under the hood the request is converted into a chat completion with the audio attached via the model's audio encoder - the same path the upstream llama.cpp server uses. Set `backend: llama-cpp` in the model YAML and point `mmproj` at the matching audio encoder. - **voxtral**: Voxtral-family models served by a dedicated backend - **[NeMo-Speech.cpp](https://github.com/NVIDIA/NeMo-Speech.cpp)**: NVIDIA's C++/ggml runtime for the Nemotron Speech models. Serves offline, streaming and live transcription, with VAD, punctuation, inverse text normalization and Sortformer speaker tags attached through model options, and covers diarization, speech synthesis and translation from the same backend. See the [NeMo-Speech.cpp backend]({{%relref "features/nemo-speech-cpp" %}}) page for the model options. @@ -190,6 +190,21 @@ curl http://localhost:8080/v1/audio/transcriptions \ For real-time use, load a cache-aware streaming model (e.g. `realtime_eou_120m-v1-*.gguf`) and pass `-F stream=true`. Deltas are emitted as the audio is decoded, with end-of-utterance events closing each segment. +### Diarization and sound classification + +The same backend also serves the `/v1/audio/diarization` and `/v1/audio/classification` endpoints, and can attach a diarization or sound model to a live transcription session. `options:` accepts paths relative to the models directory, or absolute: + +| Option | Allowed on | Used for | +|---|---|---| +| `asr_model:` | a diarization model | `include_text` on `/v1/audio/diarization` | +| `diarization_model:` | an ASR model | a `speaker` on transcript segments (and words), and speaker segments during realtime live transcription | +| `sound_model:` | an ASR model | sound events during realtime live transcription | +| `diarization_latency:` | a model with a diarization companion | latency mode for the live speaker stream; default `low` | + +With a `diarization_model` companion, `/v1/audio/transcriptions` labels each segment with its `speaker` (`"0"`, `"1"`, ... in order of first appearance) and splits segments where the speaker changes; with `timestamp_granularities[]=word` each word carries its speaker too. With `stream=true` the closing `transcript.text.done` event lists the segments with their speakers. Pass `-F diarize=false` to skip diarization for one request. The diarization GGUF can also be imported directly: `local-ai models import https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/nemotron-3-diarization-f16.gguf`. + +The loader rejects a companion whose role duplicates the primary's own (for example `asr_model:` on an already-ASR primary, or `sound_model:` on a CED primary), and rejects a companion GGUF that does not match the role its option names (for example `sound_model:` pointing at an ASR GGUF fails to load, naming the kind it expected). See [Speaker Diarization]({{% relref "audio-diarization" %}}) for the `Diarize` RPC and [Sound Classification]({{% relref "audio-classification" %}}) for `SoundDetection`, and [Realtime API]({{% relref "openai-realtime" %}}) for the live speaker/sound events emitted during a realtime session. + ### Segment timestamps Transcriptions are split into segments the same way NVIDIA NeMo does: a new segment starts after sentence-ending punctuation (`.`, `?`, `!`), and each segment carries `start`/`end` times. This is the default (NeMo's punctuation-only segmentation) and needs no configuration. While streaming, each end-of-utterance closes a segment, now with timestamps. diff --git a/docs/content/features/openai-realtime.md b/docs/content/features/openai-realtime.md index ac7afb484..150966cc4 100644 --- a/docs/content/features/openai-realtime.md +++ b/docs/content/features/openai-realtime.md @@ -129,6 +129,113 @@ A client `session.update` still overrides `type` and `eagerness` per session. - `false` (default): the transcript accumulated from the live stream is used as-is - the model runs once per utterance and the LLM starts immediately at commit. - `true`: the committed audio is re-transcribed offline. If the batch decode also ends with the end-of-utterance token the turn proceeds (using the batch transcript); if it does **not**, the commit is cancelled and the session keeps listening - treating the streaming token as a false positive. Both transcripts are compared and logged, which makes this mode a useful diagnostic for how well the streaming and batch decodes align, at the cost of one extra decode per turn. +### Live speaker and sound events (parakeet-cpp) + +When the `semantic_vad` transcription model is a parakeet-cpp model loaded with a `diarization_model` and/or `sound_model` companion (see [Audio to Text]({{% relref "audio-to-text" %}})), the realtime session also streams speaker and sound events while a turn is live, alongside the transcript deltas. Nothing needs to change on the client: unrecognized event types are ignored by standard OpenAI Realtime clients. + +The transcription model, with its companions: + +```yaml +name: parakeet-realtime-scene +backend: parakeet-cpp +parameters: + model: realtime_eou_120m-v1-f16.gguf +options: + - diarization_model:nemotron-3-diarization-q8_0.gguf + - sound_model:ced-tiny-q8_0.gguf +``` + +The realtime pipeline that uses it: + +```yaml +name: gpt-realtime +pipeline: + vad: silero-vad-ggml + transcription: parakeet-realtime-scene + llm: qwen3-4b + tts: tts-1 + turn_detection: + type: semantic_vad +``` + +Each closed speaker segment emits a `conversation.item.input_audio_transcription.segment` event under the turn's item id, with an empty `text` (the event exists to carry the speaker boundary, not a transcript - the transcript still comes from the ordinary delta/completed events): + +```json +{ + "type": "conversation.item.input_audio_transcription.segment", + "item_id": "item_abc", + "content_index": 0, + "speaker": "0", + "start": 1.92, + "end": 4.10, + "text": "" +} +``` + +Each sound event emits a `conversation.item.sound_detection` event with one tag and the detection window's `start`/`end`: + +```json +{ + "type": "conversation.item.sound_detection", + "item_id": "item_abc", + "content_index": 0, + "detections": [{"label": "Chicken, rooster", "score": 0.91, "index": 99}], + "start": 24.0, + "end": 30.0 +} +``` + +The `start`/`end` on both event types are seconds measured from the start of the current turn's own audio, not the session or the WebSocket connection - the same base the streamed transcript words use. + +The companion stream is opened fresh for each speech turn, alongside that turn's ASR live session, and closed when the turn commits: whatever it had not yet emitted is drained and sent at that point. Because the diarization model runs a brand new session every turn, its speaker indices are scoped to the turn too - `"speaker": "0"` in one turn and `"speaker": "0"` in the next are not guaranteed to be the same person, even within the same conversation. + +`score` is the peak score seen for that tag while the sound was live, not an average. + +**Limitation**: under `semantic_vad`, live transcription (and so this companion stream) only runs during speech turns - it does not see audio between turns. A sound that happens while nobody is speaking is not detected this way. If you need sound events independent of speech turns, use the pipeline's `sound_detection` model instead (see [Sound Classification]({{% relref "audio-classification" %}})), which classifies each VAD-committed utterance on its own. Use one or the other, not both, on the same session - they overlap in purpose and would emit sound detections twice. + +### Speaker and sound events with an offline model (Parakeet TDT v3) + +The live events above need a cache-aware streaming transcription model. An offline model such as Parakeet TDT 0.6B v3 (25 languages) runs under `server_vad` instead: each VAD-committed turn is transcribed as a whole. The gallery model `parakeet-cpp-realtime-scene-tdt` bundles it with Nemotron-3-Diarization and CED-Tiny, so one parakeet-cpp backend handles transcription, speakers and sounds. Point both `transcription` and `sound_detection` at it and turn on `diarization`: + +```yaml +name: gpt-realtime-scene +pipeline: + vad: silero-vad-ggml + transcription: parakeet-cpp-realtime-scene-tdt + sound_detection: parakeet-cpp-realtime-scene-tdt + diarization: true + llm: qwen3-4b + tts: tts-1 +``` + +`pipeline.diarization` asks the transcription model for speaker labels on each committed turn and emits every labelled segment as a `conversation.item.input_audio_transcription.segment` event before the turn's `completed` event. Unlike the live path, these segments carry their `text`: + +```json +{ + "type": "conversation.item.input_audio_transcription.segment", + "item_id": "item_abc", + "content_index": 0, + "id": "seg_1", + "speaker": "1", + "start": 6.85, + "end": 10.82, + "text": "Well, I don't wish to see it any more, observed Phoebe, turning away her eyes." +} +``` + +`sound_detection` classifies the same committed audio and emits one `conversation.item.sound_detection` event per turn (see [Sound Classification]({{% relref "audio-classification" %}})). As on the live path, times are relative to the turn's audio and speaker labels are only consistent within a turn. `pipeline.diarization` is off by default: it needs a transcription model that diarizes (parakeet-cpp with a `diarization_model` companion), and some other backends fail a diarization request they cannot serve. + +#### Choosing the sound model + +Both scene models ship with CED-Tiny, the cheapest to run all the time. `parakeet-cpp-realtime-scene-base` and `parakeet-cpp-realtime-scene-tdt-base` are the same pipelines with CED-Base (86M, the largest CED), which tags sounds more confidently. Any CED GGUF from [`mudler/ced-gguf`](https://huggingface.co/mudler/ced-gguf) (tiny, mini, small, base) works as `sound_model`. Measured on CPU (Ryzen 9 9950X3D) over a 37 s clip with two speakers and a rooster, as a fraction of real time: + +| | CED-Tiny | CED-Base | +|---|---|---| +| Live scene stream (diarization `low` + sound), EOU path | 0.103 | 0.125 | +| Sound detection per committed turn, TDT path | 0.005 | 0.031 | + +The EOU model's own ASR stream adds 0.016. Diarization dominates the live cost, so CED-Base keeps the live path about 7x faster than real time. + ### Disabling thinking For reasoning models, you can force the pipeline LLM's thinking off without editing the LLM model config: diff --git a/gallery/index.yaml b/gallery/index.yaml index 4e37b0e78..71c572e6b 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -53670,6 +53670,389 @@ - filename: parakeet-cpp/nemotron-3.5-asr-streaming-0.6b-q8_0.gguf uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3.5-asr-streaming-0.6b-q8_0.gguf sha256: ba2f13eccd4a5245be728f77e6149bd6a4fdcdd133ff2e08ac6005bcef7a99f1 +- name: parakeet-cpp-nemotron-3-diarization + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/mudler/parakeet-cpp-gguf + - https://huggingface.co/nvidia/Nemotron-3-Diarization + - https://github.com/mudler/parakeet.cpp + description: | + Nemotron-3-Diarization (Sortformer), Q8_0 GGUF for the parakeet-cpp backend + (C++/ggml port of NVIDIA NeMo). Speaker diarization only: served through + /v1/audio/diarization, returns per-segment start, end and speaker label + ("0", "1", ...). It does not transcribe; pair it with an ASR model and set + asr_model to get speaker-attributed text from the same call. num_speakers, + min_speakers, max_speakers and clustering_threshold are not supported by + Sortformer and are ignored. + license: openmdw-1.1 + tags: + - parakeet + - parakeet-cpp + - nemotron + - sortformer + - diarization + - speaker-diarization + - gguf + - ggml + - quantized + overrides: + backend: parakeet-cpp + known_usecases: + - diarization + name: parakeet-cpp-nemotron-3-diarization + parameters: + model: parakeet-cpp/nemotron-3-diarization-q8_0.gguf + files: + - filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf + sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22 +- name: parakeet-cpp-nemotron-3-diarization-asr + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/mudler/parakeet-cpp-gguf + - https://huggingface.co/nvidia/Nemotron-3-Diarization + - https://huggingface.co/nvidia/parakeet-tdt_ctc-110m + - https://github.com/mudler/parakeet.cpp + description: | + Nemotron-3-Diarization (Sortformer) paired with the Parakeet TDT+CTC 110M + ASR model through the asr_model option, both Q8_0/F16 GGUF for the + parakeet-cpp backend (C++/ggml port of NVIDIA NeMo). Served through + /v1/audio/diarization with include_text: each speaker segment comes back + with its transcribed text in one call. Diarization model is + OpenMDW-1.1, ASR model is CC-BY-4.0. + license: openmdw-1.1 + tags: + - parakeet + - parakeet-cpp + - nemotron + - sortformer + - asr + - diarization + - speaker-diarization + - speech-recognition + - stt + - gguf + - ggml + - quantized + overrides: + backend: parakeet-cpp + known_usecases: + - diarization + name: parakeet-cpp-nemotron-3-diarization-asr + options: + - asr_model:parakeet-cpp/tdt_ctc-110m-f16.gguf + parameters: + model: parakeet-cpp/nemotron-3-diarization-q8_0.gguf + files: + - filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf + sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22 + - filename: parakeet-cpp/tdt_ctc-110m-f16.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/tdt_ctc-110m-f16.gguf + sha256: 7f9a6376edde6a74592ace48b2ebdc27a1ac972d0be9dfcc29e668d99381faf1 +- name: parakeet-cpp-ced-tiny + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/mudler/ced-gguf + - https://huggingface.co/mispeech/ced-tiny + - https://github.com/mudler/parakeet.cpp + description: | + CED-Tiny sound event tagger, Q8_0 GGUF for the parakeet-cpp backend + (C++/ggml, loaded through third_party/ced.cpp). Served through + /v1/audio/classification: 10 s windows are scored and averaged over the + clip, then sorted by score with threshold and top_k applied. Smallest and + fastest of the CED sizes; use ced-base for higher accuracy. + license: apache-2.0 + tags: + - parakeet-cpp + - ced + - sound-classification + - audio-tagging + - gguf + - ggml + - quantized + overrides: + backend: parakeet-cpp + known_usecases: + - sound_classification + name: parakeet-cpp-ced-tiny + parameters: + model: parakeet-cpp/ced-tiny-q8_0.gguf + files: + - filename: parakeet-cpp/ced-tiny-q8_0.gguf + uri: huggingface://mudler/ced-gguf/ced-tiny-q8_0.gguf + sha256: 48bee4e2fc3cc85d7806e03471db24e77fda6c2a2e81ffe9ef67caebaf2bd674 +- name: parakeet-cpp-ced-base + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/mudler/ced-gguf + - https://huggingface.co/mispeech/ced-base + - https://github.com/mudler/parakeet.cpp + description: | + CED-Base sound event tagger, Q8_0 GGUF for the parakeet-cpp backend + (C++/ggml, loaded through third_party/ced.cpp). Served through + /v1/audio/classification: 10 s windows are scored and averaged over the + clip, then sorted by score with threshold and top_k applied. Larger and + more accurate than ced-tiny, still CPU-friendly. + license: apache-2.0 + tags: + - parakeet-cpp + - ced + - sound-classification + - audio-tagging + - gguf + - ggml + - quantized + overrides: + backend: parakeet-cpp + known_usecases: + - sound_classification + name: parakeet-cpp-ced-base + parameters: + model: parakeet-cpp/ced-base-q8_0.gguf + files: + - filename: parakeet-cpp/ced-base-q8_0.gguf + uri: huggingface://mudler/ced-gguf/ced-base-q8_0.gguf + sha256: bd34a7710169f0047fea17267965d211f967828ab25ba6fb9d3768481393f6e2 +- name: parakeet-cpp-realtime-scene + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/mudler/parakeet-cpp-gguf + - https://huggingface.co/mudler/ced-gguf + - https://huggingface.co/nvidia/parakeet_realtime_eou_120m-v1 + - https://huggingface.co/nvidia/Nemotron-3-Diarization + - https://huggingface.co/mispeech/ced-tiny + - https://github.com/mudler/parakeet.cpp + description: | + Cache-aware streaming RNNT FastConformer with end-of-utterance (EOU) + detection, 120M, paired with Nemotron-3-Diarization and CED-Tiny through + the diarization_model and sound_model options. F16/Q8_0 GGUF for the + parakeet-cpp backend (C++/ggml port of NVIDIA NeMo). Use with streaming + transcription: while a turn is live, closed speaker segments and sound + events are surfaced alongside the ASR text (realtime + conversation.item.input_audio_transcription.segment and + conversation.item.sound_detection events). Live speaker/sound events only + fire during speech turns under semantic_vad; sounds between turns are not + seen by this path. License per model: transcription model NVIDIA Open + Model License, diarization model OpenMDW-1.1, CED-Tiny Apache-2.0. + license: nvidia-open-model-license + tags: + - parakeet + - parakeet-cpp + - nemotron + - sortformer + - ced + - asr + - speech-recognition + - diarization + - sound-classification + - streaming + - realtime + - stt + - gguf + - ggml + overrides: + backend: parakeet-cpp + known_usecases: + - transcript + name: parakeet-cpp-realtime-scene + options: + - diarization_model:parakeet-cpp/nemotron-3-diarization-q8_0.gguf + - sound_model:parakeet-cpp/ced-tiny-q8_0.gguf + parameters: + model: parakeet-cpp/realtime_eou_120m-v1-f16.gguf + files: + - filename: parakeet-cpp/realtime_eou_120m-v1-f16.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/realtime_eou_120m-v1-f16.gguf + sha256: d1a2b12f12b8a096a57499c9111ed13b442a2b786e17a292c168be45088f0edc + - filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf + sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22 + - filename: parakeet-cpp/ced-tiny-q8_0.gguf + uri: huggingface://mudler/ced-gguf/ced-tiny-q8_0.gguf + sha256: 48bee4e2fc3cc85d7806e03471db24e77fda6c2a2e81ffe9ef67caebaf2bd674 +- name: parakeet-cpp-realtime-scene-tdt + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/mudler/parakeet-cpp-gguf + - https://huggingface.co/mudler/ced-gguf + - https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3 + - https://huggingface.co/nvidia/Nemotron-3-Diarization + - https://huggingface.co/mispeech/ced-tiny + - https://github.com/mudler/parakeet.cpp + description: | + Parakeet TDT 0.6B v3 (multilingual, 25 European languages) paired with + Nemotron-3-Diarization and CED-Tiny through the diarization_model and + sound_model options: one parakeet-cpp backend transcribes, labels speakers + and tags sound events. GGUF for the parakeet-cpp backend (C++/ggml port of + NVIDIA NeMo). TDT is not a streaming model, so in a realtime pipeline use + it with server_vad: set it as both transcription and sound_detection and + turn on pipeline.diarization, and each committed turn gets speaker segments + (conversation.item.input_audio_transcription.segment, with text) and + sound tags (conversation.item.sound_detection). Also labels speakers on + /v1/audio/transcriptions. Speaker labels are per turn. License per model: + transcription model CC-BY-4.0, diarization model OpenMDW-1.1, CED-Tiny + Apache-2.0. + license: cc-by-4.0 + tags: + - parakeet + - parakeet-cpp + - nemotron + - sortformer + - ced + - asr + - speech-recognition + - diarization + - sound-classification + - multilingual + - realtime + - stt + - gguf + - ggml + overrides: + backend: parakeet-cpp + known_usecases: + - transcript + - diarization + - sound_classification + name: parakeet-cpp-realtime-scene-tdt + options: + - diarization_model:parakeet-cpp/nemotron-3-diarization-q8_0.gguf + - sound_model:parakeet-cpp/ced-tiny-q8_0.gguf + parameters: + model: parakeet-cpp/tdt-0.6b-v3-f16.gguf + files: + - filename: parakeet-cpp/tdt-0.6b-v3-f16.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/tdt-0.6b-v3-f16.gguf + sha256: 8ba47343e1e919895aca90e099150a01ed203ee0942d8ed31e27295efc5abb22 + - filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf + sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22 + - filename: parakeet-cpp/ced-tiny-q8_0.gguf + uri: huggingface://mudler/ced-gguf/ced-tiny-q8_0.gguf + sha256: 48bee4e2fc3cc85d7806e03471db24e77fda6c2a2e81ffe9ef67caebaf2bd674 +- name: parakeet-cpp-realtime-scene-base + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/mudler/parakeet-cpp-gguf + - https://huggingface.co/mudler/ced-gguf + - https://huggingface.co/nvidia/parakeet_realtime_eou_120m-v1 + - https://huggingface.co/nvidia/Nemotron-3-Diarization + - https://huggingface.co/mispeech/ced-base + - https://github.com/mudler/parakeet.cpp + description: | + Cache-aware streaming RNNT FastConformer with end-of-utterance (EOU) + detection, 120M, paired with Nemotron-3-Diarization and CED-Base (86M, the largest CED; + more confident sound tags than CED-Tiny at a small extra cost: on CPU the + live diarization + sound stream runs at 0.125 of real time against 0.103 + with CED-Tiny) through + the diarization_model and sound_model options. F16/Q8_0 GGUF for the + parakeet-cpp backend (C++/ggml port of NVIDIA NeMo). Use with streaming + transcription: while a turn is live, closed speaker segments and sound + events are surfaced alongside the ASR text (realtime + conversation.item.input_audio_transcription.segment and + conversation.item.sound_detection events). Live speaker/sound events only + fire during speech turns under semantic_vad; sounds between turns are not + seen by this path. License per model: transcription model NVIDIA Open + Model License, diarization model OpenMDW-1.1, CED-Base Apache-2.0. + license: nvidia-open-model-license + tags: + - parakeet + - parakeet-cpp + - nemotron + - sortformer + - ced + - asr + - speech-recognition + - diarization + - sound-classification + - streaming + - realtime + - stt + - gguf + - ggml + overrides: + backend: parakeet-cpp + known_usecases: + - transcript + name: parakeet-cpp-realtime-scene-base + options: + - diarization_model:parakeet-cpp/nemotron-3-diarization-q8_0.gguf + - sound_model:parakeet-cpp/ced-base-q8_0.gguf + parameters: + model: parakeet-cpp/realtime_eou_120m-v1-f16.gguf + files: + - filename: parakeet-cpp/realtime_eou_120m-v1-f16.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/realtime_eou_120m-v1-f16.gguf + sha256: d1a2b12f12b8a096a57499c9111ed13b442a2b786e17a292c168be45088f0edc + - filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf + sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22 + - filename: parakeet-cpp/ced-base-q8_0.gguf + uri: huggingface://mudler/ced-gguf/ced-base-q8_0.gguf + sha256: bd34a7710169f0047fea17267965d211f967828ab25ba6fb9d3768481393f6e2 +- name: parakeet-cpp-realtime-scene-tdt-base + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/mudler/parakeet-cpp-gguf + - https://huggingface.co/mudler/ced-gguf + - https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3 + - https://huggingface.co/nvidia/Nemotron-3-Diarization + - https://huggingface.co/mispeech/ced-base + - https://github.com/mudler/parakeet.cpp + description: | + Parakeet TDT 0.6B v3 (multilingual, 25 European languages) paired with + Nemotron-3-Diarization and CED-Base (86M, the largest CED; about 0.03 s of + CPU per second of audio per committed turn, against 0.005 for CED-Tiny) + through the diarization_model and + sound_model options: one parakeet-cpp backend transcribes, labels speakers + and tags sound events. GGUF for the parakeet-cpp backend (C++/ggml port of + NVIDIA NeMo). TDT is not a streaming model, so in a realtime pipeline use + it with server_vad: set it as both transcription and sound_detection and + turn on pipeline.diarization, and each committed turn gets speaker segments + (conversation.item.input_audio_transcription.segment, with text) and + sound tags (conversation.item.sound_detection). Also labels speakers on + /v1/audio/transcriptions. Speaker labels are per turn. License per model: + transcription model CC-BY-4.0, diarization model OpenMDW-1.1, CED-Tiny + Apache-2.0. + license: cc-by-4.0 + tags: + - parakeet + - parakeet-cpp + - nemotron + - sortformer + - ced + - asr + - speech-recognition + - diarization + - sound-classification + - multilingual + - realtime + - stt + - gguf + - ggml + overrides: + backend: parakeet-cpp + known_usecases: + - transcript + - diarization + - sound_classification + name: parakeet-cpp-realtime-scene-tdt-base + options: + - diarization_model:parakeet-cpp/nemotron-3-diarization-q8_0.gguf + - sound_model:parakeet-cpp/ced-base-q8_0.gguf + parameters: + model: parakeet-cpp/tdt-0.6b-v3-f16.gguf + files: + - filename: parakeet-cpp/tdt-0.6b-v3-f16.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/tdt-0.6b-v3-f16.gguf + sha256: 8ba47343e1e919895aca90e099150a01ed203ee0942d8ed31e27295efc5abb22 + - filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf + sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22 + - filename: parakeet-cpp/ced-base-q8_0.gguf + uri: huggingface://mudler/ced-gguf/ced-base-q8_0.gguf + sha256: bd34a7710169f0047fea17267965d211f967828ab25ba6fb9d3768481393f6e2 - name: moss-transcribe-cpp-0.9b url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: From 84c83a70dd9499d93e3837909bdbe8baef3e3e2a Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Wed, 30 Sep 2026 11:26:22 +0000 Subject: [PATCH 12/33] feat(config): add systemone usecase for decision models Explicit-only, reserving usecase like score and token_classify: a declared list is authoritative and the heuristic never guesses it. vllm-cpp now lists systemone and vision as possible usecases. Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto --- core/config/backend_capabilities.go | 10 +++++++-- core/config/gguf.go | 4 ++-- core/config/model_config.go | 24 +++++++++++++++++---- core/config/model_config_test.go | 33 +++++++++++++++++++++++++++++ 4 files changed, 63 insertions(+), 8 deletions(-) diff --git a/core/config/backend_capabilities.go b/core/config/backend_capabilities.go index b48266ead..b4c40248c 100644 --- a/core/config/backend_capabilities.go +++ b/core/config/backend_capabilities.go @@ -35,6 +35,7 @@ const ( UsecaseSpeakerRecognition = "speaker_recognition" UsecaseTokenClassify = "token_classify" UsecaseScore = "score" + UsecaseSystemOne = "systemone" ) // GRPCMethod identifies a Backend service RPC from backend.proto. @@ -216,6 +217,11 @@ var UsecaseInfoMap = map[string]UsecaseInfo{ GRPCMethod: MethodScore, Description: "Joint log-probability scoring of candidate continuations via the Score RPC. Declared explicitly via known_usecases and usable alongside generation usecases.", }, + UsecaseSystemOne: { + Flag: FLAG_SYSTEMONE, + GRPCMethod: MethodScore, + Description: "SystemOne decision API (POST /v1/systemone): typed choice, noul and score questions over a state text, answered by a non-generative decision model through the Score RPC (question_type systemone). Declared explicitly via known_usecases.", + }, } // BackendCapability describes which gRPC methods and usecases a backend supports. @@ -349,10 +355,10 @@ var BackendCapabilities = map[string]BackendCapability{ // model returns an error rather than silent garbage. "vllm-cpp": { GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateVideo, MethodTokenClassify, MethodScore}, - PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVideo, UsecaseTokenClassify, UsecaseScore}, + PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVision, UsecaseVideo, UsecaseTokenClassify, UsecaseScore, UsecaseSystemOne}, DefaultUsecases: []string{UsecaseChat}, AcceptsImages: true, - Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and kev/laya decision pipelines", + Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and SystemOne decision models (kev, laya, CLM, GLiNER2.5-Decide, xor, nimble)", }, "vllm-omni": { GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateImage, MethodGenerateVideo, MethodTTS}, diff --git a/core/config/gguf.go b/core/config/gguf.go index e5f3bc5b4..f9b8c748f 100644 --- a/core/config/gguf.go +++ b/core/config/gguf.go @@ -16,14 +16,14 @@ import ( // reservedNonChatModel reports whether the operator reserved this model for an // internal primitive — the router score classifier or the PII NER -// token_classify tier. Such a model has no chat template and must not be +// token_classify tier, or a SystemOne decision head. Such a model has no chat template and must not be // given the generative-chat defaults the GGUF importer otherwise applies // (FLAG_CHAT, jinja templating): surfacing it in chat pickers defeats the // reservation. Operators who do want a combined model declare both usecases // explicitly — the combination is valid. func reservedNonChatModel(cfg *ModelConfig) bool { return cfg.KnownUsecases != nil && - (*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY)) != 0 + (*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY|FLAG_SYSTEMONE)) != 0 } // genAudioEncoderKey is the mmproj metadata flag llama.cpp's mtmd writes for a diff --git a/core/config/model_config.go b/core/config/model_config.go index c8502fae5..a78e1db6c 100644 --- a/core/config/model_config.go +++ b/core/config/model_config.go @@ -2056,6 +2056,13 @@ const ( FLAG_3D ModelConfigUsecase = 0b100000000000000000000000 FLAG_3D_ANIMATION ModelConfigUsecase = 1 << 24 + // Marks a model as wired for the SystemOne decision API (POST + // /v1/systemone: typed choice / noul / score questions over a state). + // Explicit only, like FLAG_SCORE: a decision model never generates + // text, so guessing chat or embeddings for it would surface it in + // pickers it cannot serve. + FLAG_SYSTEMONE ModelConfigUsecase = 1 << 25 + // Common Subsets FLAG_LLM ModelConfigUsecase = FLAG_CHAT | FLAG_COMPLETION | FLAG_EDIT ) @@ -2118,6 +2125,7 @@ func GetAllModelConfigUsecases() map[string]ModelConfigUsecase { "FLAG_TOKEN_CLASSIFY": FLAG_TOKEN_CLASSIFY, "FLAG_3D": FLAG_3D, "FLAG_3D_ANIMATION": FLAG_3D_ANIMATION, + "FLAG_SYSTEMONE": FLAG_SYSTEMONE, } } @@ -2146,9 +2154,9 @@ func GetUsecasesFromYAML(input []string) *ModelConfigUsecase { // // Declared known_usecases are normally additive — the guessing heuristic // still adds whatever it can infer from backend/templates. The exceptions -// are FLAG_SCORE and FLAG_TOKEN_CLASSIFY: when the operator declared -// either, they reserved the model for an internal direct-decode primitive -// (the router classifier, or the PII NER tier). Letting GuessUsecases +// are FLAG_SCORE, FLAG_TOKEN_CLASSIFY and FLAG_SYSTEMONE: when the operator +// declared any of them, they reserved the model for a direct-decode primitive +// (the router classifier, the PII NER tier, or a SystemOne decision head). Letting GuessUsecases // paint chat/completion/embeddings on top would surface it in pickers it // was deliberately kept out of. So a declared score or token_classify // list is authoritative; declare the generation usecases explicitly @@ -2158,7 +2166,7 @@ func (c *ModelConfig) HasUsecases(u ModelConfigUsecase) bool { if (u & *c.KnownUsecases) == u { return true } - if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY)) != 0 { + if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY | FLAG_SYSTEMONE)) != 0 { return false } } @@ -2381,6 +2389,14 @@ func (c *ModelConfig) GuessUsecases(u ModelConfigUsecase) bool { return false } + if (u & FLAG_SYSTEMONE) == FLAG_SYSTEMONE { + // No heuristic: SystemOne intent is a deliberate operator choice + // (the model is a non-generative decision head), so + // HasUsecases(FLAG_SYSTEMONE) is true only when KnownUsecases + // declares it explicitly. + return false + } + return true } diff --git a/core/config/model_config_test.go b/core/config/model_config_test.go index 828160fe1..1873cfef8 100644 --- a/core/config/model_config_test.go +++ b/core/config/model_config_test.go @@ -955,3 +955,36 @@ var _ = Describe("ModelConfig alias", func() { Expect(err).To(MatchError(ContainSubstring("alias"))) }) }) + +var _ = Describe("systemone usecase", func() { + // A decision model never generates text, so a declared systemone list + // must stay authoritative and the heuristic must never guess the flag. + It("is authoritative when declared and never guessed", func() { + declared := GetUsecasesFromYAML([]string{"systemone"}) + Expect(declared).NotTo(BeNil()) + Expect(*declared).NotTo(Equal(FLAG_ANY)) + + cfg := ModelConfig{ + Name: "laya", + Backend: "vllm-cpp", + KnownUsecases: declared, + TemplateConfig: TemplateConfig{ + Chat: "inherited from chatml", + ChatMessage: "inherited from chatml", + Completion: "inherited from chatml", + }, + } + Expect(cfg.HasUsecases(*declared)).To(BeTrue()) + Expect(cfg.HasUsecases(FLAG_CHAT)).To(BeFalse()) + Expect(cfg.HasUsecases(FLAG_COMPLETION)).To(BeFalse()) + Expect(cfg.HasUsecases(FLAG_EMBEDDINGS)).To(BeFalse()) + + undeclared := ModelConfig{Name: "laya", Backend: "vllm-cpp"} + Expect(undeclared.HasUsecases(*declared)).To(BeFalse()) + }) + + It("is a reserved usecase for the GGUF importer chat-default guard", func() { + declared := GetUsecasesFromYAML([]string{"systemone"}) + Expect(reservedNonChatModel(&ModelConfig{Backend: "vllm-cpp", KnownUsecases: declared})).To(BeTrue()) + }) +}) From e03cf8dac828ee112ae84beab38997139a92e07a Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Wed, 30 Sep 2026 11:27:40 +0000 Subject: [PATCH 13/33] feat(systemone): refuse models that do not declare the usecase A chat-only model now gets a 400 naming known_usecases: [systemone] instead of a backend error. Configs declaring no usecases and token_classify models stay allowed so existing laya and GLiNER setups keep working. Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto --- core/http/endpoints/localai/systemone.go | 38 +++++++++++++++++++ .../endpoints/localai/systemone_gate_test.go | 34 +++++++++++++++++ 2 files changed, 72 insertions(+) create mode 100644 core/http/endpoints/localai/systemone_gate_test.go diff --git a/core/http/endpoints/localai/systemone.go b/core/http/endpoints/localai/systemone.go index e272e1b8b..435414d39 100644 --- a/core/http/endpoints/localai/systemone.go +++ b/core/http/endpoints/localai/systemone.go @@ -371,6 +371,35 @@ func systemOneError(c echo.Context, status int, msg string) error { }) } +// systemOneModelAllowed keeps chat and embedding models out of the decision +// API with an actionable error instead of a backend failure. A config that +// declares no usecases predates the flag and stays allowed, and a +// token_classify model is allowed because the NER path serves it. +func systemOneModelAllowed(cfg config.ModelConfig) error { + if cfg.KnownUsecases == nil { + return nil + } + if *cfg.KnownUsecases&(config.FLAG_SYSTEMONE|config.FLAG_TOKEN_CLASSIFY) != 0 { + return nil + } + return fmt.Errorf("model %q does not declare the systemone usecase (known_usecases: [systemone])", cfg.Name) +} + +// checkSystemOneModel applies systemOneModelAllowed to a model looked up by +// name. An unknown model passes here so the existing not-found handling +// downstream keeps its status code. +func checkSystemOneModel(app *application.Application, modelName string) error { + cl := app.ModelConfigLoader() + if cl == nil { + return nil + } + cfg, ok := cl.GetModelConfig(modelName) + if !ok { + return nil + } + return systemOneModelAllowed(cfg) +} + // backendSupportsScore reports whether the named backend implements the // Score gRPC RPC. vllm-cpp does (kev/laya decision pipeline and cua-s1-forms // scoring via the unified vllm_decide C ABI); other backends fall through to @@ -408,6 +437,9 @@ func SystemOneEndpoint(app *application.Application) echo.HandlerFunc { if req.Model == "" { return systemOneError(c, http.StatusBadRequest, "model is required") } + if err := checkSystemOneModel(app, req.Model); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } // vllm-cpp models (kev/laya) implement the decision pipeline natively // via the vllm_decide C ABI. Forward the raw request JSON through the // Score RPC and return the backend's response as-is. @@ -474,6 +506,9 @@ func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc { if req.Request.Model == "" { return systemOneError(c, http.StatusBadRequest, "model is required") } + if err := checkSystemOneModel(app, req.Request.Model); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } if req.Question == "" { return systemOneError(c, http.StatusBadRequest, "question is required") } @@ -610,6 +645,9 @@ func SystemOneSeparateEndpoint(app *application.Application) echo.HandlerFunc { if req.Model == "" { return systemOneError(c, http.StatusBadRequest, "model is required") } + if err := checkSystemOneModel(app, req.Model); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } parsed, err := parseSystemOneRequest(&req) if err != nil { return systemOneError(c, http.StatusBadRequest, err.Error()) diff --git a/core/http/endpoints/localai/systemone_gate_test.go b/core/http/endpoints/localai/systemone_gate_test.go new file mode 100644 index 000000000..b7857da56 --- /dev/null +++ b/core/http/endpoints/localai/systemone_gate_test.go @@ -0,0 +1,34 @@ +package localai + +import ( + "github.com/mudler/LocalAI/core/config" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("systemOneModelAllowed", func() { + mk := func(usecases ...string) config.ModelConfig { + return config.ModelConfig{ + Name: "m", + Backend: "vllm-cpp", + KnownUsecases: config.GetUsecasesFromYAML(usecases), + } + } + + It("accepts a declared systemone model", func() { + Expect(systemOneModelAllowed(mk("systemone"))).To(Succeed()) + }) + + It("accepts a token_classify model, which the NER path serves", func() { + Expect(systemOneModelAllowed(mk("token_classify"))).To(Succeed()) + }) + + It("keeps configs that declare no usecases working", func() { + Expect(systemOneModelAllowed(config.ModelConfig{Name: "laya", Backend: "vllm-cpp"})).To(Succeed()) + }) + + It("refuses a chat-only model with an actionable message", func() { + Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [systemone]"))) + }) +}) From 36846466e48acb041b0634472648baa21f37e764 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Wed, 30 Sep 2026 11:29:23 +0000 Subject: [PATCH 14/33] feat(ui): show the systemone usecase on installed models Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto --- core/http/react-ui/e2e/models-lifecycle.spec.js | 13 +++++++++++++ core/http/react-ui/public/locales/de/models.json | 2 +- core/http/react-ui/public/locales/en/models.json | 2 +- core/http/react-ui/public/locales/es/models.json | 2 +- core/http/react-ui/public/locales/id/models.json | 2 +- core/http/react-ui/public/locales/it/models.json | 2 +- core/http/react-ui/public/locales/ko/models.json | 2 +- core/http/react-ui/public/locales/pt-BR/models.json | 2 +- core/http/react-ui/public/locales/zh-CN/models.json | 2 +- core/http/react-ui/src/pages/InstalledModels.jsx | 3 ++- core/http/react-ui/src/utils/capabilities.js | 1 + 11 files changed, 24 insertions(+), 9 deletions(-) diff --git a/core/http/react-ui/e2e/models-lifecycle.spec.js b/core/http/react-ui/e2e/models-lifecycle.spec.js index 3a8d125a5..6587bff8b 100644 --- a/core/http/react-ui/e2e/models-lifecycle.spec.js +++ b/core/http/react-ui/e2e/models-lifecycle.spec.js @@ -172,6 +172,19 @@ test.describe('Models lifecycle', () => { await expect(installedPane(page)).toContainText('Worker one') }) + test('shows the systemone use case on a decision model', async ({ page }) => { + await page.route('**/api/models/capabilities', route => route.fulfill({ + contentType: 'application/json', + body: JSON.stringify({ + data: [...installedModels, { id: 'decider', backend: 'vllm-cpp', capabilities: ['FLAG_SYSTEMONE'] }], + }), + })) + await page.goto('/app/models?view=installed&model=decider') + + await expect(installedPane(page)).toContainText('decider') + await expect(installedPane(page)).toContainText('SystemOne') + }) + test('stops a running model with confirmation', async ({ page }) => { await page.goto('/app/models?view=installed&model=alpha') diff --git a/core/http/react-ui/public/locales/de/models.json b/core/http/react-ui/public/locales/de/models.json index d88cb8c70..d5c58a6b9 100644 --- a/core/http/react-ui/public/locales/de/models.json +++ b/core/http/react-ui/public/locales/de/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/en/models.json b/core/http/react-ui/public/locales/en/models.json index a2150e785..b5ab38303 100644 --- a/core/http/react-ui/public/locales/en/models.json +++ b/core/http/react-ui/public/locales/en/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/es/models.json b/core/http/react-ui/public/locales/es/models.json index 989189850..27fc03752 100644 --- a/core/http/react-ui/public/locales/es/models.json +++ b/core/http/react-ui/public/locales/es/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/id/models.json b/core/http/react-ui/public/locales/id/models.json index 1dee74031..67ea168f7 100644 --- a/core/http/react-ui/public/locales/id/models.json +++ b/core/http/react-ui/public/locales/id/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/it/models.json b/core/http/react-ui/public/locales/it/models.json index edcc1b587..0f4c80e56 100644 --- a/core/http/react-ui/public/locales/it/models.json +++ b/core/http/react-ui/public/locales/it/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/ko/models.json b/core/http/react-ui/public/locales/ko/models.json index b2a20016e..74874ed70 100644 --- a/core/http/react-ui/public/locales/ko/models.json +++ b/core/http/react-ui/public/locales/ko/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/pt-BR/models.json b/core/http/react-ui/public/locales/pt-BR/models.json index 26e567a44..8a795411e 100644 --- a/core/http/react-ui/public/locales/pt-BR/models.json +++ b/core/http/react-ui/public/locales/pt-BR/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/zh-CN/models.json b/core/http/react-ui/public/locales/zh-CN/models.json index 40f78260e..e130ff78a 100644 --- a/core/http/react-ui/public/locales/zh-CN/models.json +++ b/core/http/react-ui/public/locales/zh-CN/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/src/pages/InstalledModels.jsx b/core/http/react-ui/src/pages/InstalledModels.jsx index 6013d984f..c607a61ac 100644 --- a/core/http/react-ui/src/pages/InstalledModels.jsx +++ b/core/http/react-ui/src/pages/InstalledModels.jsx @@ -22,7 +22,7 @@ import { CAP_CHAT, CAP_COMPLETION, CAP_IMAGE, CAP_VIDEO, CAP_TTS, CAP_TRANSCRIPT, CAP_SOUND_GENERATION, CAP_FACE_RECOGNITION, CAP_SPEAKER_RECOGNITION, CAP_EMBEDDINGS, CAP_RERANK, - CAP_VAD, CAP_SCORE, + CAP_VAD, CAP_SCORE, CAP_SYSTEMONE, } from '../utils/capabilities' const USE_CASES = [ @@ -39,6 +39,7 @@ const USE_CASES = [ { cap: CAP_RERANK, labelKey: 'rerank' }, { cap: CAP_VAD, labelKey: 'vad' }, { cap: CAP_SCORE, labelKey: 'score' }, + { cap: CAP_SYSTEMONE, labelKey: 'systemone' }, ] export function modelUseCases(model) { diff --git a/core/http/react-ui/src/utils/capabilities.js b/core/http/react-ui/src/utils/capabilities.js index f01cc781c..0775ef8d9 100644 --- a/core/http/react-ui/src/utils/capabilities.js +++ b/core/http/react-ui/src/utils/capabilities.js @@ -29,4 +29,5 @@ export const CAP_SPEAKER_RECOGNITION = 'FLAG_SPEAKER_RECOGNITION' export const CAP_AUDIO_TRANSFORM = 'FLAG_AUDIO_TRANSFORM' export const CAP_REALTIME_AUDIO = 'FLAG_REALTIME_AUDIO' export const CAP_SCORE = 'FLAG_SCORE' +export const CAP_SYSTEMONE = 'FLAG_SYSTEMONE' export const CAP_TOKEN_CLASSIFY = 'FLAG_TOKEN_CLASSIFY' From b4852d62d20a4b2489990b2c845e469c38214ff0 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Wed, 30 Sep 2026 11:32:03 +0000 Subject: [PATCH 15/33] feat(systemone): register the decisions API on auth and instructions Adds a default-on systemone route feature for the three /v1/systemone routes and an /api/instructions area for them. No MCP tool is added: the endpoints run inference and are not admin install/edit actions, and the route-map test still passes. Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto --- core/http/auth/features.go | 6 +++++ core/http/auth/features_systemone_test.go | 24 +++++++++++++++++++ core/http/auth/permissions.go | 3 ++- .../endpoints/localai/api_instructions.go | 6 +++++ .../localai/api_instructions_test.go | 14 ++++++++++- 5 files changed, 51 insertions(+), 2 deletions(-) create mode 100644 core/http/auth/features_systemone_test.go diff --git a/core/http/auth/features.go b/core/http/auth/features.go index 45411f824..94ae73f4d 100644 --- a/core/http/auth/features.go +++ b/core/http/auth/features.go @@ -71,6 +71,11 @@ var RouteFeatureRegistry = []RouteFeature{ // Detection {"POST", "/v1/detection", FeatureDetection}, + // SystemOne decision API + {"POST", "/v1/systemone", FeatureSystemOne}, + {"POST", "/v1/systemone/permute", FeatureSystemOne}, + {"POST", "/v1/systemone/separate", FeatureSystemOne}, + // Face recognition {"POST", "/v1/face/verify", FeatureFaceRecognition}, {"POST", "/v1/face/analyze", FeatureFaceRecognition}, @@ -209,5 +214,6 @@ func APIFeatureMetas() []FeatureMeta { {FeatureVoiceRecognition, "Voice Recognition", true}, {FeatureAudioTransform, "Audio Transform", true}, {FeaturePIIFilter, "PII Analyze / Redact", true}, + {FeatureSystemOne, "SystemOne Decisions", true}, } } diff --git a/core/http/auth/features_systemone_test.go b/core/http/auth/features_systemone_test.go new file mode 100644 index 000000000..31cfc6cb7 --- /dev/null +++ b/core/http/auth/features_systemone_test.go @@ -0,0 +1,24 @@ +package auth_test + +import ( + . "github.com/mudler/LocalAI/core/http/auth" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("SystemOne feature registration", func() { + It("gates the three decision routes behind one default-on API feature", func() { + Expect(APIFeatures).To(ContainElement(FeatureSystemOne)) + + patterns := []string{} + for _, route := range RouteFeatureRegistry { + if route.Feature == FeatureSystemOne { + Expect(route.Method).To(Equal("POST")) + patterns = append(patterns, route.Pattern) + } + } + Expect(patterns).To(ConsistOf("/v1/systemone", "/v1/systemone/permute", "/v1/systemone/separate")) + + Expect(APIFeatureMetas()).To(ContainElement(FeatureMeta{Key: FeatureSystemOne, Label: "SystemOne Decisions", DefaultValue: true})) + }) +}) diff --git a/core/http/auth/permissions.go b/core/http/auth/permissions.go index 95e76f572..c01f6e72b 100644 --- a/core/http/auth/permissions.go +++ b/core/http/auth/permissions.go @@ -59,6 +59,7 @@ const ( FeatureFaceRecognition = "face_recognition" FeatureVoiceRecognition = "voice_recognition" FeatureAudioTransform = "audio_transform" + FeatureSystemOne = "systemone" // FeaturePIIFilter gates the synchronous PII analyze/redact service // (POST /api/pii/{analyze,redact}). Default ON like the other API // features; the admin-only events log is gated separately in-handler. @@ -78,7 +79,7 @@ var APIFeatures = []string{ FeatureVAD, FeatureDetection, FeatureVideo, Feature3D, FeatureEmbeddings, FeatureSound, FeatureRealtime, FeatureModeration, FeatureRerank, FeatureTokenize, FeatureMCP, FeatureStores, FeatureFaceRecognition, FeatureVoiceRecognition, FeatureAudioTransform, - FeaturePIIFilter, + FeaturePIIFilter, FeatureSystemOne, } // AllFeatures lists all known features (used by UI and validation). diff --git a/core/http/endpoints/localai/api_instructions.go b/core/http/endpoints/localai/api_instructions.go index 8d0ea6d2f..dc60a3c21 100644 --- a/core/http/endpoints/localai/api_instructions.go +++ b/core/http/endpoints/localai/api_instructions.go @@ -105,6 +105,12 @@ var instructionDefs = []instructionDef{ Tags: []string{"voice-recognition"}, Intro: "Voice (speaker) recognition — the audio analog to /v1/face/*. Use /v1/voice/verify for 1:1 speaker comparison, /v1/voice/identify for 1:N match against the registered store, /v1/voice/{register,forget} to manage that store, /v1/voice/embed for a raw speaker-encoder vector, and /v1/voice/analyze for age / gender / emotion inferred from speech. Registrations are in-memory by default and lost on restart. Audio inputs accept URL, base64, or data-URI; /v1/embeddings remains text-only.", }, + { + Name: "systemone", + Description: "Typed decisions (choice, noul, score) over a state text with calibrated confidence", + Tags: []string{"systemone"}, + Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { : { type: choice|noul|score, instructions, criteria } } }. The model must declare known_usecases: [systemone] (or token_classify for the zero-shot NER path); a config that declares no usecases keeps working. Responses carry per-question answers with confidence and probabilities plus token usage. The wire shape matches Ollama's /v1/systemone.", + }, { Name: "branding", Description: "Whitelabel the instance: configure name, tagline, logo, and favicon", diff --git a/core/http/endpoints/localai/api_instructions_test.go b/core/http/endpoints/localai/api_instructions_test.go index f42e1c92d..727f4cb86 100644 --- a/core/http/endpoints/localai/api_instructions_test.go +++ b/core/http/endpoints/localai/api_instructions_test.go @@ -39,7 +39,7 @@ var _ = Describe("API Instructions Endpoints", func() { instructions, ok := resp["instructions"].([]any) Expect(ok).To(BeTrue()) - Expect(instructions).To(HaveLen(20)) + Expect(instructions).To(HaveLen(21)) // Verify each instruction has required fields and correct URL format for _, s := range instructions { @@ -82,6 +82,7 @@ var _ = Describe("API Instructions Endpoints", func() { "voice-library", "3d", "failover", + "systemone", )) }) }) @@ -136,6 +137,17 @@ var _ = Describe("API Instructions Endpoints", func() { Expect(string(body)).NotTo(ContainSubstring("/v1/3d/generations")) }) + It("should advertise the SystemOne decisions API", func() { + req := httptest.NewRequest(http.MethodGet, "/api/instructions/systemone", nil) + rec := httptest.NewRecorder() + app.ServeHTTP(rec, req) + + Expect(rec.Code).To(Equal(http.StatusOK)) + body, _ := io.ReadAll(rec.Body) + Expect(string(body)).To(ContainSubstring("POST /v1/systemone")) + Expect(string(body)).To(ContainSubstring("known_usecases: [systemone]")) + }) + It("should return JSON fragment when format=json", func() { req := httptest.NewRequest(http.MethodGet, "/api/instructions/chat-inference?format=json", nil) rec := httptest.NewRecorder() From c2af8d56ea916f11247ba3ee048fff8e72238a07 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Wed, 30 Sep 2026 11:37:13 +0000 Subject: [PATCH 16/33] feat(gallery): tag vllm-cpp entries by capability and add decision and vision models laya declares the systemone usecase instead of chat. The gated Qwen3.6 27B NVFP4 entries gain vision; the 35B-A3B entries gain it as experimental because image input is not token-gated. Adds GLiNER2.5-Decide and Qwen3-VL-4B. A guard test keeps capability tags and known_usecases in agreement for every vllm-cpp entry. Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto --- core/gallery/vllm_cpp_tags_test.go | 48 +++++++++++++ gallery/index.yaml | 104 ++++++++++++++++++++++++++++- 2 files changed, 151 insertions(+), 1 deletion(-) create mode 100644 core/gallery/vllm_cpp_tags_test.go diff --git a/core/gallery/vllm_cpp_tags_test.go b/core/gallery/vllm_cpp_tags_test.go new file mode 100644 index 000000000..799dd471e --- /dev/null +++ b/core/gallery/vllm_cpp_tags_test.go @@ -0,0 +1,48 @@ +package gallery_test + +import ( + "fmt" + "slices" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + "github.com/mudler/LocalAI/core/config" +) + +// A gallery tag that names a capability is what users filter on, and +// known_usecases is what the server routes on. When they disagree, the entry +// is listed under a filter it cannot serve, or is hidden from one it can. +var _ = Describe("gallery/index.yaml vllm-cpp capability tags", func() { + It("keeps capability tags and known_usecases in agreement", func() { + entries, err := loadGalleryIndex() + Expect(err).ToNot(HaveOccurred()) + + tagToFlag := map[string]config.ModelConfigUsecase{ + "systemone": config.FLAG_SYSTEMONE, + "vision": config.FLAG_VISION, + "token-classify": config.FLAG_TOKEN_CLASSIFY, + "scoring": config.FLAG_SCORE, + } + + var violations []string + seen := 0 + for i := range entries { + e := &entries[i] + if backend, _ := e.Overrides["backend"].(string); backend != "vllm-cpp" { + continue + } + seen++ + declared := e.GetKnownUsecases() + for tag, flag := range tagToFlag { + tagged := slices.Contains(e.Tags, tag) + has := declared != nil && *declared&flag == flag + if tagged != has { + violations = append(violations, fmt.Sprintf("%s: tag %q present=%v but known_usecases declares it=%v", e.Name, tag, tagged, has)) + } + } + } + Expect(seen).To(BeNumerically(">", 0)) + Expect(violations).To(BeEmpty()) + }) +}) diff --git a/gallery/index.yaml b/gallery/index.yaml index 71c572e6b..10276000a 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -19360,6 +19360,7 @@ - qwen3.6 - nvfp4 - vllm-cpp + - vision - tool-calling - reasoning - gpu @@ -19372,6 +19373,7 @@ known_usecases: - chat - completion + - vision # Tool calls and the split are parsed by the engine's own streaming # parsers, so LocalAI's Go-side grammar path stays out of the way. function: @@ -19429,6 +19431,7 @@ - qwen3.6 - nvfp4 - vllm-cpp + - vision - speculative-decoding - mtp - tool-calling @@ -19442,6 +19445,7 @@ known_usecases: - chat - completion + - vision function: grammar: disable: true @@ -19497,6 +19501,7 @@ - qwen3.6 - nvfp4 - vllm-cpp + - vision - speculative-decoding - dflash - tool-calling @@ -19510,6 +19515,7 @@ known_usecases: - chat - completion + - vision function: grammar: disable: true @@ -19557,6 +19563,9 @@ with roughly 3B parameters active per token, so it reads like a much larger model while costing about as much per token as a small one. + Image input is implemented in the engine but is not token-gated against + vLLM yet, so the vision usecase on this entry is experimental. + This is the engine's gated MoE checkpoint: token-for-token identical to vLLM over the 315-prompt battery on both the synchronous and asynchronous paths, at 0.92x to 0.97x vLLM's throughput from concurrency 1 to 32. @@ -19575,6 +19584,8 @@ - moe - nvfp4 - vllm-cpp + - vision + - experimental - tool-calling - reasoning - gpu @@ -19587,6 +19598,7 @@ known_usecases: - chat - completion + - vision function: grammar: disable: true @@ -19615,6 +19627,9 @@ description: | Qwen3.6-35B-A3B NVFP4 on vllm.cpp with MTP speculative decoding enabled. + Image input is implemented in the engine but is not token-gated against + vLLM yet, so the vision usecase on this entry is experimental. + The draft head ships inside the checkpoint's own mtp.* tensors, so there is no second model to download. On this model the speculative path is token-exact against speculation-off on both the synchronous and asynchronous @@ -19632,6 +19647,8 @@ - moe - nvfp4 - vllm-cpp + - vision + - experimental - speculative-decoding - mtp - tool-calling @@ -19645,6 +19662,7 @@ known_usecases: - chat - completion + - vision function: grammar: disable: true @@ -63645,7 +63663,7 @@ overrides: backend: vllm-cpp known_usecases: - - chat + - systemone parameters: model: convaiinnovations/laya artifacts: @@ -63654,6 +63672,90 @@ source: type: huggingface repo: convaiinnovations/laya +- name: gliner25-decide-vllm-cpp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/fastino/GLiNER2.5-Decide + - https://github.com/mudler/vllm.cpp + description: | + GLiNER2.5-Decide is a DeBERTa-v3-large encoder with a classification head + that answers typed decision questions over a state text in one forward + pass. It never generates text, so there is nothing to parse. + + In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine runs the + decision pipeline (choice, noul and score question types) through the + vllm_decide C ABI. This is the decision model, not the zero-shot NER model: + use the gliner2.5 entry for entity extraction. F32 weights, about 2 GB. + The weights are pinned to a revision so the entry keeps serving the + checkpoint it was checked against. + license: apache-2.0 + tags: + - decision + - systemone + - vllm-cpp + - cpu + - gpu + size: 2GB + last_checked: "2026-09-30" + overrides: + backend: vllm-cpp + known_usecases: + - systemone + parameters: + model: fastino/GLiNER2.5-Decide + artifacts: + - name: model + target: model + source: + type: huggingface + repo: fastino/GLiNER2.5-Decide + revision: 5a7adf72a23b4d311abae6ce050d7f0012bb3416 +- name: qwen3-vl-4b-vllm-cpp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct + - https://github.com/mudler/vllm.cpp + description: | + Qwen3-VL-4B-Instruct on vllm.cpp, in bf16: a small vision-language model + that takes images alongside text. In the engine's correctness battery the + image path matches vLLM token for token, and video input is a near tie. + + Roughly 9 GB of weights plus KV cache at the context configured here. It + runs where the flagship NVFP4 checkpoints cannot, including plain CPU. + license: apache-2.0 + tags: + - llm + - vision + - multimodal + - qwen + - qwen3-vl + - vllm-cpp + - cpu + - gpu + size: 9GB + last_checked: "2026-09-30" + overrides: + backend: vllm-cpp + known_usecases: + - chat + - completion + - vision + template: + use_tokenizer_template: true + context_size: 8192 + engine_args: + block_size: 32 + num_blocks: 512 + max_num_seqs: 4 + parameters: + model: Qwen/Qwen3-VL-4B-Instruct + artifacts: + - name: model + target: model + source: + type: huggingface + repo: Qwen/Qwen3-VL-4B-Instruct + revision: ebb281ec70b05090aa6165b016eac8ec08e71b17 - name: cua-s1-forms-vllm-cpp url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: From c7f278dd0d12468eadc273df6a4bcd686a68cbb3 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Wed, 30 Sep 2026 11:37:48 +0000 Subject: [PATCH 17/33] docs: document the systemone usecase and decisions API Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto --- docs/content/advanced/model-configuration.md | 4 +- docs/content/features/systemone.md | 103 +++++++++++++++++++ docs/content/features/vllm-cpp.md | 21 ++-- 3 files changed, 117 insertions(+), 11 deletions(-) create mode 100644 docs/content/features/systemone.md diff --git a/docs/content/advanced/model-configuration.md b/docs/content/advanced/model-configuration.md index 5cfa74ccd..10e71ef96 100644 --- a/docs/content/advanced/model-configuration.md +++ b/docs/content/advanced/model-configuration.md @@ -1066,7 +1066,9 @@ known_usecases: - embeddings ``` -Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `llm` (combination of CHAT, COMPLETION, EDIT). +Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `score`, `token_classify`, `systemone`, `llm` (combination of CHAT, COMPLETION, EDIT). + +`systemone` marks a model as a decision model for the [SystemOne API]({{% relref "features/systemone" %}}) (`POST /v1/systemone`). It is never guessed, and a model that declares it is not listed as a chat, completion or embeddings model. `token_classify` marks a model as a token-classification (NER) provider for the PII filter (e.g. an `openai-privacy-filter` GGUF). Declare it explicitly together with `embeddings: true` (the classifier loads via TOKEN_CLS pooling). It runs on the dedicated `privacy-filter` backend (`backend/cpp/privacy-filter`), a standalone GGML engine for the `openai-privacy-filter` family - separate from `llama-cpp`, which no longer carries the token-classification path. diff --git a/docs/content/features/systemone.md b/docs/content/features/systemone.md new file mode 100644 index 000000000..d58a3d436 --- /dev/null +++ b/docs/content/features/systemone.md @@ -0,0 +1,103 @@ ++++ +disableToc = false +title = "SystemOne decisions" +weight = 66 +url = "/features/systemone/" ++++ + +SystemOne is an API for fast, typed decisions. You send a piece of text (the +*state*) and a set of named questions. A decision model answers each question +with a value and a confidence, in one pass. The model does not generate text, so +there is nothing to parse and no free-form output to validate. + +The request and response shapes follow the [kev](https://github.com/jaredpalmer/kev) +project and match the `/v1/systemone` endpoint that Ollama added in 0.35. + +## Endpoints + +| Endpoint | Method | Description | +|---|---|---| +| `/v1/systemone` | POST | Answer all questions in one pass | +| `/v1/systemone/permute` | POST | Re-run one choice question under `n_perm` option orders | +| `/v1/systemone/separate` | POST | Answer each question in its own pass | + +## Question types + +| Type | Answer | Fields in the answer | +|---|---|---| +| `choice` | One option out of a named set | `choice`, `probabilities`, `confidence` | +| `noul` | Yes, no or unknown for a statement | `noul` (0 to 1), `entities` | +| `score` | One level on a scale | `score`, `legend`, `probabilities`, `confidence` | + +## Example + +```bash +curl http://localhost:8080/v1/systemone -H "Content-Type: application/json" -d '{ + "model": "laya-vllm-cpp", + "state": "My order arrived broken and I want my money back. This is the second time.", + "questions": { + "team": { + "type": "choice", + "instructions": "Which team should handle this ticket?", + "criteria": { + "billing": "Payments, invoices and refunds", + "shipping": "Delivery and damaged goods", + "product": "Questions about how the product works" + } + }, + "refund_requested": { + "type": "noul", + "instructions": "The customer explicitly asks for a refund" + }, + "urgency": { + "type": "score", + "instructions": "How urgent is this ticket?", + "criteria": ["not urgent", "somewhat urgent", "urgent", "critical"] + } + } +}' +``` + +Every answer carries a `confidence` value, and the response reports token usage +and `latency_ms`. + +## Choosing a model + +A model can serve SystemOne only if it is a decision model. Declare the usecase +in the model config: + +```yaml +name: laya +backend: vllm-cpp +known_usecases: + - systemone +parameters: + model: convaiinnovations/laya +``` + +`systemone` is never guessed, and a model that declares it is not listed as a +chat, completion or embeddings model. A model that declares usecases without +`systemone` or `token_classify` gets a `400` from these endpoints that names the +missing usecase. A config that declares no usecases at all keeps working, so +setups that predate the flag are not broken. Models that declare `token_classify` +are served by the zero-shot NER path. + +Install one from the gallery and filter on the `systemone` tag: + +| Gallery entry | Model | Notes | +|---|---|---| +| `laya-vllm-cpp` | Laya | ModernBERT-large, non-autoregressive, about 800 MB | +| `gliner25-decide-vllm-cpp` | GLiNER2.5-Decide | DeBERTa-v3-large with a classification head, about 2 GB | + +The engine, [vllm.cpp]({{% relref "features/vllm-cpp" %}}), also supports the +kev, CLM and xor decision models. Those checkpoints need a conversion step, so +they are not gallery entries yet. + +Tev1 is an autoregressive decision model. It answers through chat completions +and does not serve `/v1/systemone` yet. + +## Access control + +When authentication is on, the three routes need the `systemone` feature. It is +on by default for every user, like the other API features, and an administrator +can turn it off per user. diff --git a/docs/content/features/vllm-cpp.md b/docs/content/features/vllm-cpp.md index 74e3c0d38..6dc67c0b9 100644 --- a/docs/content/features/vllm-cpp.md +++ b/docs/content/features/vllm-cpp.md @@ -160,22 +160,23 @@ forward, which is the required contract for pooling models in vllm.cpp. A device-resident forward is tracked as a performance optimization, not a correctness gap. -### SystemOne structured-extraction API +### SystemOne decision API -The `vllm-cpp` backend also exposes kev-compatible SystemOne endpoints that -turn zero-shot NER into structured question answering. These mirror the API -from the [kev](https://github.com/jaredpalmer/kev) project: +The `vllm-cpp` backend serves the kev-compatible SystemOne endpoints: typed +`choice`, `noul` and `score` questions over a state text, answered by a +non-generative decision model in one pass. A decision model declares +`known_usecases: [systemone]`. See [SystemOne decisions]({{% relref "features/systemone" %}}) +for the request shape, the models you can install and the access rules. | Endpoint | Method | Description | |---|---|---| -| `/v1/systemone` | POST | Answer all questions in one NER pass | +| `/v1/systemone` | POST | Answer all questions in one pass | | `/v1/systemone/permute` | POST | Re-run one choice question under n_perm option orders | -| `/v1/systemone/separate` | POST | Answer each question in its own NER pass (N passes) | +| `/v1/systemone/separate` | POST | Answer each question in its own pass (N passes) | -Each question has a `type` of `noul` (binary entity presence), `choice` (pick -one option), or `score` (pick one level). The `model` field in the request body -selects the NER model. Labels are derived from the question definition, so no -`ner_labels` configuration is needed for these endpoints. +The GLiNER2.5 zero-shot NER model (`token_classify`) also serves these +endpoints. It derives its NER labels from the question definitions, so no +`ner_labels` configuration is needed. ## Beyond text generation From b3d65fd538d0b90f4e6f2ce3e4547c32ef0af27c Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Wed, 30 Sep 2026 11:55:26 +0000 Subject: [PATCH 18/33] fix(systemone): route NER models to the NER path and refuse decision models on permute and separate vllm_decide refuses NER architectures and the NER entry point refuses decision architectures, so each model kind 500ed on half of the routes. A token_classify model now goes to the NER path on /v1/systemone, and /permute and /separate return 400 for decision models. Docs and instructions state which kind serves which route. Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto --- .../endpoints/localai/api_instructions.go | 2 +- core/http/endpoints/localai/systemone.go | 57 ++++++++++++++++++- .../endpoints/localai/systemone_gate_test.go | 40 +++++++++++++ docs/content/features/systemone.md | 19 +++++-- docs/content/features/vllm-cpp.md | 8 ++- 5 files changed, 116 insertions(+), 10 deletions(-) diff --git a/core/http/endpoints/localai/api_instructions.go b/core/http/endpoints/localai/api_instructions.go index dc60a3c21..702c771ff 100644 --- a/core/http/endpoints/localai/api_instructions.go +++ b/core/http/endpoints/localai/api_instructions.go @@ -109,7 +109,7 @@ var instructionDefs = []instructionDef{ Name: "systemone", Description: "Typed decisions (choice, noul, score) over a state text with calibrated confidence", Tags: []string{"systemone"}, - Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { : { type: choice|noul|score, instructions, criteria } } }. The model must declare known_usecases: [systemone] (or token_classify for the zero-shot NER path); a config that declares no usecases keeps working. Responses carry per-question answers with confidence and probabilities plus token usage. The wire shape matches Ollama's /v1/systemone.", + Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { : { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [systemone] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. The wire shape matches Ollama's /v1/systemone.", }, { Name: "branding", diff --git a/core/http/endpoints/localai/systemone.go b/core/http/endpoints/localai/systemone.go index 435414d39..17f68a71a 100644 --- a/core/http/endpoints/localai/systemone.go +++ b/core/http/endpoints/localai/systemone.go @@ -400,6 +400,55 @@ func checkSystemOneModel(app *application.Application, modelName string) error { return systemOneModelAllowed(cfg) } +// systemOneUsesDecisionPipeline reports whether /v1/systemone forwards the +// request to the backend's Score RPC (the decision pipeline) for this model. +// A model that declares token_classify without systemone is a zero-shot NER +// model: the backend's decision entry point refuses those architectures, so it +// goes to the NER path instead. A config that declares nothing keeps the +// decision pipeline, which is what setups that predate the systemone usecase +// relied on. +func systemOneUsesDecisionPipeline(cfg config.ModelConfig) bool { + if !backendSupportsScore(cfg.Backend) { + return false + } + if cfg.KnownUsecases == nil { + return true + } + declared := *cfg.KnownUsecases + if declared&config.FLAG_SYSTEMONE != 0 { + return true + } + return declared&config.FLAG_TOKEN_CLASSIFY == 0 +} + +// systemOneNERAllowed guards /permute and /separate, which always run the NER +// path. A decision model cannot serve them: the backend's NER entry point +// refuses its architecture, and the caller would see a backend error. +func systemOneNERAllowed(cfg config.ModelConfig) error { + if cfg.KnownUsecases == nil { + return nil + } + declared := *cfg.KnownUsecases + if declared&config.FLAG_SYSTEMONE != 0 && declared&config.FLAG_TOKEN_CLASSIFY == 0 { + return fmt.Errorf("model %q is a decision model: /permute and /separate use the NER path, use POST /v1/systemone instead", cfg.Name) + } + return nil +} + +// checkSystemOneNERModel applies systemOneNERAllowed to a model looked up by +// name; an unknown model passes so the not-found handling keeps its status. +func checkSystemOneNERModel(app *application.Application, modelName string) error { + cl := app.ModelConfigLoader() + if cl == nil { + return nil + } + cfg, ok := cl.GetModelConfig(modelName) + if !ok { + return nil + } + return systemOneNERAllowed(cfg) +} + // backendSupportsScore reports whether the named backend implements the // Score gRPC RPC. vllm-cpp does (kev/laya decision pipeline and cua-s1-forms // scoring via the unified vllm_decide C ABI); other backends fall through to @@ -445,7 +494,7 @@ func SystemOneEndpoint(app *application.Application) echo.HandlerFunc { // Score RPC and return the backend's response as-is. cl := app.ModelConfigLoader() if cl != nil { - if cfg, ok := cl.GetModelConfig(req.Model); ok && backendSupportsScore(cfg.Backend) { + if cfg, ok := cl.GetModelConfig(req.Model); ok && systemOneUsesDecisionPipeline(cfg) { reqJSON, err := json.Marshal(req) if err != nil { return systemOneError(c, http.StatusInternalServerError, "failed to marshal request: "+err.Error()) @@ -509,6 +558,9 @@ func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc { if err := checkSystemOneModel(app, req.Request.Model); err != nil { return systemOneError(c, http.StatusBadRequest, err.Error()) } + if err := checkSystemOneNERModel(app, req.Request.Model); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } if req.Question == "" { return systemOneError(c, http.StatusBadRequest, "question is required") } @@ -648,6 +700,9 @@ func SystemOneSeparateEndpoint(app *application.Application) echo.HandlerFunc { if err := checkSystemOneModel(app, req.Model); err != nil { return systemOneError(c, http.StatusBadRequest, err.Error()) } + if err := checkSystemOneNERModel(app, req.Model); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } parsed, err := parseSystemOneRequest(&req) if err != nil { return systemOneError(c, http.StatusBadRequest, err.Error()) diff --git a/core/http/endpoints/localai/systemone_gate_test.go b/core/http/endpoints/localai/systemone_gate_test.go index b7857da56..b790c8df2 100644 --- a/core/http/endpoints/localai/systemone_gate_test.go +++ b/core/http/endpoints/localai/systemone_gate_test.go @@ -32,3 +32,43 @@ var _ = Describe("systemOneModelAllowed", func() { Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [systemone]"))) }) }) + +var _ = Describe("systemone routing by model kind", func() { + mk := func(backend string, usecases ...string) config.ModelConfig { + c := config.ModelConfig{Name: "m", Backend: backend} + if len(usecases) > 0 { + c.KnownUsecases = config.GetUsecasesFromYAML(usecases) + } + return c + } + + Describe("systemOneUsesDecisionPipeline", func() { + It("sends a declared decision model to the decision pipeline", func() { + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "systemone"))).To(BeTrue()) + }) + It("sends a token_classify model to the NER path, since vllm_decide refuses NER architectures", func() { + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "token_classify"))).To(BeFalse()) + }) + It("keeps configs that declare nothing on the decision pipeline", func() { + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp"))).To(BeTrue()) + }) + It("prefers the decision pipeline when both usecases are declared", func() { + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "systemone", "token_classify"))).To(BeTrue()) + }) + It("never uses it for a backend without the Score RPC", func() { + Expect(systemOneUsesDecisionPipeline(mk("no-such-backend", "systemone"))).To(BeFalse()) + }) + }) + + Describe("systemOneNERAllowed", func() { + It("refuses a decision model on the NER-only routes with an actionable message", func() { + Expect(systemOneNERAllowed(mk("vllm-cpp", "systemone"))).To(MatchError(ContainSubstring("/v1/systemone"))) + }) + It("accepts a token_classify model", func() { + Expect(systemOneNERAllowed(mk("vllm-cpp", "token_classify"))).To(Succeed()) + }) + It("accepts configs that declare nothing", func() { + Expect(systemOneNERAllowed(mk("vllm-cpp"))).To(Succeed()) + }) + }) +}) diff --git a/docs/content/features/systemone.md b/docs/content/features/systemone.md index d58a3d436..84c65a665 100644 --- a/docs/content/features/systemone.md +++ b/docs/content/features/systemone.md @@ -21,6 +21,13 @@ project and match the `/v1/systemone` endpoint that Ollama added in 0.35. | `/v1/systemone/permute` | POST | Re-run one choice question under `n_perm` option orders | | `/v1/systemone/separate` | POST | Answer each question in its own pass | +Which route a model can serve depends on its kind: + +| Model kind | `/v1/systemone` | `/permute` and `/separate` | +|---|---|---| +| Decision model (`systemone`), such as Laya or GLiNER2.5-Decide | Yes | No, returns `400` | +| Zero-shot NER model (`token_classify`), such as GLiNER2.5 | Yes, through the NER path | Yes | + ## Question types | Type | Answer | Fields in the answer | @@ -58,8 +65,8 @@ curl http://localhost:8080/v1/systemone -H "Content-Type: application/json" -d ' }' ``` -Every answer carries a `confidence` value, and the response reports token usage -and `latency_ms`. +Answers from a decision model carry a `confidence` value, and the response +reports token usage and `latency_ms`. The NER path does not report token usage. ## Choosing a model @@ -78,9 +85,11 @@ parameters: `systemone` is never guessed, and a model that declares it is not listed as a chat, completion or embeddings model. A model that declares usecases without `systemone` or `token_classify` gets a `400` from these endpoints that names the -missing usecase. A config that declares no usecases at all keeps working, so -setups that predate the flag are not broken. Models that declare `token_classify` -are served by the zero-shot NER path. +missing usecase. A model that declares `token_classify` and not `systemone` is +served by the zero-shot NER path. A vllm-cpp config that declares no usecases is +treated as a decision model, so setups that predate the flag keep working, but a +config that declares only `chat` (as an older `laya` gallery entry did) now gets +the `400` and needs `known_usecases: [systemone]`. Install one from the gallery and filter on the `systemone` tag: diff --git a/docs/content/features/vllm-cpp.md b/docs/content/features/vllm-cpp.md index 6dc67c0b9..ba9840302 100644 --- a/docs/content/features/vllm-cpp.md +++ b/docs/content/features/vllm-cpp.md @@ -174,9 +174,11 @@ for the request shape, the models you can install and the access rules. | `/v1/systemone/permute` | POST | Re-run one choice question under n_perm option orders | | `/v1/systemone/separate` | POST | Answer each question in its own pass (N passes) | -The GLiNER2.5 zero-shot NER model (`token_classify`) also serves these -endpoints. It derives its NER labels from the question definitions, so no -`ner_labels` configuration is needed. +The GLiNER2.5 zero-shot NER model (`token_classify`) also serves +`/v1/systemone`, through the NER path, and it is the model to use for +`/v1/systemone/permute` and `/v1/systemone/separate`, which decision models +refuse with a `400`. It derives its NER labels from the question definitions, so +no `ner_labels` configuration is needed. ## Beyond text generation From b8fdb50291a2f7b989aee6fe5873d741e6d5eac1 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Wed, 30 Sep 2026 15:56:27 +0200 Subject: [PATCH 19/33] chore(model-gallery): :arrow_up: update checksum (#12366) :arrow_up: Checksum updates in gallery/index.yaml Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- gallery/index.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/gallery/index.yaml b/gallery/index.yaml index 71c572e6b..58c1daba9 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -297,7 +297,7 @@ files: - filename: ds4flash.gguf uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF - sha256: 0ee532bce971b1464cce75ce74b7e5205dfd5cc4a844d27ce8f4b4de82a7136f + sha256: bfca14d287c9fd865529e02efa0ba6f572fb7bef623fa0bf4b0fb6214ee63a5e - name: "qwopus3.8-27b-flash-v2" variants: - model: qwopus3.8-27b-flash-v2-q8 From 4d0317db8d6eeb5a246ac9a2b455c2da01351491 Mon Sep 17 00:00:00 2001 From: mudler-agent Date: Wed, 30 Sep 2026 16:12:35 +0200 Subject: [PATCH 20/33] chore(deps): bump nib to v0.12.1 (#12372) nib v0.12.0 called xlog.SetLogger with a *slog.Logger, which does not build against the xlog v0.0.6 LocalAI uses. v0.12.1 fixes that (mudler/nib#137). nib's ApprovalMode is now a named string type, so the chat tests compare against nibtypes.ApprovalAuto and ApprovalPrompt instead of untyped strings, and run.go sets the exported constant. Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto --- core/cli/chat/run.go | 2 +- core/cli/chat/run_test.go | 6 ++--- go.mod | 23 ++++++++++++++++++-- go.sum | 46 +++++++++++++++++++++++++++++++++++---- 4 files changed, 67 insertions(+), 10 deletions(-) diff --git a/core/cli/chat/run.go b/core/cli/chat/run.go index 3da8e08d2..d221eab8e 100644 --- a/core/cli/chat/run.go +++ b/core/cli/chat/run.go @@ -314,7 +314,7 @@ func agentOptions(dir, model string, opts Options) app.Options { TraceDir: opts.TraceDir, } if opts.Yolo { - overrides.ApprovalMode = "auto" + overrides.ApprovalMode = nibtypes.ApprovalAuto } return app.Options{ diff --git a/core/cli/chat/run_test.go b/core/cli/chat/run_test.go index e334e0a0b..7899c46eb 100644 --- a/core/cli/chat/run_test.go +++ b/core/cli/chat/run_test.go @@ -509,7 +509,7 @@ var _ = Describe("prepare", func() { Expect(agentOptions(dir, "a-model", opts).Overrides.ApprovalMode).To(BeEmpty()) opts.Yolo = true - Expect(agentOptions(dir, "a-model", opts).Overrides.ApprovalMode).To(Equal("auto")) + Expect(agentOptions(dir, "a-model", opts).Overrides.ApprovalMode).To(Equal(nibtypes.ApprovalAuto)) }) // The specs above pin what is handed over. These pin what nib does with @@ -570,7 +570,7 @@ var _ = Describe("prepare", func() { opts.Yolo = true cfg := resolve(agentOptions(dir, "a-model", opts)) - Expect(cfg.ApprovalMode).To(Equal("auto")) + Expect(cfg.ApprovalMode).To(Equal(nibtypes.ApprovalAuto)) }) // The other half of the same rule, and the reason an unset flag is @@ -582,7 +582,7 @@ var _ = Describe("prepare", func() { cfg := resolve(agentOptions(dir, "a-model", optionsWithStreams(os.Stdin, os.Stdout, os.Stderr))) Expect(cfg.APIKey).To(Equal("saved-key")) - Expect(cfg.ApprovalMode).To(Equal("prompt")) + Expect(cfg.ApprovalMode).To(Equal(nibtypes.ApprovalPrompt)) }) }) }) diff --git a/go.mod b/go.mod index ed69ee4ea..3a5b32908 100644 --- a/go.mod +++ b/go.mod @@ -36,11 +36,11 @@ require ( github.com/mholt/archiver/v3 v3.5.1 github.com/microcosm-cc/bluemonday v1.0.27 github.com/modelcontextprotocol/go-sdk v1.5.0 - github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 + github.com/mudler/cogito v0.11.1-0.20260928072733-b40513ef5d1a github.com/mudler/edgevpn v0.34.0 github.com/mudler/go-processmanager v0.1.2-0.20260823202314-dfa0ed852db6 github.com/mudler/memory v0.0.0-20260406210934-424c1ecf2cf8 - github.com/mudler/nib v0.6.0 + github.com/mudler/nib v0.12.1 github.com/mudler/xlog v0.0.6 github.com/nats-io/jwt/v2 v2.7.4 github.com/nats-io/nats.go v1.52.0 @@ -153,6 +153,25 @@ require ( github.com/mattn/go-sqlite3 v1.14.32 // indirect github.com/moby/moby/api v1.54.2 // indirect github.com/moby/moby/client v0.4.1 // indirect + github.com/msuozzo/bonsai v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-bash v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-c v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-dockerfile v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-go v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-gotemplate v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-groovy v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-java v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-javascript v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-kotlin v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-markdown v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-markdown-inline v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-python v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-ruby v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-rust v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-terraform v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-tsx v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-typescript v0.4.0 // indirect + github.com/msuozzo/bonsai/bonsai-yaml v0.4.0 // indirect github.com/muesli/ansi v0.0.0-20230316100256-276c6243b2f6 // indirect github.com/muesli/cancelreader v0.2.2 // indirect github.com/nats-io/nuid v1.0.1 // indirect diff --git a/go.sum b/go.sum index 6a01b5bd9..bb724dd3d 100644 --- a/go.sum +++ b/go.sum @@ -994,10 +994,48 @@ github.com/mr-tron/base58 v1.3.0 h1:K6Y13R2h+dku0wOqKtecgRnBUBPrZzLZy5aIj8lCcJI= github.com/mr-tron/base58 v1.3.0/go.mod h1:2BuubE67DCSWwVfx37JWNG8emOC0sHEU4/HpcYgCLX8= github.com/mschoch/smat v0.2.0 h1:8imxQsjDm8yFEAVBe7azKmKSgzSkZXDuKkSq9374khM= github.com/mschoch/smat v0.2.0/go.mod h1:kc9mz7DoBKqDyiRL7VZN8KvXQMWeTaVnttLRXOlotKw= +github.com/msuozzo/bonsai v0.4.0 h1:WVGqsSctbGcL5ehdCdnckwRxBC3omwsmOtzKH/2tuZw= +github.com/msuozzo/bonsai v0.4.0/go.mod h1:LDo0Dmp6qaeEoJAG83UuLDHtwHn6H1knSYKqoXMxm4M= +github.com/msuozzo/bonsai/bonsai-bash v0.4.0 h1:I8pjnVYSAaKew3SmtnDf8IKAH30jHvWfymKfq1zf6bc= +github.com/msuozzo/bonsai/bonsai-bash v0.4.0/go.mod h1:6Tflp8naB4zdsvPEEujA+6bOJ/WQ5PiITZwbM4dj1gg= +github.com/msuozzo/bonsai/bonsai-c v0.4.0 h1:8uY7V/ofhIpsWWY4SXkgDEP14g8+JUhsfp9dOmGSiK0= +github.com/msuozzo/bonsai/bonsai-c v0.4.0/go.mod h1:1KW4TR0hVjP2O7iPqxmyD5vFimAbNrG87tU1srIBG5k= +github.com/msuozzo/bonsai/bonsai-dockerfile v0.4.0 h1:NDR2AuG4pFkL2VLgYiCJPbCrjqegH+LPZX/uEa1Q4yQ= +github.com/msuozzo/bonsai/bonsai-dockerfile v0.4.0/go.mod h1:fZLkQxL5zQk9J96xG7Lw+PQe4mZpU+Bqld6ToOJU3J0= +github.com/msuozzo/bonsai/bonsai-go v0.4.0 h1:mDH7ExUuH9vFKLyZ63SUk4FuJ5y2Z4gEg8WcIOCzfiY= +github.com/msuozzo/bonsai/bonsai-go v0.4.0/go.mod h1:xdTJhN+7nGFzKZM8R2fEkHM/vM9DbWYhy44nLeR+oR8= +github.com/msuozzo/bonsai/bonsai-gotemplate v0.4.0 h1:wGnquoW7t9HGmMsvMZdAnvxmULSaDrr6JPUkWtzn2Ck= +github.com/msuozzo/bonsai/bonsai-gotemplate v0.4.0/go.mod h1:DDfS5ey/vTTA25MguPVaDYP+abbmE/JZt/XWzmVfYWY= +github.com/msuozzo/bonsai/bonsai-groovy v0.4.0 h1:Dhdlsbo1gCoE51ig2uXbW/DqndCpFZyDWQjdmaOrdgc= +github.com/msuozzo/bonsai/bonsai-groovy v0.4.0/go.mod h1:5fEI1HjFSX1NJDLIccYnh2my03dtVikepPF6mVL8xc0= +github.com/msuozzo/bonsai/bonsai-java v0.4.0 h1:MNr/ShaNv688jNQByKu3OvwFUXAIUUiqhPi253M3EP8= +github.com/msuozzo/bonsai/bonsai-java v0.4.0/go.mod h1:bgcXWciHZqVQ9GBPhDkk/svvja907uzXvV/QxKmKM+s= +github.com/msuozzo/bonsai/bonsai-javascript v0.4.0 h1:hJ0Fz140Os3ZpV7Kl5ko7bsI++f5qayOl2HlLrdhXUU= +github.com/msuozzo/bonsai/bonsai-javascript v0.4.0/go.mod h1:QTqVNr8cwvdHBuMU0VsX+zW22DQfFYV0l7wClVXIVvM= +github.com/msuozzo/bonsai/bonsai-kotlin v0.4.0 h1:tTEsbgGoj1razbqvmcCMsAh2Y5/+gp7hMALSj4w1rbw= +github.com/msuozzo/bonsai/bonsai-kotlin v0.4.0/go.mod h1:cdwHqA5ys0bS32bBUv9B8AMS32UAJ6HqKNYNFHApvWw= +github.com/msuozzo/bonsai/bonsai-markdown v0.4.0 h1:D0c5FwSVu9xrujaLCYsrzO50aJ4zLJzUZCtbu8rSSTc= +github.com/msuozzo/bonsai/bonsai-markdown v0.4.0/go.mod h1:+o73zcEV/4O+8wthJC5iFdVx0M1S8SkI7YAYs7AIJng= +github.com/msuozzo/bonsai/bonsai-markdown-inline v0.4.0 h1:0XZb2aRuMrYN1Z1Tcw2J7mOWexAOPzg4VsWrcwR233E= +github.com/msuozzo/bonsai/bonsai-markdown-inline v0.4.0/go.mod h1:hjhOJSQylIlHEXT0+ns7ocHl9+qw1LgO9mits9Bt/Fc= +github.com/msuozzo/bonsai/bonsai-python v0.4.0 h1:spQNWwF6vBL4o8vfC+zoY8iZJfnLue1izNk7FKs0oXo= +github.com/msuozzo/bonsai/bonsai-python v0.4.0/go.mod h1:yMx1oeH6SQUuru9IhIioteV7cmF7S/UQNp9bDdMSfjs= +github.com/msuozzo/bonsai/bonsai-ruby v0.4.0 h1:hZygD1ev7EjLE5eDhlPRi+M0XrNKvzjMYx8RTZ0uxVA= +github.com/msuozzo/bonsai/bonsai-ruby v0.4.0/go.mod h1:GzujOi6rnLFX10zU9ntLNcetpnMaLeeUm526/5+aRHQ= +github.com/msuozzo/bonsai/bonsai-rust v0.4.0 h1:EQ2Fu6DWoGbq0OCV30acchw2VmV1WIOSNMVOLHyj2LY= +github.com/msuozzo/bonsai/bonsai-rust v0.4.0/go.mod h1:Qme4vyNPcCVkaGbLRiUW1NHERiY+bwfsPHsVyr3kZBg= +github.com/msuozzo/bonsai/bonsai-terraform v0.4.0 h1:loIUha5CgR7TgtfkaToP8PIufPrt5P6O/ySs3s0QcMo= +github.com/msuozzo/bonsai/bonsai-terraform v0.4.0/go.mod h1:xqGKDODYTigesGhHpZx3ZMuMivQ7nGGzCUmIHA4GxVE= +github.com/msuozzo/bonsai/bonsai-tsx v0.4.0 h1:/dheQMtc7luNGkL8E6DILowN/SxWyJ6W1x4msLqZiQQ= +github.com/msuozzo/bonsai/bonsai-tsx v0.4.0/go.mod h1:YW/ExrMLztBgSvoY1KTKMMQapV0OqMBb4RF62kQ/GMA= +github.com/msuozzo/bonsai/bonsai-typescript v0.4.0 h1:BD6JHfCs3rGNUttL+tMnz/iav1Y1+DkiMcy871NsKwo= +github.com/msuozzo/bonsai/bonsai-typescript v0.4.0/go.mod h1:DElq3FKqbTkJXcat1o8m98ijL3wUf+jW+DNInzh5Mgo= +github.com/msuozzo/bonsai/bonsai-yaml v0.4.0 h1:PJfyfjMQrcWU5e3ib0xGsVuOafNuimf5fK0NhTk8olU= +github.com/msuozzo/bonsai/bonsai-yaml v0.4.0/go.mod h1:z8jc0tjXDSOQRKRG0G3uymvVxHdBbrVsWd149nPOyHE= github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca h1:bHlzSuOc5cKvHGF21ZjFFUV7IlkS3wt9YkBsWtkcaC4= github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca/go.mod h1:nk6zt1s5ANgchJYTWGY1jfFPuITSn1gB5oHZ/uFeFDg= -github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 h1:eYTR8od5HdaHlh9AKCkxkRoHs2/wmx24BF5qrUh2TRY= -github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6/go.mod h1:6sfja3lcu2nWRzEc0wwqGNu/eCG3EWgij+8s7xyUeQ4= +github.com/mudler/cogito v0.11.1-0.20260928072733-b40513ef5d1a h1:b3bZ15XGd3kH+gje6ggFAciT4uC83+24xO/qc3o4viQ= +github.com/mudler/cogito v0.11.1-0.20260928072733-b40513ef5d1a/go.mod h1:UxGNMBRakV0A2uVHv2JQIHjpnVPkEWFaS7SDhyynZqQ= github.com/mudler/edgevpn v0.34.0 h1:qDrD/rCPFY/FdURbXudIZWihVKY4VOX3nMn3CcbeQEU= github.com/mudler/edgevpn v0.34.0/go.mod h1:yki7uMi5LR9gSMrw8PdPieuxsrk8BLV2Ui7VBEmbbIA= github.com/mudler/go-piper v0.0.0-20241023091659-2494246fd9fc h1:RxwneJl1VgvikiX28EkpdAyL4yQVnJMrbquKospjHyA= @@ -1008,8 +1046,8 @@ github.com/mudler/localrecall v0.6.5 h1:Q0atTJFFAyumKZG5dbGSrvQ+wsuA88hywIOfHdxp github.com/mudler/localrecall v0.6.5/go.mod h1:28k5n19raUrkuwXkacdNsBlj8yuSnGhpT16tu+2+4dU= github.com/mudler/memory v0.0.0-20260406210934-424c1ecf2cf8 h1:Ry8RiWy8fZ6Ff4E7dPmjRsBrnHOnPeOOj2LhCgyjQu0= github.com/mudler/memory v0.0.0-20260406210934-424c1ecf2cf8/go.mod h1:EA8Ashhd56o32qN7ouPKFSRUs/Z+LrRCF4v6R2Oarm8= -github.com/mudler/nib v0.6.0 h1:6l2bQJkgHT5+o6S0wmmk/ipTCbFzmiSp1pM/tGQv4Eo= -github.com/mudler/nib v0.6.0/go.mod h1:d+Ymgi7PxDLnGxpYAkugv+mwRDtR1CMcgyykKNipwWA= +github.com/mudler/nib v0.12.1 h1:7yKZeOWcIac62UoNb3J2R/oEtFYzebFwgap9HPjIJdI= +github.com/mudler/nib v0.12.1/go.mod h1:I6diFODU8ALfPwFIbvrwpLDfhyBu6wAzoseig9jJEQ4= github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145 h1:z59tA3IDYPt71nzH1jpxeaA1LuDw8aZfpTQFNU43Zb8= github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145/go.mod h1:z3yFhcL9bSykmmh6xgGu0hyoItd4CnxgtWMEWw8uFJU= github.com/mudler/water v0.0.0-20250808092830-dd90dcf09025 h1:WFLP5FHInarYGXi6B/Ze204x7Xy6q/I4nCZnWEyPHK0= From bd6863af8170cf8d0650261876db6e551a51d117 Mon Sep 17 00:00:00 2001 From: mudler-agent Date: Wed, 30 Sep 2026 16:13:23 +0200 Subject: [PATCH 21/33] fix(react-ui): extract text from PDF attachments in chat and home (#12374) The React UI read every non-media attachment with file.text(). For a PDF that decodes the binary bytes as UTF-8, so the model received raw "%PDF ... stream ... endobj" noise instead of the document. The legacy Alpine UI ran pdf.js; that step was not ported when the React UI replaced it, but both file pickers still advertise .pdf. Add a shared readAttachmentText helper that routes PDFs through pdfjs-dist and reads other files as before. pdf.js and its worker load on first use, so the main bundle does not grow. A PDF that cannot be parsed or has no text layer (scanned, encrypted, damaged) is rejected with a toast instead of being attached as an empty or garbage file. Cover the chat and home paths with Playwright specs that build a real PDF in the test. Assisted-by: Claude Code:claude-sonnet-5-5 [playwright] [eslint] Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto --- .../react-ui/e2e/chat-pdf-attachment.spec.js | 97 +++++++ core/http/react-ui/package-lock.json | 271 ++++++++++++++++++ core/http/react-ui/package.json | 1 + .../http/react-ui/public/locales/en/chat.json | 3 +- .../http/react-ui/public/locales/en/home.json | 3 +- core/http/react-ui/src/pages/Chat.jsx | 10 +- core/http/react-ui/src/pages/Home.jsx | 10 +- core/http/react-ui/src/utils/pdf.js | 48 ++++ 8 files changed, 437 insertions(+), 6 deletions(-) create mode 100644 core/http/react-ui/e2e/chat-pdf-attachment.spec.js create mode 100644 core/http/react-ui/src/utils/pdf.js diff --git a/core/http/react-ui/e2e/chat-pdf-attachment.spec.js b/core/http/react-ui/e2e/chat-pdf-attachment.spec.js new file mode 100644 index 000000000..eaf1d0fb1 --- /dev/null +++ b/core/http/react-ui/e2e/chat-pdf-attachment.spec.js @@ -0,0 +1,97 @@ +import { test, expect } from './coverage-fixtures.js' + +// Single-page PDF with a real text layer, built byte by byte so the xref +// offsets are valid and the spec needs no binary fixture. +function buildPdf(text) { + const stream = `BT /F1 18 Tf 20 100 Td (${text}) Tj ET` + const objs = [ + '<< /Type /Catalog /Pages 2 0 R >>', + '<< /Type /Pages /Kids [3 0 R] /Count 1 >>', + '<< /Type /Page /Parent 2 0 R /MediaBox [0 0 300 200] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>', + `<< /Length ${stream.length} >>\nstream\n${stream}\nendstream`, + '<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>', + ] + let out = '%PDF-1.4\n' + const offsets = [] + objs.forEach((body, i) => { + offsets.push(out.length) + out += `${i + 1} 0 obj\n${body}\nendobj\n` + }) + const xref = out.length + out += `xref\n0 ${objs.length + 1}\n0000000000 65535 f \n` + for (const o of offsets) out += `${String(o).padStart(10, '0')} 00000 n \n` + out += `trailer\n<< /Size ${objs.length + 1} /Root 1 0 R >>\nstartxref\n${xref}\n%%EOF\n` + return Buffer.from(out, 'latin1') +} + +async function openChat(page) { + await page.route('**/api/models/capabilities', (route) => { + route.fulfill({ + contentType: 'application/json', + body: JSON.stringify({ data: [{ id: 'test-model', capabilities: ['FLAG_CHAT'] }] }), + }) + }) + await page.goto('/app/chat') + await expect(page.getByRole('button', { name: 'test-model' })).toBeVisible({ timeout: 10_000 }) +} + +test.describe('Chat - PDF attachments', () => { + test('sends the extracted text layer, not the raw PDF bytes', async ({ page }) => { + let requestBody = '' + await page.route('**/v1/chat/completions', (route) => { + requestBody = route.request().postData() || '' + route.fulfill({ status: 500, contentType: 'application/json', body: JSON.stringify({ error: { message: 'stop' } }) }) + }) + await openChat(page) + + await page.locator('input[type=file]').setInputFiles({ + name: 'report.pdf', + mimeType: 'application/pdf', + buffer: buildPdf('Quarterly revenue grew 42 percent'), + }) + await expect(page.locator('.chat-file-name', { hasText: 'report.pdf' })).toBeVisible() + + await page.locator('.chat-input').fill('Summarize') + await page.locator('.chat-send-btn').click() + + await expect.poll(() => requestBody).toContain('Quarterly revenue grew 42 percent') + expect(requestBody).toContain('File: report.pdf') + expect(requestBody).not.toContain('%PDF') + }) + + test('rejects a PDF that cannot be parsed instead of attaching garbage', async ({ page }) => { + await openChat(page) + + await page.locator('input[type=file]').setInputFiles({ + name: 'broken.pdf', + mimeType: 'application/pdf', + buffer: Buffer.from('%PDF-1.4 this is not a real document'), + }) + + await expect(page.getByText('Could not read text from broken.pdf')).toBeVisible({ timeout: 10_000 }) + await expect(page.locator('.chat-file-name', { hasText: 'broken.pdf' })).toHaveCount(0) + }) +}) + +test.describe('Home - PDF attachments', () => { + test('attaches a PDF that has a text layer', async ({ page }) => { + await page.goto('/app') + await page.locator('input[type=file][accept*="pdf"]').setInputFiles({ + name: 'notes.pdf', + mimeType: 'application/pdf', + buffer: buildPdf('Meeting notes for Tuesday'), + }) + await expect(page.locator('.home-file-tag', { hasText: 'notes.pdf' })).toBeVisible({ timeout: 10_000 }) + }) + + test('rejects a PDF that cannot be parsed', async ({ page }) => { + await page.goto('/app') + await page.locator('input[type=file][accept*="pdf"]').setInputFiles({ + name: 'broken.pdf', + mimeType: 'application/pdf', + buffer: Buffer.from('%PDF-1.4 this is not a real document'), + }) + await expect(page.getByText('Could not read text from broken.pdf')).toBeVisible({ timeout: 10_000 }) + await expect(page.locator('.home-file-tag')).toHaveCount(0) + }) +}) diff --git a/core/http/react-ui/package-lock.json b/core/http/react-ui/package-lock.json index 6683ac675..9c68bd1c7 100644 --- a/core/http/react-ui/package-lock.json +++ b/core/http/react-ui/package-lock.json @@ -29,6 +29,7 @@ "i18next-browser-languagedetector": "^8.2.1", "i18next-http-backend": "^3.0.6", "marked": "^15.0.7", + "pdfjs-dist": "^5.6.205", "react": "^19.1.0", "react-dom": "^19.1.0", "react-i18next": "^17.0.6", @@ -1021,6 +1022,256 @@ "resolved": "https://registry.npmjs.org/json-schema-traverse/-/json-schema-traverse-1.0.0.tgz", "integrity": "sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug==" }, + "node_modules/@napi-rs/canvas": { + "version": "0.1.100", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas/-/canvas-0.1.100.tgz", + "integrity": "sha512-xglYA6q3XO5P3BNJYxVZ1IV7DLVjp1Py6nwag88YntrS+3vKHyYcMqXVS4ZztJmwz2uGvz1FWhI/4LgbR5uQDA==", + "license": "MIT", + "optional": true, + "workspaces": [ + "e2e/*" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + }, + "optionalDependencies": { + "@napi-rs/canvas-android-arm64": "0.1.100", + "@napi-rs/canvas-darwin-arm64": "0.1.100", + "@napi-rs/canvas-darwin-x64": "0.1.100", + "@napi-rs/canvas-linux-arm-gnueabihf": "0.1.100", + "@napi-rs/canvas-linux-arm64-gnu": "0.1.100", + "@napi-rs/canvas-linux-arm64-musl": "0.1.100", + "@napi-rs/canvas-linux-riscv64-gnu": "0.1.100", + "@napi-rs/canvas-linux-x64-gnu": "0.1.100", + "@napi-rs/canvas-linux-x64-musl": "0.1.100", + "@napi-rs/canvas-win32-arm64-msvc": "0.1.100", + "@napi-rs/canvas-win32-x64-msvc": "0.1.100" + } + }, + "node_modules/@napi-rs/canvas-android-arm64": { + "version": "0.1.100", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-android-arm64/-/canvas-android-arm64-0.1.100.tgz", + "integrity": "sha512-hjhCKhntPv9+t4ckHymdx0phYNcVW+GKQR6Lzw2zE+pOVjOplSmtx9nNNknTjbEDLcuLZqA1y8ufKg1XfgftzQ==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "android" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-darwin-arm64": { + "version": "0.1.100", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-darwin-arm64/-/canvas-darwin-arm64-0.1.100.tgz", + "integrity": "sha512-2PcswRaC7Ly645DGt88///zuFDhJxJYdKAs1uU3mfk1atYkXufgcgLfBpk6Tm12nCQBaNt1wpybuPZ4qOhTo8A==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-darwin-x64": { + "version": "0.1.100", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-darwin-x64/-/canvas-darwin-x64-0.1.100.tgz", + "integrity": "sha512-ePNZtj7pNIva/siZMg+HmbeozkIjqUIYdoymH8HaA3qK7LfzFN4WMBM8G6HQ9ZC+H3+Dnn5pqtiXpgLykaPOhw==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-linux-arm-gnueabihf": { + "version": "0.1.100", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-arm-gnueabihf/-/canvas-linux-arm-gnueabihf-0.1.100.tgz", + "integrity": "sha512-d5cDB48oWFGU8/XPhUOFAlySgb/VAu7D+s8fi55K1Pcfg8aPplHWqMgibhVLU8ky7Pyg/fuiVLz4Nf3JrSTuUA==", + "cpu": [ + "arm" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-linux-arm64-gnu": { + "version": "0.1.100", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-arm64-gnu/-/canvas-linux-arm64-gnu-0.1.100.tgz", + "integrity": "sha512-rDxgxRu69RvDlX/bh9o22DxLsGr8EqsNgotL9+RwQE1S0b0cqeatqsw6aW45mukm0B42DIAaAacKaYQ8cqS1nw==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-linux-arm64-musl": { + "version": "0.1.100", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-arm64-musl/-/canvas-linux-arm64-musl-0.1.100.tgz", + "integrity": "sha512-K3mDW66N+xT2/V439u1alFANiBUjdEx2gLiNYnCmUsva5jZMxWTjafBYwTzYK+EMFMHrUoabuU+T1BIP5CgbYQ==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-linux-riscv64-gnu": { + "version": "0.1.100", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-riscv64-gnu/-/canvas-linux-riscv64-gnu-0.1.100.tgz", + "integrity": "sha512-mooqUBTIsccZpnoQC4NgrC1v6C1vof39etLNMnBwCY+p0gajWJvAHLGQ6g/gGyS5YrpDW+GefSN4+Cvcr08UWw==", + "cpu": [ + "riscv64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-linux-x64-gnu": { + "version": "0.1.100", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-x64-gnu/-/canvas-linux-x64-gnu-0.1.100.tgz", + "integrity": "sha512-1eCvkDCazm7FFhsT7DfGOdSaHgZVK3bt/dSBl5EWHOWmnz+I7j8tPseJqqD81NF+MH21jKUK4wQSDjN0mdhnTg==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-linux-x64-musl": { + "version": "0.1.100", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-x64-musl/-/canvas-linux-x64-musl-0.1.100.tgz", + "integrity": "sha512-20arT6lnI19S68qNlii73TSEDbECNgzMz2EpldC1V3mZFuRkeujXkcebRk0LRJe9SEUAooYiLokfMViY8IX7yA==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-win32-arm64-msvc": { + "version": "0.1.100", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-win32-arm64-msvc/-/canvas-win32-arm64-msvc-0.1.100.tgz", + "integrity": "sha512-DZFFT1wIAg37LJw37yhMRFfjATd3vTQzjZ1Yki8u2vhO6Hi5VE6BVaGQ1aaDu7xb4iMErz+9EOwjpS7xcxFeBw==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, + "node_modules/@napi-rs/canvas-win32-x64-msvc": { + "version": "0.1.100", + "resolved": "https://registry.npmjs.org/@napi-rs/canvas-win32-x64-msvc/-/canvas-win32-x64-msvc-0.1.100.tgz", + "integrity": "sha512-MyT1j3mHC2+Lu4pBi9mKyMJhtP6U7k7EldY7sj/uS5gJA65gTXt8MefJQXLJo5d/vZbuWmfxzkEUNc/urV3pHA==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 10" + }, + "funding": { + "type": "github", + "url": "https://github.com/sponsors/Brooooooklyn" + } + }, "node_modules/@napi-rs/wasm-runtime": { "version": "1.1.5", "resolved": "https://registry.npmjs.org/@napi-rs/wasm-runtime/-/wasm-runtime-1.1.5.tgz", @@ -5219,6 +5470,13 @@ "node": ">=8" } }, + "node_modules/node-readable-to-web-readable-stream": { + "version": "0.4.2", + "resolved": "https://registry.npmjs.org/node-readable-to-web-readable-stream/-/node-readable-to-web-readable-stream-0.4.2.tgz", + "integrity": "sha512-/cMZNI34v//jUTrI+UIo4ieHAB5EZRY/+7OmXZgBxaWBMcW2tGdceIw06RFxWxrKZ5Jp3sI2i5TsRo+CBhtVLQ==", + "license": "MIT", + "optional": true + }, "node_modules/node-releases": { "version": "2.0.54", "resolved": "https://registry.npmjs.org/node-releases/-/node-releases-2.0.54.tgz", @@ -5752,6 +6010,19 @@ "url": "https://opencollective.com/express" } }, + "node_modules/pdfjs-dist": { + "version": "5.6.205", + "resolved": "https://registry.npmjs.org/pdfjs-dist/-/pdfjs-dist-5.6.205.tgz", + "integrity": "sha512-tlUj+2IDa7G1SbvBNN74UHRLJybZDWYom+k6p5KIZl7huBvsA4APi6mKL+zCxd3tLjN5hOOEE9Tv7VdzO88pfg==", + "license": "Apache-2.0", + "engines": { + "node": ">=20.19.0 || >=22.13.0 || >=24" + }, + "optionalDependencies": { + "@napi-rs/canvas": "^0.1.96", + "node-readable-to-web-readable-stream": "^0.4.2" + } + }, "node_modules/picocolors": { "version": "1.1.1", "resolved": "https://registry.npmjs.org/picocolors/-/picocolors-1.1.1.tgz", diff --git a/core/http/react-ui/package.json b/core/http/react-ui/package.json index 73aa8a8a0..abedc40e8 100644 --- a/core/http/react-ui/package.json +++ b/core/http/react-ui/package.json @@ -45,6 +45,7 @@ "i18next-browser-languagedetector": "^8.2.1", "i18next-http-backend": "^3.0.6", "marked": "^15.0.7", + "pdfjs-dist": "^5.6.205", "react": "^19.1.0", "react-dom": "^19.1.0", "react-i18next": "^17.0.6", diff --git a/core/http/react-ui/public/locales/en/chat.json b/core/http/react-ui/public/locales/en/chat.json index f1c058790..1f9b1ad8a 100644 --- a/core/http/react-ui/public/locales/en/chat.json +++ b/core/http/react-ui/public/locales/en/chat.json @@ -117,7 +117,8 @@ "copied": "Copied to clipboard", "copyFailed": "Could not copy to clipboard", "chatCopied": "Chat copied to clipboard", - "forked": "Created a new chat" + "forked": "Created a new chat", + "pdfReadFailed": "Could not read text from {{name}}. It may be scanned, encrypted or damaged." }, "menu": { "trigger": "Chats", diff --git a/core/http/react-ui/public/locales/en/home.json b/core/http/react-ui/public/locales/en/home.json index b8bb3dd13..b65bea822 100644 --- a/core/http/react-ui/public/locales/en/home.json +++ b/core/http/react-ui/public/locales/en/home.json @@ -35,7 +35,8 @@ "enterToSend": "Enter to send", "selectModelFirst": "Select a model first", "sendMessage": "Send message", - "selectModelToast": "Please select a model first" + "selectModelToast": "Please select a model first", + "pdfReadFailed": "Could not read text from {{name}}. It may be scanned, encrypted or damaged." }, "quickLinks": { "manageByChat": "Manage by chat", diff --git a/core/http/react-ui/src/pages/Chat.jsx b/core/http/react-ui/src/pages/Chat.jsx index d799efba0..d714d4fde 100644 --- a/core/http/react-ui/src/pages/Chat.jsx +++ b/core/http/react-ui/src/pages/Chat.jsx @@ -9,6 +9,7 @@ import { extractCodeArtifacts, renderMarkdownWithArtifacts } from '../utils/arti import CanvasPanel from '../components/CanvasPanel' import Toggle from '../components/Toggle' import { fileToBase64, modelsApi, mcpApi } from '../utils/api' +import { readAttachmentText } from '../utils/pdf' import { CAP_CHAT } from '../utils/capabilities' import { useMCPClient } from '../hooks/useMCPClient' import MCPAppFrame from '../components/MCPAppFrame' @@ -842,13 +843,18 @@ export default function Chat() { const base64 = await fileToBase64(file) const entry = { name: file.name, type: file.type, base64 } if (!file.type.startsWith('image/') && !file.type.startsWith('audio/') && !file.type.startsWith('video/')) { - entry.textContent = await file.text().catch(() => '') + try { + entry.textContent = await readAttachmentText(file) + } catch { + addToast(t('toasts.pdfReadFailed', { name: file.name }), 'error') + continue + } } newFiles.push(entry) } setFiles(prev => [...prev, ...newFiles]) e.target.value = '' - }, []) + }, [addToast, t]) const handlePaste = useCallback(async (e) => { const items = e.clipboardData?.items diff --git a/core/http/react-ui/src/pages/Home.jsx b/core/http/react-ui/src/pages/Home.jsx index a5e43b5b4..621358266 100644 --- a/core/http/react-ui/src/pages/Home.jsx +++ b/core/http/react-ui/src/pages/Home.jsx @@ -12,6 +12,7 @@ import HomeConnect from '../components/HomeConnect' import { useResources } from '../hooks/useResources' import { usePolling } from '../hooks/usePolling' import { fileToBase64, backendControlApi, systemApi, modelsApi, mcpApi, nodesApi } from '../utils/api' +import { readAttachmentText } from '../utils/pdf' import { API_CONFIG } from '../utils/config' import { greetingKey } from '../utils/greeting' import StatusPill from '../components/StatusPill' @@ -158,12 +159,17 @@ export default function Home() { const base64 = await fileToBase64(file) const entry = { name: file.name, type: file.type, base64 } if (!file.type.startsWith('image/') && !file.type.startsWith('audio/')) { - entry.textContent = await file.text().catch(() => '') + try { + entry.textContent = await readAttachmentText(file) + } catch { + addToast(t('input.pdfReadFailed', { name: file.name }), 'error') + continue + } } newFiles.push(entry) } setter(prev => [...prev, ...newFiles]) - }, []) + }, [addToast, t]) const removeFile = useCallback((file) => { const removeFn = (prev) => prev.filter(f => f !== file) diff --git a/core/http/react-ui/src/utils/pdf.js b/core/http/react-ui/src/utils/pdf.js new file mode 100644 index 000000000..2996e50db --- /dev/null +++ b/core/http/react-ui/src/utils/pdf.js @@ -0,0 +1,48 @@ +export function isPdf(file) { + return file?.type === 'application/pdf' || /\.pdf$/i.test(file?.name || '') +} + +// pdf.js and its worker are loaded on first use so the main bundle does not +// pay for them when nobody attaches a PDF. +async function loadPdfjs() { + const [pdfjs, worker] = await Promise.all([ + import('pdfjs-dist'), + import('pdfjs-dist/build/pdf.worker.min.mjs?url'), + ]) + pdfjs.GlobalWorkerOptions.workerSrc = worker.default + return pdfjs +} + +// Returns the text layer of every page, one block per page. Throws when the +// file cannot be parsed or has no text layer (scanned PDFs): sending an empty +// attachment to the model would look like success and silently lose the file. +export async function extractPdfText(file) { + const pdfjs = await loadPdfjs() + const data = new Uint8Array(await file.arrayBuffer()) + const doc = await pdfjs.getDocument({ data }).promise + try { + const pages = [] + for (let i = 1; i <= doc.numPages; i++) { + const page = await doc.getPage(i) + const content = await page.getTextContent() + let text = '' + for (const item of content.items) { + text += item.str + text += item.hasEOL ? '\n' : '' + } + pages.push(text.trim()) + } + const text = pages.filter(Boolean).join('\n\n') + if (!text) throw new Error('PDF has no extractable text') + return text + } finally { + await doc.destroy() + } +} + +// Text of an attached non-media file. PDFs go through pdf.js; everything else +// is read as UTF-8. +export async function readAttachmentText(file) { + if (isPdf(file)) return extractPdfText(file) + return file.text().catch(() => '') +} From 70ce62901ff0d44e7100d692ba2cb443f655d408 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Wed, 30 Sep 2026 14:14:09 +0000 Subject: [PATCH 22/33] refactor: name the capability decisions instead of systemone The usecase describes what a model can do, and the category is the Decisions API. SystemOne stays as the wire contract: the /v1/systemone routes, the Score RPC question_type and the swagger tag are unchanged. The usecase, flag, auth feature, UI label, gallery tags and docs page are now decisions. Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto --- core/config/backend_capabilities.go | 12 ++++---- core/config/gguf.go | 4 +-- core/config/model_config.go | 20 ++++++------- core/config/model_config_test.go | 8 ++--- core/gallery/vllm_cpp_tags_test.go | 2 +- core/http/auth/features.go | 10 +++---- ...one_test.go => features_decisions_test.go} | 8 ++--- core/http/auth/permissions.go | 4 +-- .../endpoints/localai/api_instructions.go | 4 +-- .../localai/api_instructions_test.go | 8 ++--- core/http/endpoints/localai/systemone.go | 10 +++---- .../endpoints/localai/systemone_gate_test.go | 14 ++++----- .../react-ui/e2e/models-lifecycle.spec.js | 6 ++-- .../react-ui/public/locales/de/models.json | 2 +- .../react-ui/public/locales/en/models.json | 2 +- .../react-ui/public/locales/es/models.json | 2 +- .../react-ui/public/locales/id/models.json | 2 +- .../react-ui/public/locales/it/models.json | 2 +- .../react-ui/public/locales/ko/models.json | 2 +- .../react-ui/public/locales/pt-BR/models.json | 2 +- .../react-ui/public/locales/zh-CN/models.json | 2 +- .../react-ui/src/pages/InstalledModels.jsx | 4 +-- core/http/react-ui/src/utils/capabilities.js | 2 +- docs/content/advanced/model-configuration.md | 4 +-- .../features/{systemone.md => decisions.md} | 30 ++++++++++--------- docs/content/features/vllm-cpp.md | 6 ++-- gallery/index.yaml | 8 ++--- 27 files changed, 91 insertions(+), 89 deletions(-) rename core/http/auth/{features_systemone_test.go => features_decisions_test.go} (70%) rename docs/content/features/{systemone.md => decisions.md} (78%) diff --git a/core/config/backend_capabilities.go b/core/config/backend_capabilities.go index b4c40248c..650bc3f3c 100644 --- a/core/config/backend_capabilities.go +++ b/core/config/backend_capabilities.go @@ -35,7 +35,7 @@ const ( UsecaseSpeakerRecognition = "speaker_recognition" UsecaseTokenClassify = "token_classify" UsecaseScore = "score" - UsecaseSystemOne = "systemone" + UsecaseDecisions = "decisions" ) // GRPCMethod identifies a Backend service RPC from backend.proto. @@ -217,10 +217,10 @@ var UsecaseInfoMap = map[string]UsecaseInfo{ GRPCMethod: MethodScore, Description: "Joint log-probability scoring of candidate continuations via the Score RPC. Declared explicitly via known_usecases and usable alongside generation usecases.", }, - UsecaseSystemOne: { - Flag: FLAG_SYSTEMONE, + UsecaseDecisions: { + Flag: FLAG_DECISIONS, GRPCMethod: MethodScore, - Description: "SystemOne decision API (POST /v1/systemone): typed choice, noul and score questions over a state text, answered by a non-generative decision model through the Score RPC (question_type systemone). Declared explicitly via known_usecases.", + Description: "Decision models (served by POST /v1/systemone): typed choice, noul and score questions over a state text, answered by a non-generative decision model through the Score RPC (question_type systemone). Declared explicitly via known_usecases.", }, } @@ -355,10 +355,10 @@ var BackendCapabilities = map[string]BackendCapability{ // model returns an error rather than silent garbage. "vllm-cpp": { GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateVideo, MethodTokenClassify, MethodScore}, - PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVision, UsecaseVideo, UsecaseTokenClassify, UsecaseScore, UsecaseSystemOne}, + PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVision, UsecaseVideo, UsecaseTokenClassify, UsecaseScore, UsecaseDecisions}, DefaultUsecases: []string{UsecaseChat}, AcceptsImages: true, - Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and SystemOne decision models (kev, laya, CLM, GLiNER2.5-Decide, xor, nimble)", + Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and decision models (kev, laya, CLM, GLiNER2.5-Decide, xor, nimble)", }, "vllm-omni": { GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateImage, MethodGenerateVideo, MethodTTS}, diff --git a/core/config/gguf.go b/core/config/gguf.go index f9b8c748f..fad00a6c7 100644 --- a/core/config/gguf.go +++ b/core/config/gguf.go @@ -16,14 +16,14 @@ import ( // reservedNonChatModel reports whether the operator reserved this model for an // internal primitive — the router score classifier or the PII NER -// token_classify tier, or a SystemOne decision head. Such a model has no chat template and must not be +// token_classify tier, or a decision head. Such a model has no chat template and must not be // given the generative-chat defaults the GGUF importer otherwise applies // (FLAG_CHAT, jinja templating): surfacing it in chat pickers defeats the // reservation. Operators who do want a combined model declare both usecases // explicitly — the combination is valid. func reservedNonChatModel(cfg *ModelConfig) bool { return cfg.KnownUsecases != nil && - (*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY|FLAG_SYSTEMONE)) != 0 + (*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY|FLAG_DECISIONS)) != 0 } // genAudioEncoderKey is the mmproj metadata flag llama.cpp's mtmd writes for a diff --git a/core/config/model_config.go b/core/config/model_config.go index a78e1db6c..bc084aa87 100644 --- a/core/config/model_config.go +++ b/core/config/model_config.go @@ -2056,12 +2056,12 @@ const ( FLAG_3D ModelConfigUsecase = 0b100000000000000000000000 FLAG_3D_ANIMATION ModelConfigUsecase = 1 << 24 - // Marks a model as wired for the SystemOne decision API (POST - // /v1/systemone: typed choice / noul / score questions over a state). + // Marks a model as a decision model: it answers typed choice / noul / + // score questions over a state (served by POST /v1/systemone). // Explicit only, like FLAG_SCORE: a decision model never generates // text, so guessing chat or embeddings for it would surface it in // pickers it cannot serve. - FLAG_SYSTEMONE ModelConfigUsecase = 1 << 25 + FLAG_DECISIONS ModelConfigUsecase = 1 << 25 // Common Subsets FLAG_LLM ModelConfigUsecase = FLAG_CHAT | FLAG_COMPLETION | FLAG_EDIT @@ -2125,7 +2125,7 @@ func GetAllModelConfigUsecases() map[string]ModelConfigUsecase { "FLAG_TOKEN_CLASSIFY": FLAG_TOKEN_CLASSIFY, "FLAG_3D": FLAG_3D, "FLAG_3D_ANIMATION": FLAG_3D_ANIMATION, - "FLAG_SYSTEMONE": FLAG_SYSTEMONE, + "FLAG_DECISIONS": FLAG_DECISIONS, } } @@ -2154,9 +2154,9 @@ func GetUsecasesFromYAML(input []string) *ModelConfigUsecase { // // Declared known_usecases are normally additive — the guessing heuristic // still adds whatever it can infer from backend/templates. The exceptions -// are FLAG_SCORE, FLAG_TOKEN_CLASSIFY and FLAG_SYSTEMONE: when the operator +// are FLAG_SCORE, FLAG_TOKEN_CLASSIFY and FLAG_DECISIONS: when the operator // declared any of them, they reserved the model for a direct-decode primitive -// (the router classifier, the PII NER tier, or a SystemOne decision head). Letting GuessUsecases +// (the router classifier, the PII NER tier, or a decision head). Letting GuessUsecases // paint chat/completion/embeddings on top would surface it in pickers it // was deliberately kept out of. So a declared score or token_classify // list is authoritative; declare the generation usecases explicitly @@ -2166,7 +2166,7 @@ func (c *ModelConfig) HasUsecases(u ModelConfigUsecase) bool { if (u & *c.KnownUsecases) == u { return true } - if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY | FLAG_SYSTEMONE)) != 0 { + if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY | FLAG_DECISIONS)) != 0 { return false } } @@ -2389,10 +2389,10 @@ func (c *ModelConfig) GuessUsecases(u ModelConfigUsecase) bool { return false } - if (u & FLAG_SYSTEMONE) == FLAG_SYSTEMONE { - // No heuristic: SystemOne intent is a deliberate operator choice + if (u & FLAG_DECISIONS) == FLAG_DECISIONS { + // No heuristic: decisions intent is a deliberate operator choice // (the model is a non-generative decision head), so - // HasUsecases(FLAG_SYSTEMONE) is true only when KnownUsecases + // HasUsecases(FLAG_DECISIONS) is true only when KnownUsecases // declares it explicitly. return false } diff --git a/core/config/model_config_test.go b/core/config/model_config_test.go index 1873cfef8..b9d46f1c2 100644 --- a/core/config/model_config_test.go +++ b/core/config/model_config_test.go @@ -956,11 +956,11 @@ var _ = Describe("ModelConfig alias", func() { }) }) -var _ = Describe("systemone usecase", func() { - // A decision model never generates text, so a declared systemone list +var _ = Describe("decisions usecase", func() { + // A decision model never generates text, so a declared decisions list // must stay authoritative and the heuristic must never guess the flag. It("is authoritative when declared and never guessed", func() { - declared := GetUsecasesFromYAML([]string{"systemone"}) + declared := GetUsecasesFromYAML([]string{"decisions"}) Expect(declared).NotTo(BeNil()) Expect(*declared).NotTo(Equal(FLAG_ANY)) @@ -984,7 +984,7 @@ var _ = Describe("systemone usecase", func() { }) It("is a reserved usecase for the GGUF importer chat-default guard", func() { - declared := GetUsecasesFromYAML([]string{"systemone"}) + declared := GetUsecasesFromYAML([]string{"decisions"}) Expect(reservedNonChatModel(&ModelConfig{Backend: "vllm-cpp", KnownUsecases: declared})).To(BeTrue()) }) }) diff --git a/core/gallery/vllm_cpp_tags_test.go b/core/gallery/vllm_cpp_tags_test.go index 799dd471e..b84f9a2b1 100644 --- a/core/gallery/vllm_cpp_tags_test.go +++ b/core/gallery/vllm_cpp_tags_test.go @@ -19,7 +19,7 @@ var _ = Describe("gallery/index.yaml vllm-cpp capability tags", func() { Expect(err).ToNot(HaveOccurred()) tagToFlag := map[string]config.ModelConfigUsecase{ - "systemone": config.FLAG_SYSTEMONE, + "decisions": config.FLAG_DECISIONS, "vision": config.FLAG_VISION, "token-classify": config.FLAG_TOKEN_CLASSIFY, "scoring": config.FLAG_SCORE, diff --git a/core/http/auth/features.go b/core/http/auth/features.go index 94ae73f4d..4c1f53ec2 100644 --- a/core/http/auth/features.go +++ b/core/http/auth/features.go @@ -71,10 +71,10 @@ var RouteFeatureRegistry = []RouteFeature{ // Detection {"POST", "/v1/detection", FeatureDetection}, - // SystemOne decision API - {"POST", "/v1/systemone", FeatureSystemOne}, - {"POST", "/v1/systemone/permute", FeatureSystemOne}, - {"POST", "/v1/systemone/separate", FeatureSystemOne}, + // Decisions API (SystemOne wire contract) + {"POST", "/v1/systemone", FeatureDecisions}, + {"POST", "/v1/systemone/permute", FeatureDecisions}, + {"POST", "/v1/systemone/separate", FeatureDecisions}, // Face recognition {"POST", "/v1/face/verify", FeatureFaceRecognition}, @@ -214,6 +214,6 @@ func APIFeatureMetas() []FeatureMeta { {FeatureVoiceRecognition, "Voice Recognition", true}, {FeatureAudioTransform, "Audio Transform", true}, {FeaturePIIFilter, "PII Analyze / Redact", true}, - {FeatureSystemOne, "SystemOne Decisions", true}, + {FeatureDecisions, "Decisions", true}, } } diff --git a/core/http/auth/features_systemone_test.go b/core/http/auth/features_decisions_test.go similarity index 70% rename from core/http/auth/features_systemone_test.go rename to core/http/auth/features_decisions_test.go index 31cfc6cb7..4c5528618 100644 --- a/core/http/auth/features_systemone_test.go +++ b/core/http/auth/features_decisions_test.go @@ -6,19 +6,19 @@ import ( . "github.com/onsi/gomega" ) -var _ = Describe("SystemOne feature registration", func() { +var _ = Describe("Decisions feature registration", func() { It("gates the three decision routes behind one default-on API feature", func() { - Expect(APIFeatures).To(ContainElement(FeatureSystemOne)) + Expect(APIFeatures).To(ContainElement(FeatureDecisions)) patterns := []string{} for _, route := range RouteFeatureRegistry { - if route.Feature == FeatureSystemOne { + if route.Feature == FeatureDecisions { Expect(route.Method).To(Equal("POST")) patterns = append(patterns, route.Pattern) } } Expect(patterns).To(ConsistOf("/v1/systemone", "/v1/systemone/permute", "/v1/systemone/separate")) - Expect(APIFeatureMetas()).To(ContainElement(FeatureMeta{Key: FeatureSystemOne, Label: "SystemOne Decisions", DefaultValue: true})) + Expect(APIFeatureMetas()).To(ContainElement(FeatureMeta{Key: FeatureDecisions, Label: "Decisions", DefaultValue: true})) }) }) diff --git a/core/http/auth/permissions.go b/core/http/auth/permissions.go index c01f6e72b..3f8b6deff 100644 --- a/core/http/auth/permissions.go +++ b/core/http/auth/permissions.go @@ -59,7 +59,7 @@ const ( FeatureFaceRecognition = "face_recognition" FeatureVoiceRecognition = "voice_recognition" FeatureAudioTransform = "audio_transform" - FeatureSystemOne = "systemone" + FeatureDecisions = "decisions" // FeaturePIIFilter gates the synchronous PII analyze/redact service // (POST /api/pii/{analyze,redact}). Default ON like the other API // features; the admin-only events log is gated separately in-handler. @@ -79,7 +79,7 @@ var APIFeatures = []string{ FeatureVAD, FeatureDetection, FeatureVideo, Feature3D, FeatureEmbeddings, FeatureSound, FeatureRealtime, FeatureModeration, FeatureRerank, FeatureTokenize, FeatureMCP, FeatureStores, FeatureFaceRecognition, FeatureVoiceRecognition, FeatureAudioTransform, - FeaturePIIFilter, FeatureSystemOne, + FeaturePIIFilter, FeatureDecisions, } // AllFeatures lists all known features (used by UI and validation). diff --git a/core/http/endpoints/localai/api_instructions.go b/core/http/endpoints/localai/api_instructions.go index 702c771ff..107f4fe59 100644 --- a/core/http/endpoints/localai/api_instructions.go +++ b/core/http/endpoints/localai/api_instructions.go @@ -106,10 +106,10 @@ var instructionDefs = []instructionDef{ Intro: "Voice (speaker) recognition — the audio analog to /v1/face/*. Use /v1/voice/verify for 1:1 speaker comparison, /v1/voice/identify for 1:N match against the registered store, /v1/voice/{register,forget} to manage that store, /v1/voice/embed for a raw speaker-encoder vector, and /v1/voice/analyze for age / gender / emotion inferred from speech. Registrations are in-memory by default and lost on restart. Audio inputs accept URL, base64, or data-URI; /v1/embeddings remains text-only.", }, { - Name: "systemone", + Name: "decisions", Description: "Typed decisions (choice, noul, score) over a state text with calibrated confidence", Tags: []string{"systemone"}, - Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { : { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [systemone] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. The wire shape matches Ollama's /v1/systemone.", + Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { : { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [decisions] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. The wire shape matches Ollama's /v1/systemone.", }, { Name: "branding", diff --git a/core/http/endpoints/localai/api_instructions_test.go b/core/http/endpoints/localai/api_instructions_test.go index 727f4cb86..a3b504304 100644 --- a/core/http/endpoints/localai/api_instructions_test.go +++ b/core/http/endpoints/localai/api_instructions_test.go @@ -82,7 +82,7 @@ var _ = Describe("API Instructions Endpoints", func() { "voice-library", "3d", "failover", - "systemone", + "decisions", )) }) }) @@ -137,15 +137,15 @@ var _ = Describe("API Instructions Endpoints", func() { Expect(string(body)).NotTo(ContainSubstring("/v1/3d/generations")) }) - It("should advertise the SystemOne decisions API", func() { - req := httptest.NewRequest(http.MethodGet, "/api/instructions/systemone", nil) + It("should advertise the Decisions API", func() { + req := httptest.NewRequest(http.MethodGet, "/api/instructions/decisions", nil) rec := httptest.NewRecorder() app.ServeHTTP(rec, req) Expect(rec.Code).To(Equal(http.StatusOK)) body, _ := io.ReadAll(rec.Body) Expect(string(body)).To(ContainSubstring("POST /v1/systemone")) - Expect(string(body)).To(ContainSubstring("known_usecases: [systemone]")) + Expect(string(body)).To(ContainSubstring("known_usecases: [decisions]")) }) It("should return JSON fragment when format=json", func() { diff --git a/core/http/endpoints/localai/systemone.go b/core/http/endpoints/localai/systemone.go index 17f68a71a..64ba78e1f 100644 --- a/core/http/endpoints/localai/systemone.go +++ b/core/http/endpoints/localai/systemone.go @@ -379,10 +379,10 @@ func systemOneModelAllowed(cfg config.ModelConfig) error { if cfg.KnownUsecases == nil { return nil } - if *cfg.KnownUsecases&(config.FLAG_SYSTEMONE|config.FLAG_TOKEN_CLASSIFY) != 0 { + if *cfg.KnownUsecases&(config.FLAG_DECISIONS|config.FLAG_TOKEN_CLASSIFY) != 0 { return nil } - return fmt.Errorf("model %q does not declare the systemone usecase (known_usecases: [systemone])", cfg.Name) + return fmt.Errorf("model %q does not declare the decisions usecase (known_usecases: [decisions])", cfg.Name) } // checkSystemOneModel applies systemOneModelAllowed to a model looked up by @@ -405,7 +405,7 @@ func checkSystemOneModel(app *application.Application, modelName string) error { // A model that declares token_classify without systemone is a zero-shot NER // model: the backend's decision entry point refuses those architectures, so it // goes to the NER path instead. A config that declares nothing keeps the -// decision pipeline, which is what setups that predate the systemone usecase +// decision pipeline, which is what setups that predate the decisions usecase // relied on. func systemOneUsesDecisionPipeline(cfg config.ModelConfig) bool { if !backendSupportsScore(cfg.Backend) { @@ -415,7 +415,7 @@ func systemOneUsesDecisionPipeline(cfg config.ModelConfig) bool { return true } declared := *cfg.KnownUsecases - if declared&config.FLAG_SYSTEMONE != 0 { + if declared&config.FLAG_DECISIONS != 0 { return true } return declared&config.FLAG_TOKEN_CLASSIFY == 0 @@ -429,7 +429,7 @@ func systemOneNERAllowed(cfg config.ModelConfig) error { return nil } declared := *cfg.KnownUsecases - if declared&config.FLAG_SYSTEMONE != 0 && declared&config.FLAG_TOKEN_CLASSIFY == 0 { + if declared&config.FLAG_DECISIONS != 0 && declared&config.FLAG_TOKEN_CLASSIFY == 0 { return fmt.Errorf("model %q is a decision model: /permute and /separate use the NER path, use POST /v1/systemone instead", cfg.Name) } return nil diff --git a/core/http/endpoints/localai/systemone_gate_test.go b/core/http/endpoints/localai/systemone_gate_test.go index b790c8df2..65978c463 100644 --- a/core/http/endpoints/localai/systemone_gate_test.go +++ b/core/http/endpoints/localai/systemone_gate_test.go @@ -16,8 +16,8 @@ var _ = Describe("systemOneModelAllowed", func() { } } - It("accepts a declared systemone model", func() { - Expect(systemOneModelAllowed(mk("systemone"))).To(Succeed()) + It("accepts a declared decisions model", func() { + Expect(systemOneModelAllowed(mk("decisions"))).To(Succeed()) }) It("accepts a token_classify model, which the NER path serves", func() { @@ -29,7 +29,7 @@ var _ = Describe("systemOneModelAllowed", func() { }) It("refuses a chat-only model with an actionable message", func() { - Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [systemone]"))) + Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [decisions]"))) }) }) @@ -44,7 +44,7 @@ var _ = Describe("systemone routing by model kind", func() { Describe("systemOneUsesDecisionPipeline", func() { It("sends a declared decision model to the decision pipeline", func() { - Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "systemone"))).To(BeTrue()) + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions"))).To(BeTrue()) }) It("sends a token_classify model to the NER path, since vllm_decide refuses NER architectures", func() { Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "token_classify"))).To(BeFalse()) @@ -53,16 +53,16 @@ var _ = Describe("systemone routing by model kind", func() { Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp"))).To(BeTrue()) }) It("prefers the decision pipeline when both usecases are declared", func() { - Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "systemone", "token_classify"))).To(BeTrue()) + Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions", "token_classify"))).To(BeTrue()) }) It("never uses it for a backend without the Score RPC", func() { - Expect(systemOneUsesDecisionPipeline(mk("no-such-backend", "systemone"))).To(BeFalse()) + Expect(systemOneUsesDecisionPipeline(mk("no-such-backend", "decisions"))).To(BeFalse()) }) }) Describe("systemOneNERAllowed", func() { It("refuses a decision model on the NER-only routes with an actionable message", func() { - Expect(systemOneNERAllowed(mk("vllm-cpp", "systemone"))).To(MatchError(ContainSubstring("/v1/systemone"))) + Expect(systemOneNERAllowed(mk("vllm-cpp", "decisions"))).To(MatchError(ContainSubstring("/v1/systemone"))) }) It("accepts a token_classify model", func() { Expect(systemOneNERAllowed(mk("vllm-cpp", "token_classify"))).To(Succeed()) diff --git a/core/http/react-ui/e2e/models-lifecycle.spec.js b/core/http/react-ui/e2e/models-lifecycle.spec.js index 6587bff8b..0dff49857 100644 --- a/core/http/react-ui/e2e/models-lifecycle.spec.js +++ b/core/http/react-ui/e2e/models-lifecycle.spec.js @@ -172,17 +172,17 @@ test.describe('Models lifecycle', () => { await expect(installedPane(page)).toContainText('Worker one') }) - test('shows the systemone use case on a decision model', async ({ page }) => { + test('shows the decisions use case on a decision model', async ({ page }) => { await page.route('**/api/models/capabilities', route => route.fulfill({ contentType: 'application/json', body: JSON.stringify({ - data: [...installedModels, { id: 'decider', backend: 'vllm-cpp', capabilities: ['FLAG_SYSTEMONE'] }], + data: [...installedModels, { id: 'decider', backend: 'vllm-cpp', capabilities: ['FLAG_DECISIONS'] }], }), })) await page.goto('/app/models?view=installed&model=decider') await expect(installedPane(page)).toContainText('decider') - await expect(installedPane(page)).toContainText('SystemOne') + await expect(installedPane(page)).toContainText('Decisions') }) test('stops a running model with confirmation', async ({ page }) => { diff --git a/core/http/react-ui/public/locales/de/models.json b/core/http/react-ui/public/locales/de/models.json index d5c58a6b9..af487c86e 100644 --- a/core/http/react-ui/public/locales/de/models.json +++ b/core/http/react-ui/public/locales/de/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/en/models.json b/core/http/react-ui/public/locales/en/models.json index b5ab38303..f60dc3051 100644 --- a/core/http/react-ui/public/locales/en/models.json +++ b/core/http/react-ui/public/locales/en/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/es/models.json b/core/http/react-ui/public/locales/es/models.json index 27fc03752..eddf6a0b4 100644 --- a/core/http/react-ui/public/locales/es/models.json +++ b/core/http/react-ui/public/locales/es/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/id/models.json b/core/http/react-ui/public/locales/id/models.json index 67ea168f7..4ca0bcd2a 100644 --- a/core/http/react-ui/public/locales/id/models.json +++ b/core/http/react-ui/public/locales/id/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/it/models.json b/core/http/react-ui/public/locales/it/models.json index 0f4c80e56..b67d9da75 100644 --- a/core/http/react-ui/public/locales/it/models.json +++ b/core/http/react-ui/public/locales/it/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/ko/models.json b/core/http/react-ui/public/locales/ko/models.json index 74874ed70..e47a7da87 100644 --- a/core/http/react-ui/public/locales/ko/models.json +++ b/core/http/react-ui/public/locales/ko/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/pt-BR/models.json b/core/http/react-ui/public/locales/pt-BR/models.json index 8a795411e..3f0e5c97a 100644 --- a/core/http/react-ui/public/locales/pt-BR/models.json +++ b/core/http/react-ui/public/locales/pt-BR/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/public/locales/zh-CN/models.json b/core/http/react-ui/public/locales/zh-CN/models.json index e130ff78a..524e49933 100644 --- a/core/http/react-ui/public/locales/zh-CN/models.json +++ b/core/http/react-ui/public/locales/zh-CN/models.json @@ -46,7 +46,7 @@ "open": { "title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS", "transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings", - "rerank": "Rerank", "vad": "VAD", "score": "Score", "systemone": "SystemOne" + "rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions" }, "empty": { "title": "No models installed yet", "text": "Explore the gallery or import a model to get started.", diff --git a/core/http/react-ui/src/pages/InstalledModels.jsx b/core/http/react-ui/src/pages/InstalledModels.jsx index c607a61ac..fce53a124 100644 --- a/core/http/react-ui/src/pages/InstalledModels.jsx +++ b/core/http/react-ui/src/pages/InstalledModels.jsx @@ -22,7 +22,7 @@ import { CAP_CHAT, CAP_COMPLETION, CAP_IMAGE, CAP_VIDEO, CAP_TTS, CAP_TRANSCRIPT, CAP_SOUND_GENERATION, CAP_FACE_RECOGNITION, CAP_SPEAKER_RECOGNITION, CAP_EMBEDDINGS, CAP_RERANK, - CAP_VAD, CAP_SCORE, CAP_SYSTEMONE, + CAP_VAD, CAP_SCORE, CAP_DECISIONS, } from '../utils/capabilities' const USE_CASES = [ @@ -39,7 +39,7 @@ const USE_CASES = [ { cap: CAP_RERANK, labelKey: 'rerank' }, { cap: CAP_VAD, labelKey: 'vad' }, { cap: CAP_SCORE, labelKey: 'score' }, - { cap: CAP_SYSTEMONE, labelKey: 'systemone' }, + { cap: CAP_DECISIONS, labelKey: 'decisions' }, ] export function modelUseCases(model) { diff --git a/core/http/react-ui/src/utils/capabilities.js b/core/http/react-ui/src/utils/capabilities.js index 0775ef8d9..722c85842 100644 --- a/core/http/react-ui/src/utils/capabilities.js +++ b/core/http/react-ui/src/utils/capabilities.js @@ -29,5 +29,5 @@ export const CAP_SPEAKER_RECOGNITION = 'FLAG_SPEAKER_RECOGNITION' export const CAP_AUDIO_TRANSFORM = 'FLAG_AUDIO_TRANSFORM' export const CAP_REALTIME_AUDIO = 'FLAG_REALTIME_AUDIO' export const CAP_SCORE = 'FLAG_SCORE' -export const CAP_SYSTEMONE = 'FLAG_SYSTEMONE' +export const CAP_DECISIONS = 'FLAG_DECISIONS' export const CAP_TOKEN_CLASSIFY = 'FLAG_TOKEN_CLASSIFY' diff --git a/docs/content/advanced/model-configuration.md b/docs/content/advanced/model-configuration.md index 10e71ef96..c1870e3ec 100644 --- a/docs/content/advanced/model-configuration.md +++ b/docs/content/advanced/model-configuration.md @@ -1066,9 +1066,9 @@ known_usecases: - embeddings ``` -Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `score`, `token_classify`, `systemone`, `llm` (combination of CHAT, COMPLETION, EDIT). +Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `score`, `token_classify`, `decisions`, `llm` (combination of CHAT, COMPLETION, EDIT). -`systemone` marks a model as a decision model for the [SystemOne API]({{% relref "features/systemone" %}}) (`POST /v1/systemone`). It is never guessed, and a model that declares it is not listed as a chat, completion or embeddings model. +`decisions` marks a model as a decision model for the [Decisions API]({{% relref "features/decisions" %}}) (`POST /v1/systemone`). It is never guessed, and a model that declares it is not listed as a chat, completion or embeddings model. `token_classify` marks a model as a token-classification (NER) provider for the PII filter (e.g. an `openai-privacy-filter` GGUF). Declare it explicitly together with `embeddings: true` (the classifier loads via TOKEN_CLS pooling). It runs on the dedicated `privacy-filter` backend (`backend/cpp/privacy-filter`), a standalone GGML engine for the `openai-privacy-filter` family - separate from `llama-cpp`, which no longer carries the token-classification path. diff --git a/docs/content/features/systemone.md b/docs/content/features/decisions.md similarity index 78% rename from docs/content/features/systemone.md rename to docs/content/features/decisions.md index 84c65a665..fa3422aaa 100644 --- a/docs/content/features/systemone.md +++ b/docs/content/features/decisions.md @@ -1,17 +1,19 @@ +++ disableToc = false -title = "SystemOne decisions" +title = "Decisions API" weight = 66 -url = "/features/systemone/" +url = "/features/decisions/" +++ -SystemOne is an API for fast, typed decisions. You send a piece of text (the +The Decisions API is a fast, typed decision layer. You send a piece of text (the *state*) and a set of named questions. A decision model answers each question with a value and a confidence, in one pass. The model does not generate text, so there is nothing to parse and no free-form output to validate. -The request and response shapes follow the [kev](https://github.com/jaredpalmer/kev) -project and match the `/v1/systemone` endpoint that Ollama added in 0.35. +LocalAI serves it on the `/v1/systemone` routes. The request and response shapes +follow the [kev](https://github.com/jaredpalmer/kev) project and match the +`/v1/systemone` endpoint that Ollama added in 0.35. The wire contract is called +SystemOne; the capability a model declares is called `decisions`. ## Endpoints @@ -25,7 +27,7 @@ Which route a model can serve depends on its kind: | Model kind | `/v1/systemone` | `/permute` and `/separate` | |---|---|---| -| Decision model (`systemone`), such as Laya or GLiNER2.5-Decide | Yes | No, returns `400` | +| Decision model (`decisions`), such as Laya or GLiNER2.5-Decide | Yes | No, returns `400` | | Zero-shot NER model (`token_classify`), such as GLiNER2.5 | Yes, through the NER path | Yes | ## Question types @@ -70,28 +72,28 @@ reports token usage and `latency_ms`. The NER path does not report token usage. ## Choosing a model -A model can serve SystemOne only if it is a decision model. Declare the usecase +A model can serve the Decisions API only if it is a decision model. Declare the usecase in the model config: ```yaml name: laya backend: vllm-cpp known_usecases: - - systemone + - decisions parameters: model: convaiinnovations/laya ``` -`systemone` is never guessed, and a model that declares it is not listed as a +`decisions` is never guessed, and a model that declares it is not listed as a chat, completion or embeddings model. A model that declares usecases without -`systemone` or `token_classify` gets a `400` from these endpoints that names the -missing usecase. A model that declares `token_classify` and not `systemone` is +`decisions` or `token_classify` gets a `400` from these endpoints that names the +missing usecase. A model that declares `token_classify` and not `decisions` is served by the zero-shot NER path. A vllm-cpp config that declares no usecases is treated as a decision model, so setups that predate the flag keep working, but a config that declares only `chat` (as an older `laya` gallery entry did) now gets -the `400` and needs `known_usecases: [systemone]`. +the `400` and needs `known_usecases: [decisions]`. -Install one from the gallery and filter on the `systemone` tag: +Install one from the gallery and filter on the `decisions` tag: | Gallery entry | Model | Notes | |---|---|---| @@ -107,6 +109,6 @@ and does not serve `/v1/systemone` yet. ## Access control -When authentication is on, the three routes need the `systemone` feature. It is +When authentication is on, the three routes need the `decisions` feature. It is on by default for every user, like the other API features, and an administrator can turn it off per user. diff --git a/docs/content/features/vllm-cpp.md b/docs/content/features/vllm-cpp.md index ba9840302..bf7323e0d 100644 --- a/docs/content/features/vllm-cpp.md +++ b/docs/content/features/vllm-cpp.md @@ -160,12 +160,12 @@ forward, which is the required contract for pooling models in vllm.cpp. A device-resident forward is tracked as a performance optimization, not a correctness gap. -### SystemOne decision API +### Decisions API -The `vllm-cpp` backend serves the kev-compatible SystemOne endpoints: typed +The `vllm-cpp` backend serves the kev-compatible SystemOne endpoints (the Decisions API): typed `choice`, `noul` and `score` questions over a state text, answered by a non-generative decision model in one pass. A decision model declares -`known_usecases: [systemone]`. See [SystemOne decisions]({{% relref "features/systemone" %}}) +`known_usecases: [decisions]`. See [Decisions API]({{% relref "features/decisions" %}}) for the request shape, the models you can install and the access rules. | Endpoint | Method | Description | diff --git a/gallery/index.yaml b/gallery/index.yaml index 10276000a..5bded4aaf 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -63653,7 +63653,7 @@ 512-token context. F16 weights, ~804 MB. license: apache-2.0 tags: - - decision + - decisions - systemone - vllm-cpp - cpu @@ -63663,7 +63663,7 @@ overrides: backend: vllm-cpp known_usecases: - - systemone + - decisions parameters: model: convaiinnovations/laya artifacts: @@ -63690,7 +63690,7 @@ checkpoint it was checked against. license: apache-2.0 tags: - - decision + - decisions - systemone - vllm-cpp - cpu @@ -63700,7 +63700,7 @@ overrides: backend: vllm-cpp known_usecases: - - systemone + - decisions parameters: model: fastino/GLiNER2.5-Decide artifacts: From da37a4d9299d53067da88bf03d5cd7add2bc1f53 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Wed, 30 Sep 2026 16:17:00 +0200 Subject: [PATCH 23/33] chore: :arrow_up: Update localai-org/ced.cpp to `61dec2ab0106f2047ee40062a7075dbf08c523d0` (#12365) :arrow_up: Update localai-org/ced.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/ced/Makefile | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/backend/go/ced/Makefile b/backend/go/ced/Makefile index a2676f8f2..b9497856b 100644 --- a/backend/go/ced/Makefile +++ b/backend/go/ced/Makefile @@ -1,6 +1,6 @@ # ced sound-classification backend Makefile. # -# Upstream pin lives below as CED_VERSION?=b10237678d1c3b30c77d19f2e63f6c198c7f8d09 +# Upstream pin lives below as CED_VERSION?=61dec2ab0106f2047ee40062a7075dbf08c523d0 # and update it (matches the parakeet-cpp / whisper.cpp convention). # # Local dev shortcut: symlink an out-of-tree ced.cpp shared build + header and @@ -9,7 +9,7 @@ # ln -sf /path/to/ced.cpp/include/ced_capi.h . # go build -o ced-grpc . -CED_VERSION?=b10237678d1c3b30c77d19f2e63f6c198c7f8d09 +CED_VERSION?=61dec2ab0106f2047ee40062a7075dbf08c523d0 CED_REPO?=https://github.com/localai-org/ced.cpp GOCMD?=go From c61e316d1f635a9b6ef7c2ebe4e7a1de92379324 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Wed, 30 Sep 2026 16:17:15 +0200 Subject: [PATCH 24/33] chore: :arrow_up: Update ikawrakow/ik_llama.cpp to `0821d62a8b356bd1db3c6765551a30bfcc44a6de` (#12364) :arrow_up: Update ikawrakow/ik_llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/ik-llama-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/ik-llama-cpp/Makefile b/backend/cpp/ik-llama-cpp/Makefile index 569df41fb..167bf61ac 100644 --- a/backend/cpp/ik-llama-cpp/Makefile +++ b/backend/cpp/ik-llama-cpp/Makefile @@ -1,5 +1,5 @@ -IK_LLAMA_VERSION?=d741de5074cd424dd3ba7cfc4d9b7649f1eb0463 +IK_LLAMA_VERSION?=0821d62a8b356bd1db3c6765551a30bfcc44a6de LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp CMAKE_ARGS?= From 82aaea3a4d575564265b294182c3ef4cb761783d Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Wed, 30 Sep 2026 16:17:29 +0200 Subject: [PATCH 25/33] chore: :arrow_up: Update CrispStrobe/CrispASR to `be202c472503a5c7f1d3e568c420865cad02f1c3` (#12363) :arrow_up: Update CrispStrobe/CrispASR Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/go/crispasr/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/go/crispasr/Makefile b/backend/go/crispasr/Makefile index 35f915924..947023262 100644 --- a/backend/go/crispasr/Makefile +++ b/backend/go/crispasr/Makefile @@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1) # CrispASR version (release tag) CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR -CRISPASR_VERSION?=2cd383a926e3c37334e75eb5d8b8a85bd82ed22d +CRISPASR_VERSION?=be202c472503a5c7f1d3e568c420865cad02f1c3 SO_TARGET?=libgocrispasr.so CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF From 64291c6cd02ee6be894d1a8729586bd79e9f1d2c Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Wed, 30 Sep 2026 16:17:43 +0200 Subject: [PATCH 26/33] chore: :arrow_up: Update 0xShug0/audio.cpp to `ed96b7307c8daba2ebcf7912af928825f6b14cb9` (#12362) :arrow_up: Update 0xShug0/audio.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/audio-cpp/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/audio-cpp/Makefile b/backend/cpp/audio-cpp/Makefile index 16ec08f5a..9a1928e52 100644 --- a/backend/cpp/audio-cpp/Makefile +++ b/backend/cpp/audio-cpp/Makefile @@ -9,7 +9,7 @@ # recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean # rebuild and so the bump bot can see the pin. -AUDIO_CPP_VERSION?=f825d1d1b92af309585aeb656b2a59c44fc603eb +AUDIO_CPP_VERSION?=ed96b7307c8daba2ebcf7912af928825f6b14cb9 AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST)))) From 5613572f3830b3bcb1374d6dc3770a36a6ede25e Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Wed, 30 Sep 2026 14:31:50 +0000 Subject: [PATCH 27/33] feat(systemone): validate requests and align the docs with Ollama's contract All three routes now validate the request before it reaches a model: body size (413 over 64 KiB), state, question count, blank ids, option and level counts, and noul criteria keys. Forwarded decision requests skipped this before, so a malformed question surfaced as a backend error. The docs claimed the wire shape matches Ollama's. Field names and question types do; confidence, error shape, keep_alive and state rendering differ, and the docs now say so. Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto --- .../endpoints/localai/api_instructions.go | 2 +- core/http/endpoints/localai/systemone.go | 129 +++++++++++++++++- .../localai/systemone_validate_test.go | 96 +++++++++++++ docs/content/features/decisions.md | 45 +++++- 4 files changed, 262 insertions(+), 10 deletions(-) create mode 100644 core/http/endpoints/localai/systemone_validate_test.go diff --git a/core/http/endpoints/localai/api_instructions.go b/core/http/endpoints/localai/api_instructions.go index 107f4fe59..e65e61c38 100644 --- a/core/http/endpoints/localai/api_instructions.go +++ b/core/http/endpoints/localai/api_instructions.go @@ -109,7 +109,7 @@ var instructionDefs = []instructionDef{ Name: "decisions", Description: "Typed decisions (choice, noul, score) over a state text with calibrated confidence", Tags: []string{"systemone"}, - Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { : { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [decisions] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. The wire shape matches Ollama's /v1/systemone.", + Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { : { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [decisions] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. Field names and question types follow Ollama's /v1/systemone, with differences in confidence, error shape and keep_alive (see the Decisions API docs). A request over 64 KiB, with more than 64 questions, or with a malformed question is refused.", }, { Name: "branding", diff --git a/core/http/endpoints/localai/systemone.go b/core/http/endpoints/localai/systemone.go index 64ba78e1f..7bb3e81b8 100644 --- a/core/http/endpoints/localai/systemone.go +++ b/core/http/endpoints/localai/systemone.go @@ -2,6 +2,7 @@ package localai import ( "encoding/json" + "errors" "fmt" "math" "math/rand" @@ -449,6 +450,113 @@ func checkSystemOneNERModel(app *application.Application, modelName string) erro return systemOneNERAllowed(cfg) } +// systemOneMaxBody and systemOneMaxQuestions bound one request. They keep a +// single call from pinning a decision model on an unbounded prompt, and match +// the limits Ollama documents for the same wire contract, so a client written +// for one server behaves the same on the other. The engine enforces any +// per-model option cap (letter-answer models refuse more than 26 options). +const ( + systemOneMaxBody = 64 << 10 + systemOneMaxQuestions = 64 +) + +// systemOneBind binds the JSON body with a size cap. Bind reads the whole body +// first, so the cap has to be on the reader. +func systemOneBind(c echo.Context, v any) error { + c.Request().Body = http.MaxBytesReader(c.Response(), c.Request().Body, systemOneMaxBody) + return c.Bind(v) +} + +// systemOneBindStatus maps a bind failure to its status: 413 when the body +// exceeded the cap, 400 for anything else. +func systemOneBindStatus(err error) int { + var tooLarge *http.MaxBytesError + if errors.As(err, &tooLarge) { + return http.StatusRequestEntityTooLarge + } + return http.StatusBadRequest +} + +func systemOneBindMessage(err error) string { + if systemOneBindStatus(err) == http.StatusRequestEntityTooLarge { + return fmt.Sprintf("request body exceeds %d KiB", systemOneMaxBody>>10) + } + return "invalid request body" +} + +// validateSystemOneRequest checks the structure every path needs, before the +// request is forwarded to a decision model or run through the NER path. The +// forwarded path never sees parseSystemOneRequest, so without this a malformed +// question would surface as a backend error instead of a 400. +func validateSystemOneRequest(req *schema.SystemOneRequest) error { + if len(req.State) == 0 || string(req.State) == "null" { + return fmt.Errorf("state is required") + } + var state any + if err := json.Unmarshal(req.State, &state); err != nil { + return fmt.Errorf("state is not valid JSON: %w", err) + } + if s, ok := state.(string); ok && strings.TrimSpace(s) == "" { + return fmt.Errorf("state is required") + } + if len(req.Questions) == 0 { + return fmt.Errorf("questions is required and must contain at least one question") + } + if len(req.Questions) > systemOneMaxQuestions { + return fmt.Errorf("questions must contain at most %d questions", systemOneMaxQuestions) + } + qids := make([]string, 0, len(req.Questions)) + for id := range req.Questions { + qids = append(qids, id) + } + sort.Strings(qids) + for _, id := range qids { + if strings.TrimSpace(id) == "" { + return fmt.Errorf("question ids must not be blank") + } + q := req.Questions[id] + switch q.Type { + case "choice": + var criteria map[string]json.RawMessage + if err := json.Unmarshal(q.Criteria, &criteria); err != nil { + return fmt.Errorf("question %q (choice) requires a criteria object", id) + } + if len(criteria) < 2 { + return fmt.Errorf("question %q (choice) requires at least 2 options", id) + } + for k := range criteria { + if strings.TrimSpace(k) == "" { + return fmt.Errorf("question %q (choice) has a blank option key", id) + } + } + case "score": + var criteria []json.RawMessage + if err := json.Unmarshal(q.Criteria, &criteria); err != nil { + return fmt.Errorf("question %q (score) requires a criteria array", id) + } + if len(criteria) < 2 { + return fmt.Errorf("question %q (score) requires at least 2 levels", id) + } + case "noul": + if len(q.Criteria) == 0 || string(q.Criteria) == "null" { + continue + } + var criteria map[string]json.RawMessage + if err := json.Unmarshal(q.Criteria, &criteria); err != nil { + return fmt.Errorf("question %q (noul) criteria must be an object with \"false\" and \"true\" descriptions", id) + } + for k := range criteria { + if k != "false" && k != "true" { + return fmt.Errorf("question %q (noul) criteria may only have \"false\" and \"true\" keys", id) + } + } + default: + return fmt.Errorf("question %q has unknown type: %s", id, q.Type) + } + } + return nil +} + // backendSupportsScore reports whether the named backend implements the // Score gRPC RPC. vllm-cpp does (kev/laya decision pipeline and cua-s1-forms // scoring via the unified vllm_decide C ABI); other backends fall through to @@ -480,8 +588,8 @@ func backendSupportsScore(backendName string) bool { func SystemOneEndpoint(app *application.Application) echo.HandlerFunc { return func(c echo.Context) error { var req schema.SystemOneRequest - if err := c.Bind(&req); err != nil { - return systemOneError(c, http.StatusBadRequest, "invalid request body") + if err := systemOneBind(c, &req); err != nil { + return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err)) } if req.Model == "" { return systemOneError(c, http.StatusBadRequest, "model is required") @@ -489,6 +597,9 @@ func SystemOneEndpoint(app *application.Application) echo.HandlerFunc { if err := checkSystemOneModel(app, req.Model); err != nil { return systemOneError(c, http.StatusBadRequest, err.Error()) } + if err := validateSystemOneRequest(&req); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } // vllm-cpp models (kev/laya) implement the decision pipeline natively // via the vllm_decide C ABI. Forward the raw request JSON through the // Score RPC and return the backend's response as-is. @@ -549,8 +660,8 @@ func SystemOneEndpoint(app *application.Application) echo.HandlerFunc { func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc { return func(c echo.Context) error { var req schema.SystemOnePermuteRequest - if err := c.Bind(&req); err != nil { - return systemOneError(c, http.StatusBadRequest, "invalid request body") + if err := systemOneBind(c, &req); err != nil { + return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err)) } if req.Request.Model == "" { return systemOneError(c, http.StatusBadRequest, "model is required") @@ -561,6 +672,9 @@ func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc { if err := checkSystemOneNERModel(app, req.Request.Model); err != nil { return systemOneError(c, http.StatusBadRequest, err.Error()) } + if err := validateSystemOneRequest(&req.Request); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } if req.Question == "" { return systemOneError(c, http.StatusBadRequest, "question is required") } @@ -691,8 +805,8 @@ func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc { func SystemOneSeparateEndpoint(app *application.Application) echo.HandlerFunc { return func(c echo.Context) error { var req schema.SystemOneRequest - if err := c.Bind(&req); err != nil { - return systemOneError(c, http.StatusBadRequest, "invalid request body") + if err := systemOneBind(c, &req); err != nil { + return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err)) } if req.Model == "" { return systemOneError(c, http.StatusBadRequest, "model is required") @@ -703,6 +817,9 @@ func SystemOneSeparateEndpoint(app *application.Application) echo.HandlerFunc { if err := checkSystemOneNERModel(app, req.Model); err != nil { return systemOneError(c, http.StatusBadRequest, err.Error()) } + if err := validateSystemOneRequest(&req); err != nil { + return systemOneError(c, http.StatusBadRequest, err.Error()) + } parsed, err := parseSystemOneRequest(&req) if err != nil { return systemOneError(c, http.StatusBadRequest, err.Error()) diff --git a/core/http/endpoints/localai/systemone_validate_test.go b/core/http/endpoints/localai/systemone_validate_test.go new file mode 100644 index 000000000..dfec4c27c --- /dev/null +++ b/core/http/endpoints/localai/systemone_validate_test.go @@ -0,0 +1,96 @@ +package localai + +import ( + "encoding/json" + "net/http" + "net/http/httptest" + "strings" + + "github.com/labstack/echo/v4" + "github.com/mudler/LocalAI/core/schema" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("validateSystemOneRequest", func() { + req := func(state string, questions string) *schema.SystemOneRequest { + r := &schema.SystemOneRequest{Model: "m", State: json.RawMessage(state)} + Expect(json.Unmarshal([]byte(questions), &r.Questions)).To(Succeed()) + return r + } + + It("accepts the three question types", func() { + r := req(`"ticket text"`, `{ + "team": {"type":"choice","instructions":"which","criteria":{"a":"A","b":null}}, + "refund": {"type":"noul","instructions":"refund?","criteria":{"false":"No refund","true":"Refund asked"}}, + "urgency": {"type":"score","instructions":"how urgent","criteria":["low","high"]} + }`) + Expect(validateSystemOneRequest(r)).To(Succeed()) + }) + + It("accepts a noul question with no criteria", func() { + Expect(validateSystemOneRequest(req(`"x"`, `{"q":{"type":"noul","instructions":"i"}}`))).To(Succeed()) + }) + + DescribeTable("refuses a malformed request with a message that names the problem", + func(state, questions, want string) { + Expect(validateSystemOneRequest(req(state, questions))).To(MatchError(ContainSubstring(want))) + }, + Entry("missing state", ``, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"), + Entry("null state", `null`, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"), + Entry("blank string state", `" "`, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"), + Entry("no questions", `"x"`, `{}`, "at least one question"), + Entry("blank question id", `"x"`, `{" ":{"type":"noul","instructions":"i"}}`, "blank"), + Entry("unknown type", `"x"`, `{"q":{"type":"rank","instructions":"i"}}`, "unknown type"), + Entry("choice with one option", `"x"`, `{"q":{"type":"choice","instructions":"i","criteria":{"a":"A"}}}`, "at least 2"), + Entry("choice with a blank option key", `"x"`, `{"q":{"type":"choice","instructions":"i","criteria":{"a":"A"," ":"B"}}}`, "blank"), + Entry("score with one level", `"x"`, `{"q":{"type":"score","instructions":"i","criteria":["only"]}}`, "at least 2"), + Entry("noul criteria with a stray key", `"x"`, `{"q":{"type":"noul","instructions":"i","criteria":{"maybe":"M"}}}`, `"false" and "true"`), + ) + + It("refuses more than 64 questions", func() { + var b strings.Builder + b.WriteString("{") + for i := 0; i < 65; i++ { + if i > 0 { + b.WriteString(",") + } + b.WriteString(`"q` + strings.Repeat("x", i) + `":{"type":"noul","instructions":"i"}`) + } + b.WriteString("}") + Expect(validateSystemOneRequest(req(`"x"`, b.String()))).To(MatchError(ContainSubstring("at most 64"))) + }) +}) + +var _ = Describe("systemOneBind", func() { + bind := func(body string) (int, error) { + e := echo.New() + r := httptest.NewRequest(http.MethodPost, "/v1/systemone", strings.NewReader(body)) + r.Header.Set("Content-Type", "application/json") + c := e.NewContext(r, httptest.NewRecorder()) + var out schema.SystemOneRequest + if err := systemOneBind(c, &out); err != nil { + return systemOneBindStatus(err), err + } + return http.StatusOK, nil + } + + It("binds a normal body", func() { + status, err := bind(`{"model":"m","state":"x","questions":{}}`) + Expect(err).ToNot(HaveOccurred()) + Expect(status).To(Equal(http.StatusOK)) + }) + + It("answers 413 for a body over 64 KiB", func() { + status, err := bind(`{"model":"m","state":"` + strings.Repeat("a", 65*1024) + `"}`) + Expect(err).To(HaveOccurred()) + Expect(status).To(Equal(http.StatusRequestEntityTooLarge)) + }) + + It("answers 400 for malformed JSON", func() { + status, err := bind(`{not json`) + Expect(err).To(HaveOccurred()) + Expect(status).To(Equal(http.StatusBadRequest)) + }) +}) diff --git a/docs/content/features/decisions.md b/docs/content/features/decisions.md index fa3422aaa..0118f110d 100644 --- a/docs/content/features/decisions.md +++ b/docs/content/features/decisions.md @@ -11,9 +11,15 @@ with a value and a confidence, in one pass. The model does not generate text, so there is nothing to parse and no free-form output to validate. LocalAI serves it on the `/v1/systemone` routes. The request and response shapes -follow the [kev](https://github.com/jaredpalmer/kev) project and match the -`/v1/systemone` endpoint that Ollama added in 0.35. The wire contract is called -SystemOne; the capability a model declares is called `decisions`. +follow the [kev](https://github.com/jaredpalmer/kev) project, and the field names +and question types are the same ones Ollama serves on its `/v1/systemone` +endpoint (Ollama 0.35 and later). The wire contract is called SystemOne; the +capability a model declares is called `decisions`. See +[Compatibility with Ollama](#compatibility-with-ollama) for what differs. + +OpenAI announced its own Decisions API in limited preview on 2026-09-29. It has no +public request or response schema yet, so LocalAI does not serve a `/v1/decisions` +route. ## Endpoints @@ -107,6 +113,39 @@ they are not gallery entries yet. Tev1 is an autoregressive decision model. It answers through chat completions and does not serve `/v1/systemone` yet. +## Request limits + +A request is refused with `400` (or `413` for the body size) when: + +- the body is larger than 64 KiB, +- `state` is missing or blank, +- there are no questions, or more than 64, +- a question id is blank, +- a `choice` question has fewer than 2 options or a blank option key, +- a `score` question has fewer than 2 levels, +- a `noul` question has `criteria` with keys other than `"false"` and `"true"`. + +A `noul` question may carry `criteria` with a description for each outcome, for +example `{"false": "No refund is requested", "true": "The customer requests a refund"}`. +Some models cap the number of options for a `choice` or `score` question (models +that answer with a letter accept at most 26). The engine refuses more options than +the model supports and the error names the limit. + +## Compatibility with Ollama + +The field names, question types and answer fields are the same as Ollama's +`/v1/systemone`, so a client written for one works against the other for the +common case. These behaviors differ: + +| | Ollama | LocalAI | +|---|---|---| +| `confidence` | `1 - H(p) / ln(N)`, an entropy measure | Computed by the model's pipeline. For kev and Laya it is a normalized margin, so the same probabilities give a different value | +| Errors | `{"error": "message"}` | `{"error": {"message": "...", "type": "invalid_request"}}` | +| `keep_alive` | Sets how long the model stays loaded | Accepted and ignored. Model lifetime follows the LocalAI idle and watchdog settings | +| `state` given as an object | Serialized as JSON text | Rendered as labeled lines, the way kev does it | +| `noul` answer on the NER path | `{type, noul}` | Also carries `entities` | +| Token `usage` | Full prompt lengths across all questions | Whatever the backend reports; the NER path reports 0 | + ## Access control When authentication is on, the three routes need the `decisions` feature. It is From 355c39968c5de31710ef5a05c7520d01c858524e Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Wed, 30 Sep 2026 16:53:09 +0200 Subject: [PATCH 28/33] chore: :arrow_up: Update mudler/parakeet.cpp to `623a968bccbd2214588df398fcce687cd4218dea` (#12347) :arrow_up: Update mudler/parakeet.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> From 7e0d836d21aeccd62f890a4024587812f08ccbc6 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Wed, 30 Sep 2026 16:53:22 +0200 Subject: [PATCH 29/33] chore: :arrow_up: Update TheTom/llama-cpp-turboquant to `bcb85fc3ae85efa0f5f392c6c880dfc524923860` (#12344) :arrow_up: Update TheTom/llama-cpp-turboquant Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> --- backend/cpp/turboquant/Makefile | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backend/cpp/turboquant/Makefile b/backend/cpp/turboquant/Makefile index 7a8024ebf..a3a1e7eca 100644 --- a/backend/cpp/turboquant/Makefile +++ b/backend/cpp/turboquant/Makefile @@ -1,7 +1,7 @@ # Pinned to the HEAD of feature/turboquant-kv-cache on https://github.com/TheTom/llama-cpp-turboquant. # Auto-bumped nightly by .github/workflows/bump_deps.yaml. -TURBOQUANT_VERSION?=a3d5603d110bda29222d2011596cdc84d7fa532d +TURBOQUANT_VERSION?=bcb85fc3ae85efa0f5f392c6c880dfc524923860 LLAMA_REPO?=https://github.com/TheTom/llama-cpp-turboquant CMAKE_ARGS?= From 1d7023be1a1400facf002a5751da5615c5164880 Mon Sep 17 00:00:00 2001 From: mudler-agent Date: Wed, 30 Sep 2026 19:23:24 +0200 Subject: [PATCH 30/33] test(agentpool): pin the standalone agent contract (#12378) Adds a Ginkgo contract suite for the standalone (LocalAGI-backed) agent service in core/services/agentpool, with a fake OpenAI-compatible LLM and a harness that boots a real AgentPoolService. It pins CRUD, per-user isolation, export and import, pause and resume, the chat and SSE event contract, status and observables, persistence and the raw-key agent lookup. No production code changes. Specs that pin a known gap are named known gap or known defect, so later phases flip them on purpose. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- .../agentpool/contract_helpers_test.go | 304 ++++++++++++++++++ .../standalone_chat_contract_test.go | 157 +++++++++ .../agentpool/standalone_contract_test.go | 186 +++++++++++ .../standalone_persistence_contract_test.go | 101 ++++++ .../standalone_status_contract_test.go | 232 +++++++++++++ 5 files changed, 980 insertions(+) create mode 100644 core/services/agentpool/contract_helpers_test.go create mode 100644 core/services/agentpool/standalone_chat_contract_test.go create mode 100644 core/services/agentpool/standalone_contract_test.go create mode 100644 core/services/agentpool/standalone_persistence_contract_test.go create mode 100644 core/services/agentpool/standalone_status_contract_test.go diff --git a/core/services/agentpool/contract_helpers_test.go b/core/services/agentpool/contract_helpers_test.go new file mode 100644 index 000000000..13faff6a2 --- /dev/null +++ b/core/services/agentpool/contract_helpers_test.go @@ -0,0 +1,304 @@ +package agentpool_test + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "net/http/httptest" + "strings" + "sync" + + "github.com/mudler/LocalAGI/core/sse" + "github.com/mudler/LocalAGI/core/state" + "github.com/mudler/LocalAGI/core/types" + "github.com/mudler/LocalAI/core/config" + "github.com/mudler/LocalAI/core/services/agentpool" + . "github.com/onsi/gomega" +) + +type fakeLLMRequest struct { + Model string + Stream bool + Messages []map[string]any + Tools []string + ToolChoice any +} + +// fakeLLM is an OpenAI-compatible chat endpoint. The standalone pool reaches its +// LLM over HTTP (apiURL), so a real server is the only seam that exercises the +// whole agent loop without a model. +type fakeLLM struct { + srv *httptest.Server + mu sync.Mutex + reply string + requests []fakeLLMRequest + toolName string + toolArgs string + // failStatus, when non-zero, makes every chat completion fail with that + // HTTP status so a spec can drive the agent's error path. + failStatus int +} + +func newFakeLLM(reply string) *fakeLLM { + f := &fakeLLM{reply: reply} + f.srv = httptest.NewServer(http.HandlerFunc(f.handle)) + return f +} + +func (f *fakeLLM) URL() string { return f.srv.URL } +func (f *fakeLLM) Close() { f.srv.Close() } + +func (f *fakeLLM) SetReply(r string) { + f.mu.Lock() + defer f.mu.Unlock() + f.reply = r +} + +// SetToolCall makes the fake answer with one call to the named function until +// the conversation carries a tool result, then with the plain reply. Keying on +// the tool message rather than a request counter keeps the fake independent of +// how many planning requests the agent makes before it runs the tool. Only +// the LocalAGI counter-action specs use it today; P2 keeps it for the MCP +// tool fixture that replaces them, since the native executor ignores +// Actions. The +// flip side: tool mode stays on until a request carries a role "tool" +// message, so a client that restarts with trimmed history would get the +// tool call again and loop until its iteration cap. +func (f *fakeLLM) SetToolCall(name, argsJSON string) { + f.mu.Lock() + defer f.mu.Unlock() + f.toolName = name + f.toolArgs = argsJSON +} + +// SetFailure makes every chat completion answer with the given HTTP status and +// an OpenAI-style error body, which is what an unreachable or broken backend +// looks like to the agent. Requests are still recorded so a spec can count +// retries. +func (f *fakeLLM) SetFailure(status int) { + f.mu.Lock() + defer f.mu.Unlock() + f.failStatus = status +} + +func (f *fakeLLM) Requests() []fakeLLMRequest { + f.mu.Lock() + defer f.mu.Unlock() + return append([]fakeLLMRequest(nil), f.requests...) +} + +func (f *fakeLLM) handle(w http.ResponseWriter, r *http.Request) { + body, _ := io.ReadAll(r.Body) + var req struct { + Model string `json:"model"` + Stream bool `json:"stream"` + Messages []map[string]any `json:"messages"` + Tools []struct { + Function struct { + Name string `json:"name"` + } `json:"function"` + } `json:"tools"` + ToolChoice any `json:"tool_choice"` + } + _ = json.Unmarshal(body, &req) + + var tools []string + for _, t := range req.Tools { + tools = append(tools, t.Function.Name) + } + hasToolResult := false + for _, m := range req.Messages { + if m["role"] == "tool" { + hasToolResult = true + } + } + + f.mu.Lock() + f.requests = append(f.requests, fakeLLMRequest{Model: req.Model, Stream: req.Stream, Messages: req.Messages, Tools: tools, ToolChoice: req.ToolChoice}) + reply := f.reply + failStatus := f.failStatus + var toolCall map[string]any + if f.toolName != "" && !hasToolResult { + toolCall = map[string]any{ + "index": 0, "id": "call_fake", "type": "function", + "function": map[string]any{"name": f.toolName, "arguments": f.toolArgs}, + } + } + f.mu.Unlock() + + if failStatus != 0 { + w.Header().Set("Content-Type", "application/json") + w.WriteHeader(failStatus) + _ = json.NewEncoder(w).Encode(map[string]any{ + "error": map[string]any{"message": "fake backend failure", "type": "server_error", "code": failStatus}, + }) + return + } + + message := map[string]any{"role": "assistant", "content": reply} + finish := "stop" + if toolCall != nil { + // "index" belongs only to streaming deltas, so the non-streaming + // message carries a copy without it. + plain := map[string]any{} + for k, v := range toolCall { + if k != "index" { + plain[k] = v + } + } + message = map[string]any{"role": "assistant", "content": "", "tool_calls": []map[string]any{plain}} + finish = "tool_calls" + } + + if !req.Stream { + w.Header().Set("Content-Type", "application/json") + _ = json.NewEncoder(w).Encode(map[string]any{ + "id": "chatcmpl-fake", "object": "chat.completion", "model": req.Model, + "choices": []map[string]any{{ + "index": 0, + "message": message, + "finish_reason": finish, + }}, + "usage": map[string]any{"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + }) + return + } + + w.Header().Set("Content-Type", "text/event-stream") + // A write error means the client went away mid-stream; the handler just + // stops writing so a disconnect never blocks or fails the fake. + write := func(s string) bool { + _, err := io.WriteString(w, s) + return err == nil + } + chunk := func(delta map[string]any, finish any) bool { + b, _ := json.Marshal(map[string]any{ + "id": "chatcmpl-fake", "object": "chat.completion.chunk", "model": req.Model, + "choices": []map[string]any{{"index": 0, "delta": delta, "finish_reason": finish}}, + }) + return write(fmt.Sprintf("data: %s\n\n", b)) + } + first := map[string]any{"role": "assistant", "content": reply} + if toolCall != nil { + first = map[string]any{"role": "assistant", "tool_calls": []map[string]any{toolCall}} + } + if !chunk(first, nil) || !chunk(map[string]any{}, finish) || !write("data: [DONE]\n\n") { + return + } + if fl, ok := w.(http.Flusher); ok { + fl.Flush() + } +} + +// startStandalone boots a real standalone AgentPoolService on stateDir with its +// LLM pointed at llmURL. It registers no cleanup itself: callers own Stop(). +func startStandalone(stateDir, llmURL string) *agentpool.AgentPoolService { + cfg := config.NewApplicationConfig() + cfg.AgentPool = config.AgentPoolConfig{ + Enabled: true, + StateDir: stateDir, + APIURL: llmURL, + DefaultModel: "fake-model", + Timeout: "30s", + } + svc, err := agentpool.NewAgentPoolService(cfg) + Expect(err).ToNot(HaveOccurred()) + Expect(svc.Start(context.Background())).To(Succeed()) + return svc +} + +func newAgentConfig(name string) *state.AgentConfig { + return &state.AgentConfig{ + Name: name, + Model: "fake-model", + Description: "contract test agent", + SystemPrompt: "You are a test agent.", + } +} + +// Engine-specific: awaitRunning and collectSSE reach into LocalAGI types +// (agent.Agent via GetAgentForUser, types.NewJob, sse.Manager and +// sse.NewClient). P1 must re-seat them when LocalAGI types leave the service +// signatures; the specs that call them should not need to change. + +// awaitRunning blocks until the agent's Run loop is serving jobs. The pool +// starts Run in a goroutine and LocalAGI's Scheduler.Start and Scheduler.Stop +// are unsynchronized: a Stop (update, delete, svc.Stop) that lands while Start +// is still running can nil the scheduler context under the poll goroutine and +// crash the test binary. Run starts its workers only after Scheduler.Start has +// returned, and jobQueue is unbuffered, so Execute returning proves Start is +// done. The job's context is already cancelled, so the worker finishes it as +// expired without calling the LLM or recording an observable. This depends on +// LocalAGI not short-circuiting a cancelled job before a worker receives it, +// so re-check it when LocalAGI is bumped; the timeout turns a hang into a +// failure if that ever changes. +func awaitRunning(svc *agentpool.AgentPoolService, userID, name string) { + a := svc.GetAgentForUser(userID, name) + Expect(a).ToNot(BeNil()) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + done := make(chan struct{}) + go func() { + defer close(done) + a.Execute(types.NewJob(types.WithContext(ctx))) + }() + Eventually(done, "10s").Should(BeClosed()) +} + +type sseEvent struct { + Name string + Data map[string]any +} + +// collectSSE registers a listener on the agent's SSE manager and records every +// event until stop is called. Subscribe before Chat so nothing is missed. +// The manager replays its last 10 events on Register, so a spec that chats +// twice with a fresh collector sees the first turn's completed status too. +func collectSSE(svc *agentpool.AgentPoolService, userID, name string) (events func() []sseEvent, stop func()) { + mgr := svc.GetSSEManagerForUser(userID, name) + Expect(mgr).ToNot(BeNil()) + // Include the user: the manager keys listeners by ID, so two users' + // same-named agents must never share one if a manager is ever shared. + client := sse.NewClient("contract-" + userID + "-" + name) + mgr.Register(client) + + var mu sync.Mutex + var got []sseEvent + done := make(chan struct{}) + go func() { + for { + select { + case <-done: + return + case env, ok := <-client.Chan(): + if !ok { + return + } + ev := sseEvent{} + for _, line := range strings.Split(env.String(), "\n") { + switch { + case strings.HasPrefix(line, "event:"): + ev.Name = strings.TrimSpace(strings.TrimPrefix(line, "event:")) + case strings.HasPrefix(line, "data:"): + // The parse error is ignored because the hud event's + // data is not JSON; those events keep only their name. + _ = json.Unmarshal([]byte(strings.TrimSpace(strings.TrimPrefix(line, "data:"))), &ev.Data) + } + } + mu.Lock() + got = append(got, ev) + mu.Unlock() + } + } + }() + return func() []sseEvent { + mu.Lock() + defer mu.Unlock() + return append([]sseEvent(nil), got...) + }, func() { + close(done) + mgr.Unregister(client.ID()) + } +} diff --git a/core/services/agentpool/standalone_chat_contract_test.go b/core/services/agentpool/standalone_chat_contract_test.go new file mode 100644 index 000000000..9a2a452c3 --- /dev/null +++ b/core/services/agentpool/standalone_chat_contract_test.go @@ -0,0 +1,157 @@ +package agentpool_test + +import ( + "net/http" + + "github.com/mudler/LocalAI/core/services/agentpool" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("standalone chat contract", func() { + var ( + llm *fakeLLM + svc *agentpool.AgentPoolService + ) + + BeforeEach(func() { + llm = newFakeLLM("pong") + // Ginkgo runs cleanups LIFO: registering Close first stops the pool + // before its LLM goes away. + DeferCleanup(llm.Close) + svc = startStandalone(GinkgoT().TempDir(), llm.URL()) + DeferCleanup(svc.Stop) + Expect(svc.CreateAgentForUser("alice", newAgentConfig("chatty"))).To(Succeed()) + awaitRunning(svc, "alice", "chatty") + }) + + It("streams the user message, processing, agent reply and completed status over SSE", func() { + events, stop := collectSSE(svc, "alice", "chatty") + defer stop() + + msgID, err := svc.ChatForUser("alice", "chatty", "ping") + Expect(err).ToNot(HaveOccurred()) + Expect(msgID).ToNot(BeEmpty()) + + Eventually(func() []sseEvent { return statusEvents(events(), "completed") }, "30s", "100ms"). + ShouldNot(BeEmpty(), "no completed status event") + + var user, agent, processing bool + for _, e := range events() { + switch { + case e.Name == "json_message" && e.Data["sender"] == "user": + Expect(e.Data["content"]).To(Equal("ping")) + user = true + case e.Name == "json_message" && e.Data["sender"] == "agent": + Expect(e.Data["content"]).To(ContainSubstring("pong")) + // Current standalone shape: the reply id is the id ChatForUser + // returned plus "-agent". The UI correlates on message_id + // (AgentChat.jsx), which the distributed dispatcher sends; a + // native engine may send either, so flip this deliberately. + Expect(e.Data).To(HaveKeyWithValue("id", msgID+"-agent")) + agent = true + case e.Name == "json_message_status" && e.Data["status"] == "processing": + processing = true + } + } + Expect(user).To(BeTrue(), "user json_message") + Expect(processing).To(BeTrue(), "processing status") + Expect(agent).To(BeTrue(), "agent json_message") + }) + + It("sends the user's message to the LLM under the configured model", func() { + _, err := svc.ChatForUser("alice", "chatty", "ping") + Expect(err).ToNot(HaveOccurred()) + + Eventually(llm.Requests, "30s", "100ms").ShouldNot(BeEmpty()) + req := llm.Requests()[0] + Expect(req.Model).To(Equal("fake-model")) + // Observed shape: content is a plain string, not a parts array. A + // switch to parts would change what an OpenAI-compatible backend sees. + Expect(req.Messages).To(ContainElement(And( + HaveKeyWithValue("role", "user"), + HaveKeyWithValue("content", "ping"), + ))) + }) + + // The chat page clears its "processing" state only on an agent + // json_message or a json_error, so a failed turn must end in json_error + // followed by completed, never in silence. The fake fails every request; + // cogito retries the decision 5 times with a linear 1s..5s backoff, so the + // turn settles after about 15s, hence the 60s budget. The error text is + // cogito's wrapped chain and is not pinned. + It("reports a failing LLM as json_error then completed, with no agent reply", func() { + llm.SetFailure(http.StatusInternalServerError) + events, stop := collectSSE(svc, "alice", "chatty") + defer stop() + + _, err := svc.ChatForUser("alice", "chatty", "ping") + Expect(err).ToNot(HaveOccurred()) + + Eventually(func() []sseEvent { return statusEvents(events(), "completed") }, "60s", "100ms"). + ShouldNot(BeEmpty(), "no completed status event") + + errorAt, completedAt := -1, -1 + for i, e := range events() { + switch { + case e.Name == "json_error" && errorAt < 0: + errorAt = i + Expect(e.Data).To(HaveKeyWithValue("error", And(BeAssignableToTypeOf(""), Not(BeEmpty())))) + case e.Name == "json_message_status" && e.Data["status"] == "completed" && completedAt < 0: + completedAt = i + case e.Name == "json_message" && e.Data["sender"] == "agent": + Fail("a failed turn must not produce an agent json_message") + } + } + Expect(errorAt).To(BeNumerically(">=", 0), "no json_error event") + Expect(errorAt).To(BeNumerically("<", completedAt), "json_error must precede completed") + Expect(llm.Requests()).ToNot(BeEmpty()) + }) + + It("reports chat with an unknown agent as ErrAgentNotFound", func() { + _, err := svc.ChatForUser("alice", "ghost", "hi") + Expect(err).To(MatchError(agentpool.ErrAgentNotFound)) + }) + + It("reports chat with a deleted agent as ErrAgentNotFound", func() { + Expect(svc.DeleteAgentForUser("alice", "chatty")).To(Succeed()) + _, err := svc.ChatForUser("alice", "chatty", "hi") + Expect(err).To(MatchError(agentpool.ErrAgentNotFound)) + }) + + It("does not deliver one user's chat events to another user's agent of the same name", func() { + Expect(svc.CreateAgentForUser("bob", newAgentConfig("chatty"))).To(Succeed()) + awaitRunning(svc, "bob", "chatty") + aliceEvents, stopAlice := collectSSE(svc, "alice", "chatty") + defer stopAlice() + bobEvents, stopBob := collectSSE(svc, "bob", "chatty") + defer stopBob() + + _, err := svc.ChatForUser("alice", "chatty", "ping") + Expect(err).ToNot(HaveOccurred()) + Eventually(func() []sseEvent { return statusEvents(aliceEvents(), "completed") }, "30s", "100ms").ShouldNot(BeEmpty()) + + // LocalAGI pushes a "hud" snapshot of each agent's own state every + // second, so bob's stream is not silent; what must never reach it is + // anything produced by alice's chat. + Consistently(func() []string { + var names []string + for _, e := range bobEvents() { + if e.Name != "hud" { + names = append(names, e.Name) + } + } + return names + }, "1s", "100ms").Should(BeEmpty()) + }) +}) + +func statusEvents(events []sseEvent, status string) []sseEvent { + var out []sseEvent + for _, e := range events { + if e.Name == "json_message_status" && e.Data["status"] == status { + out = append(out, e) + } + } + return out +} diff --git a/core/services/agentpool/standalone_contract_test.go b/core/services/agentpool/standalone_contract_test.go new file mode 100644 index 000000000..e88720656 --- /dev/null +++ b/core/services/agentpool/standalone_contract_test.go @@ -0,0 +1,186 @@ +package agentpool_test + +import ( + "encoding/json" + + "github.com/mudler/LocalAI/core/services/agentpool" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("standalone agent service contract", func() { + var ( + llm *fakeLLM + dir string + ) + + BeforeEach(func() { + llm = newFakeLLM("hello from the fake model") + // DeferCleanup runs after AfterEach and in LIFO order, so registering + // here lets a pool started later stop before its LLM goes away. + DeferCleanup(llm.Close) + dir = GinkgoT().TempDir() + }) + + It("boots against a fake LLM and lists no agents", func() { + svc := startStandalone(dir, llm.URL()) + DeferCleanup(svc.Stop) + + Expect(svc.ListAgentsForUser("")).To(BeEmpty()) + }) + + Context("agent CRUD", func() { + var svc *agentpool.AgentPoolService + BeforeEach(func() { + svc = startStandalone(dir, llm.URL()) + DeferCleanup(svc.Stop) + }) + + It("creates, reads back, updates and deletes an agent", func() { + Expect(svc.CreateAgentForUser("alice", newAgentConfig("helper"))).To(Succeed()) + awaitRunning(svc, "alice", "helper") + + got := svc.GetAgentConfigForUser("alice", "helper") + Expect(got).ToNot(BeNil()) + // The pool stores the key ("alice:helper"); the API must show the bare name. + Expect(got.Name).To(Equal("helper")) + Expect(got.Model).To(Equal("fake-model")) + Expect(svc.ListAgentsForUser("alice")).To(HaveKeyWithValue("helper", true)) + + updated := newAgentConfig("helper") + updated.Description = "changed" + Expect(svc.UpdateAgentForUser("alice", "helper", updated)).To(Succeed()) + // Update restarts the agent, so the new instance needs the same wait. + awaitRunning(svc, "alice", "helper") + Expect(svc.GetAgentConfigForUser("alice", "helper").Description).To(Equal("changed")) + + Expect(svc.DeleteAgentForUser("alice", "helper")).To(Succeed()) + Expect(svc.GetAgentConfigForUser("alice", "helper")).To(BeNil()) + Expect(svc.ListAgentsForUser("alice")).ToNot(HaveKey("helper")) + }) + + It("reports an update of a missing agent as ErrAgentNotFound", func() { + err := svc.UpdateAgentForUser("alice", "ghost", newAgentConfig("ghost")) + Expect(err).To(MatchError(agentpool.ErrAgentNotFound)) + }) + + It("keeps two users' agents with the same name apart", func() { + a := newAgentConfig("shared-name") + a.Description = "alice's" + b := newAgentConfig("shared-name") + b.Description = "bob's" + Expect(svc.CreateAgentForUser("alice", a)).To(Succeed()) + Expect(svc.CreateAgentForUser("bob", b)).To(Succeed()) + awaitRunning(svc, "alice", "shared-name") + awaitRunning(svc, "bob", "shared-name") + + Expect(svc.GetAgentConfigForUser("alice", "shared-name").Description).To(Equal("alice's")) + Expect(svc.GetAgentConfigForUser("bob", "shared-name").Description).To(Equal("bob's")) + + Expect(svc.DeleteAgentForUser("alice", "shared-name")).To(Succeed()) + Expect(svc.GetAgentConfigForUser("alice", "shared-name")).To(BeNil()) + Expect(svc.GetAgentConfigForUser("bob", "shared-name")).ToNot(BeNil()) + + grouped := svc.ListAllAgentsGrouped() + Expect(grouped).To(HaveKey("bob")) + Expect(grouped).ToNot(HaveKey("alice")) + }) + + It("round-trips a config through export and import without a user", func() { + cfg := newAgentConfig("portable") + cfg.Description = "carry me" + Expect(svc.CreateAgentForUser("", cfg)).To(Succeed()) + awaitRunning(svc, "", "portable") + + data, err := svc.ExportAgentForUser("", "portable") + Expect(err).ToNot(HaveOccurred()) + + Expect(svc.DeleteAgentForUser("", "portable")).To(Succeed()) + Expect(svc.ImportAgentForUser("", data)).To(Succeed()) + awaitRunning(svc, "", "portable") + + got := svc.GetAgentConfigForUser("", "portable") + Expect(got).ToNot(BeNil()) + Expect(got.Description).To(Equal("carry me")) + }) + + // Known defect pinned on purpose: export returns the stored config, whose + // name is the pool key, and import refuses ":" in names. A rewrite that + // fixes this must flip this spec rather than silently change behavior. + It("known defect: exports a user's agent under its pool key, which import then rejects", func() { + cfg := newAgentConfig("portable") + cfg.Description = "carry me" + Expect(svc.CreateAgentForUser("alice", cfg)).To(Succeed()) + awaitRunning(svc, "alice", "portable") + + data, err := svc.ExportAgentForUser("alice", "portable") + Expect(err).ToNot(HaveOccurred()) + var out map[string]any + Expect(json.Unmarshal(data, &out)).To(Succeed()) + Expect(out["name"]).To(Equal("alice:portable")) + + Expect(svc.DeleteAgentForUser("alice", "portable")).To(Succeed()) + Expect(svc.ImportAgentForUser("alice", data)).To(MatchError(ContainSubstring("invalid characters"))) + Expect(svc.GetAgentConfigForUser("alice", "portable")).To(BeNil()) + }) + + // Engine-specific: P2/P5 flips this because the native import drops + // unknown fields (connectors, actions) with a warning, so the export + // will no longer carry them. + // P5 strips connectors and actions and the P2 migration reads old configs, + // so record what a config that carries them looks like today. LocalAGI + // logs "Failed to create IRC client" for this fixture because the IRC + // config has no nickname; that is expected and is not a failure. + It("accepts and returns a config that carries connectors and actions", func() { + raw := []byte(`{ + "name": "legacy", + "model": "fake-model", + "description": "old style", + "connectors": [{"type": "irc", "config": "{}"}], + "actions": [{"name": "search", "config": "{}"}] + }`) + Expect(svc.ImportAgentForUser("alice", raw)).To(Succeed()) + awaitRunning(svc, "alice", "legacy") + + data, err := svc.ExportAgentForUser("alice", "legacy") + Expect(err).ToNot(HaveOccurred()) + var out map[string]any + Expect(json.Unmarshal(data, &out)).To(Succeed()) + Expect(out["connectors"]).To(HaveLen(1)) + Expect(out["actions"]).To(HaveLen(1)) + // The P2 migration reads these element shapes. The action name is + // what LocalAGI actually stores, observed as the name sent in, not a + // resolved alias. + Expect(out["connectors"]).To(ConsistOf(And( + HaveKeyWithValue("type", "irc"), + HaveKeyWithValue("config", BeAssignableToTypeOf("")), + ))) + Expect(out["actions"]).To(ConsistOf(And( + HaveKeyWithValue("name", "search"), + HaveKeyWithValue("config", BeAssignableToTypeOf("")), + ))) + }) + }) + + Context("pause and resume", func() { + It("toggles the active flag reported by the list", func() { + svc := startStandalone(dir, llm.URL()) + DeferCleanup(svc.Stop) + Expect(svc.CreateAgentForUser("alice", newAgentConfig("napper"))).To(Succeed()) + awaitRunning(svc, "alice", "napper") + Expect(svc.ListAgentsForUser("alice")).To(HaveKeyWithValue("napper", true)) + + Expect(svc.PauseAgentForUser("alice", "napper")).To(Succeed()) + Expect(svc.ListAgentsForUser("alice")).To(HaveKeyWithValue("napper", false)) + + Expect(svc.ResumeAgentForUser("alice", "napper")).To(Succeed()) + Expect(svc.ListAgentsForUser("alice")).To(HaveKeyWithValue("napper", true)) + }) + + It("reports pausing a missing agent as ErrAgentNotFound", func() { + svc := startStandalone(dir, llm.URL()) + DeferCleanup(svc.Stop) + Expect(svc.PauseAgentForUser("alice", "ghost")).To(MatchError(agentpool.ErrAgentNotFound)) + }) + }) +}) diff --git a/core/services/agentpool/standalone_persistence_contract_test.go b/core/services/agentpool/standalone_persistence_contract_test.go new file mode 100644 index 000000000..0fd782c27 --- /dev/null +++ b/core/services/agentpool/standalone_persistence_contract_test.go @@ -0,0 +1,101 @@ +package agentpool_test + +import ( + "encoding/json" + "os" + "path/filepath" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("standalone persistence contract", func() { + var ( + llm *fakeLLM + dir string + ) + + BeforeEach(func() { + llm = newFakeLLM("ok") + dir = GinkgoT().TempDir() + DeferCleanup(llm.Close) + }) + + // The P2 migration imports this file, so its layout is an interface. + It("writes pool.json keyed by userID:name with the agent config as value", func() { + svc := startStandalone(dir, llm.URL()) + Expect(svc.CreateAgentForUser("alice", newAgentConfig("keeper"))).To(Succeed()) + awaitRunning(svc, "alice", "keeper") + Expect(svc.CreateAgentForUser("", newAgentConfig("anon"))).To(Succeed()) + awaitRunning(svc, "", "anon") + svc.Stop() + + raw, err := os.ReadFile(filepath.Join(dir, "pool.json")) + Expect(err).ToNot(HaveOccurred()) + var pool map[string]map[string]any + Expect(json.Unmarshal(raw, &pool)).To(Succeed()) + Expect(pool).To(HaveKey("alice:keeper")) + Expect(pool).To(HaveKey("anon")) + Expect(pool["alice:keeper"]["model"]).To(Equal("fake-model")) + // The stored name repeats the key, prefix included: the P2 importer + // strips the prefix, so it depends on this. + Expect(pool["alice:keeper"]["name"]).To(Equal("alice:keeper")) + Expect(pool["anon"]["name"]).To(Equal("anon")) + }) + + It("restores agents after a restart on the same state dir", func() { + svc := startStandalone(dir, llm.URL()) + Expect(svc.CreateAgentForUser("alice", newAgentConfig("survivor"))).To(Succeed()) + awaitRunning(svc, "alice", "survivor") + svc.Stop() + + again := startStandalone(dir, llm.URL()) + DeferCleanup(again.Stop) + awaitRunning(again, "alice", "survivor") + Expect(again.GetAgentConfigForUser("alice", "survivor")).ToNot(BeNil()) + Expect(again.ListAgentsForUser("alice")).To(HaveKey("survivor")) + }) + + // Pause is only an in-memory flag on the running agent: pool.json has no + // status field and no per-agent file records pause, so a restart brings + // the agent back active. Pinned as a known gap for the native-store migration to + // close on purpose rather than by accident. + It("known gap: does not keep a paused agent paused across a restart", func() { + svc := startStandalone(dir, llm.URL()) + Expect(svc.CreateAgentForUser("alice", newAgentConfig("sleeper"))).To(Succeed()) + awaitRunning(svc, "alice", "sleeper") + Expect(svc.PauseAgentForUser("alice", "sleeper")).To(Succeed()) + Expect(svc.ListAgentsForUser("alice")).To(HaveKeyWithValue("sleeper", false)) + svc.Stop() + + again := startStandalone(dir, llm.URL()) + DeferCleanup(again.Stop) + awaitRunning(again, "alice", "sleeper") + Expect(again.ListAgentsForUser("alice")).To(HaveKeyWithValue("sleeper", true)) + }) + + // The /v1/responses interceptor decides "is this model an agent" with + // GetAgent(name) using the raw pool key, with no user prefix. + // Known gap, not a contract: any caller who sends model "alice:mine" runs + // alice's agent (the interceptor has no user check), while alice's own + // request for "mine" falls through; a later fix must not read as a break. + It("known gap: resolves an agent by its raw key for the responses interceptor", func() { + svc := startStandalone(dir, llm.URL()) + DeferCleanup(svc.Stop) + Expect(svc.CreateAgentForUser("", newAgentConfig("global-agent"))).To(Succeed()) + awaitRunning(svc, "", "global-agent") + Expect(svc.CreateAgentForUser("alice", newAgentConfig("mine"))).To(Succeed()) + awaitRunning(svc, "alice", "mine") + + Expect(svc.GetAgent("global-agent")).ToNot(BeNil()) + Expect(svc.GetAgent("alice:mine")).ToNot(BeNil()) + Expect(svc.GetAgent("mine")).To(BeNil(), "current gap: a user's own agent is not found by its bare name, only by its pool key") + Expect(svc.GetAgent("nope")).To(BeNil()) + }) + + It("exposes the state dir it was started with", func() { + svc := startStandalone(dir, llm.URL()) + DeferCleanup(svc.Stop) + Expect(svc.StateDir()).To(Equal(dir)) + }) +}) diff --git a/core/services/agentpool/standalone_status_contract_test.go b/core/services/agentpool/standalone_status_contract_test.go new file mode 100644 index 000000000..f109b32b2 --- /dev/null +++ b/core/services/agentpool/standalone_status_contract_test.go @@ -0,0 +1,232 @@ +package agentpool_test + +import ( + "encoding/json" + "fmt" + + "github.com/mudler/LocalAGI/core/state" + "github.com/mudler/LocalAGI/core/types" + + "github.com/mudler/LocalAI/core/services/agentpool" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("standalone status and observables contract", func() { + var ( + llm *fakeLLM + svc *agentpool.AgentPoolService + ) + + BeforeEach(func() { + llm = newFakeLLM("done") + // Cleanups run in reverse, so the pool stops before the fake LLM goes away. + DeferCleanup(llm.Close) + svc = startStandalone(GinkgoT().TempDir(), llm.URL()) + DeferCleanup(svc.Stop) + Expect(svc.CreateAgentForUser("alice", newAgentConfig("observed"))).To(Succeed()) + awaitRunning(svc, "alice", "observed") + }) + + // runOnce chats once and waits for the completed status. ChatForUser sends + // that status as soon as Ask returns, before the job finalizers have written + // the finished observable, so callers that read observables use + // settledObservable instead. + runOnce := func() { + events, stop := collectSSE(svc, "alice", "observed") + defer stop() + _, err := svc.ChatForUser("alice", "observed", "go") + Expect(err).ToNot(HaveOccurred()) + Eventually(func() []sseEvent { return statusEvents(events(), "completed") }, "30s", "100ms"). + ShouldNot(BeEmpty(), "no completed status event") + } + + // settleRun chats once with the named agent and returns its observables + // and SSE events after the job finalizers are done. Three observer.Update + // calls follow Finish (the Execute finalizer, the consumeJob finalizer and + // the deferred MakeLastProgressCompletion update, LocalAGI agent.go + // 1182-1187), and Update re-appends an observable whose id is gone, so + // clearing while one is still pending would bring the observable back. Each Update also sends an observable_update event, so the run is + // settled once the root observable carries a completion, the completed + // status went out, and no further observable_update arrives. + settleRun := func(name string) ([]map[string]any, []sseEvent) { + events, stop := collectSSE(svc, "alice", name) + defer stop() + _, err := svc.ChatForUser("alice", name, "go") + Expect(err).ToNot(HaveOccurred()) + + Eventually(func() []sseEvent { return statusEvents(events(), "completed") }, "30s", "100ms"). + ShouldNot(BeEmpty(), "no completed status event") + Eventually(func(g Gomega) { + raw, err := svc.GetAgentObservablesForUser("alice", name) + g.Expect(err).ToNot(HaveOccurred()) + rootDone := false + for _, r := range raw { + var o map[string]any + g.Expect(json.Unmarshal(r, &o)).To(Succeed()) + if _, child := o["parent_id"]; !child { + _, rootDone = o["completion"] + } + } + g.Expect(rootDone).To(BeTrue()) + }, "30s", "100ms").Should(Succeed()) + + updates := func() int { + n := 0 + for _, e := range events() { + if e.Name == "observable_update" { + n++ + } + } + return n + } + var last int + Eventually(func() bool { + n := updates() + stable := n == last + last = n + return stable + }, "10s", "300ms").Should(BeTrue()) + Consistently(updates, "300ms", "50ms").Should(Equal(last)) + + raw, err := svc.GetAgentObservablesForUser("alice", name) + Expect(err).ToNot(HaveOccurred()) + obs := make([]map[string]any, len(raw)) + for i, r := range raw { + Expect(json.Unmarshal(r, &obs[i])).To(Succeed()) + } + return obs, events() + } + + // settledObservable returns the first observable of a settled plain run. + settledObservable := func() map[string]any { + obs, _ := settleRun("observed") + Expect(obs).ToNot(BeEmpty()) + return obs[0] + } + + It("returns an empty observable list before any run", func() { + obs, err := svc.GetAgentObservablesForUser("alice", "observed") + Expect(err).ToNot(HaveOccurred()) + Expect(obs).To(BeEmpty()) + }) + + // Discovery: a plain-content reply (no tool call) is enough for LocalAGI + // to record a "job" observable, so the action fallback was not needed. + // parent_id is not pinned: it is omitempty and a root job has none. + It("records observables after a run with the fields the agent status UI reads", func() { + first := settledObservable() + Expect(first).To(HaveKey("id")) + Expect(first).To(HaveKey("creation")) + Expect(first).To(HaveKey("completion")) + }) + + It("clears observables", func() { + settledObservable() + + Expect(svc.ClearAgentObservablesForUser("alice", "observed")).To(Succeed()) + obs, err := svc.GetAgentObservablesForUser("alice", "observed") + Expect(err).ToNot(HaveOccurred()) + Expect(obs).To(BeEmpty()) + }) + + It("reports observables of a missing agent as ErrAgentNotFound", func() { + _, err := svc.GetAgentObservablesForUser("alice", "ghost") + Expect(err).To(MatchError(agentpool.ErrAgentNotFound)) + Expect(svc.ClearAgentObservablesForUser("alice", "ghost")).To(MatchError(agentpool.ErrAgentNotFound)) + }) + + // LocalAGI only creates a status entry when an action result is recorded, + // so an agent whose runs never called an action looks the same as a + // missing one. Pinned as current behavior, not as a desirable contract. + It("returns a nil status for an agent with no action results, as for a missing one", func() { + Expect(svc.GetAgentStatusForUser("alice", "observed")).To(BeNil()) + runOnce() + Consistently(func() any { return svc.GetAgentStatusForUser("alice", "observed") }, "500ms", "100ms").Should(BeNil()) + Expect(svc.GetAgentStatusForUser("alice", "ghost")).To(BeNil()) + }) + + // Engine-specific: P2 rewrites these three specs because they depend on + // the LocalAGI counter action and the native executor ignores Actions; + // P2 swaps in an MCP tool fixture. The status spec also reads + // types.ActionState, a LocalAGI type that P1 must re-seat when LocalAGI + // types leave the service signatures. + Context("after a run that calls a tool", func() { + BeforeEach(func() { + cfg := newAgentConfig("tooled") + // counter is pure and in-memory, so the action result is + // deterministic without any external service. + cfg.Actions = []state.ActionsConfig{{Name: "counter", Config: "{}"}} + Expect(svc.CreateAgentForUser("alice", cfg)).To(Succeed()) + awaitRunning(svc, "alice", "tooled") + llm.SetToolCall("counter", `{"name":"contract","adjustment":1}`) + }) + + It("records a status entry the status endpoint renders with action, params and result", func() { + settleRun("tooled") + + st := svc.GetAgentStatusForUser("alice", "tooled") + Expect(st).ToNot(BeNil()) + // Select by action name rather than position: the order of + // status entries is a LocalAGI detail, not part of the contract. + var h types.ActionState + Expect(st.Results()).To(ContainElement(Satisfy(func(s types.ActionState) bool { + return s.ActionCurrentState.Action != nil && + s.ActionCurrentState.Action.Definition().Name.String() == "counter" + }), &h)) + Expect(h.ActionCurrentState.Params).To(HaveKeyWithValue("name", "contract")) + Expect(h.Result).To(ContainSubstring("Created counter 'contract'")) + + // Same format string as GetAgentStatusEndpoint: this text is what + // the agent status page shows, so the params must render as JSON + // through ActionParams.String rather than as a Go map. + rendered := fmt.Sprintf("Reasoning: %s\nAction taken: %s\nParameters: %+v\nResult: %s", + h.Reasoning, h.ActionCurrentState.Action.Definition().Name.String(), h.ActionCurrentState.Params, h.Result) + Expect(rendered).To(ContainSubstring("Action taken: counter\n")) + Expect(rendered).To(ContainSubstring(`"name":"contract"`)) + Expect(rendered).To(ContainSubstring("Result: Created counter 'contract'")) + }) + + // The observables tree nests the action under the job that ran it. + // Only the link is pinned: ids, ordering and conversation contents are + // LocalAGI internals. + It("records a child action observable linked to the root job by parent_id", func() { + obs, _ := settleRun("tooled") + Expect(len(obs)).To(BeNumerically(">", 1)) + + var roots, children []map[string]any + for _, o := range obs { + if _, ok := o["parent_id"]; ok { + children = append(children, o) + } else { + roots = append(roots, o) + } + } + Expect(roots).To(HaveLen(1)) + Expect(children).ToNot(BeEmpty()) + for _, c := range children { + // Compared as decoded JSON, whatever type the ids have: the id + // type is a LocalAGI detail, the parent link is the contract. + Expect(c["parent_id"]).To(Equal(roots[0]["id"])) + } + // "action" is the LocalAGI name, visible in AgentStatus.jsx; a + // native engine may name it differently, so flip deliberately. + Expect(children).To(ContainElement(And( + HaveKeyWithValue("name", "action"), + HaveKeyWithValue("completion", HaveKeyWithValue("action_result", ContainSubstring("Created counter"))), + ))) + }) + + It("still delivers the agent's final reply over SSE", func() { + _, events := settleRun("tooled") + var replies []sseEvent + for _, e := range events { + if e.Name == "json_message" && e.Data["sender"] == "agent" { + replies = append(replies, e) + } + } + Expect(replies).ToNot(BeEmpty()) + Expect(replies[0].Data["content"]).To(ContainSubstring("done")) + }) + }) +}) From b540c3e1fd842bf3e5a5f7787e9a634ce2b2850a Mon Sep 17 00:00:00 2001 From: mudler-agent Date: Wed, 30 Sep 2026 20:22:49 +0200 Subject: [PATCH 31/33] chore(vllm-cpp): bump to 967883486 (ABI v30), add hf_overrides and Tev1 entries, fix vllm-cpp gallery installs (#12379) * chore(vllm-cpp): bump vllm.cpp to 967883486 (ABI v30) Moves the pin from c3bebc357 to 967883486. On top of the Nimble decision adapter and the Qwen3.5 vision-loader fix, this brings Tev1 on /v1/systemone and vllm_decide (opt-in through a "Tev1Model" architecture in config.json), a tokenizer/ subdirectory fallback so the Laya HF snapshot loads as downloaded, a stop-token fix, a logprobs fix under async scheduling and a pinned parakeet.cpp fetch for the diarization build. ABI v30 only adds the diarization and speaker-attributed ASR entry points; no existing struct or signature changed, so the purego mirrors keep their layout and only abiVersion moves to 30. Between 4479dc99f and 967883486 vllm.h changed only in a comment. v30 turns VLLM_CPP_WITH_DIARIZATION on by default. The fetch is pinned now, but ON still downloads parakeet.cpp at configure time and links a second ggml into libvllm for calls this backend never makes, so build with the option off: the symbols stay present as refusing stubs. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude Code:claude-sonnet-5-5 * feat(vllm-cpp): add the hf_overrides engine arg vLLM parity: engine_args.hf_overrides is a JSON object of top-level config.json keys merged over the model directory's own config.json. The main use is opting a published checkpoint into an engine adapter its config does not name, such as {"architectures": ["Tev1Model"]} on the Tev1 snapshots, which declare Qwen3_5ForConditionalGeneration. The C ABI has no override input and the engine reads config.json from the directory it is given, so Load builds a private overlay directory: the merged config.json plus a symlink to every other entry of the model directory, and passes that to the engine. The download is never written. Free, a failed load and the next Load remove the overlay. validModelPath and the DFlash draft resolution still see the real directory. A value that is not a JSON object, a .gguf model or a directory without config.json fails the load instead of being skipped like an unknown engine_args key, because loading the unmodified config would serve a different architecture than the one configured. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude Code:claude-sonnet-5-5 * fix(gallery): nest vllm-cpp artifacts under overrides artifacts: is a model-config key, and the installer reads model-config keys only from overrides:. Five vllm-cpp entries (laya, gliner25-decide, qwen3-vl-4b, cua-s1-forms and gliner2.5) declared it at the entry top level, where it is silently dropped: the install reports success, writes a config whose model is the bare HF repo id and downloads nothing, and vllm-cpp (which does not infer artifacts) then fails the first load with "model path not found". Move each block under overrides:, and add a guard test that refuses a top-level artifacts: key in the index. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude Code:claude-sonnet-5-5 * feat(gallery): add Tev1 4B and 0.8B on vllm-cpp Two decisions entries for Together AI's Tev1 checkpoints, pinned to the current HF revisions. Tev1 is autoregressive: vllm.cpp answers /v1/systemone by scoring the option letters, and the same engine still serves chat completions. The published config.json names Qwen3_5ForConditionalGeneration, so each entry sets hf_overrides: {architectures: [Tev1Model]} to enable the decision route without editing the download. known_usecases is [decisions] only, since a declared decisions list is authoritative for reservation. The descriptions state what was checked: agreement with transformers on CPU over seven questions (4B 7/7, max probability difference 0.0004; 0.8B 6/7 with one near tie), CPU-only for the decision route, and a fine-tune license the model card says is still being finalized, so no license key is set. The Decisions API page lists both entries, drops the note that Tev1 does not serve /v1/systemone and documents the 24-option limit (Ollama allows 26). Signed-off-by: Ettore Di Giacinto Assisted-by: Claude Code:claude-sonnet-5-5 --------- Signed-off-by: Ettore Di Giacinto Co-authored-by: Ettore Di Giacinto --- backend/go/vllm-cpp/Makefile | 10 +- backend/go/vllm-cpp/README.md | 29 +++- backend/go/vllm-cpp/backend.go | 35 ++++- backend/go/vllm-cpp/govllmcpp.go | 9 +- backend/go/vllm-cpp/hfoverrides.go | 124 ++++++++++++++++ backend/go/vllm-cpp/hfoverrides_test.go | 138 ++++++++++++++++++ backend/go/vllm-cpp/options.go | 10 ++ backend/go/vllm-cpp/vllmcpp_test.go | 4 +- core/gallery/vllm_cpp_tags_test.go | 25 ++++ docs/content/features/decisions.md | 17 ++- docs/content/features/text-generation.md | 1 + docs/content/features/vllm-cpp.md | 33 +++++ gallery/index.yaml | 175 ++++++++++++++++++----- 13 files changed, 565 insertions(+), 45 deletions(-) create mode 100644 backend/go/vllm-cpp/hfoverrides.go create mode 100644 backend/go/vllm-cpp/hfoverrides_test.go diff --git a/backend/go/vllm-cpp/Makefile b/backend/go/vllm-cpp/Makefile index 56a32dbaf..32733f11d 100644 --- a/backend/go/vllm-cpp/Makefile +++ b/backend/go/vllm-cpp/Makefile @@ -11,7 +11,7 @@ JOBS?=$(shell nproc --ignore=1 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || e # vllm.cpp version VLLM_CPP_REPO?=https://github.com/mudler/vllm.cpp -VLLM_CPP_VERSION?=c3bebc357385990f721af66a3a6c69328dd4fc6c +VLLM_CPP_VERSION?=96788348627b6a079fcc3ef7fc6676a970965d7b # MLX GEMM provider (darwin/metal only; see the metal branch below for why). # Consumed as the prebuilt pip wheel: building MLX from source needs `xcrun @@ -40,6 +40,14 @@ MLX_ROOT=$(shell echo $(MLX_VENV)/lib/python*/site-packages/mlx) # server, examples and tests of the engine are never built here. CMAKE_ARGS+=-DVLLM_CPP_SERVER=OFF -DVLLM_CPP_BUILD_TESTS=OFF -DVLLM_CPP_BUILD_EXAMPLES=OFF CMAKE_ARGS+=-DCMAKE_BUILD_TYPE=Release +# Diarization (ABI v30) is ON upstream and FetchContents parakeet.cpp at +# configure time, building a second (static) ggml into libvllm. The pinned +# engine now pins that fetch to a commit, so ON configures, but it still adds a +# network fetch and a second ggml to every build for calls this backend never +# makes: it binds none of the vllm_diariz*/vllm_transcribe_and_* entry points, +# and with the option OFF they compile as stubs that refuse by name, so the ABI +# stays v30-complete without the extra dependency. +CMAKE_ARGS+=-DVLLM_CPP_WITH_DIARIZATION=OFF # vllm.cpp sets no global -march: SIMD tiers are per-file with runtime dispatch, # so ONE portable library serves every CPU of the target arch (unlike the diff --git a/backend/go/vllm-cpp/README.md b/backend/go/vllm-cpp/README.md index 49804ffb4..11427f926 100644 --- a/backend/go/vllm-cpp/README.md +++ b/backend/go/vllm-cpp/README.md @@ -9,7 +9,7 @@ It serves two things: text generation, and MiniMax-H3 joint video+audio generation. The backend dlopens the engine's stable C ABI (`libvllm`, `include/vllm.h`, -ABI v20) through purego: +ABI v30) through purego: - `Load` -> `vllm_engine_load`: accepts a `.gguf` file or a HF-style model directory (`config.json` + safetensors). `context_size` maps to @@ -44,6 +44,14 @@ the Makefile therefore means updating `abiVersion` plus the mirrors (and their offsets in `vllmcpp_test.go`) in the same change; `make abi-check` compares the pinned header against the bindings and the library build runs it first. +The Makefile builds libvllm with `-DVLLM_CPP_WITH_DIARIZATION=OFF`. vllm.cpp +turns that option ON by default since ABI v30, and ON fetches a pinned +parakeet.cpp (with its own ggml) at configure time. This backend does not bind +the diarization entry points, so OFF adds no dependency and changes nothing it +serves: those calls exist in libvllm but refuse with "not compiled in". If a +future change binds them, pin the parakeet.cpp source with +`-DVLLM_CPP_PARAKEET_CPP_DIR` and make `package.sh` bundle what it links. + Model config example: ```yaml @@ -56,6 +64,25 @@ options: - max_num_seqs:16 ``` +## hf_overrides + +`engine_args.hf_overrides` (vLLM parity) is a JSON object of top-level +`config.json` keys merged over the model directory's `config.json`. The C ABI +has no override input and the engine reads `config.json` from the directory it +is given, so `Load` builds an overlay (`hfoverrides.go`): a temp dir with the +merged `config.json` plus a symlink to every other entry of the model dir, and +passes the overlay as `model_path`. `validModelPath` and the DFlash draft +resolution still run against the real model dir. `Free` (and a failed load, or +the next `Load`) removes the overlay. A value that is not an object, a `.gguf` +model, or a dir without `config.json` fails the load instead of being ignored, +because loading the unmodified config would serve another architecture. + +```yaml +engine_args: + hf_overrides: + architectures: ["Tev1Model"] # opt a Qwen3.5-declared Tev1 snapshot into the Tev1 adapter +``` + ## MiniMax-H3 video+audio generation `GenerateVideo` -> `vllm_video_generate` (ABI v12). H3 renders picture and sound diff --git a/backend/go/vllm-cpp/backend.go b/backend/go/vllm-cpp/backend.go index 40e06fe8c..a2796b105 100644 --- a/backend/go/vllm-cpp/backend.go +++ b/backend/go/vllm-cpp/backend.go @@ -37,6 +37,10 @@ type VllmCpp struct { // other's checkpoints. Exactly one of the two is ever non-zero. videoEngine uintptr opts loadOptions + // overlayDir is the hf_overrides overlay handed to the engine in place of + // the model directory. It must outlive the engine handle (the engine may + // reopen files through it), so it is removed in Free, not after Load. + overlayDir string } // Stream registry: the per-request bridge between the C token callback and @@ -141,6 +145,23 @@ func (v *VllmCpp) Load(opts *pb.ModelOptions) error { } v.opts.speculativeConfig = resolvedSpec + // A reload reuses this struct: drop any overlay from the previous model + // before building a new one so it is not leaked. + if err := removeConfigOverlay(v.overlayDir); err != nil { + xlog.Warn("[vllm-cpp] stale overlay", "error", err) + } + v.overlayDir = "" + enginePath := model + if hasHFOverrides(v.opts.hfOverrides) { + overlay, err := newConfigOverlay(model, v.opts.hfOverrides) + if err != nil { + return err + } + v.overlayDir = overlay + enginePath = overlay + xlog.Info("[vllm-cpp] hf_overrides applied through overlay", "model", model, "overlay", overlay, "overrides", v.opts.hfOverrides) + } + mp := defaultModelParams() if v.opts.blockSize > 0 { mp.BlockSize = v.opts.blockSize @@ -173,7 +194,7 @@ func (v *VllmCpp) Load(opts *pb.ModelOptions) error { // Every string below is borrowed by C for the duration of the load call // only (the library copies what it keeps), so the backing slices just have // to outlive vllmEngineLoad - hence the single KeepAlive after it. - modelC := cString(model) + modelC := cString(enginePath) mp.ModelPath = uintptr(unsafe.Pointer(&modelC[0])) // #nosec G103 -- borrowed by C for the load call only keep := [][]byte{modelC} setStr := func(dst *uintptr, s string) { @@ -205,7 +226,12 @@ func (v *VllmCpp) Load(opts *pb.ModelOptions) error { rc := vllmEngineLoad(unsafe.Pointer(&mp), unsafe.Pointer(&engine)) // #nosec G103 -- POD out-params runtime.KeepAlive(keep) if rc != vllmOK { - return fmt.Errorf("vllm-cpp: engine load failed: %s", vllmLastError()) + loadErr := fmt.Errorf("vllm-cpp: engine load failed: %s", vllmLastError()) + if err := removeConfigOverlay(v.overlayDir); err != nil { + xlog.Warn("[vllm-cpp] overlay cleanup after failed load", "error", err) + } + v.overlayDir = "" + return loadErr } v.engine = engine return nil @@ -220,7 +246,10 @@ func (v *VllmCpp) Free() error { vllmVideoEngineFree(v.videoEngine) v.videoEngine = 0 } - return nil + // After the engine is gone, so nothing still reads through the links. + err := removeConfigOverlay(v.overlayDir) + v.overlayDir = "" + return err } // samplingFromPredict lowers PredictOptions into the C sampling POD plus the diff --git a/backend/go/vllm-cpp/govllmcpp.go b/backend/go/vllm-cpp/govllmcpp.go index f3723d096..71794ff31 100644 --- a/backend/go/vllm-cpp/govllmcpp.go +++ b/backend/go/vllm-cpp/govllmcpp.go @@ -1,6 +1,6 @@ package main -// purego bindings for the vllm.cpp stable C ABI (include/vllm.h, ABI v27). +// purego bindings for the vllm.cpp stable C ABI (include/vllm.h, ABI v30). // // The structs below are hand-mirrored PODs of the C declarations, with // explicit padding so the Go layout matches the C layout on linux/darwin @@ -21,7 +21,12 @@ import ( // the header of the VLLM_CPP_VERSION pinned in the Makefile: the build checks // the two against each other, because a mismatch is only caught at runtime by // registerLib, where it takes the backend down on every load (issue #11379). -const abiVersion = 29 +// +// v30 only ADDED the diarization and speaker-attributed-ASR entry points; every +// struct mirrored here is byte-identical to v29. They are not bound because the +// Makefile builds libvllm with VLLM_CPP_WITH_DIARIZATION=OFF, where they are +// stubs that refuse every call. +const abiVersion = 30 // The ABI's tri-state toggles (enable_prefix_caching ABI v7, // enable_jump_forward ABI v10) share one encoding: 0 is NOT "off", it is diff --git a/backend/go/vllm-cpp/hfoverrides.go b/backend/go/vllm-cpp/hfoverrides.go new file mode 100644 index 000000000..2d7e3abdf --- /dev/null +++ b/backend/go/vllm-cpp/hfoverrides.go @@ -0,0 +1,124 @@ +package main + +// hf_overrides, vLLM parity: a JSON object of config.json keys laid over the +// model directory's own config.json at load time. +// +// The engine reads config.json straight from the model directory and has no +// override input on the C ABI, so the only way to change what it sees without +// editing the snapshot is to hand it a different directory. The overlay built +// here is that directory: a private temp dir holding the merged config.json +// and a symlink for every other entry of the original. The snapshot itself is +// never written, which matters because it is a content-addressed download that +// a gallery reinstall or a hash check would otherwise flag or overwrite. +// +// The canonical use is opting a published checkpoint into an engine adapter +// its config does not name, e.g. {"architectures": ["Tev1Model"]} on a Tev1 +// snapshot that declares Qwen3_5ForConditionalGeneration. + +import ( + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "strings" +) + +// newConfigOverlay builds the overlay directory for modelDir with overrides +// merged over its config.json and returns its path. Keys are merged at the top +// level only: an override replaces the whole value of its key, nested objects +// included, which is what vLLM does for a plain (non sub-config) key. +// +// Bad input is refused rather than skipped, unlike an unknown engine_args key: +// hf_overrides exists to change which architecture loads, so silently loading +// the unmodified config would serve a different model than the one configured. +func newConfigOverlay(modelDir, overrides string) (dir string, err error) { + var patch map[string]any + if err := json.Unmarshal([]byte(overrides), &patch); err != nil || patch == nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides must be a JSON object of config.json keys, got %q", overrides) + } + + info, err := os.Stat(modelDir) + if err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: %w", err) + } + if !info.IsDir() { + return "", fmt.Errorf("vllm-cpp: hf_overrides needs a model directory with a config.json, %q is a file", modelDir) + } + absDir, err := filepath.Abs(modelDir) + if err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: %w", err) + } + + raw, err := os.ReadFile(filepath.Join(absDir, "config.json")) + if err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides needs %s: %w", filepath.Join(absDir, "config.json"), err) + } + var config map[string]any + if err := json.Unmarshal(raw, &config); err != nil || config == nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: %s is not a JSON object", filepath.Join(absDir, "config.json")) + } + for k, v := range patch { + config[k] = v + } + merged, err := json.MarshalIndent(config, "", " ") + if err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: encoding the merged config.json: %w", err) + } + + entries, err := os.ReadDir(absDir) + if err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: %w", err) + } + + dir, err = os.MkdirTemp("", "vllm-cpp-hf-overrides-*") + if err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: creating the overlay: %w", err) + } + defer func() { + if err != nil { + _ = os.RemoveAll(dir) + dir = "" + } + }() + + // Links point at the entry path, not at what it resolves to: an HF cache + // snapshot is itself a tree of links into blobs/, and the engine already + // follows those. + for _, e := range entries { + if e.Name() == "config.json" { + continue + } + if err := os.Symlink(filepath.Join(absDir, e.Name()), filepath.Join(dir, e.Name())); err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: linking %s into the overlay: %w", e.Name(), err) + } + } + if err := os.WriteFile(filepath.Join(dir, "config.json"), merged, 0o600); err != nil { + return "", fmt.Errorf("vllm-cpp: hf_overrides: writing the merged config.json: %w", err) + } + return dir, nil +} + +// removeConfigOverlay deletes an overlay built by newConfigOverlay. RemoveAll +// removes the symlinks themselves and never descends into their targets, so +// the original model directory is safe. +func removeConfigOverlay(dir string) error { + if dir == "" { + return nil + } + if err := os.RemoveAll(dir); err != nil && !errors.Is(err, os.ErrNotExist) { + return fmt.Errorf("vllm-cpp: removing the hf_overrides overlay %s: %w", dir, err) + } + return nil +} + +// hasHFOverrides reports whether an overlay is needed. An empty object is a +// no-op in vLLM too, so it loads the directory directly instead of paying for +// an overlay that changes nothing. +func hasHFOverrides(overrides string) bool { + switch strings.TrimSpace(overrides) { + case "", "{}", "null": + return false + } + return true +} diff --git a/backend/go/vllm-cpp/hfoverrides_test.go b/backend/go/vllm-cpp/hfoverrides_test.go new file mode 100644 index 000000000..cf3159f84 --- /dev/null +++ b/backend/go/vllm-cpp/hfoverrides_test.go @@ -0,0 +1,138 @@ +package main + +import ( + "encoding/json" + "os" + "path/filepath" + + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + + pb "github.com/mudler/LocalAI/pkg/grpc/proto" +) + +var _ = Describe("hf_overrides", func() { + var modelDir string + const originalConfig = `{"architectures":["Qwen3_5ForConditionalGeneration"],"hidden_size":2560,"text_config":{"num_hidden_layers":36}}` + + BeforeEach(func() { + modelDir = GinkgoT().TempDir() + Expect(os.WriteFile(filepath.Join(modelDir, "config.json"), []byte(originalConfig), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelDir, "model.safetensors"), []byte("weights"), 0o644)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelDir, "tokenizer.json"), []byte("{}"), 0o644)).To(Succeed()) + Expect(os.MkdirAll(filepath.Join(modelDir, "tokenizer"), 0o755)).To(Succeed()) + Expect(os.WriteFile(filepath.Join(modelDir, "tokenizer", "vocab.json"), []byte("{}"), 0o644)).To(Succeed()) + }) + + Describe("parsing", func() { + It("reads a YAML-nested object from engine_args as a JSON document", func() { + lo := parseOptions(&pb.ModelOptions{ + EngineArgs: `{"hf_overrides":{"architectures":["Tev1Model"]},"max_num_seqs":2}`, + }) + Expect(lo.hfOverrides).To(MatchJSON(`{"architectures":["Tev1Model"]}`)) + Expect(lo.maxNumSeqs).To(Equal(int32(2))) + }) + + It("accepts a pre-encoded JSON string", func() { + lo := parseOptions(&pb.ModelOptions{ + EngineArgs: `{"hf_overrides":"{\"architectures\":[\"Tev1Model\"]}"}`, + }) + Expect(lo.hfOverrides).To(MatchJSON(`{"architectures":["Tev1Model"]}`)) + }) + }) + + Describe("config overlay", func() { + It("writes the merged config.json and symlinks every other entry to the original", func() { + overlay, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"],"new_key":1}`) + Expect(err).ToNot(HaveOccurred()) + DeferCleanup(os.RemoveAll, overlay) + Expect(overlay).ToNot(Equal(modelDir)) + + raw, err := os.ReadFile(filepath.Join(overlay, "config.json")) + Expect(err).ToNot(HaveOccurred()) + Expect(raw).To(MatchJSON(`{"architectures":["Tev1Model"],"hidden_size":2560,"text_config":{"num_hidden_layers":36},"new_key":1}`)) + info, err := os.Lstat(filepath.Join(overlay, "config.json")) + Expect(err).ToNot(HaveOccurred()) + Expect(info.Mode() & os.ModeSymlink).To(BeZero()) + + for _, name := range []string{"model.safetensors", "tokenizer.json", "tokenizer"} { + target, err := os.Readlink(filepath.Join(overlay, name)) + Expect(err).ToNot(HaveOccurred(), name) + Expect(target).To(Equal(filepath.Join(modelDir, name))) + } + // The subdir resolves through the link, so a tokenizer/ fallback + // in the engine still finds its files. + Expect(filepath.Join(overlay, "tokenizer", "vocab.json")).To(BeARegularFile()) + + entries, err := os.ReadDir(overlay) + Expect(err).ToNot(HaveOccurred()) + Expect(entries).To(HaveLen(4)) + }) + + It("leaves the original config.json untouched", func() { + overlay, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"]}`) + Expect(err).ToNot(HaveOccurred()) + DeferCleanup(os.RemoveAll, overlay) + + raw, err := os.ReadFile(filepath.Join(modelDir, "config.json")) + Expect(err).ToNot(HaveOccurred()) + Expect(string(raw)).To(Equal(originalConfig)) + }) + + It("is removed by Free", func() { + overlay, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"]}`) + Expect(err).ToNot(HaveOccurred()) + Expect(overlay).To(BeADirectory()) + + v := &VllmCpp{overlayDir: overlay} + Expect(v.Free()).To(Succeed()) + Expect(overlay).ToNot(BeAnExistingFile()) + Expect(v.overlayDir).To(BeEmpty()) + // The originals the links pointed at survive the cleanup. + Expect(filepath.Join(modelDir, "model.safetensors")).To(BeARegularFile()) + Expect(filepath.Join(modelDir, "tokenizer", "vocab.json")).To(BeARegularFile()) + }) + + DescribeTable("refuses bad input", + func(overrides string, useFile bool, substr string) { + target := modelDir + if useFile { + target = filepath.Join(modelDir, "model.safetensors") + } + overlay, err := newConfigOverlay(target, overrides) + Expect(err).To(MatchError(ContainSubstring(substr))) + Expect(overlay).To(BeEmpty()) + }, + Entry("a JSON array", `["Tev1Model"]`, false, "must be a JSON object"), + Entry("a JSON scalar", `5`, false, "must be a JSON object"), + Entry("unparseable JSON", `{"architectures":`, false, "must be a JSON object"), + Entry("a model that is not a directory", `{"architectures":["Tev1Model"]}`, true, "model directory"), + ) + + It("refuses a directory without config.json", func() { + Expect(os.Remove(filepath.Join(modelDir, "config.json"))).To(Succeed()) + _, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"]}`) + Expect(err).To(MatchError(ContainSubstring("config.json"))) + }) + + It("does not leave a half-built overlay behind on failure", func() { + Expect(os.WriteFile(filepath.Join(modelDir, "config.json"), []byte("not json"), 0o644)).To(Succeed()) + before, _ := filepath.Glob(filepath.Join(os.TempDir(), "vllm-cpp-hf-overrides-*")) + _, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"]}`) + Expect(err).To(HaveOccurred()) + after, _ := filepath.Glob(filepath.Join(os.TempDir(), "vllm-cpp-hf-overrides-*")) + Expect(after).To(ConsistOf(before)) + }) + }) + + It("keeps the merged document a valid object when overrides replace a nested key", func() { + overlay, err := newConfigOverlay(modelDir, `{"text_config":{"num_hidden_layers":2}}`) + Expect(err).ToNot(HaveOccurred()) + DeferCleanup(os.RemoveAll, overlay) + raw, err := os.ReadFile(filepath.Join(overlay, "config.json")) + Expect(err).ToNot(HaveOccurred()) + var doc map[string]any + Expect(json.Unmarshal(raw, &doc)).To(Succeed()) + Expect(doc["text_config"]).To(Equal(map[string]any{"num_hidden_layers": float64(2)})) + }) +}) diff --git a/backend/go/vllm-cpp/options.go b/backend/go/vllm-cpp/options.go index 1a36e62b3..0e95eb63a 100644 --- a/backend/go/vllm-cpp/options.go +++ b/backend/go/vllm-cpp/options.go @@ -74,6 +74,10 @@ type loadOptions struct { nerThreshold float32 // nerMaxWidth is the maximum span width in tokens (0 = engine default 12). nerMaxWidth int32 + // hf_overrides (vLLM parity): a JSON object of config.json keys merged + // over the model directory's config.json through a private overlay dir, + // see newConfigOverlay. Empty = load the directory as is. + hfOverrides string } // videoOptions is the MiniMax-H3 checkpoint SET plus its generation defaults. @@ -208,6 +212,8 @@ func applyOptionsList(lo *loadOptions, options []string) { lo.kvTransferConfig = strings.TrimSpace(v) case "tokenizer_config", "tokenizer_config_path": lo.tokenizerConfigPath = strings.TrimSpace(v) + case "hf_overrides": + lo.hfOverrides = strings.TrimSpace(v) case "enable_prefix_caching", "enable_radix_attention": if b, err := strconv.ParseBool(strings.TrimSpace(v)); err == nil { lo.enablePrefixCaching = boolTriState(b) @@ -343,6 +349,10 @@ func applyEngineArgs(lo *loadOptions, engineArgs string) { lo.speculativeConfig = jsonDocument(v, lo.speculativeConfig, k) case "kv_transfer_config": lo.kvTransferConfig = jsonDocument(v, lo.kvTransferConfig, k) + case "hf_overrides": + // Kept verbatim even when it is not an object: Load refuses a + // malformed value instead of loading the unmodified config. + lo.hfOverrides = jsonDocument(v, lo.hfOverrides, k) case "enable_prefix_caching", "enable_radix_attention": if b, ok := v.(bool); ok { lo.enablePrefixCaching = boolTriState(b) diff --git a/backend/go/vllm-cpp/vllmcpp_test.go b/backend/go/vllm-cpp/vllmcpp_test.go index 51ca81ace..1b6a3f6ea 100644 --- a/backend/go/vllm-cpp/vllmcpp_test.go +++ b/backend/go/vllm-cpp/vllmcpp_test.go @@ -16,7 +16,7 @@ func TestVllmCpp(t *testing.T) { RunSpecs(t, "vllm-cpp suite") } -// The Go POD mirrors must match the C struct layout of vllm.h (ABI v29) +// The Go POD mirrors must match the C struct layout of vllm.h (ABI v30) // byte-for-byte: these offsets are the C offsets on LP64 (linux/darwin // amd64+arm64). A failure here means govllmcpp.go drifted from vllm.h. var _ = Describe("C ABI struct mirrors", func() { @@ -24,7 +24,7 @@ var _ = Describe("C ABI struct mirrors", func() { // VLLM_ABI_VERSION in the vllm.h of VLLM_CPP_VERSION (Makefile). // Moving the pin past this without growing the mirrors below ships a // backend that refuses every load at startup (issue #11379). - Expect(abiVersion).To(Equal(29)) + Expect(abiVersion).To(Equal(30)) }) It("cModelParams matches vllm_model_params", func() { diff --git a/core/gallery/vllm_cpp_tags_test.go b/core/gallery/vllm_cpp_tags_test.go index b84f9a2b1..7c028d8a8 100644 --- a/core/gallery/vllm_cpp_tags_test.go +++ b/core/gallery/vllm_cpp_tags_test.go @@ -2,10 +2,13 @@ package gallery_test import ( "fmt" + "os" + "path/filepath" "slices" . "github.com/onsi/ginkgo/v2" . "github.com/onsi/gomega" + "gopkg.in/yaml.v3" "github.com/mudler/LocalAI/core/config" ) @@ -46,3 +49,25 @@ var _ = Describe("gallery/index.yaml vllm-cpp capability tags", func() { Expect(violations).To(BeEmpty()) }) }) + +// artifacts: is a model-config key, so the installer only sees it inside +// overrides:. At the top level of an entry it is silently dropped, the +// installed config keeps a bare HF repo id as its model, and a backend that +// does not infer artifacts (vllm-cpp among them) fails the first load with +// "model path not found" while the install itself reported success. +var _ = Describe("gallery/index.yaml artifacts placement", func() { + It("declares artifacts under overrides, never at the entry top level", func() { + data, err := os.ReadFile(filepath.Join("..", "..", "gallery", "index.yaml")) + Expect(err).ToNot(HaveOccurred()) + var raw []map[string]any + Expect(yaml.Unmarshal(data, &raw)).To(Succeed()) + + var misplaced []string + for _, e := range raw { + if _, ok := e["artifacts"]; ok { + misplaced = append(misplaced, fmt.Sprint(e["name"])) + } + } + Expect(misplaced).To(BeEmpty()) + }) +}) diff --git a/docs/content/features/decisions.md b/docs/content/features/decisions.md index 0118f110d..2893d47dc 100644 --- a/docs/content/features/decisions.md +++ b/docs/content/features/decisions.md @@ -105,13 +105,22 @@ Install one from the gallery and filter on the `decisions` tag: |---|---|---| | `laya-vllm-cpp` | Laya | ModernBERT-large, non-autoregressive, about 800 MB | | `gliner25-decide-vllm-cpp` | GLiNER2.5-Decide | DeBERTa-v3-large with a classification head, about 2 GB | +| `tev1-4b-vllm-cpp` | Tev1 4B | Autoregressive Qwen3.5-4B fine-tune that answers with an option letter, about 9.3 GB | +| `tev1-0.8b-vllm-cpp` | Tev1 0.8B | Autoregressive Qwen3.5-0.8B fine-tune that answers with an option letter, about 1.8 GB | The engine, [vllm.cpp]({{% relref "features/vllm-cpp" %}}), also supports the kev, CLM and xor decision models. Those checkpoints need a conversion step, so they are not gallery entries yet. -Tev1 is an autoregressive decision model. It answers through chat completions -and does not serve `/v1/systemone` yet. +Tev1 is an autoregressive decision model. The engine answers each question by +scoring the option letters, so its `confidence` is the entropy measure Ollama +uses. A Tev1 `choice` or `score` question accepts at most 24 options (Ollama +allows 26), because the model is trained on the letters A to X, and every +option needs a nonempty description. The published checkpoints name another +architecture in `config.json`, so the Tev1 gallery entries set +`engine_args.hf_overrides` to load them as `Tev1Model` (see +[Overriding config.json keys]({{% relref "features/vllm-cpp" %}}#overriding-configjson-keys-hf_overrides)). +The same model also answers `/v1/chat/completions` requests. ## Request limits @@ -127,8 +136,8 @@ A request is refused with `400` (or `413` for the body size) when: A `noul` question may carry `criteria` with a description for each outcome, for example `{"false": "No refund is requested", "true": "The customer requests a refund"}`. -Some models cap the number of options for a `choice` or `score` question (models -that answer with a letter accept at most 26). The engine refuses more options than +Some models cap the number of options for a `choice` or `score` question. Models +that answer with a letter accept at most 26, and Tev1 accepts at most 24. The engine refuses more options than the model supports and the error names the limit. ## Compatibility with Ollama diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md index 19dcf64b0..2b4fad656 100644 --- a/docs/content/features/text-generation.md +++ b/docs/content/features/text-generation.md @@ -1093,6 +1093,7 @@ engine_args: | `tokenizer_config` | Override the `tokenizer_config.json` the chat template is read from | `/tokenizer_config.json` | | `speculative_config` | Speculative decoding (see below) | disabled | | `kv_transfer_config` | External KV connector / LMCache (see below) | none | +| `hf_overrides` | JSON object of `config.json` keys merged over the model directory's own, as vLLM's `--hf-overrides` (see the [vllm.cpp backend page]({{% relref "features/vllm-cpp" %}}#overriding-configjson-keys-hf_overrides)) | none | Raising `max_num_batched_tokens` lets more prefill land in a single step, at the cost of decode latency for requests queued behind it. The default deliberately diff --git a/docs/content/features/vllm-cpp.md b/docs/content/features/vllm-cpp.md index bf7323e0d..1b7389f1f 100644 --- a/docs/content/features/vllm-cpp.md +++ b/docs/content/features/vllm-cpp.md @@ -133,6 +133,39 @@ engine_args: tool_parser: qwen3_coder ``` +## Overriding config.json keys (`hf_overrides`) + +`engine_args.hf_overrides` is a JSON object of top-level `config.json` keys that +the backend merges over the model directory's own `config.json` before the +engine loads it, like vLLM's `--hf-overrides`. The main use is to opt a published +checkpoint into an engine adapter that its config does not name. For example, +the Tev1 repositories declare `Qwen3_5ForConditionalGeneration`, and vllm.cpp +serves them as decision models only when the architecture is `Tev1Model`: + +```yaml +engine_args: + hf_overrides: + architectures: ["Tev1Model"] +``` + +The downloaded model files do not change. At load the backend creates a private +temporary directory. It writes the merged `config.json` there and adds a +symlink for each other entry of the model directory (weights, tokenizer files, +a `tokenizer/` subdirectory). Then it gives that directory to the engine. When +the model unloads, the backend removes the directory. + +Rules: + +- The merge is top-level only. An override key replaces the whole value of that + key, including a nested object such as `text_config`. +- The model must be a directory that contains a `config.json`. A `.gguf` file + or a directory without `config.json` fails the load. +- A value that is not a JSON object (an array, a scalar, or JSON that does not + parse) fails the load. The backend does not ignore it, because loading the + unchanged config would serve a different architecture than the one you + configured. +- An empty object (`{}`) does nothing. + ## Named entity recognition (GLiNER2.5) The `vllm-cpp` backend serves [GLiNER2.5](https://huggingface.co/fastino/gliner2.5-multi-v1), diff --git a/gallery/index.yaml b/gallery/index.yaml index 197b5f306..104585e01 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -63666,12 +63666,12 @@ - decisions parameters: model: convaiinnovations/laya - artifacts: - - name: model - target: model - source: - type: huggingface - repo: convaiinnovations/laya + artifacts: + - name: model + target: model + source: + type: huggingface + repo: convaiinnovations/laya - name: gliner25-decide-vllm-cpp url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: @@ -63703,13 +63703,124 @@ - decisions parameters: model: fastino/GLiNER2.5-Decide - artifacts: - - name: model - target: model - source: - type: huggingface - repo: fastino/GLiNER2.5-Decide - revision: 5a7adf72a23b4d311abae6ce050d7f0012bb3416 + artifacts: + - name: model + target: model + source: + type: huggingface + repo: fastino/GLiNER2.5-Decide + revision: 5a7adf72a23b4d311abae6ce050d7f0012bb3416 +- name: tev1-4b-vllm-cpp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/togethercomputer/Tev1-4B-experimental + - https://github.com/mudler/vllm.cpp + description: | + Tev1-4B-experimental is an experimental decision model from Together + AI: a supervised fine-tune of Qwen3.5-4B that picks one option letter + for a state, a question and 2 to 24 labeled options. It keeps the standard + next-token head, so it is autoregressive, unlike Laya or GLiNER2.5-Decide. + + In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine scores the + answer letters of each choice, noul and score question through the + vllm_decide C ABI and returns probabilities with an entropy confidence, as + Ollama does for tev1. A choice or score question accepts at most 24 options + (Ollama allows 26) and every option needs a nonempty description. The + published config.json names Qwen3_5ForConditionalGeneration, so this entry + sets hf_overrides to load it as Tev1Model without editing the download. + + Checked against transformers BF16 on CPU over seven + questions: 7/7 answers equal, largest probability difference 0.0004. The decision route is verified on CPU only; GPU serving has not + been measured. BF16 weights, about 9.3GB, pinned to a revision. The + fine-tune license is still being finalized by Together AI (base model + Apache-2.0). + tags: + - decisions + - systemone + - vllm-cpp + - cpu + - gpu + size: 9.3GB + last_checked: "2026-09-30" + overrides: + backend: vllm-cpp + known_usecases: + - decisions + template: + use_tokenizer_template: true + context_size: 2048 + engine_args: + hf_overrides: + architectures: + - Tev1Model + block_size: 32 + num_blocks: 256 + max_num_seqs: 4 + parameters: + model: togethercomputer/Tev1-4B-experimental + artifacts: + - name: model + target: model + source: + type: huggingface + repo: togethercomputer/Tev1-4B-experimental + revision: 0b7becf017daa0e5eb222f8ce7483c8c8259c52f +- name: tev1-0.8b-vllm-cpp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/togethercomputer/Tev1-0.8B-experimental + - https://github.com/mudler/vllm.cpp + description: | + Tev1-0.8B-experimental is an experimental decision model from Together + AI: a supervised fine-tune of Qwen3.5-0.8B that picks one option letter + for a state, a question and 2 to 24 labeled options. It keeps the standard + next-token head, so it is autoregressive, unlike Laya or GLiNER2.5-Decide. + + In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine scores the + answer letters of each choice, noul and score question through the + vllm_decide C ABI and returns probabilities with an entropy confidence, as + Ollama does for tev1. A choice or score question accepts at most 24 options + (Ollama allows 26) and every option needs a nonempty description. The + published config.json names Qwen3_5ForConditionalGeneration, so this entry + sets hf_overrides to load it as Tev1Model without editing the download. + + Checked against transformers BF16 on CPU over seven + questions: 6/7 answers equal, the miss a near tie (0.453 against 0.514 + in transformers, 0.4845 each here), largest probability difference 0.031. The decision route is verified on CPU only; GPU serving has not + been measured. BF16 weights, about 1.8GB, pinned to a revision. The + fine-tune license is still being finalized by Together AI (base model + Apache-2.0). + tags: + - decisions + - systemone + - vllm-cpp + - cpu + - gpu + size: 1.8GB + last_checked: "2026-09-30" + overrides: + backend: vllm-cpp + known_usecases: + - decisions + template: + use_tokenizer_template: true + context_size: 2048 + engine_args: + hf_overrides: + architectures: + - Tev1Model + block_size: 32 + num_blocks: 256 + max_num_seqs: 4 + parameters: + model: togethercomputer/Tev1-0.8B-experimental + artifacts: + - name: model + target: model + source: + type: huggingface + repo: togethercomputer/Tev1-0.8B-experimental + revision: 6bb2dff14b38fea90ddb14d870166ccaf77374e9 - name: qwen3-vl-4b-vllm-cpp url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: @@ -63749,13 +63860,13 @@ max_num_seqs: 4 parameters: model: Qwen/Qwen3-VL-4B-Instruct - artifacts: - - name: model - target: model - source: - type: huggingface - repo: Qwen/Qwen3-VL-4B-Instruct - revision: ebb281ec70b05090aa6165b016eac8ec08e71b17 + artifacts: + - name: model + target: model + source: + type: huggingface + repo: Qwen/Qwen3-VL-4B-Instruct + revision: ebb281ec70b05090aa6165b016eac8ec08e71b17 - name: cua-s1-forms-vllm-cpp url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: @@ -63784,12 +63895,12 @@ - score parameters: model: cua-ai/cua-s1-forms - artifacts: - - name: model - target: model - source: - type: huggingface - repo: cua-ai/cua-s1-forms + artifacts: + - name: model + target: model + source: + type: huggingface + repo: cua-ai/cua-s1-forms - name: gliner2.5-vllm-cpp url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: @@ -63821,12 +63932,12 @@ - token_classify parameters: model: fastino/gliner2.5-multi-v1 - artifacts: - - name: model - target: model - source: - type: huggingface - repo: fastino/gliner2.5-multi-v1 + artifacts: + - name: model + target: model + source: + type: huggingface + repo: fastino/gliner2.5-multi-v1 - name: nemo-speech-cpp-sortformer-diarization-v2 url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: From 246ca859d350abeeff141af2ab99f6a3ed1ca358 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Wed, 30 Sep 2026 18:31:38 +0000 Subject: [PATCH 32/33] fix(vllm-cpp): annotate the hf_overrides config.json read for gosec G304 flags reading a path built from a variable. The directory is the model directory from the operator's own model config, not a request input, so it is annotated the way the other backends do it. Assisted-by: Claude Code:claude-sonnet-5-5 Signed-off-by: Ettore Di Giacinto --- backend/go/vllm-cpp/hfoverrides.go | 1 + 1 file changed, 1 insertion(+) diff --git a/backend/go/vllm-cpp/hfoverrides.go b/backend/go/vllm-cpp/hfoverrides.go index 2d7e3abdf..57745c1e6 100644 --- a/backend/go/vllm-cpp/hfoverrides.go +++ b/backend/go/vllm-cpp/hfoverrides.go @@ -50,6 +50,7 @@ func newConfigOverlay(modelDir, overrides string) (dir string, err error) { return "", fmt.Errorf("vllm-cpp: hf_overrides: %w", err) } + // #nosec G304 -- absDir is the model directory from the operator's own model config, never a request-supplied path raw, err := os.ReadFile(filepath.Join(absDir, "config.json")) if err != nil { return "", fmt.Errorf("vllm-cpp: hf_overrides needs %s: %w", filepath.Join(absDir, "config.json"), err) From 975ff5ca42b5e2fd4d0966ea4470a6c3559f4439 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Wed, 30 Sep 2026 22:28:07 +0000 Subject: [PATCH 33/33] feat(gallery): add kev-0.8b-vllm-cpp decision model Add the kev 0.8B decision model, converted for vllm.cpp and pinned to revision c17e7366 of mudler/kev-0.8b-vllm-cpp. It is a redistribution of jaredpalmer/kev-0.8b with the LoRA merged and the PointerHead stored as head.safetensors, so only the vllm-cpp backend can load it. The artifact sits under overrides, where the installer reads it. The entry sets a 2048-token context and an explicit KV pool: with the default 4096-token context the CPU KV pool holds only 4064 tokens and the load fails. List the entry in the decisions gallery table and drop kev from the list of decision models that are not gallery entries yet. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude Code:claude-sonnet-5-5 --- docs/content/features/decisions.md | 6 ++-- gallery/index.yaml | 54 ++++++++++++++++++++++++++++++ 2 files changed, 58 insertions(+), 2 deletions(-) diff --git a/docs/content/features/decisions.md b/docs/content/features/decisions.md index 2893d47dc..2feb30c44 100644 --- a/docs/content/features/decisions.md +++ b/docs/content/features/decisions.md @@ -107,10 +107,12 @@ Install one from the gallery and filter on the `decisions` tag: | `gliner25-decide-vllm-cpp` | GLiNER2.5-Decide | DeBERTa-v3-large with a classification head, about 2 GB | | `tev1-4b-vllm-cpp` | Tev1 4B | Autoregressive Qwen3.5-4B fine-tune that answers with an option letter, about 9.3 GB | | `tev1-0.8b-vllm-cpp` | Tev1 0.8B | Autoregressive Qwen3.5-0.8B fine-tune that answers with an option letter, about 1.8 GB | +| `kev-0.8b-vllm-cpp` | kev 0.8B | Qwen3.5-0.8B-Base with a merged LoRA and a PointerHead readout, converted for vllm.cpp only, about 1.53 GB | The engine, [vllm.cpp]({{% relref "features/vllm-cpp" %}}), also supports the -kev, CLM and xor decision models. Those checkpoints need a conversion step, so -they are not gallery entries yet. +CLM and xor decision models. Those checkpoints need a conversion step, so +they are not gallery entries yet. The kev entry installs a checkpoint that was +already converted with the vllm.cpp `convert-kev.py` script. Tev1 is an autoregressive decision model. The engine answers each question by scoring the option letters, so its `confidence` is the entropy measure Ollama diff --git a/gallery/index.yaml b/gallery/index.yaml index 104585e01..eb37a7d12 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -63821,6 +63821,60 @@ type: huggingface repo: togethercomputer/Tev1-0.8B-experimental revision: 6bb2dff14b38fea90ddb14d870166ccaf77374e9 +- name: kev-0.8b-vllm-cpp + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/mudler/kev-0.8b-vllm-cpp + - https://huggingface.co/jaredpalmer/kev-0.8b + - https://github.com/mudler/vllm.cpp + description: | + kev is a System 1 decision model by Jared Palmer. It answers typed choice, + noul and score questions about a text state with one scoring pass per + question. It does not generate text. The model is a frozen + Qwen3.5-0.8B-Base backbone, a rank-16 LoRA adapter and a PointerHead + readout. + + This entry installs a converted redistribution of jaredpalmer/kev-0.8b: + the LoRA is merged into the BF16 backbone, the head is stored as + head.safetensors, and config.json names the KevModel architecture. The + checkpoint only works with vllm.cpp (the vllm-cpp backend); transformers, + vLLM and llama.cpp cannot load it. + + In LocalAI, serve it via POST /v1/systemone. The vllm.cpp project records + PointerHead golden-vector tests (25 cases) and a 5-case end-to-end + comparison against the kev reference server as equal. The upload itself + was smoke-tested with one request on CPU; there is no accuracy benchmark + and no GPU run. The entry sets a 2048-token context and an explicit KV + pool, because the default 4096-token context does not fit the default + CPU KV pool and the load fails. BF16 weights, about 1.53 GB, pinned to a + revision. + license: apache-2.0 + tags: + - decisions + - systemone + - vllm-cpp + - cpu + - gpu + size: 1.53GB + last_checked: "2026-09-30" + overrides: + backend: vllm-cpp + known_usecases: + - decisions + context_size: 2048 + engine_args: + block_size: 32 + num_blocks: 256 + max_num_seqs: 4 + parameters: + model: mudler/kev-0.8b-vllm-cpp + artifacts: + - name: model + target: model + source: + type: huggingface + repo: mudler/kev-0.8b-vllm-cpp + revision: c17e73666ded1e9d284470eae7e0de9a27294e77 - name: qwen3-vl-4b-vllm-cpp url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: