chore: merge master into distributed transport PR

Bring the distributed branch onto current master before the CI fix.

Assisted-by: Codex:gpt-6
This commit is contained in:
localai-org-maint-bot committed 2026-10-01 03:12:40 +00:00
commit 4bc0e92a3d
102 files changed
+7143 -205

No files matched your search

+17
View File
@@ -641,12 +641,29 @@ message TranscriptLiveResponse {
repeated TranscriptWord words = 4; // words finalized by this feed (stream-relative ns)
TranscriptResult final_result = 5; // terminal message only, after the send side closes
bool eob = 6; // <EOB> fired: a backchannel ("uh-huh") ended — NOT a turn boundary
repeated LiveSpeakerSegment speakers = 7; // closed speaker segments from a companion diarization/scene stream
repeated LiveSoundEvent sounds = 8; // closed sound events from a companion sound/scene stream
}
message LiveSpeakerSegment {
string speaker = 1; // decimal speaker index
int64 start = 2; // stream-relative nanoseconds
int64 end = 3;
}
message LiveSoundEvent {
string label = 1;
int32 index = 2;
float peak = 3;
int64 start = 4; // stream-relative nanoseconds
int64 end = 5;
}
message TranscriptWord {
int64 start = 1;
int64 end = 2;
string text = 3;
string speaker = 4; // backend speaker label when diarizing; empty otherwise
}
message TranscriptSegment {
+1 -1
View File
@@ -9,7 +9,7 @@
# recipe is a make target (not a prepare.sh) so 'make purge && make' is a clean
# rebuild and so the bump bot can see the pin.
AUDIO_CPP_VERSION?=77491a33c589c53ff18add050095cf35647c8213
AUDIO_CPP_VERSION?=ed96b7307c8daba2ebcf7912af928825f6b14cb9
AUDIO_CPP_REPO?=https://github.com/0xShug0/audio.cpp
CURRENT_MAKEFILE_DIR := $(dir $(abspath $(lastword $(MAKEFILE_LIST))))
+1 -1
View File
@@ -1,5 +1,5 @@
IK_LLAMA_VERSION?=ed27bf7ed25e637692e89cd341d802522a2cee8a
IK_LLAMA_VERSION?=0821d62a8b356bd1db3c6765551a30bfcc44a6de
LLAMA_REPO?=https://github.com/ikawrakow/ik_llama.cpp
CMAKE_ARGS?=
+1 -1
View File
@@ -1,5 +1,5 @@
LLAMA_VERSION?=95887577ab5fead779581a7030a83c7752ff3234
LLAMA_VERSION?=4da6337767f973e2b4d0797e5b323d77d8565e4a
LLAMA_REPO?=https://github.com/ggerganov/llama.cpp
CMAKE_ARGS?=
+1 -1
View File
@@ -1,7 +1,7 @@
# Pinned to the HEAD of feature/turboquant-kv-cache on https://github.com/TheTom/llama-cpp-turboquant.
# Auto-bumped nightly by .github/workflows/bump_deps.yaml.
TURBOQUANT_VERSION?=a3d5603d110bda29222d2011596cdc84d7fa532d
TURBOQUANT_VERSION?=bcb85fc3ae85efa0f5f392c6c880dfc524923860
LLAMA_REPO?=https://github.com/TheTom/llama-cpp-turboquant
CMAKE_ARGS?=
+2 -2
View File
@@ -1,6 +1,6 @@
# ced sound-classification backend Makefile.
#
# Upstream pin lives below as CED_VERSION?=db5aae02973a745722d6fbd2157cab1999106777
# Upstream pin lives below as CED_VERSION?=61dec2ab0106f2047ee40062a7075dbf08c523d0
# and update it (matches the parakeet-cpp / whisper.cpp convention).
#
# Local dev shortcut: symlink an out-of-tree ced.cpp shared build + header and
@@ -9,7 +9,7 @@
# ln -sf /path/to/ced.cpp/include/ced_capi.h .
# go build -o ced-grpc .
CED_VERSION?=db5aae02973a745722d6fbd2157cab1999106777
CED_VERSION?=61dec2ab0106f2047ee40062a7075dbf08c523d0
CED_REPO?=https://github.com/localai-org/ced.cpp
GOCMD?=go
+1 -1
View File
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
# CrispASR version (release tag)
CRISPASR_REPO?=https://github.com/CrispStrobe/CrispASR
CRISPASR_VERSION?=ec98831d0776ec8a16ccaf93955693eb7ecfbec3
CRISPASR_VERSION?=be202c472503a5c7f1d3e568c420865cad02f1c3
SO_TARGET?=libgocrispasr.so
CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
+1 -1
View File
@@ -12,7 +12,7 @@
# runs 'make -C backend/go/$(BACKEND) build' and then copies package/), so it
# has to produce the binary and the package, not just the shared libraries.
NEMO_SPEECH_VERSION?=97a15afa5caa9bce5baaa86c1184103877af4101
NEMO_SPEECH_VERSION?=0f706e43cf1fbc031bad1423e05460d3acaeaa1c
NEMO_SPEECH_REPO?=https://github.com/NVIDIA/NeMo-Speech.cpp
GOCMD?=go
+1 -1
View File
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
# omnivoice.cpp version
OMNIVOICE_REPO?=https://github.com/ServeurpersoCom/omnivoice.cpp
OMNIVOICE_VERSION?=ead199a2bc4c53a57cac90095ae049a111d9e98d
OMNIVOICE_VERSION?=53e6c2066150802ad3cd4b655b31c696e78e0019
SO_TARGET?=libgomnivoicecpp.so
CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
+2 -2
View File
@@ -1,6 +1,6 @@
# parakeet-cpp backend Makefile.
#
# Upstream pin lives below as PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6
# Upstream pin lives below as PARAKEET_VERSION?=623a968bccbd2214588df398fcce687cd4218dea
# (.github/bump_deps.sh) can find and update it - matches the
# whisper.cpp / ds4 / vibevoice-cpp convention.
#
@@ -15,7 +15,7 @@
# That's what the L0 smoke test uses. The default target below does the
# proper clone-at-pin + cmake build so CI doesn't need a side-checkout.
PARAKEET_VERSION?=2bf88954dc628b32835734e2e9159550a75a1dc6
PARAKEET_VERSION?=623a968bccbd2214588df398fcce687cd4218dea
PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp
GOCMD?=go
+328
View File
@@ -0,0 +1,328 @@
package main
import (
"encoding/json"
"fmt"
"sort"
"strconv"
"strings"
"github.com/mudler/LocalAI/pkg/grpc/grpcerrors"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
"github.com/mudler/xlog"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)
// diarizeSegmentJSON mirrors one element of parakeet_capi_diarize_pcm's
// "segments" array: {"speaker":0,"start":0.50,"end":5.52}.
type diarizeSegmentJSON struct {
Speaker int `json:"speaker"`
Start float64 `json:"start"`
End float64 `json:"end"`
}
// diarizePCMDoc mirrors the document parakeet_capi_diarize_pcm returns.
// "speakers" is the model's CAPACITY (e.g. 8 for Nemotron-3-Diarization),
// not the count of speakers actually present, so it is not read here; the
// response's num_speakers is computed from distinct segment labels instead.
type diarizePCMDoc struct {
Segments []diarizeSegmentJSON `json:"segments"`
}
// diarizeUtteranceJSON mirrors one element of
// parakeet_capi_transcribe_and_diarize_json's "utterances" array. Speaker is
// -1 when no diarized speaker overlaps the utterance.
type diarizeUtteranceJSON struct {
Speaker int `json:"speaker"`
Text string `json:"text"`
Start float64 `json:"start"`
End float64 `json:"end"`
}
// transcribeAndDiarizeDoc mirrors the document
// parakeet_capi_transcribe_and_diarize_json returns. Only "utterances" is
// consumed here; the per-word "words" detail belongs to a speaker-attributed
// transcript RPC, not Diarize.
type transcribeAndDiarizeDoc struct {
Utterances []diarizeUtteranceJSON `json:"utterances"`
}
// speakerLabel renders a 0-based speaker index as the decimal string
// DiarizeSegment.speaker documents, or "unknown" for -1 (no diarized speaker
// overlaps this utterance; only transcribe_and_diarize_json can report this).
func speakerLabel(speaker int) string {
if speaker < 0 {
return "unknown"
}
return strconv.Itoa(speaker)
}
// unsupportedDiarizeFields names the DiarizeRequest fields Sortformer has no
// equivalent for: it is an end-to-end model with a fixed speaker capacity and
// no clustering stage, so there is no config knob to target a speaker count
// or a clustering distance. Logged rather than rejected, so a request naming
// one of these still gets the diarization it can have.
func unsupportedDiarizeFields(req *pb.DiarizeRequest) []string {
var out []string
if req.GetNumSpeakers() != 0 {
out = append(out, "num_speakers")
}
if req.GetMinSpeakers() != 0 {
out = append(out, "min_speakers")
}
if req.GetMaxSpeakers() != 0 {
out = append(out, "max_speakers")
}
if req.GetClusteringThreshold() != 0 {
out = append(out, "clustering_threshold")
}
return out
}
// Diarize labels who spoke when in the audio at req.Dst, using the loaded
// diarization model (p.diarCtx). When req.IncludeText is set and an ASR
// companion (p.ctxPtr) is loaded, each segment also carries its transcript
// (parakeet_capi_transcribe_and_diarize_json, one utterance per speaker
// turn); otherwise, or when no ASR companion is loaded, segments carry no
// text (parakeet_capi_diarize_pcm) and no error is raised.
func (p *ParakeetCpp) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, error) {
if p.diarCtx == 0 {
return pb.DiarizeResponse{}, status.Error(codes.FailedPrecondition,
"parakeet-cpp: model is not a diarization model")
}
if CppDiarizePCM == nil {
return pb.DiarizeResponse{}, status.Error(codes.Unimplemented,
"parakeet-cpp: loaded libparakeet.so has no diarization support (parakeet_capi_diarize_pcm missing)")
}
if req.GetDst() == "" {
return pb.DiarizeResponse{}, status.Error(codes.InvalidArgument,
"parakeet-cpp: DiarizeRequest.dst (audio path) is required")
}
if dropped := unsupportedDiarizeFields(req); len(dropped) > 0 {
xlog.Debug("parakeet-cpp: ignoring diarization request fields Sortformer has no equivalent for",
"fields", dropped)
}
pcm, duration, err := decodeWavMono16k(req.GetDst())
if err != nil {
return pb.DiarizeResponse{}, status.Errorf(codes.InvalidArgument, "parakeet-cpp: decode audio: %s", err)
}
if len(pcm) == 0 {
return pb.DiarizeResponse{}, status.Error(codes.InvalidArgument, "parakeet-cpp: empty audio")
}
wantText := req.GetIncludeText() && p.ctxPtr != 0 && CppTranscribeAndDiarizeJSON != nil
raw, err := p.diarizeCall(pcm, wantText)
if err != nil {
return pb.DiarizeResponse{}, err
}
segments, err := parseDiarizeDoc(raw, wantText)
if err != nil {
return pb.DiarizeResponse{}, err
}
segments = applyDurationFilters(segments, req.GetMinDurationOn(), req.GetMinDurationOff())
renumberDiarizeSegments(segments)
return pb.DiarizeResponse{
Segments: segments,
NumSpeakers: distinctDiarizeSpeakers(segments),
Duration: duration,
}, nil
}
// diarizeCall runs the single C call Diarize needs (transcribe_and_diarize_json
// when wantText, else diarize_pcm) under engineMu, and returns the raw JSON
// document. p.diarCtx (and, on the include_text path, p.ctxPtr) is re-checked
// under the lock before the C call: Diarize's own p.diarCtx==0/wantText checks
// run before this lock is taken, so a Free() racing in between (which zeroes
// those fields under the same engineMu) would otherwise reach the C side with
// a freed context. last_error is ctx-shared, so it is read under the same
// lock as the failing call.
func (p *ParakeetCpp) diarizeCall(pcm []float32, wantText bool) (string, error) {
p.engineMu.Lock()
defer p.engineMu.Unlock()
if p.diarCtx == 0 || (wantText && p.ctxPtr == 0) {
return "", grpcerrors.ModelNotLoaded("parakeet-cpp")
}
var cstr uintptr
if wantText {
cstr = CppTranscribeAndDiarizeJSON(p.ctxPtr, p.diarCtx, &pcm[0], int32(len(pcm)), 16000)
} else {
cstr = CppDiarizePCM(p.diarCtx, &pcm[0], int32(len(pcm)), 16000)
}
if cstr == 0 {
return "", fmt.Errorf("parakeet-cpp: diarize failed: %s", diarizeLastError(p, wantText))
}
raw := goStringFromCPtr(cstr)
CppFreeString(cstr)
return raw, nil
}
// diarizeLastError reads last_error off p.diarCtx and, on the include_text
// path, p.ctxPtr too — the failing call is CppTranscribeAndDiarizeJSON there,
// and either side of the pairing may be the one that set it — then joins
// whichever came back non-empty. Called under the same engineMu as the
// failing call (last_error is ctx-shared state).
func diarizeLastError(p *ParakeetCpp, wantText bool) string {
var msgs []string
if m := CppLastError(p.diarCtx); m != "" {
msgs = append(msgs, m)
}
if wantText {
if m := CppLastError(p.ctxPtr); m != "" {
msgs = append(msgs, m)
}
}
if len(msgs) == 0 {
return "unknown error"
}
return strings.Join(msgs, "; ")
}
// parseDiarizeDoc decodes the raw JSON diarizeCall returned into
// DiarizeSegments (without ids: renumberDiarizeSegments assigns those after
// filtering).
func parseDiarizeDoc(raw string, wantText bool) ([]*pb.DiarizeSegment, error) {
if wantText {
var doc transcribeAndDiarizeDoc
if err := json.Unmarshal([]byte(raw), &doc); err != nil {
return nil, fmt.Errorf("parakeet-cpp: decode diarize json: %w", err)
}
segs := make([]*pb.DiarizeSegment, 0, len(doc.Utterances))
for _, u := range doc.Utterances {
segs = append(segs, &pb.DiarizeSegment{
Start: float32(u.Start),
End: float32(u.End),
Speaker: speakerLabel(u.Speaker),
Text: u.Text,
})
}
return segs, nil
}
var doc diarizePCMDoc
if err := json.Unmarshal([]byte(raw), &doc); err != nil {
return nil, fmt.Errorf("parakeet-cpp: decode diarize json: %w", err)
}
segs := make([]*pb.DiarizeSegment, 0, len(doc.Segments))
for _, s := range doc.Segments {
segs = append(segs, &pb.DiarizeSegment{
Start: float32(s.Start),
End: float32(s.End),
Speaker: speakerLabel(s.Speaker),
})
}
return segs, nil
}
// applyDurationFilters applies the request's postprocessing knobs, in the
// order NeMo's diarization postprocessing does: merge first
// (min_duration_off), then drop short segments (min_duration_on) — dropping
// first would leave short gaps unmerged that the drop step just created.
// Segments are assumed sorted by start time, as parakeet_capi_diarize_pcm and
// parakeet_capi_transcribe_and_diarize_json document. A non-positive value
// disables that filter (the proto's "0 = backend default" reads here as "no
// filtering").
func applyDurationFilters(segs []*pb.DiarizeSegment, minOn, minOff float32) []*pb.DiarizeSegment {
segs = mergeCloseSegments(segs, minOff)
segs = dropShortSegments(segs, minOn)
return segs
}
// mergeCloseSegments merges SAME-SPEAKER segments separated by a gap shorter
// than minOff into one segment spanning both (and concatenating any text).
// Segments from different speakers are never merged, regardless of gap: the
// gap only ever means "the same speaker paused", never "two speakers are
// actually one".
//
// Merging runs per speaker rather than on the single start-sorted list: two
// segments of the same speaker are not necessarily adjacent in that list once
// another speaker's turn falls between them (A, B, A), and a start-sorted
// walk would then never compare the two A's at all. Grouping by speaker first
// keeps each group's own start order (segs is assumed start-sorted, as
// parakeet_capi_diarize_pcm and parakeet_capi_transcribe_and_diarize_json
// document), merges within the group, then the merged segments are re-sorted
// by start so interleaved speakers come back out in timeline order.
func mergeCloseSegments(segs []*pb.DiarizeSegment, minOff float32) []*pb.DiarizeSegment {
if minOff <= 0 || len(segs) < 2 {
return segs
}
bySpeaker := make(map[string][]*pb.DiarizeSegment)
var order []string // first-seen speaker order, for a deterministic group walk
for _, s := range segs {
if _, ok := bySpeaker[s.GetSpeaker()]; !ok {
order = append(order, s.GetSpeaker())
}
bySpeaker[s.GetSpeaker()] = append(bySpeaker[s.GetSpeaker()], s)
}
out := make([]*pb.DiarizeSegment, 0, len(segs))
for _, speaker := range order {
group := bySpeaker[speaker]
merged := make([]*pb.DiarizeSegment, 0, len(group))
merged = append(merged, group[0])
for _, s := range group[1:] {
prev := merged[len(merged)-1]
if s.GetStart()-prev.GetEnd() < minOff {
if s.GetEnd() > prev.GetEnd() {
prev.End = s.End
}
if s.GetText() != "" {
if prev.GetText() != "" {
prev.Text = prev.GetText() + " " + s.GetText()
} else {
prev.Text = s.GetText()
}
}
continue
}
merged = append(merged, s)
}
out = append(out, merged...)
}
sort.Slice(out, func(i, j int) bool { return out[i].GetStart() < out[j].GetStart() })
return out
}
// dropShortSegments discards segments shorter than minOn.
func dropShortSegments(segs []*pb.DiarizeSegment, minOn float32) []*pb.DiarizeSegment {
if minOn <= 0 {
return segs
}
out := make([]*pb.DiarizeSegment, 0, len(segs))
for _, s := range segs {
if s.GetEnd()-s.GetStart() < minOn {
continue
}
out = append(out, s)
}
return out
}
// renumberDiarizeSegments assigns sequential ids (0..) to the final segment
// list, after filtering may have dropped or merged entries.
func renumberDiarizeSegments(segs []*pb.DiarizeSegment) {
for i, s := range segs {
s.Id = int32(i)
}
}
// distinctDiarizeSpeakers counts the distinct speaker labels present in segs.
// This is what DiarizeResponse.num_speakers documents — the count of speakers
// actually present in the result — and is NOT the diarize_pcm JSON's
// top-level "speakers" field, which reports the model's fixed capacity.
func distinctDiarizeSpeakers(segs []*pb.DiarizeSegment) int32 {
seen := make(map[string]struct{}, len(segs))
for _, s := range segs {
seen[s.GetSpeaker()] = struct{}{}
}
return int32(len(seen))
}
+275
View File
@@ -0,0 +1,275 @@
package main
import (
"path/filepath"
"sync"
"unsafe"
"github.com/mudler/LocalAI/pkg/grpc/grpcerrors"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)
// The Diarize specs drive it entirely against stubbed CppDiarizePCM /
// CppTranscribeAndDiarizeJSON / CppFreeString / CppLastError (the same seam
// live_test.go and roles_test.go use), so they run without libparakeet.so.
// diarizeCstrPool hands out NUL-terminated C-style strings backed by Go
// memory and keeps them alive for the duration of a spec (goStringFromCPtr
// reads through the raw pointer; mirrors live_test.go's liveCstrPool).
type diarizeCstrPool struct {
mu sync.Mutex
bufs [][]byte
}
func (p *diarizeCstrPool) cstr(s string) uintptr {
p.mu.Lock()
defer p.mu.Unlock()
b := append([]byte(s), 0)
p.bufs = append(p.bufs, b)
return uintptr(unsafe.Pointer(&b[0]))
}
// diarizeStubs swaps every C entry point Diarize touches and returns a
// restore func for AfterEach (mirrors live_test.go's liveStubs).
func diarizeStubs() (restore func()) {
savedDiarize := CppDiarizePCM
savedTranscribeAndDiarize := CppTranscribeAndDiarizeJSON
savedFreeString := CppFreeString
savedLastError := CppLastError
return func() {
CppDiarizePCM = savedDiarize
CppTranscribeAndDiarizeJSON = savedTranscribeAndDiarize
CppFreeString = savedFreeString
CppLastError = savedLastError
}
}
// diarizeWav writes a silent 16 kHz mono WAV of the given duration (seconds)
// to a fresh temp file and returns its path. decodeWavMono16k reads real
// audio bytes off disk, so Diarize needs a file on disk even though the
// stubbed C calls never look at its samples.
func diarizeWav(seconds float64) string {
GinkgoHelper()
path := filepath.Join(GinkgoT().TempDir(), "diarize.wav")
writeMono16kWav(path, int(seconds*16000))
return path
}
var _ = Describe("ParakeetCpp.Diarize", func() {
var restore func()
var pool *diarizeCstrPool
BeforeEach(func() {
restore = diarizeStubs()
pool = &diarizeCstrPool{}
})
AfterEach(func() { restore() })
It("fails with FailedPrecondition when no diarization model is loaded", func() {
p := &ParakeetCpp{}
_, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1)})
Expect(err).To(HaveOccurred())
Expect(status.Code(err)).To(Equal(codes.FailedPrecondition))
Expect(err.Error()).To(ContainSubstring("model is not a diarization model"))
})
It("fails with Unimplemented when the loaded libparakeet.so has no diarize_pcm symbol", func() {
CppDiarizePCM = nil
p := &ParakeetCpp{diarCtx: 42}
_, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1)})
Expect(err).To(HaveOccurred())
Expect(status.Code(err)).To(Equal(codes.Unimplemented))
})
It("maps plain segments with sequential ids, decimal speaker labels and distinct speaker count", func() {
CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr {
return pool.cstr(`{"speakers":8,"segments":[` +
`{"speaker":0,"start":0.00,"end":3.00},` +
`{"speaker":1,"start":3.00,"end":6.00}]}`)
}
CppFreeString = func(uintptr) {}
p := &ParakeetCpp{diarCtx: 42}
resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(6)})
Expect(err).ToNot(HaveOccurred())
Expect(resp.Segments).To(HaveLen(2))
Expect(resp.Segments[0].Id).To(Equal(int32(0)))
Expect(resp.Segments[0].Speaker).To(Equal("0"))
Expect(resp.Segments[1].Id).To(Equal(int32(1)))
Expect(resp.Segments[1].Speaker).To(Equal("1"))
Expect(resp.Segments[0].Text).To(BeEmpty())
Expect(resp.NumSpeakers).To(Equal(int32(2)))
Expect(resp.Duration).To(BeNumerically("~", 6.0, 0.01))
})
It("fills text from utterances when include_text is set with an ASR companion, and maps speaker -1 to unknown", func() {
CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr {
Fail("diarize_pcm must not be called when include_text has an ASR companion to pair with")
return 0
}
CppTranscribeAndDiarizeJSON = func(asr, diar uintptr, samples *float32, n int32, sampleRate int32) uintptr {
return pool.cstr(`{"speakers":8,"utterances":[` +
`{"speaker":0,"text":"hello there","start":0.00,"end":1.00,"conf":0.9},` +
`{"speaker":-1,"text":"mumble","start":1.00,"end":1.50,"conf":0.4}],"words":[]}`)
}
CppFreeString = func(uintptr) {}
p := &ParakeetCpp{diarCtx: 42, ctxPtr: 7}
resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(2), IncludeText: true})
Expect(err).ToNot(HaveOccurred())
Expect(resp.Segments).To(HaveLen(2))
Expect(resp.Segments[0].Speaker).To(Equal("0"))
Expect(resp.Segments[0].Text).To(Equal("hello there"))
Expect(resp.Segments[1].Speaker).To(Equal("unknown"))
Expect(resp.Segments[1].Text).To(Equal("mumble"))
})
It("falls back to plain segments without error when include_text is set but no ASR companion is loaded", func() {
diarizeCalled := false
CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr {
diarizeCalled = true
return pool.cstr(`{"speakers":8,"segments":[{"speaker":0,"start":0.00,"end":1.00}]}`)
}
CppFreeString = func(uintptr) {}
CppTranscribeAndDiarizeJSON = func(asr, diar uintptr, samples *float32, n int32, sampleRate int32) uintptr {
Fail("transcribe_and_diarize_json must not be called without an ASR companion")
return 0
}
p := &ParakeetCpp{diarCtx: 42} // no ctxPtr companion
resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1), IncludeText: true})
Expect(err).ToNot(HaveOccurred())
Expect(diarizeCalled).To(BeTrue())
Expect(resp.Segments).To(HaveLen(1))
Expect(resp.Segments[0].Text).To(BeEmpty())
})
It("drops a segment shorter than min_duration_on", func() {
CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr {
return pool.cstr(`{"speakers":8,"segments":[` +
`{"speaker":0,"start":0.30,"end":0.50},` + // 0.2s, at 0.3
`{"speaker":0,"start":1.00,"end":2.00}]}`) // 1.0s, kept
}
CppFreeString = func(uintptr) {}
p := &ParakeetCpp{diarCtx: 42}
resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(3), MinDurationOn: 0.3})
Expect(err).ToNot(HaveOccurred())
Expect(resp.Segments).To(HaveLen(1))
Expect(resp.Segments[0].Id).To(Equal(int32(0)))
Expect(resp.Segments[0].Start).To(BeNumerically("~", 1.0, 0.001))
})
It("merges same-speaker segments across a short gap but not across a speaker change", func() {
CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr {
return pool.cstr(`{"speakers":8,"segments":[` +
`{"speaker":0,"start":0.00,"end":1.00},` +
`{"speaker":0,"start":1.30,"end":2.00},` + // 0.3s gap, same speaker: merges
`{"speaker":1,"start":2.10,"end":3.00}]}`) // 0.1s gap, different speaker: stays separate
}
CppFreeString = func(uintptr) {}
p := &ParakeetCpp{diarCtx: 42}
resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(3), MinDurationOff: 0.5})
Expect(err).ToNot(HaveOccurred())
Expect(resp.Segments).To(HaveLen(2))
Expect(resp.Segments[0].Id).To(Equal(int32(0)))
Expect(resp.Segments[0].Speaker).To(Equal("0"))
Expect(resp.Segments[0].Start).To(BeNumerically("~", 0.0, 0.001))
Expect(resp.Segments[0].End).To(BeNumerically("~", 2.0, 0.001))
Expect(resp.Segments[1].Id).To(Equal(int32(1)))
Expect(resp.Segments[1].Speaker).To(Equal("1"))
})
It("surfaces last_error when the C call returns NULL", func() {
CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr {
return 0
}
CppLastError = func(ctx uintptr) string { return "boom" }
p := &ParakeetCpp{diarCtx: 42}
_, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1)})
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(ContainSubstring("boom"))
})
It("reports last_error from both contexts when the include_text C call returns NULL", func() {
// Diarize's Unimplemented gate checks CppDiarizePCM regardless of
// wantText, so it needs a non-nil (never called) stub here too.
CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr {
Fail("diarize_pcm must not be called when include_text has an ASR companion to pair with")
return 0
}
CppTranscribeAndDiarizeJSON = func(asr, diar uintptr, samples *float32, n int32, sampleRate int32) uintptr {
return 0
}
CppLastError = func(ctx uintptr) string {
if ctx == 7 {
return "asr side broke"
}
return "diar side broke"
}
p := &ParakeetCpp{diarCtx: 42, ctxPtr: 7}
_, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1), IncludeText: true})
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(ContainSubstring("asr side broke"))
Expect(err.Error()).To(ContainSubstring("diar side broke"))
})
It("wraps a decode failure as InvalidArgument", func() {
CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr {
Fail("decode must fail before any C call is made")
return 0
}
p := &ParakeetCpp{diarCtx: 42}
_, err := p.Diarize(&pb.DiarizeRequest{Dst: filepath.Join(GinkgoT().TempDir(), "missing.wav")})
Expect(err).To(HaveOccurred())
Expect(status.Code(err)).To(Equal(codes.InvalidArgument))
})
It("returns ModelNotLoaded without a C call when diarCtx is zeroed between the entry check and the call", func() {
called := false
CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr {
called = true
return pool.cstr(`{"speakers":8,"segments":[]}`)
}
CppFreeString = func(uintptr) {}
p := &ParakeetCpp{diarCtx: 42}
// Simulate a Free() racing between Diarize's own diarCtx==0 check and
// diarizeCall's lock, exactly as it zeroes diarCtx under engineMu.
p.diarCtx = 0
_, err := p.diarizeCall(make([]float32, 10), false)
Expect(grpcerrors.IsModelNotLoaded(err)).To(BeTrue())
Expect(called).To(BeFalse(), "no C call once diarCtx was cleared")
})
It("merges same-speaker segments across an intervening different speaker (A, B, A)", func() {
CppDiarizePCM = func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr {
return pool.cstr(`{"speakers":8,"segments":[` +
`{"speaker":0,"start":0.00,"end":1.00},` +
`{"speaker":1,"start":1.05,"end":1.20},` + // short B segment sits between the two A's
`{"speaker":0,"start":1.30,"end":2.00}]}`) // 0.1s gap from the first A: same speaker, merges
}
CppFreeString = func(uintptr) {}
p := &ParakeetCpp{diarCtx: 42}
resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(3), MinDurationOff: 0.5})
Expect(err).ToNot(HaveOccurred())
// The two speaker-0 segments merge into one spanning 0.00-2.00, and
// the timeline re-sort puts speaker 1's untouched segment in between.
Expect(resp.Segments).To(HaveLen(2))
Expect(resp.Segments[0].Speaker).To(Equal("0"))
Expect(resp.Segments[0].Start).To(BeNumerically("~", 0.0, 0.001))
Expect(resp.Segments[0].End).To(BeNumerically("~", 2.0, 0.001))
Expect(resp.Segments[1].Speaker).To(Equal("1"))
Expect(resp.Segments[1].Start).To(BeNumerically("~", 1.05, 0.001))
Expect(resp.Segments[1].End).To(BeNumerically("~", 1.20, 0.001))
})
})
+154 -28
View File
@@ -74,8 +74,55 @@ var (
// libparakeet.so; nil falls back to the text-only CppStreamFeed/Finalize path.
CppStreamFeedJSON func(s uintptr, pcm []float32, nSamples int32) uintptr
CppStreamFinalizeJSON func(s uintptr) uintptr
// CppModelKind reports which kind of model a loaded context holds
// (parakeet_capi_model_kind, ABI v8): see the modelKind* constants in
// roles.go. nil on an older libparakeet.so; Load then treats the primary
// as ASR (pre-v8 behavior) and rejects companion model options.
CppModelKind func(ctx uintptr) int32
// Speaker diarization (ABI v7). CppDiarizePCM runs offline diarization
// over in-memory mono float PCM; CppTranscribeAndDiarizeJSON pairs it with
// an ASR context for speaker-attributed text. Both return a malloc'd char*
// JSON document (uintptr, freed via CppFreeString).
CppDiarizePCM func(ctx uintptr, samples *float32, n int32, sampleRate int32) uintptr
CppTranscribeAndDiarizeJSON func(asr, diar uintptr, samples *float32, n int32, sampleRate int32) uintptr
// Sound-event detection (CED) and the combined scene stream (ABI v8).
// CppNumClasses/CppSoundOptsDefault/CppSoundStreamBegin.../
// CppSceneOptsDefault/CppSceneStreamBegin... are only registered when
// CppModelKind is present (see main.go); nil otherwise.
CppNumClasses func(ctx uintptr) int32
CppSoundOptsDefault func(o *cSoundOpts)
CppSoundStreamBegin func(tagger uintptr, o *cSoundOpts) uintptr
CppSoundStreamFeed func(s uintptr, pcm *float32, n int32, isLast int32, out *uintptr, nOut *int32) int32
CppSoundStreamDrainScoresJSON func(s uintptr) uintptr
CppFreeSoundSegments func(segs uintptr)
CppSoundStreamFree func(s uintptr)
CppSceneOptsDefault func(o *cSceneOpts)
CppSceneStreamBegin func(asr, diar, tagger uintptr, o *cSceneOpts) uintptr
CppSceneStreamFeedJSON func(s uintptr, pcm *float32, n int32, isLast int32) uintptr
CppSceneStreamLastError func(s uintptr) string
CppSceneStreamFree func(s uintptr)
)
// cSoundOpts and cSceneOpts mirror parakeet_sound_opts / parakeet_scene_opts
// in parakeet_capi.h field-for-field (int -> int32, float -> float32); the
// C side sizes/versions them via the leading `size` field, set by the
// matching *_opts_default call.
type cSoundOpts struct {
Size int32
WindowSec, HopSec, OnThreshold, OffThreshold, MinDurationSec float32
TopK int32
}
type cSceneOpts struct {
Size int32
DiarLatency int32
Sound cSoundOpts
Flags int32
}
// streamChunkSamples is how much 16 kHz mono PCM we hand to stream_feed per
// call (1 s). The session buffers internally and decodes once a full
// cache-aware encoder chunk is available, so this only bounds how often we
@@ -140,10 +187,23 @@ type transcriptToken struct {
// touch it concurrently.
type ParakeetCpp struct {
base.Base
ctxPtr uintptr
engineMu sync.Mutex // sole guard of the one C engine (dispatcher + streaming)
bat *batcher
batStop chan struct{}
ctxPtr uintptr // ASR context: the primary when it is an ASR model, or the asr_model companion
// diarCtx / tagCtx are the diarization and sound (CED) model contexts:
// the primary when it is that kind, or the diarization_model/sound_model
// companion. See roles.go.
diarCtx uintptr
tagCtx uintptr
// diarLatency is the PARAKEET_DIAR_LATENCY_* mode for diarization
// streaming (diarization_latency: option, default "low"). Unused until
// the diarization/scene streaming paths land.
diarLatency int32
// companions holds every context this backend loaded itself beyond the
// primary (asr_model:/diarization_model:/sound_model: options), so Free
// can release them after the primary.
companions []uintptr
engineMu sync.Mutex // sole guard of the one C engine (dispatcher + streaming)
bat *batcher
batStop chan struct{}
// segmentGapFrames is NeMo's segment_gap_threshold in ENCODER FRAMES (model
// YAML option, default 0=off). When >0 it adds NeMo's silence-gap split on
// top of the punctuation split; converted to seconds via the JSON frame_sec.
@@ -151,21 +211,17 @@ type ParakeetCpp struct {
}
// Load is the LocalAI gRPC entry point for LoadModel: it calls
// parakeet_capi_load with the GGUF path and stashes the resulting
// opaque context pointer for AudioTranscription.
// parakeet_capi_load with the GGUF path, classifies it and any companion
// models named in Options[] by role (see roles.go), and starts the dynamic
// batcher when an ASR context (primary or companion) ends up loaded.
func (p *ParakeetCpp) Load(opts *pb.ModelOptions) error {
if opts.ModelFile == "" {
return errors.New("parakeet-cpp: ModelFile is required")
}
ctx := CppLoad(opts.ModelFile)
if ctx == 0 {
// No ctx to ask for last_error (the C-API's last-error buffer
// lives on the ctx that was never returned). Surface the path
// so the operator at least knows which load failed.
return fmt.Errorf("parakeet-cpp: parakeet_capi_load failed for %q", opts.ModelFile)
if err := p.loadRoles(opts); err != nil {
return err
}
p.ctxPtr = ctx
// Dynamic batching knobs (model YAML options:, key:value form). Batching is
// OFF by default (batch_max_size:1): each request runs on its own. On GPU,
@@ -182,6 +238,12 @@ func (p *ParakeetCpp) Load(opts *pb.ModelOptions) error {
// default matches NeMo's default (punctuation-only segments); when set it
// additionally splits segments on inter-word silence (see transcriptResultFromDoc).
p.segmentGapFrames = optInt(opts, "segment_gap_threshold", 0)
// The batcher only ever drives the ASR context; a diarization/sound
// primary with no asr_model companion has no ctxPtr and needs none.
if p.ctxPtr == 0 {
return nil
}
if CppTranscribePcmBatchJSON != nil {
p.batStop = make(chan struct{})
p.bat = newBatcher(maxSize, time.Duration(maxWaitMs)*time.Millisecond, p.runBatch)
@@ -287,12 +349,16 @@ func (p *ParakeetCpp) runBatch(reqs []*batchRequest) {
// OpenAI API, whose default is segment-level); token ids always populate
// Segment.Tokens.
//
// translate/diarize/prompt/temperature/threads are not applicable to parakeet
// and are ignored; language is honored on the batched + streaming paths (see
// opts.GetLanguage() below); streaming is handled by AudioTranscriptionStream
// (L2).
// With a diarization_model companion, diarize=true labels segments with their
// speaker (speakers.go). translate/prompt/temperature/threads are not
// applicable to parakeet and are ignored; language is honored on the batched +
// streaming paths (see opts.GetLanguage() below); streaming is handled by
// AudioTranscriptionStream (L2).
func (p *ParakeetCpp) AudioTranscription(ctx context.Context, opts *pb.TranscriptRequest) (pb.TranscriptResult, error) {
if p.ctxPtr == 0 {
if err := p.notASRError(); err != nil {
return pb.TranscriptResult{}, err
}
return pb.TranscriptResult{}, grpcerrors.ModelNotLoaded("parakeet-cpp")
}
if opts.Dst == "" {
@@ -350,7 +416,17 @@ func (p *ParakeetCpp) AudioTranscription(ctx context.Context, opts *pb.Transcrip
if err := json.Unmarshal([]byte(res.json), &doc); err != nil {
return pb.TranscriptResult{}, fmt.Errorf("parakeet-cpp: decode transcript json: %w", err)
}
return transcriptResultFromDoc(doc, opts, p.segmentGapFrames), nil
// With a diarization_model companion, label each segment with its speaker.
var speakers []int
if p.wantSpeakers(opts.GetDiarize()) && len(doc.Words) > 0 {
segs, err := p.diarizeSegmentsPCM(pcm)
if err != nil {
return pb.TranscriptResult{}, err
}
speakers = assignSpeakers(doc.Words, segs)
}
return transcriptResultWithSpeakers(doc, opts, p.segmentGapFrames, speakers), nil
}
// segmentSeparators is NeMo's default segment_seperators (sentence-ending
@@ -365,6 +441,14 @@ var segmentSeparators = []rune{'.', '?', '!'}
// the caller requested word granularity; token ids populate each segment's
// Tokens by time-window membership. Shared by the batched and direct paths.
func transcriptResultFromDoc(doc transcriptJSON, opts *pb.TranscriptRequest, gapFrames int) pb.TranscriptResult {
return transcriptResultWithSpeakers(doc, opts, gapFrames, nil)
}
// transcriptResultWithSpeakers is transcriptResultFromDoc plus optional
// per-word speakers (indexed like doc.Words, -1 = none; see speakers.go):
// segments additionally split wherever the speaker changes, and segments and
// words carry the speaker's label.
func transcriptResultWithSpeakers(doc transcriptJSON, opts *pb.TranscriptRequest, gapFrames int, speakers []int) pb.TranscriptResult {
text, eou := stripEouMarker(strings.TrimSpace(doc.Text))
// Frame-unit gap threshold -> seconds (NeMo segment_gap_threshold). 0 = off.
@@ -388,6 +472,11 @@ func transcriptResultFromDoc(doc transcriptJSON, opts *pb.TranscriptRequest, gap
}
}
var groupSpeakers []int
if speakers != nil && len(speakers) == len(doc.Words) {
groups, groupSpeakers = splitAtSpeakerChanges(groups, speakers)
}
wantWords := wordsRequested(opts.TimestampGranularities)
segments := make([]*pb.TranscriptSegment, 0, len(groups))
for id, group := range groups {
@@ -402,10 +491,14 @@ func transcriptResultFromDoc(doc transcriptJSON, opts *pb.TranscriptRequest, gap
Text: strings.TrimSpace(strings.Join(parts, " ")),
Tokens: tokensInWindow(doc.Tokens, group[0].Start, group[len(group)-1].End),
}
if groupSpeakers != nil {
seg.Speaker = transcriptSpeaker(groupSpeakers[id])
}
if wantWords {
ws := make([]*pb.TranscriptWord, len(group))
for i, gw := range group {
ws[i] = &pb.TranscriptWord{Start: secondsToNanos(gw.Start), End: secondsToNanos(gw.End), Text: gw.W}
ws[i] = &pb.TranscriptWord{Start: secondsToNanos(gw.Start), End: secondsToNanos(gw.End), Text: gw.W,
Speaker: seg.Speaker}
}
seg.Words = ws
}
@@ -503,10 +596,11 @@ func tokensInWindow(tokens []transcriptToken, start, end float64) []int32 {
// text-only library (no words) it falls back to segmenting the delta text, so
// the same assembler serves both paths.
type streamSegmenter struct {
segs []*pb.TranscriptSegment
cur []transcriptWord // words for the open segment (ABI v4 JSON path)
curText []string // delta text for the open segment (text-only path)
nextID int32
segs []*pb.TranscriptSegment
segWords [][]transcriptWord // words of each segment (nil for text-only ones)
cur []transcriptWord // words for the open segment (ABI v4 JSON path)
curText []string // delta text for the open segment (text-only path)
nextID int32
}
func (s *streamSegmenter) add(r streamFeedResult) {
@@ -534,12 +628,14 @@ func (s *streamSegmenter) flush() {
End: secondsToNanos(s.cur[len(s.cur)-1].End),
Text: strings.TrimSpace(strings.Join(parts, " ")),
})
s.segWords = append(s.segWords, s.cur)
s.nextID++
case len(s.curText) > 0:
// No words this segment: emit a text-only segment (no timestamps),
// skipping a purely-whitespace one as the legacy text path did.
if t := strings.TrimSpace(strings.Join(s.curText, "")); t != "" {
s.segs = append(s.segs, &pb.TranscriptSegment{Id: s.nextID, Text: t})
s.segWords = append(s.segWords, nil)
s.nextID++
}
}
@@ -686,6 +782,9 @@ func (p *ParakeetCpp) AudioTranscriptionStream(ctx context.Context, opts *pb.Tra
defer close(results)
if p.ctxPtr == 0 {
if err := p.notASRError(); err != nil {
return err
}
return grpcerrors.ModelNotLoaded("parakeet-cpp")
}
if opts.Dst == "" {
@@ -753,6 +852,28 @@ func (p *ParakeetCpp) AudioTranscriptionStream(ctx context.Context, opts *pb.Tra
// The single-segment fallback stays trimmed.
fullText := full.String()
segments := seg.segments()
// With a diarization_model companion, label each utterance with the
// speaker who said most of it. The whole file is available, so this runs
// the same diarization as the unary path.
if p.wantSpeakers(opts.GetDiarize()) && len(seg.segWords) == len(segments) {
var all []transcriptWord
for _, ws := range seg.segWords {
all = append(all, ws...)
}
if len(all) > 0 {
segs, err := p.diarizeSegmentsPCM(data)
if err != nil {
return err
}
speakers := assignSpeakers(all, segs)
k := 0
for i, ws := range seg.segWords {
segments[i].Speaker = transcriptSpeaker(majoritySpeaker(ws, speakers[k:k+len(ws)]))
k += len(ws)
}
}
}
if trimmed := strings.TrimSpace(fullText); len(segments) == 0 && trimmed != "" {
segments = append(segments, &pb.TranscriptSegment{Id: 0, Text: trimmed})
}
@@ -817,8 +938,10 @@ func decodeWavMono16k(path string) ([]float32, float32, error) {
return data, duration, nil
}
// Free releases the underlying parakeet_ctx. Called by LocalAI when the
// model is unloaded.
// Free releases every parakeet_ctx this backend holds (the primary and any
// asr_model:/diarization_model:/sound_model: companions loaded in Load) and
// is idempotent: fields are zeroed as they are freed, so a second call frees
// nothing. Called by LocalAI when the model is unloaded.
func (p *ParakeetCpp) Free() error {
// Stop the dispatcher before releasing the engine so no in-flight runBatch
// can touch a freed ctx (close leak / use-after-free on reload).
@@ -830,10 +953,13 @@ func (p *ParakeetCpp) Free() error {
// re-checks ctxPtr under the lock) can never feed into a freed ctx.
p.engineMu.Lock()
defer p.engineMu.Unlock()
if p.ctxPtr != 0 {
CppFree(p.ctxPtr)
p.ctxPtr = 0
for _, ctxField := range [...]*uintptr{&p.ctxPtr, &p.diarCtx, &p.tagCtx} {
if *ctxField != 0 {
CppFree(*ctxField)
*ctxField = 0
}
}
p.companions = nil
return nil
}
@@ -59,6 +59,23 @@ func ensureLibLoaded() {
purego.RegisterLibFunc(&CppStreamFeedJSON, lib, "parakeet_capi_stream_feed_json")
purego.RegisterLibFunc(&CppStreamFinalizeJSON, lib, "parakeet_capi_stream_finalize_json")
}
// Diarization and model roles, probed like main.go (speakers_test.go).
if sym, err := purego.Dlsym(lib, "parakeet_capi_diarize_pcm"); err == nil && sym != 0 {
purego.RegisterLibFunc(&CppDiarizePCM, lib, "parakeet_capi_diarize_pcm")
}
if sym, err := purego.Dlsym(lib, "parakeet_capi_transcribe_and_diarize_json"); err == nil && sym != 0 {
purego.RegisterLibFunc(&CppTranscribeAndDiarizeJSON, lib, "parakeet_capi_transcribe_and_diarize_json")
}
if sym, err := purego.Dlsym(lib, "parakeet_capi_model_kind"); err == nil && sym != 0 {
purego.RegisterLibFunc(&CppModelKind, lib, "parakeet_capi_model_kind")
purego.RegisterLibFunc(&CppNumClasses, lib, "parakeet_capi_num_classes")
purego.RegisterLibFunc(&CppSoundOptsDefault, lib, "parakeet_capi_sound_opts_default")
purego.RegisterLibFunc(&CppSoundStreamBegin, lib, "parakeet_capi_sound_stream_begin")
purego.RegisterLibFunc(&CppSoundStreamFeed, lib, "parakeet_capi_sound_stream_feed")
purego.RegisterLibFunc(&CppSoundStreamDrainScoresJSON, lib, "parakeet_capi_sound_stream_drain_scores_json")
purego.RegisterLibFunc(&CppFreeSoundSegments, lib, "parakeet_capi_free_sound_segments")
purego.RegisterLibFunc(&CppSoundStreamFree, lib, "parakeet_capi_sound_stream_free")
}
purego.RegisterLibFunc(&CppFreeString, lib, "parakeet_capi_free_string")
purego.RegisterLibFunc(&CppLastError, lib, "parakeet_capi_last_error")
})
@@ -203,6 +220,24 @@ var _ = Describe("ParakeetCpp", func() {
})
Context("AudioTranscriptionStream", func() {
It("names the loaded role instead of a generic model-not-loaded error for a diarization primary", func() {
// CppStreamBegin/CppStreamBeginLang are left nil (zero value): if
// AudioTranscriptionStream tried to call either, this would panic
// instead of returning cleanly, so a clean typed error here also
// proves no C call was made.
p := &ParakeetCpp{diarCtx: 1}
results := make(chan *pb.TranscriptStreamResponse, 8)
err := p.AudioTranscriptionStream(context.Background(),
&pb.TranscriptRequest{Dst: "ignored.wav"}, results)
Expect(err).To(MatchError(ContainSubstring("diarization model")))
var emitted []*pb.TranscriptStreamResponse
for r := range results {
emitted = append(emitted, r)
}
Expect(emitted).To(BeEmpty())
})
It("returns the typed Unimplemented signal for non-streaming models (no offline fallback)", func() {
// stream_begin == 0 means the loaded model is not a cache-aware
// streaming model. The backend must surface that, not silently
+72 -14
View File
@@ -41,6 +41,9 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest
defer close(out)
if p.ctxPtr == 0 {
if err := p.notASRError(); err != nil {
return err
}
return grpcerrors.ModelNotLoaded("parakeet-cpp")
}
@@ -68,6 +71,23 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest
// current when the RPC unwinds.
defer func() { p.streamFree(stream) }()
// scene runs a no-ASR scene stream (diarization/sound only) beside the
// ASR session when a diarization_model:/sound_model: companion is loaded
// (see scene.go). A zero handle means scene events are disabled: no
// companions, or the begin/a later feed call failed (logged below / in
// feedSlicesScene), in which case live transcription continues ASR-only.
// Reassigned on a mid-stream Config reset alongside stream, which also
// brings back a scene stream the session had disabled after an earlier
// scene error.
var scene sceneStreamHandle
if p.sceneWanted() {
scene = p.sceneBegin()
if scene.s == 0 {
xlog.Warn("parakeet-cpp: scene stream begin failed; live continues without speaker/sound events")
}
}
defer func() { p.sceneFree(scene) }()
out <- &pb.TranscriptLiveResponse{Ready: true}
var (
@@ -83,22 +103,31 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest
behindWarned bool
)
// emit forwards one decode increment: it streams the per-feed tokens the
// realtime turn detector consumes (delta/eou/eob/words) and accumulates the
// running transcript for the closing FinalResult. No segmentation or
// boundary latch here — the live consumer reads only the streamed tokens
// and the final Text; per-utterance segments and the terminal <EOU> flag
// are an offline-path concern (see AudioTranscriptionStream / boundary.go).
emit := func(r streamFeedResult) error {
// emit sends one decode increment as its own response when it carries
// anything: either the ASR side (delta/eou/eob/words, accumulated into
// the running transcript for the closing FinalResult) or the scene
// side (closed speaker/sound events), never both at once — the live
// audio loop below calls it once for the ASR result right after the ASR
// feed and, separately, once more for the scene document after the
// scene feed (see feedSlicesScene), so a slice with both produces two
// responses, ASR first. No segmentation or boundary latch here — the
// live consumer reads only the streamed tokens and the final Text;
// per-utterance segments and the terminal <EOU> flag are an
// offline-path concern (see AudioTranscriptionStream / boundary.go).
emit := func(r streamFeedResult, sceneDoc sceneFeedJSON) error {
if r.Delta != "" {
full.WriteString(r.Delta)
}
if r.Delta != "" || r.Eou || r.Eob || len(r.Words) > 0 {
speakers := liveSpeakersToProto(sceneDoc.Speakers)
sounds := liveSoundsToProto(sceneDoc.Sounds)
if r.Delta != "" || r.Eou || r.Eob || len(r.Words) > 0 || len(speakers) > 0 || len(sounds) > 0 {
out <- &pb.TranscriptLiveResponse{
Delta: r.Delta,
Eou: r.Eou,
Eob: r.Eob,
Words: liveWordsToProto(r.Words),
Delta: r.Delta,
Eou: r.Eou,
Eob: r.Eob,
Words: liveWordsToProto(r.Words),
Speakers: speakers,
Sounds: sounds,
}
}
return nil
@@ -120,8 +149,20 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest
return grpcerrors.LiveTranscriptionUnsupported("parakeet-cpp",
"loaded model is not a cache-aware streaming model")
}
// The scene stream is freed and begun again alongside the ASR
// session, mirroring the reset above.
p.sceneFree(scene)
scene = sceneStreamHandle{}
if p.sceneWanted() {
scene = p.sceneBegin()
if scene.s == 0 {
xlog.Warn("parakeet-cpp: scene stream begin failed; live continues without speaker/sound events")
}
}
full.Reset()
fedSecs = 0
behindSec = 0
behindWarned = false
case *pb.TranscriptLiveRequest_Audio:
pcm := payload.Audio.GetPcm()
audioSec := float64(len(pcm)) / liveSampleRate
@@ -129,7 +170,9 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest
start := time.Now()
// nil ctx: a live session is bounded by this request channel, not a
// context — cancellation is the caller closing the stream.
if err := p.feedSlices(nil, stream, pcm, emit); err != nil {
var asrWall, sceneWall time.Duration
scene, asrWall, sceneWall, err = p.feedSlicesScene(nil, stream, scene, pcm, emit)
if err != nil {
return err
}
wallSec := time.Since(start).Seconds()
@@ -139,6 +182,7 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest
}
xlog.Debug("parakeet-cpp: live feed",
"audio_ms", int(audioSec*1000), "wall_ms", int(wallSec*1000),
"asr_wall_ms", int(asrWall.Seconds()*1000), "scene_wall_ms", int(sceneWall.Seconds()*1000),
"behind_ms", int(behindSec*1000), "fed_s", fedSecs)
if behindSec > 1 && !behindWarned {
behindWarned = true
@@ -153,9 +197,23 @@ func (p *ParakeetCpp) AudioTranscriptionLive(in <-chan *pb.TranscriptLiveRequest
// The live FinalResult carries only Text — the authoritative full-turn
// transcript the realtime core commits. Per-utterance segments, duration,
// and the terminal <EOU> flag are not produced on the live path.
if err := p.flushTail(stream, emit); err != nil {
if err := p.flushTail(stream, func(r streamFeedResult) error {
return emit(r, sceneFeedJSON{})
}); err != nil {
return err
}
// The scene stream gets its own is_last flush (it consumes no new audio
// here, so it is not part of flushTail above); its remaining events go
// out before the terminal FinalResult, then the stream is released by the
// deferred sceneFree above.
if scene.s != 0 {
doc, err := p.sceneFeed(scene, nil, true)
if err != nil {
xlog.Warn("parakeet-cpp: live scene finalize failed", "err", err)
} else if err := emit(streamFeedResult{}, doc); err != nil {
return err
}
}
out <- &pb.TranscriptLiveResponse{
FinalResult: &pb.TranscriptResult{Text: strings.TrimSpace(full.String())},
}
+304 -2
View File
@@ -42,22 +42,61 @@ func liveStubs() (restore func()) {
savedFinalize, savedFinalizeJSON := CppStreamFinalize, CppStreamFinalizeJSON
savedFree, savedLastError := CppStreamFree, CppLastError
savedFreeString := CppFreeString
savedSceneOptsDefault := CppSceneOptsDefault
savedSceneBegin := CppSceneStreamBegin
savedSceneFeedJSON := CppSceneStreamFeedJSON
savedSceneLastError := CppSceneStreamLastError
savedSceneFree := CppSceneStreamFree
return func() {
CppStreamBegin, CppStreamBeginLang = savedBegin, savedBeginLang
CppStreamFeed, CppStreamFeedJSON = savedFeed, savedFeedJSON
CppStreamFinalize, CppStreamFinalizeJSON = savedFinalize, savedFinalizeJSON
CppStreamFree, CppLastError = savedFree, savedLastError
CppFreeString = savedFreeString
CppSceneOptsDefault = savedSceneOptsDefault
CppSceneStreamBegin = savedSceneBegin
CppSceneStreamFeedJSON = savedSceneFeedJSON
CppSceneStreamLastError = savedSceneLastError
CppSceneStreamFree = savedSceneFree
}
}
// liveSceneStubs wires a minimal scene stream stub set onto p (a diarization
// and/or sound companion context so sceneWanted() is true) and returns the
// call-count trackers the specs assert on. feedJSON is called once per scene
// feed (including the is_last flush) with the stream handle the C side would
// have received (so a reset spec can tell a pre-reset feed from a post-reset
// one) and must return the canned document for that call.
func liveSceneStubs(feedJSON func(calls int, s uintptr, isLast int32) uintptr) (begun, freed *int) {
begun, freed = new(int), new(int)
CppSceneOptsDefault = func(o *cSceneOpts) { *o = cSceneOpts{} }
CppSceneStreamBegin = func(asr, diar, tagger uintptr, o *cSceneOpts) uintptr {
*begun++
return uintptr(100 + *begun)
}
calls := 0
CppSceneStreamFeedJSON = func(s uintptr, pcm *float32, n int32, isLast int32) uintptr {
calls++
return feedJSON(calls, s, isLast)
}
CppSceneStreamLastError = func(s uintptr) string { return "scene stub error" }
CppSceneStreamFree = func(s uintptr) { *freed++ }
return begun, freed
}
// runLive starts the RPC on its own goroutine and returns the request
// channel plus a collector for everything the backend emitted.
// channel plus a collector for everything the backend emitted. GinkgoRecover
// turns an Expect/Fail failure inside a stub called from this goroutine into
// a normal spec failure instead of a panic that would crash the whole test
// binary (Ginkgo's failure handling is goroutine-local).
func runLive(p *ParakeetCpp) (chan *pb.TranscriptLiveRequest, chan *pb.TranscriptLiveResponse, chan error) {
in := make(chan *pb.TranscriptLiveRequest)
out := make(chan *pb.TranscriptLiveResponse, 32)
errCh := make(chan error, 1)
go func() { errCh <- p.AudioTranscriptionLive(in, out) }()
go func() {
defer GinkgoRecover()
errCh <- p.AudioTranscriptionLive(in, out)
}()
return in, out, errCh
}
@@ -106,6 +145,19 @@ var _ = Describe("AudioTranscriptionLive (stubbed C API)", func() {
AfterEach(func() { restore() })
It("names the loaded role instead of a generic model-not-loaded error for a sound primary", func() {
// The ctxPtr==0 check returns before AudioTranscriptionLive ever reads
// from `in`, so nothing may be sent on it (unbuffered: a send would
// block forever waiting for a read that never happens).
p2 := &ParakeetCpp{tagCtx: 1}
in, out, errCh := runLive(p2)
close(in)
err := <-errCh
Expect(err).To(MatchError(ContainSubstring("sound model")))
Expect(collectLive(out)).To(BeEmpty())
})
It("rejects a stream whose first message is not a config", func() {
in, out, errCh := runLive(p)
in <- liveAudio([]float32{0.1})
@@ -368,6 +420,256 @@ var _ = Describe("AudioTranscriptionLive (stubbed C API)", func() {
Expect(got).To(HaveLen(1)) // just the ready ack
close(in)
})
It("makes no scene C call and behaves unchanged when no companion is loaded", func() {
// p has ctxPtr only (no diarCtx/tagCtx): sceneWanted() must be false,
// and none of the scene entry points may be touched.
CppSceneOptsDefault = func(o *cSceneOpts) { Fail("scene_opts_default called with no companions loaded") }
CppSceneStreamBegin = func(asr, diar, tagger uintptr, o *cSceneOpts) uintptr {
Fail("scene_stream_begin called with no companions loaded")
return 0
}
CppSceneStreamFeedJSON = func(s uintptr, pcm *float32, n int32, isLast int32) uintptr {
Fail("scene_stream_feed_json called with no companions loaded")
return 0
}
CppSceneStreamFree = func(s uintptr) { Fail("scene_stream_free called with no companions loaded") }
CppStreamFeedJSON = func(s uintptr, pcm []float32, n int32) uintptr {
return pool.cstr(`{"text":"hi","eou":0,"frame_sec":0.08,"words":[]}`)
}
CppStreamFinalizeJSON = func(s uintptr) uintptr {
return pool.cstr(`{"text":"","eou":0,"frame_sec":0.08,"words":[]}`)
}
in, out, errCh := runLive(p)
in <- liveConfig("")
in <- liveAudio(make([]float32, 10))
close(in)
Expect(<-errCh).NotTo(HaveOccurred())
got := collectLive(out)
Expect(got).To(HaveLen(3)) // ready, delta, final
Expect(got[1].Speakers).To(BeEmpty())
Expect(got[1].Sounds).To(BeEmpty())
})
})
var _ = Describe("AudioTranscriptionLive scene events (stubbed C API)", func() {
var (
pool *liveCstrPool
restore func()
p *ParakeetCpp
)
BeforeEach(func() {
pool = &liveCstrPool{}
restore = liveStubs()
p = &ParakeetCpp{ctxPtr: 1, diarCtx: 2}
CppStreamBeginLang = nil
CppStreamBegin = func(ctx uintptr) uintptr { return 7 }
CppStreamFree = func(s uintptr) {}
CppFreeString = func(s uintptr) {}
CppLastError = func(ctx uintptr) string { return "stub error" }
CppStreamFeed = nil
CppStreamFeedJSON = func(s uintptr, pcm []float32, n int32) uintptr {
return pool.cstr(`{"text":"","eou":0,"frame_sec":0.08,"words":[]}`)
}
CppStreamFinalize = nil
CppStreamFinalizeJSON = func(s uintptr) uintptr {
return pool.cstr(`{"text":"","eou":0,"frame_sec":0.08,"words":[]}`)
}
})
AfterEach(func() { restore() })
It("emits a closed speaker segment as its own response", func() {
liveSceneStubs(func(calls int, s uintptr, isLast int32) uintptr {
if calls == 1 {
return pool.cstr(`{"speakers":[{"speaker":0,"start":0.1,"end":0.6}],"sounds":[]}`)
}
return pool.cstr(`{"speakers":[],"sounds":[]}`)
})
in, out, errCh := runLive(p)
in <- liveConfig("")
in <- liveAudio(make([]float32, 10))
close(in)
Expect(<-errCh).NotTo(HaveOccurred())
got := collectLive(out)
Expect(got).To(HaveLen(3)) // ready, speaker-only response, final
Expect(got[1].Delta).To(BeEmpty())
Expect(got[1].Speakers).To(HaveLen(1))
Expect(got[1].Speakers[0].Speaker).To(Equal("0"))
Expect(got[1].Speakers[0].Start).To(Equal(int64(0.1 * 1e9)))
Expect(got[1].Speakers[0].End).To(Equal(int64(0.6 * 1e9)))
})
It("sends the ASR delta and a scene sound event as two responses, ASR first", func() {
CppStreamFeedJSON = func(s uintptr, pcm []float32, n int32) uintptr {
return pool.cstr(`{"text":"hello ","eou":0,"frame_sec":0.08,` +
`"words":[{"w":"hello","start":0.1,"end":0.4,"conf":0.9}]}`)
}
liveSceneStubs(func(calls int, s uintptr, isLast int32) uintptr {
if calls == 1 {
return pool.cstr(`{"speakers":[],"sounds":[{"index":99,` +
`"label":"Chicken, rooster","start":24.0,"end":30.0,"peak":0.86}]}`)
}
return pool.cstr(`{"speakers":[],"sounds":[]}`)
})
in, out, errCh := runLive(p)
in <- liveConfig("")
in <- liveAudio(make([]float32, 10))
close(in)
Expect(<-errCh).NotTo(HaveOccurred())
got := collectLive(out)
Expect(got).To(HaveLen(4)) // ready, ASR delta, sound-only, final
Expect(got[1].Delta).To(Equal("hello "))
Expect(got[1].Sounds).To(BeEmpty(), "the ASR response must not wait on the scene feed")
Expect(got[2].Delta).To(BeEmpty())
Expect(got[2].Sounds).To(HaveLen(1))
Expect(got[2].Sounds[0].Label).To(Equal("Chicken, rooster"))
Expect(got[2].Sounds[0].Index).To(Equal(int32(99)))
Expect(got[2].Sounds[0].Peak).To(BeNumerically("~", 0.86, 1e-6))
Expect(got[2].Sounds[0].Start).To(Equal(int64(24.0 * 1e9)))
Expect(got[2].Sounds[0].End).To(Equal(int64(30.0 * 1e9)))
})
It("flushes the scene stream is_last before the final result, then frees it", func() {
begun, freed := liveSceneStubs(func(calls int, s uintptr, isLast int32) uintptr {
if calls == 2 {
Expect(isLast).To(Equal(int32(1)))
return pool.cstr(`{"speakers":[{"speaker":1,"start":1.0,"end":2.0}],"sounds":[]}`)
}
Expect(isLast).To(Equal(int32(0)))
return pool.cstr(`{"speakers":[],"sounds":[]}`)
})
in, out, errCh := runLive(p)
in <- liveConfig("")
in <- liveAudio(make([]float32, 10))
close(in)
Expect(<-errCh).NotTo(HaveOccurred())
got := collectLive(out)
Expect(got).To(HaveLen(3)) // ready, speaker from the is_last flush, final
Expect(got[1].Speakers).To(HaveLen(1))
Expect(got[1].Speakers[0].Speaker).To(Equal("1"))
Expect(got[2].FinalResult).NotTo(BeNil())
Expect(*begun).To(Equal(1))
Expect(*freed).To(Equal(1))
})
It("frees and begins the scene stream again on a mid-stream config reset", func() {
streamBegun := 0
CppStreamBegin = func(ctx uintptr) uintptr { streamBegun++; return uintptr(10 + streamBegun) }
var seenStreams []uintptr
begun, freed := liveSceneStubs(func(calls int, s uintptr, isLast int32) uintptr {
seenStreams = append(seenStreams, s)
return pool.cstr(`{"speakers":[],"sounds":[]}`)
})
in, out, errCh := runLive(p)
in <- liveConfig("")
in <- liveAudio(make([]float32, 10))
in <- liveConfig("") // reset
in <- liveAudio(make([]float32, 10))
close(in)
Expect(<-errCh).NotTo(HaveOccurred())
collectLive(out)
Expect(*begun).To(Equal(2), "scene stream begun again on reset")
Expect(*freed).To(Equal(2), "old scene stream freed on reset, new one on unwind")
// One scene feed per audio message (pre-reset, post-reset) plus the
// close is_last flush, which runs on the post-reset stream.
Expect(seenStreams).To(HaveLen(3))
Expect(seenStreams[0]).NotTo(Equal(seenStreams[1]), "post-reset audio must go to the new scene stream handle")
Expect(seenStreams[2]).To(Equal(seenStreams[1]), "the close flush uses the post-reset stream too")
})
It("returns without a C call when Free() ran between begin and a scene feed", func() {
feedCalls := 0
CppSceneOptsDefault = func(o *cSceneOpts) { *o = cSceneOpts{} }
CppSceneStreamBegin = func(asr, diar, tagger uintptr, o *cSceneOpts) uintptr { return 999 }
CppSceneStreamFeedJSON = func(s uintptr, pcm *float32, n int32, isLast int32) uintptr {
feedCalls++
return pool.cstr(`{"speakers":[],"sounds":[]}`)
}
h := p.sceneBegin()
Expect(h.s).NotTo(BeZero())
// Simulate a Free() racing in between the begin and the next feed: it
// zeroes the companion context under engineMu, exactly as the real
// Free() does.
p.diarCtx = 0
_, err := p.sceneFeed(h, make([]float32, 10), false)
Expect(grpcerrors.IsModelNotLoaded(err)).To(BeTrue())
Expect(feedCalls).To(Equal(0), "no C call once the scene stream's contexts were freed")
})
It("degrades to ASR-only after a mid-session scene feed failure: freed once, no more scene events, ASR keeps working", func() {
CppStreamFeedJSON = func(s uintptr, pcm []float32, n int32) uintptr {
return pool.cstr(`{"text":"hi ","eou":0,"frame_sec":0.08,` +
`"words":[{"w":"hi","start":0.1,"end":0.3,"conf":0.9}]}`)
}
sceneFeedCalls := 0
begun, freed := liveSceneStubs(func(calls int, s uintptr, isLast int32) uintptr {
sceneFeedCalls++
return 0 // fails every call; only the first should ever be reached
})
in, out, errCh := runLive(p)
in <- liveConfig("")
in <- liveAudio(make([]float32, 10)) // scene feed fails here: warn, free, zero the handle
in <- liveAudio(make([]float32, 10)) // ASR-only: no scene C call at all
close(in)
Expect(<-errCh).NotTo(HaveOccurred())
got := collectLive(out)
// ready, ASR delta (msg 1), ASR delta (msg 2), final: no scene-only
// response ever appears, before or after the failure.
Expect(got).To(HaveLen(4))
Expect(got[0].Ready).To(BeTrue())
Expect(got[1].Delta).To(Equal("hi "))
Expect(got[2].Delta).To(Equal("hi "))
for _, r := range got {
Expect(r.Speakers).To(BeEmpty(), "no speaker events once the scene stream has failed")
Expect(r.Sounds).To(BeEmpty(), "no sound events once the scene stream has failed")
}
Expect(got[3].FinalResult).NotTo(BeNil())
Expect(got[3].FinalResult.Text).To(Equal("hi hi"))
Expect(*begun).To(Equal(1))
Expect(sceneFeedCalls).To(Equal(1), "the second audio message must not retry the broken scene stream")
Expect(*freed).To(Equal(1), "the broken scene stream is freed exactly once, not again at RPC unwind")
})
It("continues without scene events when scene begin fails", func() {
CppSceneOptsDefault = func(o *cSceneOpts) { *o = cSceneOpts{} }
CppSceneStreamBegin = func(asr, diar, tagger uintptr, o *cSceneOpts) uintptr { return 0 }
sceneFeedCalled := false
CppSceneStreamFeedJSON = func(s uintptr, pcm *float32, n int32, isLast int32) uintptr {
sceneFeedCalled = true
return 0
}
CppSceneStreamFree = func(s uintptr) {}
in, out, errCh := runLive(p)
in <- liveConfig("")
in <- liveAudio(make([]float32, 10))
close(in)
Expect(<-errCh).NotTo(HaveOccurred())
got := collectLive(out)
Expect(got).To(HaveLen(2)) // ready, final only: no scene events, no ASR delta this stub sends
Expect(sceneFeedCalled).To(BeFalse(), "no feed call once begin failed")
})
})
var _ = Describe("stripEouMarker", func() {
+28
View File
@@ -90,6 +90,34 @@ func main() {
purego.RegisterLibFunc(&CppStreamFinalizeJSON, lib, "parakeet_capi_stream_finalize_json")
}
// Model roles + diarization/sound (ABI v7-v8): parakeet_capi_model_kind is
// what lets Load tell an ASR/diarization/sound context apart, so it gates
// every other new symbol below (an older libparakeet.so gets none of
// them, and companion model options are rejected in roles.go). Diarization
// itself (diarize_pcm, transcribe_and_diarize_json) predates model_kind
// (ABI v7), so it is probed on its own.
if sym, err := purego.Dlsym(lib, "parakeet_capi_diarize_pcm"); err == nil && sym != 0 {
purego.RegisterLibFunc(&CppDiarizePCM, lib, "parakeet_capi_diarize_pcm")
}
if sym, err := purego.Dlsym(lib, "parakeet_capi_transcribe_and_diarize_json"); err == nil && sym != 0 {
purego.RegisterLibFunc(&CppTranscribeAndDiarizeJSON, lib, "parakeet_capi_transcribe_and_diarize_json")
}
if sym, err := purego.Dlsym(lib, "parakeet_capi_model_kind"); err == nil && sym != 0 {
purego.RegisterLibFunc(&CppModelKind, lib, "parakeet_capi_model_kind")
purego.RegisterLibFunc(&CppNumClasses, lib, "parakeet_capi_num_classes")
purego.RegisterLibFunc(&CppSoundOptsDefault, lib, "parakeet_capi_sound_opts_default")
purego.RegisterLibFunc(&CppSoundStreamBegin, lib, "parakeet_capi_sound_stream_begin")
purego.RegisterLibFunc(&CppSoundStreamFeed, lib, "parakeet_capi_sound_stream_feed")
purego.RegisterLibFunc(&CppSoundStreamDrainScoresJSON, lib, "parakeet_capi_sound_stream_drain_scores_json")
purego.RegisterLibFunc(&CppFreeSoundSegments, lib, "parakeet_capi_free_sound_segments")
purego.RegisterLibFunc(&CppSoundStreamFree, lib, "parakeet_capi_sound_stream_free")
purego.RegisterLibFunc(&CppSceneOptsDefault, lib, "parakeet_capi_scene_opts_default")
purego.RegisterLibFunc(&CppSceneStreamBegin, lib, "parakeet_capi_scene_stream_begin")
purego.RegisterLibFunc(&CppSceneStreamFeedJSON, lib, "parakeet_capi_scene_stream_feed_json")
purego.RegisterLibFunc(&CppSceneStreamLastError, lib, "parakeet_capi_scene_stream_last_error")
purego.RegisterLibFunc(&CppSceneStreamFree, lib, "parakeet_capi_scene_stream_free")
}
fmt.Fprintf(os.Stderr, "[parakeet-cpp] ABI=%d\n", CppAbiVersion())
flag.Parse()
+245
View File
@@ -0,0 +1,245 @@
package main
import (
"errors"
"fmt"
"path/filepath"
"strings"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
"github.com/mudler/xlog"
)
// Model kinds returned by parakeet_capi_model_kind (ABI v8; mirrors the
// PARAKEET_MODEL_KIND_* defines in parakeet_capi.h).
const (
modelKindNone = 0
modelKindASR = 1
modelKindDiarization = 2
modelKindSound = 3
)
// Diarization streaming latency modes (mirrors PARAKEET_DIAR_LATENCY_* in
// parakeet_capi.h). diarLatencyLow is the spec's default when
// diarization_latency: is unset.
const (
diarLatencyModel int32 = 0
diarLatencyLow int32 = 1
diarLatencyVeryLow int32 = 2
diarLatencyUltraLow int32 = 3
)
// modelKindName renders a model kind for error messages.
func modelKindName(kind int32) string {
switch kind {
case modelKindASR:
return "ASR"
case modelKindDiarization:
return "diarization"
case modelKindSound:
return "sound"
default:
return "unknown"
}
}
// optString reads a string model option (key:value form) from ModelOptions,
// returning "" when the key is absent. Same strings.Cut parsing as optInt.
func optString(opts *pb.ModelOptions, key string) string {
for _, o := range opts.GetOptions() {
k, v, ok := strings.Cut(o, ":")
if ok && strings.TrimSpace(k) == key {
return strings.TrimSpace(v)
}
}
return ""
}
// resolveModelPath resolves a companion model option's path against
// modelPath (opts.ModelPath, the LocalAI models root): an absolute p, or an
// empty modelPath, passes through unchanged; anything else is joined onto
// modelPath. Mirrors vibevoice-cpp's resolvePath for tokenizer=/voice=/etc.
func resolveModelPath(modelPath, p string) string {
if p == "" || filepath.IsAbs(p) || modelPath == "" {
return p
}
return filepath.Join(modelPath, p)
}
// parseDiarLatency maps the diarization_latency option value to a
// PARAKEET_DIAR_LATENCY_* mode. "" defaults to "low" (the spec's default);
// any other unrecognized value is a Load error.
func parseDiarLatency(s string) (int32, error) {
switch strings.ToLower(strings.TrimSpace(s)) {
case "":
return diarLatencyLow, nil
case "model":
return diarLatencyModel, nil
case "low":
return diarLatencyLow, nil
case "very_low":
return diarLatencyVeryLow, nil
case "ultra_low":
return diarLatencyUltraLow, nil
default:
return 0, fmt.Errorf("parakeet-cpp: unknown diarization_latency %q (want model|low|very_low|ultra_low)", s)
}
}
// companionSpec is one asr_model:/diarization_model:/sound_model: option: its
// name (for error messages and path resolution), the raw option value, the
// model kind the loaded companion must report, the ParakeetCpp field it is
// assigned to on success, and a getter for that same field's current value
// (used to reject a companion whose role the primary already occupies).
type companionSpec struct {
optName string
value string
wantKind int32
assign func(*ParakeetCpp, uintptr)
current func(*ParakeetCpp) uintptr
}
// indefiniteArticle returns "an" for a word starting with a vowel sound and
// "a" otherwise, for grammatical error messages built from modelKindName.
func indefiniteArticle(word string) string {
if len(word) == 0 {
return "a"
}
switch word[0] {
case 'A', 'E', 'I', 'O', 'U', 'a', 'e', 'i', 'o', 'u':
return "an"
default:
return "a"
}
}
// loadRoles loads opts.ModelFile as the primary parakeet_ctx, classifies it
// with parakeet_capi_model_kind (ABI v8) into ctxPtr/diarCtx/tagCtx, and
// loads any companion models named in Options[] (asr_model:,
// diarization_model:, sound_model:; paths resolved against opts.ModelPath).
// It also parses diarization_latency: into p.diarLatency.
//
// Against an older libparakeet.so (CppModelKind == nil) the primary is
// treated as ASR — the pre-v8 behavior — and companion model options are
// rejected outright, since there is no way to verify what they loaded.
//
// On any failure every context this call opened (primary and any companions
// loaded before the failure) is freed before the error is returned.
func (p *ParakeetCpp) loadRoles(opts *pb.ModelOptions) error {
diarModelOpt := optString(opts, "diarization_model")
asrModelOpt := optString(opts, "asr_model")
soundModelOpt := optString(opts, "sound_model")
hasCompanionOpts := diarModelOpt != "" || asrModelOpt != "" || soundModelOpt != ""
if hasCompanionOpts && CppModelKind == nil {
return errors.New("parakeet-cpp: asr_model/diarization_model/sound_model options need " +
"parakeet_capi_model_kind (ABI v8) to verify what they load; the loaded libparakeet.so " +
"is too old to report companion model roles")
}
latency, err := parseDiarLatency(optString(opts, "diarization_latency"))
if err != nil {
return err
}
primary := CppLoad(opts.ModelFile)
if primary == 0 {
// No ctx to ask for last_error (the C-API's last-error buffer lives on
// the ctx that was never returned). Surface the path so the operator
// at least knows which load failed.
return fmt.Errorf("parakeet-cpp: parakeet_capi_load failed for %q", opts.ModelFile)
}
loaded := []uintptr{primary}
// freeLoaded undoes everything loadRoles opened this call: every context
// it freed AND every ParakeetCpp field it may have assigned (the primary
// lands in one of ctxPtr/diarCtx/tagCtx before the companion loop runs,
// and an earlier companion's spec.assign runs before a later one fails).
// Leaving a role field pointing at a freed ctx would double-free it on a
// later Free() call.
freeLoaded := func() {
for _, c := range loaded {
CppFree(c)
}
p.ctxPtr, p.diarCtx, p.tagCtx = 0, 0, 0
p.companions = nil
}
primaryKind := int32(modelKindASR) // old-library default: today's behavior
if CppModelKind != nil {
primaryKind = CppModelKind(primary)
if primaryKind == modelKindNone {
xlog.Warn("parakeet-cpp: parakeet_capi_model_kind reported PARAKEET_MODEL_KIND_NONE " +
"for a successfully loaded primary; treating it as an ASR model")
}
}
switch primaryKind {
case modelKindDiarization:
p.diarCtx = primary
case modelKindSound:
p.tagCtx = primary
default:
p.ctxPtr = primary
}
specs := []companionSpec{
{"diarization_model", diarModelOpt, modelKindDiarization,
func(pp *ParakeetCpp, c uintptr) { pp.diarCtx = c },
func(pp *ParakeetCpp) uintptr { return pp.diarCtx }},
{"asr_model", asrModelOpt, modelKindASR,
func(pp *ParakeetCpp, c uintptr) { pp.ctxPtr = c },
func(pp *ParakeetCpp) uintptr { return pp.ctxPtr }},
{"sound_model", soundModelOpt, modelKindSound,
func(pp *ParakeetCpp, c uintptr) { pp.tagCtx = c },
func(pp *ParakeetCpp) uintptr { return pp.tagCtx }},
}
for _, spec := range specs {
if spec.value == "" {
continue
}
// A companion whose role the primary already occupies (e.g. asr_model:
// on an already-ASR primary) would overwrite that role field below,
// leaking the primary ctx: Free() only walks ctxPtr/diarCtx/tagCtx, so
// the overwritten pointer is never freed. Reject it before loading.
if spec.current(p) != 0 {
freeLoaded()
return fmt.Errorf("parakeet-cpp: %s is not allowed on %s %s model",
spec.optName, indefiniteArticle(modelKindName(spec.wantKind)), modelKindName(spec.wantKind))
}
resolved := resolveModelPath(opts.ModelPath, spec.value)
cctx := CppLoad(resolved)
if cctx == 0 {
freeLoaded()
return fmt.Errorf("parakeet-cpp: failed to load %s %q", spec.optName, resolved)
}
loaded = append(loaded, cctx)
if gotKind := CppModelKind(cctx); gotKind != spec.wantKind {
freeLoaded()
return fmt.Errorf("parakeet-cpp: %s %q is a %s model, expected a %s model",
spec.optName, resolved, modelKindName(gotKind), modelKindName(spec.wantKind))
}
spec.assign(p, cctx)
p.companions = append(p.companions, cctx)
}
p.diarLatency = latency
return nil
}
// notASRError reports why AudioTranscription (and the streaming/live RPCs)
// cannot run when p.ctxPtr == 0: a loaded diarization or sound primary with
// no asr_model companion, named explicitly so the caller knows to use the
// right RPC instead of a generic "model not loaded". Returns nil when
// neither role is loaded (genuinely no model), leaving the caller to report
// the ordinary ModelNotLoaded error.
func (p *ParakeetCpp) notASRError() error {
switch {
case p.diarCtx != 0:
return errors.New("parakeet-cpp: loaded model is a diarization model, not ASR " +
"(use Diarize, or load with an asr_model: companion)")
case p.tagCtx != 0:
return errors.New("parakeet-cpp: loaded model is a sound model, not ASR " +
"(use SoundDetection)")
default:
return nil
}
}
+417
View File
@@ -0,0 +1,417 @@
package main
import (
"context"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
// The role-loading specs drive Load/Free entirely against stubbed
// CppLoad/CppModelKind/CppFree (the same seam live_test.go and
// batcher_test.go use), so they run without libparakeet.so.
// fakeLib is a tiny in-memory stand-in for libparakeet.so: paths registered
// via withModel resolve to a fresh ctx handle of the given model kind on
// CppLoad, any other path fails the load, CppModelKind reads the kind back
// by ctx, and CppFree records every ctx it was asked to free, in order.
type fakeLib struct {
kinds map[string]int32
ctxKind map[uintptr]int32
next uintptr
loadedPaths []string
freed []uintptr
}
func newFakeLib() *fakeLib {
return &fakeLib{kinds: map[string]int32{}, ctxKind: map[uintptr]int32{}, next: 1}
}
func (f *fakeLib) withModel(path string, kind int32) *fakeLib {
f.kinds[path] = kind
return f
}
// install swaps CppLoad/CppModelKind/CppFree for fakes backed by f and
// returns a restore func for AfterEach (mirrors live_test.go's liveStubs).
func (f *fakeLib) install() (restore func()) {
savedLoad, savedKind, savedFree := CppLoad, CppModelKind, CppFree
CppLoad = func(path string) uintptr {
f.loadedPaths = append(f.loadedPaths, path)
kind, ok := f.kinds[path]
if !ok {
return 0
}
ctx := f.next
f.next++
f.ctxKind[ctx] = kind
return ctx
}
CppModelKind = func(ctx uintptr) int32 { return f.ctxKind[ctx] }
CppFree = func(ctx uintptr) { f.freed = append(f.freed, ctx) }
return func() {
CppLoad, CppModelKind, CppFree = savedLoad, savedKind, savedFree
}
}
var _ = Describe("model roles (stubbed C API)", func() {
var restore func()
AfterEach(func() {
if restore != nil {
restore()
restore = nil
}
})
It("loads an ASR primary with no options", func() {
f := newFakeLib().withModel("asr.gguf", modelKindASR)
restore = f.install()
p := &ParakeetCpp{}
Expect(p.Load(&pb.ModelOptions{ModelFile: "asr.gguf"})).To(Succeed())
Expect(p.ctxPtr).ToNot(BeZero())
Expect(p.diarCtx).To(BeZero())
Expect(p.tagCtx).To(BeZero())
})
It("loads a diarization primary and rejects AudioTranscription without any further C call", func() {
f := newFakeLib().withModel("diar.gguf", modelKindDiarization)
restore = f.install()
p := &ParakeetCpp{}
Expect(p.Load(&pb.ModelOptions{ModelFile: "diar.gguf"})).To(Succeed())
Expect(p.diarCtx).ToNot(BeZero())
Expect(p.ctxPtr).To(BeZero())
// CppTranscribePathJSON/CppTranscribePcmBatchJSON are left nil (the
// zero value): if AudioTranscription tried to call either, this would
// panic instead of returning cleanly, so a clean typed error here also
// proves no C call was made.
_, err := p.AudioTranscription(context.Background(), &pb.TranscriptRequest{Dst: "x.wav"})
Expect(err).To(MatchError(ContainSubstring("diarization model")))
})
It("loads a sound primary", func() {
f := newFakeLib().withModel("sound.gguf", modelKindSound)
restore = f.install()
p := &ParakeetCpp{}
Expect(p.Load(&pb.ModelOptions{ModelFile: "sound.gguf"})).To(Succeed())
Expect(p.tagCtx).ToNot(BeZero())
Expect(p.ctxPtr).To(BeZero())
Expect(p.diarCtx).To(BeZero())
})
It("loads an ASR primary plus diarization and sound companions, and Free releases all three", func() {
f := newFakeLib().
withModel("asr.gguf", modelKindASR).
withModel("/models/x.gguf", modelKindDiarization).
withModel("/abs/y.gguf", modelKindSound)
restore = f.install()
p := &ParakeetCpp{}
err := p.Load(&pb.ModelOptions{
ModelFile: "asr.gguf",
ModelPath: "/models",
Options: []string{"diarization_model:x.gguf", "sound_model:/abs/y.gguf"},
})
Expect(err).ToNot(HaveOccurred())
Expect(f.loadedPaths).To(Equal([]string{"asr.gguf", "/models/x.gguf", "/abs/y.gguf"}),
"diarization_model resolves against ModelPath, sound_model's absolute path passes through")
Expect(p.ctxPtr).ToNot(BeZero())
Expect(p.diarCtx).ToNot(BeZero())
Expect(p.tagCtx).ToNot(BeZero())
Expect(p.companions).To(HaveLen(2))
asrCtx, diarCtx, tagCtx := p.ctxPtr, p.diarCtx, p.tagCtx
Expect(p.Free()).To(Succeed())
Expect(f.freed).To(ConsistOf(asrCtx, diarCtx, tagCtx))
Expect(p.ctxPtr).To(BeZero())
Expect(p.diarCtx).To(BeZero())
Expect(p.tagCtx).To(BeZero())
})
It("loads a diarization primary plus an asr_model companion into ctxPtr", func() {
f := newFakeLib().
withModel("diar.gguf", modelKindDiarization).
withModel("/models/z.gguf", modelKindASR)
restore = f.install()
p := &ParakeetCpp{}
err := p.Load(&pb.ModelOptions{
ModelFile: "diar.gguf",
ModelPath: "/models",
Options: []string{"asr_model:z.gguf"},
})
Expect(err).ToNot(HaveOccurred())
Expect(p.diarCtx).ToNot(BeZero())
Expect(p.ctxPtr).ToNot(BeZero())
Expect(p.ctxPtr).ToNot(Equal(p.diarCtx))
})
It("fails a companion of the wrong kind and frees every ctx it opened", func() {
f := newFakeLib().
withModel("asr.gguf", modelKindASR).
withModel("/models/wrong.gguf", modelKindDiarization)
restore = f.install()
p := &ParakeetCpp{}
err := p.Load(&pb.ModelOptions{
ModelFile: "asr.gguf",
ModelPath: "/models",
Options: []string{"sound_model:wrong.gguf"},
})
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(ContainSubstring("sound_model"))
Expect(err.Error()).To(ContainSubstring("diarization"))
Expect(f.freed).To(HaveLen(2), "the primary and the wrong-kind companion must both be freed")
// Every role field the failed load may have assigned (the primary
// lands in ctxPtr before the companion loop runs) must be reset, or a
// later Free() would double-free an already-freed context.
Expect(p.ctxPtr).To(BeZero())
Expect(p.diarCtx).To(BeZero())
Expect(p.tagCtx).To(BeZero())
Expect(p.companions).To(BeEmpty())
freedBeforeFree := len(f.freed)
Expect(p.Free()).To(Succeed())
Expect(f.freed).To(HaveLen(freedBeforeFree), "Free after a failed Load must not free anything again")
})
It("rejects an asr_model companion on an already-ASR primary and frees everything it opened", func() {
f := newFakeLib().
withModel("asr.gguf", modelKindASR).
withModel("/models/other.gguf", modelKindASR)
restore = f.install()
p := &ParakeetCpp{}
err := p.Load(&pb.ModelOptions{
ModelFile: "asr.gguf",
ModelPath: "/models",
Options: []string{"asr_model:other.gguf"},
})
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(Equal(`parakeet-cpp: asr_model is not allowed on an ASR model`))
Expect(f.loadedPaths).To(Equal([]string{"asr.gguf"}))
Expect(f.freed).To(HaveLen(1), "the primary must be freed too, or it leaks")
Expect(p.ctxPtr).To(BeZero())
Expect(p.diarCtx).To(BeZero())
Expect(p.tagCtx).To(BeZero())
Expect(p.companions).To(BeEmpty())
})
It("rejects a diarization_model companion on an already-diarization primary", func() {
f := newFakeLib().
withModel("diar.gguf", modelKindDiarization).
withModel("/models/other.gguf", modelKindDiarization)
restore = f.install()
p := &ParakeetCpp{}
err := p.Load(&pb.ModelOptions{
ModelFile: "diar.gguf",
ModelPath: "/models",
Options: []string{"diarization_model:other.gguf"},
})
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(Equal(`parakeet-cpp: diarization_model is not allowed on a diarization model`))
Expect(f.loadedPaths).To(Equal([]string{"diar.gguf"}))
Expect(p.diarCtx).To(BeZero())
})
It("rejects a sound_model companion on an already-sound (CED) primary", func() {
f := newFakeLib().
withModel("sound.gguf", modelKindSound).
withModel("/models/other.gguf", modelKindSound)
restore = f.install()
p := &ParakeetCpp{}
err := p.Load(&pb.ModelOptions{
ModelFile: "sound.gguf",
ModelPath: "/models",
Options: []string{"sound_model:other.gguf"},
})
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(Equal(`parakeet-cpp: sound_model is not allowed on a sound model`))
Expect(f.loadedPaths).To(Equal([]string{"sound.gguf"}))
Expect(p.tagCtx).To(BeZero())
})
It("does not overwrite ctxPtr (and so does not leak the primary) when an asr_model companion "+
"duplicates an ASR primary loaded alongside other companions", func() {
// Regression for the leak this whole check exists to close: before
// the fix, spec.assign(p, cctx) overwrote p.ctxPtr with the
// companion's ctx, and Free() (which only walks
// ctxPtr/diarCtx/tagCtx) never saw the original primary again.
f := newFakeLib().
withModel("asr.gguf", modelKindASR).
withModel("/models/diar.gguf", modelKindDiarization).
withModel("/models/dup.gguf", modelKindASR)
restore = f.install()
p := &ParakeetCpp{}
err := p.Load(&pb.ModelOptions{
ModelFile: "asr.gguf",
ModelPath: "/models",
// diarization_model loads first (declared first in loadRoles'
// specs) and succeeds; asr_model then collides with the primary.
Options: []string{"diarization_model:diar.gguf", "asr_model:dup.gguf"},
})
Expect(err).To(HaveOccurred())
Expect(f.freed).To(HaveLen(2), "the primary and the already-loaded diarization companion")
Expect(f.loadedPaths).To(Equal([]string{"asr.gguf", "/models/diar.gguf"}),
"dup.gguf must never be loaded: the role check runs before CppLoad")
Expect(p.ctxPtr).To(BeZero())
Expect(p.diarCtx).To(BeZero())
Expect(p.companions).To(BeEmpty())
})
It("treats a primary reporting PARAKEET_MODEL_KIND_NONE as ASR", func() {
f := newFakeLib().withModel("model.gguf", modelKindNone)
restore = f.install()
p := &ParakeetCpp{}
Expect(p.Load(&pb.ModelOptions{ModelFile: "model.gguf"})).To(Succeed())
Expect(p.ctxPtr).ToNot(BeZero())
Expect(p.diarCtx).To(BeZero())
Expect(p.tagCtx).To(BeZero())
})
It("parses diarization_latency:very_low", func() {
f := newFakeLib().withModel("diar.gguf", modelKindDiarization)
restore = f.install()
p := &ParakeetCpp{}
err := p.Load(&pb.ModelOptions{
ModelFile: "diar.gguf",
Options: []string{"diarization_latency:very_low"},
})
Expect(err).ToNot(HaveOccurred())
Expect(p.diarLatency).To(Equal(int32(2)))
})
It("defaults diarization_latency to low (1) when unset", func() {
f := newFakeLib().withModel("diar.gguf", modelKindDiarization)
restore = f.install()
p := &ParakeetCpp{}
Expect(p.Load(&pb.ModelOptions{ModelFile: "diar.gguf"})).To(Succeed())
Expect(p.diarLatency).To(Equal(int32(1)))
})
It("rejects an invalid diarization_latency before any C call", func() {
f := newFakeLib().withModel("diar.gguf", modelKindDiarization)
restore = f.install()
p := &ParakeetCpp{}
err := p.Load(&pb.ModelOptions{
ModelFile: "diar.gguf",
Options: []string{"diarization_latency:bogus"},
})
Expect(err).To(HaveOccurred())
Expect(f.loadedPaths).To(BeEmpty())
})
Context("old library (no parakeet_capi_model_kind)", func() {
It("treats the primary as ASR, matching pre-v8 behavior", func() {
f := newFakeLib().withModel("model.gguf", modelKindDiarization) // kind is never consulted
restore = f.install()
CppModelKind = nil
p := &ParakeetCpp{}
Expect(p.Load(&pb.ModelOptions{ModelFile: "model.gguf"})).To(Succeed())
Expect(p.ctxPtr).ToNot(BeZero())
Expect(p.diarCtx).To(BeZero())
Expect(p.tagCtx).To(BeZero())
})
It("rejects companion model options with an error naming the library as too old", func() {
f := newFakeLib().withModel("asr.gguf", modelKindASR)
restore = f.install()
CppModelKind = nil
p := &ParakeetCpp{}
err := p.Load(&pb.ModelOptions{
ModelFile: "asr.gguf",
Options: []string{"diarization_model:diar.gguf"},
})
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(ContainSubstring("too old"))
Expect(f.loadedPaths).To(BeEmpty(), "rejected before any C call")
})
})
Context("batcher gating", func() {
It("does not start the batcher for a non-ASR primary with no asr_model companion", func() {
f := newFakeLib().withModel("diar.gguf", modelKindDiarization)
restore = f.install()
savedBatch := CppTranscribePcmBatchJSON
defer func() { CppTranscribePcmBatchJSON = savedBatch }()
CppTranscribePcmBatchJSON = func(uintptr, []float32, []int32, int32, int32, int32) uintptr { return 0 }
p := &ParakeetCpp{}
Expect(p.Load(&pb.ModelOptions{ModelFile: "diar.gguf"})).To(Succeed())
Expect(p.bat).To(BeNil())
})
It("starts the batcher for an ASR ctxPtr", func() {
f := newFakeLib().withModel("asr.gguf", modelKindASR)
restore = f.install()
savedBatch := CppTranscribePcmBatchJSON
defer func() { CppTranscribePcmBatchJSON = savedBatch }()
CppTranscribePcmBatchJSON = func(uintptr, []float32, []int32, int32, int32, int32) uintptr { return 0 }
p := &ParakeetCpp{}
Expect(p.Load(&pb.ModelOptions{ModelFile: "asr.gguf"})).To(Succeed())
Expect(p.bat).ToNot(BeNil())
Expect(p.Free()).To(Succeed()) // stop the dispatcher goroutine
})
})
})
var _ = Describe("resolveModelPath", func() {
It("keeps an absolute path unchanged", func() {
Expect(resolveModelPath("/models", "/abs/y.gguf")).To(Equal("/abs/y.gguf"))
})
It("joins a relative path onto modelPath", func() {
Expect(resolveModelPath("/models", "x.gguf")).To(Equal("/models/x.gguf"))
})
It("passes a relative path through when modelPath is empty", func() {
Expect(resolveModelPath("", "x.gguf")).To(Equal("x.gguf"))
})
})
var _ = Describe("parseDiarLatency", func() {
It("defaults to low (1) for an empty value", func() {
v, err := parseDiarLatency("")
Expect(err).ToNot(HaveOccurred())
Expect(v).To(Equal(int32(1)))
})
It("maps every named mode", func() {
for s, want := range map[string]int32{
"model": 0, "low": 1, "very_low": 2, "ultra_low": 3,
} {
v, err := parseDiarLatency(s)
Expect(err).ToNot(HaveOccurred())
Expect(v).To(Equal(want), "mode %q", s)
}
})
It("rejects an unknown value", func() {
_, err := parseDiarLatency("bogus")
Expect(err).To(HaveOccurred())
})
})
+258
View File
@@ -0,0 +1,258 @@
package main
import (
"context"
"encoding/json"
"fmt"
"time"
"github.com/mudler/LocalAI/pkg/grpc/grpcerrors"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
"github.com/mudler/xlog"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)
// sceneSpeakerJSON mirrors one element of a scene feed document's "speakers"
// array: {"speaker":0,"start":0.0,"end":0.6}.
type sceneSpeakerJSON struct {
Speaker int `json:"speaker"`
Start float64 `json:"start"`
End float64 `json:"end"`
}
// sceneSoundJSON mirrors one element of a scene feed document's "sounds"
// array: {"index":99,"label":"Chicken, rooster","start":24.0,"end":30.0,"peak":0.86}.
type sceneSoundJSON struct {
Index int `json:"index"`
Label string `json:"label"`
Start float64 `json:"start"`
End float64 `json:"end"`
Peak float32 `json:"peak"`
}
// sceneFeedJSON mirrors the subset of the document
// parakeet_capi_scene_stream_feed_json returns (docs/sound.md) that the live
// path consumes: the closed "speakers" and "sounds" arrays. "t",
// "utterances", "words" and "active" belong to an offline scene/SAS
// consumer, not the live path, and are not decoded here.
type sceneFeedJSON struct {
Speakers []sceneSpeakerJSON `json:"speakers"`
Sounds []sceneSoundJSON `json:"sounds"`
}
// sceneWanted reports whether AudioTranscriptionLive should run a companion
// scene stream beside the ASR streaming session: at least one of the
// diarization/sound companions must be loaded, and the scene C-API symbols
// must be present. In practice the nil checks are defensive rather than
// live: loadRoles only ever sets diarCtx/tagCtx when parakeet_capi_model_kind
// (ABI v8) is present, and main.go registers every scene symbol in the same
// Dlsym-gated block as model_kind, so a companion being loaded already
// guarantees the scene symbols exist.
func (p *ParakeetCpp) sceneWanted() bool {
return (p.diarCtx != 0 || p.tagCtx != 0) &&
CppSceneOptsDefault != nil && CppSceneStreamBegin != nil &&
CppSceneStreamFeedJSON != nil && CppSceneStreamFree != nil
}
// sceneStreamHandle bundles the C scene_stream pointer with the diar/tag
// contexts it was begun with. sceneFeed re-checks those against p.diarCtx/
// p.tagCtx under engineMu before every call, so a Free() racing between the
// begin and a later feed (freeing the very contexts the stream borrows) is
// caught instead of handed to the C side — mirroring streamFeedDoc's re-check
// of p.ctxPtr (see the "Per-C-call engine serialization" comment in
// goparakeetcpp.go). The zero value (s == 0) means "no scene stream".
type sceneStreamHandle struct {
s uintptr
diar uintptr
tag uintptr
}
// sceneBegin opens a no-ASR scene stream (diarization and/or sound events
// only; the live path's own ASR session already covers transcription) under
// engineMu. Call only when sceneWanted() is true. Refuses to begin with both
// contexts 0 (defensive: sceneWanted() already guards this). A zero handle
// means the C call itself failed; the caller logs a warning and continues
// the live session without speaker/sound events.
func (p *ParakeetCpp) sceneBegin() sceneStreamHandle {
p.engineMu.Lock()
defer p.engineMu.Unlock()
diar, tag := p.diarCtx, p.tagCtx
if diar == 0 && tag == 0 {
return sceneStreamHandle{}
}
var opts cSceneOpts
CppSceneOptsDefault(&opts)
opts.DiarLatency = p.diarLatency
// The live scene path never drains sound scores (unlike the offline
// SoundDetection RPC, see sound.go), so the default top_k of 5 would
// leave the C side's per-window score queue growing for the session's
// whole lifetime. 0 disables per-class score retention; sound EVENTS
// (onset/offset, what the live path actually consumes) are unaffected.
opts.Sound.TopK = 0
s := CppSceneStreamBegin(0, diar, tag, &opts)
if s == 0 {
return sceneStreamHandle{}
}
return sceneStreamHandle{s: s, diar: diar, tag: tag}
}
// sceneFree releases a scene stream opened by sceneBegin. A zero handle
// (scene events disabled or never began) is a no-op. Safe to call even after
// the contexts the stream borrowed have been freed: parakeet_scene_stream's
// destructor only releases its own buffers and never dereferences the
// borrowed asr/diar/tagger pointers (verified against
// parakeet.cpp's parakeet_capi_scene_stream_free / SceneStream::~SceneStream
// / DiarPcmStream::~DiarPcmStream, all `= default`), unlike a feed call.
func (p *ParakeetCpp) sceneFree(h sceneStreamHandle) {
if h.s == 0 {
return
}
p.engineMu.Lock()
defer p.engineMu.Unlock()
CppSceneStreamFree(h.s)
}
// sceneFeed runs one scene-stream feed (or the is_last flush) under
// engineMu and returns the parsed document. Before touching the C side it
// re-checks that p.diarCtx/p.tagCtx still match what the stream was begun
// with: Free() can run between the caller's ASR feed and this call (both
// take engineMu individually, never for a session's lifetime, so nothing
// blocks a concurrent Free()) and free the very model the stream borrows.
// A mismatch returns ModelNotLoaded without making the C call; last_error is
// otherwise stream-scoped (parakeet_capi_scene_stream_last_error), read
// under the same lock as the failing call.
func (p *ParakeetCpp) sceneFeed(h sceneStreamHandle, pcm []float32, isLast bool) (sceneFeedJSON, error) {
p.engineMu.Lock()
defer p.engineMu.Unlock()
if p.diarCtx != h.diar || p.tagCtx != h.tag {
return sceneFeedJSON{}, grpcerrors.ModelNotLoaded("parakeet-cpp")
}
var last int32
if isLast {
last = 1
}
var ptr *float32
if len(pcm) > 0 {
ptr = &pcm[0]
}
ret := CppSceneStreamFeedJSON(h.s, ptr, int32(len(pcm)), last)
if ret == 0 {
msg := ""
if CppSceneStreamLastError != nil {
msg = CppSceneStreamLastError(h.s)
}
if msg == "" {
msg = "unknown error"
}
return sceneFeedJSON{}, fmt.Errorf("parakeet-cpp: scene stream feed failed: %s", msg)
}
raw := goStringFromCPtr(ret)
CppFreeString(ret)
var doc sceneFeedJSON
if err := json.Unmarshal([]byte(raw), &doc); err != nil {
return sceneFeedJSON{}, fmt.Errorf("parakeet-cpp: decode scene json: %w", err)
}
return doc, nil
}
// feedSlicesScene mirrors driver.go's feedSlices but also feeds the same pcm
// slice to an optional companion scene stream right after each ASR slice, so
// the live path's speaker/sound events stay time-aligned with the ASR decode
// increments. scene.s == 0 disables scene feeding for this call (no
// companions, or a previous scene feed already disabled it this session).
//
// The ASR result is emitted immediately after the ASR feed — the same
// response contents/timing a no-companion session would produce — before the
// scene feed for that slice runs, so a companion model never adds scene
// compute latency in front of the ASR delta/<EOU> that drives realtime turn
// detection. Any closed speakers/sounds from the scene feed are emitted
// afterward as their own response, so a slice with both produces two
// responses, ASR first.
//
// A scene feed failure degrades gracefully rather than aborting live
// transcription over a secondary feature: it frees the broken stream, warns
// once, and zeroes the handle so the caller carries the ASR-only session
// forward. Returns the (possibly now-zeroed) scene handle plus the
// cumulative ASR and scene wall time this call spent in feedChunk/sceneFeed,
// for the caller's lag log line.
func (p *ParakeetCpp) feedSlicesScene(ctx context.Context, stream uintptr, scene sceneStreamHandle, pcm []float32, onFeed func(streamFeedResult, sceneFeedJSON) error) (sceneStreamHandle, time.Duration, time.Duration, error) {
var asrWall, sceneWall time.Duration
for off := 0; off < len(pcm); off += streamChunkSamples {
if ctx != nil {
if err := ctx.Err(); err != nil {
return scene, asrWall, sceneWall, status.Error(codes.Canceled, "transcription cancelled")
}
}
end := min(off+streamChunkSamples, len(pcm))
chunk := pcm[off:end]
asrStart := time.Now()
res, err := p.feedChunk(stream, chunk, false)
asrWall += time.Since(asrStart)
if err != nil {
return scene, asrWall, sceneWall, err
}
if err := onFeed(res, sceneFeedJSON{}); err != nil {
return scene, asrWall, sceneWall, err
}
if scene.s == 0 {
continue
}
sceneStart := time.Now()
sceneDoc, serr := p.sceneFeed(scene, chunk, false)
sceneWall += time.Since(sceneStart)
if serr != nil {
xlog.Warn("parakeet-cpp: live scene feed failed; disabling speaker/sound events for this session",
"err", serr)
p.sceneFree(scene)
scene = sceneStreamHandle{}
continue
}
if err := onFeed(streamFeedResult{}, sceneDoc); err != nil {
return scene, asrWall, sceneWall, err
}
}
return scene, asrWall, sceneWall, nil
}
// liveSpeakersToProto maps a scene feed document's closed "speakers" into
// TranscriptLiveResponse.speakers (stream-relative nanoseconds). Reuses
// diarize.go's speakerLabel so the live path renders speaker indices the
// same way the offline Diarize RPC does.
func liveSpeakersToProto(speakers []sceneSpeakerJSON) []*pb.LiveSpeakerSegment {
if len(speakers) == 0 {
return nil
}
out := make([]*pb.LiveSpeakerSegment, len(speakers))
for i, s := range speakers {
out[i] = &pb.LiveSpeakerSegment{
Speaker: speakerLabel(s.Speaker),
Start: secondsToNanos(s.Start),
End: secondsToNanos(s.End),
}
}
return out
}
// liveSoundsToProto maps a scene feed document's closed "sounds" into
// TranscriptLiveResponse.sounds (stream-relative nanoseconds).
func liveSoundsToProto(sounds []sceneSoundJSON) []*pb.LiveSoundEvent {
if len(sounds) == 0 {
return nil
}
out := make([]*pb.LiveSoundEvent, len(sounds))
for i, s := range sounds {
out[i] = &pb.LiveSoundEvent{
Label: s.Label,
Index: int32(s.Index),
Peak: s.Peak,
Start: secondsToNanos(s.Start),
End: secondsToNanos(s.End),
}
}
return out
}
+42
View File
@@ -0,0 +1,42 @@
package main
import (
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
// The sceneBegin spec drives it entirely against stubbed
// CppSceneOptsDefault/CppSceneStreamBegin (the same seam live_test.go uses
// for the full live-session scene specs), so it runs without libparakeet.so.
var _ = Describe("ParakeetCpp.sceneBegin", func() {
It("forces Sound.TopK to 0 and carries p.diarLatency into the begin opts", func() {
savedOptsDefault := CppSceneOptsDefault
savedBegin := CppSceneStreamBegin
defer func() {
CppSceneOptsDefault = savedOptsDefault
CppSceneStreamBegin = savedBegin
}()
// parakeet_capi_scene_opts_default's real default is top_k = 5 (a
// sound-window score history); simulate that here so the test proves
// sceneBegin overrides it rather than merely never setting it.
CppSceneOptsDefault = func(o *cSceneOpts) {
*o = cSceneOpts{Sound: cSoundOpts{TopK: 5}}
}
var gotOpts cSceneOpts
CppSceneStreamBegin = func(asr, diar, tagger uintptr, o *cSceneOpts) uintptr {
gotOpts = *o
return 1
}
p := &ParakeetCpp{diarCtx: 42, diarLatency: diarLatencyVeryLow}
h := p.sceneBegin()
Expect(h.s).ToNot(BeZero())
Expect(gotOpts.Sound.TopK).To(Equal(int32(0)),
"the live scene path never drains sound scores (see sound.go's SoundDetection, "+
"which does); a nonzero top_k leaves the C side's per-window score queue "+
"growing for the session's lifetime")
Expect(gotOpts.DiarLatency).To(Equal(diarLatencyVeryLow))
})
})
+261
View File
@@ -0,0 +1,261 @@
package main
import (
"context"
"encoding/json"
"fmt"
"sort"
"github.com/mudler/LocalAI/pkg/grpc/grpcerrors"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)
// soundFeedChunkSamples is how much 16 kHz mono PCM soundStreamScores hands
// to sound_stream_feed per call (10 s), matching the window/hop it asks for
// below. The clip is fed in these pieces with is_last set on the final one,
// mirroring the streaming ASR path's chunked feed.
const soundFeedChunkSamples = 10 * 16000
// soundTagJSON mirrors one element of a soundWindowJSON's "tags" array.
type soundTagJSON struct {
Index int `json:"index"`
Label string `json:"label"`
Score float32 `json:"score"`
}
// soundWindowJSON mirrors one element of the array
// parakeet_capi_sound_stream_drain_scores_json returns:
//
// [{"start":0.0,"end":10.0,"tags":[{"index":0,"label":"Speech","score":0.93}, ...]}]
type soundWindowJSON struct {
Start float64 `json:"start"`
End float64 `json:"end"`
Tags []soundTagJSON `json:"tags"`
}
// classAvg is one class's score averaged across the drained windows, plus
// the label the tagger reported for it.
type classAvg struct {
Index int
Label string
Score float32
}
// SoundDetection runs the loaded CED model (p.tagCtx) over the clip at
// req.Src through a one-shot sound stream (window 10 s, hop 10 s, top_k set
// to the tagger's full class count so every window's drain carries a score
// for every class), averages each class's score across the drained windows,
// sorts descending, applies req.Threshold, then req.TopK (0 = all classes).
func (p *ParakeetCpp) SoundDetection(ctx context.Context, req *pb.SoundDetectionRequest) (*pb.SoundDetectionResponse, error) {
if p.tagCtx == 0 {
return nil, status.Error(codes.FailedPrecondition,
"parakeet-cpp: model is not a sound (CED) model")
}
if CppSoundStreamBegin == nil || CppSoundStreamFeed == nil || CppSoundStreamDrainScoresJSON == nil ||
CppSoundStreamFree == nil || CppSoundOptsDefault == nil || CppNumClasses == nil {
return nil, status.Error(codes.Unimplemented,
"parakeet-cpp: loaded libparakeet.so has no sound-event detection support "+
"(parakeet_capi_sound_stream_* missing)")
}
if req.GetSrc() == "" {
return nil, status.Error(codes.InvalidArgument,
"parakeet-cpp: SoundDetectionRequest.src (audio path) is required")
}
pcm, _, err := decodeWavMono16k(req.GetSrc())
if err != nil {
return nil, status.Errorf(codes.InvalidArgument, "parakeet-cpp: decode audio: %s", err)
}
windows, nClasses, err := p.soundStreamScores(ctx, pcm)
if err != nil {
return nil, err
}
avgs := averageWindowScores(windows, nClasses)
sortSoundDetectionsDesc(avgs)
avgs = filterSoundDetections(avgs, req.GetThreshold(), req.GetTopK())
resp := &pb.SoundDetectionResponse{Detections: make([]*pb.SoundClass, 0, len(avgs))}
for _, a := range avgs {
resp.Detections = append(resp.Detections, &pb.SoundClass{
Label: a.Label,
Score: a.Score,
Index: int32(a.Index),
})
}
return resp, nil
}
// soundStreamScores runs pcm through a fresh sound stream and returns the
// drained per-window scores plus the tagger's class count. The C calls run
// under engineMu (see soundStreamDrain); JSON decoding happens after the
// lock is released.
func (p *ParakeetCpp) soundStreamScores(ctx context.Context, pcm []float32) ([]soundWindowJSON, int, error) {
doc, nClasses, err := p.soundStreamDrain(ctx, pcm)
if err != nil {
return nil, nClasses, err
}
var windows []soundWindowJSON
if err := json.Unmarshal([]byte(doc), &windows); err != nil {
return nil, nClasses, fmt.Errorf("parakeet-cpp: decode sound scores json: %w", err)
}
return windows, nClasses, nil
}
// soundStreamDrain runs pcm through a fresh sound stream and returns the
// raw JSON document parakeet_capi_sound_stream_drain_scores_json drained,
// plus the tagger's class count. Every C call (opts default, begin, feed,
// free, drain) runs under engineMu; the stream is freed (deferred right
// after a successful begin) even when a later feed or drain call fails, or
// ctx is cancelled mid-feed. Each feed's returned segments array is freed
// with parakeet_capi_free_sound_segments even though SoundDetection has no
// use for the segments themselves (it only reads the drained window
// scores). ctx.Err() is checked before each feed slice, mirroring
// driver.go's feedSlices, so a long clip can be cancelled mid-feed; the
// caller decodes the returned JSON outside the lock.
func (p *ParakeetCpp) soundStreamDrain(ctx context.Context, pcm []float32) (string, int, error) {
p.engineMu.Lock()
defer p.engineMu.Unlock()
// SoundDetection's own p.tagCtx==0 check runs before this lock is taken;
// re-check here so a Free() racing in between (which zeroes p.tagCtx
// under this same engineMu) is caught instead of handed to the C side,
// mirroring streamFeedDoc's/sceneFeed's re-check.
if p.tagCtx == 0 {
return "", 0, grpcerrors.ModelNotLoaded("parakeet-cpp")
}
nClasses := int(CppNumClasses(p.tagCtx))
var opts cSoundOpts
CppSoundOptsDefault(&opts)
opts.WindowSec = 10
opts.HopSec = 10
opts.TopK = int32(nClasses)
stream := CppSoundStreamBegin(p.tagCtx, &opts)
if stream == 0 {
return "", nClasses, fmt.Errorf("parakeet-cpp: sound_stream_begin failed: %s", soundLastError(p.tagCtx))
}
defer CppSoundStreamFree(stream)
offset := 0
for {
if ctx != nil {
if err := ctx.Err(); err != nil {
return "", nClasses, status.Error(codes.Canceled, "parakeet-cpp: sound detection cancelled")
}
}
end := offset + soundFeedChunkSamples
isLast := int32(0)
if end >= len(pcm) {
end = len(pcm)
isLast = 1
}
var samplePtr *float32
if end > offset {
samplePtr = &pcm[offset]
}
var segsOut uintptr
var nOut int32
rc := CppSoundStreamFeed(stream, samplePtr, int32(end-offset), isLast, &segsOut, &nOut)
if segsOut != 0 && CppFreeSoundSegments != nil {
CppFreeSoundSegments(segsOut)
}
if rc != 0 {
return "", nClasses, fmt.Errorf("parakeet-cpp: sound_stream_feed failed: %s", soundLastError(p.tagCtx))
}
offset = end
if isLast == 1 {
break
}
}
raw := CppSoundStreamDrainScoresJSON(stream)
if raw == 0 {
return "", nClasses, fmt.Errorf("parakeet-cpp: sound_stream_drain_scores_json failed: %s", soundLastError(p.tagCtx))
}
doc := goStringFromCPtr(raw)
CppFreeString(raw)
return doc, nClasses, nil
}
// soundLastError reads ctx's last_error, substituting a fallback message
// when the C side left it empty.
func soundLastError(ctx uintptr) string {
msg := CppLastError(ctx)
if msg == "" {
msg = "unknown error"
}
return msg
}
// averageWindowScores averages each class's score across the drained
// per-window scores: CED's own long-clip method, summing a class's score
// over every window and dividing by the window count (a class absent from a
// window's tags counts as 0 in that window). Only classes that appeared in
// at least one window are returned, in no particular order; callers sort and
// filter afterward. A pure function so it is easy to unit test in isolation
// from the C stream.
func averageWindowScores(windows []soundWindowJSON, nClasses int) []classAvg {
if len(windows) == 0 {
return nil
}
sums := make(map[int]float32)
labels := make(map[int]string)
for _, w := range windows {
for _, t := range w.Tags {
if t.Index < 0 || (nClasses > 0 && t.Index >= nClasses) {
continue
}
sums[t.Index] += t.Score
if _, ok := labels[t.Index]; !ok {
labels[t.Index] = t.Label
}
}
}
n := float32(len(windows))
out := make([]classAvg, 0, len(sums))
for idx, sum := range sums {
out = append(out, classAvg{Index: idx, Label: labels[idx], Score: sum / n})
}
return out
}
// sortSoundDetectionsDesc sorts avgs by score descending, breaking ties by
// class index for a deterministic order (map iteration in
// averageWindowScores is otherwise unordered).
func sortSoundDetectionsDesc(avgs []classAvg) {
sort.Slice(avgs, func(i, j int) bool {
if avgs[i].Score != avgs[j].Score {
return avgs[i].Score > avgs[j].Score
}
return avgs[i].Index < avgs[j].Index
})
}
// filterSoundDetections drops entries scoring below threshold, then keeps
// only the first topK entries (0 = keep all). avgs is assumed already sorted
// descending by score.
func filterSoundDetections(avgs []classAvg, threshold float32, topK int32) []classAvg {
out := avgs[:0:0]
for _, a := range avgs {
if a.Score < threshold {
continue
}
out = append(out, a)
}
if topK > 0 && int32(len(out)) > topK {
out = out[:topK]
}
return out
}
+383
View File
@@ -0,0 +1,383 @@
package main
import (
"context"
"path/filepath"
"sync"
"unsafe"
"github.com/mudler/LocalAI/pkg/grpc/grpcerrors"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)
// The SoundDetection specs drive it entirely against stubbed
// CppSoundStreamBegin / CppSoundStreamFeed / CppSoundStreamDrainScoresJSON /
// CppSoundStreamFree / CppFreeSoundSegments / CppSoundOptsDefault /
// CppNumClasses / CppFreeString / CppLastError (the same seam diarize_test.go
// and live_test.go use), so they run without libparakeet.so.
// soundCstrPool hands out NUL-terminated C-style strings backed by Go memory
// and keeps them alive for the duration of a spec (goStringFromCPtr reads
// through the raw pointer; mirrors diarize_test.go's diarizeCstrPool).
type soundCstrPool struct {
mu sync.Mutex
bufs [][]byte
}
func (p *soundCstrPool) cstr(s string) uintptr {
p.mu.Lock()
defer p.mu.Unlock()
b := append([]byte(s), 0)
p.bufs = append(p.bufs, b)
return uintptr(unsafe.Pointer(&b[0]))
}
// soundStubs swaps every C entry point SoundDetection touches and returns a
// restore func for AfterEach (mirrors diarize_test.go's diarizeStubs).
func soundStubs() (restore func()) {
savedBegin := CppSoundStreamBegin
savedFeed := CppSoundStreamFeed
savedDrain := CppSoundStreamDrainScoresJSON
savedFree := CppSoundStreamFree
savedFreeSegs := CppFreeSoundSegments
savedOptsDefault := CppSoundOptsDefault
savedNumClasses := CppNumClasses
savedFreeString := CppFreeString
savedLastError := CppLastError
return func() {
CppSoundStreamBegin = savedBegin
CppSoundStreamFeed = savedFeed
CppSoundStreamDrainScoresJSON = savedDrain
CppSoundStreamFree = savedFree
CppFreeSoundSegments = savedFreeSegs
CppSoundOptsDefault = savedOptsDefault
CppNumClasses = savedNumClasses
CppFreeString = savedFreeString
CppLastError = savedLastError
}
}
// soundWav writes a silent 16 kHz mono WAV of the given duration (seconds) to
// a fresh temp file and returns its path. decodeWavMono16k reads real audio
// bytes off disk, so SoundDetection needs a file on disk even though the
// stubbed C calls never look at its samples.
func soundWav(seconds float64) string {
GinkgoHelper()
path := filepath.Join(GinkgoT().TempDir(), "sound.wav")
writeMono16kWav(path, int(seconds*16000))
return path
}
// noopFeed is a CppSoundStreamFeed stub that always succeeds and returns no
// segments, for specs that only care about the drained window scores.
func noopFeed(s uintptr, pcm *float32, n int32, isLast int32, out *uintptr, nOut *int32) int32 {
*out = 0
*nOut = 0
return 0
}
var _ = Describe("ParakeetCpp.SoundDetection", func() {
var restore func()
var pool *soundCstrPool
BeforeEach(func() {
restore = soundStubs()
pool = &soundCstrPool{}
CppFreeString = func(uintptr) {}
CppFreeSoundSegments = func(uintptr) {}
CppSoundOptsDefault = func(o *cSoundOpts) { *o = cSoundOpts{} }
})
AfterEach(func() { restore() })
It("fails with FailedPrecondition when no sound model is loaded", func() {
p := &ParakeetCpp{}
_, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: soundWav(1)})
Expect(err).To(HaveOccurred())
Expect(status.Code(err)).To(Equal(codes.FailedPrecondition))
Expect(err.Error()).To(ContainSubstring("model is not a sound"))
})
It("fails with Unimplemented when the loaded libparakeet.so has no sound_stream_begin symbol", func() {
CppSoundStreamBegin = nil
p := &ParakeetCpp{tagCtx: 42}
_, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: soundWav(1)})
Expect(err).To(HaveOccurred())
Expect(status.Code(err)).To(Equal(codes.Unimplemented))
})
It("averages two windows per class and sorts descending", func() {
CppNumClasses = func(uintptr) int32 { return 2 }
CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 }
CppSoundStreamFeed = noopFeed
CppSoundStreamFree = func(uintptr) {}
CppSoundStreamDrainScoresJSON = func(uintptr) uintptr {
return pool.cstr(`[` +
`{"start":0,"end":10,"tags":[{"index":0,"label":"Speech","score":0.8},{"index":1,"label":"Music","score":0.2}]},` +
`{"start":10,"end":20,"tags":[{"index":0,"label":"Speech","score":0.4},{"index":1,"label":"Music","score":0.6}]}` +
`]`)
}
p := &ParakeetCpp{tagCtx: 42}
resp, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: soundWav(1)})
Expect(err).ToNot(HaveOccurred())
Expect(resp.Detections).To(HaveLen(2))
Expect(resp.Detections[0].Label).To(Equal("Speech"))
Expect(resp.Detections[0].Score).To(BeNumerically("~", 0.6, 1e-6))
Expect(resp.Detections[1].Label).To(Equal("Music"))
Expect(resp.Detections[1].Score).To(BeNumerically("~", 0.4, 1e-6))
})
It("drops classes scoring below threshold", func() {
CppNumClasses = func(uintptr) int32 { return 2 }
CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 }
CppSoundStreamFeed = noopFeed
CppSoundStreamFree = func(uintptr) {}
CppSoundStreamDrainScoresJSON = func(uintptr) uintptr {
return pool.cstr(`[{"start":0,"end":10,"tags":[` +
`{"index":0,"label":"Speech","score":0.8},` +
`{"index":1,"label":"Music","score":0.2}]}]`)
}
p := &ParakeetCpp{tagCtx: 42}
resp, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{
Src: soundWav(1), Threshold: 0.5,
})
Expect(err).ToNot(HaveOccurred())
Expect(resp.Detections).To(HaveLen(1))
Expect(resp.Detections[0].Label).To(Equal("Speech"))
})
It("keeps only the top_k entries", func() {
CppNumClasses = func(uintptr) int32 { return 4 }
CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 }
CppSoundStreamFeed = noopFeed
CppSoundStreamFree = func(uintptr) {}
CppSoundStreamDrainScoresJSON = func(uintptr) uintptr {
return pool.cstr(`[{"start":0,"end":10,"tags":[` +
`{"index":0,"label":"A","score":0.9},` +
`{"index":1,"label":"B","score":0.7},` +
`{"index":2,"label":"C","score":0.5},` +
`{"index":3,"label":"D","score":0.3}]}]`)
}
p := &ParakeetCpp{tagCtx: 42}
resp, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{
Src: soundWav(1), TopK: 3,
})
Expect(err).ToNot(HaveOccurred())
Expect(resp.Detections).To(HaveLen(3))
Expect(resp.Detections[0].Label).To(Equal("A"))
Expect(resp.Detections[1].Label).To(Equal("B"))
Expect(resp.Detections[2].Label).To(Equal("C"))
})
It("keeps all classes when top_k is 0", func() {
CppNumClasses = func(uintptr) int32 { return 4 }
CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 }
CppSoundStreamFeed = noopFeed
CppSoundStreamFree = func(uintptr) {}
CppSoundStreamDrainScoresJSON = func(uintptr) uintptr {
return pool.cstr(`[{"start":0,"end":10,"tags":[` +
`{"index":0,"label":"A","score":0.9},` +
`{"index":1,"label":"B","score":0.7},` +
`{"index":2,"label":"C","score":0.5},` +
`{"index":3,"label":"D","score":0.3}]}]`)
}
p := &ParakeetCpp{tagCtx: 42}
resp, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{
Src: soundWav(1), TopK: 0,
})
Expect(err).ToNot(HaveOccurred())
Expect(resp.Detections).To(HaveLen(4))
})
It("passes window 10s, hop 10s and top_k = the tagger's class count to sound_stream_begin", func() {
CppNumClasses = func(uintptr) int32 { return 527 }
var gotOpts cSoundOpts
CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr {
gotOpts = *o
return 1
}
CppSoundStreamFeed = noopFeed
CppSoundStreamFree = func(uintptr) {}
CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { return pool.cstr(`[]`) }
p := &ParakeetCpp{tagCtx: 42}
_, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: soundWav(1)})
Expect(err).ToNot(HaveOccurred())
Expect(gotOpts.WindowSec).To(BeNumerically("==", 10))
Expect(gotOpts.HopSec).To(BeNumerically("==", 10))
Expect(gotOpts.TopK).To(Equal(int32(527)))
})
It("returns no detections without error for a short clip whose drain is empty", func() {
CppNumClasses = func(uintptr) int32 { return 2 }
CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 }
CppSoundStreamFeed = noopFeed
CppSoundStreamFree = func(uintptr) {}
CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { return pool.cstr(`[]`) }
p := &ParakeetCpp{tagCtx: 42}
resp, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: soundWav(0.1)})
Expect(err).ToNot(HaveOccurred())
Expect(resp.Detections).To(BeEmpty())
})
It("surfaces last_error and still frees the stream when feed fails", func() {
freed := false
CppNumClasses = func(uintptr) int32 { return 2 }
CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 }
CppSoundStreamFeed = func(s uintptr, pcm *float32, n int32, isLast int32, out *uintptr, nOut *int32) int32 {
*out = 0
*nOut = 0
return 1
}
CppSoundStreamDrainScoresJSON = func(uintptr) uintptr {
Fail("drain_scores_json must not be called when feed failed")
return 0
}
CppSoundStreamFree = func(uintptr) { freed = true }
CppLastError = func(uintptr) string { return "boom" }
p := &ParakeetCpp{tagCtx: 42}
_, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: soundWav(1)})
Expect(err).To(HaveOccurred())
Expect(err.Error()).To(ContainSubstring("boom"))
Expect(freed).To(BeTrue())
})
It("returns Canceled without feeding when ctx is already cancelled, and frees the stream", func() {
freed := false
ctx, cancel := context.WithCancel(context.Background())
cancel()
CppNumClasses = func(uintptr) int32 { return 2 }
CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 }
CppSoundStreamFeed = func(s uintptr, pcm *float32, n int32, isLast int32, out *uintptr, nOut *int32) int32 {
Fail("sound_stream_feed must not be called when ctx is already cancelled")
return 0
}
CppSoundStreamDrainScoresJSON = func(uintptr) uintptr {
Fail("drain_scores_json must not be called when ctx is already cancelled")
return 0
}
CppSoundStreamFree = func(uintptr) { freed = true }
p := &ParakeetCpp{tagCtx: 42}
_, err := p.SoundDetection(ctx, &pb.SoundDetectionRequest{Src: soundWav(15)})
Expect(err).To(HaveOccurred())
Expect(status.Code(err)).To(Equal(codes.Canceled))
Expect(freed).To(BeTrue())
})
It("stops feeding and frees the stream when ctx is cancelled mid-feed", func() {
freed := false
feedCount := 0
ctx, cancel := context.WithCancel(context.Background())
CppNumClasses = func(uintptr) int32 { return 2 }
CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { return 1 }
CppSoundStreamFeed = func(s uintptr, pcm *float32, n int32, isLast int32, out *uintptr, nOut *int32) int32 {
feedCount++
cancel() // cancel after the first feed so a second chunk would exist if not stopped
*out = 0
*nOut = 0
return 0
}
CppSoundStreamDrainScoresJSON = func(uintptr) uintptr {
Fail("drain_scores_json must not be called when the feed loop was cancelled")
return 0
}
CppSoundStreamFree = func(uintptr) { freed = true }
// Two 10 s chunks, so a second feed call would happen without the cancel.
p := &ParakeetCpp{tagCtx: 42}
_, err := p.SoundDetection(ctx, &pb.SoundDetectionRequest{Src: soundWav(15)})
Expect(err).To(HaveOccurred())
Expect(status.Code(err)).To(Equal(codes.Canceled))
Expect(feedCount).To(Equal(1))
Expect(freed).To(BeTrue())
})
It("wraps a decode failure as InvalidArgument", func() {
// Every required symbol must be non-nil to clear SoundDetection's own
// Unimplemented gate and reach the decode step this spec targets;
// none of them may actually be called.
fail := func(string) { Fail("no C call once the decode itself has failed") }
CppNumClasses = func(uintptr) int32 { fail("num_classes"); return 0 }
CppSoundStreamBegin = func(tagger uintptr, o *cSoundOpts) uintptr { fail("begin"); return 0 }
CppSoundStreamFeed = func(s uintptr, pcm *float32, n int32, isLast int32, out *uintptr, nOut *int32) int32 {
fail("feed")
return 0
}
CppSoundStreamDrainScoresJSON = func(uintptr) uintptr { fail("drain"); return 0 }
CppSoundStreamFree = func(uintptr) { fail("free") }
p := &ParakeetCpp{tagCtx: 42}
_, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{
Src: filepath.Join(GinkgoT().TempDir(), "missing.wav"),
})
Expect(err).To(HaveOccurred())
Expect(status.Code(err)).To(Equal(codes.InvalidArgument))
})
It("returns ModelNotLoaded without a C call when tagCtx is zeroed between the entry check and the call", func() {
called := false
CppNumClasses = func(uintptr) int32 { called = true; return 2 }
p := &ParakeetCpp{tagCtx: 42}
// Simulate a Free() racing between SoundDetection's own tagCtx==0
// check and soundStreamDrain's lock, exactly as it zeroes tagCtx
// under engineMu.
p.tagCtx = 0
_, _, err := p.soundStreamDrain(context.Background(), make([]float32, 10))
Expect(grpcerrors.IsModelNotLoaded(err)).To(BeTrue())
Expect(called).To(BeFalse(), "no C call once tagCtx was cleared")
})
})
var _ = Describe("averageWindowScores", func() {
It("returns nil for no windows", func() {
Expect(averageWindowScores(nil, 2)).To(BeNil())
})
It("treats a class absent from a window as 0 in that window's contribution", func() {
windows := []soundWindowJSON{
{Tags: []soundTagJSON{{Index: 0, Label: "Speech", Score: 1.0}}},
{Tags: []soundTagJSON{}}, // Speech absent this window
}
out := averageWindowScores(windows, 2)
Expect(out).To(HaveLen(1))
Expect(out[0].Index).To(Equal(0))
Expect(out[0].Score).To(BeNumerically("~", 0.5, 1e-6))
})
It("ignores an out-of-range class index", func() {
windows := []soundWindowJSON{
{Tags: []soundTagJSON{{Index: 5, Label: "Bogus", Score: 1.0}}},
}
Expect(averageWindowScores(windows, 2)).To(BeEmpty())
})
})
var _ = Describe("filterSoundDetections", func() {
It("keeps everything when top_k is 0 and threshold is 0", func() {
in := []classAvg{{Index: 0, Score: 0.1}, {Index: 1, Score: 0.9}}
Expect(filterSoundDetections(in, 0, 0)).To(HaveLen(2))
})
It("drops entries below threshold before applying top_k", func() {
in := []classAvg{
{Index: 0, Score: 0.9},
{Index: 1, Score: 0.4},
{Index: 2, Score: 0.1},
}
out := filterSoundDetections(in, 0.3, 5)
Expect(out).To(HaveLen(2))
})
})
+121
View File
@@ -0,0 +1,121 @@
package main
import (
"encoding/json"
"fmt"
"strconv"
)
// Speaker labels on transcripts. With a diarization_model companion attached
// to an ASR model, unary transcription tags each segment (and, with word
// timestamps, each word) with its speaker, and the stream=true final result
// tags each utterance. Live transcription carries speakers through the scene
// stream instead (scene.go).
// speakerSnapSeconds mirrors parakeet.cpp's merge_asr_diarization: a word that
// overlaps no speaker segment takes the nearest segment's speaker when that
// segment is this close. ASR word boundaries and diarization boundaries can
// disagree by a frame or two; a word farther than this has no speaker.
const speakerSnapSeconds = 0.5
// transcriptSpeaker renders a 0-based speaker for a transcript segment or
// word; -1 (no speaker) is left empty so the field is omitted.
func transcriptSpeaker(spk int) string {
if spk < 0 {
return ""
}
return strconv.Itoa(spk)
}
// wantSpeakers reports whether a transcription should carry speaker labels:
// a diarization companion is attached, the library can diarize, and the
// request did not turn it off (the OpenAI endpoint sends diarize=true unless
// the client passes diarize=false).
func (p *ParakeetCpp) wantSpeakers(diarize bool) bool {
return diarize && p.ctxPtr != 0 && p.diarCtx != 0 && CppDiarizePCM != nil
}
// diarizeSegmentsPCM runs the diarization companion over 16 kHz PCM
// (parakeet_capi_diarize_pcm, the checkpoint's own mode, as NeMo's
// diarize()) and returns its segments.
func (p *ParakeetCpp) diarizeSegmentsPCM(pcm []float32) ([]diarizeSegmentJSON, error) {
if len(pcm) == 0 {
return nil, nil
}
raw, err := p.diarizeCall(pcm, false)
if err != nil {
return nil, err
}
var doc diarizePCMDoc
if err := json.Unmarshal([]byte(raw), &doc); err != nil {
return nil, fmt.Errorf("parakeet-cpp: decode diarization json: %w", err)
}
return doc.Segments, nil
}
// assignSpeakers gives each word the speaker whose segments overlap it most,
// falling back to the nearest segment within speakerSnapSeconds; -1 when none.
// Same rule as parakeet.cpp's merge_asr_diarization, so the labels match the
// library's own speaker-attributed ASR.
func assignSpeakers(words []transcriptWord, segs []diarizeSegmentJSON) []int {
out := make([]int, len(words))
for i, w := range words {
best, bestOverlap := -1, 0.0
for _, s := range segs {
if ov := min(w.End, s.End) - max(w.Start, s.Start); ov > bestOverlap {
best, bestOverlap = s.Speaker, ov
}
}
if best < 0 {
bestDist := speakerSnapSeconds
for _, s := range segs {
dist := s.Start - w.End
if s.End <= w.Start {
dist = w.Start - s.End
}
if dist >= 0 && dist <= bestDist {
best, bestDist = s.Speaker, dist
}
}
}
out[i] = best
}
return out
}
// splitAtSpeakerChanges splits each word group wherever the speaker changes,
// so every segment has one speaker. speakers is indexed like the
// concatenation of groups. Returns the new groups and each group's speaker.
func splitAtSpeakerChanges(groups [][]transcriptWord, speakers []int) ([][]transcriptWord, []int) {
var outGroups [][]transcriptWord
var outSpk []int
k := 0
for _, g := range groups {
start := 0
for i := 1; i <= len(g); i++ {
if i == len(g) || speakers[k+i] != speakers[k+start] {
outGroups = append(outGroups, g[start:i])
outSpk = append(outSpk, speakers[k+start])
start = i
}
}
k += len(g)
}
return outGroups, outSpk
}
// majoritySpeaker is the speaker covering most of the words' duration, or -1.
func majoritySpeaker(words []transcriptWord, speakers []int) int {
dur := map[int]float64{}
best, bestDur := -1, 0.0
for i, w := range words {
if speakers[i] < 0 {
continue
}
dur[speakers[i]] += max(w.End-w.Start, 1e-3)
if dur[speakers[i]] > bestDur {
best, bestDur = speakers[i], dur[speakers[i]]
}
}
return best
}
+166
View File
@@ -0,0 +1,166 @@
package main
import (
"context"
"os"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
func sseg(spk int, start, end float64) diarizeSegmentJSON {
return diarizeSegmentJSON{Speaker: spk, Start: start, End: end}
}
// speakerTurnsOf collapses consecutive repeats: [0 0 1 0] -> [0 1 0].
func speakerTurnsOf(labels []string) []string {
var out []string
for _, l := range labels {
if len(out) == 0 || out[len(out)-1] != l {
out = append(out, l)
}
}
return out
}
var _ = Describe("transcript speaker labels", func() {
Context("assignSpeakers", func() {
It("picks the speaker with the largest overlap", func() {
words := []transcriptWord{tw("a", 0.5, 0.8)}
Expect(assignSpeakers(words, []diarizeSegmentJSON{sseg(0, 0.0, 0.6), sseg(1, 0.55, 2.0)})).
To(Equal([]int{1}))
})
It("snaps a word just outside a segment to it, but not a far one", func() {
words := []transcriptWord{tw("well", 19.92, 20.00), tw("far", 25.0, 25.2)}
segs := []diarizeSegmentJSON{sseg(0, 14.78, 18.75), sseg(1, 20.10, 23.60)}
Expect(assignSpeakers(words, segs)).To(Equal([]int{1, -1}))
})
})
It("splits segments at speaker turns and labels segments and words", func() {
doc := transcriptJSON{
Text: "hi there. hello",
Words: []transcriptWord{tw("hi", 0.0, 0.3), tw("there.", 0.3, 0.6), tw("hello", 1.0, 1.4)},
}
// Punctuation gives [hi there.] [hello]; the turn after "hi" splits the first.
opts := &pb.TranscriptRequest{TimestampGranularities: []string{"word"}}
res := transcriptResultWithSpeakers(doc, opts, 0, []int{0, 1, 1})
Expect(res.Segments).To(HaveLen(3))
Expect([]string{res.Segments[0].Speaker, res.Segments[1].Speaker, res.Segments[2].Speaker}).
To(Equal([]string{"0", "1", "1"}))
Expect(res.Segments[1].Text).To(Equal("there."))
Expect(res.Segments[2].Words[0].Speaker).To(Equal("1"))
for i, seg := range res.Segments {
Expect(seg.Id).To(Equal(int32(i)))
}
})
It("leaves speakers empty without diarization and for unknown words", func() {
doc := transcriptJSON{Text: "hi.", Words: []transcriptWord{tw("hi.", 0, 0.3)}}
Expect(transcriptResultFromDoc(doc, &pb.TranscriptRequest{}, 0).Segments[0].Speaker).To(BeEmpty())
Expect(transcriptResultWithSpeakers(doc, &pb.TranscriptRequest{}, 0, []int{-1}).Segments[0].Speaker).
To(BeEmpty())
})
It("picks the speaker covering most of an utterance", func() {
words := []transcriptWord{tw("a", 0, 0.2), tw("b", 0.2, 1.5), tw("c", 1.5, 1.6)}
Expect(majoritySpeaker(words, []int{0, 1, 0})).To(Equal(1))
Expect(majoritySpeaker(words, []int{-1, -1, -1})).To(Equal(-1))
})
It("only diarizes with a companion, a capable library and diarize=true", func() {
saved := CppDiarizePCM
defer func() { CppDiarizePCM = saved }()
CppDiarizePCM = func(uintptr, *float32, int32, int32) uintptr { return 0 }
p := &ParakeetCpp{ctxPtr: 1, diarCtx: 2}
Expect(p.wantSpeakers(true)).To(BeTrue())
Expect(p.wantSpeakers(false)).To(BeFalse())
Expect((&ParakeetCpp{ctxPtr: 1}).wantSpeakers(true)).To(BeFalse())
CppDiarizePCM = nil
Expect(p.wantSpeakers(true)).To(BeFalse())
})
})
var _ = Describe("ParakeetCpp transcript speakers (real models)", func() {
It("labels unary and stream=true transcripts with a diarization_model companion", func() {
diarModel := os.Getenv("PARAKEET_BACKEND_TEST_DIAR_MODEL")
asrModel := os.Getenv("PARAKEET_BACKEND_TEST_MODEL")
wavPath := os.Getenv("PARAKEET_BACKEND_TEST_DIAR_WAV")
if diarModel == "" || asrModel == "" || wavPath == "" {
Skip("set PARAKEET_BACKEND_TEST_DIAR_MODEL, PARAKEET_BACKEND_TEST_MODEL and " +
"PARAKEET_BACKEND_TEST_DIAR_WAV (a multi-speaker 16 kHz WAV)")
}
ensureLibLoaded()
if CppDiarizePCM == nil || CppModelKind == nil {
Skip("libparakeet.so has no diarization / model-kind C-API")
}
p := &ParakeetCpp{}
Expect(p.Load(&pb.ModelOptions{
ModelFile: asrModel,
Options: []string{"diarization_model:" + diarModel},
})).To(Succeed())
defer func() { _ = p.Free() }()
res, err := p.AudioTranscription(context.Background(), &pb.TranscriptRequest{Dst: wavPath, Diarize: true})
Expect(err).ToNot(HaveOccurred())
labels := make([]string, len(res.Segments))
for i, s := range res.Segments {
Expect(s.Speaker).ToNot(BeEmpty(), "segment %d %q", i, s.Text)
labels[i] = s.Speaker
}
// parakeet.cpp's tests/fixtures/two_speakers.wav alternates A-B-A-B.
Expect(speakerTurnsOf(labels)).To(Equal([]string{"0", "1", "0", "1"}))
plain, err := p.AudioTranscription(context.Background(), &pb.TranscriptRequest{Dst: wavPath})
Expect(err).ToNot(HaveOccurred())
for _, s := range plain.Segments {
Expect(s.Speaker).To(BeEmpty())
}
Expect(plain.Text).To(Equal(res.Text))
})
})
var _ = Describe("ParakeetCpp scene companions on an offline ASR model (real models)", func() {
It("labels speakers and detects sounds from one model with both companions", func() {
asrModel := os.Getenv("PARAKEET_BACKEND_TEST_SCENE_ASR_MODEL") // e.g. tdt-0.6b-v3
diarModel := os.Getenv("PARAKEET_BACKEND_TEST_DIAR_MODEL")
soundModel := os.Getenv("PARAKEET_BACKEND_TEST_SOUND_MODEL")
wavPath := os.Getenv("PARAKEET_BACKEND_TEST_SCENE_WAV") // speech + a non-speech sound
wantLabel := os.Getenv("PARAKEET_BACKEND_TEST_SCENE_LABEL") // e.g. "Chicken, rooster"
if asrModel == "" || diarModel == "" || soundModel == "" || wavPath == "" || wantLabel == "" {
Skip("set PARAKEET_BACKEND_TEST_SCENE_ASR_MODEL, _DIAR_MODEL, _SOUND_MODEL, " +
"PARAKEET_BACKEND_TEST_SCENE_WAV and PARAKEET_BACKEND_TEST_SCENE_LABEL")
}
ensureLibLoaded()
if CppDiarizePCM == nil || CppModelKind == nil {
Skip("libparakeet.so has no diarization / model-kind C-API")
}
p := &ParakeetCpp{}
Expect(p.Load(&pb.ModelOptions{
ModelFile: asrModel,
Options: []string{"diarization_model:" + diarModel, "sound_model:" + soundModel},
})).To(Succeed())
defer func() { _ = p.Free() }()
res, err := p.AudioTranscription(context.Background(), &pb.TranscriptRequest{Dst: wavPath, Diarize: true})
Expect(err).ToNot(HaveOccurred())
labels := make([]string, 0, len(res.Segments))
for _, s := range res.Segments {
GinkgoWriter.Printf("[%5.1f-%5.1f] spk %q: %s\n", float64(s.Start)/1e9, float64(s.End)/1e9, s.Speaker, s.Text)
labels = append(labels, s.Speaker)
}
Expect(len(speakerTurnsOf(labels))).To(BeNumerically(">=", 3), "speaker turns: %v", labels)
Expect(labels).To(ContainElements("0", "1"))
sd, err := p.SoundDetection(context.Background(), &pb.SoundDetectionRequest{Src: wavPath, TopK: 5})
Expect(err).ToNot(HaveOccurred())
var got []string
for _, d := range sd.GetDetections() {
GinkgoWriter.Printf("sound %q %.2f\n", d.GetLabel(), d.GetScore())
got = append(got, d.GetLabel())
}
Expect(got).To(ContainElement(wantLabel))
})
})
+9 -1
View File
@@ -11,7 +11,7 @@ JOBS?=$(shell nproc --ignore=1 2>/dev/null || sysctl -n hw.ncpu 2>/dev/null || e
# vllm.cpp version
VLLM_CPP_REPO?=https://github.com/mudler/vllm.cpp
VLLM_CPP_VERSION?=c3bebc357385990f721af66a3a6c69328dd4fc6c
VLLM_CPP_VERSION?=96788348627b6a079fcc3ef7fc6676a970965d7b
# MLX GEMM provider (darwin/metal only; see the metal branch below for why).
# Consumed as the prebuilt pip wheel: building MLX from source needs `xcrun
@@ -40,6 +40,14 @@ MLX_ROOT=$(shell echo $(MLX_VENV)/lib/python*/site-packages/mlx)
# server, examples and tests of the engine are never built here.
CMAKE_ARGS+=-DVLLM_CPP_SERVER=OFF -DVLLM_CPP_BUILD_TESTS=OFF -DVLLM_CPP_BUILD_EXAMPLES=OFF
CMAKE_ARGS+=-DCMAKE_BUILD_TYPE=Release
# Diarization (ABI v30) is ON upstream and FetchContents parakeet.cpp at
# configure time, building a second (static) ggml into libvllm. The pinned
# engine now pins that fetch to a commit, so ON configures, but it still adds a
# network fetch and a second ggml to every build for calls this backend never
# makes: it binds none of the vllm_diariz*/vllm_transcribe_and_* entry points,
# and with the option OFF they compile as stubs that refuse by name, so the ABI
# stays v30-complete without the extra dependency.
CMAKE_ARGS+=-DVLLM_CPP_WITH_DIARIZATION=OFF
# vllm.cpp sets no global -march: SIMD tiers are per-file with runtime dispatch,
# so ONE portable library serves every CPU of the target arch (unlike the
+28 -1
View File
@@ -9,7 +9,7 @@ It serves two things: text generation, and MiniMax-H3 joint video+audio
generation.
The backend dlopens the engine's stable C ABI (`libvllm`, `include/vllm.h`,
ABI v20) through purego:
ABI v30) through purego:
- `Load` -> `vllm_engine_load`: accepts a `.gguf` file or a HF-style model
directory (`config.json` + safetensors). `context_size` maps to
@@ -44,6 +44,14 @@ the Makefile therefore means updating `abiVersion` plus the mirrors (and their
offsets in `vllmcpp_test.go`) in the same change; `make abi-check` compares the
pinned header against the bindings and the library build runs it first.
The Makefile builds libvllm with `-DVLLM_CPP_WITH_DIARIZATION=OFF`. vllm.cpp
turns that option ON by default since ABI v30, and ON fetches a pinned
parakeet.cpp (with its own ggml) at configure time. This backend does not bind
the diarization entry points, so OFF adds no dependency and changes nothing it
serves: those calls exist in libvllm but refuse with "not compiled in". If a
future change binds them, pin the parakeet.cpp source with
`-DVLLM_CPP_PARAKEET_CPP_DIR` and make `package.sh` bundle what it links.
Model config example:
```yaml
@@ -56,6 +64,25 @@ options:
- max_num_seqs:16
```
## hf_overrides
`engine_args.hf_overrides` (vLLM parity) is a JSON object of top-level
`config.json` keys merged over the model directory's `config.json`. The C ABI
has no override input and the engine reads `config.json` from the directory it
is given, so `Load` builds an overlay (`hfoverrides.go`): a temp dir with the
merged `config.json` plus a symlink to every other entry of the model dir, and
passes the overlay as `model_path`. `validModelPath` and the DFlash draft
resolution still run against the real model dir. `Free` (and a failed load, or
the next `Load`) removes the overlay. A value that is not an object, a `.gguf`
model, or a dir without `config.json` fails the load instead of being ignored,
because loading the unmodified config would serve another architecture.
```yaml
engine_args:
hf_overrides:
architectures: ["Tev1Model"] # opt a Qwen3.5-declared Tev1 snapshot into the Tev1 adapter
```
## MiniMax-H3 video+audio generation
`GenerateVideo` -> `vllm_video_generate` (ABI v12). H3 renders picture and sound
+32 -3
View File
@@ -37,6 +37,10 @@ type VllmCpp struct {
// other's checkpoints. Exactly one of the two is ever non-zero.
videoEngine uintptr
opts loadOptions
// overlayDir is the hf_overrides overlay handed to the engine in place of
// the model directory. It must outlive the engine handle (the engine may
// reopen files through it), so it is removed in Free, not after Load.
overlayDir string
}
// Stream registry: the per-request bridge between the C token callback and
@@ -141,6 +145,23 @@ func (v *VllmCpp) Load(opts *pb.ModelOptions) error {
}
v.opts.speculativeConfig = resolvedSpec
// A reload reuses this struct: drop any overlay from the previous model
// before building a new one so it is not leaked.
if err := removeConfigOverlay(v.overlayDir); err != nil {
xlog.Warn("[vllm-cpp] stale overlay", "error", err)
}
v.overlayDir = ""
enginePath := model
if hasHFOverrides(v.opts.hfOverrides) {
overlay, err := newConfigOverlay(model, v.opts.hfOverrides)
if err != nil {
return err
}
v.overlayDir = overlay
enginePath = overlay
xlog.Info("[vllm-cpp] hf_overrides applied through overlay", "model", model, "overlay", overlay, "overrides", v.opts.hfOverrides)
}
mp := defaultModelParams()
if v.opts.blockSize > 0 {
mp.BlockSize = v.opts.blockSize
@@ -173,7 +194,7 @@ func (v *VllmCpp) Load(opts *pb.ModelOptions) error {
// Every string below is borrowed by C for the duration of the load call
// only (the library copies what it keeps), so the backing slices just have
// to outlive vllmEngineLoad - hence the single KeepAlive after it.
modelC := cString(model)
modelC := cString(enginePath)
mp.ModelPath = uintptr(unsafe.Pointer(&modelC[0])) // #nosec G103 -- borrowed by C for the load call only
keep := [][]byte{modelC}
setStr := func(dst *uintptr, s string) {
@@ -205,7 +226,12 @@ func (v *VllmCpp) Load(opts *pb.ModelOptions) error {
rc := vllmEngineLoad(unsafe.Pointer(&mp), unsafe.Pointer(&engine)) // #nosec G103 -- POD out-params
runtime.KeepAlive(keep)
if rc != vllmOK {
return fmt.Errorf("vllm-cpp: engine load failed: %s", vllmLastError())
loadErr := fmt.Errorf("vllm-cpp: engine load failed: %s", vllmLastError())
if err := removeConfigOverlay(v.overlayDir); err != nil {
xlog.Warn("[vllm-cpp] overlay cleanup after failed load", "error", err)
}
v.overlayDir = ""
return loadErr
}
v.engine = engine
return nil
@@ -220,7 +246,10 @@ func (v *VllmCpp) Free() error {
vllmVideoEngineFree(v.videoEngine)
v.videoEngine = 0
}
return nil
// After the engine is gone, so nothing still reads through the links.
err := removeConfigOverlay(v.overlayDir)
v.overlayDir = ""
return err
}
// samplingFromPredict lowers PredictOptions into the C sampling POD plus the
+7 -2
View File
@@ -1,6 +1,6 @@
package main
// purego bindings for the vllm.cpp stable C ABI (include/vllm.h, ABI v27).
// purego bindings for the vllm.cpp stable C ABI (include/vllm.h, ABI v30).
//
// The structs below are hand-mirrored PODs of the C declarations, with
// explicit padding so the Go layout matches the C layout on linux/darwin
@@ -21,7 +21,12 @@ import (
// the header of the VLLM_CPP_VERSION pinned in the Makefile: the build checks
// the two against each other, because a mismatch is only caught at runtime by
// registerLib, where it takes the backend down on every load (issue #11379).
const abiVersion = 29
//
// v30 only ADDED the diarization and speaker-attributed-ASR entry points; every
// struct mirrored here is byte-identical to v29. They are not bound because the
// Makefile builds libvllm with VLLM_CPP_WITH_DIARIZATION=OFF, where they are
// stubs that refuse every call.
const abiVersion = 30
// The ABI's tri-state toggles (enable_prefix_caching ABI v7,
// enable_jump_forward ABI v10) share one encoding: 0 is NOT "off", it is
+125
View File
@@ -0,0 +1,125 @@
package main
// hf_overrides, vLLM parity: a JSON object of config.json keys laid over the
// model directory's own config.json at load time.
//
// The engine reads config.json straight from the model directory and has no
// override input on the C ABI, so the only way to change what it sees without
// editing the snapshot is to hand it a different directory. The overlay built
// here is that directory: a private temp dir holding the merged config.json
// and a symlink for every other entry of the original. The snapshot itself is
// never written, which matters because it is a content-addressed download that
// a gallery reinstall or a hash check would otherwise flag or overwrite.
//
// The canonical use is opting a published checkpoint into an engine adapter
// its config does not name, e.g. {"architectures": ["Tev1Model"]} on a Tev1
// snapshot that declares Qwen3_5ForConditionalGeneration.
import (
"encoding/json"
"errors"
"fmt"
"os"
"path/filepath"
"strings"
)
// newConfigOverlay builds the overlay directory for modelDir with overrides
// merged over its config.json and returns its path. Keys are merged at the top
// level only: an override replaces the whole value of its key, nested objects
// included, which is what vLLM does for a plain (non sub-config) key.
//
// Bad input is refused rather than skipped, unlike an unknown engine_args key:
// hf_overrides exists to change which architecture loads, so silently loading
// the unmodified config would serve a different model than the one configured.
func newConfigOverlay(modelDir, overrides string) (dir string, err error) {
var patch map[string]any
if err := json.Unmarshal([]byte(overrides), &patch); err != nil || patch == nil {
return "", fmt.Errorf("vllm-cpp: hf_overrides must be a JSON object of config.json keys, got %q", overrides)
}
info, err := os.Stat(modelDir)
if err != nil {
return "", fmt.Errorf("vllm-cpp: hf_overrides: %w", err)
}
if !info.IsDir() {
return "", fmt.Errorf("vllm-cpp: hf_overrides needs a model directory with a config.json, %q is a file", modelDir)
}
absDir, err := filepath.Abs(modelDir)
if err != nil {
return "", fmt.Errorf("vllm-cpp: hf_overrides: %w", err)
}
// #nosec G304 -- absDir is the model directory from the operator's own model config, never a request-supplied path
raw, err := os.ReadFile(filepath.Join(absDir, "config.json"))
if err != nil {
return "", fmt.Errorf("vllm-cpp: hf_overrides needs %s: %w", filepath.Join(absDir, "config.json"), err)
}
var config map[string]any
if err := json.Unmarshal(raw, &config); err != nil || config == nil {
return "", fmt.Errorf("vllm-cpp: hf_overrides: %s is not a JSON object", filepath.Join(absDir, "config.json"))
}
for k, v := range patch {
config[k] = v
}
merged, err := json.MarshalIndent(config, "", " ")
if err != nil {
return "", fmt.Errorf("vllm-cpp: hf_overrides: encoding the merged config.json: %w", err)
}
entries, err := os.ReadDir(absDir)
if err != nil {
return "", fmt.Errorf("vllm-cpp: hf_overrides: %w", err)
}
dir, err = os.MkdirTemp("", "vllm-cpp-hf-overrides-*")
if err != nil {
return "", fmt.Errorf("vllm-cpp: hf_overrides: creating the overlay: %w", err)
}
defer func() {
if err != nil {
_ = os.RemoveAll(dir)
dir = ""
}
}()
// Links point at the entry path, not at what it resolves to: an HF cache
// snapshot is itself a tree of links into blobs/, and the engine already
// follows those.
for _, e := range entries {
if e.Name() == "config.json" {
continue
}
if err := os.Symlink(filepath.Join(absDir, e.Name()), filepath.Join(dir, e.Name())); err != nil {
return "", fmt.Errorf("vllm-cpp: hf_overrides: linking %s into the overlay: %w", e.Name(), err)
}
}
if err := os.WriteFile(filepath.Join(dir, "config.json"), merged, 0o600); err != nil {
return "", fmt.Errorf("vllm-cpp: hf_overrides: writing the merged config.json: %w", err)
}
return dir, nil
}
// removeConfigOverlay deletes an overlay built by newConfigOverlay. RemoveAll
// removes the symlinks themselves and never descends into their targets, so
// the original model directory is safe.
func removeConfigOverlay(dir string) error {
if dir == "" {
return nil
}
if err := os.RemoveAll(dir); err != nil && !errors.Is(err, os.ErrNotExist) {
return fmt.Errorf("vllm-cpp: removing the hf_overrides overlay %s: %w", dir, err)
}
return nil
}
// hasHFOverrides reports whether an overlay is needed. An empty object is a
// no-op in vLLM too, so it loads the directory directly instead of paying for
// an overlay that changes nothing.
func hasHFOverrides(overrides string) bool {
switch strings.TrimSpace(overrides) {
case "", "{}", "null":
return false
}
return true
}
+138
View File
@@ -0,0 +1,138 @@
package main
import (
"encoding/json"
"os"
"path/filepath"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
)
var _ = Describe("hf_overrides", func() {
var modelDir string
const originalConfig = `{"architectures":["Qwen3_5ForConditionalGeneration"],"hidden_size":2560,"text_config":{"num_hidden_layers":36}}`
BeforeEach(func() {
modelDir = GinkgoT().TempDir()
Expect(os.WriteFile(filepath.Join(modelDir, "config.json"), []byte(originalConfig), 0o644)).To(Succeed())
Expect(os.WriteFile(filepath.Join(modelDir, "model.safetensors"), []byte("weights"), 0o644)).To(Succeed())
Expect(os.WriteFile(filepath.Join(modelDir, "tokenizer.json"), []byte("{}"), 0o644)).To(Succeed())
Expect(os.MkdirAll(filepath.Join(modelDir, "tokenizer"), 0o755)).To(Succeed())
Expect(os.WriteFile(filepath.Join(modelDir, "tokenizer", "vocab.json"), []byte("{}"), 0o644)).To(Succeed())
})
Describe("parsing", func() {
It("reads a YAML-nested object from engine_args as a JSON document", func() {
lo := parseOptions(&pb.ModelOptions{
EngineArgs: `{"hf_overrides":{"architectures":["Tev1Model"]},"max_num_seqs":2}`,
})
Expect(lo.hfOverrides).To(MatchJSON(`{"architectures":["Tev1Model"]}`))
Expect(lo.maxNumSeqs).To(Equal(int32(2)))
})
It("accepts a pre-encoded JSON string", func() {
lo := parseOptions(&pb.ModelOptions{
EngineArgs: `{"hf_overrides":"{\"architectures\":[\"Tev1Model\"]}"}`,
})
Expect(lo.hfOverrides).To(MatchJSON(`{"architectures":["Tev1Model"]}`))
})
})
Describe("config overlay", func() {
It("writes the merged config.json and symlinks every other entry to the original", func() {
overlay, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"],"new_key":1}`)
Expect(err).ToNot(HaveOccurred())
DeferCleanup(os.RemoveAll, overlay)
Expect(overlay).ToNot(Equal(modelDir))
raw, err := os.ReadFile(filepath.Join(overlay, "config.json"))
Expect(err).ToNot(HaveOccurred())
Expect(raw).To(MatchJSON(`{"architectures":["Tev1Model"],"hidden_size":2560,"text_config":{"num_hidden_layers":36},"new_key":1}`))
info, err := os.Lstat(filepath.Join(overlay, "config.json"))
Expect(err).ToNot(HaveOccurred())
Expect(info.Mode() & os.ModeSymlink).To(BeZero())
for _, name := range []string{"model.safetensors", "tokenizer.json", "tokenizer"} {
target, err := os.Readlink(filepath.Join(overlay, name))
Expect(err).ToNot(HaveOccurred(), name)
Expect(target).To(Equal(filepath.Join(modelDir, name)))
}
// The subdir resolves through the link, so a tokenizer/ fallback
// in the engine still finds its files.
Expect(filepath.Join(overlay, "tokenizer", "vocab.json")).To(BeARegularFile())
entries, err := os.ReadDir(overlay)
Expect(err).ToNot(HaveOccurred())
Expect(entries).To(HaveLen(4))
})
It("leaves the original config.json untouched", func() {
overlay, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"]}`)
Expect(err).ToNot(HaveOccurred())
DeferCleanup(os.RemoveAll, overlay)
raw, err := os.ReadFile(filepath.Join(modelDir, "config.json"))
Expect(err).ToNot(HaveOccurred())
Expect(string(raw)).To(Equal(originalConfig))
})
It("is removed by Free", func() {
overlay, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"]}`)
Expect(err).ToNot(HaveOccurred())
Expect(overlay).To(BeADirectory())
v := &VllmCpp{overlayDir: overlay}
Expect(v.Free()).To(Succeed())
Expect(overlay).ToNot(BeAnExistingFile())
Expect(v.overlayDir).To(BeEmpty())
// The originals the links pointed at survive the cleanup.
Expect(filepath.Join(modelDir, "model.safetensors")).To(BeARegularFile())
Expect(filepath.Join(modelDir, "tokenizer", "vocab.json")).To(BeARegularFile())
})
DescribeTable("refuses bad input",
func(overrides string, useFile bool, substr string) {
target := modelDir
if useFile {
target = filepath.Join(modelDir, "model.safetensors")
}
overlay, err := newConfigOverlay(target, overrides)
Expect(err).To(MatchError(ContainSubstring(substr)))
Expect(overlay).To(BeEmpty())
},
Entry("a JSON array", `["Tev1Model"]`, false, "must be a JSON object"),
Entry("a JSON scalar", `5`, false, "must be a JSON object"),
Entry("unparseable JSON", `{"architectures":`, false, "must be a JSON object"),
Entry("a model that is not a directory", `{"architectures":["Tev1Model"]}`, true, "model directory"),
)
It("refuses a directory without config.json", func() {
Expect(os.Remove(filepath.Join(modelDir, "config.json"))).To(Succeed())
_, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"]}`)
Expect(err).To(MatchError(ContainSubstring("config.json")))
})
It("does not leave a half-built overlay behind on failure", func() {
Expect(os.WriteFile(filepath.Join(modelDir, "config.json"), []byte("not json"), 0o644)).To(Succeed())
before, _ := filepath.Glob(filepath.Join(os.TempDir(), "vllm-cpp-hf-overrides-*"))
_, err := newConfigOverlay(modelDir, `{"architectures":["Tev1Model"]}`)
Expect(err).To(HaveOccurred())
after, _ := filepath.Glob(filepath.Join(os.TempDir(), "vllm-cpp-hf-overrides-*"))
Expect(after).To(ConsistOf(before))
})
})
It("keeps the merged document a valid object when overrides replace a nested key", func() {
overlay, err := newConfigOverlay(modelDir, `{"text_config":{"num_hidden_layers":2}}`)
Expect(err).ToNot(HaveOccurred())
DeferCleanup(os.RemoveAll, overlay)
raw, err := os.ReadFile(filepath.Join(overlay, "config.json"))
Expect(err).ToNot(HaveOccurred())
var doc map[string]any
Expect(json.Unmarshal(raw, &doc)).To(Succeed())
Expect(doc["text_config"]).To(Equal(map[string]any{"num_hidden_layers": float64(2)}))
})
})
+10
View File
@@ -74,6 +74,10 @@ type loadOptions struct {
nerThreshold float32
// nerMaxWidth is the maximum span width in tokens (0 = engine default 12).
nerMaxWidth int32
// hf_overrides (vLLM parity): a JSON object of config.json keys merged
// over the model directory's config.json through a private overlay dir,
// see newConfigOverlay. Empty = load the directory as is.
hfOverrides string
}
// videoOptions is the MiniMax-H3 checkpoint SET plus its generation defaults.
@@ -208,6 +212,8 @@ func applyOptionsList(lo *loadOptions, options []string) {
lo.kvTransferConfig = strings.TrimSpace(v)
case "tokenizer_config", "tokenizer_config_path":
lo.tokenizerConfigPath = strings.TrimSpace(v)
case "hf_overrides":
lo.hfOverrides = strings.TrimSpace(v)
case "enable_prefix_caching", "enable_radix_attention":
if b, err := strconv.ParseBool(strings.TrimSpace(v)); err == nil {
lo.enablePrefixCaching = boolTriState(b)
@@ -343,6 +349,10 @@ func applyEngineArgs(lo *loadOptions, engineArgs string) {
lo.speculativeConfig = jsonDocument(v, lo.speculativeConfig, k)
case "kv_transfer_config":
lo.kvTransferConfig = jsonDocument(v, lo.kvTransferConfig, k)
case "hf_overrides":
// Kept verbatim even when it is not an object: Load refuses a
// malformed value instead of loading the unmodified config.
lo.hfOverrides = jsonDocument(v, lo.hfOverrides, k)
case "enable_prefix_caching", "enable_radix_attention":
if b, ok := v.(bool); ok {
lo.enablePrefixCaching = boolTriState(b)
+2 -2
View File
@@ -16,7 +16,7 @@ func TestVllmCpp(t *testing.T) {
RunSpecs(t, "vllm-cpp suite")
}
// The Go POD mirrors must match the C struct layout of vllm.h (ABI v29)
// The Go POD mirrors must match the C struct layout of vllm.h (ABI v30)
// byte-for-byte: these offsets are the C offsets on LP64 (linux/darwin
// amd64+arm64). A failure here means govllmcpp.go drifted from vllm.h.
var _ = Describe("C ABI struct mirrors", func() {
@@ -24,7 +24,7 @@ var _ = Describe("C ABI struct mirrors", func() {
// VLLM_ABI_VERSION in the vllm.h of VLLM_CPP_VERSION (Makefile).
// Moving the pin past this without growing the mirrors below ships a
// backend that refuses every load at startup (issue #11379).
Expect(abiVersion).To(Equal(29))
Expect(abiVersion).To(Equal(30))
})
It("cModelParams matches vllm_model_params", func() {
+1 -1
View File
@@ -8,7 +8,7 @@ JOBS?=$(shell nproc --ignore=1)
# whisper.cpp version
WHISPER_REPO?=https://github.com/ggml-org/whisper.cpp
WHISPER_CPP_VERSION?=d09f61a708f3487afa956ff578e60eae5e7a233c
WHISPER_CPP_VERSION?=6e4ab854f67f743900934a703d5603419384c961
SO_TARGET?=libgowhisper.so
CMAKE_ARGS+=-DBUILD_SHARED_LIBS=OFF
+4 -3
View File
@@ -213,9 +213,10 @@ func transcriptResultFromProto(r *proto.TranscriptResult) *schema.TranscriptionR
var words []schema.TranscriptionWord
for _, w := range s.Words {
var word = schema.TranscriptionWord{
Start: time.Duration(w.Start),
End: time.Duration(w.End),
Text: w.Text,
Start: time.Duration(w.Start),
End: time.Duration(w.End),
Text: w.Text,
Speaker: w.Speaker,
}
words = append(words, word)
tr.Words = append(tr.Words, word)
+47 -8
View File
@@ -26,11 +26,33 @@ import (
// backchannel ("uh-huh") ended — callers must NOT treat Eob as a turn
// boundary.
type LiveTranscriptionEvent struct {
Delta string
Eou bool
Eob bool
Words []schema.TranscriptionWord
Final *schema.TranscriptionResult
Delta string
Eou bool
Eob bool
Words []schema.TranscriptionWord
Speakers []LiveSpeakerSegment
Sounds []LiveSoundEvent
Final *schema.TranscriptionResult
}
// LiveSpeakerSegment is one closed speaker segment from a companion
// diarization/scene stream running alongside live transcription. Start/End
// are stream-relative seconds (mapped from the backend's nanoseconds).
type LiveSpeakerSegment struct {
Speaker string
Start float64
End float64
}
// LiveSoundEvent is one closed sound event from a companion sound/scene
// stream running alongside live transcription. Start/End are stream-relative
// seconds (mapped from the backend's nanoseconds).
type LiveSoundEvent struct {
Label string
Index int
Peak float32
Start float64
End float64
}
// LiveTranscriptionSession is a handle on an open live transcription stream.
@@ -298,9 +320,26 @@ func liveEventFromProto(r *proto.TranscriptLiveResponse) LiveTranscriptionEvent
}
for _, w := range r.GetWords() {
ev.Words = append(ev.Words, schema.TranscriptionWord{
Start: time.Duration(w.Start),
End: time.Duration(w.End),
Text: w.Text,
Start: time.Duration(w.Start),
End: time.Duration(w.End),
Text: w.Text,
Speaker: w.Speaker,
})
}
for _, s := range r.GetSpeakers() {
ev.Speakers = append(ev.Speakers, LiveSpeakerSegment{
Speaker: s.GetSpeaker(),
Start: time.Duration(s.GetStart()).Seconds(),
End: time.Duration(s.GetEnd()).Seconds(),
})
}
for _, s := range r.GetSounds() {
ev.Sounds = append(ev.Sounds, LiveSoundEvent{
Label: s.GetLabel(),
Index: int(s.GetIndex()),
Peak: s.GetPeak(),
Start: time.Duration(s.GetStart()).Seconds(),
End: time.Duration(s.GetEnd()).Seconds(),
})
}
if r.GetFinalResult() != nil {
@@ -54,11 +54,53 @@ var _ = Describe("liveEventFromProto", func() {
Expect(ev.Final).To(BeNil())
})
It("carries word speakers and final segment speakers from a diarizing backend", func() {
ev := liveEventFromProto(&proto.TranscriptLiveResponse{
Words: []*proto.TranscriptWord{{Text: "hi", Speaker: "1"}},
})
Expect(ev.Words[0].Speaker).To(Equal("1"))
ev = liveEventFromProto(&proto.TranscriptLiveResponse{
FinalResult: &proto.TranscriptResult{
Text: "hi there",
Segments: []*proto.TranscriptSegment{{Text: "hi", Speaker: "0"}, {Text: "there", Speaker: "1"}},
},
})
Expect(ev.Final.Segments[1].Speaker).To(Equal("1"))
})
It("maps the eob backchannel flag separately from eou", func() {
ev := liveEventFromProto(&proto.TranscriptLiveResponse{Delta: "uh-huh", Eob: true})
Expect(ev.Eob).To(BeTrue())
Expect(ev.Eou).To(BeFalse())
})
It("maps speaker segments and sound events (ns -> seconds)", func() {
ev := liveEventFromProto(&proto.TranscriptLiveResponse{
Speakers: []*proto.LiveSpeakerSegment{
{Speaker: "1", Start: int64(1500 * time.Millisecond), End: int64(3200 * time.Millisecond)},
},
Sounds: []*proto.LiveSoundEvent{
{Label: "Dog bark", Index: 5, Peak: 0.8, Start: int64(500 * time.Millisecond), End: int64(900 * time.Millisecond)},
},
})
Expect(ev.Speakers).To(HaveLen(1))
Expect(ev.Speakers[0].Speaker).To(Equal("1"))
Expect(ev.Speakers[0].Start).To(BeNumerically("~", 1.5, 1e-9))
Expect(ev.Speakers[0].End).To(BeNumerically("~", 3.2, 1e-9))
Expect(ev.Sounds).To(HaveLen(1))
Expect(ev.Sounds[0].Label).To(Equal("Dog bark"))
Expect(ev.Sounds[0].Index).To(Equal(5))
Expect(ev.Sounds[0].Peak).To(BeNumerically("~", 0.8, 1e-6))
Expect(ev.Sounds[0].Start).To(BeNumerically("~", 0.5, 1e-9))
Expect(ev.Sounds[0].End).To(BeNumerically("~", 0.9, 1e-9))
})
It("leaves speakers and sounds nil when the proto carries none", func() {
ev := liveEventFromProto(&proto.TranscriptLiveResponse{Delta: "hi"})
Expect(ev.Speakers).To(BeNil())
Expect(ev.Sounds).To(BeNil())
})
})
// liveTraceState is what makes streaming-only pipelines visible on the
+1 -1
View File
@@ -314,7 +314,7 @@ func agentOptions(dir, model string, opts Options) app.Options {
TraceDir: opts.TraceDir,
}
if opts.Yolo {
overrides.ApprovalMode = "auto"
overrides.ApprovalMode = nibtypes.ApprovalAuto
}
return app.Options{
+3 -3
View File
@@ -509,7 +509,7 @@ var _ = Describe("prepare", func() {
Expect(agentOptions(dir, "a-model", opts).Overrides.ApprovalMode).To(BeEmpty())
opts.Yolo = true
Expect(agentOptions(dir, "a-model", opts).Overrides.ApprovalMode).To(Equal("auto"))
Expect(agentOptions(dir, "a-model", opts).Overrides.ApprovalMode).To(Equal(nibtypes.ApprovalAuto))
})
// The specs above pin what is handed over. These pin what nib does with
@@ -570,7 +570,7 @@ var _ = Describe("prepare", func() {
opts.Yolo = true
cfg := resolve(agentOptions(dir, "a-model", opts))
Expect(cfg.ApprovalMode).To(Equal("auto"))
Expect(cfg.ApprovalMode).To(Equal(nibtypes.ApprovalAuto))
})
// The other half of the same rule, and the reason an unset flag is
@@ -582,7 +582,7 @@ var _ = Describe("prepare", func() {
cfg := resolve(agentOptions(dir, "a-model", optionsWithStreams(os.Stdin, os.Stdout, os.Stderr)))
Expect(cfg.APIKey).To(Equal("saved-key"))
Expect(cfg.ApprovalMode).To(Equal("prompt"))
Expect(cfg.ApprovalMode).To(Equal(nibtypes.ApprovalPrompt))
})
})
})
+8 -6
View File
@@ -93,18 +93,20 @@ func (t *TranscriptCMD) Run(ctx *cliContext.Context) error {
}
for _, word := range(tr.Words) {
trs.Words = append(trs.Words, schema.TranscriptionWordSeconds{
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Speaker: word.Speaker,
})
}
for _, seg := range(tr.Segments) {
segWords := []schema.TranscriptionWordSeconds{}
for _, word := range(seg.Words) {
segWords = append(segWords, schema.TranscriptionWordSeconds{
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Speaker: word.Speaker,
})
}
trs.Segments = append(trs.Segments, schema.TranscriptionSegmentSeconds{
+15 -5
View File
@@ -35,6 +35,7 @@ const (
UsecaseSpeakerRecognition = "speaker_recognition"
UsecaseTokenClassify = "token_classify"
UsecaseScore = "score"
UsecaseDecisions = "decisions"
)
// GRPCMethod identifies a Backend service RPC from backend.proto.
@@ -216,6 +217,11 @@ var UsecaseInfoMap = map[string]UsecaseInfo{
GRPCMethod: MethodScore,
Description: "Joint log-probability scoring of candidate continuations via the Score RPC. Declared explicitly via known_usecases and usable alongside generation usecases.",
},
UsecaseDecisions: {
Flag: FLAG_DECISIONS,
GRPCMethod: MethodScore,
Description: "Decision models (served by POST /v1/systemone): typed choice, noul and score questions over a state text, answered by a non-generative decision model through the Score RPC (question_type systemone). Declared explicitly via known_usecases.",
},
}
// BackendCapability describes which gRPC methods and usecases a backend supports.
@@ -349,10 +355,10 @@ var BackendCapabilities = map[string]BackendCapability{
// model returns an error rather than silent garbage.
"vllm-cpp": {
GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateVideo, MethodTokenClassify, MethodScore},
PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVideo, UsecaseTokenClassify, UsecaseScore},
PossibleUsecases: []string{UsecaseChat, UsecaseCompletion, UsecaseVision, UsecaseVideo, UsecaseTokenClassify, UsecaseScore, UsecaseDecisions},
DefaultUsecases: []string{UsecaseChat},
AcceptsImages: true,
Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and kev/laya decision pipelines",
Description: "vllm.cpp — the LocalAI team's C++20 port of vLLM; text generation, MiniMax-H3 video+audio generation, GLiNER2.5 zero-shot NER, cua-s1-forms scoring, and decision models (kev, laya, CLM, GLiNER2.5-Decide, xor, nimble)",
},
"vllm-omni": {
GRPCMethods: []GRPCMethod{MethodPredict, MethodPredictStream, MethodGenerateImage, MethodGenerateVideo, MethodTTS},
@@ -477,11 +483,15 @@ var BackendCapabilities = map[string]BackendCapability{
DefaultUsecases: []string{UsecaseTranscript},
Description: "NVIDIA NeMo speech recognition",
},
// parakeet-cpp loads three model kinds, picked from the GGUF: an ASR model
// transcribes (and labels speakers when a diarization_model companion is
// attached), a Nemotron-3-Diarization model answers Diarize, and a CED model
// answers SoundDetection. PossibleUsecases is their union.
"parakeet-cpp": {
GRPCMethods: []GRPCMethod{MethodAudioTranscription},
PossibleUsecases: []string{UsecaseTranscript},
GRPCMethods: []GRPCMethod{MethodAudioTranscription, MethodDiarize, MethodSoundDetection},
PossibleUsecases: []string{UsecaseTranscript, UsecaseDiarization, UsecaseSoundClassification},
DefaultUsecases: []string{UsecaseTranscript},
Description: "NVIDIA NeMo Parakeet ASR (parakeet.cpp)",
Description: "NVIDIA NeMo Parakeet ASR, Nemotron-3-Diarization speaker diarization and CED sound-event detection (parakeet.cpp)",
},
// nemo-speech-cpp is one gRPC server in front of four NeMo-Speech.cpp model
// families, picked at load time from the GGUF general.architecture key, so
+2 -2
View File
@@ -16,14 +16,14 @@ import (
// reservedNonChatModel reports whether the operator reserved this model for an
// internal primitive — the router score classifier or the PII NER
// token_classify tier. Such a model has no chat template and must not be
// token_classify tier, or a decision head. Such a model has no chat template and must not be
// given the generative-chat defaults the GGUF importer otherwise applies
// (FLAG_CHAT, jinja templating): surfacing it in chat pickers defeats the
// reservation. Operators who do want a combined model declare both usecases
// explicitly — the combination is valid.
func reservedNonChatModel(cfg *ModelConfig) bool {
return cfg.KnownUsecases != nil &&
(*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY)) != 0
(*cfg.KnownUsecases&(FLAG_SCORE|FLAG_TOKEN_CLASSIFY|FLAG_DECISIONS)) != 0
}
// genAudioEncoderKey is the mmproj metadata flag llama.cpp's mtmd writes for a
+7
View File
@@ -509,6 +509,13 @@ func DefaultRegistry() map[string]FieldMetaOverride {
Min: f64(0),
Order: 66,
},
"pipeline.diarization": {
Section: "pipeline",
Label: "Speaker Diarization",
Description: "Label speakers on each committed utterance and emit every labelled segment as a conversation.item.input_audio_transcription.segment event. Needs a transcription model that diarizes (e.g. parakeet-cpp with a diarization_model companion). Speaker labels are per turn.",
Component: "toggle",
Order: 67,
},
"pipeline.reasoning_effort": {
Section: "pipeline",
Label: "Reasoning Effort",
+28 -4
View File
@@ -833,6 +833,14 @@ type Pipeline struct {
SoundDetectionWindowMs int `yaml:"sound_detection_window_ms,omitempty" json:"sound_detection_window_ms,omitempty"`
SoundDetectionHopMs int `yaml:"sound_detection_hop_ms,omitempty" json:"sound_detection_hop_ms,omitempty"`
// Diarization asks the transcription model for speaker labels on each
// VAD-committed utterance and emits every labelled segment as a
// conversation.item.input_audio_transcription.segment event. It needs a
// transcription model that diarizes (e.g. parakeet-cpp with a
// diarization_model companion); off by default because some backends fail
// a diarize request they cannot serve. Speaker labels are per turn.
Diarization bool `yaml:"diarization,omitempty" json:"diarization,omitempty"`
// ReasoningEffort sets the reasoning effort (none|minimal|low|medium|high) for
// the pipeline's LLM without editing the LLM model config. Overrides the LLM's
// own reasoning_effort. Unset leaves the LLM model config in charge.
@@ -2048,6 +2056,13 @@ const (
FLAG_3D ModelConfigUsecase = 0b100000000000000000000000
FLAG_3D_ANIMATION ModelConfigUsecase = 1 << 24
// Marks a model as a decision model: it answers typed choice / noul /
// score questions over a state (served by POST /v1/systemone).
// Explicit only, like FLAG_SCORE: a decision model never generates
// text, so guessing chat or embeddings for it would surface it in
// pickers it cannot serve.
FLAG_DECISIONS ModelConfigUsecase = 1 << 25
// Common Subsets
FLAG_LLM ModelConfigUsecase = FLAG_CHAT | FLAG_COMPLETION | FLAG_EDIT
)
@@ -2110,6 +2125,7 @@ func GetAllModelConfigUsecases() map[string]ModelConfigUsecase {
"FLAG_TOKEN_CLASSIFY": FLAG_TOKEN_CLASSIFY,
"FLAG_3D": FLAG_3D,
"FLAG_3D_ANIMATION": FLAG_3D_ANIMATION,
"FLAG_DECISIONS": FLAG_DECISIONS,
}
}
@@ -2138,9 +2154,9 @@ func GetUsecasesFromYAML(input []string) *ModelConfigUsecase {
//
// Declared known_usecases are normally additive — the guessing heuristic
// still adds whatever it can infer from backend/templates. The exceptions
// are FLAG_SCORE and FLAG_TOKEN_CLASSIFY: when the operator declared
// either, they reserved the model for an internal direct-decode primitive
// (the router classifier, or the PII NER tier). Letting GuessUsecases
// are FLAG_SCORE, FLAG_TOKEN_CLASSIFY and FLAG_DECISIONS: when the operator
// declared any of them, they reserved the model for a direct-decode primitive
// (the router classifier, the PII NER tier, or a decision head). Letting GuessUsecases
// paint chat/completion/embeddings on top would surface it in pickers it
// was deliberately kept out of. So a declared score or token_classify
// list is authoritative; declare the generation usecases explicitly
@@ -2150,7 +2166,7 @@ func (c *ModelConfig) HasUsecases(u ModelConfigUsecase) bool {
if (u & *c.KnownUsecases) == u {
return true
}
if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY)) != 0 {
if (*c.KnownUsecases & (FLAG_SCORE | FLAG_TOKEN_CLASSIFY | FLAG_DECISIONS)) != 0 {
return false
}
}
@@ -2373,6 +2389,14 @@ func (c *ModelConfig) GuessUsecases(u ModelConfigUsecase) bool {
return false
}
if (u & FLAG_DECISIONS) == FLAG_DECISIONS {
// No heuristic: decisions intent is a deliberate operator choice
// (the model is a non-generative decision head), so
// HasUsecases(FLAG_DECISIONS) is true only when KnownUsecases
// declares it explicitly.
return false
}
return true
}
+33
View File
@@ -955,3 +955,36 @@ var _ = Describe("ModelConfig alias", func() {
Expect(err).To(MatchError(ContainSubstring("alias")))
})
})
var _ = Describe("decisions usecase", func() {
// A decision model never generates text, so a declared decisions list
// must stay authoritative and the heuristic must never guess the flag.
It("is authoritative when declared and never guessed", func() {
declared := GetUsecasesFromYAML([]string{"decisions"})
Expect(declared).NotTo(BeNil())
Expect(*declared).NotTo(Equal(FLAG_ANY))
cfg := ModelConfig{
Name: "laya",
Backend: "vllm-cpp",
KnownUsecases: declared,
TemplateConfig: TemplateConfig{
Chat: "inherited from chatml",
ChatMessage: "inherited from chatml",
Completion: "inherited from chatml",
},
}
Expect(cfg.HasUsecases(*declared)).To(BeTrue())
Expect(cfg.HasUsecases(FLAG_CHAT)).To(BeFalse())
Expect(cfg.HasUsecases(FLAG_COMPLETION)).To(BeFalse())
Expect(cfg.HasUsecases(FLAG_EMBEDDINGS)).To(BeFalse())
undeclared := ModelConfig{Name: "laya", Backend: "vllm-cpp"}
Expect(undeclared.HasUsecases(*declared)).To(BeFalse())
})
It("is a reserved usecase for the GGUF importer chat-default guard", func() {
declared := GetUsecasesFromYAML([]string{"decisions"})
Expect(reservedNonChatModel(&ModelConfig{Backend: "vllm-cpp", KnownUsecases: declared})).To(BeTrue())
})
})
+28 -3
View File
@@ -108,6 +108,11 @@ func (i *ParakeetCppImporter) Import(details Details) (gallery.ModelConfig, erro
uri := downloader.URI(details.URI)
directGGUF := isParakeetGGUF(filepath.Base(details.URI))
// A speaker diarization GGUF is served by the same backend but answers
// /v1/audio/diarization, not transcription.
if directGGUF && isParakeetDiarGGUF(filepath.Base(details.URI)) {
modelConfig.KnownUsecaseStrings = []string{"diarization"}
}
switch {
case uri.LooksLikeURL() && directGGUF:
// Direct file URL (e.g. .../resolve/main/tdt_ctc-110m-f16.gguf). The
@@ -128,12 +133,23 @@ func (i *ParakeetCppImporter) Import(details Details) (gallery.ModelConfig, erro
// HF repo: collect every parakeet GGUF, pick the preferred quant, and
// nest under parakeet-cpp/models/<name>/ so a multi-quant repo doesn't
// collide on disk.
var ggufFiles []hfapi.ModelFile
// Prefer ASR weights: a repo that also ships the diarization model
// (mudler/parakeet-cpp-gguf) imports as a transcription model, and the
// diarization GGUF is imported by its direct URL. A repo with only
// diarization weights imports as a diarization model.
var ggufFiles, diarFiles []hfapi.ModelFile
for _, f := range details.HuggingFace.Files {
if isParakeetGGUF(filepath.Base(f.Path)) {
switch base := filepath.Base(f.Path); {
case isParakeetDiarGGUF(base):
diarFiles = append(diarFiles, f)
case isParakeetGGUF(base):
ggufFiles = append(ggufFiles, f)
}
}
if len(ggufFiles) == 0 && len(diarFiles) > 0 {
ggufFiles = diarFiles
modelConfig.KnownUsecaseStrings = []string{"diarization"}
}
if chosen, ok := pickPreferredGGMLFile(ggufFiles, quants); ok {
target := filepath.Join("parakeet-cpp", "models", name, filepath.Base(chosen.Path))
cfg.Files = append(cfg.Files, gallery.File{
@@ -176,5 +192,14 @@ func isParakeetGGUF(name string) bool {
return true
}
}
return false
return isParakeetDiarGGUF(name)
}
// isParakeetDiarGGUF reports whether name is the parakeet.cpp speaker
// diarization GGUF (nemotron-3-diarization-<quant>.gguf). Matched by its
// published name only, so diarization weights for other backends are not
// claimed.
func isParakeetDiarGGUF(name string) bool {
lower := strings.ToLower(name)
return strings.HasSuffix(lower, ".gguf") && strings.Contains(lower, "nemotron-3-diarization")
}
@@ -50,6 +50,11 @@ var _ = Describe("ParakeetCppImporter", func() {
Expect(imp.Match(d)).To(BeTrue())
})
It("matches a direct URL to the diarization GGUF", func() {
d := parakeetDetails("https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/nemotron-3-diarization-q8_0.gguf", `{}`)
Expect(imp.Match(d)).To(BeTrue())
})
It("does NOT claim a generic llama-style GGUF", func() {
d := parakeetDetails("huggingface://someorg/some-llm-gguf", `{}`,
hfapi.ModelFile{Path: "llama-3-8b-instruct-q4_k_m.gguf"},
@@ -66,6 +71,43 @@ var _ = Describe("ParakeetCppImporter", func() {
})
Context("import (Import)", func() {
It("imports the diarization GGUF as a diarization model", func() {
d := parakeetDetails("https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/nemotron-3-diarization-q8_0.gguf",
`{"name":"nemotron-diarization"}`)
cfg, err := imp.Import(d)
Expect(err).ToNot(HaveOccurred())
Expect(cfg.ConfigFile).To(ContainSubstring("backend: parakeet-cpp"))
Expect(cfg.ConfigFile).To(ContainSubstring("diarization"))
Expect(cfg.ConfigFile).ToNot(ContainSubstring("transcript"))
Expect(cfg.Files).To(HaveLen(1))
Expect(cfg.Files[0].Filename).To(HaveSuffix("nemotron-3-diarization-q8_0.gguf"))
})
It("keeps picking ASR weights from a repo that also ships the diarization model", func() {
d := parakeetDetails("huggingface://mudler/parakeet-cpp-gguf", `{"name":"parakeet-110m"}`,
hfapi.ModelFile{Path: "nemotron-3-diarization-f16.gguf", URL: "https://hf/diar-f16", SHA256: "ddd"},
hfapi.ModelFile{Path: "tdt_ctc-110m-f16.gguf", URL: "https://hf/f16", SHA256: "aaa"},
hfapi.ModelFile{Path: "nemotron-3-diarization-q8_0.gguf", URL: "https://hf/diar-q8", SHA256: "eee"},
)
cfg, err := imp.Import(d)
Expect(err).ToNot(HaveOccurred())
Expect(cfg.Files).To(HaveLen(1))
Expect(cfg.Files[0].URI).To(Equal("https://hf/f16"))
Expect(cfg.ConfigFile).To(ContainSubstring("transcript"))
})
It("imports a diarization-only repo as a diarization model", func() {
d := parakeetDetails("huggingface://someone/diar-gguf", `{"name":"diar"}`,
hfapi.ModelFile{Path: "nemotron-3-diarization-f16.gguf", URL: "https://hf/diar-f16", SHA256: "ddd"},
hfapi.ModelFile{Path: "nemotron-3-diarization-q8_0.gguf", URL: "https://hf/diar-q8", SHA256: "eee"},
)
cfg, err := imp.Import(d)
Expect(err).ToNot(HaveOccurred())
Expect(cfg.Files).To(HaveLen(1))
Expect(cfg.Files[0].URI).To(Equal("https://hf/diar-q8"), "default quant ladder picks q8_0 before f16")
Expect(cfg.ConfigFile).To(ContainSubstring("diarization"))
})
It("picks the default quant (q4_k) from a multi-quant HF repo", func() {
d := parakeetDetails("huggingface://mudler/parakeet-cpp-gguf", `{"name":"parakeet-110m"}`,
hfapi.ModelFile{Path: "tdt_ctc-110m-f16.gguf", URL: "https://hf/f16", SHA256: "aaa"},
+73
View File
@@ -0,0 +1,73 @@
package gallery_test
import (
"fmt"
"os"
"path/filepath"
"slices"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"gopkg.in/yaml.v3"
"github.com/mudler/LocalAI/core/config"
)
// A gallery tag that names a capability is what users filter on, and
// known_usecases is what the server routes on. When they disagree, the entry
// is listed under a filter it cannot serve, or is hidden from one it can.
var _ = Describe("gallery/index.yaml vllm-cpp capability tags", func() {
It("keeps capability tags and known_usecases in agreement", func() {
entries, err := loadGalleryIndex()
Expect(err).ToNot(HaveOccurred())
tagToFlag := map[string]config.ModelConfigUsecase{
"decisions": config.FLAG_DECISIONS,
"vision": config.FLAG_VISION,
"token-classify": config.FLAG_TOKEN_CLASSIFY,
"scoring": config.FLAG_SCORE,
}
var violations []string
seen := 0
for i := range entries {
e := &entries[i]
if backend, _ := e.Overrides["backend"].(string); backend != "vllm-cpp" {
continue
}
seen++
declared := e.GetKnownUsecases()
for tag, flag := range tagToFlag {
tagged := slices.Contains(e.Tags, tag)
has := declared != nil && *declared&flag == flag
if tagged != has {
violations = append(violations, fmt.Sprintf("%s: tag %q present=%v but known_usecases declares it=%v", e.Name, tag, tagged, has))
}
}
}
Expect(seen).To(BeNumerically(">", 0))
Expect(violations).To(BeEmpty())
})
})
// artifacts: is a model-config key, so the installer only sees it inside
// overrides:. At the top level of an entry it is silently dropped, the
// installed config keeps a bare HF repo id as its model, and a backend that
// does not infer artifacts (vllm-cpp among them) fails the first load with
// "model path not found" while the install itself reported success.
var _ = Describe("gallery/index.yaml artifacts placement", func() {
It("declares artifacts under overrides, never at the entry top level", func() {
data, err := os.ReadFile(filepath.Join("..", "..", "gallery", "index.yaml"))
Expect(err).ToNot(HaveOccurred())
var raw []map[string]any
Expect(yaml.Unmarshal(data, &raw)).To(Succeed())
var misplaced []string
for _, e := range raw {
if _, ok := e["artifacts"]; ok {
misplaced = append(misplaced, fmt.Sprint(e["name"]))
}
}
Expect(misplaced).To(BeEmpty())
})
})
+6
View File
@@ -71,6 +71,11 @@ var RouteFeatureRegistry = []RouteFeature{
// Detection
{"POST", "/v1/detection", FeatureDetection},
// Decisions API (SystemOne wire contract)
{"POST", "/v1/systemone", FeatureDecisions},
{"POST", "/v1/systemone/permute", FeatureDecisions},
{"POST", "/v1/systemone/separate", FeatureDecisions},
// Face recognition
{"POST", "/v1/face/verify", FeatureFaceRecognition},
{"POST", "/v1/face/analyze", FeatureFaceRecognition},
@@ -209,5 +214,6 @@ func APIFeatureMetas() []FeatureMeta {
{FeatureVoiceRecognition, "Voice Recognition", true},
{FeatureAudioTransform, "Audio Transform", true},
{FeaturePIIFilter, "PII Analyze / Redact", true},
{FeatureDecisions, "Decisions", true},
}
}
+24
View File
@@ -0,0 +1,24 @@
package auth_test
import (
. "github.com/mudler/LocalAI/core/http/auth"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("Decisions feature registration", func() {
It("gates the three decision routes behind one default-on API feature", func() {
Expect(APIFeatures).To(ContainElement(FeatureDecisions))
patterns := []string{}
for _, route := range RouteFeatureRegistry {
if route.Feature == FeatureDecisions {
Expect(route.Method).To(Equal("POST"))
patterns = append(patterns, route.Pattern)
}
}
Expect(patterns).To(ConsistOf("/v1/systemone", "/v1/systemone/permute", "/v1/systemone/separate"))
Expect(APIFeatureMetas()).To(ContainElement(FeatureMeta{Key: FeatureDecisions, Label: "Decisions", DefaultValue: true}))
})
})
+2 -1
View File
@@ -59,6 +59,7 @@ const (
FeatureFaceRecognition = "face_recognition"
FeatureVoiceRecognition = "voice_recognition"
FeatureAudioTransform = "audio_transform"
FeatureDecisions = "decisions"
// FeaturePIIFilter gates the synchronous PII analyze/redact service
// (POST /api/pii/{analyze,redact}). Default ON like the other API
// features; the admin-only events log is gated separately in-handler.
@@ -78,7 +79,7 @@ var APIFeatures = []string{
FeatureVAD, FeatureDetection, FeatureVideo, Feature3D, FeatureEmbeddings, FeatureSound,
FeatureRealtime, FeatureModeration, FeatureRerank, FeatureTokenize, FeatureMCP, FeatureStores,
FeatureFaceRecognition, FeatureVoiceRecognition, FeatureAudioTransform,
FeaturePIIFilter,
FeaturePIIFilter, FeatureDecisions,
}
// AllFeatures lists all known features (used by UI and validation).
@@ -105,6 +105,12 @@ var instructionDefs = []instructionDef{
Tags: []string{"voice-recognition"},
Intro: "Voice (speaker) recognition — the audio analog to /v1/face/*. Use /v1/voice/verify for 1:1 speaker comparison, /v1/voice/identify for 1:N match against the registered store, /v1/voice/{register,forget} to manage that store, /v1/voice/embed for a raw speaker-encoder vector, and /v1/voice/analyze for age / gender / emotion inferred from speech. Registrations are in-memory by default and lost on restart. Audio inputs accept URL, base64, or data-URI; /v1/embeddings remains text-only.",
},
{
Name: "decisions",
Description: "Typed decisions (choice, noul, score) over a state text with calibrated confidence",
Tags: []string{"systemone"},
Intro: "POST /v1/systemone answers every question in one pass; /v1/systemone/permute re-runs one choice question under n_perm option orders; /v1/systemone/separate answers each question in its own pass. Request: { model, state, questions: { <id>: { type: choice|noul|score, instructions, criteria } } }. A decision model declares known_usecases: [decisions] and serves only /v1/systemone; a zero-shot NER model declares token_classify and serves all three routes (through the NER path); /permute and /separate return 400 for decision models. A vllm-cpp config that declares no usecases is treated as a decision model. Responses carry per-question answers with confidence and probabilities plus token usage. Field names and question types follow Ollama's /v1/systemone, with differences in confidence, error shape and keep_alive (see the Decisions API docs). A request over 64 KiB, with more than 64 questions, or with a malformed question is refused.",
},
{
Name: "branding",
Description: "Whitelabel the instance: configure name, tagline, logo, and favicon",
@@ -39,7 +39,7 @@ var _ = Describe("API Instructions Endpoints", func() {
instructions, ok := resp["instructions"].([]any)
Expect(ok).To(BeTrue())
Expect(instructions).To(HaveLen(20))
Expect(instructions).To(HaveLen(21))
// Verify each instruction has required fields and correct URL format
for _, s := range instructions {
@@ -82,6 +82,7 @@ var _ = Describe("API Instructions Endpoints", func() {
"voice-library",
"3d",
"failover",
"decisions",
))
})
})
@@ -136,6 +137,17 @@ var _ = Describe("API Instructions Endpoints", func() {
Expect(string(body)).NotTo(ContainSubstring("/v1/3d/generations"))
})
It("should advertise the Decisions API", func() {
req := httptest.NewRequest(http.MethodGet, "/api/instructions/decisions", nil)
rec := httptest.NewRecorder()
app.ServeHTTP(rec, req)
Expect(rec.Code).To(Equal(http.StatusOK))
body, _ := io.ReadAll(rec.Body)
Expect(string(body)).To(ContainSubstring("POST /v1/systemone"))
Expect(string(body)).To(ContainSubstring("known_usecases: [decisions]"))
})
It("should return JSON fragment when format=json", func() {
req := httptest.NewRequest(http.MethodGet, "/api/instructions/chat-inference?format=json", nil)
rec := httptest.NewRecorder()
+217 -7
View File
@@ -2,6 +2,7 @@ package localai
import (
"encoding/json"
"errors"
"fmt"
"math"
"math/rand"
@@ -371,6 +372,191 @@ func systemOneError(c echo.Context, status int, msg string) error {
})
}
// systemOneModelAllowed keeps chat and embedding models out of the decision
// API with an actionable error instead of a backend failure. A config that
// declares no usecases predates the flag and stays allowed, and a
// token_classify model is allowed because the NER path serves it.
func systemOneModelAllowed(cfg config.ModelConfig) error {
if cfg.KnownUsecases == nil {
return nil
}
if *cfg.KnownUsecases&(config.FLAG_DECISIONS|config.FLAG_TOKEN_CLASSIFY) != 0 {
return nil
}
return fmt.Errorf("model %q does not declare the decisions usecase (known_usecases: [decisions])", cfg.Name)
}
// checkSystemOneModel applies systemOneModelAllowed to a model looked up by
// name. An unknown model passes here so the existing not-found handling
// downstream keeps its status code.
func checkSystemOneModel(app *application.Application, modelName string) error {
cl := app.ModelConfigLoader()
if cl == nil {
return nil
}
cfg, ok := cl.GetModelConfig(modelName)
if !ok {
return nil
}
return systemOneModelAllowed(cfg)
}
// systemOneUsesDecisionPipeline reports whether /v1/systemone forwards the
// request to the backend's Score RPC (the decision pipeline) for this model.
// A model that declares token_classify without systemone is a zero-shot NER
// model: the backend's decision entry point refuses those architectures, so it
// goes to the NER path instead. A config that declares nothing keeps the
// decision pipeline, which is what setups that predate the decisions usecase
// relied on.
func systemOneUsesDecisionPipeline(cfg config.ModelConfig) bool {
if !backendSupportsScore(cfg.Backend) {
return false
}
if cfg.KnownUsecases == nil {
return true
}
declared := *cfg.KnownUsecases
if declared&config.FLAG_DECISIONS != 0 {
return true
}
return declared&config.FLAG_TOKEN_CLASSIFY == 0
}
// systemOneNERAllowed guards /permute and /separate, which always run the NER
// path. A decision model cannot serve them: the backend's NER entry point
// refuses its architecture, and the caller would see a backend error.
func systemOneNERAllowed(cfg config.ModelConfig) error {
if cfg.KnownUsecases == nil {
return nil
}
declared := *cfg.KnownUsecases
if declared&config.FLAG_DECISIONS != 0 && declared&config.FLAG_TOKEN_CLASSIFY == 0 {
return fmt.Errorf("model %q is a decision model: /permute and /separate use the NER path, use POST /v1/systemone instead", cfg.Name)
}
return nil
}
// checkSystemOneNERModel applies systemOneNERAllowed to a model looked up by
// name; an unknown model passes so the not-found handling keeps its status.
func checkSystemOneNERModel(app *application.Application, modelName string) error {
cl := app.ModelConfigLoader()
if cl == nil {
return nil
}
cfg, ok := cl.GetModelConfig(modelName)
if !ok {
return nil
}
return systemOneNERAllowed(cfg)
}
// systemOneMaxBody and systemOneMaxQuestions bound one request. They keep a
// single call from pinning a decision model on an unbounded prompt, and match
// the limits Ollama documents for the same wire contract, so a client written
// for one server behaves the same on the other. The engine enforces any
// per-model option cap (letter-answer models refuse more than 26 options).
const (
systemOneMaxBody = 64 << 10
systemOneMaxQuestions = 64
)
// systemOneBind binds the JSON body with a size cap. Bind reads the whole body
// first, so the cap has to be on the reader.
func systemOneBind(c echo.Context, v any) error {
c.Request().Body = http.MaxBytesReader(c.Response(), c.Request().Body, systemOneMaxBody)
return c.Bind(v)
}
// systemOneBindStatus maps a bind failure to its status: 413 when the body
// exceeded the cap, 400 for anything else.
func systemOneBindStatus(err error) int {
var tooLarge *http.MaxBytesError
if errors.As(err, &tooLarge) {
return http.StatusRequestEntityTooLarge
}
return http.StatusBadRequest
}
func systemOneBindMessage(err error) string {
if systemOneBindStatus(err) == http.StatusRequestEntityTooLarge {
return fmt.Sprintf("request body exceeds %d KiB", systemOneMaxBody>>10)
}
return "invalid request body"
}
// validateSystemOneRequest checks the structure every path needs, before the
// request is forwarded to a decision model or run through the NER path. The
// forwarded path never sees parseSystemOneRequest, so without this a malformed
// question would surface as a backend error instead of a 400.
func validateSystemOneRequest(req *schema.SystemOneRequest) error {
if len(req.State) == 0 || string(req.State) == "null" {
return fmt.Errorf("state is required")
}
var state any
if err := json.Unmarshal(req.State, &state); err != nil {
return fmt.Errorf("state is not valid JSON: %w", err)
}
if s, ok := state.(string); ok && strings.TrimSpace(s) == "" {
return fmt.Errorf("state is required")
}
if len(req.Questions) == 0 {
return fmt.Errorf("questions is required and must contain at least one question")
}
if len(req.Questions) > systemOneMaxQuestions {
return fmt.Errorf("questions must contain at most %d questions", systemOneMaxQuestions)
}
qids := make([]string, 0, len(req.Questions))
for id := range req.Questions {
qids = append(qids, id)
}
sort.Strings(qids)
for _, id := range qids {
if strings.TrimSpace(id) == "" {
return fmt.Errorf("question ids must not be blank")
}
q := req.Questions[id]
switch q.Type {
case "choice":
var criteria map[string]json.RawMessage
if err := json.Unmarshal(q.Criteria, &criteria); err != nil {
return fmt.Errorf("question %q (choice) requires a criteria object", id)
}
if len(criteria) < 2 {
return fmt.Errorf("question %q (choice) requires at least 2 options", id)
}
for k := range criteria {
if strings.TrimSpace(k) == "" {
return fmt.Errorf("question %q (choice) has a blank option key", id)
}
}
case "score":
var criteria []json.RawMessage
if err := json.Unmarshal(q.Criteria, &criteria); err != nil {
return fmt.Errorf("question %q (score) requires a criteria array", id)
}
if len(criteria) < 2 {
return fmt.Errorf("question %q (score) requires at least 2 levels", id)
}
case "noul":
if len(q.Criteria) == 0 || string(q.Criteria) == "null" {
continue
}
var criteria map[string]json.RawMessage
if err := json.Unmarshal(q.Criteria, &criteria); err != nil {
return fmt.Errorf("question %q (noul) criteria must be an object with \"false\" and \"true\" descriptions", id)
}
for k := range criteria {
if k != "false" && k != "true" {
return fmt.Errorf("question %q (noul) criteria may only have \"false\" and \"true\" keys", id)
}
}
default:
return fmt.Errorf("question %q has unknown type: %s", id, q.Type)
}
}
return nil
}
// backendSupportsScore reports whether the named backend implements the
// Score gRPC RPC. vllm-cpp does (kev/laya decision pipeline and cua-s1-forms
// scoring via the unified vllm_decide C ABI); other backends fall through to
@@ -402,18 +588,24 @@ func backendSupportsScore(backendName string) bool {
func SystemOneEndpoint(app *application.Application) echo.HandlerFunc {
return func(c echo.Context) error {
var req schema.SystemOneRequest
if err := c.Bind(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, "invalid request body")
if err := systemOneBind(c, &req); err != nil {
return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err))
}
if req.Model == "" {
return systemOneError(c, http.StatusBadRequest, "model is required")
}
if err := checkSystemOneModel(app, req.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := validateSystemOneRequest(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
// vllm-cpp models (kev/laya) implement the decision pipeline natively
// via the vllm_decide C ABI. Forward the raw request JSON through the
// Score RPC and return the backend's response as-is.
cl := app.ModelConfigLoader()
if cl != nil {
if cfg, ok := cl.GetModelConfig(req.Model); ok && backendSupportsScore(cfg.Backend) {
if cfg, ok := cl.GetModelConfig(req.Model); ok && systemOneUsesDecisionPipeline(cfg) {
reqJSON, err := json.Marshal(req)
if err != nil {
return systemOneError(c, http.StatusInternalServerError, "failed to marshal request: "+err.Error())
@@ -468,12 +660,21 @@ func SystemOneEndpoint(app *application.Application) echo.HandlerFunc {
func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc {
return func(c echo.Context) error {
var req schema.SystemOnePermuteRequest
if err := c.Bind(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, "invalid request body")
if err := systemOneBind(c, &req); err != nil {
return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err))
}
if req.Request.Model == "" {
return systemOneError(c, http.StatusBadRequest, "model is required")
}
if err := checkSystemOneModel(app, req.Request.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := checkSystemOneNERModel(app, req.Request.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := validateSystemOneRequest(&req.Request); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if req.Question == "" {
return systemOneError(c, http.StatusBadRequest, "question is required")
}
@@ -604,12 +805,21 @@ func SystemOnePermuteEndpoint(app *application.Application) echo.HandlerFunc {
func SystemOneSeparateEndpoint(app *application.Application) echo.HandlerFunc {
return func(c echo.Context) error {
var req schema.SystemOneRequest
if err := c.Bind(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, "invalid request body")
if err := systemOneBind(c, &req); err != nil {
return systemOneError(c, systemOneBindStatus(err), systemOneBindMessage(err))
}
if req.Model == "" {
return systemOneError(c, http.StatusBadRequest, "model is required")
}
if err := checkSystemOneModel(app, req.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := checkSystemOneNERModel(app, req.Model); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
if err := validateSystemOneRequest(&req); err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
}
parsed, err := parseSystemOneRequest(&req)
if err != nil {
return systemOneError(c, http.StatusBadRequest, err.Error())
@@ -0,0 +1,74 @@
package localai
import (
"github.com/mudler/LocalAI/core/config"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("systemOneModelAllowed", func() {
mk := func(usecases ...string) config.ModelConfig {
return config.ModelConfig{
Name: "m",
Backend: "vllm-cpp",
KnownUsecases: config.GetUsecasesFromYAML(usecases),
}
}
It("accepts a declared decisions model", func() {
Expect(systemOneModelAllowed(mk("decisions"))).To(Succeed())
})
It("accepts a token_classify model, which the NER path serves", func() {
Expect(systemOneModelAllowed(mk("token_classify"))).To(Succeed())
})
It("keeps configs that declare no usecases working", func() {
Expect(systemOneModelAllowed(config.ModelConfig{Name: "laya", Backend: "vllm-cpp"})).To(Succeed())
})
It("refuses a chat-only model with an actionable message", func() {
Expect(systemOneModelAllowed(mk("chat"))).To(MatchError(ContainSubstring("known_usecases: [decisions]")))
})
})
var _ = Describe("systemone routing by model kind", func() {
mk := func(backend string, usecases ...string) config.ModelConfig {
c := config.ModelConfig{Name: "m", Backend: backend}
if len(usecases) > 0 {
c.KnownUsecases = config.GetUsecasesFromYAML(usecases)
}
return c
}
Describe("systemOneUsesDecisionPipeline", func() {
It("sends a declared decision model to the decision pipeline", func() {
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions"))).To(BeTrue())
})
It("sends a token_classify model to the NER path, since vllm_decide refuses NER architectures", func() {
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "token_classify"))).To(BeFalse())
})
It("keeps configs that declare nothing on the decision pipeline", func() {
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp"))).To(BeTrue())
})
It("prefers the decision pipeline when both usecases are declared", func() {
Expect(systemOneUsesDecisionPipeline(mk("vllm-cpp", "decisions", "token_classify"))).To(BeTrue())
})
It("never uses it for a backend without the Score RPC", func() {
Expect(systemOneUsesDecisionPipeline(mk("no-such-backend", "decisions"))).To(BeFalse())
})
})
Describe("systemOneNERAllowed", func() {
It("refuses a decision model on the NER-only routes with an actionable message", func() {
Expect(systemOneNERAllowed(mk("vllm-cpp", "decisions"))).To(MatchError(ContainSubstring("/v1/systemone")))
})
It("accepts a token_classify model", func() {
Expect(systemOneNERAllowed(mk("vllm-cpp", "token_classify"))).To(Succeed())
})
It("accepts configs that declare nothing", func() {
Expect(systemOneNERAllowed(mk("vllm-cpp"))).To(Succeed())
})
})
})
@@ -0,0 +1,96 @@
package localai
import (
"encoding/json"
"net/http"
"net/http/httptest"
"strings"
"github.com/labstack/echo/v4"
"github.com/mudler/LocalAI/core/schema"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("validateSystemOneRequest", func() {
req := func(state string, questions string) *schema.SystemOneRequest {
r := &schema.SystemOneRequest{Model: "m", State: json.RawMessage(state)}
Expect(json.Unmarshal([]byte(questions), &r.Questions)).To(Succeed())
return r
}
It("accepts the three question types", func() {
r := req(`"ticket text"`, `{
"team": {"type":"choice","instructions":"which","criteria":{"a":"A","b":null}},
"refund": {"type":"noul","instructions":"refund?","criteria":{"false":"No refund","true":"Refund asked"}},
"urgency": {"type":"score","instructions":"how urgent","criteria":["low","high"]}
}`)
Expect(validateSystemOneRequest(r)).To(Succeed())
})
It("accepts a noul question with no criteria", func() {
Expect(validateSystemOneRequest(req(`"x"`, `{"q":{"type":"noul","instructions":"i"}}`))).To(Succeed())
})
DescribeTable("refuses a malformed request with a message that names the problem",
func(state, questions, want string) {
Expect(validateSystemOneRequest(req(state, questions))).To(MatchError(ContainSubstring(want)))
},
Entry("missing state", ``, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"),
Entry("null state", `null`, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"),
Entry("blank string state", `" "`, `{"q":{"type":"noul","instructions":"i"}}`, "state is required"),
Entry("no questions", `"x"`, `{}`, "at least one question"),
Entry("blank question id", `"x"`, `{" ":{"type":"noul","instructions":"i"}}`, "blank"),
Entry("unknown type", `"x"`, `{"q":{"type":"rank","instructions":"i"}}`, "unknown type"),
Entry("choice with one option", `"x"`, `{"q":{"type":"choice","instructions":"i","criteria":{"a":"A"}}}`, "at least 2"),
Entry("choice with a blank option key", `"x"`, `{"q":{"type":"choice","instructions":"i","criteria":{"a":"A"," ":"B"}}}`, "blank"),
Entry("score with one level", `"x"`, `{"q":{"type":"score","instructions":"i","criteria":["only"]}}`, "at least 2"),
Entry("noul criteria with a stray key", `"x"`, `{"q":{"type":"noul","instructions":"i","criteria":{"maybe":"M"}}}`, `"false" and "true"`),
)
It("refuses more than 64 questions", func() {
var b strings.Builder
b.WriteString("{")
for i := 0; i < 65; i++ {
if i > 0 {
b.WriteString(",")
}
b.WriteString(`"q` + strings.Repeat("x", i) + `":{"type":"noul","instructions":"i"}`)
}
b.WriteString("}")
Expect(validateSystemOneRequest(req(`"x"`, b.String()))).To(MatchError(ContainSubstring("at most 64")))
})
})
var _ = Describe("systemOneBind", func() {
bind := func(body string) (int, error) {
e := echo.New()
r := httptest.NewRequest(http.MethodPost, "/v1/systemone", strings.NewReader(body))
r.Header.Set("Content-Type", "application/json")
c := e.NewContext(r, httptest.NewRecorder())
var out schema.SystemOneRequest
if err := systemOneBind(c, &out); err != nil {
return systemOneBindStatus(err), err
}
return http.StatusOK, nil
}
It("binds a normal body", func() {
status, err := bind(`{"model":"m","state":"x","questions":{}}`)
Expect(err).ToNot(HaveOccurred())
Expect(status).To(Equal(http.StatusOK))
})
It("answers 413 for a body over 64 KiB", func() {
status, err := bind(`{"model":"m","state":"` + strings.Repeat("a", 65*1024) + `"}`)
Expect(err).To(HaveOccurred())
Expect(status).To(Equal(http.StatusRequestEntityTooLarge))
})
It("answers 400 for malformed JSON", func() {
status, err := bind(`{not json`)
Expect(err).To(HaveOccurred())
Expect(status).To(Equal(http.StatusBadRequest))
})
})
@@ -99,6 +99,7 @@ type fakeModel struct {
transcribeDeltas []string
transcribeFinal *schema.TranscriptionResult
transcribeErr error
lastDiarize bool // diarize flag of the last Transcribe/TranscribeStream call
// TranscribeLive scripting: liveErr makes the open fail (degrade path);
// liveEvents are delivered to onEvent synchronously at open;
@@ -200,7 +201,8 @@ func (m *fakeModel) VAD(_ context.Context, req *schema.VADRequest) (*schema.VADR
return &schema.VADResponse{Segments: m.vadSegments}, nil
}
func (m *fakeModel) Transcribe(context.Context, string, string, bool, bool, string) (*schema.TranscriptionResult, error) {
func (m *fakeModel) Transcribe(_ context.Context, _, _ string, _, diarize bool, _ string) (*schema.TranscriptionResult, error) {
m.lastDiarize = diarize
return m.transcribeFinal, m.transcribeErr
}
@@ -247,7 +249,8 @@ func (m *fakeModel) TTSStream(_ context.Context, _, _, _ string, onAudio func(pc
return nil
}
func (m *fakeModel) TranscribeStream(_ context.Context, _, _ string, _, _ bool, _ string, onDelta func(text string)) (*schema.TranscriptionResult, error) {
func (m *fakeModel) TranscribeStream(_ context.Context, _, _ string, _, diarize bool, _ string, onDelta func(text string)) (*schema.TranscriptionResult, error) {
m.lastDiarize = diarize
for _, d := range m.transcribeDeltas {
onDelta(d)
}
@@ -209,6 +209,35 @@ func (l *liveTurnState) drainEvents(audioSec float64) {
if ev.Final != nil && strings.TrimSpace(ev.Final.Text) != "" {
l.finalText = ev.Final.Text
}
// Speaker and sound events from a companion diarization/scene
// stream: forward each as its own event under the turn's item
// id, same as caption deltas. Text is empty — the event exists
// to carry the speaker/segment boundary, not transcript text.
if l.transport != nil && l.itemID != "" {
for _, seg := range ev.Speakers {
sendEvent(l.transport, types.ConversationItemInputAudioTranscriptionSegmentEvent{
ServerEventBase: types.ServerEventBase{EventID: "event_TODO"},
ItemID: l.itemID,
ContentIndex: 0,
Speaker: seg.Speaker,
Start: seg.Start,
End: seg.End,
})
}
for _, sound := range ev.Sounds {
start, end := sound.Start, sound.End
sendEvent(l.transport, types.ConversationItemSoundDetectionEvent{
ServerEventBase: types.ServerEventBase{EventID: "event_TODO"},
ItemID: l.itemID,
ContentIndex: 0,
Detections: []types.SoundDetectionTag{
{Label: sound.Label, Score: sound.Peak, Index: sound.Index},
},
Start: &start,
End: &end,
})
}
}
default:
return
}
@@ -291,6 +291,67 @@ var _ = Describe("liveTurnState", func() {
Expect(ftr.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionFailed)).To(Equal(0))
})
})
Describe("scene events (speakers and sounds)", func() {
It("emits a segment event per speaker with empty text under the turn's item id", func() {
Expect(lts.openTurn(context.Background(), "item1")).To(BeTrue())
turnID := lts.itemID
m.liveSession.onEvent(backend.LiveTranscriptionEvent{
Speakers: []backend.LiveSpeakerSegment{{Speaker: "1", Start: 1.2, End: 3.4}},
})
lts.drainEvents(3.4)
var got []types.ConversationItemInputAudioTranscriptionSegmentEvent
for _, e := range ftr.events() {
if seg, ok := e.(types.ConversationItemInputAudioTranscriptionSegmentEvent); ok {
got = append(got, seg)
}
}
Expect(got).To(HaveLen(1))
Expect(got[0].ItemID).To(Equal(turnID))
Expect(got[0].Speaker).To(Equal("1"))
Expect(got[0].Start).To(BeNumerically("~", 1.2, 1e-9))
Expect(got[0].End).To(BeNumerically("~", 3.4, 1e-9))
Expect(got[0].Text).To(BeEmpty())
})
It("emits a sound_detection event per sound with one tag and start/end", func() {
Expect(lts.openTurn(context.Background(), "item1")).To(BeTrue())
turnID := lts.itemID
m.liveSession.onEvent(backend.LiveTranscriptionEvent{
Sounds: []backend.LiveSoundEvent{{Label: "Dog bark", Index: 5, Peak: 0.8, Start: 0.5, End: 0.9}},
})
lts.drainEvents(1.0)
var got []types.ConversationItemSoundDetectionEvent
for _, e := range ftr.events() {
if sd, ok := e.(types.ConversationItemSoundDetectionEvent); ok {
got = append(got, sd)
}
}
Expect(got).To(HaveLen(1))
Expect(got[0].ItemID).To(Equal(turnID))
Expect(got[0].Detections).To(HaveLen(1))
Expect(got[0].Detections[0].Label).To(Equal("Dog bark"))
Expect(got[0].Detections[0].Score).To(BeNumerically("~", 0.8, 1e-6))
Expect(got[0].Detections[0].Index).To(Equal(5))
Expect(got[0].Start).NotTo(BeNil())
Expect(*got[0].Start).To(BeNumerically("~", 0.5, 1e-9))
Expect(got[0].End).NotTo(BeNil())
Expect(*got[0].End).To(BeNumerically("~", 0.9, 1e-9))
})
It("sends neither event when a live event carries no speakers or sounds", func() {
Expect(lts.openTurn(context.Background(), "item1")).To(BeTrue())
m.liveSession.onEvent(backend.LiveTranscriptionEvent{Delta: "hi"})
lts.drainEvents(1.0)
Expect(ftr.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionSegment)).To(Equal(0))
Expect(ftr.countEvents(types.ServerEventTypeConversationItemSoundDetection)).To(Equal(0))
})
})
})
// commitUtteranceWithTranscript routes the three transcript sources: the
@@ -3,6 +3,7 @@ package openai
import (
"context"
"encoding/binary"
"encoding/json"
"errors"
"os"
@@ -14,6 +15,69 @@ import (
"github.com/mudler/LocalAI/core/schema"
)
// ConversationItemSoundDetectionEvent gained optional Start/End (seconds)
// for the live scene-event path; the unary/windowed paths never set them,
// so existing consumers must see no start/end keys at all.
var _ = Describe("ConversationItemSoundDetectionEvent JSON", func() {
It("omits start and end when nil", func() {
ev := types.ConversationItemSoundDetectionEvent{
ItemID: "item1",
Detections: []types.SoundDetectionTag{{Label: "Speech", Score: 0.5, Index: 7}},
}
b, err := json.Marshal(ev)
Expect(err).ToNot(HaveOccurred())
var got map[string]any
Expect(json.Unmarshal(b, &got)).To(Succeed())
_, hasStart := got["start"]
_, hasEnd := got["end"]
Expect(hasStart).To(BeFalse())
Expect(hasEnd).To(BeFalse())
})
It("includes start and end when set", func() {
start, end := 0.5, 0.9
ev := types.ConversationItemSoundDetectionEvent{
ItemID: "item1",
Start: &start,
End: &end,
}
b, err := json.Marshal(ev)
Expect(err).ToNot(HaveOccurred())
var got map[string]any
Expect(json.Unmarshal(b, &got)).To(Succeed())
Expect(got["start"]).To(BeNumerically("~", 0.5, 1e-9))
Expect(got["end"]).To(BeNumerically("~", 0.9, 1e-9))
})
})
// ConversationItemInputAudioTranscriptionSegmentEvent.Start/End are plain
// float64 (no omitempty): a speaker segment starting at 0.0s must still
// carry "start" in the JSON, unlike the sound-detection event's optional
// pointer fields above.
var _ = Describe("ConversationItemInputAudioTranscriptionSegmentEvent JSON", func() {
It("marshals start:0 and end:1.5 even when start is the zero value", func() {
ev := types.ConversationItemInputAudioTranscriptionSegmentEvent{
ItemID: "item1",
Speaker: "1",
Start: 0,
End: 1.5,
}
b, err := json.Marshal(ev)
Expect(err).ToNot(HaveOccurred())
var got map[string]any
Expect(json.Unmarshal(b, &got)).To(Succeed())
_, hasStart := got["start"]
_, hasEnd := got["end"]
Expect(hasStart).To(BeTrue())
Expect(hasEnd).To(BeTrue())
Expect(got["start"]).To(BeNumerically("~", 0.0, 1e-9))
Expect(got["end"]).To(BeNumerically("~", 1.5, 1e-9))
})
})
// emitSoundDetection classifies a committed utterance and emits a single
// conversation.item.sound_detection event carrying the scored AudioSet tags.
var _ = Describe("emitSoundDetection", func() {
@@ -5,6 +5,7 @@ import (
"fmt"
"github.com/mudler/LocalAI/core/http/endpoints/openai/types"
"github.com/mudler/LocalAI/core/schema"
)
// emitPrecomputedTranscription emits the transcription events for a turn
@@ -42,9 +43,10 @@ func emitPrecomputedTranscription(t Transport, itemID string, deltas []string, t
// a single completed event. delta and completed events share itemID.
func emitTranscription(ctx context.Context, t Transport, session *Session, itemID, audioPath string) (string, error) {
cfg := session.InputAudioTranscription
diarize := session.ModelConfig != nil && session.ModelConfig.Pipeline.Diarization
if session.ModelConfig != nil && session.ModelConfig.Pipeline.StreamTranscription() {
final, err := session.ModelInterface.TranscribeStream(ctx, audioPath, cfg.Language, false, false, cfg.Prompt, func(delta string) {
final, err := session.ModelInterface.TranscribeStream(ctx, audioPath, cfg.Language, false, diarize, cfg.Prompt, func(delta string) {
_ = t.SendEvent(types.ConversationItemInputAudioTranscriptionDeltaEvent{
ServerEventBase: types.ServerEventBase{EventID: "event_TODO"},
ItemID: itemID,
@@ -58,6 +60,11 @@ func emitTranscription(ctx context.Context, t Transport, session *Session, itemI
transcript := ""
if final != nil {
transcript = final.Text
if diarize {
if err := emitSpeakerSegments(t, itemID, final); err != nil {
return "", err
}
}
}
if err := t.SendEvent(types.ConversationItemInputAudioTranscriptionCompletedEvent{
ServerEventBase: types.ServerEventBase{EventID: "event_TODO"},
@@ -71,13 +78,18 @@ func emitTranscription(ctx context.Context, t Transport, session *Session, itemI
}
// Unary fallback: transcribe the whole utterance, emit one completed event.
tr, err := session.ModelInterface.Transcribe(ctx, audioPath, cfg.Language, false, false, cfg.Prompt)
tr, err := session.ModelInterface.Transcribe(ctx, audioPath, cfg.Language, false, diarize, cfg.Prompt)
if err != nil {
return "", err
}
if tr == nil {
return "", fmt.Errorf("transcribe result is nil")
}
if diarize {
if err := emitSpeakerSegments(t, itemID, tr); err != nil {
return "", err
}
}
if err := t.SendEvent(types.ConversationItemInputAudioTranscriptionCompletedEvent{
ServerEventBase: types.ServerEventBase{EventID: "event_TODO"},
ItemID: itemID,
@@ -88,3 +100,29 @@ func emitTranscription(ctx context.Context, t Transport, session *Session, itemI
}
return tr.Text, nil
}
// emitSpeakerSegments forwards each speaker-labelled segment of a committed
// turn's transcript as a conversation.item.input_audio_transcription.segment
// event (pipeline.diarization), before the turn's completed event. Times are
// relative to the turn's audio and speaker labels are only consistent within
// the turn, as on the live path. Segments without a speaker are skipped.
func emitSpeakerSegments(t Transport, itemID string, tr *schema.TranscriptionResult) error {
for _, seg := range tr.Segments {
if seg.Speaker == "" {
continue
}
if err := t.SendEvent(types.ConversationItemInputAudioTranscriptionSegmentEvent{
ServerEventBase: types.ServerEventBase{EventID: "event_TODO"},
ItemID: itemID,
ContentIndex: 0,
ID: fmt.Sprintf("seg_%d", seg.Id),
Speaker: seg.Speaker,
Start: seg.Start.Seconds(),
End: seg.End.Seconds(),
Text: seg.Text,
}); err != nil {
return err
}
}
return nil
}
@@ -2,6 +2,7 @@ package openai
import (
"context"
"time"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
@@ -51,4 +52,86 @@ var _ = Describe("emitTranscription", func() {
Expect(t.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionDelta)).To(Equal(0))
Expect(t.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionCompleted)).To(Equal(1))
})
Context("pipeline.diarization", func() {
labelled := &schema.TranscriptionResult{
Text: "hi there. hello",
Segments: []schema.TranscriptionSegment{
{Id: 0, Text: "hi there.", Start: 0, End: 600 * time.Millisecond, Speaker: "0"},
{Id: 1, Text: "hello", Start: time.Second, End: 1400 * time.Millisecond, Speaker: "1"},
{Id: 2, Text: "unlabelled"},
},
}
segmentEvents := func(t *fakeTransport) []types.ConversationItemInputAudioTranscriptionSegmentEvent {
var out []types.ConversationItemInputAudioTranscriptionSegmentEvent
for _, e := range t.sent {
if seg, ok := e.(types.ConversationItemInputAudioTranscriptionSegmentEvent); ok {
out = append(out, seg)
}
}
return out
}
It("requests speakers and emits one segment event per labelled segment", func() {
m := &fakeModel{transcribeFinal: labelled}
session := &Session{
InputAudioTranscription: &types.AudioTranscription{},
ModelConfig: &config.ModelConfig{Pipeline: config.Pipeline{Diarization: true}},
ModelInterface: m,
}
t := &fakeTransport{}
transcript, err := emitTranscription(context.Background(), t, session, "item1", "/tmp/x.wav")
Expect(err).ToNot(HaveOccurred())
Expect(transcript).To(Equal("hi there. hello"))
Expect(m.lastDiarize).To(BeTrue())
segs := segmentEvents(t)
Expect(segs).To(HaveLen(2))
Expect(segs[0].ItemID).To(Equal("item1"))
Expect(segs[0].Speaker).To(Equal("0"))
Expect(segs[0].Text).To(Equal("hi there."))
Expect(segs[1].Speaker).To(Equal("1"))
Expect(segs[1].Start).To(BeNumerically("~", 1.0, 1e-9))
Expect(segs[1].End).To(BeNumerically("~", 1.4, 1e-9))
Expect(t.countEvents(types.ServerEventTypeConversationItemInputAudioTranscriptionCompleted)).To(Equal(1))
})
It("also emits segment events on the streaming transcription path", func() {
on := true
m := &fakeModel{transcribeDeltas: []string{"hi"}, transcribeFinal: labelled}
session := &Session{
InputAudioTranscription: &types.AudioTranscription{},
ModelConfig: &config.ModelConfig{Pipeline: config.Pipeline{
Diarization: true,
Streaming: config.PipelineStreaming{Transcription: &on},
}},
ModelInterface: m,
}
t := &fakeTransport{}
_, err := emitTranscription(context.Background(), t, session, "item1", "/tmp/x.wav")
Expect(err).ToNot(HaveOccurred())
Expect(m.lastDiarize).To(BeTrue())
Expect(segmentEvents(t)).To(HaveLen(2))
})
It("neither asks for speakers nor emits segments when off", func() {
m := &fakeModel{transcribeFinal: labelled}
session := &Session{
InputAudioTranscription: &types.AudioTranscription{},
ModelConfig: &config.ModelConfig{},
ModelInterface: m,
}
t := &fakeTransport{}
_, err := emitTranscription(context.Background(), t, session, "item1", "/tmp/x.wav")
Expect(err).ToNot(HaveOccurred())
Expect(m.lastDiarize).To(BeFalse())
Expect(segmentEvents(t)).To(BeEmpty())
})
})
})
+14 -8
View File
@@ -210,18 +210,20 @@ func TranscriptEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, app
}
for _, word := range tr.Words {
trs.Words = append(trs.Words, schema.TranscriptionWordSeconds{
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Speaker: word.Speaker,
})
}
for _, seg := range tr.Segments {
segWords := []schema.TranscriptionWordSeconds{}
for _, word := range seg.Words {
segWords = append(segWords, schema.TranscriptionWordSeconds{
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Start: word.Start.Seconds(),
End: word.End.Seconds(),
Text: word.Text,
Speaker: word.Speaker,
})
}
trs.Segments = append(trs.Segments, schema.TranscriptionSegmentSeconds{
@@ -338,12 +340,16 @@ func streamTranscription(c echo.Context, req backend.TranscriptionRequest, ml *m
if len(finalResult.Segments) > 0 {
segs := make([]map[string]any, 0, len(finalResult.Segments))
for _, seg := range finalResult.Segments {
segs = append(segs, map[string]any{
entry := map[string]any{
"id": seg.Id,
"start": seg.Start.Seconds(),
"end": seg.End.Seconds(),
"text": seg.Text,
})
}
if seg.Speaker != "" {
entry["speaker"] = seg.Speaker
}
segs = append(segs, entry)
}
doneEvent["segments"] = segs
}
@@ -512,6 +512,15 @@ type ConversationItemSoundDetectionEvent struct {
// The scored sound-event tags, in score-descending order.
Detections []SoundDetectionTag `json:"detections"`
// The start time of the detection window in seconds, when known. Set by
// the live scene-event path (a companion sound stream alongside live
// transcription); omitted by the unary/windowed sound-detection paths,
// which have no per-event timing.
Start *float64 `json:"start,omitempty"`
// The end time of the detection window in seconds, when known.
End *float64 `json:"end,omitempty"`
}
func (m ConversationItemSoundDetectionEvent) ServerEventType() ServerEventType {
@@ -586,11 +595,13 @@ type ConversationItemInputAudioTranscriptionSegmentEvent struct {
// The speaker label for the segment, if available.
Speaker string `json:"speaker,omitempty"`
// The start time of the segment in seconds.
Start float64 `json:"start,omitempty"`
// The start time of the segment in seconds. Always present (not
// omitempty: a segment starting at 0.0s must still carry "start").
Start float64 `json:"start"`
// The end time of the segment in seconds.
End float64 `json:"end,omitempty"`
// The end time of the segment in seconds. Always present (not
// omitempty: see Start).
End float64 `json:"end"`
// The text content of the segment.
Text string `json:"text,omitempty"`
@@ -0,0 +1,97 @@
import { test, expect } from './coverage-fixtures.js'
// Single-page PDF with a real text layer, built byte by byte so the xref
// offsets are valid and the spec needs no binary fixture.
function buildPdf(text) {
const stream = `BT /F1 18 Tf 20 100 Td (${text}) Tj ET`
const objs = [
'<< /Type /Catalog /Pages 2 0 R >>',
'<< /Type /Pages /Kids [3 0 R] /Count 1 >>',
'<< /Type /Page /Parent 2 0 R /MediaBox [0 0 300 200] /Contents 4 0 R /Resources << /Font << /F1 5 0 R >> >> >>',
`<< /Length ${stream.length} >>\nstream\n${stream}\nendstream`,
'<< /Type /Font /Subtype /Type1 /BaseFont /Helvetica >>',
]
let out = '%PDF-1.4\n'
const offsets = []
objs.forEach((body, i) => {
offsets.push(out.length)
out += `${i + 1} 0 obj\n${body}\nendobj\n`
})
const xref = out.length
out += `xref\n0 ${objs.length + 1}\n0000000000 65535 f \n`
for (const o of offsets) out += `${String(o).padStart(10, '0')} 00000 n \n`
out += `trailer\n<< /Size ${objs.length + 1} /Root 1 0 R >>\nstartxref\n${xref}\n%%EOF\n`
return Buffer.from(out, 'latin1')
}
async function openChat(page) {
await page.route('**/api/models/capabilities', (route) => {
route.fulfill({
contentType: 'application/json',
body: JSON.stringify({ data: [{ id: 'test-model', capabilities: ['FLAG_CHAT'] }] }),
})
})
await page.goto('/app/chat')
await expect(page.getByRole('button', { name: 'test-model' })).toBeVisible({ timeout: 10_000 })
}
test.describe('Chat - PDF attachments', () => {
test('sends the extracted text layer, not the raw PDF bytes', async ({ page }) => {
let requestBody = ''
await page.route('**/v1/chat/completions', (route) => {
requestBody = route.request().postData() || ''
route.fulfill({ status: 500, contentType: 'application/json', body: JSON.stringify({ error: { message: 'stop' } }) })
})
await openChat(page)
await page.locator('input[type=file]').setInputFiles({
name: 'report.pdf',
mimeType: 'application/pdf',
buffer: buildPdf('Quarterly revenue grew 42 percent'),
})
await expect(page.locator('.chat-file-name', { hasText: 'report.pdf' })).toBeVisible()
await page.locator('.chat-input').fill('Summarize')
await page.locator('.chat-send-btn').click()
await expect.poll(() => requestBody).toContain('Quarterly revenue grew 42 percent')
expect(requestBody).toContain('File: report.pdf')
expect(requestBody).not.toContain('%PDF')
})
test('rejects a PDF that cannot be parsed instead of attaching garbage', async ({ page }) => {
await openChat(page)
await page.locator('input[type=file]').setInputFiles({
name: 'broken.pdf',
mimeType: 'application/pdf',
buffer: Buffer.from('%PDF-1.4 this is not a real document'),
})
await expect(page.getByText('Could not read text from broken.pdf')).toBeVisible({ timeout: 10_000 })
await expect(page.locator('.chat-file-name', { hasText: 'broken.pdf' })).toHaveCount(0)
})
})
test.describe('Home - PDF attachments', () => {
test('attaches a PDF that has a text layer', async ({ page }) => {
await page.goto('/app')
await page.locator('input[type=file][accept*="pdf"]').setInputFiles({
name: 'notes.pdf',
mimeType: 'application/pdf',
buffer: buildPdf('Meeting notes for Tuesday'),
})
await expect(page.locator('.home-file-tag', { hasText: 'notes.pdf' })).toBeVisible({ timeout: 10_000 })
})
test('rejects a PDF that cannot be parsed', async ({ page }) => {
await page.goto('/app')
await page.locator('input[type=file][accept*="pdf"]').setInputFiles({
name: 'broken.pdf',
mimeType: 'application/pdf',
buffer: Buffer.from('%PDF-1.4 this is not a real document'),
})
await expect(page.getByText('Could not read text from broken.pdf')).toBeVisible({ timeout: 10_000 })
await expect(page.locator('.home-file-tag')).toHaveCount(0)
})
})
@@ -172,6 +172,19 @@ test.describe('Models lifecycle', () => {
await expect(installedPane(page)).toContainText('Worker one')
})
test('shows the decisions use case on a decision model', async ({ page }) => {
await page.route('**/api/models/capabilities', route => route.fulfill({
contentType: 'application/json',
body: JSON.stringify({
data: [...installedModels, { id: 'decider', backend: 'vllm-cpp', capabilities: ['FLAG_DECISIONS'] }],
}),
}))
await page.goto('/app/models?view=installed&model=decider')
await expect(installedPane(page)).toContainText('decider')
await expect(installedPane(page)).toContainText('Decisions')
})
test('stops a running model with confirmation', async ({ page }) => {
await page.goto('/app/models?view=installed&model=alpha')
+271
View File
@@ -29,6 +29,7 @@
"i18next-browser-languagedetector": "^8.2.1",
"i18next-http-backend": "^3.0.6",
"marked": "^15.0.7",
"pdfjs-dist": "^5.6.205",
"react": "^19.1.0",
"react-dom": "^19.1.0",
"react-i18next": "^17.0.6",
@@ -1021,6 +1022,256 @@
"resolved": "https://registry.npmjs.org/json-schema-traverse/-/json-schema-traverse-1.0.0.tgz",
"integrity": "sha512-NM8/P9n3XjXhIZn1lLhkFaACTOURQXjWhV4BA/RnOv8xvgqtqpAX9IO4mRQxSx1Rlo4tqzeqb0sOlruaOy3dug=="
},
"node_modules/@napi-rs/canvas": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas/-/canvas-0.1.100.tgz",
"integrity": "sha512-xglYA6q3XO5P3BNJYxVZ1IV7DLVjp1Py6nwag88YntrS+3vKHyYcMqXVS4ZztJmwz2uGvz1FWhI/4LgbR5uQDA==",
"license": "MIT",
"optional": true,
"workspaces": [
"e2e/*"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
},
"optionalDependencies": {
"@napi-rs/canvas-android-arm64": "0.1.100",
"@napi-rs/canvas-darwin-arm64": "0.1.100",
"@napi-rs/canvas-darwin-x64": "0.1.100",
"@napi-rs/canvas-linux-arm-gnueabihf": "0.1.100",
"@napi-rs/canvas-linux-arm64-gnu": "0.1.100",
"@napi-rs/canvas-linux-arm64-musl": "0.1.100",
"@napi-rs/canvas-linux-riscv64-gnu": "0.1.100",
"@napi-rs/canvas-linux-x64-gnu": "0.1.100",
"@napi-rs/canvas-linux-x64-musl": "0.1.100",
"@napi-rs/canvas-win32-arm64-msvc": "0.1.100",
"@napi-rs/canvas-win32-x64-msvc": "0.1.100"
}
},
"node_modules/@napi-rs/canvas-android-arm64": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-android-arm64/-/canvas-android-arm64-0.1.100.tgz",
"integrity": "sha512-hjhCKhntPv9+t4ckHymdx0phYNcVW+GKQR6Lzw2zE+pOVjOplSmtx9nNNknTjbEDLcuLZqA1y8ufKg1XfgftzQ==",
"cpu": [
"arm64"
],
"license": "MIT",
"optional": true,
"os": [
"android"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-darwin-arm64": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-darwin-arm64/-/canvas-darwin-arm64-0.1.100.tgz",
"integrity": "sha512-2PcswRaC7Ly645DGt88///zuFDhJxJYdKAs1uU3mfk1atYkXufgcgLfBpk6Tm12nCQBaNt1wpybuPZ4qOhTo8A==",
"cpu": [
"arm64"
],
"license": "MIT",
"optional": true,
"os": [
"darwin"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-darwin-x64": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-darwin-x64/-/canvas-darwin-x64-0.1.100.tgz",
"integrity": "sha512-ePNZtj7pNIva/siZMg+HmbeozkIjqUIYdoymH8HaA3qK7LfzFN4WMBM8G6HQ9ZC+H3+Dnn5pqtiXpgLykaPOhw==",
"cpu": [
"x64"
],
"license": "MIT",
"optional": true,
"os": [
"darwin"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-linux-arm-gnueabihf": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-arm-gnueabihf/-/canvas-linux-arm-gnueabihf-0.1.100.tgz",
"integrity": "sha512-d5cDB48oWFGU8/XPhUOFAlySgb/VAu7D+s8fi55K1Pcfg8aPplHWqMgibhVLU8ky7Pyg/fuiVLz4Nf3JrSTuUA==",
"cpu": [
"arm"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-linux-arm64-gnu": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-arm64-gnu/-/canvas-linux-arm64-gnu-0.1.100.tgz",
"integrity": "sha512-rDxgxRu69RvDlX/bh9o22DxLsGr8EqsNgotL9+RwQE1S0b0cqeatqsw6aW45mukm0B42DIAaAacKaYQ8cqS1nw==",
"cpu": [
"arm64"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-linux-arm64-musl": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-arm64-musl/-/canvas-linux-arm64-musl-0.1.100.tgz",
"integrity": "sha512-K3mDW66N+xT2/V439u1alFANiBUjdEx2gLiNYnCmUsva5jZMxWTjafBYwTzYK+EMFMHrUoabuU+T1BIP5CgbYQ==",
"cpu": [
"arm64"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-linux-riscv64-gnu": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-riscv64-gnu/-/canvas-linux-riscv64-gnu-0.1.100.tgz",
"integrity": "sha512-mooqUBTIsccZpnoQC4NgrC1v6C1vof39etLNMnBwCY+p0gajWJvAHLGQ6g/gGyS5YrpDW+GefSN4+Cvcr08UWw==",
"cpu": [
"riscv64"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-linux-x64-gnu": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-x64-gnu/-/canvas-linux-x64-gnu-0.1.100.tgz",
"integrity": "sha512-1eCvkDCazm7FFhsT7DfGOdSaHgZVK3bt/dSBl5EWHOWmnz+I7j8tPseJqqD81NF+MH21jKUK4wQSDjN0mdhnTg==",
"cpu": [
"x64"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-linux-x64-musl": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-linux-x64-musl/-/canvas-linux-x64-musl-0.1.100.tgz",
"integrity": "sha512-20arT6lnI19S68qNlii73TSEDbECNgzMz2EpldC1V3mZFuRkeujXkcebRk0LRJe9SEUAooYiLokfMViY8IX7yA==",
"cpu": [
"x64"
],
"license": "MIT",
"optional": true,
"os": [
"linux"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-win32-arm64-msvc": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-win32-arm64-msvc/-/canvas-win32-arm64-msvc-0.1.100.tgz",
"integrity": "sha512-DZFFT1wIAg37LJw37yhMRFfjATd3vTQzjZ1Yki8u2vhO6Hi5VE6BVaGQ1aaDu7xb4iMErz+9EOwjpS7xcxFeBw==",
"cpu": [
"arm64"
],
"license": "MIT",
"optional": true,
"os": [
"win32"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/canvas-win32-x64-msvc": {
"version": "0.1.100",
"resolved": "https://registry.npmjs.org/@napi-rs/canvas-win32-x64-msvc/-/canvas-win32-x64-msvc-0.1.100.tgz",
"integrity": "sha512-MyT1j3mHC2+Lu4pBi9mKyMJhtP6U7k7EldY7sj/uS5gJA65gTXt8MefJQXLJo5d/vZbuWmfxzkEUNc/urV3pHA==",
"cpu": [
"x64"
],
"license": "MIT",
"optional": true,
"os": [
"win32"
],
"engines": {
"node": ">= 10"
},
"funding": {
"type": "github",
"url": "https://github.com/sponsors/Brooooooklyn"
}
},
"node_modules/@napi-rs/wasm-runtime": {
"version": "1.1.5",
"resolved": "https://registry.npmjs.org/@napi-rs/wasm-runtime/-/wasm-runtime-1.1.5.tgz",
@@ -5219,6 +5470,13 @@
"node": ">=8"
}
},
"node_modules/node-readable-to-web-readable-stream": {
"version": "0.4.2",
"resolved": "https://registry.npmjs.org/node-readable-to-web-readable-stream/-/node-readable-to-web-readable-stream-0.4.2.tgz",
"integrity": "sha512-/cMZNI34v//jUTrI+UIo4ieHAB5EZRY/+7OmXZgBxaWBMcW2tGdceIw06RFxWxrKZ5Jp3sI2i5TsRo+CBhtVLQ==",
"license": "MIT",
"optional": true
},
"node_modules/node-releases": {
"version": "2.0.54",
"resolved": "https://registry.npmjs.org/node-releases/-/node-releases-2.0.54.tgz",
@@ -5752,6 +6010,19 @@
"url": "https://opencollective.com/express"
}
},
"node_modules/pdfjs-dist": {
"version": "5.6.205",
"resolved": "https://registry.npmjs.org/pdfjs-dist/-/pdfjs-dist-5.6.205.tgz",
"integrity": "sha512-tlUj+2IDa7G1SbvBNN74UHRLJybZDWYom+k6p5KIZl7huBvsA4APi6mKL+zCxd3tLjN5hOOEE9Tv7VdzO88pfg==",
"license": "Apache-2.0",
"engines": {
"node": ">=20.19.0 || >=22.13.0 || >=24"
},
"optionalDependencies": {
"@napi-rs/canvas": "^0.1.96",
"node-readable-to-web-readable-stream": "^0.4.2"
}
},
"node_modules/picocolors": {
"version": "1.1.1",
"resolved": "https://registry.npmjs.org/picocolors/-/picocolors-1.1.1.tgz",
+1
View File
@@ -45,6 +45,7 @@
"i18next-browser-languagedetector": "^8.2.1",
"i18next-http-backend": "^3.0.6",
"marked": "^15.0.7",
"pdfjs-dist": "^5.6.205",
"react": "^19.1.0",
"react-dom": "^19.1.0",
"react-i18next": "^17.0.6",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -117,7 +117,8 @@
"copied": "Copied to clipboard",
"copyFailed": "Could not copy to clipboard",
"chatCopied": "Chat copied to clipboard",
"forked": "Created a new chat"
"forked": "Created a new chat",
"pdfReadFailed": "Could not read text from {{name}}. It may be scanned, encrypted or damaged."
},
"menu": {
"trigger": "Chats",
@@ -35,7 +35,8 @@
"enterToSend": "Enter to send",
"selectModelFirst": "Select a model first",
"sendMessage": "Send message",
"selectModelToast": "Please select a model first"
"selectModelToast": "Please select a model first",
"pdfReadFailed": "Could not read text from {{name}}. It may be scanned, encrypted or damaged."
},
"quickLinks": {
"manageByChat": "Manage by chat",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
@@ -46,7 +46,7 @@
"open": {
"title": "Open", "chat": "Chat", "completion": "Completion", "image": "Image", "video": "Video", "tts": "TTS",
"transcribe": "Transcribe", "sound": "Sound", "face": "Face", "voice": "Voice", "embeddings": "Embeddings",
"rerank": "Rerank", "vad": "VAD", "score": "Score"
"rerank": "Rerank", "vad": "VAD", "score": "Score", "decisions": "Decisions"
},
"empty": {
"title": "No models installed yet", "text": "Explore the gallery or import a model to get started.",
+8 -2
View File
@@ -9,6 +9,7 @@ import { extractCodeArtifacts, renderMarkdownWithArtifacts } from '../utils/arti
import CanvasPanel from '../components/CanvasPanel'
import Toggle from '../components/Toggle'
import { fileToBase64, modelsApi, mcpApi } from '../utils/api'
import { readAttachmentText } from '../utils/pdf'
import { CAP_CHAT } from '../utils/capabilities'
import { useMCPClient } from '../hooks/useMCPClient'
import MCPAppFrame from '../components/MCPAppFrame'
@@ -842,13 +843,18 @@ export default function Chat() {
const base64 = await fileToBase64(file)
const entry = { name: file.name, type: file.type, base64 }
if (!file.type.startsWith('image/') && !file.type.startsWith('audio/') && !file.type.startsWith('video/')) {
entry.textContent = await file.text().catch(() => '')
try {
entry.textContent = await readAttachmentText(file)
} catch {
addToast(t('toasts.pdfReadFailed', { name: file.name }), 'error')
continue
}
}
newFiles.push(entry)
}
setFiles(prev => [...prev, ...newFiles])
e.target.value = ''
}, [])
}, [addToast, t])
const handlePaste = useCallback(async (e) => {
const items = e.clipboardData?.items
+8 -2
View File
@@ -12,6 +12,7 @@ import HomeConnect from '../components/HomeConnect'
import { useResources } from '../hooks/useResources'
import { usePolling } from '../hooks/usePolling'
import { fileToBase64, backendControlApi, systemApi, modelsApi, mcpApi, nodesApi } from '../utils/api'
import { readAttachmentText } from '../utils/pdf'
import { API_CONFIG } from '../utils/config'
import { greetingKey } from '../utils/greeting'
import StatusPill from '../components/StatusPill'
@@ -158,12 +159,17 @@ export default function Home() {
const base64 = await fileToBase64(file)
const entry = { name: file.name, type: file.type, base64 }
if (!file.type.startsWith('image/') && !file.type.startsWith('audio/')) {
entry.textContent = await file.text().catch(() => '')
try {
entry.textContent = await readAttachmentText(file)
} catch {
addToast(t('input.pdfReadFailed', { name: file.name }), 'error')
continue
}
}
newFiles.push(entry)
}
setter(prev => [...prev, ...newFiles])
}, [])
}, [addToast, t])
const removeFile = useCallback((file) => {
const removeFn = (prev) => prev.filter(f => f !== file)
@@ -22,7 +22,7 @@ import {
CAP_CHAT, CAP_COMPLETION, CAP_IMAGE, CAP_VIDEO, CAP_TTS,
CAP_TRANSCRIPT, CAP_SOUND_GENERATION, CAP_FACE_RECOGNITION,
CAP_SPEAKER_RECOGNITION, CAP_EMBEDDINGS, CAP_RERANK,
CAP_VAD, CAP_SCORE,
CAP_VAD, CAP_SCORE, CAP_DECISIONS,
} from '../utils/capabilities'
const USE_CASES = [
@@ -39,6 +39,7 @@ const USE_CASES = [
{ cap: CAP_RERANK, labelKey: 'rerank' },
{ cap: CAP_VAD, labelKey: 'vad' },
{ cap: CAP_SCORE, labelKey: 'score' },
{ cap: CAP_DECISIONS, labelKey: 'decisions' },
]
export function modelUseCases(model) {
+1
View File
@@ -29,4 +29,5 @@ export const CAP_SPEAKER_RECOGNITION = 'FLAG_SPEAKER_RECOGNITION'
export const CAP_AUDIO_TRANSFORM = 'FLAG_AUDIO_TRANSFORM'
export const CAP_REALTIME_AUDIO = 'FLAG_REALTIME_AUDIO'
export const CAP_SCORE = 'FLAG_SCORE'
export const CAP_DECISIONS = 'FLAG_DECISIONS'
export const CAP_TOKEN_CLASSIFY = 'FLAG_TOKEN_CLASSIFY'
+48
View File
@@ -0,0 +1,48 @@
export function isPdf(file) {
return file?.type === 'application/pdf' || /\.pdf$/i.test(file?.name || '')
}
// pdf.js and its worker are loaded on first use so the main bundle does not
// pay for them when nobody attaches a PDF.
async function loadPdfjs() {
const [pdfjs, worker] = await Promise.all([
import('pdfjs-dist'),
import('pdfjs-dist/build/pdf.worker.min.mjs?url'),
])
pdfjs.GlobalWorkerOptions.workerSrc = worker.default
return pdfjs
}
// Returns the text layer of every page, one block per page. Throws when the
// file cannot be parsed or has no text layer (scanned PDFs): sending an empty
// attachment to the model would look like success and silently lose the file.
export async function extractPdfText(file) {
const pdfjs = await loadPdfjs()
const data = new Uint8Array(await file.arrayBuffer())
const doc = await pdfjs.getDocument({ data }).promise
try {
const pages = []
for (let i = 1; i <= doc.numPages; i++) {
const page = await doc.getPage(i)
const content = await page.getTextContent()
let text = ''
for (const item of content.items) {
text += item.str
text += item.hasEOL ? '\n' : ''
}
pages.push(text.trim())
}
const text = pages.filter(Boolean).join('\n\n')
if (!text) throw new Error('PDF has no extractable text')
return text
} finally {
await doc.destroy()
}
}
// Text of an attached non-media file. PDFs go through pdf.js; everything else
// is read as UTF-8.
export async function readAttachmentText(file) {
if (isPdf(file)) return extractPdfText(file)
return file.text().catch(() => '')
}
+8 -6
View File
@@ -13,9 +13,10 @@ type TranscriptionSegment struct {
}
type TranscriptionWord struct {
Start time.Duration `json:"start"`
End time.Duration `json:"end"`
Text string `json:"text"`
Start time.Duration `json:"start"`
End time.Duration `json:"end"`
Text string `json:"text"`
Speaker string `json:"speaker,omitempty"`
}
type TranscriptionResult struct {
@@ -42,9 +43,10 @@ type TranscriptionSegmentSeconds struct {
}
type TranscriptionWordSeconds struct {
Start float64 `json:"start"`
End float64 `json:"end"`
Text string `json:"text"`
Start float64 `json:"start"`
End float64 `json:"end"`
Text string `json:"text"`
Speaker string `json:"speaker,omitempty"`
}
type TranscriptionResultSeconds struct {
@@ -0,0 +1,304 @@
package agentpool_test
import (
"context"
"encoding/json"
"fmt"
"io"
"net/http"
"net/http/httptest"
"strings"
"sync"
"github.com/mudler/LocalAGI/core/sse"
"github.com/mudler/LocalAGI/core/state"
"github.com/mudler/LocalAGI/core/types"
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/services/agentpool"
. "github.com/onsi/gomega"
)
type fakeLLMRequest struct {
Model string
Stream bool
Messages []map[string]any
Tools []string
ToolChoice any
}
// fakeLLM is an OpenAI-compatible chat endpoint. The standalone pool reaches its
// LLM over HTTP (apiURL), so a real server is the only seam that exercises the
// whole agent loop without a model.
type fakeLLM struct {
srv *httptest.Server
mu sync.Mutex
reply string
requests []fakeLLMRequest
toolName string
toolArgs string
// failStatus, when non-zero, makes every chat completion fail with that
// HTTP status so a spec can drive the agent's error path.
failStatus int
}
func newFakeLLM(reply string) *fakeLLM {
f := &fakeLLM{reply: reply}
f.srv = httptest.NewServer(http.HandlerFunc(f.handle))
return f
}
func (f *fakeLLM) URL() string { return f.srv.URL }
func (f *fakeLLM) Close() { f.srv.Close() }
func (f *fakeLLM) SetReply(r string) {
f.mu.Lock()
defer f.mu.Unlock()
f.reply = r
}
// SetToolCall makes the fake answer with one call to the named function until
// the conversation carries a tool result, then with the plain reply. Keying on
// the tool message rather than a request counter keeps the fake independent of
// how many planning requests the agent makes before it runs the tool. Only
// the LocalAGI counter-action specs use it today; P2 keeps it for the MCP
// tool fixture that replaces them, since the native executor ignores
// Actions. The
// flip side: tool mode stays on until a request carries a role "tool"
// message, so a client that restarts with trimmed history would get the
// tool call again and loop until its iteration cap.
func (f *fakeLLM) SetToolCall(name, argsJSON string) {
f.mu.Lock()
defer f.mu.Unlock()
f.toolName = name
f.toolArgs = argsJSON
}
// SetFailure makes every chat completion answer with the given HTTP status and
// an OpenAI-style error body, which is what an unreachable or broken backend
// looks like to the agent. Requests are still recorded so a spec can count
// retries.
func (f *fakeLLM) SetFailure(status int) {
f.mu.Lock()
defer f.mu.Unlock()
f.failStatus = status
}
func (f *fakeLLM) Requests() []fakeLLMRequest {
f.mu.Lock()
defer f.mu.Unlock()
return append([]fakeLLMRequest(nil), f.requests...)
}
func (f *fakeLLM) handle(w http.ResponseWriter, r *http.Request) {
body, _ := io.ReadAll(r.Body)
var req struct {
Model string `json:"model"`
Stream bool `json:"stream"`
Messages []map[string]any `json:"messages"`
Tools []struct {
Function struct {
Name string `json:"name"`
} `json:"function"`
} `json:"tools"`
ToolChoice any `json:"tool_choice"`
}
_ = json.Unmarshal(body, &req)
var tools []string
for _, t := range req.Tools {
tools = append(tools, t.Function.Name)
}
hasToolResult := false
for _, m := range req.Messages {
if m["role"] == "tool" {
hasToolResult = true
}
}
f.mu.Lock()
f.requests = append(f.requests, fakeLLMRequest{Model: req.Model, Stream: req.Stream, Messages: req.Messages, Tools: tools, ToolChoice: req.ToolChoice})
reply := f.reply
failStatus := f.failStatus
var toolCall map[string]any
if f.toolName != "" && !hasToolResult {
toolCall = map[string]any{
"index": 0, "id": "call_fake", "type": "function",
"function": map[string]any{"name": f.toolName, "arguments": f.toolArgs},
}
}
f.mu.Unlock()
if failStatus != 0 {
w.Header().Set("Content-Type", "application/json")
w.WriteHeader(failStatus)
_ = json.NewEncoder(w).Encode(map[string]any{
"error": map[string]any{"message": "fake backend failure", "type": "server_error", "code": failStatus},
})
return
}
message := map[string]any{"role": "assistant", "content": reply}
finish := "stop"
if toolCall != nil {
// "index" belongs only to streaming deltas, so the non-streaming
// message carries a copy without it.
plain := map[string]any{}
for k, v := range toolCall {
if k != "index" {
plain[k] = v
}
}
message = map[string]any{"role": "assistant", "content": "", "tool_calls": []map[string]any{plain}}
finish = "tool_calls"
}
if !req.Stream {
w.Header().Set("Content-Type", "application/json")
_ = json.NewEncoder(w).Encode(map[string]any{
"id": "chatcmpl-fake", "object": "chat.completion", "model": req.Model,
"choices": []map[string]any{{
"index": 0,
"message": message,
"finish_reason": finish,
}},
"usage": map[string]any{"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
})
return
}
w.Header().Set("Content-Type", "text/event-stream")
// A write error means the client went away mid-stream; the handler just
// stops writing so a disconnect never blocks or fails the fake.
write := func(s string) bool {
_, err := io.WriteString(w, s)
return err == nil
}
chunk := func(delta map[string]any, finish any) bool {
b, _ := json.Marshal(map[string]any{
"id": "chatcmpl-fake", "object": "chat.completion.chunk", "model": req.Model,
"choices": []map[string]any{{"index": 0, "delta": delta, "finish_reason": finish}},
})
return write(fmt.Sprintf("data: %s\n\n", b))
}
first := map[string]any{"role": "assistant", "content": reply}
if toolCall != nil {
first = map[string]any{"role": "assistant", "tool_calls": []map[string]any{toolCall}}
}
if !chunk(first, nil) || !chunk(map[string]any{}, finish) || !write("data: [DONE]\n\n") {
return
}
if fl, ok := w.(http.Flusher); ok {
fl.Flush()
}
}
// startStandalone boots a real standalone AgentPoolService on stateDir with its
// LLM pointed at llmURL. It registers no cleanup itself: callers own Stop().
func startStandalone(stateDir, llmURL string) *agentpool.AgentPoolService {
cfg := config.NewApplicationConfig()
cfg.AgentPool = config.AgentPoolConfig{
Enabled: true,
StateDir: stateDir,
APIURL: llmURL,
DefaultModel: "fake-model",
Timeout: "30s",
}
svc, err := agentpool.NewAgentPoolService(cfg)
Expect(err).ToNot(HaveOccurred())
Expect(svc.Start(context.Background())).To(Succeed())
return svc
}
func newAgentConfig(name string) *state.AgentConfig {
return &state.AgentConfig{
Name: name,
Model: "fake-model",
Description: "contract test agent",
SystemPrompt: "You are a test agent.",
}
}
// Engine-specific: awaitRunning and collectSSE reach into LocalAGI types
// (agent.Agent via GetAgentForUser, types.NewJob, sse.Manager and
// sse.NewClient). P1 must re-seat them when LocalAGI types leave the service
// signatures; the specs that call them should not need to change.
// awaitRunning blocks until the agent's Run loop is serving jobs. The pool
// starts Run in a goroutine and LocalAGI's Scheduler.Start and Scheduler.Stop
// are unsynchronized: a Stop (update, delete, svc.Stop) that lands while Start
// is still running can nil the scheduler context under the poll goroutine and
// crash the test binary. Run starts its workers only after Scheduler.Start has
// returned, and jobQueue is unbuffered, so Execute returning proves Start is
// done. The job's context is already cancelled, so the worker finishes it as
// expired without calling the LLM or recording an observable. This depends on
// LocalAGI not short-circuiting a cancelled job before a worker receives it,
// so re-check it when LocalAGI is bumped; the timeout turns a hang into a
// failure if that ever changes.
func awaitRunning(svc *agentpool.AgentPoolService, userID, name string) {
a := svc.GetAgentForUser(userID, name)
Expect(a).ToNot(BeNil())
ctx, cancel := context.WithCancel(context.Background())
cancel()
done := make(chan struct{})
go func() {
defer close(done)
a.Execute(types.NewJob(types.WithContext(ctx)))
}()
Eventually(done, "10s").Should(BeClosed())
}
type sseEvent struct {
Name string
Data map[string]any
}
// collectSSE registers a listener on the agent's SSE manager and records every
// event until stop is called. Subscribe before Chat so nothing is missed.
// The manager replays its last 10 events on Register, so a spec that chats
// twice with a fresh collector sees the first turn's completed status too.
func collectSSE(svc *agentpool.AgentPoolService, userID, name string) (events func() []sseEvent, stop func()) {
mgr := svc.GetSSEManagerForUser(userID, name)
Expect(mgr).ToNot(BeNil())
// Include the user: the manager keys listeners by ID, so two users'
// same-named agents must never share one if a manager is ever shared.
client := sse.NewClient("contract-" + userID + "-" + name)
mgr.Register(client)
var mu sync.Mutex
var got []sseEvent
done := make(chan struct{})
go func() {
for {
select {
case <-done:
return
case env, ok := <-client.Chan():
if !ok {
return
}
ev := sseEvent{}
for _, line := range strings.Split(env.String(), "\n") {
switch {
case strings.HasPrefix(line, "event:"):
ev.Name = strings.TrimSpace(strings.TrimPrefix(line, "event:"))
case strings.HasPrefix(line, "data:"):
// The parse error is ignored because the hud event's
// data is not JSON; those events keep only their name.
_ = json.Unmarshal([]byte(strings.TrimSpace(strings.TrimPrefix(line, "data:"))), &ev.Data)
}
}
mu.Lock()
got = append(got, ev)
mu.Unlock()
}
}
}()
return func() []sseEvent {
mu.Lock()
defer mu.Unlock()
return append([]sseEvent(nil), got...)
}, func() {
close(done)
mgr.Unregister(client.ID())
}
}
@@ -0,0 +1,157 @@
package agentpool_test
import (
"net/http"
"github.com/mudler/LocalAI/core/services/agentpool"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("standalone chat contract", func() {
var (
llm *fakeLLM
svc *agentpool.AgentPoolService
)
BeforeEach(func() {
llm = newFakeLLM("pong")
// Ginkgo runs cleanups LIFO: registering Close first stops the pool
// before its LLM goes away.
DeferCleanup(llm.Close)
svc = startStandalone(GinkgoT().TempDir(), llm.URL())
DeferCleanup(svc.Stop)
Expect(svc.CreateAgentForUser("alice", newAgentConfig("chatty"))).To(Succeed())
awaitRunning(svc, "alice", "chatty")
})
It("streams the user message, processing, agent reply and completed status over SSE", func() {
events, stop := collectSSE(svc, "alice", "chatty")
defer stop()
msgID, err := svc.ChatForUser("alice", "chatty", "ping")
Expect(err).ToNot(HaveOccurred())
Expect(msgID).ToNot(BeEmpty())
Eventually(func() []sseEvent { return statusEvents(events(), "completed") }, "30s", "100ms").
ShouldNot(BeEmpty(), "no completed status event")
var user, agent, processing bool
for _, e := range events() {
switch {
case e.Name == "json_message" && e.Data["sender"] == "user":
Expect(e.Data["content"]).To(Equal("ping"))
user = true
case e.Name == "json_message" && e.Data["sender"] == "agent":
Expect(e.Data["content"]).To(ContainSubstring("pong"))
// Current standalone shape: the reply id is the id ChatForUser
// returned plus "-agent". The UI correlates on message_id
// (AgentChat.jsx), which the distributed dispatcher sends; a
// native engine may send either, so flip this deliberately.
Expect(e.Data).To(HaveKeyWithValue("id", msgID+"-agent"))
agent = true
case e.Name == "json_message_status" && e.Data["status"] == "processing":
processing = true
}
}
Expect(user).To(BeTrue(), "user json_message")
Expect(processing).To(BeTrue(), "processing status")
Expect(agent).To(BeTrue(), "agent json_message")
})
It("sends the user's message to the LLM under the configured model", func() {
_, err := svc.ChatForUser("alice", "chatty", "ping")
Expect(err).ToNot(HaveOccurred())
Eventually(llm.Requests, "30s", "100ms").ShouldNot(BeEmpty())
req := llm.Requests()[0]
Expect(req.Model).To(Equal("fake-model"))
// Observed shape: content is a plain string, not a parts array. A
// switch to parts would change what an OpenAI-compatible backend sees.
Expect(req.Messages).To(ContainElement(And(
HaveKeyWithValue("role", "user"),
HaveKeyWithValue("content", "ping"),
)))
})
// The chat page clears its "processing" state only on an agent
// json_message or a json_error, so a failed turn must end in json_error
// followed by completed, never in silence. The fake fails every request;
// cogito retries the decision 5 times with a linear 1s..5s backoff, so the
// turn settles after about 15s, hence the 60s budget. The error text is
// cogito's wrapped chain and is not pinned.
It("reports a failing LLM as json_error then completed, with no agent reply", func() {
llm.SetFailure(http.StatusInternalServerError)
events, stop := collectSSE(svc, "alice", "chatty")
defer stop()
_, err := svc.ChatForUser("alice", "chatty", "ping")
Expect(err).ToNot(HaveOccurred())
Eventually(func() []sseEvent { return statusEvents(events(), "completed") }, "60s", "100ms").
ShouldNot(BeEmpty(), "no completed status event")
errorAt, completedAt := -1, -1
for i, e := range events() {
switch {
case e.Name == "json_error" && errorAt < 0:
errorAt = i
Expect(e.Data).To(HaveKeyWithValue("error", And(BeAssignableToTypeOf(""), Not(BeEmpty()))))
case e.Name == "json_message_status" && e.Data["status"] == "completed" && completedAt < 0:
completedAt = i
case e.Name == "json_message" && e.Data["sender"] == "agent":
Fail("a failed turn must not produce an agent json_message")
}
}
Expect(errorAt).To(BeNumerically(">=", 0), "no json_error event")
Expect(errorAt).To(BeNumerically("<", completedAt), "json_error must precede completed")
Expect(llm.Requests()).ToNot(BeEmpty())
})
It("reports chat with an unknown agent as ErrAgentNotFound", func() {
_, err := svc.ChatForUser("alice", "ghost", "hi")
Expect(err).To(MatchError(agentpool.ErrAgentNotFound))
})
It("reports chat with a deleted agent as ErrAgentNotFound", func() {
Expect(svc.DeleteAgentForUser("alice", "chatty")).To(Succeed())
_, err := svc.ChatForUser("alice", "chatty", "hi")
Expect(err).To(MatchError(agentpool.ErrAgentNotFound))
})
It("does not deliver one user's chat events to another user's agent of the same name", func() {
Expect(svc.CreateAgentForUser("bob", newAgentConfig("chatty"))).To(Succeed())
awaitRunning(svc, "bob", "chatty")
aliceEvents, stopAlice := collectSSE(svc, "alice", "chatty")
defer stopAlice()
bobEvents, stopBob := collectSSE(svc, "bob", "chatty")
defer stopBob()
_, err := svc.ChatForUser("alice", "chatty", "ping")
Expect(err).ToNot(HaveOccurred())
Eventually(func() []sseEvent { return statusEvents(aliceEvents(), "completed") }, "30s", "100ms").ShouldNot(BeEmpty())
// LocalAGI pushes a "hud" snapshot of each agent's own state every
// second, so bob's stream is not silent; what must never reach it is
// anything produced by alice's chat.
Consistently(func() []string {
var names []string
for _, e := range bobEvents() {
if e.Name != "hud" {
names = append(names, e.Name)
}
}
return names
}, "1s", "100ms").Should(BeEmpty())
})
})
func statusEvents(events []sseEvent, status string) []sseEvent {
var out []sseEvent
for _, e := range events {
if e.Name == "json_message_status" && e.Data["status"] == status {
out = append(out, e)
}
}
return out
}
@@ -0,0 +1,186 @@
package agentpool_test
import (
"encoding/json"
"github.com/mudler/LocalAI/core/services/agentpool"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("standalone agent service contract", func() {
var (
llm *fakeLLM
dir string
)
BeforeEach(func() {
llm = newFakeLLM("hello from the fake model")
// DeferCleanup runs after AfterEach and in LIFO order, so registering
// here lets a pool started later stop before its LLM goes away.
DeferCleanup(llm.Close)
dir = GinkgoT().TempDir()
})
It("boots against a fake LLM and lists no agents", func() {
svc := startStandalone(dir, llm.URL())
DeferCleanup(svc.Stop)
Expect(svc.ListAgentsForUser("")).To(BeEmpty())
})
Context("agent CRUD", func() {
var svc *agentpool.AgentPoolService
BeforeEach(func() {
svc = startStandalone(dir, llm.URL())
DeferCleanup(svc.Stop)
})
It("creates, reads back, updates and deletes an agent", func() {
Expect(svc.CreateAgentForUser("alice", newAgentConfig("helper"))).To(Succeed())
awaitRunning(svc, "alice", "helper")
got := svc.GetAgentConfigForUser("alice", "helper")
Expect(got).ToNot(BeNil())
// The pool stores the key ("alice:helper"); the API must show the bare name.
Expect(got.Name).To(Equal("helper"))
Expect(got.Model).To(Equal("fake-model"))
Expect(svc.ListAgentsForUser("alice")).To(HaveKeyWithValue("helper", true))
updated := newAgentConfig("helper")
updated.Description = "changed"
Expect(svc.UpdateAgentForUser("alice", "helper", updated)).To(Succeed())
// Update restarts the agent, so the new instance needs the same wait.
awaitRunning(svc, "alice", "helper")
Expect(svc.GetAgentConfigForUser("alice", "helper").Description).To(Equal("changed"))
Expect(svc.DeleteAgentForUser("alice", "helper")).To(Succeed())
Expect(svc.GetAgentConfigForUser("alice", "helper")).To(BeNil())
Expect(svc.ListAgentsForUser("alice")).ToNot(HaveKey("helper"))
})
It("reports an update of a missing agent as ErrAgentNotFound", func() {
err := svc.UpdateAgentForUser("alice", "ghost", newAgentConfig("ghost"))
Expect(err).To(MatchError(agentpool.ErrAgentNotFound))
})
It("keeps two users' agents with the same name apart", func() {
a := newAgentConfig("shared-name")
a.Description = "alice's"
b := newAgentConfig("shared-name")
b.Description = "bob's"
Expect(svc.CreateAgentForUser("alice", a)).To(Succeed())
Expect(svc.CreateAgentForUser("bob", b)).To(Succeed())
awaitRunning(svc, "alice", "shared-name")
awaitRunning(svc, "bob", "shared-name")
Expect(svc.GetAgentConfigForUser("alice", "shared-name").Description).To(Equal("alice's"))
Expect(svc.GetAgentConfigForUser("bob", "shared-name").Description).To(Equal("bob's"))
Expect(svc.DeleteAgentForUser("alice", "shared-name")).To(Succeed())
Expect(svc.GetAgentConfigForUser("alice", "shared-name")).To(BeNil())
Expect(svc.GetAgentConfigForUser("bob", "shared-name")).ToNot(BeNil())
grouped := svc.ListAllAgentsGrouped()
Expect(grouped).To(HaveKey("bob"))
Expect(grouped).ToNot(HaveKey("alice"))
})
It("round-trips a config through export and import without a user", func() {
cfg := newAgentConfig("portable")
cfg.Description = "carry me"
Expect(svc.CreateAgentForUser("", cfg)).To(Succeed())
awaitRunning(svc, "", "portable")
data, err := svc.ExportAgentForUser("", "portable")
Expect(err).ToNot(HaveOccurred())
Expect(svc.DeleteAgentForUser("", "portable")).To(Succeed())
Expect(svc.ImportAgentForUser("", data)).To(Succeed())
awaitRunning(svc, "", "portable")
got := svc.GetAgentConfigForUser("", "portable")
Expect(got).ToNot(BeNil())
Expect(got.Description).To(Equal("carry me"))
})
// Known defect pinned on purpose: export returns the stored config, whose
// name is the pool key, and import refuses ":" in names. A rewrite that
// fixes this must flip this spec rather than silently change behavior.
It("known defect: exports a user's agent under its pool key, which import then rejects", func() {
cfg := newAgentConfig("portable")
cfg.Description = "carry me"
Expect(svc.CreateAgentForUser("alice", cfg)).To(Succeed())
awaitRunning(svc, "alice", "portable")
data, err := svc.ExportAgentForUser("alice", "portable")
Expect(err).ToNot(HaveOccurred())
var out map[string]any
Expect(json.Unmarshal(data, &out)).To(Succeed())
Expect(out["name"]).To(Equal("alice:portable"))
Expect(svc.DeleteAgentForUser("alice", "portable")).To(Succeed())
Expect(svc.ImportAgentForUser("alice", data)).To(MatchError(ContainSubstring("invalid characters")))
Expect(svc.GetAgentConfigForUser("alice", "portable")).To(BeNil())
})
// Engine-specific: P2/P5 flips this because the native import drops
// unknown fields (connectors, actions) with a warning, so the export
// will no longer carry them.
// P5 strips connectors and actions and the P2 migration reads old configs,
// so record what a config that carries them looks like today. LocalAGI
// logs "Failed to create IRC client" for this fixture because the IRC
// config has no nickname; that is expected and is not a failure.
It("accepts and returns a config that carries connectors and actions", func() {
raw := []byte(`{
"name": "legacy",
"model": "fake-model",
"description": "old style",
"connectors": [{"type": "irc", "config": "{}"}],
"actions": [{"name": "search", "config": "{}"}]
}`)
Expect(svc.ImportAgentForUser("alice", raw)).To(Succeed())
awaitRunning(svc, "alice", "legacy")
data, err := svc.ExportAgentForUser("alice", "legacy")
Expect(err).ToNot(HaveOccurred())
var out map[string]any
Expect(json.Unmarshal(data, &out)).To(Succeed())
Expect(out["connectors"]).To(HaveLen(1))
Expect(out["actions"]).To(HaveLen(1))
// The P2 migration reads these element shapes. The action name is
// what LocalAGI actually stores, observed as the name sent in, not a
// resolved alias.
Expect(out["connectors"]).To(ConsistOf(And(
HaveKeyWithValue("type", "irc"),
HaveKeyWithValue("config", BeAssignableToTypeOf("")),
)))
Expect(out["actions"]).To(ConsistOf(And(
HaveKeyWithValue("name", "search"),
HaveKeyWithValue("config", BeAssignableToTypeOf("")),
)))
})
})
Context("pause and resume", func() {
It("toggles the active flag reported by the list", func() {
svc := startStandalone(dir, llm.URL())
DeferCleanup(svc.Stop)
Expect(svc.CreateAgentForUser("alice", newAgentConfig("napper"))).To(Succeed())
awaitRunning(svc, "alice", "napper")
Expect(svc.ListAgentsForUser("alice")).To(HaveKeyWithValue("napper", true))
Expect(svc.PauseAgentForUser("alice", "napper")).To(Succeed())
Expect(svc.ListAgentsForUser("alice")).To(HaveKeyWithValue("napper", false))
Expect(svc.ResumeAgentForUser("alice", "napper")).To(Succeed())
Expect(svc.ListAgentsForUser("alice")).To(HaveKeyWithValue("napper", true))
})
It("reports pausing a missing agent as ErrAgentNotFound", func() {
svc := startStandalone(dir, llm.URL())
DeferCleanup(svc.Stop)
Expect(svc.PauseAgentForUser("alice", "ghost")).To(MatchError(agentpool.ErrAgentNotFound))
})
})
})
@@ -0,0 +1,101 @@
package agentpool_test
import (
"encoding/json"
"os"
"path/filepath"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("standalone persistence contract", func() {
var (
llm *fakeLLM
dir string
)
BeforeEach(func() {
llm = newFakeLLM("ok")
dir = GinkgoT().TempDir()
DeferCleanup(llm.Close)
})
// The P2 migration imports this file, so its layout is an interface.
It("writes pool.json keyed by userID:name with the agent config as value", func() {
svc := startStandalone(dir, llm.URL())
Expect(svc.CreateAgentForUser("alice", newAgentConfig("keeper"))).To(Succeed())
awaitRunning(svc, "alice", "keeper")
Expect(svc.CreateAgentForUser("", newAgentConfig("anon"))).To(Succeed())
awaitRunning(svc, "", "anon")
svc.Stop()
raw, err := os.ReadFile(filepath.Join(dir, "pool.json"))
Expect(err).ToNot(HaveOccurred())
var pool map[string]map[string]any
Expect(json.Unmarshal(raw, &pool)).To(Succeed())
Expect(pool).To(HaveKey("alice:keeper"))
Expect(pool).To(HaveKey("anon"))
Expect(pool["alice:keeper"]["model"]).To(Equal("fake-model"))
// The stored name repeats the key, prefix included: the P2 importer
// strips the prefix, so it depends on this.
Expect(pool["alice:keeper"]["name"]).To(Equal("alice:keeper"))
Expect(pool["anon"]["name"]).To(Equal("anon"))
})
It("restores agents after a restart on the same state dir", func() {
svc := startStandalone(dir, llm.URL())
Expect(svc.CreateAgentForUser("alice", newAgentConfig("survivor"))).To(Succeed())
awaitRunning(svc, "alice", "survivor")
svc.Stop()
again := startStandalone(dir, llm.URL())
DeferCleanup(again.Stop)
awaitRunning(again, "alice", "survivor")
Expect(again.GetAgentConfigForUser("alice", "survivor")).ToNot(BeNil())
Expect(again.ListAgentsForUser("alice")).To(HaveKey("survivor"))
})
// Pause is only an in-memory flag on the running agent: pool.json has no
// status field and no per-agent file records pause, so a restart brings
// the agent back active. Pinned as a known gap for the native-store migration to
// close on purpose rather than by accident.
It("known gap: does not keep a paused agent paused across a restart", func() {
svc := startStandalone(dir, llm.URL())
Expect(svc.CreateAgentForUser("alice", newAgentConfig("sleeper"))).To(Succeed())
awaitRunning(svc, "alice", "sleeper")
Expect(svc.PauseAgentForUser("alice", "sleeper")).To(Succeed())
Expect(svc.ListAgentsForUser("alice")).To(HaveKeyWithValue("sleeper", false))
svc.Stop()
again := startStandalone(dir, llm.URL())
DeferCleanup(again.Stop)
awaitRunning(again, "alice", "sleeper")
Expect(again.ListAgentsForUser("alice")).To(HaveKeyWithValue("sleeper", true))
})
// The /v1/responses interceptor decides "is this model an agent" with
// GetAgent(name) using the raw pool key, with no user prefix.
// Known gap, not a contract: any caller who sends model "alice:mine" runs
// alice's agent (the interceptor has no user check), while alice's own
// request for "mine" falls through; a later fix must not read as a break.
It("known gap: resolves an agent by its raw key for the responses interceptor", func() {
svc := startStandalone(dir, llm.URL())
DeferCleanup(svc.Stop)
Expect(svc.CreateAgentForUser("", newAgentConfig("global-agent"))).To(Succeed())
awaitRunning(svc, "", "global-agent")
Expect(svc.CreateAgentForUser("alice", newAgentConfig("mine"))).To(Succeed())
awaitRunning(svc, "alice", "mine")
Expect(svc.GetAgent("global-agent")).ToNot(BeNil())
Expect(svc.GetAgent("alice:mine")).ToNot(BeNil())
Expect(svc.GetAgent("mine")).To(BeNil(), "current gap: a user's own agent is not found by its bare name, only by its pool key")
Expect(svc.GetAgent("nope")).To(BeNil())
})
It("exposes the state dir it was started with", func() {
svc := startStandalone(dir, llm.URL())
DeferCleanup(svc.Stop)
Expect(svc.StateDir()).To(Equal(dir))
})
})
@@ -0,0 +1,232 @@
package agentpool_test
import (
"encoding/json"
"fmt"
"github.com/mudler/LocalAGI/core/state"
"github.com/mudler/LocalAGI/core/types"
"github.com/mudler/LocalAI/core/services/agentpool"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("standalone status and observables contract", func() {
var (
llm *fakeLLM
svc *agentpool.AgentPoolService
)
BeforeEach(func() {
llm = newFakeLLM("done")
// Cleanups run in reverse, so the pool stops before the fake LLM goes away.
DeferCleanup(llm.Close)
svc = startStandalone(GinkgoT().TempDir(), llm.URL())
DeferCleanup(svc.Stop)
Expect(svc.CreateAgentForUser("alice", newAgentConfig("observed"))).To(Succeed())
awaitRunning(svc, "alice", "observed")
})
// runOnce chats once and waits for the completed status. ChatForUser sends
// that status as soon as Ask returns, before the job finalizers have written
// the finished observable, so callers that read observables use
// settledObservable instead.
runOnce := func() {
events, stop := collectSSE(svc, "alice", "observed")
defer stop()
_, err := svc.ChatForUser("alice", "observed", "go")
Expect(err).ToNot(HaveOccurred())
Eventually(func() []sseEvent { return statusEvents(events(), "completed") }, "30s", "100ms").
ShouldNot(BeEmpty(), "no completed status event")
}
// settleRun chats once with the named agent and returns its observables
// and SSE events after the job finalizers are done. Three observer.Update
// calls follow Finish (the Execute finalizer, the consumeJob finalizer and
// the deferred MakeLastProgressCompletion update, LocalAGI agent.go
// 1182-1187), and Update re-appends an observable whose id is gone, so
// clearing while one is still pending would bring the observable back. Each Update also sends an observable_update event, so the run is
// settled once the root observable carries a completion, the completed
// status went out, and no further observable_update arrives.
settleRun := func(name string) ([]map[string]any, []sseEvent) {
events, stop := collectSSE(svc, "alice", name)
defer stop()
_, err := svc.ChatForUser("alice", name, "go")
Expect(err).ToNot(HaveOccurred())
Eventually(func() []sseEvent { return statusEvents(events(), "completed") }, "30s", "100ms").
ShouldNot(BeEmpty(), "no completed status event")
Eventually(func(g Gomega) {
raw, err := svc.GetAgentObservablesForUser("alice", name)
g.Expect(err).ToNot(HaveOccurred())
rootDone := false
for _, r := range raw {
var o map[string]any
g.Expect(json.Unmarshal(r, &o)).To(Succeed())
if _, child := o["parent_id"]; !child {
_, rootDone = o["completion"]
}
}
g.Expect(rootDone).To(BeTrue())
}, "30s", "100ms").Should(Succeed())
updates := func() int {
n := 0
for _, e := range events() {
if e.Name == "observable_update" {
n++
}
}
return n
}
var last int
Eventually(func() bool {
n := updates()
stable := n == last
last = n
return stable
}, "10s", "300ms").Should(BeTrue())
Consistently(updates, "300ms", "50ms").Should(Equal(last))
raw, err := svc.GetAgentObservablesForUser("alice", name)
Expect(err).ToNot(HaveOccurred())
obs := make([]map[string]any, len(raw))
for i, r := range raw {
Expect(json.Unmarshal(r, &obs[i])).To(Succeed())
}
return obs, events()
}
// settledObservable returns the first observable of a settled plain run.
settledObservable := func() map[string]any {
obs, _ := settleRun("observed")
Expect(obs).ToNot(BeEmpty())
return obs[0]
}
It("returns an empty observable list before any run", func() {
obs, err := svc.GetAgentObservablesForUser("alice", "observed")
Expect(err).ToNot(HaveOccurred())
Expect(obs).To(BeEmpty())
})
// Discovery: a plain-content reply (no tool call) is enough for LocalAGI
// to record a "job" observable, so the action fallback was not needed.
// parent_id is not pinned: it is omitempty and a root job has none.
It("records observables after a run with the fields the agent status UI reads", func() {
first := settledObservable()
Expect(first).To(HaveKey("id"))
Expect(first).To(HaveKey("creation"))
Expect(first).To(HaveKey("completion"))
})
It("clears observables", func() {
settledObservable()
Expect(svc.ClearAgentObservablesForUser("alice", "observed")).To(Succeed())
obs, err := svc.GetAgentObservablesForUser("alice", "observed")
Expect(err).ToNot(HaveOccurred())
Expect(obs).To(BeEmpty())
})
It("reports observables of a missing agent as ErrAgentNotFound", func() {
_, err := svc.GetAgentObservablesForUser("alice", "ghost")
Expect(err).To(MatchError(agentpool.ErrAgentNotFound))
Expect(svc.ClearAgentObservablesForUser("alice", "ghost")).To(MatchError(agentpool.ErrAgentNotFound))
})
// LocalAGI only creates a status entry when an action result is recorded,
// so an agent whose runs never called an action looks the same as a
// missing one. Pinned as current behavior, not as a desirable contract.
It("returns a nil status for an agent with no action results, as for a missing one", func() {
Expect(svc.GetAgentStatusForUser("alice", "observed")).To(BeNil())
runOnce()
Consistently(func() any { return svc.GetAgentStatusForUser("alice", "observed") }, "500ms", "100ms").Should(BeNil())
Expect(svc.GetAgentStatusForUser("alice", "ghost")).To(BeNil())
})
// Engine-specific: P2 rewrites these three specs because they depend on
// the LocalAGI counter action and the native executor ignores Actions;
// P2 swaps in an MCP tool fixture. The status spec also reads
// types.ActionState, a LocalAGI type that P1 must re-seat when LocalAGI
// types leave the service signatures.
Context("after a run that calls a tool", func() {
BeforeEach(func() {
cfg := newAgentConfig("tooled")
// counter is pure and in-memory, so the action result is
// deterministic without any external service.
cfg.Actions = []state.ActionsConfig{{Name: "counter", Config: "{}"}}
Expect(svc.CreateAgentForUser("alice", cfg)).To(Succeed())
awaitRunning(svc, "alice", "tooled")
llm.SetToolCall("counter", `{"name":"contract","adjustment":1}`)
})
It("records a status entry the status endpoint renders with action, params and result", func() {
settleRun("tooled")
st := svc.GetAgentStatusForUser("alice", "tooled")
Expect(st).ToNot(BeNil())
// Select by action name rather than position: the order of
// status entries is a LocalAGI detail, not part of the contract.
var h types.ActionState
Expect(st.Results()).To(ContainElement(Satisfy(func(s types.ActionState) bool {
return s.ActionCurrentState.Action != nil &&
s.ActionCurrentState.Action.Definition().Name.String() == "counter"
}), &h))
Expect(h.ActionCurrentState.Params).To(HaveKeyWithValue("name", "contract"))
Expect(h.Result).To(ContainSubstring("Created counter 'contract'"))
// Same format string as GetAgentStatusEndpoint: this text is what
// the agent status page shows, so the params must render as JSON
// through ActionParams.String rather than as a Go map.
rendered := fmt.Sprintf("Reasoning: %s\nAction taken: %s\nParameters: %+v\nResult: %s",
h.Reasoning, h.ActionCurrentState.Action.Definition().Name.String(), h.ActionCurrentState.Params, h.Result)
Expect(rendered).To(ContainSubstring("Action taken: counter\n"))
Expect(rendered).To(ContainSubstring(`"name":"contract"`))
Expect(rendered).To(ContainSubstring("Result: Created counter 'contract'"))
})
// The observables tree nests the action under the job that ran it.
// Only the link is pinned: ids, ordering and conversation contents are
// LocalAGI internals.
It("records a child action observable linked to the root job by parent_id", func() {
obs, _ := settleRun("tooled")
Expect(len(obs)).To(BeNumerically(">", 1))
var roots, children []map[string]any
for _, o := range obs {
if _, ok := o["parent_id"]; ok {
children = append(children, o)
} else {
roots = append(roots, o)
}
}
Expect(roots).To(HaveLen(1))
Expect(children).ToNot(BeEmpty())
for _, c := range children {
// Compared as decoded JSON, whatever type the ids have: the id
// type is a LocalAGI detail, the parent link is the contract.
Expect(c["parent_id"]).To(Equal(roots[0]["id"]))
}
// "action" is the LocalAGI name, visible in AgentStatus.jsx; a
// native engine may name it differently, so flip deliberately.
Expect(children).To(ContainElement(And(
HaveKeyWithValue("name", "action"),
HaveKeyWithValue("completion", HaveKeyWithValue("action_result", ContainSubstring("Created counter"))),
)))
})
It("still delivers the agent's final reply over SSE", func() {
_, events := settleRun("tooled")
var replies []sseEvent
for _, e := range events {
if e.Name == "json_message" && e.Data["sender"] == "agent" {
replies = append(replies, e)
}
}
Expect(replies).ToNot(BeEmpty())
Expect(replies[0].Data["content"]).To(ContainSubstring("done"))
})
})
})
+3 -1
View File
@@ -1066,7 +1066,9 @@ known_usecases:
- embeddings
```
Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `llm` (combination of CHAT, COMPLETION, EDIT).
Available flags: `chat`, `completion`, `edit`, `embeddings`, `rerank`, `image`, `transcript`, `tts`, `sound_generation`, `tokenize`, `vad`, `video`, `detection`, `score`, `token_classify`, `decisions`, `llm` (combination of CHAT, COMPLETION, EDIT).
`decisions` marks a model as a decision model for the [Decisions API]({{% relref "features/decisions" %}}) (`POST /v1/systemone`). It is never guessed, and a model that declares it is not listed as a chat, completion or embeddings model.
`token_classify` marks a model as a token-classification (NER) provider for the PII filter (e.g. an `openai-privacy-filter` GGUF). Declare it explicitly together with `embeddings: true` (the classifier loads via TOKEN_CLS pooling). It runs on the dedicated `privacy-filter` backend (`backend/cpp/privacy-filter`), a standalone GGML engine for the `openai-privacy-filter` family - separate from `llama-cpp`, which no longer carries the token-classification path.
@@ -9,6 +9,8 @@ Sound-event classification (audio tagging) answers the question **"what am I hea
LocalAI exposes this through the `/v1/audio/classification` endpoint, modelled after `/v1/audio/transcriptions`. The reference backend is **[ced.cpp](https://github.com/localai-org/ced.cpp)** (CED, a 527-class AudioSet tagger), a small ViT over a log-mel spectrogram ported to ggml with full PyTorch parity. Apache-2.0 weights are redistributable as GGUF.
**[parakeet.cpp](https://github.com/mudler/parakeet.cpp)** can also load a CED model (through `third_party/ced.cpp`) and serve `/v1/audio/classification` from the same backend used for ASR and diarization. It scores the clip in 10 s windows and averages each class's score across the windows before sorting and applying `top_k`/`threshold` - CED's own method for clips longer than one window. Install `parakeet-cpp-ced-tiny` or `parakeet-cpp-ced-base` from the gallery, or point `parameters.model` at a CED GGUF under `backend: parakeet-cpp`. A parakeet-cpp ASR model can also point `sound_model` at a CED GGUF to add live sound events during realtime transcription - see [Realtime API]({{% relref "openai-realtime" %}}).
Because classification is exposed as a regular OpenAI-style endpoint, any HTTP client works - there is no Python dependency on the consumer side.
In distributed mode, LocalAI stages uploaded audio and realtime sound-detection
@@ -59,6 +61,25 @@ curl http://localhost:8080/v1/audio/classification \
-F top_k=10
```
The same request works unchanged against a parakeet-cpp CED model:
```yaml
name: parakeet-ced-tiny
backend: parakeet-cpp
parameters:
model: ced-tiny-q8_0.gguf
known_usecases:
- sound_classification
```
```bash
curl http://localhost:8080/v1/audio/classification \
-H "Content-Type: multipart/form-data" \
-F file="@/path/to/clip.wav" \
-F model="parakeet-ced-tiny" \
-F top_k=10
```
## See also
- [Audio to Text]({{% relref "audio-to-text" %}}) - speech transcription
+31 -2
View File
@@ -9,12 +9,13 @@ url = "/features/audio-diarization/"
Speaker diarization answers the question **"who spoke when?"** - given an audio clip with multiple speakers, it returns time-stamped segments labelled with a stable speaker ID (`SPEAKER_00`, `SPEAKER_01`, …).
LocalAI exposes this through the `/v1/audio/diarization` endpoint, modelled after `/v1/audio/transcriptions`. Four backends are supported today:
LocalAI exposes this through the `/v1/audio/diarization` endpoint, modelled after `/v1/audio/transcriptions`. Five backends are supported today:
- **[sherpa-onnx](https://github.com/k2-fsa/sherpa-onnx)** - pyannote-3.0 segmentation + a speaker-embedding extractor (3D-Speaker, NeMo, WeSpeaker) + fast clustering. Pure diarization - no transcription cost. Recommended when you only need speaker turns.
- **[vibevoice.cpp](https://github.com/microsoft/VibeVoice)** - produces speaker-labelled segments as a by-product of its long-form ASR pass, so you can optionally get a transcript per segment for free.
- **[NeMo-Speech.cpp](https://github.com/NVIDIA/NeMo-Speech.cpp)** - NVIDIA Sortformer, served standalone by the [NeMo-Speech.cpp backend]({{%relref "features/nemo-speech-cpp" %}}). It is end to end, so the speaker capacity is fixed by the checkpoint and the count hints are ignored. The same backend can instead put speaker tags on a transcript, by attaching a Sortformer model to an ASR one.
- **[audio.cpp](https://github.com/0xShug0/audio.cpp)** - the `sortformer_diar` family, served by the multi-modality [audio.cpp backend]({{%relref "features/audio-cpp" %}}).
- **[parakeet.cpp](https://github.com/mudler/parakeet.cpp)** - NVIDIA Nemotron-3-Diarization (Sortformer, up to 8 speakers), served standalone or paired with a Parakeet ASR model for per-segment text. See the [Audio to Text]({{% relref "audio-to-text" %}}) page for the parakeet-cpp option reference.
Because diarization is exposed as a regular OpenAI-compatible endpoint, any HTTP client works. There is no Python dependency on pyannote or NeMo on the consumer side.
@@ -157,10 +158,38 @@ curl http://localhost:8080/v1/audio/diarization \
-F response_format=verbose_json
```
## Backend setup - parakeet-cpp (Nemotron-3-Diarization)
Nemotron-3-Diarization is Sortformer, served standalone or paired with a Parakeet ASR model. Install `parakeet-cpp-nemotron-3-diarization` from the gallery for diarization only, or `parakeet-cpp-nemotron-3-diarization-asr` for the same model paired with `parakeet-cpp-tdt_ctc-110m` through the `asr_model` option:
```yaml
name: parakeet-diarize
backend: parakeet-cpp
parameters:
model: nemotron-3-diarization-q8_0.gguf
options:
- asr_model:tdt_ctc-110m-f16.gguf
known_usecases:
- diarization
```
Getting text on each segment needs both: an `asr_model` companion loaded on the model, and `include_text=true` on the request. With only one of the two, segments carry no text and no error is raised. Sortformer has a fixed speaker capacity and no clustering stage, so `num_speakers`, `min_speakers`, `max_speakers` and `clustering_threshold` are ignored (logged at debug); `min_duration_on` and `min_duration_off` are honored. Speaker labels are the decimal index the model assigned (`"0"`, `"1"`, …), or `"unknown"` when a segment has no diarized speaker.
```bash
curl http://localhost:8080/v1/audio/diarization \
-H "Content-Type: multipart/form-data" \
-F file="@meeting.wav" \
-F model="parakeet-diarize" \
-F include_text=true \
-F response_format=verbose_json
```
Sortformer clusters on voice-like characteristics, not on "is this a human". A loud non-speech sound with voice-like pitch and rhythm (a rooster crow, in one test clip) can come back as its own speaker segment alongside the real speakers. This is model behavior, not a bug in the LocalAI integration: treat an unexpected extra speaker as a hint the clip may contain a non-speech sound, and use [Sound Classification]({{% relref "audio-classification" %}}) to confirm what it is.
## Notes
- **Speaker identity across files**: speaker IDs (`SPEAKER_00`, `SPEAKER_01`, …) are local to each request. To track the same person across multiple recordings, combine `/v1/audio/diarization` with `/v1/voice/embed` (speaker embedding) and maintain your own embedding store.
- **Hints vs. forces**: `num_speakers` overrides clustering when set; `min_speakers` / `max_speakers` are advisory and only honored by backends that expose a range hint. vibevoice.cpp ignores them - its model picks the count itself.
- **Hints vs. forces**: `num_speakers` overrides clustering when set; `min_speakers` / `max_speakers` are advisory and only honored by backends that expose a range hint. vibevoice.cpp and parakeet-cpp (Sortformer) ignore them - the model picks the count itself.
- **Sample rate**: input is automatically converted to 16 kHz mono via ffmpeg before the backend sees it; sherpa-onnx pyannote-3.0 requires 16 kHz.
## See also
+16 -1
View File
@@ -12,7 +12,7 @@ The transcription endpoint allows to convert audio files to text. The endpoint s
- **moonshine**: Ultra-fast transcription engine optimized for low-end devices
- **faster-whisper**: Fast Whisper implementation with CTranslate2
- **WhisperX**: Whisper transcription with word alignment and optional speaker diarization. Set `HF_TOKEN` and pass `diarize=true` to load WhisperX's gated pyannote diarization pipeline.
- **[parakeet-cpp](https://github.com/mudler/parakeet.cpp)**: A C++/ggml port of NVIDIA NeMo Parakeet (FastConformer TDT/CTC/RNNT/hybrid). Runs quantized GGUFs on CPU or GPU, emits word-level timestamps, and supports cache-aware streaming (the `realtime_eou` model surfaces end-of-utterance events).
- **[parakeet-cpp](https://github.com/mudler/parakeet.cpp)**: A C++/ggml port of NVIDIA NeMo Parakeet (FastConformer TDT/CTC/RNNT/hybrid). Runs quantized GGUFs on CPU or GPU, emits word-level timestamps, and supports cache-aware streaming (the `realtime_eou` model surfaces end-of-utterance events). The same backend also loads Nemotron-3-Diarization (`/v1/audio/diarization`) and CED sound models (`/v1/audio/classification`), and can attach either as a companion to a transcription model.
- **llama-cpp**: Route transcription to any multimodal-audio GGUF model served by the `llama-cpp` backend (e.g. [Qwen3-ASR](https://huggingface.co/ggml-org/Qwen3-ASR-0.6B-GGUF), Voxtral, Qwen2-Audio). Under the hood the request is converted into a chat completion with the audio attached via the model's audio encoder - the same path the upstream llama.cpp server uses. Set `backend: llama-cpp` in the model YAML and point `mmproj` at the matching audio encoder.
- **voxtral**: Voxtral-family models served by a dedicated backend
- **[NeMo-Speech.cpp](https://github.com/NVIDIA/NeMo-Speech.cpp)**: NVIDIA's C++/ggml runtime for the Nemotron Speech models. Serves offline, streaming and live transcription, with VAD, punctuation, inverse text normalization and Sortformer speaker tags attached through model options, and covers diarization, speech synthesis and translation from the same backend. See the [NeMo-Speech.cpp backend]({{%relref "features/nemo-speech-cpp" %}}) page for the model options.
@@ -190,6 +190,21 @@ curl http://localhost:8080/v1/audio/transcriptions \
For real-time use, load a cache-aware streaming model (e.g. `realtime_eou_120m-v1-*.gguf`) and pass `-F stream=true`. Deltas are emitted as the audio is decoded, with end-of-utterance events closing each segment.
### Diarization and sound classification
The same backend also serves the `/v1/audio/diarization` and `/v1/audio/classification` endpoints, and can attach a diarization or sound model to a live transcription session. `options:` accepts paths relative to the models directory, or absolute:
| Option | Allowed on | Used for |
|---|---|---|
| `asr_model:<path>` | a diarization model | `include_text` on `/v1/audio/diarization` |
| `diarization_model:<path>` | an ASR model | a `speaker` on transcript segments (and words), and speaker segments during realtime live transcription |
| `sound_model:<path>` | an ASR model | sound events during realtime live transcription |
| `diarization_latency:<model\|low\|very_low\|ultra_low>` | a model with a diarization companion | latency mode for the live speaker stream; default `low` |
With a `diarization_model` companion, `/v1/audio/transcriptions` labels each segment with its `speaker` (`"0"`, `"1"`, ... in order of first appearance) and splits segments where the speaker changes; with `timestamp_granularities[]=word` each word carries its speaker too. With `stream=true` the closing `transcript.text.done` event lists the segments with their speakers. Pass `-F diarize=false` to skip diarization for one request. The diarization GGUF can also be imported directly: `local-ai models import https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/nemotron-3-diarization-f16.gguf`.
The loader rejects a companion whose role duplicates the primary's own (for example `asr_model:` on an already-ASR primary, or `sound_model:` on a CED primary), and rejects a companion GGUF that does not match the role its option names (for example `sound_model:` pointing at an ASR GGUF fails to load, naming the kind it expected). See [Speaker Diarization]({{% relref "audio-diarization" %}}) for the `Diarize` RPC and [Sound Classification]({{% relref "audio-classification" %}}) for `SoundDetection`, and [Realtime API]({{% relref "openai-realtime" %}}) for the live speaker/sound events emitted during a realtime session.
### Segment timestamps
Transcriptions are split into segments the same way NVIDIA NeMo does: a new segment starts after sentence-ending punctuation (`.`, `?`, `!`), and each segment carries `start`/`end` times. This is the default (NeMo's punctuation-only segmentation) and needs no configuration. While streaming, each end-of-utterance closes a segment, now with timestamps.
+164
View File
@@ -0,0 +1,164 @@
+++
disableToc = false
title = "Decisions API"
weight = 66
url = "/features/decisions/"
+++
The Decisions API is a fast, typed decision layer. You send a piece of text (the
*state*) and a set of named questions. A decision model answers each question
with a value and a confidence, in one pass. The model does not generate text, so
there is nothing to parse and no free-form output to validate.
LocalAI serves it on the `/v1/systemone` routes. The request and response shapes
follow the [kev](https://github.com/jaredpalmer/kev) project, and the field names
and question types are the same ones Ollama serves on its `/v1/systemone`
endpoint (Ollama 0.35 and later). The wire contract is called SystemOne; the
capability a model declares is called `decisions`. See
[Compatibility with Ollama](#compatibility-with-ollama) for what differs.
OpenAI announced its own Decisions API in limited preview on 2026-09-29. It has no
public request or response schema yet, so LocalAI does not serve a `/v1/decisions`
route.
## Endpoints
| Endpoint | Method | Description |
|---|---|---|
| `/v1/systemone` | POST | Answer all questions in one pass |
| `/v1/systemone/permute` | POST | Re-run one choice question under `n_perm` option orders |
| `/v1/systemone/separate` | POST | Answer each question in its own pass |
Which route a model can serve depends on its kind:
| Model kind | `/v1/systemone` | `/permute` and `/separate` |
|---|---|---|
| Decision model (`decisions`), such as Laya or GLiNER2.5-Decide | Yes | No, returns `400` |
| Zero-shot NER model (`token_classify`), such as GLiNER2.5 | Yes, through the NER path | Yes |
## Question types
| Type | Answer | Fields in the answer |
|---|---|---|
| `choice` | One option out of a named set | `choice`, `probabilities`, `confidence` |
| `noul` | Yes, no or unknown for a statement | `noul` (0 to 1), `entities` |
| `score` | One level on a scale | `score`, `legend`, `probabilities`, `confidence` |
## Example
```bash
curl http://localhost:8080/v1/systemone -H "Content-Type: application/json" -d '{
"model": "laya-vllm-cpp",
"state": "My order arrived broken and I want my money back. This is the second time.",
"questions": {
"team": {
"type": "choice",
"instructions": "Which team should handle this ticket?",
"criteria": {
"billing": "Payments, invoices and refunds",
"shipping": "Delivery and damaged goods",
"product": "Questions about how the product works"
}
},
"refund_requested": {
"type": "noul",
"instructions": "The customer explicitly asks for a refund"
},
"urgency": {
"type": "score",
"instructions": "How urgent is this ticket?",
"criteria": ["not urgent", "somewhat urgent", "urgent", "critical"]
}
}
}'
```
Answers from a decision model carry a `confidence` value, and the response
reports token usage and `latency_ms`. The NER path does not report token usage.
## Choosing a model
A model can serve the Decisions API only if it is a decision model. Declare the usecase
in the model config:
```yaml
name: laya
backend: vllm-cpp
known_usecases:
- decisions
parameters:
model: convaiinnovations/laya
```
`decisions` is never guessed, and a model that declares it is not listed as a
chat, completion or embeddings model. A model that declares usecases without
`decisions` or `token_classify` gets a `400` from these endpoints that names the
missing usecase. A model that declares `token_classify` and not `decisions` is
served by the zero-shot NER path. A vllm-cpp config that declares no usecases is
treated as a decision model, so setups that predate the flag keep working, but a
config that declares only `chat` (as an older `laya` gallery entry did) now gets
the `400` and needs `known_usecases: [decisions]`.
Install one from the gallery and filter on the `decisions` tag:
| Gallery entry | Model | Notes |
|---|---|---|
| `laya-vllm-cpp` | Laya | ModernBERT-large, non-autoregressive, about 800 MB |
| `gliner25-decide-vllm-cpp` | GLiNER2.5-Decide | DeBERTa-v3-large with a classification head, about 2 GB |
| `tev1-4b-vllm-cpp` | Tev1 4B | Autoregressive Qwen3.5-4B fine-tune that answers with an option letter, about 9.3 GB |
| `tev1-0.8b-vllm-cpp` | Tev1 0.8B | Autoregressive Qwen3.5-0.8B fine-tune that answers with an option letter, about 1.8 GB |
| `kev-0.8b-vllm-cpp` | kev 0.8B | Qwen3.5-0.8B-Base with a merged LoRA and a PointerHead readout, converted for vllm.cpp only, about 1.53 GB |
The engine, [vllm.cpp]({{% relref "features/vllm-cpp" %}}), also supports the
CLM and xor decision models. Those checkpoints need a conversion step, so
they are not gallery entries yet. The kev entry installs a checkpoint that was
already converted with the vllm.cpp `convert-kev.py` script.
Tev1 is an autoregressive decision model. The engine answers each question by
scoring the option letters, so its `confidence` is the entropy measure Ollama
uses. A Tev1 `choice` or `score` question accepts at most 24 options (Ollama
allows 26), because the model is trained on the letters A to X, and every
option needs a nonempty description. The published checkpoints name another
architecture in `config.json`, so the Tev1 gallery entries set
`engine_args.hf_overrides` to load them as `Tev1Model` (see
[Overriding config.json keys]({{% relref "features/vllm-cpp" %}}#overriding-configjson-keys-hf_overrides)).
The same model also answers `/v1/chat/completions` requests.
## Request limits
A request is refused with `400` (or `413` for the body size) when:
- the body is larger than 64 KiB,
- `state` is missing or blank,
- there are no questions, or more than 64,
- a question id is blank,
- a `choice` question has fewer than 2 options or a blank option key,
- a `score` question has fewer than 2 levels,
- a `noul` question has `criteria` with keys other than `"false"` and `"true"`.
A `noul` question may carry `criteria` with a description for each outcome, for
example `{"false": "No refund is requested", "true": "The customer requests a refund"}`.
Some models cap the number of options for a `choice` or `score` question. Models
that answer with a letter accept at most 26, and Tev1 accepts at most 24. The engine refuses more options than
the model supports and the error names the limit.
## Compatibility with Ollama
The field names, question types and answer fields are the same as Ollama's
`/v1/systemone`, so a client written for one works against the other for the
common case. These behaviors differ:
| | Ollama | LocalAI |
|---|---|---|
| `confidence` | `1 - H(p) / ln(N)`, an entropy measure | Computed by the model's pipeline. For kev and Laya it is a normalized margin, so the same probabilities give a different value |
| Errors | `{"error": "message"}` | `{"error": {"message": "...", "type": "invalid_request"}}` |
| `keep_alive` | Sets how long the model stays loaded | Accepted and ignored. Model lifetime follows the LocalAI idle and watchdog settings |
| `state` given as an object | Serialized as JSON text | Rendered as labeled lines, the way kev does it |
| `noul` answer on the NER path | `{type, noul}` | Also carries `entities` |
| Token `usage` | Full prompt lengths across all questions | Whatever the backend reports; the NER path reports 0 |
## Access control
When authentication is on, the three routes need the `decisions` feature. It is
on by default for every user, like the other API features, and an administrator
can turn it off per user.
+107
View File
@@ -129,6 +129,113 @@ A client `session.update` still overrides `type` and `eagerness` per session.
- `false` (default): the transcript accumulated from the live stream is used as-is - the model runs once per utterance and the LLM starts immediately at commit.
- `true`: the committed audio is re-transcribed offline. If the batch decode also ends with the end-of-utterance token the turn proceeds (using the batch transcript); if it does **not**, the commit is cancelled and the session keeps listening - treating the streaming token as a false positive. Both transcripts are compared and logged, which makes this mode a useful diagnostic for how well the streaming and batch decodes align, at the cost of one extra decode per turn.
### Live speaker and sound events (parakeet-cpp)
When the `semantic_vad` transcription model is a parakeet-cpp model loaded with a `diarization_model` and/or `sound_model` companion (see [Audio to Text]({{% relref "audio-to-text" %}})), the realtime session also streams speaker and sound events while a turn is live, alongside the transcript deltas. Nothing needs to change on the client: unrecognized event types are ignored by standard OpenAI Realtime clients.
The transcription model, with its companions:
```yaml
name: parakeet-realtime-scene
backend: parakeet-cpp
parameters:
model: realtime_eou_120m-v1-f16.gguf
options:
- diarization_model:nemotron-3-diarization-q8_0.gguf
- sound_model:ced-tiny-q8_0.gguf
```
The realtime pipeline that uses it:
```yaml
name: gpt-realtime
pipeline:
vad: silero-vad-ggml
transcription: parakeet-realtime-scene
llm: qwen3-4b
tts: tts-1
turn_detection:
type: semantic_vad
```
Each closed speaker segment emits a `conversation.item.input_audio_transcription.segment` event under the turn's item id, with an empty `text` (the event exists to carry the speaker boundary, not a transcript - the transcript still comes from the ordinary delta/completed events):
```json
{
"type": "conversation.item.input_audio_transcription.segment",
"item_id": "item_abc",
"content_index": 0,
"speaker": "0",
"start": 1.92,
"end": 4.10,
"text": ""
}
```
Each sound event emits a `conversation.item.sound_detection` event with one tag and the detection window's `start`/`end`:
```json
{
"type": "conversation.item.sound_detection",
"item_id": "item_abc",
"content_index": 0,
"detections": [{"label": "Chicken, rooster", "score": 0.91, "index": 99}],
"start": 24.0,
"end": 30.0
}
```
The `start`/`end` on both event types are seconds measured from the start of the current turn's own audio, not the session or the WebSocket connection - the same base the streamed transcript words use.
The companion stream is opened fresh for each speech turn, alongside that turn's ASR live session, and closed when the turn commits: whatever it had not yet emitted is drained and sent at that point. Because the diarization model runs a brand new session every turn, its speaker indices are scoped to the turn too - `"speaker": "0"` in one turn and `"speaker": "0"` in the next are not guaranteed to be the same person, even within the same conversation.
`score` is the peak score seen for that tag while the sound was live, not an average.
**Limitation**: under `semantic_vad`, live transcription (and so this companion stream) only runs during speech turns - it does not see audio between turns. A sound that happens while nobody is speaking is not detected this way. If you need sound events independent of speech turns, use the pipeline's `sound_detection` model instead (see [Sound Classification]({{% relref "audio-classification" %}})), which classifies each VAD-committed utterance on its own. Use one or the other, not both, on the same session - they overlap in purpose and would emit sound detections twice.
### Speaker and sound events with an offline model (Parakeet TDT v3)
The live events above need a cache-aware streaming transcription model. An offline model such as Parakeet TDT 0.6B v3 (25 languages) runs under `server_vad` instead: each VAD-committed turn is transcribed as a whole. The gallery model `parakeet-cpp-realtime-scene-tdt` bundles it with Nemotron-3-Diarization and CED-Tiny, so one parakeet-cpp backend handles transcription, speakers and sounds. Point both `transcription` and `sound_detection` at it and turn on `diarization`:
```yaml
name: gpt-realtime-scene
pipeline:
vad: silero-vad-ggml
transcription: parakeet-cpp-realtime-scene-tdt
sound_detection: parakeet-cpp-realtime-scene-tdt
diarization: true
llm: qwen3-4b
tts: tts-1
```
`pipeline.diarization` asks the transcription model for speaker labels on each committed turn and emits every labelled segment as a `conversation.item.input_audio_transcription.segment` event before the turn's `completed` event. Unlike the live path, these segments carry their `text`:
```json
{
"type": "conversation.item.input_audio_transcription.segment",
"item_id": "item_abc",
"content_index": 0,
"id": "seg_1",
"speaker": "1",
"start": 6.85,
"end": 10.82,
"text": "Well, I don't wish to see it any more, observed Phoebe, turning away her eyes."
}
```
`sound_detection` classifies the same committed audio and emits one `conversation.item.sound_detection` event per turn (see [Sound Classification]({{% relref "audio-classification" %}})). As on the live path, times are relative to the turn's audio and speaker labels are only consistent within a turn. `pipeline.diarization` is off by default: it needs a transcription model that diarizes (parakeet-cpp with a `diarization_model` companion), and some other backends fail a diarization request they cannot serve.
#### Choosing the sound model
Both scene models ship with CED-Tiny, the cheapest to run all the time. `parakeet-cpp-realtime-scene-base` and `parakeet-cpp-realtime-scene-tdt-base` are the same pipelines with CED-Base (86M, the largest CED), which tags sounds more confidently. Any CED GGUF from [`mudler/ced-gguf`](https://huggingface.co/mudler/ced-gguf) (tiny, mini, small, base) works as `sound_model`. Measured on CPU (Ryzen 9 9950X3D) over a 37 s clip with two speakers and a rooster, as a fraction of real time:
| | CED-Tiny | CED-Base |
|---|---|---|
| Live scene stream (diarization `low` + sound), EOU path | 0.103 | 0.125 |
| Sound detection per committed turn, TDT path | 0.005 | 0.031 |
The EOU model's own ASR stream adds 0.016. Diarization dominates the live cost, so CED-Base keeps the live path about 7x faster than real time.
### Disabling thinking
For reasoning models, you can force the pipeline LLM's thinking off without editing the LLM model config:
+1
View File
@@ -1094,6 +1094,7 @@ engine_args:
| `tokenizer_config` | Override the `tokenizer_config.json` the chat template is read from | `<model_dir>/tokenizer_config.json` |
| `speculative_config` | Speculative decoding (see below) | disabled |
| `kv_transfer_config` | External KV connector / LMCache (see below) | none |
| `hf_overrides` | JSON object of `config.json` keys merged over the model directory's own, as vLLM's `--hf-overrides` (see the [vllm.cpp backend page]({{% relref "features/vllm-cpp" %}}#overriding-configjson-keys-hf_overrides)) | none |
Raising `max_num_batched_tokens` lets more prefill land in a single step, at the
cost of decode latency for requests queued behind it. The default deliberately
+46 -10
View File
@@ -133,6 +133,39 @@ engine_args:
tool_parser: qwen3_coder
```
## Overriding config.json keys (`hf_overrides`)
`engine_args.hf_overrides` is a JSON object of top-level `config.json` keys that
the backend merges over the model directory's own `config.json` before the
engine loads it, like vLLM's `--hf-overrides`. The main use is to opt a published
checkpoint into an engine adapter that its config does not name. For example,
the Tev1 repositories declare `Qwen3_5ForConditionalGeneration`, and vllm.cpp
serves them as decision models only when the architecture is `Tev1Model`:
```yaml
engine_args:
hf_overrides:
architectures: ["Tev1Model"]
```
The downloaded model files do not change. At load the backend creates a private
temporary directory. It writes the merged `config.json` there and adds a
symlink for each other entry of the model directory (weights, tokenizer files,
a `tokenizer/` subdirectory). Then it gives that directory to the engine. When
the model unloads, the backend removes the directory.
Rules:
- The merge is top-level only. An override key replaces the whole value of that
key, including a nested object such as `text_config`.
- The model must be a directory that contains a `config.json`. A `.gguf` file
or a directory without `config.json` fails the load.
- A value that is not a JSON object (an array, a scalar, or JSON that does not
parse) fails the load. The backend does not ignore it, because loading the
unchanged config would serve a different architecture than the one you
configured.
- An empty object (`{}`) does nothing.
## Named entity recognition (GLiNER2.5)
The `vllm-cpp` backend serves [GLiNER2.5](https://huggingface.co/fastino/gliner2.5-multi-v1),
@@ -160,22 +193,25 @@ forward, which is the required contract for pooling models in vllm.cpp. A
device-resident forward is tracked as a performance optimization, not a
correctness gap.
### SystemOne structured-extraction API
### Decisions API
The `vllm-cpp` backend also exposes kev-compatible SystemOne endpoints that
turn zero-shot NER into structured question answering. These mirror the API
from the [kev](https://github.com/jaredpalmer/kev) project:
The `vllm-cpp` backend serves the kev-compatible SystemOne endpoints (the Decisions API): typed
`choice`, `noul` and `score` questions over a state text, answered by a
non-generative decision model in one pass. A decision model declares
`known_usecases: [decisions]`. See [Decisions API]({{% relref "features/decisions" %}})
for the request shape, the models you can install and the access rules.
| Endpoint | Method | Description |
|---|---|---|
| `/v1/systemone` | POST | Answer all questions in one NER pass |
| `/v1/systemone` | POST | Answer all questions in one pass |
| `/v1/systemone/permute` | POST | Re-run one choice question under n_perm option orders |
| `/v1/systemone/separate` | POST | Answer each question in its own NER pass (N passes) |
| `/v1/systemone/separate` | POST | Answer each question in its own pass (N passes) |
Each question has a `type` of `noul` (binary entity presence), `choice` (pick
one option), or `score` (pick one level). The `model` field in the request body
selects the NER model. Labels are derived from the question definition, so no
`ner_labels` configuration is needed for these endpoints.
The GLiNER2.5 zero-shot NER model (`token_classify`) also serves
`/v1/systemone`, through the NER path, and it is the model to use for
`/v1/systemone/permute` and `/v1/systemone/separate`, which decision models
refuse with a `400`. It derives its NER labels from the question definitions, so
no `ner_labels` configuration is needed.
## Beyond text generation
+672 -22
View File
@@ -297,7 +297,7 @@
files:
- filename: ds4flash.gguf
uri: https://huggingface.co/unsloth/DeepSeek-V4-Flash-Vision-Exp-GGUF
sha256: f33633d55f5379e8571db06674bf7a07a2ea7bb7b44287e1a9bdc3686d69a5c6
sha256: bfca14d287c9fd865529e02efa0ba6f572fb7bef623fa0bf4b0fb6214ee63a5e
- name: "qwopus3.8-27b-flash-v2"
variants:
- model: qwopus3.8-27b-flash-v2-q8
@@ -19360,6 +19360,7 @@
- qwen3.6
- nvfp4
- vllm-cpp
- vision
- tool-calling
- reasoning
- gpu
@@ -19372,6 +19373,7 @@
known_usecases:
- chat
- completion
- vision
# Tool calls and the <think> split are parsed by the engine's own streaming
# parsers, so LocalAI's Go-side grammar path stays out of the way.
function:
@@ -19429,6 +19431,7 @@
- qwen3.6
- nvfp4
- vllm-cpp
- vision
- speculative-decoding
- mtp
- tool-calling
@@ -19442,6 +19445,7 @@
known_usecases:
- chat
- completion
- vision
function:
grammar:
disable: true
@@ -19497,6 +19501,7 @@
- qwen3.6
- nvfp4
- vllm-cpp
- vision
- speculative-decoding
- dflash
- tool-calling
@@ -19510,6 +19515,7 @@
known_usecases:
- chat
- completion
- vision
function:
grammar:
disable: true
@@ -19557,6 +19563,9 @@
with roughly 3B parameters active per token, so it reads like a much larger
model while costing about as much per token as a small one.
Image input is implemented in the engine but is not token-gated against
vLLM yet, so the vision usecase on this entry is experimental.
This is the engine's gated MoE checkpoint: token-for-token identical to vLLM
over the 315-prompt battery on both the synchronous and asynchronous paths,
at 0.92x to 0.97x vLLM's throughput from concurrency 1 to 32.
@@ -19575,6 +19584,8 @@
- moe
- nvfp4
- vllm-cpp
- vision
- experimental
- tool-calling
- reasoning
- gpu
@@ -19587,6 +19598,7 @@
known_usecases:
- chat
- completion
- vision
function:
grammar:
disable: true
@@ -19615,6 +19627,9 @@
description: |
Qwen3.6-35B-A3B NVFP4 on vllm.cpp with MTP speculative decoding enabled.
Image input is implemented in the engine but is not token-gated against
vLLM yet, so the vision usecase on this entry is experimental.
The draft head ships inside the checkpoint's own mtp.* tensors, so there is
no second model to download. On this model the speculative path is
token-exact against speculation-off on both the synchronous and asynchronous
@@ -19632,6 +19647,8 @@
- moe
- nvfp4
- vllm-cpp
- vision
- experimental
- speculative-decoding
- mtp
- tool-calling
@@ -19645,6 +19662,7 @@
known_usecases:
- chat
- completion
- vision
function:
grammar:
disable: true
@@ -53670,6 +53688,389 @@
- filename: parakeet-cpp/nemotron-3.5-asr-streaming-0.6b-q8_0.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3.5-asr-streaming-0.6b-q8_0.gguf
sha256: ba2f13eccd4a5245be728f77e6149bd6a4fdcdd133ff2e08ac6005bcef7a99f1
- name: parakeet-cpp-nemotron-3-diarization
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/mudler/parakeet-cpp-gguf
- https://huggingface.co/nvidia/Nemotron-3-Diarization
- https://github.com/mudler/parakeet.cpp
description: |
Nemotron-3-Diarization (Sortformer), Q8_0 GGUF for the parakeet-cpp backend
(C++/ggml port of NVIDIA NeMo). Speaker diarization only: served through
/v1/audio/diarization, returns per-segment start, end and speaker label
("0", "1", ...). It does not transcribe; pair it with an ASR model and set
asr_model to get speaker-attributed text from the same call. num_speakers,
min_speakers, max_speakers and clustering_threshold are not supported by
Sortformer and are ignored.
license: openmdw-1.1
tags:
- parakeet
- parakeet-cpp
- nemotron
- sortformer
- diarization
- speaker-diarization
- gguf
- ggml
- quantized
overrides:
backend: parakeet-cpp
known_usecases:
- diarization
name: parakeet-cpp-nemotron-3-diarization
parameters:
model: parakeet-cpp/nemotron-3-diarization-q8_0.gguf
files:
- filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf
sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22
- name: parakeet-cpp-nemotron-3-diarization-asr
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/mudler/parakeet-cpp-gguf
- https://huggingface.co/nvidia/Nemotron-3-Diarization
- https://huggingface.co/nvidia/parakeet-tdt_ctc-110m
- https://github.com/mudler/parakeet.cpp
description: |
Nemotron-3-Diarization (Sortformer) paired with the Parakeet TDT+CTC 110M
ASR model through the asr_model option, both Q8_0/F16 GGUF for the
parakeet-cpp backend (C++/ggml port of NVIDIA NeMo). Served through
/v1/audio/diarization with include_text: each speaker segment comes back
with its transcribed text in one call. Diarization model is
OpenMDW-1.1, ASR model is CC-BY-4.0.
license: openmdw-1.1
tags:
- parakeet
- parakeet-cpp
- nemotron
- sortformer
- asr
- diarization
- speaker-diarization
- speech-recognition
- stt
- gguf
- ggml
- quantized
overrides:
backend: parakeet-cpp
known_usecases:
- diarization
name: parakeet-cpp-nemotron-3-diarization-asr
options:
- asr_model:parakeet-cpp/tdt_ctc-110m-f16.gguf
parameters:
model: parakeet-cpp/nemotron-3-diarization-q8_0.gguf
files:
- filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf
sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22
- filename: parakeet-cpp/tdt_ctc-110m-f16.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/tdt_ctc-110m-f16.gguf
sha256: 7f9a6376edde6a74592ace48b2ebdc27a1ac972d0be9dfcc29e668d99381faf1
- name: parakeet-cpp-ced-tiny
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/mudler/ced-gguf
- https://huggingface.co/mispeech/ced-tiny
- https://github.com/mudler/parakeet.cpp
description: |
CED-Tiny sound event tagger, Q8_0 GGUF for the parakeet-cpp backend
(C++/ggml, loaded through third_party/ced.cpp). Served through
/v1/audio/classification: 10 s windows are scored and averaged over the
clip, then sorted by score with threshold and top_k applied. Smallest and
fastest of the CED sizes; use ced-base for higher accuracy.
license: apache-2.0
tags:
- parakeet-cpp
- ced
- sound-classification
- audio-tagging
- gguf
- ggml
- quantized
overrides:
backend: parakeet-cpp
known_usecases:
- sound_classification
name: parakeet-cpp-ced-tiny
parameters:
model: parakeet-cpp/ced-tiny-q8_0.gguf
files:
- filename: parakeet-cpp/ced-tiny-q8_0.gguf
uri: huggingface://mudler/ced-gguf/ced-tiny-q8_0.gguf
sha256: 48bee4e2fc3cc85d7806e03471db24e77fda6c2a2e81ffe9ef67caebaf2bd674
- name: parakeet-cpp-ced-base
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/mudler/ced-gguf
- https://huggingface.co/mispeech/ced-base
- https://github.com/mudler/parakeet.cpp
description: |
CED-Base sound event tagger, Q8_0 GGUF for the parakeet-cpp backend
(C++/ggml, loaded through third_party/ced.cpp). Served through
/v1/audio/classification: 10 s windows are scored and averaged over the
clip, then sorted by score with threshold and top_k applied. Larger and
more accurate than ced-tiny, still CPU-friendly.
license: apache-2.0
tags:
- parakeet-cpp
- ced
- sound-classification
- audio-tagging
- gguf
- ggml
- quantized
overrides:
backend: parakeet-cpp
known_usecases:
- sound_classification
name: parakeet-cpp-ced-base
parameters:
model: parakeet-cpp/ced-base-q8_0.gguf
files:
- filename: parakeet-cpp/ced-base-q8_0.gguf
uri: huggingface://mudler/ced-gguf/ced-base-q8_0.gguf
sha256: bd34a7710169f0047fea17267965d211f967828ab25ba6fb9d3768481393f6e2
- name: parakeet-cpp-realtime-scene
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/mudler/parakeet-cpp-gguf
- https://huggingface.co/mudler/ced-gguf
- https://huggingface.co/nvidia/parakeet_realtime_eou_120m-v1
- https://huggingface.co/nvidia/Nemotron-3-Diarization
- https://huggingface.co/mispeech/ced-tiny
- https://github.com/mudler/parakeet.cpp
description: |
Cache-aware streaming RNNT FastConformer with end-of-utterance (EOU)
detection, 120M, paired with Nemotron-3-Diarization and CED-Tiny through
the diarization_model and sound_model options. F16/Q8_0 GGUF for the
parakeet-cpp backend (C++/ggml port of NVIDIA NeMo). Use with streaming
transcription: while a turn is live, closed speaker segments and sound
events are surfaced alongside the ASR text (realtime
conversation.item.input_audio_transcription.segment and
conversation.item.sound_detection events). Live speaker/sound events only
fire during speech turns under semantic_vad; sounds between turns are not
seen by this path. License per model: transcription model NVIDIA Open
Model License, diarization model OpenMDW-1.1, CED-Tiny Apache-2.0.
license: nvidia-open-model-license
tags:
- parakeet
- parakeet-cpp
- nemotron
- sortformer
- ced
- asr
- speech-recognition
- diarization
- sound-classification
- streaming
- realtime
- stt
- gguf
- ggml
overrides:
backend: parakeet-cpp
known_usecases:
- transcript
name: parakeet-cpp-realtime-scene
options:
- diarization_model:parakeet-cpp/nemotron-3-diarization-q8_0.gguf
- sound_model:parakeet-cpp/ced-tiny-q8_0.gguf
parameters:
model: parakeet-cpp/realtime_eou_120m-v1-f16.gguf
files:
- filename: parakeet-cpp/realtime_eou_120m-v1-f16.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/realtime_eou_120m-v1-f16.gguf
sha256: d1a2b12f12b8a096a57499c9111ed13b442a2b786e17a292c168be45088f0edc
- filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf
sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22
- filename: parakeet-cpp/ced-tiny-q8_0.gguf
uri: huggingface://mudler/ced-gguf/ced-tiny-q8_0.gguf
sha256: 48bee4e2fc3cc85d7806e03471db24e77fda6c2a2e81ffe9ef67caebaf2bd674
- name: parakeet-cpp-realtime-scene-tdt
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/mudler/parakeet-cpp-gguf
- https://huggingface.co/mudler/ced-gguf
- https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3
- https://huggingface.co/nvidia/Nemotron-3-Diarization
- https://huggingface.co/mispeech/ced-tiny
- https://github.com/mudler/parakeet.cpp
description: |
Parakeet TDT 0.6B v3 (multilingual, 25 European languages) paired with
Nemotron-3-Diarization and CED-Tiny through the diarization_model and
sound_model options: one parakeet-cpp backend transcribes, labels speakers
and tags sound events. GGUF for the parakeet-cpp backend (C++/ggml port of
NVIDIA NeMo). TDT is not a streaming model, so in a realtime pipeline use
it with server_vad: set it as both transcription and sound_detection and
turn on pipeline.diarization, and each committed turn gets speaker segments
(conversation.item.input_audio_transcription.segment, with text) and
sound tags (conversation.item.sound_detection). Also labels speakers on
/v1/audio/transcriptions. Speaker labels are per turn. License per model:
transcription model CC-BY-4.0, diarization model OpenMDW-1.1, CED-Tiny
Apache-2.0.
license: cc-by-4.0
tags:
- parakeet
- parakeet-cpp
- nemotron
- sortformer
- ced
- asr
- speech-recognition
- diarization
- sound-classification
- multilingual
- realtime
- stt
- gguf
- ggml
overrides:
backend: parakeet-cpp
known_usecases:
- transcript
- diarization
- sound_classification
name: parakeet-cpp-realtime-scene-tdt
options:
- diarization_model:parakeet-cpp/nemotron-3-diarization-q8_0.gguf
- sound_model:parakeet-cpp/ced-tiny-q8_0.gguf
parameters:
model: parakeet-cpp/tdt-0.6b-v3-f16.gguf
files:
- filename: parakeet-cpp/tdt-0.6b-v3-f16.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/tdt-0.6b-v3-f16.gguf
sha256: 8ba47343e1e919895aca90e099150a01ed203ee0942d8ed31e27295efc5abb22
- filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf
sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22
- filename: parakeet-cpp/ced-tiny-q8_0.gguf
uri: huggingface://mudler/ced-gguf/ced-tiny-q8_0.gguf
sha256: 48bee4e2fc3cc85d7806e03471db24e77fda6c2a2e81ffe9ef67caebaf2bd674
- name: parakeet-cpp-realtime-scene-base
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/mudler/parakeet-cpp-gguf
- https://huggingface.co/mudler/ced-gguf
- https://huggingface.co/nvidia/parakeet_realtime_eou_120m-v1
- https://huggingface.co/nvidia/Nemotron-3-Diarization
- https://huggingface.co/mispeech/ced-base
- https://github.com/mudler/parakeet.cpp
description: |
Cache-aware streaming RNNT FastConformer with end-of-utterance (EOU)
detection, 120M, paired with Nemotron-3-Diarization and CED-Base (86M, the largest CED;
more confident sound tags than CED-Tiny at a small extra cost: on CPU the
live diarization + sound stream runs at 0.125 of real time against 0.103
with CED-Tiny) through
the diarization_model and sound_model options. F16/Q8_0 GGUF for the
parakeet-cpp backend (C++/ggml port of NVIDIA NeMo). Use with streaming
transcription: while a turn is live, closed speaker segments and sound
events are surfaced alongside the ASR text (realtime
conversation.item.input_audio_transcription.segment and
conversation.item.sound_detection events). Live speaker/sound events only
fire during speech turns under semantic_vad; sounds between turns are not
seen by this path. License per model: transcription model NVIDIA Open
Model License, diarization model OpenMDW-1.1, CED-Base Apache-2.0.
license: nvidia-open-model-license
tags:
- parakeet
- parakeet-cpp
- nemotron
- sortformer
- ced
- asr
- speech-recognition
- diarization
- sound-classification
- streaming
- realtime
- stt
- gguf
- ggml
overrides:
backend: parakeet-cpp
known_usecases:
- transcript
name: parakeet-cpp-realtime-scene-base
options:
- diarization_model:parakeet-cpp/nemotron-3-diarization-q8_0.gguf
- sound_model:parakeet-cpp/ced-base-q8_0.gguf
parameters:
model: parakeet-cpp/realtime_eou_120m-v1-f16.gguf
files:
- filename: parakeet-cpp/realtime_eou_120m-v1-f16.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/realtime_eou_120m-v1-f16.gguf
sha256: d1a2b12f12b8a096a57499c9111ed13b442a2b786e17a292c168be45088f0edc
- filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf
sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22
- filename: parakeet-cpp/ced-base-q8_0.gguf
uri: huggingface://mudler/ced-gguf/ced-base-q8_0.gguf
sha256: bd34a7710169f0047fea17267965d211f967828ab25ba6fb9d3768481393f6e2
- name: parakeet-cpp-realtime-scene-tdt-base
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/mudler/parakeet-cpp-gguf
- https://huggingface.co/mudler/ced-gguf
- https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3
- https://huggingface.co/nvidia/Nemotron-3-Diarization
- https://huggingface.co/mispeech/ced-base
- https://github.com/mudler/parakeet.cpp
description: |
Parakeet TDT 0.6B v3 (multilingual, 25 European languages) paired with
Nemotron-3-Diarization and CED-Base (86M, the largest CED; about 0.03 s of
CPU per second of audio per committed turn, against 0.005 for CED-Tiny)
through the diarization_model and
sound_model options: one parakeet-cpp backend transcribes, labels speakers
and tags sound events. GGUF for the parakeet-cpp backend (C++/ggml port of
NVIDIA NeMo). TDT is not a streaming model, so in a realtime pipeline use
it with server_vad: set it as both transcription and sound_detection and
turn on pipeline.diarization, and each committed turn gets speaker segments
(conversation.item.input_audio_transcription.segment, with text) and
sound tags (conversation.item.sound_detection). Also labels speakers on
/v1/audio/transcriptions. Speaker labels are per turn. License per model:
transcription model CC-BY-4.0, diarization model OpenMDW-1.1, CED-Tiny
Apache-2.0.
license: cc-by-4.0
tags:
- parakeet
- parakeet-cpp
- nemotron
- sortformer
- ced
- asr
- speech-recognition
- diarization
- sound-classification
- multilingual
- realtime
- stt
- gguf
- ggml
overrides:
backend: parakeet-cpp
known_usecases:
- transcript
- diarization
- sound_classification
name: parakeet-cpp-realtime-scene-tdt-base
options:
- diarization_model:parakeet-cpp/nemotron-3-diarization-q8_0.gguf
- sound_model:parakeet-cpp/ced-base-q8_0.gguf
parameters:
model: parakeet-cpp/tdt-0.6b-v3-f16.gguf
files:
- filename: parakeet-cpp/tdt-0.6b-v3-f16.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/tdt-0.6b-v3-f16.gguf
sha256: 8ba47343e1e919895aca90e099150a01ed203ee0942d8ed31e27295efc5abb22
- filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf
sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22
- filename: parakeet-cpp/ced-base-q8_0.gguf
uri: huggingface://mudler/ced-gguf/ced-base-q8_0.gguf
sha256: bd34a7710169f0047fea17267965d211f967828ab25ba6fb9d3768481393f6e2
- name: moss-transcribe-cpp-0.9b
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
@@ -54137,7 +54538,7 @@
files:
- filename: cohere-transcribe-q4_k.gguf
uri: huggingface://cstr/cohere-transcribe-03-2026-GGUF/cohere-transcribe-q4_k.gguf
sha256: 237261c543dc9124a3f08f95b48c9c672896ef0d79dc8cadce3fb4ddc09a2ef8
sha256: 116f4c4f7ff1b03997100d3a097e13fecd84350555b9979867b241cd6803e4f5
- name: wav2vec2-crispasr
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
@@ -63252,7 +63653,7 @@
512-token context. F16 weights, ~804 MB.
license: apache-2.0
tags:
- decision
- decisions
- systemone
- vllm-cpp
- cpu
@@ -63262,15 +63663,264 @@
overrides:
backend: vllm-cpp
known_usecases:
- chat
- decisions
parameters:
model: convaiinnovations/laya
artifacts:
- name: model
target: model
source:
type: huggingface
repo: convaiinnovations/laya
artifacts:
- name: model
target: model
source:
type: huggingface
repo: convaiinnovations/laya
- name: gliner25-decide-vllm-cpp
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/fastino/GLiNER2.5-Decide
- https://github.com/mudler/vllm.cpp
description: |
GLiNER2.5-Decide is a DeBERTa-v3-large encoder with a classification head
that answers typed decision questions over a state text in one forward
pass. It never generates text, so there is nothing to parse.
In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine runs the
decision pipeline (choice, noul and score question types) through the
vllm_decide C ABI. This is the decision model, not the zero-shot NER model:
use the gliner2.5 entry for entity extraction. F32 weights, about 2 GB.
The weights are pinned to a revision so the entry keeps serving the
checkpoint it was checked against.
license: apache-2.0
tags:
- decisions
- systemone
- vllm-cpp
- cpu
- gpu
size: 2GB
last_checked: "2026-09-30"
overrides:
backend: vllm-cpp
known_usecases:
- decisions
parameters:
model: fastino/GLiNER2.5-Decide
artifacts:
- name: model
target: model
source:
type: huggingface
repo: fastino/GLiNER2.5-Decide
revision: 5a7adf72a23b4d311abae6ce050d7f0012bb3416
- name: tev1-4b-vllm-cpp
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/togethercomputer/Tev1-4B-experimental
- https://github.com/mudler/vllm.cpp
description: |
Tev1-4B-experimental is an experimental decision model from Together
AI: a supervised fine-tune of Qwen3.5-4B that picks one option letter
for a state, a question and 2 to 24 labeled options. It keeps the standard
next-token head, so it is autoregressive, unlike Laya or GLiNER2.5-Decide.
In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine scores the
answer letters of each choice, noul and score question through the
vllm_decide C ABI and returns probabilities with an entropy confidence, as
Ollama does for tev1. A choice or score question accepts at most 24 options
(Ollama allows 26) and every option needs a nonempty description. The
published config.json names Qwen3_5ForConditionalGeneration, so this entry
sets hf_overrides to load it as Tev1Model without editing the download.
Checked against transformers BF16 on CPU over seven
questions: 7/7 answers equal, largest probability difference 0.0004. The decision route is verified on CPU only; GPU serving has not
been measured. BF16 weights, about 9.3GB, pinned to a revision. The
fine-tune license is still being finalized by Together AI (base model
Apache-2.0).
tags:
- decisions
- systemone
- vllm-cpp
- cpu
- gpu
size: 9.3GB
last_checked: "2026-09-30"
overrides:
backend: vllm-cpp
known_usecases:
- decisions
template:
use_tokenizer_template: true
context_size: 2048
engine_args:
hf_overrides:
architectures:
- Tev1Model
block_size: 32
num_blocks: 256
max_num_seqs: 4
parameters:
model: togethercomputer/Tev1-4B-experimental
artifacts:
- name: model
target: model
source:
type: huggingface
repo: togethercomputer/Tev1-4B-experimental
revision: 0b7becf017daa0e5eb222f8ce7483c8c8259c52f
- name: tev1-0.8b-vllm-cpp
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/togethercomputer/Tev1-0.8B-experimental
- https://github.com/mudler/vllm.cpp
description: |
Tev1-0.8B-experimental is an experimental decision model from Together
AI: a supervised fine-tune of Qwen3.5-0.8B that picks one option letter
for a state, a question and 2 to 24 labeled options. It keeps the standard
next-token head, so it is autoregressive, unlike Laya or GLiNER2.5-Decide.
In LocalAI, serve it via POST /v1/systemone. The vllm.cpp engine scores the
answer letters of each choice, noul and score question through the
vllm_decide C ABI and returns probabilities with an entropy confidence, as
Ollama does for tev1. A choice or score question accepts at most 24 options
(Ollama allows 26) and every option needs a nonempty description. The
published config.json names Qwen3_5ForConditionalGeneration, so this entry
sets hf_overrides to load it as Tev1Model without editing the download.
Checked against transformers BF16 on CPU over seven
questions: 6/7 answers equal, the miss a near tie (0.453 against 0.514
in transformers, 0.4845 each here), largest probability difference 0.031. The decision route is verified on CPU only; GPU serving has not
been measured. BF16 weights, about 1.8GB, pinned to a revision. The
fine-tune license is still being finalized by Together AI (base model
Apache-2.0).
tags:
- decisions
- systemone
- vllm-cpp
- cpu
- gpu
size: 1.8GB
last_checked: "2026-09-30"
overrides:
backend: vllm-cpp
known_usecases:
- decisions
template:
use_tokenizer_template: true
context_size: 2048
engine_args:
hf_overrides:
architectures:
- Tev1Model
block_size: 32
num_blocks: 256
max_num_seqs: 4
parameters:
model: togethercomputer/Tev1-0.8B-experimental
artifacts:
- name: model
target: model
source:
type: huggingface
repo: togethercomputer/Tev1-0.8B-experimental
revision: 6bb2dff14b38fea90ddb14d870166ccaf77374e9
- name: kev-0.8b-vllm-cpp
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/mudler/kev-0.8b-vllm-cpp
- https://huggingface.co/jaredpalmer/kev-0.8b
- https://github.com/mudler/vllm.cpp
description: |
kev is a System 1 decision model by Jared Palmer. It answers typed choice,
noul and score questions about a text state with one scoring pass per
question. It does not generate text. The model is a frozen
Qwen3.5-0.8B-Base backbone, a rank-16 LoRA adapter and a PointerHead
readout.
This entry installs a converted redistribution of jaredpalmer/kev-0.8b:
the LoRA is merged into the BF16 backbone, the head is stored as
head.safetensors, and config.json names the KevModel architecture. The
checkpoint only works with vllm.cpp (the vllm-cpp backend); transformers,
vLLM and llama.cpp cannot load it.
In LocalAI, serve it via POST /v1/systemone. The vllm.cpp project records
PointerHead golden-vector tests (25 cases) and a 5-case end-to-end
comparison against the kev reference server as equal. The upload itself
was smoke-tested with one request on CPU; there is no accuracy benchmark
and no GPU run. The entry sets a 2048-token context and an explicit KV
pool, because the default 4096-token context does not fit the default
CPU KV pool and the load fails. BF16 weights, about 1.53 GB, pinned to a
revision.
license: apache-2.0
tags:
- decisions
- systemone
- vllm-cpp
- cpu
- gpu
size: 1.53GB
last_checked: "2026-09-30"
overrides:
backend: vllm-cpp
known_usecases:
- decisions
context_size: 2048
engine_args:
block_size: 32
num_blocks: 256
max_num_seqs: 4
parameters:
model: mudler/kev-0.8b-vllm-cpp
artifacts:
- name: model
target: model
source:
type: huggingface
repo: mudler/kev-0.8b-vllm-cpp
revision: c17e73666ded1e9d284470eae7e0de9a27294e77
- name: qwen3-vl-4b-vllm-cpp
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct
- https://github.com/mudler/vllm.cpp
description: |
Qwen3-VL-4B-Instruct on vllm.cpp, in bf16: a small vision-language model
that takes images alongside text. In the engine's correctness battery the
image path matches vLLM token for token, and video input is a near tie.
Roughly 9 GB of weights plus KV cache at the context configured here. It
runs where the flagship NVFP4 checkpoints cannot, including plain CPU.
license: apache-2.0
tags:
- llm
- vision
- multimodal
- qwen
- qwen3-vl
- vllm-cpp
- cpu
- gpu
size: 9GB
last_checked: "2026-09-30"
overrides:
backend: vllm-cpp
known_usecases:
- chat
- completion
- vision
template:
use_tokenizer_template: true
context_size: 8192
engine_args:
block_size: 32
num_blocks: 512
max_num_seqs: 4
parameters:
model: Qwen/Qwen3-VL-4B-Instruct
artifacts:
- name: model
target: model
source:
type: huggingface
repo: Qwen/Qwen3-VL-4B-Instruct
revision: ebb281ec70b05090aa6165b016eac8ec08e71b17
- name: cua-s1-forms-vllm-cpp
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
@@ -63299,12 +63949,12 @@
- score
parameters:
model: cua-ai/cua-s1-forms
artifacts:
- name: model
target: model
source:
type: huggingface
repo: cua-ai/cua-s1-forms
artifacts:
- name: model
target: model
source:
type: huggingface
repo: cua-ai/cua-s1-forms
- name: gliner2.5-vllm-cpp
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
@@ -63336,12 +63986,12 @@
- token_classify
parameters:
model: fastino/gliner2.5-multi-v1
artifacts:
- name: model
target: model
source:
type: huggingface
repo: fastino/gliner2.5-multi-v1
artifacts:
- name: model
target: model
source:
type: huggingface
repo: fastino/gliner2.5-multi-v1
- name: nemo-speech-cpp-sortformer-diarization-v2
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
+21 -2
View File
@@ -38,11 +38,11 @@ require (
github.com/mholt/archiver/v3 v3.5.1
github.com/microcosm-cc/bluemonday v1.0.27
github.com/modelcontextprotocol/go-sdk v1.5.0
github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6
github.com/mudler/cogito v0.11.1-0.20260928072733-b40513ef5d1a
github.com/mudler/edgevpn v0.34.0
github.com/mudler/go-processmanager v0.1.2-0.20260823202314-dfa0ed852db6
github.com/mudler/memory v0.0.0-20260406210934-424c1ecf2cf8
github.com/mudler/nib v0.6.0
github.com/mudler/nib v0.12.1
github.com/mudler/xlog v0.0.6
github.com/ollama/ollama v0.20.4
github.com/onsi/ginkgo/v2 v2.29.0
@@ -152,6 +152,25 @@ require (
github.com/mattn/go-sqlite3 v1.14.32 // indirect
github.com/moby/moby/api v1.54.2 // indirect
github.com/moby/moby/client v0.4.1 // indirect
github.com/msuozzo/bonsai v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-bash v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-c v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-dockerfile v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-go v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-gotemplate v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-groovy v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-java v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-javascript v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-kotlin v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-markdown v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-markdown-inline v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-python v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-ruby v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-rust v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-terraform v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-tsx v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-typescript v0.4.0 // indirect
github.com/msuozzo/bonsai/bonsai-yaml v0.4.0 // indirect
github.com/muesli/ansi v0.0.0-20230316100256-276c6243b2f6 // indirect
github.com/muesli/cancelreader v0.2.2 // indirect
github.com/oklog/ulid v1.3.1 // indirect
+42 -4
View File
@@ -994,10 +994,48 @@ github.com/mr-tron/base58 v1.3.0 h1:K6Y13R2h+dku0wOqKtecgRnBUBPrZzLZy5aIj8lCcJI=
github.com/mr-tron/base58 v1.3.0/go.mod h1:2BuubE67DCSWwVfx37JWNG8emOC0sHEU4/HpcYgCLX8=
github.com/mschoch/smat v0.2.0 h1:8imxQsjDm8yFEAVBe7azKmKSgzSkZXDuKkSq9374khM=
github.com/mschoch/smat v0.2.0/go.mod h1:kc9mz7DoBKqDyiRL7VZN8KvXQMWeTaVnttLRXOlotKw=
github.com/msuozzo/bonsai v0.4.0 h1:WVGqsSctbGcL5ehdCdnckwRxBC3omwsmOtzKH/2tuZw=
github.com/msuozzo/bonsai v0.4.0/go.mod h1:LDo0Dmp6qaeEoJAG83UuLDHtwHn6H1knSYKqoXMxm4M=
github.com/msuozzo/bonsai/bonsai-bash v0.4.0 h1:I8pjnVYSAaKew3SmtnDf8IKAH30jHvWfymKfq1zf6bc=
github.com/msuozzo/bonsai/bonsai-bash v0.4.0/go.mod h1:6Tflp8naB4zdsvPEEujA+6bOJ/WQ5PiITZwbM4dj1gg=
github.com/msuozzo/bonsai/bonsai-c v0.4.0 h1:8uY7V/ofhIpsWWY4SXkgDEP14g8+JUhsfp9dOmGSiK0=
github.com/msuozzo/bonsai/bonsai-c v0.4.0/go.mod h1:1KW4TR0hVjP2O7iPqxmyD5vFimAbNrG87tU1srIBG5k=
github.com/msuozzo/bonsai/bonsai-dockerfile v0.4.0 h1:NDR2AuG4pFkL2VLgYiCJPbCrjqegH+LPZX/uEa1Q4yQ=
github.com/msuozzo/bonsai/bonsai-dockerfile v0.4.0/go.mod h1:fZLkQxL5zQk9J96xG7Lw+PQe4mZpU+Bqld6ToOJU3J0=
github.com/msuozzo/bonsai/bonsai-go v0.4.0 h1:mDH7ExUuH9vFKLyZ63SUk4FuJ5y2Z4gEg8WcIOCzfiY=
github.com/msuozzo/bonsai/bonsai-go v0.4.0/go.mod h1:xdTJhN+7nGFzKZM8R2fEkHM/vM9DbWYhy44nLeR+oR8=
github.com/msuozzo/bonsai/bonsai-gotemplate v0.4.0 h1:wGnquoW7t9HGmMsvMZdAnvxmULSaDrr6JPUkWtzn2Ck=
github.com/msuozzo/bonsai/bonsai-gotemplate v0.4.0/go.mod h1:DDfS5ey/vTTA25MguPVaDYP+abbmE/JZt/XWzmVfYWY=
github.com/msuozzo/bonsai/bonsai-groovy v0.4.0 h1:Dhdlsbo1gCoE51ig2uXbW/DqndCpFZyDWQjdmaOrdgc=
github.com/msuozzo/bonsai/bonsai-groovy v0.4.0/go.mod h1:5fEI1HjFSX1NJDLIccYnh2my03dtVikepPF6mVL8xc0=
github.com/msuozzo/bonsai/bonsai-java v0.4.0 h1:MNr/ShaNv688jNQByKu3OvwFUXAIUUiqhPi253M3EP8=
github.com/msuozzo/bonsai/bonsai-java v0.4.0/go.mod h1:bgcXWciHZqVQ9GBPhDkk/svvja907uzXvV/QxKmKM+s=
github.com/msuozzo/bonsai/bonsai-javascript v0.4.0 h1:hJ0Fz140Os3ZpV7Kl5ko7bsI++f5qayOl2HlLrdhXUU=
github.com/msuozzo/bonsai/bonsai-javascript v0.4.0/go.mod h1:QTqVNr8cwvdHBuMU0VsX+zW22DQfFYV0l7wClVXIVvM=
github.com/msuozzo/bonsai/bonsai-kotlin v0.4.0 h1:tTEsbgGoj1razbqvmcCMsAh2Y5/+gp7hMALSj4w1rbw=
github.com/msuozzo/bonsai/bonsai-kotlin v0.4.0/go.mod h1:cdwHqA5ys0bS32bBUv9B8AMS32UAJ6HqKNYNFHApvWw=
github.com/msuozzo/bonsai/bonsai-markdown v0.4.0 h1:D0c5FwSVu9xrujaLCYsrzO50aJ4zLJzUZCtbu8rSSTc=
github.com/msuozzo/bonsai/bonsai-markdown v0.4.0/go.mod h1:+o73zcEV/4O+8wthJC5iFdVx0M1S8SkI7YAYs7AIJng=
github.com/msuozzo/bonsai/bonsai-markdown-inline v0.4.0 h1:0XZb2aRuMrYN1Z1Tcw2J7mOWexAOPzg4VsWrcwR233E=
github.com/msuozzo/bonsai/bonsai-markdown-inline v0.4.0/go.mod h1:hjhOJSQylIlHEXT0+ns7ocHl9+qw1LgO9mits9Bt/Fc=
github.com/msuozzo/bonsai/bonsai-python v0.4.0 h1:spQNWwF6vBL4o8vfC+zoY8iZJfnLue1izNk7FKs0oXo=
github.com/msuozzo/bonsai/bonsai-python v0.4.0/go.mod h1:yMx1oeH6SQUuru9IhIioteV7cmF7S/UQNp9bDdMSfjs=
github.com/msuozzo/bonsai/bonsai-ruby v0.4.0 h1:hZygD1ev7EjLE5eDhlPRi+M0XrNKvzjMYx8RTZ0uxVA=
github.com/msuozzo/bonsai/bonsai-ruby v0.4.0/go.mod h1:GzujOi6rnLFX10zU9ntLNcetpnMaLeeUm526/5+aRHQ=
github.com/msuozzo/bonsai/bonsai-rust v0.4.0 h1:EQ2Fu6DWoGbq0OCV30acchw2VmV1WIOSNMVOLHyj2LY=
github.com/msuozzo/bonsai/bonsai-rust v0.4.0/go.mod h1:Qme4vyNPcCVkaGbLRiUW1NHERiY+bwfsPHsVyr3kZBg=
github.com/msuozzo/bonsai/bonsai-terraform v0.4.0 h1:loIUha5CgR7TgtfkaToP8PIufPrt5P6O/ySs3s0QcMo=
github.com/msuozzo/bonsai/bonsai-terraform v0.4.0/go.mod h1:xqGKDODYTigesGhHpZx3ZMuMivQ7nGGzCUmIHA4GxVE=
github.com/msuozzo/bonsai/bonsai-tsx v0.4.0 h1:/dheQMtc7luNGkL8E6DILowN/SxWyJ6W1x4msLqZiQQ=
github.com/msuozzo/bonsai/bonsai-tsx v0.4.0/go.mod h1:YW/ExrMLztBgSvoY1KTKMMQapV0OqMBb4RF62kQ/GMA=
github.com/msuozzo/bonsai/bonsai-typescript v0.4.0 h1:BD6JHfCs3rGNUttL+tMnz/iav1Y1+DkiMcy871NsKwo=
github.com/msuozzo/bonsai/bonsai-typescript v0.4.0/go.mod h1:DElq3FKqbTkJXcat1o8m98ijL3wUf+jW+DNInzh5Mgo=
github.com/msuozzo/bonsai/bonsai-yaml v0.4.0 h1:PJfyfjMQrcWU5e3ib0xGsVuOafNuimf5fK0NhTk8olU=
github.com/msuozzo/bonsai/bonsai-yaml v0.4.0/go.mod h1:z8jc0tjXDSOQRKRG0G3uymvVxHdBbrVsWd149nPOyHE=
github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca h1:bHlzSuOc5cKvHGF21ZjFFUV7IlkS3wt9YkBsWtkcaC4=
github.com/mudler/LocalAGI v0.0.0-20260927202351-7e0947d7ebca/go.mod h1:nk6zt1s5ANgchJYTWGY1jfFPuITSn1gB5oHZ/uFeFDg=
github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6 h1:eYTR8od5HdaHlh9AKCkxkRoHs2/wmx24BF5qrUh2TRY=
github.com/mudler/cogito v0.11.1-0.20260721122412-6eece18a6bb6/go.mod h1:6sfja3lcu2nWRzEc0wwqGNu/eCG3EWgij+8s7xyUeQ4=
github.com/mudler/cogito v0.11.1-0.20260928072733-b40513ef5d1a h1:b3bZ15XGd3kH+gje6ggFAciT4uC83+24xO/qc3o4viQ=
github.com/mudler/cogito v0.11.1-0.20260928072733-b40513ef5d1a/go.mod h1:UxGNMBRakV0A2uVHv2JQIHjpnVPkEWFaS7SDhyynZqQ=
github.com/mudler/edgevpn v0.34.0 h1:qDrD/rCPFY/FdURbXudIZWihVKY4VOX3nMn3CcbeQEU=
github.com/mudler/edgevpn v0.34.0/go.mod h1:yki7uMi5LR9gSMrw8PdPieuxsrk8BLV2Ui7VBEmbbIA=
github.com/mudler/go-piper v0.0.0-20241023091659-2494246fd9fc h1:RxwneJl1VgvikiX28EkpdAyL4yQVnJMrbquKospjHyA=
@@ -1008,8 +1046,8 @@ github.com/mudler/localrecall v0.6.5 h1:Q0atTJFFAyumKZG5dbGSrvQ+wsuA88hywIOfHdxp
github.com/mudler/localrecall v0.6.5/go.mod h1:28k5n19raUrkuwXkacdNsBlj8yuSnGhpT16tu+2+4dU=
github.com/mudler/memory v0.0.0-20260406210934-424c1ecf2cf8 h1:Ry8RiWy8fZ6Ff4E7dPmjRsBrnHOnPeOOj2LhCgyjQu0=
github.com/mudler/memory v0.0.0-20260406210934-424c1ecf2cf8/go.mod h1:EA8Ashhd56o32qN7ouPKFSRUs/Z+LrRCF4v6R2Oarm8=
github.com/mudler/nib v0.6.0 h1:6l2bQJkgHT5+o6S0wmmk/ipTCbFzmiSp1pM/tGQv4Eo=
github.com/mudler/nib v0.6.0/go.mod h1:d+Ymgi7PxDLnGxpYAkugv+mwRDtR1CMcgyykKNipwWA=
github.com/mudler/nib v0.12.1 h1:7yKZeOWcIac62UoNb3J2R/oEtFYzebFwgap9HPjIJdI=
github.com/mudler/nib v0.12.1/go.mod h1:I6diFODU8ALfPwFIbvrwpLDfhyBu6wAzoseig9jJEQ4=
github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145 h1:z59tA3IDYPt71nzH1jpxeaA1LuDw8aZfpTQFNU43Zb8=
github.com/mudler/skillserver v0.0.7-0.20260520220837-a7317cbf9145/go.mod h1:z3yFhcL9bSykmmh6xgGu0hyoItd4CnxgtWMEWw8uFJU=
github.com/mudler/water v0.0.0-20250808092830-dd90dcf09025 h1:WFLP5FHInarYGXi6B/Ze204x7Xy6q/I4nCZnWEyPHK0=
Loaded 100 of 102 files, more files were not shown because too many files have changed in this diff. Show more