From 6b794651a4594fca2075a5c40fdad0445b815b11 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Wed, 7 Oct 2026 14:10:45 +0200 Subject: [PATCH] feat(diarization): return sound events with include_sounds (#12544) * feat(diarization): return sound events with include_sounds A client that wants text, speakers, voice prints and sound events had to make a diarization call and a separate sound call. Add an include_sounds request field to /v1/audio/diarization that adds a sounds array of closed events {start, end, label, confidence}, in seconds. The parakeet-cpp backend runs a tagger-only scene stream over the clip, the same stream and thresholds the live path uses, so a clip gives the same events offline and live. A model with no sound_model companion, or a backend that does not report sound events, fails with 501 and the stable code include_sounds_unsupported instead of an empty list. The proto carries sounds_included so an empty list still means "nothing heard". The localai-proxy backend forwards the field. Swagger, docs and the e2e mock backend are updated. Assisted-by: Claude:claude-sonnet-5-5 [protoc swag go] * feat(gallery): add parakeet-cpp-multilingual-diarization-speakers-sounds Same as parakeet-cpp-multilingual-diarization-speakers (TDT 0.6B v3, Nemotron-3-Diarization, WeSpeaker) plus a CED-Tiny sound_model, so one model name serves /v1/audio/diarization with include_text, include_speaker_profiles and include_sounds. It declares the sound_classification usecase like the realtime scene entries. Assisted-by: Claude:claude-sonnet-5-5 --------- Co-authored-by: Ettore Di Giacinto --- backend/backend.proto | 12 ++ backend/go/localai-proxy/audio.go | 18 +++ backend/go/localai-proxy/audio_test.go | 17 +++ backend/go/parakeet-cpp/diarize.go | 17 +++ backend/go/parakeet-cpp/diarize_sounds.go | 113 ++++++++++++++ .../go/parakeet-cpp/diarize_sounds_test.go | 141 ++++++++++++++++++ backend/go/parakeet-cpp/scene.go | 17 ++- core/backend/diarization.go | 26 ++++ core/backend/diarization_sounds_test.go | 40 +++++ .../endpoints/localai/api_instructions.go | 2 +- .../endpoints/localai/portable_http_test.go | 83 ++++++++++- core/http/endpoints/openai/diarization.go | 7 +- core/schema/diarization.go | 14 ++ core/schema/openai.go | 1 + docs/content/features/audio-classification.md | 2 +- docs/content/features/audio-diarization.md | 42 ++++++ docs/content/features/audio-to-text.md | 2 +- docs/content/features/openai-realtime.md | 2 +- docs/content/features/voice-recognition.md | 4 +- gallery/index.yaml | 66 ++++++++ pkg/grpc/grpcerrors/errors.go | 15 ++ pkg/grpc/grpcerrors/errors_test.go | 9 ++ swagger/docs.go | 35 ++++- swagger/swagger.json | 35 ++++- swagger/swagger.yaml | 33 +++- tests/e2e/mock-backend/main.go | 13 +- tests/e2e/mock_backend_test.go | 22 +++ 27 files changed, 771 insertions(+), 17 deletions(-) create mode 100644 backend/go/parakeet-cpp/diarize_sounds.go create mode 100644 backend/go/parakeet-cpp/diarize_sounds_test.go create mode 100644 core/backend/diarization_sounds_test.go diff --git a/backend/backend.proto b/backend/backend.proto index aaa9a63c8..d7b1e112d 100644 --- a/backend/backend.proto +++ b/backend/backend.proto @@ -831,6 +831,7 @@ message DiarizeRequest { // identify speakers themselves read this; others ignore it. repeated KnownVoice known_voices = 12; bool include_speaker_profiles = 13; // opt-in sensitive embeddings; unsupported backends must reject + bool include_sounds = 14; // ask for closed sound events; backends without a sound model must reject, never return an empty list } message DiarizeSegment { @@ -858,8 +859,19 @@ message KnownVoice { string encoder_weights = 6; } +// DiarizeSound is one closed sound event over the whole clip, merged the same +// way the live scene stream merges them. +message DiarizeSound { + float start = 1; // seconds + float end = 2; // seconds + string label = 3; // AudioSet label + float confidence = 4; // peak score of the event, 0..1 +} + message DiarizeResponse { string speaker_profiles_json = 5; // versioned speaker_profiles object only; absent by default + repeated DiarizeSound sounds = 6; // closed sound events; only filled when include_sounds was set + bool sounds_included = 7; // true when the backend ran sound detection for this request (an empty `sounds` is then a real "nothing heard") repeated DiarizeSegment segments = 1; int32 num_speakers = 2; // count of distinct speaker labels in `segments` float duration = 3; // total audio duration in seconds (0 if unknown) diff --git a/backend/go/localai-proxy/audio.go b/backend/go/localai-proxy/audio.go index 07e011209..b3fef1f59 100644 --- a/backend/go/localai-proxy/audio.go +++ b/backend/go/localai-proxy/audio.go @@ -293,6 +293,9 @@ func (p *LocalAIProxy) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, erro if req.GetIncludeText() { f.Set("include_text", "true") } + if req.GetIncludeSounds() { + f.Set("include_sounds", "true") + } var resp struct { Duration float64 `json:"duration"` @@ -306,6 +309,14 @@ func (p *LocalAIProxy) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, erro End float32 `json:"end"` Text string `json:"text"` } `json:"segments"` + // A pointer tells an absent field (the upstream did not report sound + // events) from an empty list (it ran and heard nothing). + Sounds *[]struct { + Start float32 `json:"start"` + End float32 `json:"end"` + Label string `json:"label"` + Confidence float32 `json:"confidence"` + } `json:"sounds"` } if err := p.postForm(context.Background(), "/v1/audio/diarization", multipartForm{fields: f, files: []formFile{{field: "file", path: req.GetDst()}}}, &resp); err != nil { @@ -324,8 +335,15 @@ func (p *LocalAIProxy) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, erro Id: s.ID, Start: s.Start, End: s.End, Speaker: speaker, Text: s.Text, }) } + var sounds []*pb.DiarizeSound + if resp.Sounds != nil { + for _, s := range *resp.Sounds { + sounds = append(sounds, &pb.DiarizeSound{Start: s.Start, End: s.End, Label: s.Label, Confidence: s.Confidence}) + } + } return pb.DiarizeResponse{ Segments: segments, NumSpeakers: resp.NumSpeakers, Duration: float32(resp.Duration), Language: resp.Language, + Sounds: sounds, SoundsIncluded: resp.Sounds != nil, }, nil } diff --git a/backend/go/localai-proxy/audio_test.go b/backend/go/localai-proxy/audio_test.go index 8f5013c7b..18f582cad 100644 --- a/backend/go/localai-proxy/audio_test.go +++ b/backend/go/localai-proxy/audio_test.go @@ -375,6 +375,23 @@ var _ = Describe("audio methods", func() { Expect(res.Segments[1].Speaker).To(Equal("SPEAKER_01")) Expect(res.Segments[1].Start).To(BeNumerically("==", 1.5)) Expect(res.Segments[1].Id).To(Equal(int32(1))) + Expect(res.SoundsIncluded).To(BeFalse(), "an upstream that sends no sounds field reports none") + }) + + It("forwards include_sounds and keeps the upstream's sound events", func() { + p := loadProxy(up, nil) + up.replyJSON("/v1/audio/diarization", map[string]any{ + "task": "diarize", "duration": 3.0, "num_speakers": 1, + "segments": []any{map[string]any{"id": 0, "speaker": "SPEAKER_00", "start": 0.0, "end": 3.0}}, + "sounds": []any{map[string]any{"start": 1.0, "end": 2.0, "label": "Dog", "confidence": 0.75}}, + }) + res, err := p.Diarize(&pb.DiarizeRequest{Dst: writeInput("talk.wav", "RIFF-talk"), IncludeSounds: true}) + Expect(err).NotTo(HaveOccurred()) + Expect(up.last().Fields).To(HaveKeyWithValue("include_sounds", "true")) + Expect(res.SoundsIncluded).To(BeTrue()) + Expect(res.Sounds).To(HaveLen(1)) + Expect(res.Sounds[0].Label).To(Equal("Dog")) + Expect(res.Sounds[0].Confidence).To(BeNumerically("==", 0.75)) }) }) diff --git a/backend/go/parakeet-cpp/diarize.go b/backend/go/parakeet-cpp/diarize.go index c70afcd99..d9a89df45 100644 --- a/backend/go/parakeet-cpp/diarize.go +++ b/backend/go/parakeet-cpp/diarize.go @@ -1,6 +1,7 @@ package main import ( + "context" "encoding/json" "fmt" "sort" @@ -123,6 +124,11 @@ func (p *ParakeetCpp) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, error return pb.DiarizeResponse{}, status.Error(codes.Unimplemented, "parakeet-cpp: speaker profiles require a loaded speaker encoder and profile-capable library") } } + if req.GetIncludeSounds() { + if err := p.checkSoundEventsAvailable(); err != nil { + return pb.DiarizeResponse{}, err + } + } if CppDiarizePCM == nil { return pb.DiarizeResponse{}, status.Error(codes.Unimplemented, "parakeet-cpp: loaded libparakeet.so has no diarization support (parakeet_capi_diarize_pcm missing)") @@ -190,7 +196,18 @@ func (p *ParakeetCpp) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, error segments = applyDurationFilters(segments, req.GetMinDurationOn(), req.GetMinDurationOff()) renumberDiarizeSegments(segments) + var sounds []*pb.DiarizeSound + if req.GetIncludeSounds() { + events, err := p.diarizeSoundEvents(context.Background(), pcm) + if err != nil { + return pb.DiarizeResponse{}, err + } + sounds = diarizeSoundsToProto(events) + } + return pb.DiarizeResponse{ + Sounds: sounds, + SoundsIncluded: req.GetIncludeSounds(), SpeakerProfilesJson: profiles, Segments: segments, NumSpeakers: distinctDiarizeSpeakers(segments), diff --git a/backend/go/parakeet-cpp/diarize_sounds.go b/backend/go/parakeet-cpp/diarize_sounds.go new file mode 100644 index 000000000..ff7b88d9a --- /dev/null +++ b/backend/go/parakeet-cpp/diarize_sounds.go @@ -0,0 +1,113 @@ +package main + +import ( + "context" + "fmt" + "sort" + + "github.com/mudler/LocalAI/pkg/grpc/grpcerrors" + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" +) + +// checkSoundEventsAvailable reports why include_sounds cannot be served, or +// nil when it can. It is checked before any audio work so a model without a +// sound companion fails fast and never returns an empty list that a client +// would read as "nothing was heard". +func (p *ParakeetCpp) checkSoundEventsAvailable() error { + if p.tagCtx == 0 { + return grpcerrors.SoundEventsUnsupported("parakeet-cpp", + "the model has no sound model; add a sound_model: companion option"+p.roleHint(componentSound, "sound_component")) + } + if CppSceneOptsDefault == nil || CppSceneStreamBegin == nil || CppSceneStreamFeedJSON == nil || CppSceneStreamFree == nil { + return grpcerrors.SoundEventsUnsupported("parakeet-cpp", + "the loaded libparakeet.so has no scene stream support (parakeet_capi_scene_stream_* missing)") + } + return nil +} + +// diarizeSoundEvents runs pcm through a tagger-only scene stream and returns +// the closed sound events. It is the same stream, with the same on/off +// thresholds and minimum duration, the live path opens beside an ASR session +// (see sceneBegin), so an offline request and a live session report the same +// events for the same audio. The clip is fed in 10 s pieces with the last one +// flushing events still open at the end of the audio. +// +// Every C call runs under engineMu, held for the whole clip like +// soundStreamDrain does; the stream is freed even when a feed fails or ctx is +// cancelled. +func (p *ParakeetCpp) diarizeSoundEvents(ctx context.Context, pcm []float32) ([]sceneSoundJSON, error) { + p.engineMu.Lock() + defer p.engineMu.Unlock() + + // The caller's check ran before this lock was taken; a Free() racing in + // between zeroes tagCtx under the same lock, so check again. + if p.tagCtx == 0 { + return nil, grpcerrors.ModelNotLoaded("parakeet-cpp") + } + + var opts cSceneOpts + CppSceneOptsDefault(&opts) + // Only closed events are used, never per-class scores, so keep none (see + // sceneBegin for why a non-zero top_k would grow a queue). + opts.Sound.TopK = 0 + + stream := CppSceneStreamBegin(0, 0, p.tagCtx, &opts) + if stream == 0 { + return nil, fmt.Errorf("parakeet-cpp: sound scene stream begin failed: %s", soundLastError(p.tagCtx)) + } + defer CppSceneStreamFree(stream) + + var events []sceneSoundJSON + for off := 0; ; { + if ctx != nil { + if err := ctx.Err(); err != nil { + return nil, status.Error(codes.Canceled, "parakeet-cpp: sound detection cancelled") + } + } + end := off + soundFeedChunkSamples + last := end >= len(pcm) + if last { + end = len(pcm) + } + doc, err := sceneFeedLocked(stream, pcm[off:end], last) + if err != nil { + return nil, err + } + events = append(events, doc.Sounds...) + if last { + break + } + off = end + } + return events, nil +} + +// diarizeSoundsToProto maps closed scene sound events to DiarizeSound, sorted +// by start (then end, then label) so the order does not depend on how the +// stream closed them. Confidence is the event's peak score. The result is +// never nil: an empty clip of sound still serialises as an empty list next to +// sounds_included=true. +func diarizeSoundsToProto(sounds []sceneSoundJSON) []*pb.DiarizeSound { + out := make([]*pb.DiarizeSound, 0, len(sounds)) + for _, s := range sounds { + out = append(out, &pb.DiarizeSound{ + Start: float32(s.Start), + End: float32(s.End), + Label: s.Label, + Confidence: s.Peak, + }) + } + sort.SliceStable(out, func(i, j int) bool { + a, b := out[i], out[j] + if a.Start != b.Start { + return a.Start < b.Start + } + if a.End != b.End { + return a.End < b.End + } + return a.Label < b.Label + }) + return out +} diff --git a/backend/go/parakeet-cpp/diarize_sounds_test.go b/backend/go/parakeet-cpp/diarize_sounds_test.go new file mode 100644 index 000000000..1c4c32ca7 --- /dev/null +++ b/backend/go/parakeet-cpp/diarize_sounds_test.go @@ -0,0 +1,141 @@ +package main + +import ( + "github.com/mudler/LocalAI/pkg/grpc/grpcerrors" + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" + "google.golang.org/grpc/codes" + "google.golang.org/grpc/status" +) + +// These specs drive Diarize with include_sounds against stubbed scene stream +// entry points, the same seam live_test.go uses, so they run without +// libparakeet.so or model files. + +var _ = Describe("ParakeetCpp.Diarize include_sounds", func() { + var ( + restoreDiar func() + restore func() + pool *diarizeCstrPool + ) + + BeforeEach(func() { + restoreDiar = diarizeStubs() + pool = &diarizeCstrPool{} + sOpts, sBegin, sFeed, sFree := CppSceneOptsDefault, CppSceneStreamBegin, CppSceneStreamFeedJSON, CppSceneStreamFree + restore = func() { + CppSceneOptsDefault, CppSceneStreamBegin, CppSceneStreamFeedJSON, CppSceneStreamFree = sOpts, sBegin, sFeed, sFree + } + CppSceneOptsDefault = func(o *cSceneOpts) { *o = cSceneOpts{Sound: cSoundOpts{TopK: 5}} } + CppFreeString = func(uintptr) {} + CppDiarizePCM = func(uintptr, *float32, int32, int32) uintptr { + return pool.cstr(`{"speakers":8,"segments":[{"speaker":0,"start":0.00,"end":3.00}]}`) + } + }) + AfterEach(func() { + restore() + restoreDiar() + }) + + It("rejects the request with Unimplemented and a stable code when no sound model is loaded", func() { + p := &ParakeetCpp{diarCtx: 42} + _, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1), IncludeSounds: true}) + Expect(err).To(HaveOccurred()) + Expect(status.Code(err)).To(Equal(codes.Unimplemented)) + Expect(err.Error()).To(ContainSubstring(grpcerrors.SoundEventsUnsupportedCode)) + Expect(err.Error()).To(ContainSubstring("sound_model")) + }) + + It("does not touch the sound path when include_sounds is off", func() { + CppSceneStreamBegin = func(uintptr, uintptr, uintptr, *cSceneOpts) uintptr { + Fail("no scene stream may start without include_sounds") + return 0 + } + p := &ParakeetCpp{diarCtx: 42, tagCtx: 43} + resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1)}) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.SoundsIncluded).To(BeFalse()) + Expect(resp.Sounds).To(BeEmpty()) + }) + + It("returns closed events from a tagger-only scene stream, sorted, with the peak as confidence", func() { + var gotDiar, gotTag uintptr + var gotOpts cSceneOpts + CppSceneStreamBegin = func(asr, diar, tag uintptr, o *cSceneOpts) uintptr { + gotDiar, gotTag, gotOpts = diar, tag, *o + return 77 + } + freed := uintptr(0) + CppSceneStreamFree = func(s uintptr) { freed = s } + feeds := 0 + var lastFlags []bool + CppSceneStreamFeedJSON = func(s uintptr, _ *float32, n int32, isLast int32) uintptr { + feeds++ + lastFlags = append(lastFlags, isLast == 1) + if isLast == 1 { + return pool.cstr(`{"speakers":[],"sounds":[{"index":99,"label":"Dog","start":12.5,"end":14.0,"peak":0.7}]}`) + } + return pool.cstr(`{"speakers":[],"sounds":[{"index":3,"label":"Cough","start":2.0,"end":2.5,"peak":0.91}]}`) + } + + p := &ParakeetCpp{diarCtx: 42, tagCtx: 43} + // 15 s of audio is two 10 s feeds, the second flushing open events. + resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(15), IncludeSounds: true}) + Expect(err).ToNot(HaveOccurred()) + Expect(gotDiar).To(BeZero(), "the sound stream must not borrow the diarization model") + Expect(gotTag).To(Equal(uintptr(43))) + Expect(gotOpts.Sound.TopK).To(Equal(int32(0))) + Expect(feeds).To(Equal(2)) + Expect(lastFlags).To(Equal([]bool{false, true})) + Expect(freed).To(Equal(uintptr(77))) + + Expect(resp.SoundsIncluded).To(BeTrue()) + Expect(resp.Sounds).To(HaveLen(2)) + Expect(resp.Sounds[0].Label).To(Equal("Cough")) + Expect(resp.Sounds[0].Start).To(BeNumerically("~", 2.0, 0.001)) + Expect(resp.Sounds[0].End).To(BeNumerically("~", 2.5, 0.001)) + Expect(resp.Sounds[0].Confidence).To(BeNumerically("~", 0.91, 0.001)) + Expect(resp.Sounds[1].Label).To(Equal("Dog")) + Expect(resp.Segments).To(HaveLen(1), "speaker segments are unaffected") + }) + + It("reports an empty list as included when nothing was heard", func() { + CppSceneStreamBegin = func(uintptr, uintptr, uintptr, *cSceneOpts) uintptr { return 1 } + CppSceneStreamFree = func(uintptr) {} + CppSceneStreamFeedJSON = func(uintptr, *float32, int32, int32) uintptr { + return pool.cstr(`{"speakers":[],"sounds":[]}`) + } + p := &ParakeetCpp{diarCtx: 42, tagCtx: 43} + resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(2), IncludeSounds: true}) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.SoundsIncluded).To(BeTrue()) + Expect(resp.Sounds).To(BeEmpty()) + }) + + It("frees the stream and returns the error when a feed fails", func() { + CppSceneStreamBegin = func(uintptr, uintptr, uintptr, *cSceneOpts) uintptr { return 5 } + freed := false + CppSceneStreamFree = func(uintptr) { freed = true } + CppSceneStreamFeedJSON = func(uintptr, *float32, int32, int32) uintptr { return 0 } + p := &ParakeetCpp{diarCtx: 42, tagCtx: 43} + _, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(2), IncludeSounds: true}) + Expect(err).To(HaveOccurred()) + Expect(freed).To(BeTrue()) + }) +}) + +var _ = Describe("diarizeSoundsToProto", func() { + It("sorts by start then end then label and never returns nil", func() { + Expect(diarizeSoundsToProto(nil)).ToNot(BeNil()) + out := diarizeSoundsToProto([]sceneSoundJSON{ + {Label: "b", Start: 5, End: 6, Peak: 0.5}, + {Label: "a", Start: 1, End: 4, Peak: 0.6}, + {Label: "a", Start: 1, End: 2, Peak: 0.7}, + }) + Expect(out).To(HaveLen(3)) + Expect(out[0].End).To(BeNumerically("~", 2, 0.001)) + Expect(out[1].End).To(BeNumerically("~", 4, 0.001)) + Expect(out[2].Label).To(Equal("b")) + }) +}) diff --git a/backend/go/parakeet-cpp/scene.go b/backend/go/parakeet-cpp/scene.go index 4bac462fe..193afcbdc 100644 --- a/backend/go/parakeet-cpp/scene.go +++ b/backend/go/parakeet-cpp/scene.go @@ -172,6 +172,18 @@ func (p *ParakeetCpp) sceneFeed(h sceneStreamHandle, pcm []float32, isLast bool) return sceneFeedJSON{}, grpcerrors.ModelNotLoaded("parakeet-cpp") } + doc, err := sceneFeedLocked(h.s, pcm, isLast) + if err != nil { + return sceneFeedJSON{}, err + } + translateNames(doc.Names, h.names) + return doc, nil +} + +// sceneFeedLocked runs one scene_stream_feed_json call and decodes its +// document. The caller holds engineMu and has already checked that the +// contexts the stream borrows are still loaded. +func sceneFeedLocked(stream uintptr, pcm []float32, isLast bool) (sceneFeedJSON, error) { var last int32 if isLast { last = 1 @@ -180,11 +192,11 @@ func (p *ParakeetCpp) sceneFeed(h sceneStreamHandle, pcm []float32, isLast bool) if len(pcm) > 0 { ptr = &pcm[0] } - ret := CppSceneStreamFeedJSON(h.s, ptr, int32(len(pcm)), last) + ret := CppSceneStreamFeedJSON(stream, ptr, int32(len(pcm)), last) if ret == 0 { msg := "" if CppSceneStreamLastError != nil { - msg = CppSceneStreamLastError(h.s) + msg = CppSceneStreamLastError(stream) } if msg == "" { msg = "unknown error" @@ -197,7 +209,6 @@ func (p *ParakeetCpp) sceneFeed(h sceneStreamHandle, pcm []float32, isLast bool) if err := json.Unmarshal([]byte(raw), &doc); err != nil { return sceneFeedJSON{}, fmt.Errorf("parakeet-cpp: decode scene json: %w", err) } - translateNames(doc.Names, h.names) return doc, nil } diff --git a/core/backend/diarization.go b/core/backend/diarization.go index 16c9fb4fd..6bc24ed66 100644 --- a/core/backend/diarization.go +++ b/core/backend/diarization.go @@ -13,6 +13,7 @@ import ( "github.com/mudler/LocalAI/core/config" "github.com/mudler/LocalAI/core/schema" "github.com/mudler/LocalAI/core/services/voicerecognition" + "github.com/mudler/LocalAI/pkg/grpc/grpcerrors" grpcPkg "github.com/mudler/LocalAI/pkg/grpc" "github.com/mudler/LocalAI/pkg/grpc/proto" @@ -35,6 +36,9 @@ type DiarizationRequest struct { MinDurationOff float32 IncludeText bool IncludeSpeakerProfiles bool + // IncludeSounds asks for closed sound events (needs a sound companion on + // the model). A backend that cannot produce them must reject the request. + IncludeSounds bool // KnownVoices are registered voices a speaker-identifying backend may use // to name the speakers. Empty for every other backend and model. KnownVoices []voicerecognition.KnownVoice @@ -59,6 +63,7 @@ func (r *DiarizationRequest) toProto(threads uint32, modelIdentity string) *prot MinDurationOff: r.MinDurationOff, IncludeText: r.IncludeText, IncludeSpeakerProfiles: r.IncludeSpeakerProfiles, + IncludeSounds: r.IncludeSounds, KnownVoices: known, } } @@ -97,6 +102,12 @@ func ModelDiarization(ctx context.Context, req DiarizationRequest, ml *model.Mod if err != nil { return nil, err } + // A backend that ignores include_sounds returns no sounds_included mark. + // Reject here rather than hand the client an empty list it would read as + // "nothing was heard". + if req.IncludeSounds && !r.GetSoundsIncluded() { + return nil, grpcerrors.SoundEventsUnsupported(modelConfig.Backend, "the backend did not report sound events for this model") + } out := diarizationResultFromProto(r) if req.IncludeSpeakerProfiles { trusted, err := speakerEncoderFromBackend(ctx, m) @@ -172,6 +183,21 @@ func diarizationResultFromProto(r *proto.DiarizeResponse) *schema.DiarizationRes }) } + if r.GetSoundsIncluded() { + out.Sounds = make([]schema.DiarizationSound, 0, len(r.Sounds)) + for _, s := range r.Sounds { + if s == nil { + continue + } + out.Sounds = append(out.Sounds, schema.DiarizationSound{ + Start: float64(s.Start), + End: float64(s.End), + Label: s.Label, + Confidence: s.Confidence, + }) + } + } + out.NumSpeakers = len(order) if out.NumSpeakers == 0 && r.NumSpeakers > 0 { out.NumSpeakers = int(r.NumSpeakers) diff --git a/core/backend/diarization_sounds_test.go b/core/backend/diarization_sounds_test.go new file mode 100644 index 000000000..163ba9841 --- /dev/null +++ b/core/backend/diarization_sounds_test.go @@ -0,0 +1,40 @@ +// SPDX-License-Identifier: MIT +package backend + +import ( + "encoding/json" + + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("diarization sound events", func() { + It("sends include_sounds to the backend only when asked", func() { + Expect((&DiarizationRequest{IncludeSounds: true}).toProto(1, "m").IncludeSounds).To(BeTrue()) + Expect((&DiarizationRequest{}).toProto(1, "m").IncludeSounds).To(BeFalse()) + }) + + It("maps proto sounds to the result and keeps an empty list when the backend heard nothing", func() { + out := diarizationResultFromProto(&pb.DiarizeResponse{ + SoundsIncluded: true, + Sounds: []*pb.DiarizeSound{{Start: 1, End: 2.5, Label: "Cough", Confidence: 0.5}}, + }) + Expect(out.Sounds).To(HaveLen(1)) + Expect(out.Sounds[0].Label).To(Equal("Cough")) + Expect(out.Sounds[0].Start).To(BeNumerically("~", 1, 1e-6)) + Expect(out.Sounds[0].End).To(BeNumerically("~", 2.5, 1e-6)) + Expect(out.Sounds[0].Confidence).To(BeNumerically("~", 0.5, 1e-6)) + + empty := diarizationResultFromProto(&pb.DiarizeResponse{SoundsIncluded: true}) + raw, err := json.Marshal(empty) + Expect(err).ToNot(HaveOccurred()) + Expect(string(raw)).To(ContainSubstring(`"sounds":[]`)) + }) + + It("leaves sounds out of the payload when they were not requested", func() { + raw, err := json.Marshal(diarizationResultFromProto(&pb.DiarizeResponse{})) + Expect(err).ToNot(HaveOccurred()) + Expect(string(raw)).ToNot(ContainSubstring("sounds")) + }) +}) diff --git a/core/http/endpoints/localai/api_instructions.go b/core/http/endpoints/localai/api_instructions.go index e65e61c38..49f2bd0b9 100644 --- a/core/http/endpoints/localai/api_instructions.go +++ b/core/http/endpoints/localai/api_instructions.go @@ -40,7 +40,7 @@ var instructionDefs = []instructionDef{ Name: "audio", Description: "Text-to-speech, voice activity detection, transcription, speaker diarization, sound classification, and sound generation", Tags: []string{"audio"}, - Intro: "GET /v1/audio/voices lists named voices for installed TTS models and accepts an optional model filter. Diarization (/v1/audio/diarization) returns speaker-labelled time segments. Backends with native ASR-diarization (vibevoice-cpp) can also emit per-segment text via include_text=true; backends with a dedicated pipeline (sherpa-onnx + pyannote) emit segmentation only. Response formats: json (default), verbose_json (adds speakers summary + text), rttm (NIST format). Sound classification (/v1/audio/classification) returns scored AudioSet sound-event tags (audio tagging via the ced backend); top_k and threshold control the returned set.", + Intro: "GET /v1/audio/voices lists named voices for installed TTS models and accepts an optional model filter. Diarization (/v1/audio/diarization) returns speaker-labelled time segments. Backends with native ASR-diarization (vibevoice-cpp) can also emit per-segment text via include_text=true; backends with a dedicated pipeline (sherpa-onnx + pyannote) emit segmentation only. include_sounds=true adds timed sound events (start, end, label, confidence) when the model has a sound companion, otherwise 501 include_sounds_unsupported. Response formats: json (default), verbose_json (adds speakers summary + text), rttm (NIST format). Sound classification (/v1/audio/classification) returns scored AudioSet sound-event tags (audio tagging via the ced backend); top_k and threshold control the returned set.", }, { Name: "voice-library", diff --git a/core/http/endpoints/localai/portable_http_test.go b/core/http/endpoints/localai/portable_http_test.go index 8009c5cb9..ad603e52e 100644 --- a/core/http/endpoints/localai/portable_http_test.go +++ b/core/http/endpoints/localai/portable_http_test.go @@ -25,6 +25,7 @@ import ( "github.com/mudler/LocalAI/core/schema" "github.com/mudler/LocalAI/core/services/voicerecognition" grpcpkg "github.com/mudler/LocalAI/pkg/grpc" + "github.com/mudler/LocalAI/pkg/grpc/grpcerrors" pb "github.com/mudler/LocalAI/pkg/grpc/proto" "github.com/mudler/LocalAI/pkg/model" "github.com/mudler/LocalAI/pkg/system" @@ -41,6 +42,10 @@ type profileHTTPBackend struct { embeds int unsupported bool values [][]byte + // soundsSupported makes Diarize honour include_sounds with sounds; left + // false it behaves like a backend that ignores the field. + soundsSupported bool + sounds []*pb.DiarizeSound } func (b *profileHTTPBackend) Status(context.Context) (*pb.StatusResponse, error) { @@ -52,7 +57,12 @@ func (b *profileHTTPBackend) Status(context.Context) (*pb.StatusResponse, error) func (b *profileHTTPBackend) Diarize(_ context.Context, r *pb.DiarizeRequest, _ ...ggrpc.CallOption) (*pb.DiarizeResponse, error) { b.last = r raw, _ := json.Marshal(b.profiles) - return &pb.DiarizeResponse{Segments: []*pb.DiarizeSegment{{Speaker: "7", Start: 0, End: 3, Text: "Hello"}}, SpeakerProfilesJson: string(raw)}, nil + resp := &pb.DiarizeResponse{Segments: []*pb.DiarizeSegment{{Speaker: "7", Start: 0, End: 3, Text: "Hello"}}, SpeakerProfilesJson: string(raw)} + if r.IncludeSounds && b.soundsSupported { + resp.SoundsIncluded = true + resp.Sounds = b.sounds + } + return resp, nil } func (b *profileHTTPBackend) VoiceEmbed(context.Context, *pb.VoiceEmbedRequest, ...ggrpc.CallOption) (*pb.VoiceEmbedResponse, error) { b.embeds++ @@ -372,3 +382,74 @@ func TestPortableProfilesNeverPersistInAPITraces(t *testing.T) { time.Sleep(10 * time.Millisecond) } } + +func TestDiarizationSoundsHTTP(t *testing.T) { + b := &profileHTTPBackend{profiles: profileFixture(), soundsSupported: true, sounds: []*pb.DiarizeSound{{Start: 1.5, End: 2.25, Label: "Dog", Confidence: 0.75}}} + e, _ := profileServer(b, false) + + for _, format := range []string{"json", "verbose_json"} { + w := profileJSON(e, "/v1/audio/diarization", map[string]any{"model": "test", "file": "YXVkaW8=", "include_sounds": true, "response_format": format}) + if w.Code != 200 || !b.last.IncludeSounds { + t.Fatal(format, w.Code, w.Body.String()) + } + var got struct { + Sounds []schema.DiarizationSound `json:"sounds"` + } + if err := json.Unmarshal(w.Body.Bytes(), &got); err != nil { + t.Fatal(err) + } + want := schema.DiarizationSound{Start: 1.5, End: 2.25, Label: "Dog", Confidence: 0.75} + if len(got.Sounds) != 1 || got.Sounds[0] != want { + t.Fatal(format, w.Body.String()) + } + } + + // Multipart form field. + body := &bytes.Buffer{} + mw := multipart.NewWriter(body) + mw.WriteField("model", "test") + mw.WriteField("include_sounds", "true") + f, _ := mw.CreateFormFile("file", "sample.wav") + f.Write([]byte("audio")) + mw.Close() + r := httptest.NewRequest("POST", "/v1/audio/diarization", body) + r.Header.Set("Content-Type", mw.FormDataContentType()) + w := httptest.NewRecorder() + e.ServeHTTP(w, r) + if w.Code != 200 || !bytes.Contains(w.Body.Bytes(), []byte(`"label":"Dog"`)) { + t.Fatal(w.Code, w.Body.String()) + } + + // Not requested: the field is absent and the backend is told so. + w = profileJSON(e, "/v1/audio/diarization", map[string]any{"model": "test", "file": "YXVkaW8="}) + if w.Code != 200 || b.last.IncludeSounds || bytes.Contains(w.Body.Bytes(), []byte(`"sounds"`)) { + t.Fatal(w.Code, w.Body.String()) + } + + // Requested and heard nothing: an empty list, not an absent field. + b.sounds = nil + w = profileJSON(e, "/v1/audio/diarization", map[string]any{"model": "test", "file": "YXVkaW8=", "include_sounds": true}) + if w.Code != 200 || !bytes.Contains(w.Body.Bytes(), []byte(`"sounds":[]`)) { + t.Fatal(w.Code, w.Body.String()) + } + + // RTTM cannot carry sounds: rejected before the backend runs. + b.last = nil + w = profileJSON(e, "/v1/audio/diarization", map[string]any{"model": "test", "file": "YXVkaW8=", "include_sounds": true, "response_format": "rttm"}) + if w.Code != 400 || b.last != nil { + t.Fatal(w.Code, w.Body.String()) + } +} + +func TestDiarizationSoundsUnsupported(t *testing.T) { + // A backend that ignores include_sounds ends in a 501 carrying the stable + // code, never a 200 with an empty list. + e, _ := profileServer(&profileHTTPBackend{profiles: profileFixture()}, false) + w := profileJSON(e, "/v1/audio/diarization", map[string]any{"model": "test", "file": "YXVkaW8=", "include_sounds": true}) + if w.Code != 501 || !bytes.Contains(w.Body.Bytes(), []byte(grpcerrors.SoundEventsUnsupportedCode)) { + t.Fatal(w.Code, w.Body.String()) + } + if bytes.Contains(w.Body.Bytes(), []byte(`"sounds"`)) { + t.Fatal("unsupported response carried a sounds field", w.Body.String()) + } +} diff --git a/core/http/endpoints/openai/diarization.go b/core/http/endpoints/openai/diarization.go index db3329dee..b75500334 100644 --- a/core/http/endpoints/openai/diarization.go +++ b/core/http/endpoints/openai/diarization.go @@ -41,7 +41,7 @@ import ( // (NIST RTTM, the standard interchange format used by pyannote/dscore). // // @Summary Identify speakers in audio (who spoke when). -// @Description JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles and response_format. Profiles require voice-recognition permission and json or verbose_json; unsupported backends return 501. +// @Description JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles, include_sounds and response_format. Profiles require voice-recognition permission and json or verbose_json; unsupported backends return 501. // @Tags audio // @accept multipart/form-data,json // @Param model formData string true "model" @@ -55,6 +55,7 @@ import ( // @Param language formData string false "audio language hint (only meaningful for backends that bundle ASR)" // @Param include_speaker_profiles formData boolean false "export portable biometric profiles (voice-recognition permission; JSON formats only)" // @Param include_text formData boolean false "include per-segment transcript when the backend supports it" +// @Param include_sounds formData boolean false "include closed sound events (start, end, label, confidence) when the model has a sound_model companion; otherwise 501 include_sounds_unsupported (JSON formats only)" // @Param response_format formData string false "json (default), verbose_json, or rttm" // @Success 200 {object} schema.DiarizationResult // @Router /v1/audio/diarization [post] @@ -74,6 +75,7 @@ func DiarizationEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, ap Language: input.Language, IncludeText: parseFormBool(c, "include_text", input.IncludeText), IncludeSpeakerProfiles: parseFormBool(c, "include_speaker_profiles", input.IncludeSpeakerProfiles), + IncludeSounds: parseFormBool(c, "include_sounds", input.IncludeSounds), } if req.IncludeSpeakerProfiles { var db *gorm.DB @@ -118,6 +120,9 @@ func DiarizationEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, ap if req.IncludeSpeakerProfiles && responseFormat == schema.DiarizationResponseFormatRTTM { return echo.NewHTTPError(http.StatusBadRequest, "speaker_profiles requires json or verbose_json") } + if req.IncludeSounds && responseFormat == schema.DiarizationResponseFormatRTTM { + return echo.NewHTTPError(http.StatusBadRequest, "include_sounds requires json or verbose_json") + } var sourceName = "audio.wav" var reader io.ReadCloser if strings.HasPrefix(c.Request().Header.Get(echo.HeaderContentType), echo.MIMEApplicationJSON) { diff --git a/core/schema/diarization.go b/core/schema/diarization.go index 5ad88c20e..7bd1cf15f 100644 --- a/core/schema/diarization.go +++ b/core/schema/diarization.go @@ -30,6 +30,16 @@ type DiarizationSpeaker struct { SegmentCount int `json:"segment_count"` } +// DiarizationSound is one closed sound event found over the whole clip. Times +// are in seconds. Label is an AudioSet class name and Confidence the peak score +// of the event (0 to 1). +type DiarizationSound struct { + Start float64 `json:"start"` + End float64 `json:"end"` + Label string `json:"label"` + Confidence float32 `json:"confidence"` +} + // DiarizationResult is the JSON payload returned by /v1/audio/diarization. // Speakers and segment text are omitted when empty so the default `json` // response stays minimal; verbose_json keeps both populated. @@ -41,6 +51,10 @@ type DiarizationResult struct { NumSpeakers int `json:"num_speakers"` Segments []DiarizationSegment `json:"segments"` Speakers []DiarizationSpeaker `json:"speakers,omitempty"` + // Sounds is present only when the request set include_sounds. An empty + // list then means the model ran and heard no event; omitzero keeps a nil + // list (not requested) out of the payload while an empty one stays. + Sounds []DiarizationSound `json:"sounds,omitzero"` } // DiarizationResponseFormatType mirrors transcription's response_format diff --git a/core/schema/openai.go b/core/schema/openai.go index 4646c94c5..ff8b3b5c6 100644 --- a/core/schema/openai.go +++ b/core/schema/openai.go @@ -189,6 +189,7 @@ type JsonSchema struct { type OpenAIRequest struct { IncludeSpeakerProfiles bool `json:"include_speaker_profiles,omitempty"` IncludeText bool `json:"include_text,omitempty"` + IncludeSounds bool `json:"include_sounds,omitempty"` PredictionOptions Context context.Context `json:"-"` diff --git a/docs/content/features/audio-classification.md b/docs/content/features/audio-classification.md index 572983651..c50d7e66f 100644 --- a/docs/content/features/audio-classification.md +++ b/docs/content/features/audio-classification.md @@ -9,7 +9,7 @@ Sound-event classification (audio tagging) answers the question **"what am I hea LocalAI exposes this through the `/v1/audio/classification` endpoint, modelled after `/v1/audio/transcriptions`. The reference backend is **[ced.cpp](https://github.com/localai-org/ced.cpp)** (CED, a 527-class AudioSet tagger), a small ViT over a log-mel spectrogram ported to ggml with full PyTorch parity. Apache-2.0 weights are redistributable as GGUF. -**[parakeet.cpp](https://github.com/mudler/parakeet.cpp)** can also load a CED model (through `third_party/ced.cpp`) and serve `/v1/audio/classification` from the same backend used for ASR and diarization. It scores the clip in 10 s windows and averages each class's score across the windows before sorting and applying `top_k`/`threshold` - CED's own method for clips longer than one window. Install `parakeet-cpp-ced-tiny` or `parakeet-cpp-ced-base` from the gallery, or point `parameters.model` at a CED GGUF under `backend: parakeet-cpp`. A parakeet-cpp ASR model can also point `sound_model` at a CED GGUF to add live sound events during realtime transcription - see [Realtime API]({{% relref "openai-realtime" %}}). +**[parakeet.cpp](https://github.com/mudler/parakeet.cpp)** can also load a CED model (through `third_party/ced.cpp`) and serve `/v1/audio/classification` from the same backend used for ASR and diarization. It scores the clip in 10 s windows and averages each class's score across the windows before sorting and applying `top_k`/`threshold` - CED's own method for clips longer than one window. Install `parakeet-cpp-ced-tiny` or `parakeet-cpp-ced-base` from the gallery, or point `parameters.model` at a CED GGUF under `backend: parakeet-cpp`. A parakeet-cpp ASR model can also point `sound_model` at a CED GGUF to add live sound events during realtime transcription - see [Realtime API]({{% relref "openai-realtime" %}}) - and to add timed sound events to an offline diarization call with `include_sounds=true` - see [Speaker Diarization]({{% relref "audio-diarization" %}}#sound-events). Such a model also answers `/v1/audio/classification` itself when it declares the `sound_classification` usecase, as the gallery entry `parakeet-cpp-multilingual-diarization-speakers-sounds` does. Because classification is exposed as a regular OpenAI-style endpoint, any HTTP client works - there is no Python dependency on the consumer side. diff --git a/docs/content/features/audio-diarization.md b/docs/content/features/audio-diarization.md index 649da38e9..0ab4ca697 100644 --- a/docs/content/features/audio-diarization.md +++ b/docs/content/features/audio-diarization.md @@ -42,6 +42,7 @@ Content-Type: multipart/form-data | `min_duration_off` | float | merge gaps shorter than this many seconds | | `language` | string | only meaningful for backends that bundle ASR (e.g. vibevoice) | | `include_text` | bool | when the backend can emit per-segment transcript for free, populate it | +| `include_sounds` | bool | add the closed sound events of the clip as `sounds`. Needs a model with a sound companion; see [Sound events](#sound-events). Default `false` | | `response_format` | string | `json` (default), `verbose_json`, or `rttm` | ### Response - `json` (default) @@ -103,6 +104,44 @@ With a parakeet-cpp model that has a `speaker_model:` (or a `speaker_component:` } ``` +### Sound events + +With `include_sounds=true` the response gains a `sounds` array: the sound events (AudioSet labels such as `Dog`, `Applause`, `Cough`) found anywhere in the clip, in the same call that returns the speakers, the text and the voice prints. Each item is `{start, end, label, confidence}`: `start` and `end` are in seconds from the start of the audio, and `confidence` is the peak score the tagger reached while the event lasted (0 to 1). Items are sorted by `start`. Both `json` and `verbose_json` carry the array; `rttm` has no place for it and returns 400. + +```bash +curl http://localhost:8080/v1/audio/diarization \ + -F model=parakeet-cpp-multilingual-diarization-speakers-sounds \ + -F file=@meeting.wav \ + -F include_text=true -F include_speaker_profiles=true -F include_sounds=true \ + -F response_format=verbose_json +``` + +```json +{ + "task": "diarize", + "duration": 31.2, + "language": "en", + "num_speakers": 2, + "segments": [ + {"id": 0, "speaker": "SPEAKER_00", "label": "0", "start": 0.0, "end": 6.4, "text": "Good morning, everyone."}, + {"id": 1, "speaker": "SPEAKER_01", "label": "1", "start": 6.8, "end": 11.2, "text": "Morning."} + ], + "speakers": [ + {"id": "SPEAKER_00", "label": "0", "total_speech_duration": 6.4, "segment_count": 1}, + {"id": "SPEAKER_01", "label": "1", "total_speech_duration": 4.4, "segment_count": 1} + ], + "speaker_profiles": {"version": 1, "encoder": {"identity": "sha256:...", "dimension": 256}, "speakers": ["..."]}, + "sounds": [ + {"start": 5.5, "end": 6.75, "label": "Cough", "confidence": 0.91}, + {"start": 12.0, "end": 14.5, "label": "Applause", "confidence": 0.68} + ] +} +``` + +An empty `sounds` array means the sound model ran and found no event. The field is absent when `include_sounds` is not set. The events come from the same sound stream, with the same on and off thresholds, that a [realtime session]({{% relref "openai-realtime" %}}) runs, so a clip gives the same events offline and live. The thresholds are not request fields. + +The model needs a sound companion, a `sound_model:` option pointing at a CED GGUF (the gallery entry `parakeet-cpp-multilingual-diarization-speakers-sounds` has one). Without it the request fails with HTTP 501 and a message that starts with the stable code `include_sounds_unsupported`; the same happens for a backend that cannot report sound events. LocalAI never answers with an empty list in place of that error. The speaker segments, text and voice prints are independent of the sound model and cost nothing extra when `include_sounds` is off. + ### Response - `rttm` NIST RTTM, the standard interchange format used by `pyannote.metrics` / `dscore`: @@ -191,11 +230,14 @@ Choose an existing gallery entry for the output you need: | Speaker turns only | `parakeet-cpp-nemotron-3-diarization` | Default options | | Speaker turns and transcript | `parakeet-cpp-nemotron-3-diarization-asr` | `include_text=true`, `response_format=verbose_json` | | Speaker turns, transcript, and identification | `parakeet-cpp-nemotron-3-diarization-asr-speakers` | Same transcript options; explicitly enroll voices for names | +| Multilingual transcript, speakers, voice prints and sound events in one call | `parakeet-cpp-multilingual-diarization-speakers-sounds` | `include_text=true`, `include_speaker_profiles=true`, `include_sounds=true`, `response_format=verbose_json` | The complete `-asr-speakers` entry downloads Nemotron-3-Diarization, Parakeet TDT+CTC 110M ASR, and the WeSpeaker ResNet34 speaker encoder. It configures both `asr_model` and `speaker_model`; no custom gallery configuration is needed. See [Remember speakers in the Web UI](#remember-speakers-in-the-web-ui) for installation and enrollment. +The `parakeet-cpp-multilingual-diarization-speakers-sounds` entry downloads Parakeet TDT 0.6B v3 (25 European languages), Nemotron-3-Diarization, the WeSpeaker ResNet34 speaker encoder and CED-Tiny, and sets `diarization_model`, `speaker_model` and `sound_model`. It answers the whole request above from one model name. See [Sound events](#sound-events). + The entries `parakeet-cpp-bundle-small` and `parakeet-cpp-bundle-standard` hold Nemotron-3-Diarization, an ASR model and the WeSpeaker speaker encoder in one file (`diar_component:diar` and `speaker_component:voice`), so one install serves the transcript and the identification options above. See [Bundle GGUF files]({{% relref "audio-to-text" %}}#bundle-gguf-files-several-models-in-one-file). For manual configuration, this example pairs Sortformer with ASR: diff --git a/docs/content/features/audio-to-text.md b/docs/content/features/audio-to-text.md index de0eb3ec2..ceeabc646 100644 --- a/docs/content/features/audio-to-text.md +++ b/docs/content/features/audio-to-text.md @@ -200,7 +200,7 @@ The same backend also serves the `/v1/audio/diarization` and `/v1/audio/classifi |---|---|---| | `asr_model:` | a diarization model | `include_text` on `/v1/audio/diarization` | | `diarization_model:` | an ASR model | a `speaker` on transcript segments (and words), and speaker segments during realtime live transcription | -| `sound_model:` | an ASR model | sound events during realtime live transcription | +| `sound_model:` | an ASR or diarization model | sound events during realtime live transcription, `sounds` on `/v1/audio/diarization` with `include_sounds=true`, and `/v1/audio/classification` on the loaded model | | `diarization_latency:` | a model with a diarization companion | latency mode for the live speaker stream; default `low` | | `speaker_model:` | a model with a diarization model | names registered speakers; a bundle can use `speaker_component:` instead (see [Bundle GGUF files](#bundle-gguf-files-several-models-in-one-file)) (see [Voice Recognition]({{% relref "voice-recognition" %}}#naming-speakers-in-diarization-and-live-transcription)) | | `speaker_tag:` | a model with `speaker_component` | extra encoder tag for registered voices that carry only a file-name tag (see [Voice Recognition]({{% relref "voice-recognition" %}}#naming-speakers-from-a-bundle)) | diff --git a/docs/content/features/openai-realtime.md b/docs/content/features/openai-realtime.md index f4ba114b2..0679042ad 100644 --- a/docs/content/features/openai-realtime.md +++ b/docs/content/features/openai-realtime.md @@ -230,7 +230,7 @@ pipeline: #### Choosing the sound model -Both scene models ship with CED-Tiny, the cheapest to run all the time. `parakeet-cpp-realtime-scene-base` and `parakeet-cpp-realtime-scene-tdt-base` are the same pipelines with CED-Base (86M, the largest CED), which tags sounds more confidently. Any CED GGUF from [`mudler/ced-gguf`](https://huggingface.co/mudler/ced-gguf) (tiny, mini, small, base) works as `sound_model`. Measured on CPU (Ryzen 9 9950X3D) over a 37 s clip with two speakers and a rooster, as a fraction of real time: +Both scene models ship with CED-Tiny, the cheapest to run all the time. `parakeet-cpp-realtime-scene-base` and `parakeet-cpp-realtime-scene-tdt-base` are the same pipelines with CED-Base (86M, the largest CED), which tags sounds more confidently. `parakeet-cpp-multilingual-diarization-speakers-sounds` loads the same TDT, Nemotron-3-Diarization and CED-Tiny trio plus the WeSpeaker encoder, and is meant for offline `/v1/audio/diarization` calls with `include_sounds=true` (see [Speaker Diarization]({{% relref "audio-diarization" %}}#sound-events)). Any CED GGUF from [`mudler/ced-gguf`](https://huggingface.co/mudler/ced-gguf) (tiny, mini, small, base) works as `sound_model`. Measured on CPU (Ryzen 9 9950X3D) over a 37 s clip with two speakers and a rooster, as a fraction of real time: | | CED-Tiny | CED-Base | |---|---|---| diff --git a/docs/content/features/voice-recognition.md b/docs/content/features/voice-recognition.md index b9512ae90..be1e12043 100644 --- a/docs/content/features/voice-recognition.md +++ b/docs/content/features/voice-recognition.md @@ -256,7 +256,9 @@ this, speakers only carry labels such as `SPEAKER_00`. 2. Install one of the gallery models that loads the same encoder: `parakeet-cpp-nemotron-3-diarization-speakers` (diarization), `parakeet-cpp-nemotron-3-diarization-asr-speakers` (diarization with - `include_text`) or `parakeet-cpp-realtime-scene-speakers` (live + `include_text`), `parakeet-cpp-multilingual-diarization-speakers-sounds` + (multilingual diarization with `include_text`, voice prints and sound events) + or `parakeet-cpp-realtime-scene-speakers` (live transcription). Each one adds `speaker_model:voice-detect-wespeaker-resnet34.gguf` to a parakeet-cpp model config. diff --git a/gallery/index.yaml b/gallery/index.yaml index 31e6a6aa1..2e6c8ada6 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -54538,6 +54538,72 @@ - filename: voice-detect-wespeaker-resnet34.gguf uri: https://huggingface.co/mudler/voice-detect-gguf/resolve/main/wespeaker-resnet34-voxceleb.gguf sha256: 72040372494eafec299836bc1977cfc13c603cb486674ed59b0f4c03758d29da +- name: parakeet-cpp-multilingual-diarization-speakers-sounds + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/mudler/parakeet-cpp-gguf + - https://huggingface.co/mudler/voice-detect-gguf + - https://huggingface.co/mudler/ced-gguf + - https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3 + - https://huggingface.co/nvidia/Nemotron-3-Diarization + - https://huggingface.co/mispeech/ced-tiny + - https://github.com/mudler/parakeet.cpp + description: | + Parakeet TDT 0.6B v3 (multilingual, 25 European languages) paired with + Nemotron-3-Diarization (Sortformer) through the diarization_model option, + WeSpeaker ResNet34 through the speaker_model option and CED-Tiny through + the sound_model option, all GGUF for the parakeet-cpp backend (C++/ggml + port of NVIDIA NeMo). One call to /v1/audio/diarization with include_text, + include_speaker_profiles and include_sounds returns the speaker turns, the + text of each turn in the spoken language, one voice-print embedding per + speaker and the sound events (AudioSet labels with start and end times), + so a client does not need separate diarization, transcription and sound + tagging calls. Also serves /v1/audio/transcriptions and + /v1/audio/classification. License per model: transcription model + CC-BY-4.0, diarization model OpenMDW-1.1, WeSpeaker encoder CC-BY-4.0, + CED-Tiny Apache-2.0. + license: cc-by-4.0 + tags: + - parakeet + - parakeet-cpp + - nemotron + - sortformer + - ced + - asr + - diarization + - speaker-diarization + - sound-classification + - speech-recognition + - multilingual + - stt + - gguf + - ggml + overrides: + backend: parakeet-cpp + known_usecases: + - transcript + - diarization + - sound_classification + name: parakeet-cpp-multilingual-diarization-speakers-sounds + options: + - diarization_model:parakeet-cpp/nemotron-3-diarization-q8_0.gguf + - speaker_model:voice-detect-wespeaker-resnet34.gguf + - sound_model:parakeet-cpp/ced-tiny-q8_0.gguf + parameters: + model: parakeet-cpp/tdt-0.6b-v3-f16.gguf + files: + - filename: parakeet-cpp/tdt-0.6b-v3-f16.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/tdt-0.6b-v3-f16.gguf + sha256: 8ba47343e1e919895aca90e099150a01ed203ee0942d8ed31e27295efc5abb22 + - filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf + uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf + sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22 + - filename: voice-detect-wespeaker-resnet34.gguf + uri: https://huggingface.co/mudler/voice-detect-gguf/resolve/main/wespeaker-resnet34-voxceleb.gguf + sha256: 72040372494eafec299836bc1977cfc13c603cb486674ed59b0f4c03758d29da + - filename: parakeet-cpp/ced-tiny-q8_0.gguf + uri: huggingface://mudler/ced-gguf/ced-tiny-q8_0.gguf + sha256: 48bee4e2fc3cc85d7806e03471db24e77fda6c2a2e81ffe9ef67caebaf2bd674 - name: parakeet-cpp-realtime-scene-speakers url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: diff --git a/pkg/grpc/grpcerrors/errors.go b/pkg/grpc/grpcerrors/errors.go index 4a0306338..8a3a261ac 100644 --- a/pkg/grpc/grpcerrors/errors.go +++ b/pkg/grpc/grpcerrors/errors.go @@ -125,3 +125,18 @@ func IsUnimplemented(err error) bool { func StreamTranscriptionUnsupported(backend, reason string) error { return status.Errorf(codes.Unimplemented, "%s: streaming transcription unsupported: %s", backend, reason) } + +// SoundEventsUnsupportedCode is the stable code carried in the message of the +// error a diarization request gets when it asks for include_sounds and the +// model cannot produce sound events. Clients match on this string, so it is +// part of the API contract. +const SoundEventsUnsupportedCode = "include_sounds_unsupported" + +// SoundEventsUnsupported returns the canonical error a backend returns when a +// diarization request sets include_sounds but the loaded model has no sound +// (CED) companion. It carries codes.Unimplemented, which the HTTP layer maps to +// 501, so the caller learns the capability is missing instead of reading an +// empty sound list as "nothing was heard". +func SoundEventsUnsupported(backend, reason string) error { + return status.Errorf(codes.Unimplemented, "%s: %s: %s", backend, SoundEventsUnsupportedCode, reason) +} diff --git a/pkg/grpc/grpcerrors/errors_test.go b/pkg/grpc/grpcerrors/errors_test.go index cb7be2978..2daddc693 100644 --- a/pkg/grpc/grpcerrors/errors_test.go +++ b/pkg/grpc/grpcerrors/errors_test.go @@ -106,3 +106,12 @@ var _ = Describe("grpcerrors", func() { Expect(grpcerrors.IsModelNotLoaded(err)).To(BeFalse()) }) }) + +var _ = Describe("SoundEventsUnsupported", func() { + It("is Unimplemented and carries the stable code", func() { + err := grpcerrors.SoundEventsUnsupported("parakeet-cpp", "no sound model") + Expect(status.Code(err)).To(Equal(codes.Unimplemented)) + Expect(err.Error()).To(ContainSubstring("include_sounds_unsupported")) + Expect(grpcerrors.SoundEventsUnsupportedCode).To(Equal("include_sounds_unsupported")) + }) +}) diff --git a/swagger/docs.go b/swagger/docs.go index b08a320b8..13c9304f7 100644 --- a/swagger/docs.go +++ b/swagger/docs.go @@ -2909,7 +2909,7 @@ const docTemplate = `{ }, "/v1/audio/diarization": { "post": { - "description": "JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles and response_format. Profiles require voice-recognition permission and json or verbose_json; unsupported backends return 501.", + "description": "JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles, include_sounds and response_format. Profiles require voice-recognition permission and json or verbose_json; unsupported backends return 501.", "consumes": [ "multipart/form-data", "application/json" @@ -2987,6 +2987,12 @@ const docTemplate = `{ "name": "include_text", "in": "formData" }, + { + "type": "boolean", + "description": "include closed sound events (start, end, label, confidence) when the model has a sound_model companion; otherwise 501 include_sounds_unsupported (JSON formats only)", + "name": "include_sounds", + "in": "formData" + }, { "type": "string", "description": "json (default), verbose_json, or rttm", @@ -5861,6 +5867,13 @@ const docTemplate = `{ "$ref": "#/definitions/schema.DiarizationSegment" } }, + "sounds": { + "description": "Sounds is present only when the request set include_sounds. An empty\nlist then means the model ran and heard no event; omitzero keeps a nil\nlist (not requested) out of the payload while an empty one stays.", + "type": "array", + "items": { + "$ref": "#/definitions/schema.DiarizationSound" + } + }, "speaker_profiles": { "$ref": "#/definitions/schema.SpeakerProfiles" }, @@ -5905,6 +5918,23 @@ const docTemplate = `{ } } }, + "schema.DiarizationSound": { + "type": "object", + "properties": { + "confidence": { + "type": "number" + }, + "end": { + "type": "number" + }, + "label": { + "type": "string" + }, + "start": { + "type": "number" + } + } + }, "schema.DiarizationSpeaker": { "type": "object", "properties": { @@ -7528,6 +7558,9 @@ const docTemplate = `{ "ignore_eos": { "type": "boolean" }, + "include_sounds": { + "type": "boolean" + }, "include_speaker_profiles": { "type": "boolean" }, diff --git a/swagger/swagger.json b/swagger/swagger.json index c328f4f09..bff76511f 100644 --- a/swagger/swagger.json +++ b/swagger/swagger.json @@ -2906,7 +2906,7 @@ }, "/v1/audio/diarization": { "post": { - "description": "JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles and response_format. Profiles require voice-recognition permission and json or verbose_json; unsupported backends return 501.", + "description": "JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles, include_sounds and response_format. Profiles require voice-recognition permission and json or verbose_json; unsupported backends return 501.", "consumes": [ "multipart/form-data", "application/json" @@ -2984,6 +2984,12 @@ "name": "include_text", "in": "formData" }, + { + "type": "boolean", + "description": "include closed sound events (start, end, label, confidence) when the model has a sound_model companion; otherwise 501 include_sounds_unsupported (JSON formats only)", + "name": "include_sounds", + "in": "formData" + }, { "type": "string", "description": "json (default), verbose_json, or rttm", @@ -5858,6 +5864,13 @@ "$ref": "#/definitions/schema.DiarizationSegment" } }, + "sounds": { + "description": "Sounds is present only when the request set include_sounds. An empty\nlist then means the model ran and heard no event; omitzero keeps a nil\nlist (not requested) out of the payload while an empty one stays.", + "type": "array", + "items": { + "$ref": "#/definitions/schema.DiarizationSound" + } + }, "speaker_profiles": { "$ref": "#/definitions/schema.SpeakerProfiles" }, @@ -5902,6 +5915,23 @@ } } }, + "schema.DiarizationSound": { + "type": "object", + "properties": { + "confidence": { + "type": "number" + }, + "end": { + "type": "number" + }, + "label": { + "type": "string" + }, + "start": { + "type": "number" + } + } + }, "schema.DiarizationSpeaker": { "type": "object", "properties": { @@ -7525,6 +7555,9 @@ "ignore_eos": { "type": "boolean" }, + "include_sounds": { + "type": "boolean" + }, "include_speaker_profiles": { "type": "boolean" }, diff --git a/swagger/swagger.yaml b/swagger/swagger.yaml index 4527627c8..244507827 100644 --- a/swagger/swagger.yaml +++ b/swagger/swagger.yaml @@ -1078,6 +1078,14 @@ definitions: items: $ref: '#/definitions/schema.DiarizationSegment' type: array + sounds: + description: |- + Sounds is present only when the request set include_sounds. An empty + list then means the model ran and heard no event; omitzero keeps a nil + list (not requested) out of the payload while an empty one stays. + items: + $ref: '#/definitions/schema.DiarizationSound' + type: array speaker_profiles: $ref: '#/definitions/schema.SpeakerProfiles' speakers: @@ -1110,6 +1118,17 @@ definitions: text: type: string type: object + schema.DiarizationSound: + properties: + confidence: + type: number + end: + type: number + label: + type: string + start: + type: number + type: object schema.DiarizationSpeaker: properties: id: @@ -2226,6 +2245,8 @@ definitions: $ref: '#/definitions/functions.JSONFunctionStructure' ignore_eos: type: boolean + include_sounds: + type: boolean include_speaker_profiles: type: boolean include_text: @@ -5413,9 +5434,9 @@ paths: consumes: - multipart/form-data - application/json - description: JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles - and response_format. Profiles require voice-recognition permission and json - or verbose_json; unsupported backends return 501. + description: JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles, + include_sounds and response_format. Profiles require voice-recognition permission + and json or verbose_json; unsupported backends return 501. parameters: - description: model in: formData @@ -5465,6 +5486,12 @@ paths: in: formData name: include_text type: boolean + - description: include closed sound events (start, end, label, confidence) when + the model has a sound_model companion; otherwise 501 include_sounds_unsupported + (JSON formats only) + in: formData + name: include_sounds + type: boolean - description: json (default), verbose_json, or rttm in: formData name: response_format diff --git a/tests/e2e/mock-backend/main.go b/tests/e2e/mock-backend/main.go index 59819e627..c5b403d9b 100644 --- a/tests/e2e/mock-backend/main.go +++ b/tests/e2e/mock-backend/main.go @@ -975,7 +975,7 @@ func (m *MockBackend) Diarize(ctx context.Context, in *pb.DiarizeRequest) (*pb.D } return out } - return &pb.DiarizeResponse{ + resp := &pb.DiarizeResponse{ Segments: []*pb.DiarizeSegment{ seg(0.0, 1.0, "5", "hello there"), seg(1.0, 2.0, "2", "general kenobi"), @@ -984,7 +984,16 @@ func (m *MockBackend) Diarize(ctx context.Context, in *pb.DiarizeRequest) (*pb.D NumSpeakers: 2, Duration: 3.5, Language: in.Language, - }, nil + } + // IncludeSounds gates the sound events; the mock always has a sound model. + if in.IncludeSounds { + resp.SoundsIncluded = true + resp.Sounds = []*pb.DiarizeSound{ + {Start: 0.5, End: 1.25, Label: "Door", Confidence: 0.8}, + {Start: 2.0, End: 3.0, Label: "Applause", Confidence: 0.6}, + } + } + return resp, nil } func (m *MockBackend) AudioEncode(ctx context.Context, in *pb.AudioEncodeRequest) (*pb.AudioEncodeResult, error) { diff --git a/tests/e2e/mock_backend_test.go b/tests/e2e/mock_backend_test.go index 043a320e0..3711de280 100644 --- a/tests/e2e/mock_backend_test.go +++ b/tests/e2e/mock_backend_test.go @@ -406,6 +406,28 @@ var _ = Describe("Mock Backend E2E Tests", Label("MockBackend"), func() { Expect(segs[1].(map[string]any)["text"]).To(Equal("general kenobi")) }) + It("returns closed sound events only when include_sounds is set", func() { + resp, data := postDiarize(map[string]string{"response_format": "verbose_json", "include_sounds": "true"}) + Expect(resp.StatusCode).To(Equal(http.StatusOK)) + var got map[string]any + Expect(json.Unmarshal(data, &got)).To(Succeed()) + sounds, ok := got["sounds"].([]any) + Expect(ok).To(BeTrue(), "include_sounds must add a sounds array") + Expect(sounds).To(HaveLen(2)) + first := sounds[0].(map[string]any) + Expect(first["label"]).To(Equal("Door")) + Expect(first["start"]).To(BeNumerically("~", 0.5, 0.001)) + Expect(first["end"]).To(BeNumerically("~", 1.25, 0.001)) + Expect(first["confidence"]).To(BeNumerically("~", 0.8, 0.001)) + + resp, data = postDiarize(map[string]string{"response_format": "verbose_json"}) + Expect(resp.StatusCode).To(Equal(http.StatusOK)) + Expect(string(data)).ToNot(ContainSubstring(`"sounds"`)) + + resp, _ = postDiarize(map[string]string{"response_format": "rttm", "include_sounds": "true"}) + Expect(resp.StatusCode).To(Equal(http.StatusBadRequest)) + }) + It("rttm response_format returns NIST RTTM rows", func() { resp, data := postDiarize(map[string]string{"response_format": "rttm"}) Expect(resp.StatusCode).To(Equal(http.StatusOK))