feat(diarization): return sound events with include_sounds (#12544)

* feat(diarization): return sound events with include_sounds

A client that wants text, speakers, voice prints and sound events had to
make a diarization call and a separate sound call. Add an include_sounds
request field to /v1/audio/diarization that adds a sounds array of closed
events {start, end, label, confidence}, in seconds.

The parakeet-cpp backend runs a tagger-only scene stream over the clip,
the same stream and thresholds the live path uses, so a clip gives the
same events offline and live. A model with no sound_model companion, or a
backend that does not report sound events, fails with 501 and the stable
code include_sounds_unsupported instead of an empty list. The proto
carries sounds_included so an empty list still means "nothing heard".

The localai-proxy backend forwards the field. Swagger, docs and the
e2e mock backend are updated.

Assisted-by: Claude:claude-sonnet-5-5 [protoc swag go]

* feat(gallery): add parakeet-cpp-multilingual-diarization-speakers-sounds

Same as parakeet-cpp-multilingual-diarization-speakers (TDT 0.6B v3,
Nemotron-3-Diarization, WeSpeaker) plus a CED-Tiny sound_model, so one
model name serves /v1/audio/diarization with include_text,
include_speaker_profiles and include_sounds. It declares the
sound_classification usecase like the realtime scene entries.

Assisted-by: Claude:claude-sonnet-5-5

---------

Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
This commit is contained in:
localai-org-maint-botandEttore Di Giacinto authored and GitHub committed 2026-10-07 14:10:45 +02:00
1 parent a3d555653c
commit 6b794651a4
27 files changed
+771 -17

No files matched your search

+12
View File
@@ -831,6 +831,7 @@ message DiarizeRequest {
// identify speakers themselves read this; others ignore it.
repeated KnownVoice known_voices = 12;
bool include_speaker_profiles = 13; // opt-in sensitive embeddings; unsupported backends must reject
bool include_sounds = 14; // ask for closed sound events; backends without a sound model must reject, never return an empty list
}
message DiarizeSegment {
@@ -858,8 +859,19 @@ message KnownVoice {
string encoder_weights = 6;
}
// DiarizeSound is one closed sound event over the whole clip, merged the same
// way the live scene stream merges them.
message DiarizeSound {
float start = 1; // seconds
float end = 2; // seconds
string label = 3; // AudioSet label
float confidence = 4; // peak score of the event, 0..1
}
message DiarizeResponse {
string speaker_profiles_json = 5; // versioned speaker_profiles object only; absent by default
repeated DiarizeSound sounds = 6; // closed sound events; only filled when include_sounds was set
bool sounds_included = 7; // true when the backend ran sound detection for this request (an empty `sounds` is then a real "nothing heard")
repeated DiarizeSegment segments = 1;
int32 num_speakers = 2; // count of distinct speaker labels in `segments`
float duration = 3; // total audio duration in seconds (0 if unknown)
+18
View File
@@ -293,6 +293,9 @@ func (p *LocalAIProxy) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, erro
if req.GetIncludeText() {
f.Set("include_text", "true")
}
if req.GetIncludeSounds() {
f.Set("include_sounds", "true")
}
var resp struct {
Duration float64 `json:"duration"`
@@ -306,6 +309,14 @@ func (p *LocalAIProxy) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, erro
End float32 `json:"end"`
Text string `json:"text"`
} `json:"segments"`
// A pointer tells an absent field (the upstream did not report sound
// events) from an empty list (it ran and heard nothing).
Sounds *[]struct {
Start float32 `json:"start"`
End float32 `json:"end"`
Label string `json:"label"`
Confidence float32 `json:"confidence"`
} `json:"sounds"`
}
if err := p.postForm(context.Background(), "/v1/audio/diarization",
multipartForm{fields: f, files: []formFile{{field: "file", path: req.GetDst()}}}, &resp); err != nil {
@@ -324,8 +335,15 @@ func (p *LocalAIProxy) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, erro
Id: s.ID, Start: s.Start, End: s.End, Speaker: speaker, Text: s.Text,
})
}
var sounds []*pb.DiarizeSound
if resp.Sounds != nil {
for _, s := range *resp.Sounds {
sounds = append(sounds, &pb.DiarizeSound{Start: s.Start, End: s.End, Label: s.Label, Confidence: s.Confidence})
}
}
return pb.DiarizeResponse{
Segments: segments, NumSpeakers: resp.NumSpeakers, Duration: float32(resp.Duration), Language: resp.Language,
Sounds: sounds, SoundsIncluded: resp.Sounds != nil,
}, nil
}
+17
View File
@@ -375,6 +375,23 @@ var _ = Describe("audio methods", func() {
Expect(res.Segments[1].Speaker).To(Equal("SPEAKER_01"))
Expect(res.Segments[1].Start).To(BeNumerically("==", 1.5))
Expect(res.Segments[1].Id).To(Equal(int32(1)))
Expect(res.SoundsIncluded).To(BeFalse(), "an upstream that sends no sounds field reports none")
})
It("forwards include_sounds and keeps the upstream's sound events", func() {
p := loadProxy(up, nil)
up.replyJSON("/v1/audio/diarization", map[string]any{
"task": "diarize", "duration": 3.0, "num_speakers": 1,
"segments": []any{map[string]any{"id": 0, "speaker": "SPEAKER_00", "start": 0.0, "end": 3.0}},
"sounds": []any{map[string]any{"start": 1.0, "end": 2.0, "label": "Dog", "confidence": 0.75}},
})
res, err := p.Diarize(&pb.DiarizeRequest{Dst: writeInput("talk.wav", "RIFF-talk"), IncludeSounds: true})
Expect(err).NotTo(HaveOccurred())
Expect(up.last().Fields).To(HaveKeyWithValue("include_sounds", "true"))
Expect(res.SoundsIncluded).To(BeTrue())
Expect(res.Sounds).To(HaveLen(1))
Expect(res.Sounds[0].Label).To(Equal("Dog"))
Expect(res.Sounds[0].Confidence).To(BeNumerically("==", 0.75))
})
})
+17
View File
@@ -1,6 +1,7 @@
package main
import (
"context"
"encoding/json"
"fmt"
"sort"
@@ -123,6 +124,11 @@ func (p *ParakeetCpp) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, error
return pb.DiarizeResponse{}, status.Error(codes.Unimplemented, "parakeet-cpp: speaker profiles require a loaded speaker encoder and profile-capable library")
}
}
if req.GetIncludeSounds() {
if err := p.checkSoundEventsAvailable(); err != nil {
return pb.DiarizeResponse{}, err
}
}
if CppDiarizePCM == nil {
return pb.DiarizeResponse{}, status.Error(codes.Unimplemented,
"parakeet-cpp: loaded libparakeet.so has no diarization support (parakeet_capi_diarize_pcm missing)")
@@ -190,7 +196,18 @@ func (p *ParakeetCpp) Diarize(req *pb.DiarizeRequest) (pb.DiarizeResponse, error
segments = applyDurationFilters(segments, req.GetMinDurationOn(), req.GetMinDurationOff())
renumberDiarizeSegments(segments)
var sounds []*pb.DiarizeSound
if req.GetIncludeSounds() {
events, err := p.diarizeSoundEvents(context.Background(), pcm)
if err != nil {
return pb.DiarizeResponse{}, err
}
sounds = diarizeSoundsToProto(events)
}
return pb.DiarizeResponse{
Sounds: sounds,
SoundsIncluded: req.GetIncludeSounds(),
SpeakerProfilesJson: profiles,
Segments: segments,
NumSpeakers: distinctDiarizeSpeakers(segments),
+113
View File
@@ -0,0 +1,113 @@
package main
import (
"context"
"fmt"
"sort"
"github.com/mudler/LocalAI/pkg/grpc/grpcerrors"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)
// checkSoundEventsAvailable reports why include_sounds cannot be served, or
// nil when it can. It is checked before any audio work so a model without a
// sound companion fails fast and never returns an empty list that a client
// would read as "nothing was heard".
func (p *ParakeetCpp) checkSoundEventsAvailable() error {
if p.tagCtx == 0 {
return grpcerrors.SoundEventsUnsupported("parakeet-cpp",
"the model has no sound model; add a sound_model: companion option"+p.roleHint(componentSound, "sound_component"))
}
if CppSceneOptsDefault == nil || CppSceneStreamBegin == nil || CppSceneStreamFeedJSON == nil || CppSceneStreamFree == nil {
return grpcerrors.SoundEventsUnsupported("parakeet-cpp",
"the loaded libparakeet.so has no scene stream support (parakeet_capi_scene_stream_* missing)")
}
return nil
}
// diarizeSoundEvents runs pcm through a tagger-only scene stream and returns
// the closed sound events. It is the same stream, with the same on/off
// thresholds and minimum duration, the live path opens beside an ASR session
// (see sceneBegin), so an offline request and a live session report the same
// events for the same audio. The clip is fed in 10 s pieces with the last one
// flushing events still open at the end of the audio.
//
// Every C call runs under engineMu, held for the whole clip like
// soundStreamDrain does; the stream is freed even when a feed fails or ctx is
// cancelled.
func (p *ParakeetCpp) diarizeSoundEvents(ctx context.Context, pcm []float32) ([]sceneSoundJSON, error) {
p.engineMu.Lock()
defer p.engineMu.Unlock()
// The caller's check ran before this lock was taken; a Free() racing in
// between zeroes tagCtx under the same lock, so check again.
if p.tagCtx == 0 {
return nil, grpcerrors.ModelNotLoaded("parakeet-cpp")
}
var opts cSceneOpts
CppSceneOptsDefault(&opts)
// Only closed events are used, never per-class scores, so keep none (see
// sceneBegin for why a non-zero top_k would grow a queue).
opts.Sound.TopK = 0
stream := CppSceneStreamBegin(0, 0, p.tagCtx, &opts)
if stream == 0 {
return nil, fmt.Errorf("parakeet-cpp: sound scene stream begin failed: %s", soundLastError(p.tagCtx))
}
defer CppSceneStreamFree(stream)
var events []sceneSoundJSON
for off := 0; ; {
if ctx != nil {
if err := ctx.Err(); err != nil {
return nil, status.Error(codes.Canceled, "parakeet-cpp: sound detection cancelled")
}
}
end := off + soundFeedChunkSamples
last := end >= len(pcm)
if last {
end = len(pcm)
}
doc, err := sceneFeedLocked(stream, pcm[off:end], last)
if err != nil {
return nil, err
}
events = append(events, doc.Sounds...)
if last {
break
}
off = end
}
return events, nil
}
// diarizeSoundsToProto maps closed scene sound events to DiarizeSound, sorted
// by start (then end, then label) so the order does not depend on how the
// stream closed them. Confidence is the event's peak score. The result is
// never nil: an empty clip of sound still serialises as an empty list next to
// sounds_included=true.
func diarizeSoundsToProto(sounds []sceneSoundJSON) []*pb.DiarizeSound {
out := make([]*pb.DiarizeSound, 0, len(sounds))
for _, s := range sounds {
out = append(out, &pb.DiarizeSound{
Start: float32(s.Start),
End: float32(s.End),
Label: s.Label,
Confidence: s.Peak,
})
}
sort.SliceStable(out, func(i, j int) bool {
a, b := out[i], out[j]
if a.Start != b.Start {
return a.Start < b.Start
}
if a.End != b.End {
return a.End < b.End
}
return a.Label < b.Label
})
return out
}
@@ -0,0 +1,141 @@
package main
import (
"github.com/mudler/LocalAI/pkg/grpc/grpcerrors"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
"google.golang.org/grpc/codes"
"google.golang.org/grpc/status"
)
// These specs drive Diarize with include_sounds against stubbed scene stream
// entry points, the same seam live_test.go uses, so they run without
// libparakeet.so or model files.
var _ = Describe("ParakeetCpp.Diarize include_sounds", func() {
var (
restoreDiar func()
restore func()
pool *diarizeCstrPool
)
BeforeEach(func() {
restoreDiar = diarizeStubs()
pool = &diarizeCstrPool{}
sOpts, sBegin, sFeed, sFree := CppSceneOptsDefault, CppSceneStreamBegin, CppSceneStreamFeedJSON, CppSceneStreamFree
restore = func() {
CppSceneOptsDefault, CppSceneStreamBegin, CppSceneStreamFeedJSON, CppSceneStreamFree = sOpts, sBegin, sFeed, sFree
}
CppSceneOptsDefault = func(o *cSceneOpts) { *o = cSceneOpts{Sound: cSoundOpts{TopK: 5}} }
CppFreeString = func(uintptr) {}
CppDiarizePCM = func(uintptr, *float32, int32, int32) uintptr {
return pool.cstr(`{"speakers":8,"segments":[{"speaker":0,"start":0.00,"end":3.00}]}`)
}
})
AfterEach(func() {
restore()
restoreDiar()
})
It("rejects the request with Unimplemented and a stable code when no sound model is loaded", func() {
p := &ParakeetCpp{diarCtx: 42}
_, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1), IncludeSounds: true})
Expect(err).To(HaveOccurred())
Expect(status.Code(err)).To(Equal(codes.Unimplemented))
Expect(err.Error()).To(ContainSubstring(grpcerrors.SoundEventsUnsupportedCode))
Expect(err.Error()).To(ContainSubstring("sound_model"))
})
It("does not touch the sound path when include_sounds is off", func() {
CppSceneStreamBegin = func(uintptr, uintptr, uintptr, *cSceneOpts) uintptr {
Fail("no scene stream may start without include_sounds")
return 0
}
p := &ParakeetCpp{diarCtx: 42, tagCtx: 43}
resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(1)})
Expect(err).ToNot(HaveOccurred())
Expect(resp.SoundsIncluded).To(BeFalse())
Expect(resp.Sounds).To(BeEmpty())
})
It("returns closed events from a tagger-only scene stream, sorted, with the peak as confidence", func() {
var gotDiar, gotTag uintptr
var gotOpts cSceneOpts
CppSceneStreamBegin = func(asr, diar, tag uintptr, o *cSceneOpts) uintptr {
gotDiar, gotTag, gotOpts = diar, tag, *o
return 77
}
freed := uintptr(0)
CppSceneStreamFree = func(s uintptr) { freed = s }
feeds := 0
var lastFlags []bool
CppSceneStreamFeedJSON = func(s uintptr, _ *float32, n int32, isLast int32) uintptr {
feeds++
lastFlags = append(lastFlags, isLast == 1)
if isLast == 1 {
return pool.cstr(`{"speakers":[],"sounds":[{"index":99,"label":"Dog","start":12.5,"end":14.0,"peak":0.7}]}`)
}
return pool.cstr(`{"speakers":[],"sounds":[{"index":3,"label":"Cough","start":2.0,"end":2.5,"peak":0.91}]}`)
}
p := &ParakeetCpp{diarCtx: 42, tagCtx: 43}
// 15 s of audio is two 10 s feeds, the second flushing open events.
resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(15), IncludeSounds: true})
Expect(err).ToNot(HaveOccurred())
Expect(gotDiar).To(BeZero(), "the sound stream must not borrow the diarization model")
Expect(gotTag).To(Equal(uintptr(43)))
Expect(gotOpts.Sound.TopK).To(Equal(int32(0)))
Expect(feeds).To(Equal(2))
Expect(lastFlags).To(Equal([]bool{false, true}))
Expect(freed).To(Equal(uintptr(77)))
Expect(resp.SoundsIncluded).To(BeTrue())
Expect(resp.Sounds).To(HaveLen(2))
Expect(resp.Sounds[0].Label).To(Equal("Cough"))
Expect(resp.Sounds[0].Start).To(BeNumerically("~", 2.0, 0.001))
Expect(resp.Sounds[0].End).To(BeNumerically("~", 2.5, 0.001))
Expect(resp.Sounds[0].Confidence).To(BeNumerically("~", 0.91, 0.001))
Expect(resp.Sounds[1].Label).To(Equal("Dog"))
Expect(resp.Segments).To(HaveLen(1), "speaker segments are unaffected")
})
It("reports an empty list as included when nothing was heard", func() {
CppSceneStreamBegin = func(uintptr, uintptr, uintptr, *cSceneOpts) uintptr { return 1 }
CppSceneStreamFree = func(uintptr) {}
CppSceneStreamFeedJSON = func(uintptr, *float32, int32, int32) uintptr {
return pool.cstr(`{"speakers":[],"sounds":[]}`)
}
p := &ParakeetCpp{diarCtx: 42, tagCtx: 43}
resp, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(2), IncludeSounds: true})
Expect(err).ToNot(HaveOccurred())
Expect(resp.SoundsIncluded).To(BeTrue())
Expect(resp.Sounds).To(BeEmpty())
})
It("frees the stream and returns the error when a feed fails", func() {
CppSceneStreamBegin = func(uintptr, uintptr, uintptr, *cSceneOpts) uintptr { return 5 }
freed := false
CppSceneStreamFree = func(uintptr) { freed = true }
CppSceneStreamFeedJSON = func(uintptr, *float32, int32, int32) uintptr { return 0 }
p := &ParakeetCpp{diarCtx: 42, tagCtx: 43}
_, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(2), IncludeSounds: true})
Expect(err).To(HaveOccurred())
Expect(freed).To(BeTrue())
})
})
var _ = Describe("diarizeSoundsToProto", func() {
It("sorts by start then end then label and never returns nil", func() {
Expect(diarizeSoundsToProto(nil)).ToNot(BeNil())
out := diarizeSoundsToProto([]sceneSoundJSON{
{Label: "b", Start: 5, End: 6, Peak: 0.5},
{Label: "a", Start: 1, End: 4, Peak: 0.6},
{Label: "a", Start: 1, End: 2, Peak: 0.7},
})
Expect(out).To(HaveLen(3))
Expect(out[0].End).To(BeNumerically("~", 2, 0.001))
Expect(out[1].End).To(BeNumerically("~", 4, 0.001))
Expect(out[2].Label).To(Equal("b"))
})
})
+14 -3
View File
@@ -172,6 +172,18 @@ func (p *ParakeetCpp) sceneFeed(h sceneStreamHandle, pcm []float32, isLast bool)
return sceneFeedJSON{}, grpcerrors.ModelNotLoaded("parakeet-cpp")
}
doc, err := sceneFeedLocked(h.s, pcm, isLast)
if err != nil {
return sceneFeedJSON{}, err
}
translateNames(doc.Names, h.names)
return doc, nil
}
// sceneFeedLocked runs one scene_stream_feed_json call and decodes its
// document. The caller holds engineMu and has already checked that the
// contexts the stream borrows are still loaded.
func sceneFeedLocked(stream uintptr, pcm []float32, isLast bool) (sceneFeedJSON, error) {
var last int32
if isLast {
last = 1
@@ -180,11 +192,11 @@ func (p *ParakeetCpp) sceneFeed(h sceneStreamHandle, pcm []float32, isLast bool)
if len(pcm) > 0 {
ptr = &pcm[0]
}
ret := CppSceneStreamFeedJSON(h.s, ptr, int32(len(pcm)), last)
ret := CppSceneStreamFeedJSON(stream, ptr, int32(len(pcm)), last)
if ret == 0 {
msg := ""
if CppSceneStreamLastError != nil {
msg = CppSceneStreamLastError(h.s)
msg = CppSceneStreamLastError(stream)
}
if msg == "" {
msg = "unknown error"
@@ -197,7 +209,6 @@ func (p *ParakeetCpp) sceneFeed(h sceneStreamHandle, pcm []float32, isLast bool)
if err := json.Unmarshal([]byte(raw), &doc); err != nil {
return sceneFeedJSON{}, fmt.Errorf("parakeet-cpp: decode scene json: %w", err)
}
translateNames(doc.Names, h.names)
return doc, nil
}
+26
View File
@@ -13,6 +13,7 @@ import (
"github.com/mudler/LocalAI/core/config"
"github.com/mudler/LocalAI/core/schema"
"github.com/mudler/LocalAI/core/services/voicerecognition"
"github.com/mudler/LocalAI/pkg/grpc/grpcerrors"
grpcPkg "github.com/mudler/LocalAI/pkg/grpc"
"github.com/mudler/LocalAI/pkg/grpc/proto"
@@ -35,6 +36,9 @@ type DiarizationRequest struct {
MinDurationOff float32
IncludeText bool
IncludeSpeakerProfiles bool
// IncludeSounds asks for closed sound events (needs a sound companion on
// the model). A backend that cannot produce them must reject the request.
IncludeSounds bool
// KnownVoices are registered voices a speaker-identifying backend may use
// to name the speakers. Empty for every other backend and model.
KnownVoices []voicerecognition.KnownVoice
@@ -59,6 +63,7 @@ func (r *DiarizationRequest) toProto(threads uint32, modelIdentity string) *prot
MinDurationOff: r.MinDurationOff,
IncludeText: r.IncludeText,
IncludeSpeakerProfiles: r.IncludeSpeakerProfiles,
IncludeSounds: r.IncludeSounds,
KnownVoices: known,
}
}
@@ -97,6 +102,12 @@ func ModelDiarization(ctx context.Context, req DiarizationRequest, ml *model.Mod
if err != nil {
return nil, err
}
// A backend that ignores include_sounds returns no sounds_included mark.
// Reject here rather than hand the client an empty list it would read as
// "nothing was heard".
if req.IncludeSounds && !r.GetSoundsIncluded() {
return nil, grpcerrors.SoundEventsUnsupported(modelConfig.Backend, "the backend did not report sound events for this model")
}
out := diarizationResultFromProto(r)
if req.IncludeSpeakerProfiles {
trusted, err := speakerEncoderFromBackend(ctx, m)
@@ -172,6 +183,21 @@ func diarizationResultFromProto(r *proto.DiarizeResponse) *schema.DiarizationRes
})
}
if r.GetSoundsIncluded() {
out.Sounds = make([]schema.DiarizationSound, 0, len(r.Sounds))
for _, s := range r.Sounds {
if s == nil {
continue
}
out.Sounds = append(out.Sounds, schema.DiarizationSound{
Start: float64(s.Start),
End: float64(s.End),
Label: s.Label,
Confidence: s.Confidence,
})
}
}
out.NumSpeakers = len(order)
if out.NumSpeakers == 0 && r.NumSpeakers > 0 {
out.NumSpeakers = int(r.NumSpeakers)
+40
View File
@@ -0,0 +1,40 @@
// SPDX-License-Identifier: MIT
package backend
import (
"encoding/json"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
. "github.com/onsi/ginkgo/v2"
. "github.com/onsi/gomega"
)
var _ = Describe("diarization sound events", func() {
It("sends include_sounds to the backend only when asked", func() {
Expect((&DiarizationRequest{IncludeSounds: true}).toProto(1, "m").IncludeSounds).To(BeTrue())
Expect((&DiarizationRequest{}).toProto(1, "m").IncludeSounds).To(BeFalse())
})
It("maps proto sounds to the result and keeps an empty list when the backend heard nothing", func() {
out := diarizationResultFromProto(&pb.DiarizeResponse{
SoundsIncluded: true,
Sounds: []*pb.DiarizeSound{{Start: 1, End: 2.5, Label: "Cough", Confidence: 0.5}},
})
Expect(out.Sounds).To(HaveLen(1))
Expect(out.Sounds[0].Label).To(Equal("Cough"))
Expect(out.Sounds[0].Start).To(BeNumerically("~", 1, 1e-6))
Expect(out.Sounds[0].End).To(BeNumerically("~", 2.5, 1e-6))
Expect(out.Sounds[0].Confidence).To(BeNumerically("~", 0.5, 1e-6))
empty := diarizationResultFromProto(&pb.DiarizeResponse{SoundsIncluded: true})
raw, err := json.Marshal(empty)
Expect(err).ToNot(HaveOccurred())
Expect(string(raw)).To(ContainSubstring(`"sounds":[]`))
})
It("leaves sounds out of the payload when they were not requested", func() {
raw, err := json.Marshal(diarizationResultFromProto(&pb.DiarizeResponse{}))
Expect(err).ToNot(HaveOccurred())
Expect(string(raw)).ToNot(ContainSubstring("sounds"))
})
})
@@ -40,7 +40,7 @@ var instructionDefs = []instructionDef{
Name: "audio",
Description: "Text-to-speech, voice activity detection, transcription, speaker diarization, sound classification, and sound generation",
Tags: []string{"audio"},
Intro: "GET /v1/audio/voices lists named voices for installed TTS models and accepts an optional model filter. Diarization (/v1/audio/diarization) returns speaker-labelled time segments. Backends with native ASR-diarization (vibevoice-cpp) can also emit per-segment text via include_text=true; backends with a dedicated pipeline (sherpa-onnx + pyannote) emit segmentation only. Response formats: json (default), verbose_json (adds speakers summary + text), rttm (NIST format). Sound classification (/v1/audio/classification) returns scored AudioSet sound-event tags (audio tagging via the ced backend); top_k and threshold control the returned set.",
Intro: "GET /v1/audio/voices lists named voices for installed TTS models and accepts an optional model filter. Diarization (/v1/audio/diarization) returns speaker-labelled time segments. Backends with native ASR-diarization (vibevoice-cpp) can also emit per-segment text via include_text=true; backends with a dedicated pipeline (sherpa-onnx + pyannote) emit segmentation only. include_sounds=true adds timed sound events (start, end, label, confidence) when the model has a sound companion, otherwise 501 include_sounds_unsupported. Response formats: json (default), verbose_json (adds speakers summary + text), rttm (NIST format). Sound classification (/v1/audio/classification) returns scored AudioSet sound-event tags (audio tagging via the ced backend); top_k and threshold control the returned set.",
},
{
Name: "voice-library",
@@ -25,6 +25,7 @@ import (
"github.com/mudler/LocalAI/core/schema"
"github.com/mudler/LocalAI/core/services/voicerecognition"
grpcpkg "github.com/mudler/LocalAI/pkg/grpc"
"github.com/mudler/LocalAI/pkg/grpc/grpcerrors"
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
"github.com/mudler/LocalAI/pkg/model"
"github.com/mudler/LocalAI/pkg/system"
@@ -41,6 +42,10 @@ type profileHTTPBackend struct {
embeds int
unsupported bool
values [][]byte
// soundsSupported makes Diarize honour include_sounds with sounds; left
// false it behaves like a backend that ignores the field.
soundsSupported bool
sounds []*pb.DiarizeSound
}
func (b *profileHTTPBackend) Status(context.Context) (*pb.StatusResponse, error) {
@@ -52,7 +57,12 @@ func (b *profileHTTPBackend) Status(context.Context) (*pb.StatusResponse, error)
func (b *profileHTTPBackend) Diarize(_ context.Context, r *pb.DiarizeRequest, _ ...ggrpc.CallOption) (*pb.DiarizeResponse, error) {
b.last = r
raw, _ := json.Marshal(b.profiles)
return &pb.DiarizeResponse{Segments: []*pb.DiarizeSegment{{Speaker: "7", Start: 0, End: 3, Text: "Hello"}}, SpeakerProfilesJson: string(raw)}, nil
resp := &pb.DiarizeResponse{Segments: []*pb.DiarizeSegment{{Speaker: "7", Start: 0, End: 3, Text: "Hello"}}, SpeakerProfilesJson: string(raw)}
if r.IncludeSounds && b.soundsSupported {
resp.SoundsIncluded = true
resp.Sounds = b.sounds
}
return resp, nil
}
func (b *profileHTTPBackend) VoiceEmbed(context.Context, *pb.VoiceEmbedRequest, ...ggrpc.CallOption) (*pb.VoiceEmbedResponse, error) {
b.embeds++
@@ -372,3 +382,74 @@ func TestPortableProfilesNeverPersistInAPITraces(t *testing.T) {
time.Sleep(10 * time.Millisecond)
}
}
func TestDiarizationSoundsHTTP(t *testing.T) {
b := &profileHTTPBackend{profiles: profileFixture(), soundsSupported: true, sounds: []*pb.DiarizeSound{{Start: 1.5, End: 2.25, Label: "Dog", Confidence: 0.75}}}
e, _ := profileServer(b, false)
for _, format := range []string{"json", "verbose_json"} {
w := profileJSON(e, "/v1/audio/diarization", map[string]any{"model": "test", "file": "YXVkaW8=", "include_sounds": true, "response_format": format})
if w.Code != 200 || !b.last.IncludeSounds {
t.Fatal(format, w.Code, w.Body.String())
}
var got struct {
Sounds []schema.DiarizationSound `json:"sounds"`
}
if err := json.Unmarshal(w.Body.Bytes(), &got); err != nil {
t.Fatal(err)
}
want := schema.DiarizationSound{Start: 1.5, End: 2.25, Label: "Dog", Confidence: 0.75}
if len(got.Sounds) != 1 || got.Sounds[0] != want {
t.Fatal(format, w.Body.String())
}
}
// Multipart form field.
body := &bytes.Buffer{}
mw := multipart.NewWriter(body)
mw.WriteField("model", "test")
mw.WriteField("include_sounds", "true")
f, _ := mw.CreateFormFile("file", "sample.wav")
f.Write([]byte("audio"))
mw.Close()
r := httptest.NewRequest("POST", "/v1/audio/diarization", body)
r.Header.Set("Content-Type", mw.FormDataContentType())
w := httptest.NewRecorder()
e.ServeHTTP(w, r)
if w.Code != 200 || !bytes.Contains(w.Body.Bytes(), []byte(`"label":"Dog"`)) {
t.Fatal(w.Code, w.Body.String())
}
// Not requested: the field is absent and the backend is told so.
w = profileJSON(e, "/v1/audio/diarization", map[string]any{"model": "test", "file": "YXVkaW8="})
if w.Code != 200 || b.last.IncludeSounds || bytes.Contains(w.Body.Bytes(), []byte(`"sounds"`)) {
t.Fatal(w.Code, w.Body.String())
}
// Requested and heard nothing: an empty list, not an absent field.
b.sounds = nil
w = profileJSON(e, "/v1/audio/diarization", map[string]any{"model": "test", "file": "YXVkaW8=", "include_sounds": true})
if w.Code != 200 || !bytes.Contains(w.Body.Bytes(), []byte(`"sounds":[]`)) {
t.Fatal(w.Code, w.Body.String())
}
// RTTM cannot carry sounds: rejected before the backend runs.
b.last = nil
w = profileJSON(e, "/v1/audio/diarization", map[string]any{"model": "test", "file": "YXVkaW8=", "include_sounds": true, "response_format": "rttm"})
if w.Code != 400 || b.last != nil {
t.Fatal(w.Code, w.Body.String())
}
}
func TestDiarizationSoundsUnsupported(t *testing.T) {
// A backend that ignores include_sounds ends in a 501 carrying the stable
// code, never a 200 with an empty list.
e, _ := profileServer(&profileHTTPBackend{profiles: profileFixture()}, false)
w := profileJSON(e, "/v1/audio/diarization", map[string]any{"model": "test", "file": "YXVkaW8=", "include_sounds": true})
if w.Code != 501 || !bytes.Contains(w.Body.Bytes(), []byte(grpcerrors.SoundEventsUnsupportedCode)) {
t.Fatal(w.Code, w.Body.String())
}
if bytes.Contains(w.Body.Bytes(), []byte(`"sounds"`)) {
t.Fatal("unsupported response carried a sounds field", w.Body.String())
}
}
+6 -1
View File
@@ -41,7 +41,7 @@ import (
// (NIST RTTM, the standard interchange format used by pyannote/dscore).
//
// @Summary Identify speakers in audio (who spoke when).
// @Description JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles and response_format. Profiles require voice-recognition permission and json or verbose_json; unsupported backends return 501.
// @Description JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles, include_sounds and response_format. Profiles require voice-recognition permission and json or verbose_json; unsupported backends return 501.
// @Tags audio
// @accept multipart/form-data,json
// @Param model formData string true "model"
@@ -55,6 +55,7 @@ import (
// @Param language formData string false "audio language hint (only meaningful for backends that bundle ASR)"
// @Param include_speaker_profiles formData boolean false "export portable biometric profiles (voice-recognition permission; JSON formats only)"
// @Param include_text formData boolean false "include per-segment transcript when the backend supports it"
// @Param include_sounds formData boolean false "include closed sound events (start, end, label, confidence) when the model has a sound_model companion; otherwise 501 include_sounds_unsupported (JSON formats only)"
// @Param response_format formData string false "json (default), verbose_json, or rttm"
// @Success 200 {object} schema.DiarizationResult
// @Router /v1/audio/diarization [post]
@@ -74,6 +75,7 @@ func DiarizationEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, ap
Language: input.Language,
IncludeText: parseFormBool(c, "include_text", input.IncludeText),
IncludeSpeakerProfiles: parseFormBool(c, "include_speaker_profiles", input.IncludeSpeakerProfiles),
IncludeSounds: parseFormBool(c, "include_sounds", input.IncludeSounds),
}
if req.IncludeSpeakerProfiles {
var db *gorm.DB
@@ -118,6 +120,9 @@ func DiarizationEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, ap
if req.IncludeSpeakerProfiles && responseFormat == schema.DiarizationResponseFormatRTTM {
return echo.NewHTTPError(http.StatusBadRequest, "speaker_profiles requires json or verbose_json")
}
if req.IncludeSounds && responseFormat == schema.DiarizationResponseFormatRTTM {
return echo.NewHTTPError(http.StatusBadRequest, "include_sounds requires json or verbose_json")
}
var sourceName = "audio.wav"
var reader io.ReadCloser
if strings.HasPrefix(c.Request().Header.Get(echo.HeaderContentType), echo.MIMEApplicationJSON) {
+14
View File
@@ -30,6 +30,16 @@ type DiarizationSpeaker struct {
SegmentCount int `json:"segment_count"`
}
// DiarizationSound is one closed sound event found over the whole clip. Times
// are in seconds. Label is an AudioSet class name and Confidence the peak score
// of the event (0 to 1).
type DiarizationSound struct {
Start float64 `json:"start"`
End float64 `json:"end"`
Label string `json:"label"`
Confidence float32 `json:"confidence"`
}
// DiarizationResult is the JSON payload returned by /v1/audio/diarization.
// Speakers and segment text are omitted when empty so the default `json`
// response stays minimal; verbose_json keeps both populated.
@@ -41,6 +51,10 @@ type DiarizationResult struct {
NumSpeakers int `json:"num_speakers"`
Segments []DiarizationSegment `json:"segments"`
Speakers []DiarizationSpeaker `json:"speakers,omitempty"`
// Sounds is present only when the request set include_sounds. An empty
// list then means the model ran and heard no event; omitzero keeps a nil
// list (not requested) out of the payload while an empty one stays.
Sounds []DiarizationSound `json:"sounds,omitzero"`
}
// DiarizationResponseFormatType mirrors transcription's response_format
+1
View File
@@ -189,6 +189,7 @@ type JsonSchema struct {
type OpenAIRequest struct {
IncludeSpeakerProfiles bool `json:"include_speaker_profiles,omitempty"`
IncludeText bool `json:"include_text,omitempty"`
IncludeSounds bool `json:"include_sounds,omitempty"`
PredictionOptions
Context context.Context `json:"-"`
@@ -9,7 +9,7 @@ Sound-event classification (audio tagging) answers the question **"what am I hea
LocalAI exposes this through the `/v1/audio/classification` endpoint, modelled after `/v1/audio/transcriptions`. The reference backend is **[ced.cpp](https://github.com/localai-org/ced.cpp)** (CED, a 527-class AudioSet tagger), a small ViT over a log-mel spectrogram ported to ggml with full PyTorch parity. Apache-2.0 weights are redistributable as GGUF.
**[parakeet.cpp](https://github.com/mudler/parakeet.cpp)** can also load a CED model (through `third_party/ced.cpp`) and serve `/v1/audio/classification` from the same backend used for ASR and diarization. It scores the clip in 10 s windows and averages each class's score across the windows before sorting and applying `top_k`/`threshold` - CED's own method for clips longer than one window. Install `parakeet-cpp-ced-tiny` or `parakeet-cpp-ced-base` from the gallery, or point `parameters.model` at a CED GGUF under `backend: parakeet-cpp`. A parakeet-cpp ASR model can also point `sound_model` at a CED GGUF to add live sound events during realtime transcription - see [Realtime API]({{% relref "openai-realtime" %}}).
**[parakeet.cpp](https://github.com/mudler/parakeet.cpp)** can also load a CED model (through `third_party/ced.cpp`) and serve `/v1/audio/classification` from the same backend used for ASR and diarization. It scores the clip in 10 s windows and averages each class's score across the windows before sorting and applying `top_k`/`threshold` - CED's own method for clips longer than one window. Install `parakeet-cpp-ced-tiny` or `parakeet-cpp-ced-base` from the gallery, or point `parameters.model` at a CED GGUF under `backend: parakeet-cpp`. A parakeet-cpp ASR model can also point `sound_model` at a CED GGUF to add live sound events during realtime transcription - see [Realtime API]({{% relref "openai-realtime" %}}) - and to add timed sound events to an offline diarization call with `include_sounds=true` - see [Speaker Diarization]({{% relref "audio-diarization" %}}#sound-events). Such a model also answers `/v1/audio/classification` itself when it declares the `sound_classification` usecase, as the gallery entry `parakeet-cpp-multilingual-diarization-speakers-sounds` does.
Because classification is exposed as a regular OpenAI-style endpoint, any HTTP client works - there is no Python dependency on the consumer side.
@@ -42,6 +42,7 @@ Content-Type: multipart/form-data
| `min_duration_off` | float | merge gaps shorter than this many seconds |
| `language` | string | only meaningful for backends that bundle ASR (e.g. vibevoice) |
| `include_text` | bool | when the backend can emit per-segment transcript for free, populate it |
| `include_sounds` | bool | add the closed sound events of the clip as `sounds`. Needs a model with a sound companion; see [Sound events](#sound-events). Default `false` |
| `response_format` | string | `json` (default), `verbose_json`, or `rttm` |
### Response - `json` (default)
@@ -103,6 +104,44 @@ With a parakeet-cpp model that has a `speaker_model:` (or a `speaker_component:`
}
```
### Sound events
With `include_sounds=true` the response gains a `sounds` array: the sound events (AudioSet labels such as `Dog`, `Applause`, `Cough`) found anywhere in the clip, in the same call that returns the speakers, the text and the voice prints. Each item is `{start, end, label, confidence}`: `start` and `end` are in seconds from the start of the audio, and `confidence` is the peak score the tagger reached while the event lasted (0 to 1). Items are sorted by `start`. Both `json` and `verbose_json` carry the array; `rttm` has no place for it and returns 400.
```bash
curl http://localhost:8080/v1/audio/diarization \
-F model=parakeet-cpp-multilingual-diarization-speakers-sounds \
-F file=@meeting.wav \
-F include_text=true -F include_speaker_profiles=true -F include_sounds=true \
-F response_format=verbose_json
```
```json
{
"task": "diarize",
"duration": 31.2,
"language": "en",
"num_speakers": 2,
"segments": [
{"id": 0, "speaker": "SPEAKER_00", "label": "0", "start": 0.0, "end": 6.4, "text": "Good morning, everyone."},
{"id": 1, "speaker": "SPEAKER_01", "label": "1", "start": 6.8, "end": 11.2, "text": "Morning."}
],
"speakers": [
{"id": "SPEAKER_00", "label": "0", "total_speech_duration": 6.4, "segment_count": 1},
{"id": "SPEAKER_01", "label": "1", "total_speech_duration": 4.4, "segment_count": 1}
],
"speaker_profiles": {"version": 1, "encoder": {"identity": "sha256:...", "dimension": 256}, "speakers": ["..."]},
"sounds": [
{"start": 5.5, "end": 6.75, "label": "Cough", "confidence": 0.91},
{"start": 12.0, "end": 14.5, "label": "Applause", "confidence": 0.68}
]
}
```
An empty `sounds` array means the sound model ran and found no event. The field is absent when `include_sounds` is not set. The events come from the same sound stream, with the same on and off thresholds, that a [realtime session]({{% relref "openai-realtime" %}}) runs, so a clip gives the same events offline and live. The thresholds are not request fields.
The model needs a sound companion, a `sound_model:` option pointing at a CED GGUF (the gallery entry `parakeet-cpp-multilingual-diarization-speakers-sounds` has one). Without it the request fails with HTTP 501 and a message that starts with the stable code `include_sounds_unsupported`; the same happens for a backend that cannot report sound events. LocalAI never answers with an empty list in place of that error. The speaker segments, text and voice prints are independent of the sound model and cost nothing extra when `include_sounds` is off.
### Response - `rttm`
NIST RTTM, the standard interchange format used by `pyannote.metrics` / `dscore`:
@@ -191,11 +230,14 @@ Choose an existing gallery entry for the output you need:
| Speaker turns only | `parakeet-cpp-nemotron-3-diarization` | Default options |
| Speaker turns and transcript | `parakeet-cpp-nemotron-3-diarization-asr` | `include_text=true`, `response_format=verbose_json` |
| Speaker turns, transcript, and identification | `parakeet-cpp-nemotron-3-diarization-asr-speakers` | Same transcript options; explicitly enroll voices for names |
| Multilingual transcript, speakers, voice prints and sound events in one call | `parakeet-cpp-multilingual-diarization-speakers-sounds` | `include_text=true`, `include_speaker_profiles=true`, `include_sounds=true`, `response_format=verbose_json` |
The complete `-asr-speakers` entry downloads Nemotron-3-Diarization, Parakeet TDT+CTC 110M ASR, and the WeSpeaker ResNet34 speaker encoder.
It configures both `asr_model` and `speaker_model`; no custom gallery configuration is needed.
See [Remember speakers in the Web UI](#remember-speakers-in-the-web-ui) for installation and enrollment.
The `parakeet-cpp-multilingual-diarization-speakers-sounds` entry downloads Parakeet TDT 0.6B v3 (25 European languages), Nemotron-3-Diarization, the WeSpeaker ResNet34 speaker encoder and CED-Tiny, and sets `diarization_model`, `speaker_model` and `sound_model`. It answers the whole request above from one model name. See [Sound events](#sound-events).
The entries `parakeet-cpp-bundle-small` and `parakeet-cpp-bundle-standard` hold Nemotron-3-Diarization, an ASR model and the WeSpeaker speaker encoder in one file (`diar_component:diar` and `speaker_component:voice`), so one install serves the transcript and the identification options above. See [Bundle GGUF files]({{% relref "audio-to-text" %}}#bundle-gguf-files-several-models-in-one-file).
For manual configuration, this example pairs Sortformer with ASR:
+1 -1
View File
@@ -200,7 +200,7 @@ The same backend also serves the `/v1/audio/diarization` and `/v1/audio/classifi
|---|---|---|
| `asr_model:<path>` | a diarization model | `include_text` on `/v1/audio/diarization` |
| `diarization_model:<path>` | an ASR model | a `speaker` on transcript segments (and words), and speaker segments during realtime live transcription |
| `sound_model:<path>` | an ASR model | sound events during realtime live transcription |
| `sound_model:<path>` | an ASR or diarization model | sound events during realtime live transcription, `sounds` on `/v1/audio/diarization` with `include_sounds=true`, and `/v1/audio/classification` on the loaded model |
| `diarization_latency:<model\|low\|very_low\|ultra_low>` | a model with a diarization companion | latency mode for the live speaker stream; default `low` |
| `speaker_model:<path>` | a model with a diarization model | names registered speakers; a bundle can use `speaker_component:<name>` instead (see [Bundle GGUF files](#bundle-gguf-files-several-models-in-one-file)) (see [Voice Recognition]({{% relref "voice-recognition" %}}#naming-speakers-in-diarization-and-live-transcription)) |
| `speaker_tag:<tag>` | a model with `speaker_component` | extra encoder tag for registered voices that carry only a file-name tag (see [Voice Recognition]({{% relref "voice-recognition" %}}#naming-speakers-from-a-bundle)) |
+1 -1
View File
@@ -230,7 +230,7 @@ pipeline:
#### Choosing the sound model
Both scene models ship with CED-Tiny, the cheapest to run all the time. `parakeet-cpp-realtime-scene-base` and `parakeet-cpp-realtime-scene-tdt-base` are the same pipelines with CED-Base (86M, the largest CED), which tags sounds more confidently. Any CED GGUF from [`mudler/ced-gguf`](https://huggingface.co/mudler/ced-gguf) (tiny, mini, small, base) works as `sound_model`. Measured on CPU (Ryzen 9 9950X3D) over a 37 s clip with two speakers and a rooster, as a fraction of real time:
Both scene models ship with CED-Tiny, the cheapest to run all the time. `parakeet-cpp-realtime-scene-base` and `parakeet-cpp-realtime-scene-tdt-base` are the same pipelines with CED-Base (86M, the largest CED), which tags sounds more confidently. `parakeet-cpp-multilingual-diarization-speakers-sounds` loads the same TDT, Nemotron-3-Diarization and CED-Tiny trio plus the WeSpeaker encoder, and is meant for offline `/v1/audio/diarization` calls with `include_sounds=true` (see [Speaker Diarization]({{% relref "audio-diarization" %}}#sound-events)). Any CED GGUF from [`mudler/ced-gguf`](https://huggingface.co/mudler/ced-gguf) (tiny, mini, small, base) works as `sound_model`. Measured on CPU (Ryzen 9 9950X3D) over a 37 s clip with two speakers and a rooster, as a fraction of real time:
| | CED-Tiny | CED-Base |
|---|---|---|
+3 -1
View File
@@ -256,7 +256,9 @@ this, speakers only carry labels such as `SPEAKER_00`.
2. Install one of the gallery models that loads the same encoder:
`parakeet-cpp-nemotron-3-diarization-speakers` (diarization),
`parakeet-cpp-nemotron-3-diarization-asr-speakers` (diarization with
`include_text`) or `parakeet-cpp-realtime-scene-speakers` (live
`include_text`), `parakeet-cpp-multilingual-diarization-speakers-sounds`
(multilingual diarization with `include_text`, voice prints and sound events)
or `parakeet-cpp-realtime-scene-speakers` (live
transcription). Each one adds
`speaker_model:voice-detect-wespeaker-resnet34.gguf` to a
parakeet-cpp model config.
+66
View File
@@ -54538,6 +54538,72 @@
- filename: voice-detect-wespeaker-resnet34.gguf
uri: https://huggingface.co/mudler/voice-detect-gguf/resolve/main/wespeaker-resnet34-voxceleb.gguf
sha256: 72040372494eafec299836bc1977cfc13c603cb486674ed59b0f4c03758d29da
- name: parakeet-cpp-multilingual-diarization-speakers-sounds
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
- https://huggingface.co/mudler/parakeet-cpp-gguf
- https://huggingface.co/mudler/voice-detect-gguf
- https://huggingface.co/mudler/ced-gguf
- https://huggingface.co/nvidia/parakeet-tdt-0.6b-v3
- https://huggingface.co/nvidia/Nemotron-3-Diarization
- https://huggingface.co/mispeech/ced-tiny
- https://github.com/mudler/parakeet.cpp
description: |
Parakeet TDT 0.6B v3 (multilingual, 25 European languages) paired with
Nemotron-3-Diarization (Sortformer) through the diarization_model option,
WeSpeaker ResNet34 through the speaker_model option and CED-Tiny through
the sound_model option, all GGUF for the parakeet-cpp backend (C++/ggml
port of NVIDIA NeMo). One call to /v1/audio/diarization with include_text,
include_speaker_profiles and include_sounds returns the speaker turns, the
text of each turn in the spoken language, one voice-print embedding per
speaker and the sound events (AudioSet labels with start and end times),
so a client does not need separate diarization, transcription and sound
tagging calls. Also serves /v1/audio/transcriptions and
/v1/audio/classification. License per model: transcription model
CC-BY-4.0, diarization model OpenMDW-1.1, WeSpeaker encoder CC-BY-4.0,
CED-Tiny Apache-2.0.
license: cc-by-4.0
tags:
- parakeet
- parakeet-cpp
- nemotron
- sortformer
- ced
- asr
- diarization
- speaker-diarization
- sound-classification
- speech-recognition
- multilingual
- stt
- gguf
- ggml
overrides:
backend: parakeet-cpp
known_usecases:
- transcript
- diarization
- sound_classification
name: parakeet-cpp-multilingual-diarization-speakers-sounds
options:
- diarization_model:parakeet-cpp/nemotron-3-diarization-q8_0.gguf
- speaker_model:voice-detect-wespeaker-resnet34.gguf
- sound_model:parakeet-cpp/ced-tiny-q8_0.gguf
parameters:
model: parakeet-cpp/tdt-0.6b-v3-f16.gguf
files:
- filename: parakeet-cpp/tdt-0.6b-v3-f16.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/tdt-0.6b-v3-f16.gguf
sha256: 8ba47343e1e919895aca90e099150a01ed203ee0942d8ed31e27295efc5abb22
- filename: parakeet-cpp/nemotron-3-diarization-q8_0.gguf
uri: huggingface://mudler/parakeet-cpp-gguf/nemotron-3-diarization-q8_0.gguf
sha256: 76c5bb1fb20d82706142ad32769b7ab496d2458489473a000fd7074c52ceec22
- filename: voice-detect-wespeaker-resnet34.gguf
uri: https://huggingface.co/mudler/voice-detect-gguf/resolve/main/wespeaker-resnet34-voxceleb.gguf
sha256: 72040372494eafec299836bc1977cfc13c603cb486674ed59b0f4c03758d29da
- filename: parakeet-cpp/ced-tiny-q8_0.gguf
uri: huggingface://mudler/ced-gguf/ced-tiny-q8_0.gguf
sha256: 48bee4e2fc3cc85d7806e03471db24e77fda6c2a2e81ffe9ef67caebaf2bd674
- name: parakeet-cpp-realtime-scene-speakers
url: github:mudler/LocalAI/gallery/virtual.yaml@master
urls:
+15
View File
@@ -125,3 +125,18 @@ func IsUnimplemented(err error) bool {
func StreamTranscriptionUnsupported(backend, reason string) error {
return status.Errorf(codes.Unimplemented, "%s: streaming transcription unsupported: %s", backend, reason)
}
// SoundEventsUnsupportedCode is the stable code carried in the message of the
// error a diarization request gets when it asks for include_sounds and the
// model cannot produce sound events. Clients match on this string, so it is
// part of the API contract.
const SoundEventsUnsupportedCode = "include_sounds_unsupported"
// SoundEventsUnsupported returns the canonical error a backend returns when a
// diarization request sets include_sounds but the loaded model has no sound
// (CED) companion. It carries codes.Unimplemented, which the HTTP layer maps to
// 501, so the caller learns the capability is missing instead of reading an
// empty sound list as "nothing was heard".
func SoundEventsUnsupported(backend, reason string) error {
return status.Errorf(codes.Unimplemented, "%s: %s: %s", backend, SoundEventsUnsupportedCode, reason)
}
+9
View File
@@ -106,3 +106,12 @@ var _ = Describe("grpcerrors", func() {
Expect(grpcerrors.IsModelNotLoaded(err)).To(BeFalse())
})
})
var _ = Describe("SoundEventsUnsupported", func() {
It("is Unimplemented and carries the stable code", func() {
err := grpcerrors.SoundEventsUnsupported("parakeet-cpp", "no sound model")
Expect(status.Code(err)).To(Equal(codes.Unimplemented))
Expect(err.Error()).To(ContainSubstring("include_sounds_unsupported"))
Expect(grpcerrors.SoundEventsUnsupportedCode).To(Equal("include_sounds_unsupported"))
})
})
+34 -1
View File
@@ -2909,7 +2909,7 @@ const docTemplate = `{
},
"/v1/audio/diarization": {
"post": {
"description": "JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles and response_format. Profiles require voice-recognition permission and json or verbose_json; unsupported backends return 501.",
"description": "JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles, include_sounds and response_format. Profiles require voice-recognition permission and json or verbose_json; unsupported backends return 501.",
"consumes": [
"multipart/form-data",
"application/json"
@@ -2987,6 +2987,12 @@ const docTemplate = `{
"name": "include_text",
"in": "formData"
},
{
"type": "boolean",
"description": "include closed sound events (start, end, label, confidence) when the model has a sound_model companion; otherwise 501 include_sounds_unsupported (JSON formats only)",
"name": "include_sounds",
"in": "formData"
},
{
"type": "string",
"description": "json (default), verbose_json, or rttm",
@@ -5861,6 +5867,13 @@ const docTemplate = `{
"$ref": "#/definitions/schema.DiarizationSegment"
}
},
"sounds": {
"description": "Sounds is present only when the request set include_sounds. An empty\nlist then means the model ran and heard no event; omitzero keeps a nil\nlist (not requested) out of the payload while an empty one stays.",
"type": "array",
"items": {
"$ref": "#/definitions/schema.DiarizationSound"
}
},
"speaker_profiles": {
"$ref": "#/definitions/schema.SpeakerProfiles"
},
@@ -5905,6 +5918,23 @@ const docTemplate = `{
}
}
},
"schema.DiarizationSound": {
"type": "object",
"properties": {
"confidence": {
"type": "number"
},
"end": {
"type": "number"
},
"label": {
"type": "string"
},
"start": {
"type": "number"
}
}
},
"schema.DiarizationSpeaker": {
"type": "object",
"properties": {
@@ -7528,6 +7558,9 @@ const docTemplate = `{
"ignore_eos": {
"type": "boolean"
},
"include_sounds": {
"type": "boolean"
},
"include_speaker_profiles": {
"type": "boolean"
},
+34 -1
View File
@@ -2906,7 +2906,7 @@
},
"/v1/audio/diarization": {
"post": {
"description": "JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles and response_format. Profiles require voice-recognition permission and json or verbose_json; unsupported backends return 501.",
"description": "JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles, include_sounds and response_format. Profiles require voice-recognition permission and json or verbose_json; unsupported backends return 501.",
"consumes": [
"multipart/form-data",
"application/json"
@@ -2984,6 +2984,12 @@
"name": "include_text",
"in": "formData"
},
{
"type": "boolean",
"description": "include closed sound events (start, end, label, confidence) when the model has a sound_model companion; otherwise 501 include_sounds_unsupported (JSON formats only)",
"name": "include_sounds",
"in": "formData"
},
{
"type": "string",
"description": "json (default), verbose_json, or rttm",
@@ -5858,6 +5864,13 @@
"$ref": "#/definitions/schema.DiarizationSegment"
}
},
"sounds": {
"description": "Sounds is present only when the request set include_sounds. An empty\nlist then means the model ran and heard no event; omitzero keeps a nil\nlist (not requested) out of the payload while an empty one stays.",
"type": "array",
"items": {
"$ref": "#/definitions/schema.DiarizationSound"
}
},
"speaker_profiles": {
"$ref": "#/definitions/schema.SpeakerProfiles"
},
@@ -5902,6 +5915,23 @@
}
}
},
"schema.DiarizationSound": {
"type": "object",
"properties": {
"confidence": {
"type": "number"
},
"end": {
"type": "number"
},
"label": {
"type": "string"
},
"start": {
"type": "number"
}
}
},
"schema.DiarizationSpeaker": {
"type": "object",
"properties": {
@@ -7525,6 +7555,9 @@
"ignore_eos": {
"type": "boolean"
},
"include_sounds": {
"type": "boolean"
},
"include_speaker_profiles": {
"type": "boolean"
},
+30 -3
View File
@@ -1078,6 +1078,14 @@ definitions:
items:
$ref: '#/definitions/schema.DiarizationSegment'
type: array
sounds:
description: |-
Sounds is present only when the request set include_sounds. An empty
list then means the model ran and heard no event; omitzero keeps a nil
list (not requested) out of the payload while an empty one stays.
items:
$ref: '#/definitions/schema.DiarizationSound'
type: array
speaker_profiles:
$ref: '#/definitions/schema.SpeakerProfiles'
speakers:
@@ -1110,6 +1118,17 @@ definitions:
text:
type: string
type: object
schema.DiarizationSound:
properties:
confidence:
type: number
end:
type: number
label:
type: string
start:
type: number
type: object
schema.DiarizationSpeaker:
properties:
id:
@@ -2226,6 +2245,8 @@ definitions:
$ref: '#/definitions/functions.JSONFunctionStructure'
ignore_eos:
type: boolean
include_sounds:
type: boolean
include_speaker_profiles:
type: boolean
include_text:
@@ -5413,9 +5434,9 @@ paths:
consumes:
- multipart/form-data
- application/json
description: JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles
and response_format. Profiles require voice-recognition permission and json
or verbose_json; unsupported backends return 501.
description: JSON accepts model, file (raw base64 audio), include_text, include_speaker_profiles,
include_sounds and response_format. Profiles require voice-recognition permission
and json or verbose_json; unsupported backends return 501.
parameters:
- description: model
in: formData
@@ -5465,6 +5486,12 @@ paths:
in: formData
name: include_text
type: boolean
- description: include closed sound events (start, end, label, confidence) when
the model has a sound_model companion; otherwise 501 include_sounds_unsupported
(JSON formats only)
in: formData
name: include_sounds
type: boolean
- description: json (default), verbose_json, or rttm
in: formData
name: response_format
+11 -2
View File
@@ -975,7 +975,7 @@ func (m *MockBackend) Diarize(ctx context.Context, in *pb.DiarizeRequest) (*pb.D
}
return out
}
return &pb.DiarizeResponse{
resp := &pb.DiarizeResponse{
Segments: []*pb.DiarizeSegment{
seg(0.0, 1.0, "5", "hello there"),
seg(1.0, 2.0, "2", "general kenobi"),
@@ -984,7 +984,16 @@ func (m *MockBackend) Diarize(ctx context.Context, in *pb.DiarizeRequest) (*pb.D
NumSpeakers: 2,
Duration: 3.5,
Language: in.Language,
}, nil
}
// IncludeSounds gates the sound events; the mock always has a sound model.
if in.IncludeSounds {
resp.SoundsIncluded = true
resp.Sounds = []*pb.DiarizeSound{
{Start: 0.5, End: 1.25, Label: "Door", Confidence: 0.8},
{Start: 2.0, End: 3.0, Label: "Applause", Confidence: 0.6},
}
}
return resp, nil
}
func (m *MockBackend) AudioEncode(ctx context.Context, in *pb.AudioEncodeRequest) (*pb.AudioEncodeResult, error) {
+22
View File
@@ -406,6 +406,28 @@ var _ = Describe("Mock Backend E2E Tests", Label("MockBackend"), func() {
Expect(segs[1].(map[string]any)["text"]).To(Equal("general kenobi"))
})
It("returns closed sound events only when include_sounds is set", func() {
resp, data := postDiarize(map[string]string{"response_format": "verbose_json", "include_sounds": "true"})
Expect(resp.StatusCode).To(Equal(http.StatusOK))
var got map[string]any
Expect(json.Unmarshal(data, &got)).To(Succeed())
sounds, ok := got["sounds"].([]any)
Expect(ok).To(BeTrue(), "include_sounds must add a sounds array")
Expect(sounds).To(HaveLen(2))
first := sounds[0].(map[string]any)
Expect(first["label"]).To(Equal("Door"))
Expect(first["start"]).To(BeNumerically("~", 0.5, 0.001))
Expect(first["end"]).To(BeNumerically("~", 1.25, 0.001))
Expect(first["confidence"]).To(BeNumerically("~", 0.8, 0.001))
resp, data = postDiarize(map[string]string{"response_format": "verbose_json"})
Expect(resp.StatusCode).To(Equal(http.StatusOK))
Expect(string(data)).ToNot(ContainSubstring(`"sounds"`))
resp, _ = postDiarize(map[string]string{"response_format": "rttm", "include_sounds": "true"})
Expect(resp.StatusCode).To(Equal(http.StatusBadRequest))
})
It("rttm response_format returns NIST RTTM rows", func() {
resp, data := postDiarize(map[string]string{"response_format": "rttm"})
Expect(resp.StatusCode).To(Equal(http.StatusOK))