mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-10 15:52:29 -04:00
* feat(diarization): return sound events with include_sounds
A client that wants text, speakers, voice prints and sound events had to
make a diarization call and a separate sound call. Add an include_sounds
request field to /v1/audio/diarization that adds a sounds array of closed
events {start, end, label, confidence}, in seconds.
The parakeet-cpp backend runs a tagger-only scene stream over the clip,
the same stream and thresholds the live path uses, so a clip gives the
same events offline and live. A model with no sound_model companion, or a
backend that does not report sound events, fails with 501 and the stable
code include_sounds_unsupported instead of an empty list. The proto
carries sounds_included so an empty list still means "nothing heard".
The localai-proxy backend forwards the field. Swagger, docs and the
e2e mock backend are updated.
Assisted-by: Claude:claude-sonnet-5-5 [protoc swag go]
* feat(gallery): add parakeet-cpp-multilingual-diarization-speakers-sounds
Same as parakeet-cpp-multilingual-diarization-speakers (TDT 0.6B v3,
Nemotron-3-Diarization, WeSpeaker) plus a CED-Tiny sound_model, so one
model name serves /v1/audio/diarization with include_text,
include_speaker_profiles and include_sounds. It declares the
sound_classification usecase like the realtime scene entries.
Assisted-by: Claude:claude-sonnet-5-5
---------
Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
70 lines
3.1 KiB
Go
70 lines
3.1 KiB
Go
package schema
|
|
|
|
// DiarizationSegment is one continuous span of speech attributed to a
|
|
// single speaker. Times are in seconds. Speaker is the normalized label
|
|
// (SPEAKER_NN, zero-padded, stable across segments); Label preserves the
|
|
// raw backend-emitted identifier for clients that already track their
|
|
// own speaker dictionary.
|
|
type DiarizationSegment struct {
|
|
Id int `json:"id"`
|
|
Speaker string `json:"speaker"`
|
|
Label string `json:"label,omitempty"`
|
|
Start float64 `json:"start"`
|
|
End float64 `json:"end"`
|
|
Text string `json:"text,omitempty"`
|
|
// Name is the registered speaker this segment was matched to, and NameScore
|
|
// the cosine similarity of the match. Both are omitted when the backend did
|
|
// not identify the speaker. Speaker stays the normalized SPEAKER_NN label.
|
|
Name string `json:"name,omitempty"`
|
|
NameScore float32 `json:"name_score,omitempty"`
|
|
}
|
|
|
|
// DiarizationSpeaker summarizes one speaker across the whole audio so
|
|
// clients can build per-speaker UIs (timeline strips, talk-time charts)
|
|
// without re-aggregating the segment list.
|
|
type DiarizationSpeaker struct {
|
|
Id string `json:"id"`
|
|
Label string `json:"label,omitempty"`
|
|
Name string `json:"name,omitempty"`
|
|
TotalSpeechDuration float64 `json:"total_speech_duration"`
|
|
SegmentCount int `json:"segment_count"`
|
|
}
|
|
|
|
// DiarizationSound is one closed sound event found over the whole clip. Times
|
|
// are in seconds. Label is an AudioSet class name and Confidence the peak score
|
|
// of the event (0 to 1).
|
|
type DiarizationSound struct {
|
|
Start float64 `json:"start"`
|
|
End float64 `json:"end"`
|
|
Label string `json:"label"`
|
|
Confidence float32 `json:"confidence"`
|
|
}
|
|
|
|
// DiarizationResult is the JSON payload returned by /v1/audio/diarization.
|
|
// Speakers and segment text are omitted when empty so the default `json`
|
|
// response stays minimal; verbose_json keeps both populated.
|
|
type DiarizationResult struct {
|
|
SpeakerProfiles *SpeakerProfiles `json:"speaker_profiles,omitempty"`
|
|
Task string `json:"task"`
|
|
Duration float64 `json:"duration,omitempty"`
|
|
Language string `json:"language,omitempty"`
|
|
NumSpeakers int `json:"num_speakers"`
|
|
Segments []DiarizationSegment `json:"segments"`
|
|
Speakers []DiarizationSpeaker `json:"speakers,omitempty"`
|
|
// Sounds is present only when the request set include_sounds. An empty
|
|
// list then means the model ran and heard no event; omitzero keeps a nil
|
|
// list (not requested) out of the payload while an empty one stays.
|
|
Sounds []DiarizationSound `json:"sounds,omitzero"`
|
|
}
|
|
|
|
// DiarizationResponseFormatType mirrors transcription's response_format
|
|
// pattern: json (default, no per-segment text), verbose_json (adds
|
|
// speakers summary + text when available), and rttm (NIST RTTM rows).
|
|
type DiarizationResponseFormatType string
|
|
|
|
const (
|
|
DiarizationResponseFormatJson DiarizationResponseFormatType = "json"
|
|
DiarizationResponseFormatJsonVerbose DiarizationResponseFormatType = "verbose_json"
|
|
DiarizationResponseFormatRTTM DiarizationResponseFormatType = "rttm"
|
|
)
|