Files
LocalAI/core/schema/speaker_profiles.go
T
mudler-agentandEttore Di Giacinto 79a7631cc5 feat(parakeet-cpp): encoder fingerprint for speaker naming, VAD trim and word filter options, pin bump (#12491)
* chore(parakeet-cpp): bump parakeet.cpp to 2de154c

Brings in the speaker registry encoder fingerprint, the VAD segment trim
and the opt-in word filter, a fix for a per-call thread count that stayed
set on the process-wide backend after a Silero VAD pass, and bundle
components loaded from memory.

Assisted-by: Claude:claude-sonnet-5-5 [Claude Code]

* feat(parakeet-cpp): encoder fingerprint for speaker naming, vad_trim and guard_* options

Speaker naming. A registered voice now carries the encoder that made it:
the embedding family (voicedetect:<arch>:<name>:<dim>) and the sha256 of
the weights. The backend reports the family of the loaded speaker model in
Status, voice enrollment from speaker_profiles stores it as encoder_family
(old entries load without it), and the registry sent to parakeet.cpp is
built with parakeet_capi_speaker_registry_add_embedding_fp. The library
then refuses a registry of another encoder family and the error names both
families; another quantization of the same family only warns. A voice with
only a weights hash gets the loaded family when the hashes are equal.

Voices without a fingerprint (registered from audio: libvoicedetect cannot
report one) keep the file-name rule and are used with a warning. The
library cannot mix them with fingerprinted voices in one registry, so a
request that has any uses the old registry for all. speaker_strict:true
drops them instead. A library without the symbols behaves as before.

Transcription. vad_trim (seconds, 0 keeps the whole cuts) goes through the
VAD options JSON, so it reaches /v1/vad and the segmenter. The guard_*
options guard_min_local_conf, guard_local_radius and guard_drop_punct_only
turn on the word filter through parakeet_capi_transcribe_path_json_with,
or through the segmenter with vad:true. They are off by default, bad
values fail the load, and a library without the symbol fails it with a
clear message. The dropped word count is logged at debug level.

Assisted-by: Claude:claude-sonnet-5-5 [Claude Code]

---------

Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
2026-10-05 15:57:16 +02:00

136 lines
4.7 KiB
Go

// SPDX-License-Identifier: MIT
package schema
import (
"fmt"
"math"
"regexp"
"strings"
)
// SpeakerEncoder identifies the exact encoder weights, not a model filename.
type SpeakerEncoder struct {
Identity string `json:"identity"`
Dimension int `json:"dimension"`
// Family is the embedding space of the encoder. The server fills it from the
// loaded encoder; exported profiles do not carry it and it is not matched.
Family string `json:"family,omitempty"`
}
// SpeakerProfileInterval locates retained clean audio in the original recording,
// in seconds. It does not describe separated or synthesized audio.
type SpeakerProfileInterval struct {
Start float64 `json:"start"`
End float64 `json:"end"`
}
// SpeakerProfile contains one sensitive voice vector per discovered speaker.
type SpeakerProfile struct {
Speaker int `json:"speaker"`
CleanDuration float64 `json:"clean_duration"`
Intervals []SpeakerProfileInterval `json:"intervals"`
UnavailableReason *string `json:"unavailable_reason"`
Embedding []float32 `json:"embedding,omitempty"`
}
// SpeakerProfiles is the versioned speaker_profiles object exported by parakeet.
// These unsigned profiles are not proof of identity or consent to enrollment.
type SpeakerProfiles struct {
Version int `json:"version"`
Encoder SpeakerEncoder `json:"encoder"`
Speakers []SpeakerProfile `json:"speakers"`
}
var speakerEncoderIdentity = regexp.MustCompile(`^sha256:[0-9a-f]{64}$`)
// Validate checks portable data against metadata from the server's loaded encoder.
// trusted must never come from the request itself. This does not authorize export
// or enrollment, verify provenance, or check intervals against recording length.
func (p SpeakerProfiles) Validate(trusted SpeakerEncoder) error {
if p.Version != 1 {
return fmt.Errorf("unsupported speaker profile version: %d", p.Version)
}
if !speakerEncoderIdentity.MatchString(trusted.Identity) || trusted.Dimension <= 0 {
return fmt.Errorf("invalid trusted speaker encoder metadata")
}
if p.Encoder.Identity != trusted.Identity || p.Encoder.Dimension != trusted.Dimension {
return fmt.Errorf("speaker profile encoder does not match loaded encoder")
}
seen := make(map[int]bool, len(p.Speakers))
for _, s := range p.Speakers {
if s.Speaker < 0 || seen[s.Speaker] {
return fmt.Errorf("invalid or duplicate speaker slot: %d", s.Speaker)
}
seen[s.Speaker] = true
if err := s.validate(trusted.Dimension); err != nil {
return fmt.Errorf("speaker %d: %w", s.Speaker, err)
}
}
return nil
}
// Select validates the complete export and returns a usable speaker for explicit
// enrollment. Callers must supply trusted loaded-encoder metadata, not p.Encoder.
func (p SpeakerProfiles) Select(speaker int, trusted SpeakerEncoder) (SpeakerProfile, error) {
if err := p.Validate(trusted); err != nil {
return SpeakerProfile{}, err
}
for _, s := range p.Speakers {
if s.Speaker == speaker {
if s.UnavailableReason != nil {
return SpeakerProfile{}, fmt.Errorf("speaker %d is unavailable", speaker)
}
return s, nil
}
}
return SpeakerProfile{}, fmt.Errorf("speaker %d not found", speaker)
}
func (s SpeakerProfile) validate(dimension int) error {
// Native JSON rounds timestamps; allow a millisecond of serialization drift.
const tolerance = 0.001
if !finiteProfileNumber(s.CleanDuration) || s.CleanDuration < 0 || s.CleanDuration > 30+tolerance {
return fmt.Errorf("invalid clean duration")
}
var duration, previousEnd float64
for _, interval := range s.Intervals {
if !finiteProfileNumber(interval.Start) || !finiteProfileNumber(interval.End) || interval.Start < previousEnd || interval.End <= interval.Start {
return fmt.Errorf("invalid clean interval")
}
duration += interval.End - interval.Start
previousEnd = interval.End
}
if !finiteProfileNumber(duration) || math.Abs(duration-s.CleanDuration) > tolerance {
return fmt.Errorf("clean duration does not match intervals")
}
if s.UnavailableReason != nil {
if strings.TrimSpace(*s.UnavailableReason) == "" || len(s.Embedding) != 0 {
return fmt.Errorf("invalid unavailable profile")
}
return nil
}
if s.CleanDuration < 2-tolerance {
return fmt.Errorf("insufficient clean speech")
}
if len(s.Embedding) != dimension {
return fmt.Errorf("speaker embedding dimension mismatch")
}
var norm float64
for _, value := range s.Embedding {
v := float64(value)
if !finiteProfileNumber(v) {
return fmt.Errorf("speaker embedding must be finite")
}
norm += v * v
}
if norm == 0 {
return fmt.Errorf("speaker embedding must be nonzero")
}
return nil
}
func finiteProfileNumber(v float64) bool {
return !math.IsNaN(v) && !math.IsInf(v, 0)
}