mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-10 07:47:29 -04:00
* chore(parakeet-cpp): bump parakeet.cpp to 2de154c Brings in the speaker registry encoder fingerprint, the VAD segment trim and the opt-in word filter, a fix for a per-call thread count that stayed set on the process-wide backend after a Silero VAD pass, and bundle components loaded from memory. Assisted-by: Claude:claude-sonnet-5-5 [Claude Code] * feat(parakeet-cpp): encoder fingerprint for speaker naming, vad_trim and guard_* options Speaker naming. A registered voice now carries the encoder that made it: the embedding family (voicedetect:<arch>:<name>:<dim>) and the sha256 of the weights. The backend reports the family of the loaded speaker model in Status, voice enrollment from speaker_profiles stores it as encoder_family (old entries load without it), and the registry sent to parakeet.cpp is built with parakeet_capi_speaker_registry_add_embedding_fp. The library then refuses a registry of another encoder family and the error names both families; another quantization of the same family only warns. A voice with only a weights hash gets the loaded family when the hashes are equal. Voices without a fingerprint (registered from audio: libvoicedetect cannot report one) keep the file-name rule and are used with a warning. The library cannot mix them with fingerprinted voices in one registry, so a request that has any uses the old registry for all. speaker_strict:true drops them instead. A library without the symbols behaves as before. Transcription. vad_trim (seconds, 0 keeps the whole cuts) goes through the VAD options JSON, so it reaches /v1/vad and the segmenter. The guard_* options guard_min_local_conf, guard_local_radius and guard_drop_punct_only turn on the word filter through parakeet_capi_transcribe_path_json_with, or through the segmenter with vad:true. They are off by default, bad values fail the load, and a library without the symbol fails it with a clear message. The dropped word count is logged at debug level. Assisted-by: Claude:claude-sonnet-5-5 [Claude Code] --------- Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
136 lines
4.7 KiB
Go
136 lines
4.7 KiB
Go
// SPDX-License-Identifier: MIT
|
|
|
|
package schema
|
|
|
|
import (
|
|
"fmt"
|
|
"math"
|
|
"regexp"
|
|
"strings"
|
|
)
|
|
|
|
// SpeakerEncoder identifies the exact encoder weights, not a model filename.
|
|
type SpeakerEncoder struct {
|
|
Identity string `json:"identity"`
|
|
Dimension int `json:"dimension"`
|
|
// Family is the embedding space of the encoder. The server fills it from the
|
|
// loaded encoder; exported profiles do not carry it and it is not matched.
|
|
Family string `json:"family,omitempty"`
|
|
}
|
|
|
|
// SpeakerProfileInterval locates retained clean audio in the original recording,
|
|
// in seconds. It does not describe separated or synthesized audio.
|
|
type SpeakerProfileInterval struct {
|
|
Start float64 `json:"start"`
|
|
End float64 `json:"end"`
|
|
}
|
|
|
|
// SpeakerProfile contains one sensitive voice vector per discovered speaker.
|
|
type SpeakerProfile struct {
|
|
Speaker int `json:"speaker"`
|
|
CleanDuration float64 `json:"clean_duration"`
|
|
Intervals []SpeakerProfileInterval `json:"intervals"`
|
|
UnavailableReason *string `json:"unavailable_reason"`
|
|
Embedding []float32 `json:"embedding,omitempty"`
|
|
}
|
|
|
|
// SpeakerProfiles is the versioned speaker_profiles object exported by parakeet.
|
|
// These unsigned profiles are not proof of identity or consent to enrollment.
|
|
type SpeakerProfiles struct {
|
|
Version int `json:"version"`
|
|
Encoder SpeakerEncoder `json:"encoder"`
|
|
Speakers []SpeakerProfile `json:"speakers"`
|
|
}
|
|
|
|
var speakerEncoderIdentity = regexp.MustCompile(`^sha256:[0-9a-f]{64}$`)
|
|
|
|
// Validate checks portable data against metadata from the server's loaded encoder.
|
|
// trusted must never come from the request itself. This does not authorize export
|
|
// or enrollment, verify provenance, or check intervals against recording length.
|
|
func (p SpeakerProfiles) Validate(trusted SpeakerEncoder) error {
|
|
if p.Version != 1 {
|
|
return fmt.Errorf("unsupported speaker profile version: %d", p.Version)
|
|
}
|
|
if !speakerEncoderIdentity.MatchString(trusted.Identity) || trusted.Dimension <= 0 {
|
|
return fmt.Errorf("invalid trusted speaker encoder metadata")
|
|
}
|
|
if p.Encoder.Identity != trusted.Identity || p.Encoder.Dimension != trusted.Dimension {
|
|
return fmt.Errorf("speaker profile encoder does not match loaded encoder")
|
|
}
|
|
seen := make(map[int]bool, len(p.Speakers))
|
|
for _, s := range p.Speakers {
|
|
if s.Speaker < 0 || seen[s.Speaker] {
|
|
return fmt.Errorf("invalid or duplicate speaker slot: %d", s.Speaker)
|
|
}
|
|
seen[s.Speaker] = true
|
|
if err := s.validate(trusted.Dimension); err != nil {
|
|
return fmt.Errorf("speaker %d: %w", s.Speaker, err)
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// Select validates the complete export and returns a usable speaker for explicit
|
|
// enrollment. Callers must supply trusted loaded-encoder metadata, not p.Encoder.
|
|
func (p SpeakerProfiles) Select(speaker int, trusted SpeakerEncoder) (SpeakerProfile, error) {
|
|
if err := p.Validate(trusted); err != nil {
|
|
return SpeakerProfile{}, err
|
|
}
|
|
for _, s := range p.Speakers {
|
|
if s.Speaker == speaker {
|
|
if s.UnavailableReason != nil {
|
|
return SpeakerProfile{}, fmt.Errorf("speaker %d is unavailable", speaker)
|
|
}
|
|
return s, nil
|
|
}
|
|
}
|
|
return SpeakerProfile{}, fmt.Errorf("speaker %d not found", speaker)
|
|
}
|
|
|
|
func (s SpeakerProfile) validate(dimension int) error {
|
|
// Native JSON rounds timestamps; allow a millisecond of serialization drift.
|
|
const tolerance = 0.001
|
|
if !finiteProfileNumber(s.CleanDuration) || s.CleanDuration < 0 || s.CleanDuration > 30+tolerance {
|
|
return fmt.Errorf("invalid clean duration")
|
|
}
|
|
var duration, previousEnd float64
|
|
for _, interval := range s.Intervals {
|
|
if !finiteProfileNumber(interval.Start) || !finiteProfileNumber(interval.End) || interval.Start < previousEnd || interval.End <= interval.Start {
|
|
return fmt.Errorf("invalid clean interval")
|
|
}
|
|
duration += interval.End - interval.Start
|
|
previousEnd = interval.End
|
|
}
|
|
if !finiteProfileNumber(duration) || math.Abs(duration-s.CleanDuration) > tolerance {
|
|
return fmt.Errorf("clean duration does not match intervals")
|
|
}
|
|
if s.UnavailableReason != nil {
|
|
if strings.TrimSpace(*s.UnavailableReason) == "" || len(s.Embedding) != 0 {
|
|
return fmt.Errorf("invalid unavailable profile")
|
|
}
|
|
return nil
|
|
}
|
|
if s.CleanDuration < 2-tolerance {
|
|
return fmt.Errorf("insufficient clean speech")
|
|
}
|
|
if len(s.Embedding) != dimension {
|
|
return fmt.Errorf("speaker embedding dimension mismatch")
|
|
}
|
|
var norm float64
|
|
for _, value := range s.Embedding {
|
|
v := float64(value)
|
|
if !finiteProfileNumber(v) {
|
|
return fmt.Errorf("speaker embedding must be finite")
|
|
}
|
|
norm += v * v
|
|
}
|
|
if norm == 0 {
|
|
return fmt.Errorf("speaker embedding must be nonzero")
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func finiteProfileNumber(v float64) bool {
|
|
return !math.IsNaN(v) && !math.IsInf(v, 0)
|
|
}
|