mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-05 12:34:43 -04:00
* feat(schema): validate portable speaker profiles Add the versioned profile schema for explicit speaker enrollment. Validate compatibility against separately supplied loaded-encoder metadata. Reject unusable speakers, invalid vectors, and inconsistent clean spans. This slice does not change HTTP routes, backend integration, or the UI. Assisted-by: OpenAI:unknown Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(parakeet): export profiles with transcripts Export opt-in speaker profiles and trusted encoder metadata. Replay registrations by ID so duplicate display names keep independent vectors. Use one profile-capable diarization for slots, names, and clean spans. Assign timestamped ASR words to those slots without a second diarization. Preserve legacy opt-out and no-ASR behavior, and propagate failures. Assisted-by: OpenAI:unknown Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(audio): enroll portable speaker profiles Gate profile exports with voice-recognition permission and validate registration against metadata from the loaded encoder. Preserve audio enrollment and independent registrations with duplicate display names. Exclude diarization and registration exchanges before API trace capture so persisted traces cannot retain profile vectors or JSON audio. Defer candidate dimensions to trusted loaded metadata. Sort candidates by registration ID so incompatible profiles cannot suppress legacy voices through registry iteration order. Keep portable identity checks closed when trusted metadata is unavailable. Test persisted traces, explicit slot zero, and selection through offline and live transport. Document privacy and the ephemeral registry lifecycle. Assisted-by: OpenAI:unknown Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * feat(ui): remember speakers from diarization Add a Studio page for diarization and opt-in speaker profiles. Preview clean intervals from the original recording before explicit registration. Join profiles by raw speaker labels, preserve duplicate names, and relabel turns only after a successful save. Discard stale results when the model or recording changes. Share registration metadata with voice management without storing vectors or recordings from this flow. Document permissions and the global, ephemeral registry. Cover enrollment, permissions, previews, and asynchronous races with mocked Playwright tests. Assisted-by: OpenAI:unknown Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * docs: clarify HTTP speaker enrollment support Replace the stale enrollment limitation with the current HTTP workflow. Distinguish native transport from explicit registration and link its docs. Assisted-by: OpenAI:unknown Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * chore(parakeet): pin merged speaker profile support Use the merged commit from mudler/parakeet.cpp#80. Its tree matches the previously accepted native pin. Assisted-by: OpenAI:unknown Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * docs: add diarization enrollment setup example Connect the existing gallery modes to the speaker enrollment workflow. Show installation, private profile export, explicit raw-slot registration, and later recognition without another export. Assisted-by: OpenAI:unknown Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * docs(blog): explain diarization speaker profiles Put the diarization walkthrough on the LocalAI website in the feature PR. Cover the three gallery modes, explicit enrollment, and privacy limits. Link setup instructions and keep availability conditional on feature support. Assisted-by: OpenAI:unknown Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * docs(blog): focus diarization on everyday use Explain what users can do with recordings before the setup steps. Replace the technical walkthrough with a short Studio guide and link readers to the existing reference for model names and developer use. Assisted-by: OpenAI:unknown Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * docs(blog): lead with speaker capabilities Present speaker recognition through everyday uses and a short UI flow. Keep technical reference details in the existing documentation. Assisted-by: OpenAI:unknown Signed-off-by: Ettore Di Giacinto <mudler@localai.io> * fix(diarization): satisfy Go lint checks Avoid copying protobuf message state when extending backend status, check the multipart reader close result, and document the focused testing.T lint exemptions. Assisted-by: nib:gpt-5.6-sol Signed-off-by: Ettore Di Giacinto <mudler@localai.io> --------- Signed-off-by: Ettore Di Giacinto <mudler@localai.io> Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
133 lines
4.5 KiB
Go
133 lines
4.5 KiB
Go
// SPDX-License-Identifier: MIT
|
|
|
|
package schema
|
|
|
|
import (
|
|
"fmt"
|
|
"math"
|
|
"regexp"
|
|
"strings"
|
|
)
|
|
|
|
// SpeakerEncoder identifies the exact encoder weights, not a model filename.
|
|
type SpeakerEncoder struct {
|
|
Identity string `json:"identity"`
|
|
Dimension int `json:"dimension"`
|
|
}
|
|
|
|
// SpeakerProfileInterval locates retained clean audio in the original recording,
|
|
// in seconds. It does not describe separated or synthesized audio.
|
|
type SpeakerProfileInterval struct {
|
|
Start float64 `json:"start"`
|
|
End float64 `json:"end"`
|
|
}
|
|
|
|
// SpeakerProfile contains one sensitive voice vector per discovered speaker.
|
|
type SpeakerProfile struct {
|
|
Speaker int `json:"speaker"`
|
|
CleanDuration float64 `json:"clean_duration"`
|
|
Intervals []SpeakerProfileInterval `json:"intervals"`
|
|
UnavailableReason *string `json:"unavailable_reason"`
|
|
Embedding []float32 `json:"embedding,omitempty"`
|
|
}
|
|
|
|
// SpeakerProfiles is the versioned speaker_profiles object exported by parakeet.
|
|
// These unsigned profiles are not proof of identity or consent to enrollment.
|
|
type SpeakerProfiles struct {
|
|
Version int `json:"version"`
|
|
Encoder SpeakerEncoder `json:"encoder"`
|
|
Speakers []SpeakerProfile `json:"speakers"`
|
|
}
|
|
|
|
var speakerEncoderIdentity = regexp.MustCompile(`^sha256:[0-9a-f]{64}$`)
|
|
|
|
// Validate checks portable data against metadata from the server's loaded encoder.
|
|
// trusted must never come from the request itself. This does not authorize export
|
|
// or enrollment, verify provenance, or check intervals against recording length.
|
|
func (p SpeakerProfiles) Validate(trusted SpeakerEncoder) error {
|
|
if p.Version != 1 {
|
|
return fmt.Errorf("unsupported speaker profile version: %d", p.Version)
|
|
}
|
|
if !speakerEncoderIdentity.MatchString(trusted.Identity) || trusted.Dimension <= 0 {
|
|
return fmt.Errorf("invalid trusted speaker encoder metadata")
|
|
}
|
|
if p.Encoder != trusted {
|
|
return fmt.Errorf("speaker profile encoder does not match loaded encoder")
|
|
}
|
|
seen := make(map[int]bool, len(p.Speakers))
|
|
for _, s := range p.Speakers {
|
|
if s.Speaker < 0 || seen[s.Speaker] {
|
|
return fmt.Errorf("invalid or duplicate speaker slot: %d", s.Speaker)
|
|
}
|
|
seen[s.Speaker] = true
|
|
if err := s.validate(trusted.Dimension); err != nil {
|
|
return fmt.Errorf("speaker %d: %w", s.Speaker, err)
|
|
}
|
|
}
|
|
return nil
|
|
}
|
|
|
|
// Select validates the complete export and returns a usable speaker for explicit
|
|
// enrollment. Callers must supply trusted loaded-encoder metadata, not p.Encoder.
|
|
func (p SpeakerProfiles) Select(speaker int, trusted SpeakerEncoder) (SpeakerProfile, error) {
|
|
if err := p.Validate(trusted); err != nil {
|
|
return SpeakerProfile{}, err
|
|
}
|
|
for _, s := range p.Speakers {
|
|
if s.Speaker == speaker {
|
|
if s.UnavailableReason != nil {
|
|
return SpeakerProfile{}, fmt.Errorf("speaker %d is unavailable", speaker)
|
|
}
|
|
return s, nil
|
|
}
|
|
}
|
|
return SpeakerProfile{}, fmt.Errorf("speaker %d not found", speaker)
|
|
}
|
|
|
|
func (s SpeakerProfile) validate(dimension int) error {
|
|
// Native JSON rounds timestamps; allow a millisecond of serialization drift.
|
|
const tolerance = 0.001
|
|
if !finiteProfileNumber(s.CleanDuration) || s.CleanDuration < 0 || s.CleanDuration > 30+tolerance {
|
|
return fmt.Errorf("invalid clean duration")
|
|
}
|
|
var duration, previousEnd float64
|
|
for _, interval := range s.Intervals {
|
|
if !finiteProfileNumber(interval.Start) || !finiteProfileNumber(interval.End) || interval.Start < previousEnd || interval.End <= interval.Start {
|
|
return fmt.Errorf("invalid clean interval")
|
|
}
|
|
duration += interval.End - interval.Start
|
|
previousEnd = interval.End
|
|
}
|
|
if !finiteProfileNumber(duration) || math.Abs(duration-s.CleanDuration) > tolerance {
|
|
return fmt.Errorf("clean duration does not match intervals")
|
|
}
|
|
if s.UnavailableReason != nil {
|
|
if strings.TrimSpace(*s.UnavailableReason) == "" || len(s.Embedding) != 0 {
|
|
return fmt.Errorf("invalid unavailable profile")
|
|
}
|
|
return nil
|
|
}
|
|
if s.CleanDuration < 2-tolerance {
|
|
return fmt.Errorf("insufficient clean speech")
|
|
}
|
|
if len(s.Embedding) != dimension {
|
|
return fmt.Errorf("speaker embedding dimension mismatch")
|
|
}
|
|
var norm float64
|
|
for _, value := range s.Embedding {
|
|
v := float64(value)
|
|
if !finiteProfileNumber(v) {
|
|
return fmt.Errorf("speaker embedding must be finite")
|
|
}
|
|
norm += v * v
|
|
}
|
|
if norm == 0 {
|
|
return fmt.Errorf("speaker embedding must be nonzero")
|
|
}
|
|
return nil
|
|
}
|
|
|
|
func finiteProfileNumber(v float64) bool {
|
|
return !math.IsNaN(v) && !math.IsInf(v, 0)
|
|
}
|