mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-10 07:47:29 -04:00
feat(voice): name registered speakers through a bundle speaker component
Naming read only speaker_model:, so a parakeet-cpp bundle that sets speaker_component:voice never received the registered voices, in live transcription or in diarization. Resolve speaker_component: when speaker_model: is not set. A component is not an encoder file, so no file-name tag is derived for it: voices with a hash or family identity and untagged voices reach the backend, and a voice with only a file-name tag matches through the new optional speaker_tag: alias. speaker_model: still wins, including when it points at the bundle file. The bundle gallery entries do not declare speaker_recognition: that usecase means the VoiceEmbed and VoiceVerify RPCs, which the backend does not serve, and it selects the default model for /v1/voice/*. The pinning test now says so. Assisted-by: Claude Code:claude-sonnet-5-5 go golangci-lint
This commit is contained in:
7 files changed
+201
-18
No files matched your search
@@ -65,12 +65,18 @@ var _ = Describe("gallery/index.yaml parakeet-cpp bundle entries", func() {
|
||||
}
|
||||
})
|
||||
|
||||
// speaker_recognition is deliberately absent: that usecase means the model
|
||||
// answers the /v1/voice/* RPCs (VoiceEmbed, VoiceVerify), which the
|
||||
// parakeet-cpp backend does not implement, and it picks the default model
|
||||
// for those routes. Naming registered speakers works from the
|
||||
// speaker_component option alone, with no usecase.
|
||||
It("declares the usecases and options of the roles each bundle covers", func() {
|
||||
all := entries()
|
||||
for name, w := range want {
|
||||
e := all[name]
|
||||
Expect(e.Overrides["known_usecases"]).To(ConsistOf(toAny(w.usecases)...), name)
|
||||
Expect(e.Overrides["options"]).To(ConsistOf(toAny(w.options)...), name)
|
||||
Expect(e.Overrides["known_usecases"]).ToNot(ContainElement("speaker_recognition"), name)
|
||||
}
|
||||
})
|
||||
|
||||
|
||||
@@ -211,25 +211,31 @@ func warnOnce(key string) bool {
|
||||
}
|
||||
|
||||
// selectKnownVoices returns the registered voices a backend may use to name
|
||||
// speakers, or nil when the model has no speaker_model, there is no registry,
|
||||
// or the registry cannot be read. It never fails the caller: unnamed speakers
|
||||
// speakers, or nil when the model names no speaker encoder (speaker_model: or,
|
||||
// for a bundle file, speaker_component:), there is no registry, or the
|
||||
// registry cannot be read. It never fails the caller: unnamed speakers
|
||||
// are the fallback. feature only prefixes the log messages.
|
||||
func selectKnownVoices(ctx context.Context, feature string, options []string, registry voicerecognition.Registry) []voicerecognition.KnownVoice {
|
||||
sm := voicerecognition.SpeakerModelFromOptions(options)
|
||||
if sm == "" || registry == nil {
|
||||
enc := voicerecognition.SpeakerEncoderFromOptions(options)
|
||||
if enc.Ref == "" || registry == nil {
|
||||
return nil
|
||||
}
|
||||
sel, err := voicerecognition.KnownVoicesFor(ctx, registry, sm)
|
||||
sm := enc.Ref
|
||||
// A speaker_component names a part of the bundle, not an encoder file, so
|
||||
// enc.File is empty then: voices with a hash or family identity and
|
||||
// untagged voices reach the backend, and a file-name tag matches only
|
||||
// through the speaker_tag alias.
|
||||
sel, err := voicerecognition.KnownVoicesFor(ctx, registry, enc.File, enc.Tags...)
|
||||
if err != nil {
|
||||
xlog.Warn(feature+": could not read the voice registry; speakers stay unnamed", "error", err)
|
||||
return nil
|
||||
}
|
||||
if len(sel.Voices) == 0 && sel.OtherEncoder > 0 {
|
||||
msg := feature + ": registered voices were made with a different encoder than this model's speaker_model; speakers stay unnamed"
|
||||
msg := feature + ": registered voices were made with a different encoder than this model's speaker_model or speaker_component; speakers stay unnamed"
|
||||
if warnOnce(feature + "|" + sm) {
|
||||
xlog.Warn(msg, "speaker_model", sm, "voices_from_other_encoder", sel.OtherEncoder)
|
||||
xlog.Warn(msg, "speaker", sm, "voices_from_other_encoder", sel.OtherEncoder)
|
||||
} else {
|
||||
xlog.Debug(msg, "speaker_model", sm, "voices_from_other_encoder", sel.OtherEncoder)
|
||||
xlog.Debug(msg, "speaker", sm, "voices_from_other_encoder", sel.OtherEncoder)
|
||||
}
|
||||
}
|
||||
return sel.Voices
|
||||
|
||||
@@ -109,4 +109,46 @@ var _ = Describe("attachKnownVoices", func() {
|
||||
fakeVoiceRegistry{entries: []voicerecognition.Entry{ada}})
|
||||
Expect(req.KnownVoices).To(BeEmpty())
|
||||
})
|
||||
|
||||
Context("with a bundle model", func() {
|
||||
const hash = "sha256:72040372aa"
|
||||
hashed := voicerecognition.Entry{Metadata: voicerecognition.Metadata{ID: "h", Name: "Hashed", Model: hash}, Embedding: []float32{1, 0}}
|
||||
legacy := voicerecognition.Entry{Metadata: voicerecognition.Metadata{ID: "l", Name: "Legacy", Model: "voice-detect-wespeaker-resnet34.gguf"}, Embedding: []float32{1, 0}}
|
||||
other := voicerecognition.Entry{Metadata: voicerecognition.Metadata{ID: "o", Name: "Other", Model: "voice-detect-ecapa-tdnn-voxceleb.gguf"}, Embedding: []float32{1, 0}}
|
||||
reg := fakeVoiceRegistry{entries: []voicerecognition.Entry{hashed, legacy, other}}
|
||||
names := func(options ...string) []string {
|
||||
var out []string
|
||||
for _, v := range selectKnownVoices(context.Background(), "test", options, reg) {
|
||||
out = append(out, v.Name)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
It("sends only the hash-tagged voice for a speaker_component", func() {
|
||||
Expect(names("diar_component:diar", "speaker_component:voice")).To(Equal([]string{"Hashed"}))
|
||||
})
|
||||
It("adds the voices tagged with the speaker_tag alias", func() {
|
||||
Expect(names("speaker_component:voice", "speaker_tag:voice-detect-wespeaker-resnet34.gguf")).
|
||||
To(Equal([]string{"Hashed", "Legacy"}))
|
||||
})
|
||||
It("does not guess a tag from the bundle file for a speaker_component", func() {
|
||||
Expect(names("speaker_component:voice", "speaker_tag:bundle.gguf")).To(Equal([]string{"Hashed"}))
|
||||
})
|
||||
It("keeps speaker_model pointing at the bundle file working", func() {
|
||||
Expect(names("speaker_model:parakeet-cpp/bundle.gguf")).To(Equal([]string{"Hashed"}))
|
||||
})
|
||||
It("lets speaker_model win over speaker_component", func() {
|
||||
Expect(names("speaker_component:voice", "speaker_model:voice-detect-ecapa-tdnn-voxceleb.gguf")).
|
||||
To(Equal([]string{"Hashed", "Other"}))
|
||||
})
|
||||
It("names nothing for a bare speaker_tag", func() {
|
||||
Expect(names("speaker_tag:voice-detect-wespeaker-resnet34.gguf")).To(BeEmpty())
|
||||
})
|
||||
It("fills a diarization request", func() {
|
||||
req := backend.DiarizationRequest{}
|
||||
attachKnownVoices(context.Background(), &req, []string{"speaker_component:voice"}, reg)
|
||||
Expect(req.KnownVoices).To(HaveLen(1))
|
||||
Expect(req.KnownVoices[0].Name).To(Equal("Hashed"))
|
||||
})
|
||||
})
|
||||
})
|
||||
@@ -29,6 +29,16 @@ var _ = Describe("liveVoiceOptions", func() {
|
||||
Expect(liveVoiceOptions(context.Background(), nil, cfgWith("speaker_model:enc.gguf"))).To(BeEmpty())
|
||||
Expect(liveVoiceOptions(context.Background(), fakeVoiceRegistry{err: errors.New("boom")}, cfgWith("speaker_model:enc.gguf"))).To(BeEmpty())
|
||||
})
|
||||
It("opens a bundle session through its speaker_component", func() {
|
||||
hashed := voicerecognition.Entry{Metadata: voicerecognition.Metadata{Name: "Ada", Model: "sha256:72040372aa"}, Embedding: []float32{1, 0}}
|
||||
legacy := voicerecognition.Entry{Metadata: voicerecognition.Metadata{Name: "Bob", Model: "voice-detect-wespeaker-resnet34.gguf"}, Embedding: []float32{1, 0}}
|
||||
reg := fakeVoiceRegistry{entries: []voicerecognition.Entry{legacy, hashed}}
|
||||
Expect(liveVoiceOptions(context.Background(), reg, cfgWith("vad:true", "speaker_component:voice"))).To(HaveLen(1))
|
||||
// A tag-only voice alone is refused without the alias, accepted with it.
|
||||
tagOnly := fakeVoiceRegistry{entries: []voicerecognition.Entry{legacy}}
|
||||
Expect(liveVoiceOptions(context.Background(), tagOnly, cfgWith("speaker_component:voice"))).To(BeEmpty())
|
||||
Expect(liveVoiceOptions(context.Background(), tagOnly, cfgWith("speaker_component:voice", "speaker_tag:voice-detect-wespeaker-resnet34.gguf"))).To(HaveLen(1))
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("transcription segment event", func() {
|
||||
|
||||
@@ -1220,7 +1220,7 @@ func newModel(pipeline *config.Pipeline, cl *config.ModelConfigLoader, ml *model
|
||||
}
|
||||
|
||||
// liveVoiceOptions selects the registered voices a live session may name speakers
|
||||
// with. It stays empty without a speaker_model or a voice registry.
|
||||
// with. It stays empty without a speaker_model or speaker_component, or a voice registry.
|
||||
func liveVoiceOptions(ctx context.Context, registry voicerecognition.Registry, cfg *config.ModelConfig) []backend.LiveOption {
|
||||
voices := selectKnownVoices(ctx, "live transcription", cfg.Options, registry)
|
||||
if len(voices) == 0 {
|
||||
|
||||
@@ -22,18 +22,61 @@ type KnownVoice struct {
|
||||
Weights string
|
||||
}
|
||||
|
||||
// SpeakerModelFromOptions returns the value of a speaker_model:<file> entry in
|
||||
// a model config's options, or "" when there is none.
|
||||
func SpeakerModelFromOptions(options []string) string {
|
||||
// optionValue returns the trimmed value of the first key:value entry in
|
||||
// options whose key is key, or "" when there is none.
|
||||
func optionValue(options []string, key string) string {
|
||||
for _, o := range options {
|
||||
k, v, ok := strings.Cut(o, ":")
|
||||
if ok && strings.TrimSpace(k) == "speaker_model" {
|
||||
if ok && strings.TrimSpace(k) == key {
|
||||
return strings.TrimSpace(v)
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// SpeakerModelFromOptions returns the speaker encoder a model config names, or
|
||||
// "" when it names none. A speaker_model:<file> entry wins. Without one, a
|
||||
// speaker_component:<name> entry names a component of the model's own bundle
|
||||
// file (parakeet-cpp), and its value is returned. A component name is not a
|
||||
// file name: it must not be used as an encoder tag (see SpeakerTagFromOptions).
|
||||
func SpeakerModelFromOptions(options []string) string {
|
||||
if v := optionValue(options, "speaker_model"); v != "" {
|
||||
return v
|
||||
}
|
||||
return optionValue(options, "speaker_component")
|
||||
}
|
||||
|
||||
// SpeakerTagFromOptions returns the lowercased value of a speaker_tag:<tag>
|
||||
// entry, or "". It is an extra encoder tag for legacy voices that carry only
|
||||
// a file-name tag. A bundle has no encoder file name of its own, so such a
|
||||
// voice matches a bundle's speaker component only through this alias.
|
||||
func SpeakerTagFromOptions(options []string) string {
|
||||
return strings.ToLower(optionValue(options, "speaker_tag"))
|
||||
}
|
||||
|
||||
// SpeakerEncoder is the speaker encoder a model config names, as far as voice
|
||||
// selection needs it.
|
||||
type SpeakerEncoder struct {
|
||||
// Ref is what the config names: the speaker_model file, else the
|
||||
// speaker_component name. Empty when the config names no encoder.
|
||||
Ref string
|
||||
// File is the speaker_model path. Empty for a speaker_component, which
|
||||
// names a part of the model's own bundle file, not an encoder file.
|
||||
File string
|
||||
// Tags are extra encoder tags from speaker_tag.
|
||||
Tags []string
|
||||
}
|
||||
|
||||
// SpeakerEncoderFromOptions reads the speaker encoder from a model config's
|
||||
// options. See SpeakerModelFromOptions and SpeakerTagFromOptions.
|
||||
func SpeakerEncoderFromOptions(options []string) SpeakerEncoder {
|
||||
enc := SpeakerEncoder{Ref: SpeakerModelFromOptions(options), File: optionValue(options, "speaker_model")}
|
||||
if t := SpeakerTagFromOptions(options); t != "" {
|
||||
enc.Tags = []string{t}
|
||||
}
|
||||
return enc
|
||||
}
|
||||
|
||||
// EncoderTag is how an encoder is identified: the lowercased base name of its
|
||||
// model file. A voice registered through the voice-detect backend carries the
|
||||
// backend's model name, which defaults to that base name.
|
||||
@@ -53,8 +96,19 @@ type KnownVoiceSelection struct {
|
||||
// filtering belongs to the loaded backend. Tagged candidates precede untagged
|
||||
// ones, each ordered by registration ID so registry iteration order cannot
|
||||
// change replay order. The input is not modified.
|
||||
func SelectKnownVoices(entries []Entry, speakerModelPath string) KnownVoiceSelection {
|
||||
tag := EncoderTag(speakerModelPath)
|
||||
//
|
||||
// extraTags are further tags that count as the loaded encoder (the speaker_tag
|
||||
// option). They are compared as lowercase names, like EncoderTag.
|
||||
func SelectKnownVoices(entries []Entry, speakerModelPath string, extraTags ...string) KnownVoiceSelection {
|
||||
tags := make(map[string]bool, len(extraTags)+1)
|
||||
if speakerModelPath != "" {
|
||||
tags[EncoderTag(speakerModelPath)] = true
|
||||
}
|
||||
for _, t := range extraTags {
|
||||
if t = strings.ToLower(strings.TrimSpace(t)); t != "" {
|
||||
tags[t] = true
|
||||
}
|
||||
}
|
||||
var sel KnownVoiceSelection
|
||||
var untagged []Entry
|
||||
for _, e := range entries {
|
||||
@@ -66,7 +120,7 @@ func SelectKnownVoices(entries []Entry, speakerModelPath string) KnownVoiceSelec
|
||||
untagged = append(untagged, e)
|
||||
// Hash-tagged portable registrations are checked against the loaded
|
||||
// encoder by the backend, never against a filename or dimension alone.
|
||||
case strings.HasPrefix(e.Metadata.Model, "sha256:"), EncoderTag(e.Metadata.Model) == tag:
|
||||
case strings.HasPrefix(e.Metadata.Model, "sha256:"), tags[EncoderTag(e.Metadata.Model)]:
|
||||
sel.Voices = append(sel.Voices, knownVoice(e))
|
||||
default:
|
||||
sel.OtherEncoder++
|
||||
@@ -90,10 +144,10 @@ func knownVoice(e Entry) KnownVoice {
|
||||
}
|
||||
|
||||
// KnownVoicesFor lists the registry and selects the voices for a speaker model.
|
||||
func KnownVoicesFor(ctx context.Context, reg Registry, speakerModelPath string) (KnownVoiceSelection, error) {
|
||||
func KnownVoicesFor(ctx context.Context, reg Registry, speakerModelPath string, extraTags ...string) (KnownVoiceSelection, error) {
|
||||
entries, err := reg.List(ctx)
|
||||
if err != nil {
|
||||
return KnownVoiceSelection{}, err
|
||||
}
|
||||
return SelectKnownVoices(entries, speakerModelPath), nil
|
||||
return SelectKnownVoices(entries, speakerModelPath, extraTags...), nil
|
||||
}
|
||||
@@ -26,6 +26,34 @@ var _ = Describe("SpeakerModelFromOptions", func() {
|
||||
Expect(voicerecognition.SpeakerModelFromOptions([]string{"diarization_model:d.gguf"})).To(BeEmpty())
|
||||
Expect(voicerecognition.SpeakerModelFromOptions(nil)).To(BeEmpty())
|
||||
})
|
||||
It("falls back to speaker_component when there is no speaker_model", func() {
|
||||
Expect(voicerecognition.SpeakerModelFromOptions([]string{"diar_component:diar", "speaker_component: voice "})).To(Equal("voice"))
|
||||
})
|
||||
It("prefers speaker_model over speaker_component, in any order", func() {
|
||||
Expect(voicerecognition.SpeakerModelFromOptions([]string{"speaker_component:voice", "speaker_model:bundle.gguf"})).To(Equal("bundle.gguf"))
|
||||
Expect(voicerecognition.SpeakerModelFromOptions([]string{"speaker_model:bundle.gguf", "speaker_component:voice"})).To(Equal("bundle.gguf"))
|
||||
})
|
||||
It("ignores empty values", func() {
|
||||
Expect(voicerecognition.SpeakerModelFromOptions([]string{"speaker_model:", "speaker_component:voice"})).To(Equal("voice"))
|
||||
Expect(voicerecognition.SpeakerModelFromOptions([]string{"speaker_model:", "speaker_component:"})).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("SpeakerEncoderFromOptions", func() {
|
||||
It("has a file for speaker_model and none for a component", func() {
|
||||
Expect(voicerecognition.SpeakerEncoderFromOptions([]string{"speaker_model:enc.gguf"})).
|
||||
To(Equal(voicerecognition.SpeakerEncoder{Ref: "enc.gguf", File: "enc.gguf"}))
|
||||
Expect(voicerecognition.SpeakerEncoderFromOptions([]string{"speaker_component:voice"})).
|
||||
To(Equal(voicerecognition.SpeakerEncoder{Ref: "voice"}))
|
||||
})
|
||||
It("reads the speaker_tag alias in lowercase", func() {
|
||||
Expect(voicerecognition.SpeakerEncoderFromOptions([]string{"speaker_component:voice", "speaker_tag: Voice-Detect-WeSpeaker.gguf"}).Tags).
|
||||
To(Equal([]string{"voice-detect-wespeaker.gguf"}))
|
||||
Expect(voicerecognition.SpeakerEncoderFromOptions([]string{"speaker_component:voice"}).Tags).To(BeEmpty())
|
||||
})
|
||||
It("names no encoder for a speaker_tag alone", func() {
|
||||
Expect(voicerecognition.SpeakerEncoderFromOptions([]string{"speaker_tag:x.gguf"}).Ref).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("EncoderTag", func() {
|
||||
@@ -146,6 +174,43 @@ var _ = Describe("encoder fingerprint of selected voices", func() {
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("SelectKnownVoices for a bundle component", func() {
|
||||
const hash = "sha256:72040372aa"
|
||||
mixed := func() []voicerecognition.Entry {
|
||||
return []voicerecognition.Entry{
|
||||
entry("hashed", hash, 1, 0),
|
||||
entry("legacy-wespeaker", "voice-detect-wespeaker-resnet34.gguf", 1, 0),
|
||||
entry("legacy-ecapa", "voice-detect-ecapa-tdnn-voxceleb.gguf", 1, 0),
|
||||
entry("old", "", 1, 0),
|
||||
}
|
||||
}
|
||||
ids := func(sel voicerecognition.KnownVoiceSelection) []string {
|
||||
var out []string
|
||||
for _, v := range sel.Voices {
|
||||
out = append(out, v.ID)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
It("sends hash-tagged and untagged voices, and refuses tag-only voices, without a tag alias", func() {
|
||||
sel := voicerecognition.SelectKnownVoices(mixed(), "")
|
||||
Expect(ids(sel)).To(Equal([]string{"hashed", "old"}))
|
||||
Expect(sel.OtherEncoder).To(Equal(2))
|
||||
})
|
||||
It("matches a tag-only voice through the alias and no other", func() {
|
||||
sel := voicerecognition.SelectKnownVoices(mixed(), "", "Voice-Detect-WeSpeaker-ResNet34.gguf")
|
||||
Expect(ids(sel)).To(Equal([]string{"hashed", "legacy-wespeaker", "old"}))
|
||||
Expect(sel.OtherEncoder).To(Equal(1))
|
||||
})
|
||||
It("adds the alias to the file tag of a speaker_model", func() {
|
||||
sel := voicerecognition.SelectKnownVoices(mixed(), "bundle.gguf", "voice-detect-ecapa-tdnn-voxceleb.gguf")
|
||||
Expect(ids(sel)).To(Equal([]string{"hashed", "legacy-ecapa", "old"}))
|
||||
})
|
||||
It("ignores a blank alias", func() {
|
||||
Expect(ids(voicerecognition.SelectKnownVoices(mixed(), "", " ", ""))).To(Equal([]string{"hashed", "old"}))
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("KnownVoicesFor", func() {
|
||||
It("selects from the registry listing", func() {
|
||||
sel, err := voicerecognition.KnownVoicesFor(context.Background(), listRegistry{entries: []voicerecognition.Entry{entry("ada", "m.gguf", 1)}}, "m.gguf")
|
||||
|
||||
Reference in new issue
Block a user