From 1d117292b15a3b7c657c2e02d3446beacd3aaca1 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Mon, 5 Oct 2026 23:43:23 +0000 Subject: [PATCH] feat(voice): name registered speakers through a bundle speaker component Naming read only speaker_model:, so a parakeet-cpp bundle that sets speaker_component:voice never received the registered voices, in live transcription or in diarization. Resolve speaker_component: when speaker_model: is not set. A component is not an encoder file, so no file-name tag is derived for it: voices with a hash or family identity and untagged voices reach the backend, and a voice with only a file-name tag matches through the new optional speaker_tag: alias. speaker_model: still wins, including when it points at the bundle file. The bundle gallery entries do not declare speaker_recognition: that usecase means the VoiceEmbed and VoiceVerify RPCs, which the backend does not serve, and it selects the default model for /v1/voice/*. The pinning test now says so. Assisted-by: Claude Code:claude-sonnet-5-5 go golangci-lint --- core/gallery/parakeet_bundle_entries_test.go | 6 ++ core/http/endpoints/openai/diarization.go | 22 +++--- .../http/endpoints/openai/diarization_test.go | 42 +++++++++++ .../openai/realtime_live_voices_test.go | 10 +++ core/http/endpoints/openai/realtime_model.go | 2 +- .../services/voicerecognition/known_voices.go | 72 ++++++++++++++++--- .../voicerecognition/known_voices_test.go | 65 +++++++++++++++++ 7 files changed, 201 insertions(+), 18 deletions(-) diff --git a/core/gallery/parakeet_bundle_entries_test.go b/core/gallery/parakeet_bundle_entries_test.go index f7558ae72..95c96a8ea 100644 --- a/core/gallery/parakeet_bundle_entries_test.go +++ b/core/gallery/parakeet_bundle_entries_test.go @@ -65,12 +65,18 @@ var _ = Describe("gallery/index.yaml parakeet-cpp bundle entries", func() { } }) + // speaker_recognition is deliberately absent: that usecase means the model + // answers the /v1/voice/* RPCs (VoiceEmbed, VoiceVerify), which the + // parakeet-cpp backend does not implement, and it picks the default model + // for those routes. Naming registered speakers works from the + // speaker_component option alone, with no usecase. It("declares the usecases and options of the roles each bundle covers", func() { all := entries() for name, w := range want { e := all[name] Expect(e.Overrides["known_usecases"]).To(ConsistOf(toAny(w.usecases)...), name) Expect(e.Overrides["options"]).To(ConsistOf(toAny(w.options)...), name) + Expect(e.Overrides["known_usecases"]).ToNot(ContainElement("speaker_recognition"), name) } }) diff --git a/core/http/endpoints/openai/diarization.go b/core/http/endpoints/openai/diarization.go index bd12bfadb..db3329dee 100644 --- a/core/http/endpoints/openai/diarization.go +++ b/core/http/endpoints/openai/diarization.go @@ -211,25 +211,31 @@ func warnOnce(key string) bool { } // selectKnownVoices returns the registered voices a backend may use to name -// speakers, or nil when the model has no speaker_model, there is no registry, -// or the registry cannot be read. It never fails the caller: unnamed speakers +// speakers, or nil when the model names no speaker encoder (speaker_model: or, +// for a bundle file, speaker_component:), there is no registry, or the +// registry cannot be read. It never fails the caller: unnamed speakers // are the fallback. feature only prefixes the log messages. func selectKnownVoices(ctx context.Context, feature string, options []string, registry voicerecognition.Registry) []voicerecognition.KnownVoice { - sm := voicerecognition.SpeakerModelFromOptions(options) - if sm == "" || registry == nil { + enc := voicerecognition.SpeakerEncoderFromOptions(options) + if enc.Ref == "" || registry == nil { return nil } - sel, err := voicerecognition.KnownVoicesFor(ctx, registry, sm) + sm := enc.Ref + // A speaker_component names a part of the bundle, not an encoder file, so + // enc.File is empty then: voices with a hash or family identity and + // untagged voices reach the backend, and a file-name tag matches only + // through the speaker_tag alias. + sel, err := voicerecognition.KnownVoicesFor(ctx, registry, enc.File, enc.Tags...) if err != nil { xlog.Warn(feature+": could not read the voice registry; speakers stay unnamed", "error", err) return nil } if len(sel.Voices) == 0 && sel.OtherEncoder > 0 { - msg := feature + ": registered voices were made with a different encoder than this model's speaker_model; speakers stay unnamed" + msg := feature + ": registered voices were made with a different encoder than this model's speaker_model or speaker_component; speakers stay unnamed" if warnOnce(feature + "|" + sm) { - xlog.Warn(msg, "speaker_model", sm, "voices_from_other_encoder", sel.OtherEncoder) + xlog.Warn(msg, "speaker", sm, "voices_from_other_encoder", sel.OtherEncoder) } else { - xlog.Debug(msg, "speaker_model", sm, "voices_from_other_encoder", sel.OtherEncoder) + xlog.Debug(msg, "speaker", sm, "voices_from_other_encoder", sel.OtherEncoder) } } return sel.Voices diff --git a/core/http/endpoints/openai/diarization_test.go b/core/http/endpoints/openai/diarization_test.go index 6a27a04d5..bdab43200 100644 --- a/core/http/endpoints/openai/diarization_test.go +++ b/core/http/endpoints/openai/diarization_test.go @@ -109,4 +109,46 @@ var _ = Describe("attachKnownVoices", func() { fakeVoiceRegistry{entries: []voicerecognition.Entry{ada}}) Expect(req.KnownVoices).To(BeEmpty()) }) + + Context("with a bundle model", func() { + const hash = "sha256:72040372aa" + hashed := voicerecognition.Entry{Metadata: voicerecognition.Metadata{ID: "h", Name: "Hashed", Model: hash}, Embedding: []float32{1, 0}} + legacy := voicerecognition.Entry{Metadata: voicerecognition.Metadata{ID: "l", Name: "Legacy", Model: "voice-detect-wespeaker-resnet34.gguf"}, Embedding: []float32{1, 0}} + other := voicerecognition.Entry{Metadata: voicerecognition.Metadata{ID: "o", Name: "Other", Model: "voice-detect-ecapa-tdnn-voxceleb.gguf"}, Embedding: []float32{1, 0}} + reg := fakeVoiceRegistry{entries: []voicerecognition.Entry{hashed, legacy, other}} + names := func(options ...string) []string { + var out []string + for _, v := range selectKnownVoices(context.Background(), "test", options, reg) { + out = append(out, v.Name) + } + return out + } + + It("sends only the hash-tagged voice for a speaker_component", func() { + Expect(names("diar_component:diar", "speaker_component:voice")).To(Equal([]string{"Hashed"})) + }) + It("adds the voices tagged with the speaker_tag alias", func() { + Expect(names("speaker_component:voice", "speaker_tag:voice-detect-wespeaker-resnet34.gguf")). + To(Equal([]string{"Hashed", "Legacy"})) + }) + It("does not guess a tag from the bundle file for a speaker_component", func() { + Expect(names("speaker_component:voice", "speaker_tag:bundle.gguf")).To(Equal([]string{"Hashed"})) + }) + It("keeps speaker_model pointing at the bundle file working", func() { + Expect(names("speaker_model:parakeet-cpp/bundle.gguf")).To(Equal([]string{"Hashed"})) + }) + It("lets speaker_model win over speaker_component", func() { + Expect(names("speaker_component:voice", "speaker_model:voice-detect-ecapa-tdnn-voxceleb.gguf")). + To(Equal([]string{"Hashed", "Other"})) + }) + It("names nothing for a bare speaker_tag", func() { + Expect(names("speaker_tag:voice-detect-wespeaker-resnet34.gguf")).To(BeEmpty()) + }) + It("fills a diarization request", func() { + req := backend.DiarizationRequest{} + attachKnownVoices(context.Background(), &req, []string{"speaker_component:voice"}, reg) + Expect(req.KnownVoices).To(HaveLen(1)) + Expect(req.KnownVoices[0].Name).To(Equal("Hashed")) + }) + }) }) diff --git a/core/http/endpoints/openai/realtime_live_voices_test.go b/core/http/endpoints/openai/realtime_live_voices_test.go index c30679a69..8e9c589bf 100644 --- a/core/http/endpoints/openai/realtime_live_voices_test.go +++ b/core/http/endpoints/openai/realtime_live_voices_test.go @@ -29,6 +29,16 @@ var _ = Describe("liveVoiceOptions", func() { Expect(liveVoiceOptions(context.Background(), nil, cfgWith("speaker_model:enc.gguf"))).To(BeEmpty()) Expect(liveVoiceOptions(context.Background(), fakeVoiceRegistry{err: errors.New("boom")}, cfgWith("speaker_model:enc.gguf"))).To(BeEmpty()) }) + It("opens a bundle session through its speaker_component", func() { + hashed := voicerecognition.Entry{Metadata: voicerecognition.Metadata{Name: "Ada", Model: "sha256:72040372aa"}, Embedding: []float32{1, 0}} + legacy := voicerecognition.Entry{Metadata: voicerecognition.Metadata{Name: "Bob", Model: "voice-detect-wespeaker-resnet34.gguf"}, Embedding: []float32{1, 0}} + reg := fakeVoiceRegistry{entries: []voicerecognition.Entry{legacy, hashed}} + Expect(liveVoiceOptions(context.Background(), reg, cfgWith("vad:true", "speaker_component:voice"))).To(HaveLen(1)) + // A tag-only voice alone is refused without the alias, accepted with it. + tagOnly := fakeVoiceRegistry{entries: []voicerecognition.Entry{legacy}} + Expect(liveVoiceOptions(context.Background(), tagOnly, cfgWith("speaker_component:voice"))).To(BeEmpty()) + Expect(liveVoiceOptions(context.Background(), tagOnly, cfgWith("speaker_component:voice", "speaker_tag:voice-detect-wespeaker-resnet34.gguf"))).To(HaveLen(1)) + }) }) var _ = Describe("transcription segment event", func() { diff --git a/core/http/endpoints/openai/realtime_model.go b/core/http/endpoints/openai/realtime_model.go index dfc49112a..0e66e7751 100644 --- a/core/http/endpoints/openai/realtime_model.go +++ b/core/http/endpoints/openai/realtime_model.go @@ -1220,7 +1220,7 @@ func newModel(pipeline *config.Pipeline, cl *config.ModelConfigLoader, ml *model } // liveVoiceOptions selects the registered voices a live session may name speakers -// with. It stays empty without a speaker_model or a voice registry. +// with. It stays empty without a speaker_model or speaker_component, or a voice registry. func liveVoiceOptions(ctx context.Context, registry voicerecognition.Registry, cfg *config.ModelConfig) []backend.LiveOption { voices := selectKnownVoices(ctx, "live transcription", cfg.Options, registry) if len(voices) == 0 { diff --git a/core/services/voicerecognition/known_voices.go b/core/services/voicerecognition/known_voices.go index c01dfbd53..aa823ea55 100644 --- a/core/services/voicerecognition/known_voices.go +++ b/core/services/voicerecognition/known_voices.go @@ -22,18 +22,61 @@ type KnownVoice struct { Weights string } -// SpeakerModelFromOptions returns the value of a speaker_model: entry in -// a model config's options, or "" when there is none. -func SpeakerModelFromOptions(options []string) string { +// optionValue returns the trimmed value of the first key:value entry in +// options whose key is key, or "" when there is none. +func optionValue(options []string, key string) string { for _, o := range options { k, v, ok := strings.Cut(o, ":") - if ok && strings.TrimSpace(k) == "speaker_model" { + if ok && strings.TrimSpace(k) == key { return strings.TrimSpace(v) } } return "" } +// SpeakerModelFromOptions returns the speaker encoder a model config names, or +// "" when it names none. A speaker_model: entry wins. Without one, a +// speaker_component: entry names a component of the model's own bundle +// file (parakeet-cpp), and its value is returned. A component name is not a +// file name: it must not be used as an encoder tag (see SpeakerTagFromOptions). +func SpeakerModelFromOptions(options []string) string { + if v := optionValue(options, "speaker_model"); v != "" { + return v + } + return optionValue(options, "speaker_component") +} + +// SpeakerTagFromOptions returns the lowercased value of a speaker_tag: +// entry, or "". It is an extra encoder tag for legacy voices that carry only +// a file-name tag. A bundle has no encoder file name of its own, so such a +// voice matches a bundle's speaker component only through this alias. +func SpeakerTagFromOptions(options []string) string { + return strings.ToLower(optionValue(options, "speaker_tag")) +} + +// SpeakerEncoder is the speaker encoder a model config names, as far as voice +// selection needs it. +type SpeakerEncoder struct { + // Ref is what the config names: the speaker_model file, else the + // speaker_component name. Empty when the config names no encoder. + Ref string + // File is the speaker_model path. Empty for a speaker_component, which + // names a part of the model's own bundle file, not an encoder file. + File string + // Tags are extra encoder tags from speaker_tag. + Tags []string +} + +// SpeakerEncoderFromOptions reads the speaker encoder from a model config's +// options. See SpeakerModelFromOptions and SpeakerTagFromOptions. +func SpeakerEncoderFromOptions(options []string) SpeakerEncoder { + enc := SpeakerEncoder{Ref: SpeakerModelFromOptions(options), File: optionValue(options, "speaker_model")} + if t := SpeakerTagFromOptions(options); t != "" { + enc.Tags = []string{t} + } + return enc +} + // EncoderTag is how an encoder is identified: the lowercased base name of its // model file. A voice registered through the voice-detect backend carries the // backend's model name, which defaults to that base name. @@ -53,8 +96,19 @@ type KnownVoiceSelection struct { // filtering belongs to the loaded backend. Tagged candidates precede untagged // ones, each ordered by registration ID so registry iteration order cannot // change replay order. The input is not modified. -func SelectKnownVoices(entries []Entry, speakerModelPath string) KnownVoiceSelection { - tag := EncoderTag(speakerModelPath) +// +// extraTags are further tags that count as the loaded encoder (the speaker_tag +// option). They are compared as lowercase names, like EncoderTag. +func SelectKnownVoices(entries []Entry, speakerModelPath string, extraTags ...string) KnownVoiceSelection { + tags := make(map[string]bool, len(extraTags)+1) + if speakerModelPath != "" { + tags[EncoderTag(speakerModelPath)] = true + } + for _, t := range extraTags { + if t = strings.ToLower(strings.TrimSpace(t)); t != "" { + tags[t] = true + } + } var sel KnownVoiceSelection var untagged []Entry for _, e := range entries { @@ -66,7 +120,7 @@ func SelectKnownVoices(entries []Entry, speakerModelPath string) KnownVoiceSelec untagged = append(untagged, e) // Hash-tagged portable registrations are checked against the loaded // encoder by the backend, never against a filename or dimension alone. - case strings.HasPrefix(e.Metadata.Model, "sha256:"), EncoderTag(e.Metadata.Model) == tag: + case strings.HasPrefix(e.Metadata.Model, "sha256:"), tags[EncoderTag(e.Metadata.Model)]: sel.Voices = append(sel.Voices, knownVoice(e)) default: sel.OtherEncoder++ @@ -90,10 +144,10 @@ func knownVoice(e Entry) KnownVoice { } // KnownVoicesFor lists the registry and selects the voices for a speaker model. -func KnownVoicesFor(ctx context.Context, reg Registry, speakerModelPath string) (KnownVoiceSelection, error) { +func KnownVoicesFor(ctx context.Context, reg Registry, speakerModelPath string, extraTags ...string) (KnownVoiceSelection, error) { entries, err := reg.List(ctx) if err != nil { return KnownVoiceSelection{}, err } - return SelectKnownVoices(entries, speakerModelPath), nil + return SelectKnownVoices(entries, speakerModelPath, extraTags...), nil } diff --git a/core/services/voicerecognition/known_voices_test.go b/core/services/voicerecognition/known_voices_test.go index 6039a1a47..7b0ad7217 100644 --- a/core/services/voicerecognition/known_voices_test.go +++ b/core/services/voicerecognition/known_voices_test.go @@ -26,6 +26,34 @@ var _ = Describe("SpeakerModelFromOptions", func() { Expect(voicerecognition.SpeakerModelFromOptions([]string{"diarization_model:d.gguf"})).To(BeEmpty()) Expect(voicerecognition.SpeakerModelFromOptions(nil)).To(BeEmpty()) }) + It("falls back to speaker_component when there is no speaker_model", func() { + Expect(voicerecognition.SpeakerModelFromOptions([]string{"diar_component:diar", "speaker_component: voice "})).To(Equal("voice")) + }) + It("prefers speaker_model over speaker_component, in any order", func() { + Expect(voicerecognition.SpeakerModelFromOptions([]string{"speaker_component:voice", "speaker_model:bundle.gguf"})).To(Equal("bundle.gguf")) + Expect(voicerecognition.SpeakerModelFromOptions([]string{"speaker_model:bundle.gguf", "speaker_component:voice"})).To(Equal("bundle.gguf")) + }) + It("ignores empty values", func() { + Expect(voicerecognition.SpeakerModelFromOptions([]string{"speaker_model:", "speaker_component:voice"})).To(Equal("voice")) + Expect(voicerecognition.SpeakerModelFromOptions([]string{"speaker_model:", "speaker_component:"})).To(BeEmpty()) + }) +}) + +var _ = Describe("SpeakerEncoderFromOptions", func() { + It("has a file for speaker_model and none for a component", func() { + Expect(voicerecognition.SpeakerEncoderFromOptions([]string{"speaker_model:enc.gguf"})). + To(Equal(voicerecognition.SpeakerEncoder{Ref: "enc.gguf", File: "enc.gguf"})) + Expect(voicerecognition.SpeakerEncoderFromOptions([]string{"speaker_component:voice"})). + To(Equal(voicerecognition.SpeakerEncoder{Ref: "voice"})) + }) + It("reads the speaker_tag alias in lowercase", func() { + Expect(voicerecognition.SpeakerEncoderFromOptions([]string{"speaker_component:voice", "speaker_tag: Voice-Detect-WeSpeaker.gguf"}).Tags). + To(Equal([]string{"voice-detect-wespeaker.gguf"})) + Expect(voicerecognition.SpeakerEncoderFromOptions([]string{"speaker_component:voice"}).Tags).To(BeEmpty()) + }) + It("names no encoder for a speaker_tag alone", func() { + Expect(voicerecognition.SpeakerEncoderFromOptions([]string{"speaker_tag:x.gguf"}).Ref).To(BeEmpty()) + }) }) var _ = Describe("EncoderTag", func() { @@ -146,6 +174,43 @@ var _ = Describe("encoder fingerprint of selected voices", func() { }) }) +var _ = Describe("SelectKnownVoices for a bundle component", func() { + const hash = "sha256:72040372aa" + mixed := func() []voicerecognition.Entry { + return []voicerecognition.Entry{ + entry("hashed", hash, 1, 0), + entry("legacy-wespeaker", "voice-detect-wespeaker-resnet34.gguf", 1, 0), + entry("legacy-ecapa", "voice-detect-ecapa-tdnn-voxceleb.gguf", 1, 0), + entry("old", "", 1, 0), + } + } + ids := func(sel voicerecognition.KnownVoiceSelection) []string { + var out []string + for _, v := range sel.Voices { + out = append(out, v.ID) + } + return out + } + + It("sends hash-tagged and untagged voices, and refuses tag-only voices, without a tag alias", func() { + sel := voicerecognition.SelectKnownVoices(mixed(), "") + Expect(ids(sel)).To(Equal([]string{"hashed", "old"})) + Expect(sel.OtherEncoder).To(Equal(2)) + }) + It("matches a tag-only voice through the alias and no other", func() { + sel := voicerecognition.SelectKnownVoices(mixed(), "", "Voice-Detect-WeSpeaker-ResNet34.gguf") + Expect(ids(sel)).To(Equal([]string{"hashed", "legacy-wespeaker", "old"})) + Expect(sel.OtherEncoder).To(Equal(1)) + }) + It("adds the alias to the file tag of a speaker_model", func() { + sel := voicerecognition.SelectKnownVoices(mixed(), "bundle.gguf", "voice-detect-ecapa-tdnn-voxceleb.gguf") + Expect(ids(sel)).To(Equal([]string{"hashed", "legacy-ecapa", "old"})) + }) + It("ignores a blank alias", func() { + Expect(ids(voicerecognition.SelectKnownVoices(mixed(), "", " ", ""))).To(Equal([]string{"hashed", "old"})) + }) +}) + var _ = Describe("KnownVoicesFor", func() { It("selects from the registry listing", func() { sel, err := voicerecognition.KnownVoicesFor(context.Background(), listRegistry{entries: []voicerecognition.Entry{entry("ada", "m.gguf", 1)}}, "m.gguf")