diff --git a/backend/backend.proto b/backend/backend.proto index d7b1e112d..614f006e9 100644 --- a/backend/backend.proto +++ b/backend/backend.proto @@ -1160,6 +1160,12 @@ message VoiceEmbedRequest { message VoiceEmbedResponse { repeated float embedding = 1; string model = 2; + // Fingerprint of the encoder that made the embedding, when the backend can + // tell: the embedding space ("voicedetect:::") and the + // weights ("sha256:" of the model file). Both empty when unknown, and + // old backends never set them. + string encoder_family = 3; + string encoder_weights = 4; } message ToolFormatMarkers { diff --git a/backend/go/voice-detect/govoicedetect.go b/backend/go/voice-detect/govoicedetect.go index 2bbe74bd0..3a780fc87 100644 --- a/backend/go/voice-detect/govoicedetect.go +++ b/backend/go/voice-detect/govoicedetect.go @@ -1,9 +1,12 @@ package main import ( + "crypto/sha256" + "encoding/hex" "encoding/json" "errors" "fmt" + "io" "math" "os" "path/filepath" @@ -36,6 +39,15 @@ var ( CppEmbedPCM func(ctx uintptr, pcm []float32, nSamples, sampleRate int32, outVec, outDim unsafe.Pointer) int32 CppVerifyPaths func(ctx uintptr, a, b string, threshold float32, outDistance, outVerified unsafe.Pointer) int32 CppAnalyzeJSON func(ctx uintptr, wavPath string) uintptr + + // Encoder identity (additive in voice-detect.cpp, ABI version stays 1). They + // stay nil on a libvoicedetect.so from before it, which is probed in main.go. + // Each returns a pointer BORROWED from the context: read it with + // goStringFromCPtr right away and never free it. It is valid until + // voicedetect_capi_free. NULL means unavailable. + CppEncoderArch func(ctx uintptr) uintptr + CppEncoderName func(ctx uintptr) uintptr + CppEncoderFamily func(ctx uintptr) uintptr ) // VoiceDetect implements the speaker-recognition voice subset of the Backend @@ -46,6 +58,12 @@ type VoiceDetect struct { base.SingleThread opts loadOptions ctxPtr uintptr + + // Encoder fingerprint, read once at load. Empty when the library cannot + // report it (older libvoicedetect.so) or, for weights, when the model is + // not a readable file. + encoderFamily string + encoderWeights string } func (v *VoiceDetect) Load(opts *pb.ModelOptions) error { @@ -89,9 +107,55 @@ func (v *VoiceDetect) Load(opts *pb.ModelOptions) error { return fmt.Errorf("voice-detect: voicedetect_capi_load failed for %q", model) } v.ctxPtr = ctx + v.encoderFamily = readEncoderFamily(ctx) + v.encoderWeights = fileIdentity(model) + xlog.Info("voice-detect: encoder fingerprint", "family", v.encoderFamily, "weights", v.encoderWeights, + "arch", readBorrowed(CppEncoderArch, ctx), "name", readBorrowed(CppEncoderName, ctx)) return nil } +// readBorrowed copies a string the library keeps on the context. fn is nil on a +// library without the accessor. A NULL or empty result means unavailable. The +// pointer is borrowed: it is copied here and never freed. +func readBorrowed(fn func(ctx uintptr) uintptr, ctx uintptr) string { + if fn == nil || ctx == 0 { + return "" + } + return goStringFromCPtr(fn(ctx)) +} + +// readEncoderFamily returns "voicedetect:::", or "" when the +// library cannot report it. A family with no part at all (":::") carries no +// information and counts as unavailable. +func readEncoderFamily(ctx uintptr) string { + family := readBorrowed(CppEncoderFamily, ctx) + if strings.Trim(family, ":") == "" { + return "" + } + return family +} + +// fileIdentity is "sha256:" of the bytes of the model file, streamed so a +// large file is never held in memory. It runs once per model load. It returns "" +// when the path is not a readable regular file; the backend then reports no +// weights identity and only the family fingerprints the voice. +func fileIdentity(path string) string { + f, err := os.Open(path) // #nosec G304 -- the model file LocalAI itself passes to the load + if err != nil { + return "" + } + defer func() { _ = f.Close() }() + if st, err := f.Stat(); err != nil || !st.Mode().IsRegular() { + return "" + } + h := sha256.New() + if _, err := io.Copy(h, f); err != nil { + xlog.Warn("voice-detect: could not hash the model file", "error", err) + return "" + } + return "sha256:" + hex.EncodeToString(h.Sum(nil)) +} + // VoiceEmbed returns the L2-normalized speaker embedding for an audio clip. // The request carries a filesystem PATH; the HTTP layer materializes // base64/URL/data-URI inputs to a temp file before the gRPC call. @@ -106,7 +170,12 @@ func (v *VoiceDetect) VoiceEmbed(req *pb.VoiceEmbedRequest) (pb.VoiceEmbedRespon if err != nil { return pb.VoiceEmbedResponse{}, err } - return pb.VoiceEmbedResponse{Embedding: emb, Model: v.opts.modelName}, nil + return pb.VoiceEmbedResponse{ + Embedding: emb, + Model: v.opts.modelName, + EncoderFamily: v.encoderFamily, + EncoderWeights: v.encoderWeights, + }, nil } func (v *VoiceDetect) embedPath(path string) ([]float32, error) { diff --git a/backend/go/voice-detect/govoicedetect_test.go b/backend/go/voice-detect/govoicedetect_test.go index 2de7fcc8a..e9d8d99fb 100644 --- a/backend/go/voice-detect/govoicedetect_test.go +++ b/backend/go/voice-detect/govoicedetect_test.go @@ -1,9 +1,13 @@ package main import ( + "crypto/sha256" + "encoding/hex" "os" + "path/filepath" "sync" "testing" + "unsafe" "github.com/ebitengine/purego" pb "github.com/mudler/LocalAI/pkg/grpc/proto" @@ -142,3 +146,118 @@ var _ = Describe("VoiceDetect end-to-end", Ordered, func() { Expect(resp.Distance).To(BeNumerically("<=", resp.Threshold)) }) }) + +// cstr returns a NUL-terminated copy of s and its address, the way the library +// hands out a borrowed char*. keep holds the buffer so the GC does not drop it. +func cstr(s string) (ptr uintptr, keep []byte) { + b := append([]byte(s), 0) + return uintptr(unsafe.Pointer(&b[0])), b +} + +// stubEncoder replaces the encoder accessors with fakes that return the given +// strings (nil pointer when the value is nil) and restores them after the spec. +func stubEncoder(arch, name, family *string) { + oldA, oldN, oldF := CppEncoderArch, CppEncoderName, CppEncoderFamily + DeferCleanup(func() { CppEncoderArch, CppEncoderName, CppEncoderFamily = oldA, oldN, oldF }) + mk := func(s *string) func(uintptr) uintptr { + if s == nil { + return func(uintptr) uintptr { return 0 } + } + ptr, keep := cstr(*s) + return func(uintptr) uintptr { _ = keep; return ptr } + } + CppEncoderArch, CppEncoderName, CppEncoderFamily = mk(arch), mk(name), mk(family) +} + +var _ = Describe("encoder family", func() { + str := func(s string) *string { return &s } + + It("reads the family the library reports", func() { + stubEncoder(str("ecapa_tdnn"), str("speechbrain/spkrec-ecapa-voxceleb"), str("voicedetect:ecapa_tdnn:speechbrain/spkrec-ecapa-voxceleb:192")) + Expect(readEncoderFamily(1)).To(Equal("voicedetect:ecapa_tdnn:speechbrain/spkrec-ecapa-voxceleb:192")) + Expect(readBorrowed(CppEncoderArch, 1)).To(Equal("ecapa_tdnn")) + Expect(readBorrowed(CppEncoderName, 1)).To(Equal("speechbrain/spkrec-ecapa-voxceleb")) + }) + + It("is empty when the library lacks the symbols", func() { + oldA, oldN, oldF := CppEncoderArch, CppEncoderName, CppEncoderFamily + DeferCleanup(func() { CppEncoderArch, CppEncoderName, CppEncoderFamily = oldA, oldN, oldF }) + CppEncoderArch, CppEncoderName, CppEncoderFamily = nil, nil, nil + Expect(readEncoderFamily(1)).To(BeEmpty()) + Expect(readBorrowed(CppEncoderArch, 1)).To(BeEmpty()) + }) + + It("is empty on a NULL pointer", func() { + stubEncoder(nil, nil, nil) + Expect(readEncoderFamily(1)).To(BeEmpty()) + }) + + It("is empty on an empty string and on a family with every field missing", func() { + stubEncoder(str(""), str(""), str("")) + Expect(readEncoderFamily(1)).To(BeEmpty()) + stubEncoder(nil, nil, str(":::")) + Expect(readEncoderFamily(1)).To(BeEmpty()) + }) + + It("keeps a family with a missing field (colons kept)", func() { + stubEncoder(nil, nil, str("voicedetect:ecapa_tdnn::192")) + Expect(readEncoderFamily(1)).To(Equal("voicedetect:ecapa_tdnn::192")) + }) + + It("does not read for a NULL context", func() { + called := false + CppEncoderFamily = func(uintptr) uintptr { called = true; return 0 } + DeferCleanup(func() { CppEncoderFamily = nil }) + Expect(readEncoderFamily(0)).To(BeEmpty()) + Expect(called).To(BeFalse()) + }) +}) + +var _ = Describe("fileIdentity", func() { + It("is the sha256 of the file bytes", func() { + path := filepath.Join(GinkgoT().TempDir(), "m.gguf") + Expect(os.WriteFile(path, []byte("weights"), 0o600)).To(Succeed()) + sum := sha256.Sum256([]byte("weights")) + Expect(fileIdentity(path)).To(Equal("sha256:" + hex.EncodeToString(sum[:]))) + }) + It("is empty for a missing path and for a directory", func() { + dir := GinkgoT().TempDir() + Expect(fileIdentity(filepath.Join(dir, "none.gguf"))).To(BeEmpty()) + Expect(fileIdentity(dir)).To(BeEmpty()) + }) +}) + +var _ = Describe("VoiceEmbed fingerprint", func() { + stubEmbed := func() { + oldP, oldF := CppEmbedPath, CppFreeVec + DeferCleanup(func() { CppEmbedPath, CppFreeVec = oldP, oldF }) + vec := []float32{0.6, 0.8} + CppEmbedPath = func(_ uintptr, _ string, outVec, outDim unsafe.Pointer) int32 { + *(*uintptr)(outVec) = uintptr(unsafe.Pointer(&vec[0])) + *(*int32)(outDim) = int32(len(vec)) + return 0 + } + CppFreeVec = func(uintptr) {} + } + + It("returns the family and the weights next to the embedding", func() { + stubEmbed() + v := &VoiceDetect{ctxPtr: 1, encoderFamily: "voicedetect:a:b:2", encoderWeights: "sha256:ab"} + v.opts.modelName = "m.gguf" + resp, err := v.VoiceEmbed(&pb.VoiceEmbedRequest{Audio: "x.wav"}) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.Embedding).To(Equal([]float32{0.6, 0.8})) + Expect(resp.Model).To(Equal("m.gguf")) + Expect(resp.EncoderFamily).To(Equal("voicedetect:a:b:2")) + Expect(resp.EncoderWeights).To(Equal("sha256:ab")) + }) + + It("leaves both empty on a library that cannot report them", func() { + stubEmbed() + v := &VoiceDetect{ctxPtr: 1} + resp, err := v.VoiceEmbed(&pb.VoiceEmbedRequest{Audio: "x.wav"}) + Expect(err).ToNot(HaveOccurred()) + Expect(resp.EncoderFamily).To(BeEmpty()) + Expect(resp.EncoderWeights).To(BeEmpty()) + }) +}) diff --git a/backend/go/voice-detect/main.go b/backend/go/voice-detect/main.go index 35421b5c3..617b8264c 100644 --- a/backend/go/voice-detect/main.go +++ b/backend/go/voice-detect/main.go @@ -54,6 +54,19 @@ func main() { purego.RegisterLibFunc(lf.FuncPtr, lib, lf.Name) } + // Encoder identity accessors (additive in voice-detect.cpp, no ABI bump). + // Probed so a libvoicedetect.so from before them still loads: the encoder + // family then stays empty and audio-registered voices stay unfingerprinted. + for _, lf := range []LibFuncs{ + {&CppEncoderArch, "voicedetect_capi_encoder_arch"}, + {&CppEncoderName, "voicedetect_capi_encoder_name"}, + {&CppEncoderFamily, "voicedetect_capi_encoder_family"}, + } { + if sym, err := purego.Dlsym(lib, lf.Name); err == nil && sym != 0 { + purego.RegisterLibFunc(lf.FuncPtr, lib, lf.Name) + } + } + fmt.Fprintf(os.Stderr, "[voice-detect] ABI=%d\n", CppAbiVersion()) flag.Parse() diff --git a/core/http/endpoints/localai/voice_identify.go b/core/http/endpoints/localai/voice_identify.go index dac259b59..c1bf52deb 100644 --- a/core/http/endpoints/localai/voice_identify.go +++ b/core/http/endpoints/localai/voice_identify.go @@ -74,6 +74,12 @@ func VoiceIdentifyEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, if trustedErr != nil || m.Metadata.Model != trusted.Identity || len(embed.GetEmbedding()) != trusted.Dimension { continue } + } else if m.Metadata.EncoderFamily != "" && embed.GetEncoderFamily() != "" { + // Both sides carry a fingerprint: the embedding space decides, not + // the file name. + if m.Metadata.EncoderFamily != embed.GetEncoderFamily() { + continue + } } else if m.Metadata.Model != "" && voicerecognition.EncoderTag(m.Metadata.Model) != voicerecognition.EncoderTag(embed.GetModel()) { continue } diff --git a/core/http/endpoints/localai/voice_register.go b/core/http/endpoints/localai/voice_register.go index d4a96dd42..0c81bf0fa 100644 --- a/core/http/endpoints/localai/voice_register.go +++ b/core/http/endpoints/localai/voice_register.go @@ -34,7 +34,7 @@ func VoiceRegisterEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, } var embedding []float32 - var encoder, family string + var encoder, family, weights string if input.SpeakerProfiles != nil { if input.Audio != "" || input.SpeakerSlot == nil { return echo.NewHTTPError(http.StatusBadRequest, "speaker_profiles requires speaker_slot and excludes audio") @@ -62,11 +62,14 @@ func VoiceRegisterEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, return mapBackendError(err) } embedding, encoder = res.GetEmbedding(), res.GetModel() + // Fingerprint reported by the voice backend. Empty from a backend that + // cannot report it: the voice then stays unfingerprinted. + family, weights = res.GetEncoderFamily(), res.GetEncoderWeights() } - meta := voiceMetadata(input.Name, input.Labels, encoder) - // Only the portable route knows the family: it comes from the loaded - // encoder. A voice-detect embedding has none, so it stays unfingerprinted. - meta.EncoderFamily = family + // The family comes from the loaded encoder (portable route) or from the + // voice backend that embedded the audio. A backend that cannot report + // one leaves it empty and the voice stays unfingerprinted. + meta := voiceMetadata(input.Name, input.Labels, encoder, family, weights) stored, err := registry.Register(c.Request().Context(), embedding, meta) if err != nil { return err @@ -81,7 +84,8 @@ func VoiceRegisterEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, // voiceMetadata is what a registration stores next to the embedding. Model is // the speaker encoder that produced it, so a consumer with a different encoder -// can tell the vectors are not comparable. -func voiceMetadata(name string, labels map[string]string, embedderModel string) voicerecognition.Metadata { - return voicerecognition.Metadata{Name: name, Labels: labels, Model: embedderModel} +// can tell the vectors are not comparable. family and weights fingerprint the +// encoder when it reported them (see voicerecognition.Metadata), "" otherwise. +func voiceMetadata(name string, labels map[string]string, embedderModel, family, weights string) voicerecognition.Metadata { + return voicerecognition.Metadata{Name: name, Labels: labels, Model: embedderModel, EncoderFamily: family, EncoderWeights: weights} } diff --git a/core/http/endpoints/localai/voice_register_test.go b/core/http/endpoints/localai/voice_register_test.go index b66d0e48d..9d97f516a 100644 --- a/core/http/endpoints/localai/voice_register_test.go +++ b/core/http/endpoints/localai/voice_register_test.go @@ -7,12 +7,23 @@ import ( var _ = Describe("voiceMetadata", func() { It("carries the name, the labels and the encoder that embedded the voice", func() { - m := voiceMetadata("ada", map[string]string{"team": "a"}, "voice-detect-wespeaker-resnet34.gguf") + m := voiceMetadata("ada", map[string]string{"team": "a"}, "voice-detect-wespeaker-resnet34.gguf", "", "") Expect(m.Name).To(Equal("ada")) Expect(m.Labels).To(Equal(map[string]string{"team": "a"})) Expect(m.Model).To(Equal("voice-detect-wespeaker-resnet34.gguf")) }) It("leaves the tag empty when the backend did not say", func() { - Expect(voiceMetadata("ada", nil, "").Model).To(BeEmpty()) + Expect(voiceMetadata("ada", nil, "", "", "").Model).To(BeEmpty()) + }) + It("records the family and the weights the voice backend reported, and keeps the name in model", func() { + m := voiceMetadata("ada", nil, "ecapa.gguf", "voicedetect:ecapa_tdnn:speechbrain/spkrec-ecapa-voxceleb:192", "sha256:ab") + Expect(m.Model).To(Equal("ecapa.gguf")) + Expect(m.EncoderFamily).To(Equal("voicedetect:ecapa_tdnn:speechbrain/spkrec-ecapa-voxceleb:192")) + Expect(m.EncoderWeights).To(Equal("sha256:ab")) + }) + It("leaves the voice unfingerprinted when the backend reported nothing", func() { + m := voiceMetadata("ada", nil, "ecapa.gguf", "", "") + Expect(m.EncoderFamily).To(BeEmpty()) + Expect(m.EncoderWeights).To(BeEmpty()) }) }) diff --git a/core/services/voicerecognition/known_voices.go b/core/services/voicerecognition/known_voices.go index aa823ea55..bf0ba233c 100644 --- a/core/services/voicerecognition/known_voices.go +++ b/core/services/voicerecognition/known_voices.go @@ -118,9 +118,13 @@ func SelectKnownVoices(entries []Entry, speakerModelPath string, extraTags ...st switch { case e.Metadata.Model == "": untagged = append(untagged, e) - // Hash-tagged portable registrations are checked against the loaded - // encoder by the backend, never against a filename or dimension alone. - case strings.HasPrefix(e.Metadata.Model, "sha256:"), tags[EncoderTag(e.Metadata.Model)]: + case e.Metadata.EncoderFamily != "", strings.HasPrefix(e.Metadata.Model, "sha256:"): + // A voice with an encoder family is checked by the backend against the + // loaded encoder's family, not by file name: the same encoder can be + // converted under another name, and a different one must be refused + // by the backend with a clear error. + sel.Voices = append(sel.Voices, knownVoice(e)) + case tags[EncoderTag(e.Metadata.Model)]: sel.Voices = append(sel.Voices, knownVoice(e)) default: sel.OtherEncoder++ @@ -137,7 +141,10 @@ func SelectKnownVoices(entries []Entry, speakerModelPath string, extraTags ...st func knownVoice(e Entry) KnownVoice { v := KnownVoice{ID: e.Metadata.ID, Name: e.Metadata.Name, Embedding: e.Embedding, Model: e.Metadata.Model, Family: e.Metadata.EncoderFamily} - if strings.HasPrefix(e.Metadata.Model, "sha256:") { + switch { + case e.Metadata.EncoderWeights != "": + v.Weights = e.Metadata.EncoderWeights + case strings.HasPrefix(e.Metadata.Model, "sha256:"): v.Weights = e.Metadata.Model } return v diff --git a/core/services/voicerecognition/known_voices_test.go b/core/services/voicerecognition/known_voices_test.go index 7b0ad7217..52a3b40a6 100644 --- a/core/services/voicerecognition/known_voices_test.go +++ b/core/services/voicerecognition/known_voices_test.go @@ -156,6 +156,23 @@ var _ = Describe("encoder fingerprint of selected voices", func() { Expect(sel.Voices[0].Family).To(Equal("voicedetect:ecapa_tdnn:ecapa:192")) Expect(sel.Voices[0].Weights).To(Equal(hash)) }) + It("forwards an audio-registered voice by its family and weights, whatever its file name", func() { + e := entry("ada", "voice-detect-ecapa.gguf", 1, 0) + e.Metadata.EncoderFamily = "voicedetect:ecapa_tdnn:ecapa:192" + e.Metadata.EncoderWeights = hash + sel := voicerecognition.SelectKnownVoices([]voicerecognition.Entry{e}, "other-encoder.gguf") + Expect(sel.OtherEncoder).To(BeZero()) + Expect(sel.Voices).To(HaveLen(1)) + Expect(sel.Voices[0].Model).To(Equal("voice-detect-ecapa.gguf")) + Expect(sel.Voices[0].Family).To(Equal("voicedetect:ecapa_tdnn:ecapa:192")) + Expect(sel.Voices[0].Weights).To(Equal(hash)) + }) + It("keeps an old audio voice without a family on the file-name filter", func() { + old := entry("ada", "voice-detect-ecapa.gguf", 1, 0) + sel := voicerecognition.SelectKnownVoices([]voicerecognition.Entry{old}, "other-encoder.gguf") + Expect(sel.Voices).To(BeEmpty()) + Expect(sel.OtherEncoder).To(Equal(1)) + }) It("leaves a file-name tagged voice unfingerprinted", func() { sel := voicerecognition.SelectKnownVoices([]voicerecognition.Entry{entry("ada", "spk.gguf", 1, 0)}, "spk.gguf") Expect(sel.Voices[0].Family).To(BeEmpty()) @@ -171,6 +188,7 @@ var _ = Describe("encoder fingerprint of selected voices", func() { Expect(fresh.EncoderFamily).To(Equal("f")) raw, _ = json.Marshal(old) Expect(string(raw)).ToNot(ContainSubstring("encoder_family")) + Expect(string(raw)).ToNot(ContainSubstring("encoder_weights")) }) }) diff --git a/core/services/voicerecognition/registry.go b/core/services/voicerecognition/registry.go index 6e51272f2..20b07db7a 100644 --- a/core/services/voicerecognition/registry.go +++ b/core/services/voicerecognition/registry.go @@ -61,8 +61,14 @@ type Metadata struct { // EncoderFamily is the embedding space of the encoder ("voicedetect:::"), // recorded when the encoder reports it (portable enrollment from speaker // profiles). Empty when unknown, and for voices registered before it existed. - // Model then holds the weights identity ("sha256:") for the same voices. + // For the speaker-profiles route Model then holds the weights identity + // ("sha256:") of the same voices. EncoderFamily string `json:"encoder_family,omitempty"` + // EncoderWeights is the "sha256:" identity of the encoder's model file, + // recorded for voices registered from audio when the voice backend reports + // it. Model keeps the encoder's name for those voices, so the 1:N filter on + // the name keeps working. Empty when unknown. + EncoderWeights string `json:"encoder_weights,omitempty"` } // Match is a single result from Identify, ranked by similarity. diff --git a/docs/content/features/voice-recognition.md b/docs/content/features/voice-recognition.md index be1e12043..500b4b948 100644 --- a/docs/content/features/voice-recognition.md +++ b/docs/content/features/voice-recognition.md @@ -344,21 +344,48 @@ A voice with only a weights identity takes the family of the loaded encoder when the weights are the same file. A voice with a different weights hash and no family is dropped, as before. -A voice registered from audio through the voice-detect backend has no -fingerprint: libvoicedetect reports no architecture or model name, so the -backend cannot tell the family, and only the file-name tag described above -applies. Such voices and fingerprinted voices cannot share one registry in the -library. When a request has any unfingerprinted voice, all of its voices are -used without the fingerprint check (the old behaviour). To get the check, enroll -every voice from `speaker_profiles`. With `speaker_strict:true` the backend -ignores unfingerprinted voices, and a request that has only those fails with -the library's message. The family is also reported in the internal backend -status next to the identity. +A voice registered from audio through the voice-detect backend (`POST +/v1/voice/register` with `audio`) is fingerprinted too, when the libvoicedetect +in the backend can report its encoder (voice-detect.cpp from the pin that adds +`voicedetect_capi_encoder_family`). The backend reads the family from the +loaded GGUF, and hashes the model file once at load to get the weights. LocalAI +stores the family as `encoder_family` and the weights as `encoder_weights`; the +`model` field keeps the encoder name as before. If the voice-detect model is +not a plain file the backend cannot hash it and the weights stay empty, so the +family alone fingerprints the voice. When the family is known, LocalAI sends +the voice to the backend whatever its file name, so the family decides. -{{% notice warning %}} -Do not set a `model_name:` option on the voice-detect model config. It -replaces the default name, the voices are then tagged with it, and they no -longer match the `speaker_model:` file. Keep the default name. +A voice-detect backend built against an older libvoicedetect cannot report +the family, but it still records the weights. Such a voice is treated as made +by the loaded `speaker_model:` when the weights are the same file, and stays +unfingerprinted when they differ. + +Voices registered before this change have no fingerprint, and so do voices +from a backend that reports neither (the Python speaker-recognition backend). +They are used as described above, by file-name +tag. Such voices and fingerprinted voices cannot share one registry in the +library. When a request has any unfingerprinted voice, all of its voices are +used without the fingerprint check. **Register those voices again** to get +the check. With `speaker_strict:true` the backend ignores unfingerprinted +voices, and a request that has only those fails with the library's message. +The family is also reported in the internal backend status next to the +identity. `/v1/voice/identify` compares the family of a stored voice with the +family of the probe when both have one, and falls back to the file-name tag +when either has none. + +The family is built from the encoder GGUF metadata. For CAM++, WeSpeaker and +ERes2Net the GGUF `general.name` is the path the model was converted from, so +the same encoder converted again under another name has another family, and +voices registered with the first file are refused with the second. Two +fine-tunes that share an architecture, name and embedding size have the same +family; only the weights hash tells them apart, and a hash mismatch alone only +logs a warning. + +{{% notice note %}} +With a fingerprint, the `model_name:` option on the voice-detect model config +no longer breaks naming: the family decides, not the name. Voices registered +without a fingerprint still need the default name to match the +`speaker_model:` file. {{% /notice %}} ### Options @@ -463,7 +490,8 @@ are independent. | `store` | string, optional | vector store model; defaults to local-store | Returns `{id, name, registered_at}`. The `id` is an opaque UUID used -by `/v1/voice/identify` and `/v1/voice/forget`. +by `/v1/voice/identify` and `/v1/voice/forget`. The stored voice records the +encoder [fingerprint](#encoder-fingerprint) when the backend reports it. ### `POST /v1/voice/identify` (1:N recognition)