mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-09 22:54:42 -04:00
feat(voice-detect): report the encoder family so audio-registered voices are fingerprinted (#12499)
* feat(voice-detect): report the encoder family so audio-registered voices are fingerprinted
A voice registered from audio through the voice-detect backend had no
encoder fingerprint, so the parakeet-cpp backend could not tell whether
it was comparable with the loaded speaker model and could only fall back
to the file-name rule.
The voice-detect backend now binds the three new libvoicedetect accessors
with a symbol probe (an older library still loads and reports nothing),
copies the borrowed strings at once and never frees them. It also hashes
the model file once at load. VoiceEmbedResponse gains two optional
fields, encoder_family and encoder_weights ("sha256:<hex>", empty when
the model is not a plain file).
/v1/voice/register stores them as encoder_family and a new
encoder_weights field in the registry entry; model keeps the encoder
name, so the 1:N identify filter by name is unchanged for old entries.
A voice with a family is sent to the backend whatever its file name, and
the backend decides by family. /v1/voice/identify compares the family
when both the stored voice and the probe have one. Old entries load
without the fields and stay unfingerprinted.
Assisted-by: Claude:claude-sonnet-5-5 [Claude Code]
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
* docs(voice-detect): say how a voice with weights but no family is treated
Assisted-by: Claude:claude-sonnet-5-5 [Claude Code]
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
* fix(voice-detect): mark the model file open as intended for gosec
Assisted-by: Claude:claude-sonnet-5-5 [Claude Code]
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
---------
Signed-off-by: Ettore Di Giacinto <mudler@localai.io>
Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
This commit is contained in:
1 parent
0102751b31
commit
6343a2dc6f
11 files changed
+318
-31
No files matched your search
@@ -1160,6 +1160,12 @@ message VoiceEmbedRequest {
|
||||
message VoiceEmbedResponse {
|
||||
repeated float embedding = 1;
|
||||
string model = 2;
|
||||
// Fingerprint of the encoder that made the embedding, when the backend can
|
||||
// tell: the embedding space ("voicedetect:<arch>:<name>:<dim>") and the
|
||||
// weights ("sha256:<hex>" of the model file). Both empty when unknown, and
|
||||
// old backends never set them.
|
||||
string encoder_family = 3;
|
||||
string encoder_weights = 4;
|
||||
}
|
||||
|
||||
message ToolFormatMarkers {
|
||||
|
||||
@@ -1,9 +1,12 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"math"
|
||||
"os"
|
||||
"path/filepath"
|
||||
@@ -36,6 +39,15 @@ var (
|
||||
CppEmbedPCM func(ctx uintptr, pcm []float32, nSamples, sampleRate int32, outVec, outDim unsafe.Pointer) int32
|
||||
CppVerifyPaths func(ctx uintptr, a, b string, threshold float32, outDistance, outVerified unsafe.Pointer) int32
|
||||
CppAnalyzeJSON func(ctx uintptr, wavPath string) uintptr
|
||||
|
||||
// Encoder identity (additive in voice-detect.cpp, ABI version stays 1). They
|
||||
// stay nil on a libvoicedetect.so from before it, which is probed in main.go.
|
||||
// Each returns a pointer BORROWED from the context: read it with
|
||||
// goStringFromCPtr right away and never free it. It is valid until
|
||||
// voicedetect_capi_free. NULL means unavailable.
|
||||
CppEncoderArch func(ctx uintptr) uintptr
|
||||
CppEncoderName func(ctx uintptr) uintptr
|
||||
CppEncoderFamily func(ctx uintptr) uintptr
|
||||
)
|
||||
|
||||
// VoiceDetect implements the speaker-recognition voice subset of the Backend
|
||||
@@ -46,6 +58,12 @@ type VoiceDetect struct {
|
||||
base.SingleThread
|
||||
opts loadOptions
|
||||
ctxPtr uintptr
|
||||
|
||||
// Encoder fingerprint, read once at load. Empty when the library cannot
|
||||
// report it (older libvoicedetect.so) or, for weights, when the model is
|
||||
// not a readable file.
|
||||
encoderFamily string
|
||||
encoderWeights string
|
||||
}
|
||||
|
||||
func (v *VoiceDetect) Load(opts *pb.ModelOptions) error {
|
||||
@@ -89,9 +107,55 @@ func (v *VoiceDetect) Load(opts *pb.ModelOptions) error {
|
||||
return fmt.Errorf("voice-detect: voicedetect_capi_load failed for %q", model)
|
||||
}
|
||||
v.ctxPtr = ctx
|
||||
v.encoderFamily = readEncoderFamily(ctx)
|
||||
v.encoderWeights = fileIdentity(model)
|
||||
xlog.Info("voice-detect: encoder fingerprint", "family", v.encoderFamily, "weights", v.encoderWeights,
|
||||
"arch", readBorrowed(CppEncoderArch, ctx), "name", readBorrowed(CppEncoderName, ctx))
|
||||
return nil
|
||||
}
|
||||
|
||||
// readBorrowed copies a string the library keeps on the context. fn is nil on a
|
||||
// library without the accessor. A NULL or empty result means unavailable. The
|
||||
// pointer is borrowed: it is copied here and never freed.
|
||||
func readBorrowed(fn func(ctx uintptr) uintptr, ctx uintptr) string {
|
||||
if fn == nil || ctx == 0 {
|
||||
return ""
|
||||
}
|
||||
return goStringFromCPtr(fn(ctx))
|
||||
}
|
||||
|
||||
// readEncoderFamily returns "voicedetect:<arch>:<name>:<dim>", or "" when the
|
||||
// library cannot report it. A family with no part at all (":::") carries no
|
||||
// information and counts as unavailable.
|
||||
func readEncoderFamily(ctx uintptr) string {
|
||||
family := readBorrowed(CppEncoderFamily, ctx)
|
||||
if strings.Trim(family, ":") == "" {
|
||||
return ""
|
||||
}
|
||||
return family
|
||||
}
|
||||
|
||||
// fileIdentity is "sha256:<hex>" of the bytes of the model file, streamed so a
|
||||
// large file is never held in memory. It runs once per model load. It returns ""
|
||||
// when the path is not a readable regular file; the backend then reports no
|
||||
// weights identity and only the family fingerprints the voice.
|
||||
func fileIdentity(path string) string {
|
||||
f, err := os.Open(path) // #nosec G304 -- the model file LocalAI itself passes to the load
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
defer func() { _ = f.Close() }()
|
||||
if st, err := f.Stat(); err != nil || !st.Mode().IsRegular() {
|
||||
return ""
|
||||
}
|
||||
h := sha256.New()
|
||||
if _, err := io.Copy(h, f); err != nil {
|
||||
xlog.Warn("voice-detect: could not hash the model file", "error", err)
|
||||
return ""
|
||||
}
|
||||
return "sha256:" + hex.EncodeToString(h.Sum(nil))
|
||||
}
|
||||
|
||||
// VoiceEmbed returns the L2-normalized speaker embedding for an audio clip.
|
||||
// The request carries a filesystem PATH; the HTTP layer materializes
|
||||
// base64/URL/data-URI inputs to a temp file before the gRPC call.
|
||||
@@ -106,7 +170,12 @@ func (v *VoiceDetect) VoiceEmbed(req *pb.VoiceEmbedRequest) (pb.VoiceEmbedRespon
|
||||
if err != nil {
|
||||
return pb.VoiceEmbedResponse{}, err
|
||||
}
|
||||
return pb.VoiceEmbedResponse{Embedding: emb, Model: v.opts.modelName}, nil
|
||||
return pb.VoiceEmbedResponse{
|
||||
Embedding: emb,
|
||||
Model: v.opts.modelName,
|
||||
EncoderFamily: v.encoderFamily,
|
||||
EncoderWeights: v.encoderWeights,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func (v *VoiceDetect) embedPath(path string) ([]float32, error) {
|
||||
|
||||
@@ -1,9 +1,13 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sync"
|
||||
"testing"
|
||||
"unsafe"
|
||||
|
||||
"github.com/ebitengine/purego"
|
||||
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
|
||||
@@ -142,3 +146,118 @@ var _ = Describe("VoiceDetect end-to-end", Ordered, func() {
|
||||
Expect(resp.Distance).To(BeNumerically("<=", resp.Threshold))
|
||||
})
|
||||
})
|
||||
|
||||
// cstr returns a NUL-terminated copy of s and its address, the way the library
|
||||
// hands out a borrowed char*. keep holds the buffer so the GC does not drop it.
|
||||
func cstr(s string) (ptr uintptr, keep []byte) {
|
||||
b := append([]byte(s), 0)
|
||||
return uintptr(unsafe.Pointer(&b[0])), b
|
||||
}
|
||||
|
||||
// stubEncoder replaces the encoder accessors with fakes that return the given
|
||||
// strings (nil pointer when the value is nil) and restores them after the spec.
|
||||
func stubEncoder(arch, name, family *string) {
|
||||
oldA, oldN, oldF := CppEncoderArch, CppEncoderName, CppEncoderFamily
|
||||
DeferCleanup(func() { CppEncoderArch, CppEncoderName, CppEncoderFamily = oldA, oldN, oldF })
|
||||
mk := func(s *string) func(uintptr) uintptr {
|
||||
if s == nil {
|
||||
return func(uintptr) uintptr { return 0 }
|
||||
}
|
||||
ptr, keep := cstr(*s)
|
||||
return func(uintptr) uintptr { _ = keep; return ptr }
|
||||
}
|
||||
CppEncoderArch, CppEncoderName, CppEncoderFamily = mk(arch), mk(name), mk(family)
|
||||
}
|
||||
|
||||
var _ = Describe("encoder family", func() {
|
||||
str := func(s string) *string { return &s }
|
||||
|
||||
It("reads the family the library reports", func() {
|
||||
stubEncoder(str("ecapa_tdnn"), str("speechbrain/spkrec-ecapa-voxceleb"), str("voicedetect:ecapa_tdnn:speechbrain/spkrec-ecapa-voxceleb:192"))
|
||||
Expect(readEncoderFamily(1)).To(Equal("voicedetect:ecapa_tdnn:speechbrain/spkrec-ecapa-voxceleb:192"))
|
||||
Expect(readBorrowed(CppEncoderArch, 1)).To(Equal("ecapa_tdnn"))
|
||||
Expect(readBorrowed(CppEncoderName, 1)).To(Equal("speechbrain/spkrec-ecapa-voxceleb"))
|
||||
})
|
||||
|
||||
It("is empty when the library lacks the symbols", func() {
|
||||
oldA, oldN, oldF := CppEncoderArch, CppEncoderName, CppEncoderFamily
|
||||
DeferCleanup(func() { CppEncoderArch, CppEncoderName, CppEncoderFamily = oldA, oldN, oldF })
|
||||
CppEncoderArch, CppEncoderName, CppEncoderFamily = nil, nil, nil
|
||||
Expect(readEncoderFamily(1)).To(BeEmpty())
|
||||
Expect(readBorrowed(CppEncoderArch, 1)).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("is empty on a NULL pointer", func() {
|
||||
stubEncoder(nil, nil, nil)
|
||||
Expect(readEncoderFamily(1)).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("is empty on an empty string and on a family with every field missing", func() {
|
||||
stubEncoder(str(""), str(""), str(""))
|
||||
Expect(readEncoderFamily(1)).To(BeEmpty())
|
||||
stubEncoder(nil, nil, str(":::"))
|
||||
Expect(readEncoderFamily(1)).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("keeps a family with a missing field (colons kept)", func() {
|
||||
stubEncoder(nil, nil, str("voicedetect:ecapa_tdnn::192"))
|
||||
Expect(readEncoderFamily(1)).To(Equal("voicedetect:ecapa_tdnn::192"))
|
||||
})
|
||||
|
||||
It("does not read for a NULL context", func() {
|
||||
called := false
|
||||
CppEncoderFamily = func(uintptr) uintptr { called = true; return 0 }
|
||||
DeferCleanup(func() { CppEncoderFamily = nil })
|
||||
Expect(readEncoderFamily(0)).To(BeEmpty())
|
||||
Expect(called).To(BeFalse())
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("fileIdentity", func() {
|
||||
It("is the sha256 of the file bytes", func() {
|
||||
path := filepath.Join(GinkgoT().TempDir(), "m.gguf")
|
||||
Expect(os.WriteFile(path, []byte("weights"), 0o600)).To(Succeed())
|
||||
sum := sha256.Sum256([]byte("weights"))
|
||||
Expect(fileIdentity(path)).To(Equal("sha256:" + hex.EncodeToString(sum[:])))
|
||||
})
|
||||
It("is empty for a missing path and for a directory", func() {
|
||||
dir := GinkgoT().TempDir()
|
||||
Expect(fileIdentity(filepath.Join(dir, "none.gguf"))).To(BeEmpty())
|
||||
Expect(fileIdentity(dir)).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("VoiceEmbed fingerprint", func() {
|
||||
stubEmbed := func() {
|
||||
oldP, oldF := CppEmbedPath, CppFreeVec
|
||||
DeferCleanup(func() { CppEmbedPath, CppFreeVec = oldP, oldF })
|
||||
vec := []float32{0.6, 0.8}
|
||||
CppEmbedPath = func(_ uintptr, _ string, outVec, outDim unsafe.Pointer) int32 {
|
||||
*(*uintptr)(outVec) = uintptr(unsafe.Pointer(&vec[0]))
|
||||
*(*int32)(outDim) = int32(len(vec))
|
||||
return 0
|
||||
}
|
||||
CppFreeVec = func(uintptr) {}
|
||||
}
|
||||
|
||||
It("returns the family and the weights next to the embedding", func() {
|
||||
stubEmbed()
|
||||
v := &VoiceDetect{ctxPtr: 1, encoderFamily: "voicedetect:a:b:2", encoderWeights: "sha256:ab"}
|
||||
v.opts.modelName = "m.gguf"
|
||||
resp, err := v.VoiceEmbed(&pb.VoiceEmbedRequest{Audio: "x.wav"})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(resp.Embedding).To(Equal([]float32{0.6, 0.8}))
|
||||
Expect(resp.Model).To(Equal("m.gguf"))
|
||||
Expect(resp.EncoderFamily).To(Equal("voicedetect:a:b:2"))
|
||||
Expect(resp.EncoderWeights).To(Equal("sha256:ab"))
|
||||
})
|
||||
|
||||
It("leaves both empty on a library that cannot report them", func() {
|
||||
stubEmbed()
|
||||
v := &VoiceDetect{ctxPtr: 1}
|
||||
resp, err := v.VoiceEmbed(&pb.VoiceEmbedRequest{Audio: "x.wav"})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(resp.EncoderFamily).To(BeEmpty())
|
||||
Expect(resp.EncoderWeights).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
@@ -54,6 +54,19 @@ func main() {
|
||||
purego.RegisterLibFunc(lf.FuncPtr, lib, lf.Name)
|
||||
}
|
||||
|
||||
// Encoder identity accessors (additive in voice-detect.cpp, no ABI bump).
|
||||
// Probed so a libvoicedetect.so from before them still loads: the encoder
|
||||
// family then stays empty and audio-registered voices stay unfingerprinted.
|
||||
for _, lf := range []LibFuncs{
|
||||
{&CppEncoderArch, "voicedetect_capi_encoder_arch"},
|
||||
{&CppEncoderName, "voicedetect_capi_encoder_name"},
|
||||
{&CppEncoderFamily, "voicedetect_capi_encoder_family"},
|
||||
} {
|
||||
if sym, err := purego.Dlsym(lib, lf.Name); err == nil && sym != 0 {
|
||||
purego.RegisterLibFunc(lf.FuncPtr, lib, lf.Name)
|
||||
}
|
||||
}
|
||||
|
||||
fmt.Fprintf(os.Stderr, "[voice-detect] ABI=%d\n", CppAbiVersion())
|
||||
|
||||
flag.Parse()
|
||||
|
||||
@@ -74,6 +74,12 @@ func VoiceIdentifyEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader,
|
||||
if trustedErr != nil || m.Metadata.Model != trusted.Identity || len(embed.GetEmbedding()) != trusted.Dimension {
|
||||
continue
|
||||
}
|
||||
} else if m.Metadata.EncoderFamily != "" && embed.GetEncoderFamily() != "" {
|
||||
// Both sides carry a fingerprint: the embedding space decides, not
|
||||
// the file name.
|
||||
if m.Metadata.EncoderFamily != embed.GetEncoderFamily() {
|
||||
continue
|
||||
}
|
||||
} else if m.Metadata.Model != "" && voicerecognition.EncoderTag(m.Metadata.Model) != voicerecognition.EncoderTag(embed.GetModel()) {
|
||||
continue
|
||||
}
|
||||
|
||||
@@ -34,7 +34,7 @@ func VoiceRegisterEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader,
|
||||
}
|
||||
|
||||
var embedding []float32
|
||||
var encoder, family string
|
||||
var encoder, family, weights string
|
||||
if input.SpeakerProfiles != nil {
|
||||
if input.Audio != "" || input.SpeakerSlot == nil {
|
||||
return echo.NewHTTPError(http.StatusBadRequest, "speaker_profiles requires speaker_slot and excludes audio")
|
||||
@@ -62,11 +62,14 @@ func VoiceRegisterEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader,
|
||||
return mapBackendError(err)
|
||||
}
|
||||
embedding, encoder = res.GetEmbedding(), res.GetModel()
|
||||
// Fingerprint reported by the voice backend. Empty from a backend that
|
||||
// cannot report it: the voice then stays unfingerprinted.
|
||||
family, weights = res.GetEncoderFamily(), res.GetEncoderWeights()
|
||||
}
|
||||
meta := voiceMetadata(input.Name, input.Labels, encoder)
|
||||
// Only the portable route knows the family: it comes from the loaded
|
||||
// encoder. A voice-detect embedding has none, so it stays unfingerprinted.
|
||||
meta.EncoderFamily = family
|
||||
// The family comes from the loaded encoder (portable route) or from the
|
||||
// voice backend that embedded the audio. A backend that cannot report
|
||||
// one leaves it empty and the voice stays unfingerprinted.
|
||||
meta := voiceMetadata(input.Name, input.Labels, encoder, family, weights)
|
||||
stored, err := registry.Register(c.Request().Context(), embedding, meta)
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -81,7 +84,8 @@ func VoiceRegisterEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader,
|
||||
|
||||
// voiceMetadata is what a registration stores next to the embedding. Model is
|
||||
// the speaker encoder that produced it, so a consumer with a different encoder
|
||||
// can tell the vectors are not comparable.
|
||||
func voiceMetadata(name string, labels map[string]string, embedderModel string) voicerecognition.Metadata {
|
||||
return voicerecognition.Metadata{Name: name, Labels: labels, Model: embedderModel}
|
||||
// can tell the vectors are not comparable. family and weights fingerprint the
|
||||
// encoder when it reported them (see voicerecognition.Metadata), "" otherwise.
|
||||
func voiceMetadata(name string, labels map[string]string, embedderModel, family, weights string) voicerecognition.Metadata {
|
||||
return voicerecognition.Metadata{Name: name, Labels: labels, Model: embedderModel, EncoderFamily: family, EncoderWeights: weights}
|
||||
}
|
||||
@@ -7,12 +7,23 @@ import (
|
||||
|
||||
var _ = Describe("voiceMetadata", func() {
|
||||
It("carries the name, the labels and the encoder that embedded the voice", func() {
|
||||
m := voiceMetadata("ada", map[string]string{"team": "a"}, "voice-detect-wespeaker-resnet34.gguf")
|
||||
m := voiceMetadata("ada", map[string]string{"team": "a"}, "voice-detect-wespeaker-resnet34.gguf", "", "")
|
||||
Expect(m.Name).To(Equal("ada"))
|
||||
Expect(m.Labels).To(Equal(map[string]string{"team": "a"}))
|
||||
Expect(m.Model).To(Equal("voice-detect-wespeaker-resnet34.gguf"))
|
||||
})
|
||||
It("leaves the tag empty when the backend did not say", func() {
|
||||
Expect(voiceMetadata("ada", nil, "").Model).To(BeEmpty())
|
||||
Expect(voiceMetadata("ada", nil, "", "", "").Model).To(BeEmpty())
|
||||
})
|
||||
It("records the family and the weights the voice backend reported, and keeps the name in model", func() {
|
||||
m := voiceMetadata("ada", nil, "ecapa.gguf", "voicedetect:ecapa_tdnn:speechbrain/spkrec-ecapa-voxceleb:192", "sha256:ab")
|
||||
Expect(m.Model).To(Equal("ecapa.gguf"))
|
||||
Expect(m.EncoderFamily).To(Equal("voicedetect:ecapa_tdnn:speechbrain/spkrec-ecapa-voxceleb:192"))
|
||||
Expect(m.EncoderWeights).To(Equal("sha256:ab"))
|
||||
})
|
||||
It("leaves the voice unfingerprinted when the backend reported nothing", func() {
|
||||
m := voiceMetadata("ada", nil, "ecapa.gguf", "", "")
|
||||
Expect(m.EncoderFamily).To(BeEmpty())
|
||||
Expect(m.EncoderWeights).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
@@ -118,9 +118,13 @@ func SelectKnownVoices(entries []Entry, speakerModelPath string, extraTags ...st
|
||||
switch {
|
||||
case e.Metadata.Model == "":
|
||||
untagged = append(untagged, e)
|
||||
// Hash-tagged portable registrations are checked against the loaded
|
||||
// encoder by the backend, never against a filename or dimension alone.
|
||||
case strings.HasPrefix(e.Metadata.Model, "sha256:"), tags[EncoderTag(e.Metadata.Model)]:
|
||||
case e.Metadata.EncoderFamily != "", strings.HasPrefix(e.Metadata.Model, "sha256:"):
|
||||
// A voice with an encoder family is checked by the backend against the
|
||||
// loaded encoder's family, not by file name: the same encoder can be
|
||||
// converted under another name, and a different one must be refused
|
||||
// by the backend with a clear error.
|
||||
sel.Voices = append(sel.Voices, knownVoice(e))
|
||||
case tags[EncoderTag(e.Metadata.Model)]:
|
||||
sel.Voices = append(sel.Voices, knownVoice(e))
|
||||
default:
|
||||
sel.OtherEncoder++
|
||||
@@ -137,7 +141,10 @@ func SelectKnownVoices(entries []Entry, speakerModelPath string, extraTags ...st
|
||||
|
||||
func knownVoice(e Entry) KnownVoice {
|
||||
v := KnownVoice{ID: e.Metadata.ID, Name: e.Metadata.Name, Embedding: e.Embedding, Model: e.Metadata.Model, Family: e.Metadata.EncoderFamily}
|
||||
if strings.HasPrefix(e.Metadata.Model, "sha256:") {
|
||||
switch {
|
||||
case e.Metadata.EncoderWeights != "":
|
||||
v.Weights = e.Metadata.EncoderWeights
|
||||
case strings.HasPrefix(e.Metadata.Model, "sha256:"):
|
||||
v.Weights = e.Metadata.Model
|
||||
}
|
||||
return v
|
||||
|
||||
@@ -156,6 +156,23 @@ var _ = Describe("encoder fingerprint of selected voices", func() {
|
||||
Expect(sel.Voices[0].Family).To(Equal("voicedetect:ecapa_tdnn:ecapa:192"))
|
||||
Expect(sel.Voices[0].Weights).To(Equal(hash))
|
||||
})
|
||||
It("forwards an audio-registered voice by its family and weights, whatever its file name", func() {
|
||||
e := entry("ada", "voice-detect-ecapa.gguf", 1, 0)
|
||||
e.Metadata.EncoderFamily = "voicedetect:ecapa_tdnn:ecapa:192"
|
||||
e.Metadata.EncoderWeights = hash
|
||||
sel := voicerecognition.SelectKnownVoices([]voicerecognition.Entry{e}, "other-encoder.gguf")
|
||||
Expect(sel.OtherEncoder).To(BeZero())
|
||||
Expect(sel.Voices).To(HaveLen(1))
|
||||
Expect(sel.Voices[0].Model).To(Equal("voice-detect-ecapa.gguf"))
|
||||
Expect(sel.Voices[0].Family).To(Equal("voicedetect:ecapa_tdnn:ecapa:192"))
|
||||
Expect(sel.Voices[0].Weights).To(Equal(hash))
|
||||
})
|
||||
It("keeps an old audio voice without a family on the file-name filter", func() {
|
||||
old := entry("ada", "voice-detect-ecapa.gguf", 1, 0)
|
||||
sel := voicerecognition.SelectKnownVoices([]voicerecognition.Entry{old}, "other-encoder.gguf")
|
||||
Expect(sel.Voices).To(BeEmpty())
|
||||
Expect(sel.OtherEncoder).To(Equal(1))
|
||||
})
|
||||
It("leaves a file-name tagged voice unfingerprinted", func() {
|
||||
sel := voicerecognition.SelectKnownVoices([]voicerecognition.Entry{entry("ada", "spk.gguf", 1, 0)}, "spk.gguf")
|
||||
Expect(sel.Voices[0].Family).To(BeEmpty())
|
||||
@@ -171,6 +188,7 @@ var _ = Describe("encoder fingerprint of selected voices", func() {
|
||||
Expect(fresh.EncoderFamily).To(Equal("f"))
|
||||
raw, _ = json.Marshal(old)
|
||||
Expect(string(raw)).ToNot(ContainSubstring("encoder_family"))
|
||||
Expect(string(raw)).ToNot(ContainSubstring("encoder_weights"))
|
||||
})
|
||||
})
|
||||
|
||||
|
||||
@@ -61,8 +61,14 @@ type Metadata struct {
|
||||
// EncoderFamily is the embedding space of the encoder ("voicedetect:<arch>:<name>:<dim>"),
|
||||
// recorded when the encoder reports it (portable enrollment from speaker
|
||||
// profiles). Empty when unknown, and for voices registered before it existed.
|
||||
// Model then holds the weights identity ("sha256:<hex>") for the same voices.
|
||||
// For the speaker-profiles route Model then holds the weights identity
|
||||
// ("sha256:<hex>") of the same voices.
|
||||
EncoderFamily string `json:"encoder_family,omitempty"`
|
||||
// EncoderWeights is the "sha256:<hex>" identity of the encoder's model file,
|
||||
// recorded for voices registered from audio when the voice backend reports
|
||||
// it. Model keeps the encoder's name for those voices, so the 1:N filter on
|
||||
// the name keeps working. Empty when unknown.
|
||||
EncoderWeights string `json:"encoder_weights,omitempty"`
|
||||
}
|
||||
|
||||
// Match is a single result from Identify, ranked by similarity.
|
||||
|
||||
@@ -344,21 +344,48 @@ A voice with only a weights identity takes the family of the loaded encoder
|
||||
when the weights are the same file. A voice with a different weights hash and
|
||||
no family is dropped, as before.
|
||||
|
||||
A voice registered from audio through the voice-detect backend has no
|
||||
fingerprint: libvoicedetect reports no architecture or model name, so the
|
||||
backend cannot tell the family, and only the file-name tag described above
|
||||
applies. Such voices and fingerprinted voices cannot share one registry in the
|
||||
library. When a request has any unfingerprinted voice, all of its voices are
|
||||
used without the fingerprint check (the old behaviour). To get the check, enroll
|
||||
every voice from `speaker_profiles`. With `speaker_strict:true` the backend
|
||||
ignores unfingerprinted voices, and a request that has only those fails with
|
||||
the library's message. The family is also reported in the internal backend
|
||||
status next to the identity.
|
||||
A voice registered from audio through the voice-detect backend (`POST
|
||||
/v1/voice/register` with `audio`) is fingerprinted too, when the libvoicedetect
|
||||
in the backend can report its encoder (voice-detect.cpp from the pin that adds
|
||||
`voicedetect_capi_encoder_family`). The backend reads the family from the
|
||||
loaded GGUF, and hashes the model file once at load to get the weights. LocalAI
|
||||
stores the family as `encoder_family` and the weights as `encoder_weights`; the
|
||||
`model` field keeps the encoder name as before. If the voice-detect model is
|
||||
not a plain file the backend cannot hash it and the weights stay empty, so the
|
||||
family alone fingerprints the voice. When the family is known, LocalAI sends
|
||||
the voice to the backend whatever its file name, so the family decides.
|
||||
|
||||
{{% notice warning %}}
|
||||
Do not set a `model_name:` option on the voice-detect model config. It
|
||||
replaces the default name, the voices are then tagged with it, and they no
|
||||
longer match the `speaker_model:` file. Keep the default name.
|
||||
A voice-detect backend built against an older libvoicedetect cannot report
|
||||
the family, but it still records the weights. Such a voice is treated as made
|
||||
by the loaded `speaker_model:` when the weights are the same file, and stays
|
||||
unfingerprinted when they differ.
|
||||
|
||||
Voices registered before this change have no fingerprint, and so do voices
|
||||
from a backend that reports neither (the Python speaker-recognition backend).
|
||||
They are used as described above, by file-name
|
||||
tag. Such voices and fingerprinted voices cannot share one registry in the
|
||||
library. When a request has any unfingerprinted voice, all of its voices are
|
||||
used without the fingerprint check. **Register those voices again** to get
|
||||
the check. With `speaker_strict:true` the backend ignores unfingerprinted
|
||||
voices, and a request that has only those fails with the library's message.
|
||||
The family is also reported in the internal backend status next to the
|
||||
identity. `/v1/voice/identify` compares the family of a stored voice with the
|
||||
family of the probe when both have one, and falls back to the file-name tag
|
||||
when either has none.
|
||||
|
||||
The family is built from the encoder GGUF metadata. For CAM++, WeSpeaker and
|
||||
ERes2Net the GGUF `general.name` is the path the model was converted from, so
|
||||
the same encoder converted again under another name has another family, and
|
||||
voices registered with the first file are refused with the second. Two
|
||||
fine-tunes that share an architecture, name and embedding size have the same
|
||||
family; only the weights hash tells them apart, and a hash mismatch alone only
|
||||
logs a warning.
|
||||
|
||||
{{% notice note %}}
|
||||
With a fingerprint, the `model_name:` option on the voice-detect model config
|
||||
no longer breaks naming: the family decides, not the name. Voices registered
|
||||
without a fingerprint still need the default name to match the
|
||||
`speaker_model:` file.
|
||||
{{% /notice %}}
|
||||
|
||||
### Options
|
||||
@@ -463,7 +490,8 @@ are independent.
|
||||
| `store` | string, optional | vector store model; defaults to local-store |
|
||||
|
||||
Returns `{id, name, registered_at}`. The `id` is an opaque UUID used
|
||||
by `/v1/voice/identify` and `/v1/voice/forget`.
|
||||
by `/v1/voice/identify` and `/v1/voice/forget`. The stored voice records the
|
||||
encoder [fingerprint](#encoder-fingerprint) when the backend reports it.
|
||||
|
||||
### `POST /v1/voice/identify` (1:N recognition)
|
||||
|
||||
|
||||
Reference in new issue
Block a user