From 79a7631cc55bc51a60d8d972b3d72c6aa4933ced Mon Sep 17 00:00:00 2001 From: mudler-agent Date: Mon, 5 Oct 2026 15:57:16 +0200 Subject: [PATCH] feat(parakeet-cpp): encoder fingerprint for speaker naming, VAD trim and word filter options, pin bump (#12491) * chore(parakeet-cpp): bump parakeet.cpp to 2de154c Brings in the speaker registry encoder fingerprint, the VAD segment trim and the opt-in word filter, a fix for a per-call thread count that stayed set on the process-wide backend after a Silero VAD pass, and bundle components loaded from memory. Assisted-by: Claude:claude-sonnet-5-5 [Claude Code] * feat(parakeet-cpp): encoder fingerprint for speaker naming, vad_trim and guard_* options Speaker naming. A registered voice now carries the encoder that made it: the embedding family (voicedetect:::) and the sha256 of the weights. The backend reports the family of the loaded speaker model in Status, voice enrollment from speaker_profiles stores it as encoder_family (old entries load without it), and the registry sent to parakeet.cpp is built with parakeet_capi_speaker_registry_add_embedding_fp. The library then refuses a registry of another encoder family and the error names both families; another quantization of the same family only warns. A voice with only a weights hash gets the loaded family when the hashes are equal. Voices without a fingerprint (registered from audio: libvoicedetect cannot report one) keep the file-name rule and are used with a warning. The library cannot mix them with fingerprinted voices in one registry, so a request that has any uses the old registry for all. speaker_strict:true drops them instead. A library without the symbols behaves as before. Transcription. vad_trim (seconds, 0 keeps the whole cuts) goes through the VAD options JSON, so it reaches /v1/vad and the segmenter. The guard_* options guard_min_local_conf, guard_local_radius and guard_drop_punct_only turn on the word filter through parakeet_capi_transcribe_path_json_with, or through the segmenter with vad:true. They are off by default, bad values fail the load, and a library without the symbol fails it with a clear message. The dropped word count is logged at debug level. Assisted-by: Claude:claude-sonnet-5-5 [Claude Code] --------- Co-authored-by: Ettore Di Giacinto --- backend/backend.proto | 6 + backend/go/parakeet-cpp/Makefile | 4 +- backend/go/parakeet-cpp/goparakeetcpp.go | 56 ++++- backend/go/parakeet-cpp/guard.go | 91 +++++++ backend/go/parakeet-cpp/guard_test.go | 145 ++++++++++++ backend/go/parakeet-cpp/main.go | 11 + backend/go/parakeet-cpp/profiles.go | 5 + backend/go/parakeet-cpp/roles.go | 9 + backend/go/parakeet-cpp/scene.go | 3 + backend/go/parakeet-cpp/scene_test.go | 3 + .../parakeet-cpp/speaker_fingerprint_test.go | 224 ++++++++++++++++++ backend/go/parakeet-cpp/speaker_registry.go | 116 ++++++++- backend/go/parakeet-cpp/vad.go | 5 +- backend/go/parakeet-cpp/vad_rpc_test.go | 19 ++ core/backend/diarization.go | 12 +- core/backend/diarization_profiles_test.go | 39 ++- core/backend/transcript_live.go | 2 +- core/http/endpoints/localai/voice_register.go | 10 +- core/schema/speaker_profiles.go | 5 +- core/schema/speaker_profiles_test.go | 6 + .../services/voicerecognition/known_voices.go | 15 +- .../voicerecognition/known_voices_test.go | 29 +++ core/services/voicerecognition/registry.go | 5 + docs/content/features/audio-to-text.md | 20 ++ .../features/voice-activity-detection.md | 1 + docs/content/features/voice-recognition.md | 37 +++ swagger/docs.go | 8 + swagger/swagger.json | 8 + swagger/swagger.yaml | 6 + 29 files changed, 876 insertions(+), 24 deletions(-) create mode 100644 backend/go/parakeet-cpp/guard.go create mode 100644 backend/go/parakeet-cpp/guard_test.go create mode 100644 backend/go/parakeet-cpp/speaker_fingerprint_test.go diff --git a/backend/backend.proto b/backend/backend.proto index 676e6222c..aaa9a63c8 100644 --- a/backend/backend.proto +++ b/backend/backend.proto @@ -851,6 +851,11 @@ message KnownVoice { string name = 1; repeated float embedding = 2; string model = 3; + // Fingerprint of the encoder that made the embedding, when it is known: the + // embedding space ("voicedetect:::") and the exact weights + // ("sha256:"). Empty for voices registered before it was recorded. + string encoder_family = 5; + string encoder_weights = 6; } message DiarizeResponse { @@ -913,6 +918,7 @@ message MemoryUsageData { message SpeakerEncoder { string identity = 1; // sha256 of loaded GGUF bytes int32 dimension = 2; + string family = 3; // embedding space of the encoder; empty when the backend cannot tell } message StatusResponse { diff --git a/backend/go/parakeet-cpp/Makefile b/backend/go/parakeet-cpp/Makefile index 54db93ddb..7fdad6cd0 100644 --- a/backend/go/parakeet-cpp/Makefile +++ b/backend/go/parakeet-cpp/Makefile @@ -1,6 +1,6 @@ # parakeet-cpp backend Makefile. # -# Upstream pin lives below as PARAKEET_VERSION?=0cca477249ffb16c1623fb947d5bac0624961d41 +# Upstream pin lives below as PARAKEET_VERSION?=2de154c622830b62bfcfb92556dbb71bf0263ddd # (.github/bump_deps.sh) can find and update it - matches the # whisper.cpp / ds4 / vibevoice-cpp convention. # @@ -15,7 +15,7 @@ # That's what the L0 smoke test uses. The default target below does the # proper clone-at-pin + cmake build so CI doesn't need a side-checkout. -PARAKEET_VERSION?=0cca477249ffb16c1623fb947d5bac0624961d41 +PARAKEET_VERSION?=2de154c622830b62bfcfb92556dbb71bf0263ddd PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp GOCMD?=go diff --git a/backend/go/parakeet-cpp/goparakeetcpp.go b/backend/go/parakeet-cpp/goparakeetcpp.go index 270678da4..c8d8b9b68 100644 --- a/backend/go/parakeet-cpp/goparakeetcpp.go +++ b/backend/go/parakeet-cpp/goparakeetcpp.go @@ -36,8 +36,12 @@ var ( CppFree func(ctx uintptr) CppTranscribePath func(ctx uintptr, wavPath string, decoder int32) uintptr CppTranscribePathJSON func(ctx uintptr, wavPath string, decoder int32) uintptr - CppFreeString func(s uintptr) - CppLastError func(ctx uintptr) string + // CppTranscribePathJSONWith is CppTranscribePathJSON with the optional word + // filter (min_local_conf, local_radius, drop_punct_only as a JSON object). + // nil on a libparakeet.so from before the filter. + CppTranscribePathJSONWith func(ctx uintptr, wavPath string, decoder int32, optionsJSON string) uintptr + CppFreeString func(s uintptr) + CppLastError func(ctx uintptr) string // Bundle GGUF (additive in the C-API, no ABI bump; see bundle.go). All three // are registered together and nil on an older libparakeet.so, where a @@ -151,6 +155,12 @@ var ( CppSpeakerRegistryAddEmbedding func(reg uintptr, name string, emb *float32, dim int32) int32 CppSpeakerRegistryLastError func(reg uintptr) string CppSceneStreamBeginSpeaker func(asr, diar, tagger, speaker, reg uintptr, o *cSceneOpts) uintptr + // Encoder fingerprint (additive, ABI 10). Probed as a group; nil on a library + // from before it. CppSpeakerEncoderFamily returns a borrowed char*, read it with + // goStringFromCPtr and do not free it. A family or weights string "" means none. + CppSpeakerRegistryAddEmbeddingFP func(reg uintptr, name string, emb *float32, dim int32, family, weights string) int32 + CppSpeakerRegistrySetStrict func(reg uintptr, strict int32) + CppSpeakerEncoderFamily func(ctx uintptr) uintptr // CppDiarizeNamedPCMJSON takes two float32 arguments (acceptThreshold, margin), which // purego passes in floating-point registers. Not exercised without the real library. CppDiarizeNamedPCMJSON func(diar, speaker, reg uintptr, samples *float32, n, sampleRate int32, acceptThreshold, margin float32) uintptr @@ -201,6 +211,10 @@ type transcriptJSON struct { FrameSec float64 `json:"frame_sec"` Words []transcriptWord `json:"words"` Tokens []transcriptToken `json:"tokens"` + // Guard is present only when the word filter ran (guard_* options). + Guard *struct { + DroppedWords int `json:"dropped_words"` + } `json:"guard"` } // streamFeedJSON mirrors the document returned by @@ -258,6 +272,9 @@ type ParakeetCpp struct { spkCtx uintptr speakerAccept float32 speakerMargin float32 + // speakerStrict (speaker_strict:true) refuses registered voices that carry no + // encoder fingerprint instead of using them unverified. + speakerStrict bool // diarLatency is the PARAKEET_DIAR_LATENCY_* mode for diarization // streaming (diarization_latency: option, default "low"). Unused until // the diarization/scene streaming paths land. @@ -285,9 +302,13 @@ type ParakeetCpp struct { // RPC then falls back to the ASR context's own VAD head. vadCtx uintptr // vadOptions is the JSON object built from the vad_threshold, vad_min_pause, - // vad_min_speech, vad_speech_pad and vad_max_segment model options ("" when + // vad_min_speech, vad_speech_pad, vad_max_segment and vad_trim model options ("" when // none is set, so the library picks the defaults of the detector in use). vadOptions string + // guardOptions is the JSON options object of the word filter (guard_* + // model options), "" when it is off. Offline transcription then takes the + // file-path route like vad:true, because the batched entry point has no filter. + guardOptions string } // Load is the LocalAI gRPC entry point for LoadModel: it calls @@ -310,6 +331,14 @@ func (p *ParakeetCpp) Load(opts *pb.ModelOptions) error { return err } p.vadOptions = vadOpts + guardOpts, err := parseGuardOptions(opts) + if err != nil { + return err + } + p.guardOptions = guardOpts + if guardOpts != "" && p.vad && CppTranscribePathJSONVadWith == nil { + return errors.New("parakeet-cpp: the guard_* options with vad:true need a libparakeet.so with parakeet_capi_transcribe_path_json_vad_with; rebuild the backend against a newer parakeet.cpp") + } if optString(opts, "vad_model") != "" || optString(opts, "vad_component") != "" { if CppTranscribePathJSONVadWith == nil { return errors.New("parakeet-cpp: vad_model and vad_component need a libparakeet.so with parakeet_capi_transcribe_path_json_vad_with; rebuild the backend against a newer parakeet.cpp") @@ -505,7 +534,7 @@ func (p *ParakeetCpp) AudioTranscription(ctx context.Context, opts *pb.Transcrip // With vad:true the same file-path route is taken through the // VAD-segmented entry point, which cuts long audio at pauses. The batcher // has no VAD variant, so this path replaces it for offline requests. - if p.bat == nil || p.vad { + if p.bat == nil || p.vad || p.guardOptions != "" { converted, cleanup, err := convertToWavMono16k(opts.Dst) if err != nil { return pb.TranscriptResult{}, err @@ -515,7 +544,7 @@ func (p *ParakeetCpp) AudioTranscription(ctx context.Context, opts *pb.Transcrip if err != nil { return pb.TranscriptResult{}, err } - if p.vad && p.wantSpeakers(opts.GetDiarize()) && len(doc.Words) > 0 { + if (p.vad || p.guardOptions != "") && p.wantSpeakers(opts.GetDiarize()) && len(doc.Words) > 0 { pcm, _, err := decodeWavMono16k(converted) if err != nil { return pb.TranscriptResult{}, err @@ -575,13 +604,20 @@ func (p *ParakeetCpp) AudioTranscription(ctx context.Context, opts *pb.Transcrip func (p *ParakeetCpp) transcribePathDoc(path string) (transcriptJSON, error) { call, name := func() uintptr { return CppTranscribePathJSON(p.ctxPtr, path, 0) }, "transcribe_path_json" switch { - case p.vad && (p.vadCtx != 0 || p.vadOptions != "") && CppTranscribePathJSONVadWith != nil: - // An external Silero, or tuned segmenter options on the model's own head. + case p.vad && (p.vadCtx != 0 || p.vadOptions != "" || p.guardOptions != "") && CppTranscribePathJSONVadWith != nil: + // An external Silero, tuned segmenter options on the model's own head, or the + // word filter: the segmenter takes the VAD keys and the filter keys in one object. + opts, err := mergeJSONObjects(p.vadOptions, p.guardOptions) + if err != nil { + return transcriptJSON{}, fmt.Errorf("parakeet-cpp: build vad options: %w", err) + } call, name = func() uintptr { - return CppTranscribePathJSONVadWith(p.ctxPtr, p.vadCtx, path, 0, p.vadOptions) + return CppTranscribePathJSONVadWith(p.ctxPtr, p.vadCtx, path, 0, opts) }, "transcribe_path_json_vad_with" case p.vad: call, name = func() uintptr { return CppTranscribePathJSONVad(p.ctxPtr, path, 0) }, "transcribe_path_json_vad" + case p.guardOptions != "" && CppTranscribePathJSONWith != nil: + call, name = func() uintptr { return CppTranscribePathJSONWith(p.ctxPtr, path, 0, p.guardOptions) }, "transcribe_path_json_with" } p.engineMu.Lock() cstr := call() @@ -599,6 +635,10 @@ func (p *ParakeetCpp) transcribePathDoc(path string) (transcriptJSON, error) { if err := json.Unmarshal([]byte(raw), &doc); err != nil { return transcriptJSON{}, fmt.Errorf("parakeet-cpp: decode transcript json: %w", err) } + if doc.Guard != nil { + // TranscriptResult has no field for it, so the count is only logged. + xlog.Debug("parakeet-cpp: word filter", "dropped_words", doc.Guard.DroppedWords) + } return doc, nil } diff --git a/backend/go/parakeet-cpp/guard.go b/backend/go/parakeet-cpp/guard.go new file mode 100644 index 000000000..6bc4c3fe0 --- /dev/null +++ b/backend/go/parakeet-cpp/guard.go @@ -0,0 +1,91 @@ +package main + +import ( + "encoding/json" + "errors" + "fmt" + "math" + "strconv" + + pb "github.com/mudler/LocalAI/pkg/grpc/proto" +) + +// The word filter of libparakeet (the "guard"): an opt-in pass over a finished +// decode that drops words which stand alone or sit among low-confidence words, +// as noise tends to give, and words that are only punctuation. It is off unless +// one of these model options is set: +// +// guard_min_local_conf:0.5 0 to 1; 0 = off +// guard_local_radius:5 seconds > 0; the library default is 5 +// guard_drop_punct_only:true +// +// The names carry the guard_ prefix because the bare library names +// (min_local_conf, local_radius) say nothing next to vad_*, speaker_* and the +// other options of this backend, and the library's result calls the section +// "guard". + +// parseGuardOptions reads the guard_* model options and returns the JSON +// options object of parakeet_capi_transcribe_path_json_with (the keys the +// library names min_local_conf, local_radius and drop_punct_only), or "" when +// none is set. A value that does not parse or is out of range fails the load, +// and so does the filter on a libparakeet.so that cannot run it. +func parseGuardOptions(opts *pb.ModelOptions) (string, error) { + obj := map[string]any{} + if raw := optString(opts, "guard_min_local_conf"); raw != "" { + v, err := strconv.ParseFloat(raw, 64) + if err != nil || math.IsNaN(v) || math.IsInf(v, 0) { + return "", fmt.Errorf("parakeet-cpp: option guard_min_local_conf: %q is not a number", raw) + } + if v < 0 || v > 1 { + return "", fmt.Errorf("parakeet-cpp: option guard_min_local_conf: %v is out of range, want 0 to 1", v) + } + obj["min_local_conf"] = v + } + if raw := optString(opts, "guard_local_radius"); raw != "" { + v, err := strconv.ParseFloat(raw, 64) + if err != nil || math.IsNaN(v) || math.IsInf(v, 0) { + return "", fmt.Errorf("parakeet-cpp: option guard_local_radius: %q is not a number", raw) + } + if v <= 0 { + return "", fmt.Errorf("parakeet-cpp: option guard_local_radius: %v is out of range, want seconds > 0", v) + } + obj["local_radius"] = v + } + if optString(opts, "guard_drop_punct_only") != "" { + b, err := optBool(opts, "guard_drop_punct_only", false) + if err != nil { + return "", err + } + obj["drop_punct_only"] = b + } + if len(obj) == 0 { + return "", nil + } + if CppTranscribePathJSONWith == nil { + return "", errors.New("parakeet-cpp: the guard_* options need a libparakeet.so with parakeet_capi_transcribe_path_json_with; rebuild the backend against a newer parakeet.cpp") + } + b, err := json.Marshal(obj) + if err != nil { + return "", err + } + return string(b), nil +} + +// mergeJSONObjects joins two flat JSON objects ("" is an empty one). The +// segmenter entry point takes the VAD keys and the guard keys in one object. +func mergeJSONObjects(a, b string) (string, error) { + if a == "" { + return b, nil + } + if b == "" { + return a, nil + } + m := map[string]json.RawMessage{} + for _, s := range []string{a, b} { + if err := json.Unmarshal([]byte(s), &m); err != nil { + return "", err + } + } + out, err := json.Marshal(m) + return string(out), err +} diff --git a/backend/go/parakeet-cpp/guard_test.go b/backend/go/parakeet-cpp/guard_test.go new file mode 100644 index 000000000..7b830986d --- /dev/null +++ b/backend/go/parakeet-cpp/guard_test.go @@ -0,0 +1,145 @@ +package main + +import ( + "context" + "path/filepath" + + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +var _ = Describe("guard_* model options (word filter)", func() { + var ( + savedPlain func(uintptr, string, int32) uintptr + savedWithOpts func(uintptr, string, int32, string) uintptr + savedVadWith func(ctx, vadCtx uintptr, p string, d int32, o string) uintptr + savedFree func(uintptr) + pool *diarizeCstrPool + gotOpts string + usedWith, usedVadWith bool + ) + opts := func(o ...string) *pb.ModelOptions { return &pb.ModelOptions{Options: o} } + + BeforeEach(func() { + savedPlain, savedWithOpts, savedVadWith, savedFree = CppTranscribePathJSON, CppTranscribePathJSONWith, CppTranscribePathJSONVadWith, CppFreeString + pool = &diarizeCstrPool{} + usedWith, usedVadWith, gotOpts = false, false, "" + CppFreeString = func(uintptr) {} + CppTranscribePathJSON = func(uintptr, string, int32) uintptr { + Fail("the plain entry point must not be used when the filter is on") + return 0 + } + CppTranscribePathJSONWith = func(_ uintptr, _ string, _ int32, o string) uintptr { + usedWith, gotOpts = true, o + return pool.cstr(`{"text":"hello.","frame_sec":0.08,"words":[{"w":"hello.","start":0.1,"end":0.4,"conf":0.9}],"tokens":[],"guard":{"dropped_words":2}}`) + } + CppTranscribePathJSONVadWith = func(_, _ uintptr, _ string, _ int32, o string) uintptr { + usedVadWith, gotOpts = true, o + return pool.cstr(`{"text":"hello.","frame_sec":0.08,"words":[],"tokens":[],"guard":{"dropped_words":0}}`) + } + }) + AfterEach(func() { + CppTranscribePathJSON, CppTranscribePathJSONWith, CppTranscribePathJSONVadWith, CppFreeString = savedPlain, savedWithOpts, savedVadWith, savedFree + }) + + It("is off when no option is set, and then needs no library symbol", func() { + CppTranscribePathJSONWith = nil + s, err := parseGuardOptions(opts("vad:true")) + Expect(err).ToNot(HaveOccurred()) + Expect(s).To(BeEmpty()) + }) + + It("maps the options to the library keys", func() { + s, err := parseGuardOptions(opts("guard_min_local_conf:0.5", "guard_local_radius:3", "guard_drop_punct_only:true")) + Expect(err).ToNot(HaveOccurred()) + Expect(s).To(MatchJSON(`{"min_local_conf":0.5,"local_radius":3,"drop_punct_only":true}`)) + }) + + It("keeps an explicit 0 (off) and false as given", func() { + s, err := parseGuardOptions(opts("guard_min_local_conf:0", "guard_drop_punct_only:false")) + Expect(err).ToNot(HaveOccurred()) + Expect(s).To(MatchJSON(`{"min_local_conf":0,"drop_punct_only":false}`)) + }) + + DescribeTable("rejects bad values at load", + func(opt, msg string) { + _, err := parseGuardOptions(opts(opt)) + Expect(err).To(MatchError(ContainSubstring(msg))) + }, + Entry("confidence not a number", "guard_min_local_conf:high", "is not a number"), + Entry("confidence above one", "guard_min_local_conf:1.5", "out of range"), + Entry("negative confidence", "guard_min_local_conf:-0.1", "out of range"), + Entry("radius zero", "guard_local_radius:0", "out of range"), + Entry("radius NaN", "guard_local_radius:NaN", "is not a number"), + Entry("bool typo", "guard_drop_punct_only:ture", "is not a boolean"), + ) + + It("fails the load, naming the symbol, on a library without the filter", func() { + CppTranscribePathJSONWith = nil + _, err := parseGuardOptions(opts("guard_min_local_conf:0.5")) + Expect(err).To(MatchError(ContainSubstring("parakeet_capi_transcribe_path_json_with"))) + }) + + It("fails Load for guard_* with vad:true when the segmenter entry point is missing", func() { + CppTranscribePathJSONVadWith = nil + f := newFakeLib().withModel("asr.gguf", modelKindASR) + restore := f.install() + defer restore() + savedVad := CppTranscribePathJSONVad + CppTranscribePathJSONVad = func(uintptr, string, int32) uintptr { return 0 } + defer func() { CppTranscribePathJSONVad = savedVad }() + err := (&ParakeetCpp{}).Load(&pb.ModelOptions{ModelFile: "asr.gguf", Options: []string{"vad:true", "guard_min_local_conf:0.5"}}) + Expect(err).To(MatchError(ContainSubstring("parakeet_capi_transcribe_path_json_vad_with"))) + }) + + It("fails Load on a bad value before any model is loaded", func() { + f := newFakeLib().withModel("asr.gguf", modelKindASR) + restore := f.install() + defer restore() + err := (&ParakeetCpp{}).Load(&pb.ModelOptions{ModelFile: "asr.gguf", Options: []string{"guard_min_local_conf:2"}}) + Expect(err).To(MatchError(ContainSubstring("guard_min_local_conf"))) + Expect(f.loadedPaths).To(BeEmpty()) + }) + + It("routes offline transcription through the filter entry point, past the batcher", func() { + wav := filepath.Join(GinkgoT().TempDir(), "a.wav") + writeMono16kWav(wav, 16000) + // A non-nil batcher: without the filter the request would go to it. + p := &ParakeetCpp{ctxPtr: 7, bat: &batcher{}, guardOptions: `{"min_local_conf":0.5}`} + res, err := p.AudioTranscription(context.Background(), &pb.TranscriptRequest{Dst: wav}) + Expect(err).ToNot(HaveOccurred()) + Expect(usedWith).To(BeTrue()) + Expect(gotOpts).To(Equal(`{"min_local_conf":0.5}`)) + Expect(res.Text).To(Equal("hello.")) + }) + + It("sends the filter keys together with the VAD keys to the segmenter when vad is on", func() { + p := &ParakeetCpp{ctxPtr: 7, vad: true, vadOptions: `{"trim":0}`, guardOptions: `{"min_local_conf":0.5}`} + _, err := p.transcribePathDoc("/x/long.wav") + Expect(err).ToNot(HaveOccurred()) + Expect(usedVadWith).To(BeTrue()) + Expect(usedWith).To(BeFalse()) + Expect(gotOpts).To(MatchJSON(`{"trim":0,"min_local_conf":0.5}`)) + }) + + It("does not reach the filter entry point when the filter is off", func() { + CppTranscribePathJSON = func(uintptr, string, int32) uintptr { + return pool.cstr(`{"text":"plain.","frame_sec":0.08,"words":[],"tokens":[]}`) + } + p := &ParakeetCpp{ctxPtr: 7} + doc, err := p.transcribePathDoc("/x/a.wav") + Expect(err).ToNot(HaveOccurred()) + Expect(doc.Text).To(Equal("plain.")) + Expect(usedWith).To(BeFalse()) + Expect(doc.Guard).To(BeNil()) + }) + + It("reads the dropped word count of the document", func() { + p := &ParakeetCpp{ctxPtr: 7, guardOptions: `{"drop_punct_only":true}`} + doc, err := p.transcribePathDoc("/x/a.wav") + Expect(err).ToNot(HaveOccurred()) + Expect(doc.Guard).ToNot(BeNil()) + Expect(doc.Guard.DroppedWords).To(Equal(2)) + }) +}) diff --git a/backend/go/parakeet-cpp/main.go b/backend/go/parakeet-cpp/main.go index fe1bb8ce6..fbc6cb6ee 100644 --- a/backend/go/parakeet-cpp/main.go +++ b/backend/go/parakeet-cpp/main.go @@ -163,6 +163,17 @@ func main() { purego.RegisterLibFunc(&CppDiarizeNamedPCMJSON, lib, "parakeet_capi_diarize_named_pcm_json") } + // Encoder fingerprint of the speaker registry (additive, no ABI bump). + if sym, err := purego.Dlsym(lib, "parakeet_capi_speaker_registry_add_embedding_fp"); err == nil && sym != 0 { + purego.RegisterLibFunc(&CppSpeakerRegistryAddEmbeddingFP, lib, "parakeet_capi_speaker_registry_add_embedding_fp") + purego.RegisterLibFunc(&CppSpeakerRegistrySetStrict, lib, "parakeet_capi_speaker_registry_set_strict") + purego.RegisterLibFunc(&CppSpeakerEncoderFamily, lib, "parakeet_capi_speaker_encoder_family") + } + // Word filter on transcription (additive, no ABI bump). + if sym, err := purego.Dlsym(lib, "parakeet_capi_transcribe_path_json_with"); err == nil && sym != 0 { + purego.RegisterLibFunc(&CppTranscribePathJSONWith, lib, "parakeet_capi_transcribe_path_json_with") + } + for _, lf := range []LibFuncs{ {&CppSpeakerIdentity, "parakeet_capi_speaker_identity"}, {&CppSpeakerDim, "parakeet_capi_speaker_dim"}, diff --git a/backend/go/parakeet-cpp/profiles.go b/backend/go/parakeet-cpp/profiles.go index b902b84e3..13d9fb92b 100644 --- a/backend/go/parakeet-cpp/profiles.go +++ b/backend/go/parakeet-cpp/profiles.go @@ -17,6 +17,11 @@ func (p *ParakeetCpp) Status() (pb.StatusResponse, error) { dim := CppSpeakerDim(p.spkCtx) if identity != 0 && dim > 0 { result.SpeakerEncoder = &pb.SpeakerEncoder{Identity: goStringFromCPtr(identity), Dimension: dim} + if CppSpeakerEncoderFamily != nil { + if family := CppSpeakerEncoderFamily(p.spkCtx); family != 0 { + result.SpeakerEncoder.Family = goStringFromCPtr(family) + } + } } } return pb.StatusResponse{ diff --git a/backend/go/parakeet-cpp/roles.go b/backend/go/parakeet-cpp/roles.go index de31f7cb9..e34a51cbe 100644 --- a/backend/go/parakeet-cpp/roles.go +++ b/backend/go/parakeet-cpp/roles.go @@ -171,6 +171,14 @@ func (p *ParakeetCpp) loadRoles(opts *pb.ModelOptions) error { return err } + strict, err := optBool(opts, "speaker_strict", false) + if err != nil { + return err + } + if strict && CppSpeakerRegistrySetStrict == nil { + return errors.New("parakeet-cpp: speaker_strict needs a libparakeet.so with parakeet_capi_speaker_registry_set_strict; rebuild the backend against a newer parakeet.cpp") + } + latency, err := parseDiarLatency(optString(opts, "diarization_latency")) if err != nil { return err @@ -311,6 +319,7 @@ func (p *ParakeetCpp) loadRoles(opts *pb.ModelOptions) error { return errors.New("parakeet-cpp: speaker_model needs a diarization model (the primary, diarization_model: or diar_component:)") } p.speakerAccept, p.speakerMargin = accept, margin + p.speakerStrict = strict p.diarLatency = latency return nil } diff --git a/backend/go/parakeet-cpp/scene.go b/backend/go/parakeet-cpp/scene.go index 099b968e2..4bac462fe 100644 --- a/backend/go/parakeet-cpp/scene.go +++ b/backend/go/parakeet-cpp/scene.go @@ -120,6 +120,9 @@ func (p *ParakeetCpp) sceneBegin(voices []*pb.KnownVoice) sceneStreamHandle { opts.SpeakerMargin = p.speakerMargin s := CppSceneStreamBeginSpeaker(0, diar, tag, p.spkCtx, reg, &opts) if s == 0 { + // The library also refuses a registry from another encoder (or without a + // fingerprint under speaker_strict) here and says why on the speaker context. + xlog.Warn("parakeet-cpp: could not start a live session with speaker names", "error", CppLastError(p.spkCtx)) p.freeSpeakerRegistry(reg) return sceneStreamHandle{} } diff --git a/backend/go/parakeet-cpp/scene_test.go b/backend/go/parakeet-cpp/scene_test.go index ca4005707..877d40423 100644 --- a/backend/go/parakeet-cpp/scene_test.go +++ b/backend/go/parakeet-cpp/scene_test.go @@ -123,6 +123,9 @@ var _ = Describe("scene stream with speaker names", func() { It("frees the registry when the speaker begin fails, and degrades to no scene stream", func() { CppSceneStreamBeginSpeaker = func(asr, diar, tag, spk, reg uintptr, o *cSceneOpts) uintptr { return 0 } + lastErr := CppLastError + defer func() { CppLastError = lastErr }() + CppLastError = func(uintptr) string { return "registry encoder family differs" } p := &ParakeetCpp{diarCtx: 1, spkCtx: 2} h := p.sceneBegin([]*pb.KnownVoice{{Name: "Ada", Embedding: []float32{1, 0}}}) Expect(h.s).To(Equal(uintptr(0))) diff --git a/backend/go/parakeet-cpp/speaker_fingerprint_test.go b/backend/go/parakeet-cpp/speaker_fingerprint_test.go new file mode 100644 index 000000000..03a53b347 --- /dev/null +++ b/backend/go/parakeet-cpp/speaker_fingerprint_test.go @@ -0,0 +1,224 @@ +package main + +import ( + "fmt" + + pb "github.com/mudler/LocalAI/pkg/grpc/proto" + . "github.com/onsi/ginkgo/v2" + . "github.com/onsi/gomega" +) + +type fpAdd struct{ key, family, weights string } + +const ( + famECAPA = "voicedetect:ecapa_tdnn:speechbrain/spkrec-ecapa-voxceleb:192" + famCAMPP = "voicedetect:campplus:3dspeaker/campplus:192" + hashLoad = "sha256:aaaa" + hashOther = "sha256:bbbb" +) + +// The fingerprint specs run against stubbed C entry points, like the registry +// specs in speaker_registry_test.go. +var _ = Describe("speaker registry encoder fingerprint", func() { + var ( + restore func() + pool *diarizeCstrPool + plain []string + fp []fpAdd + strictSet []int32 + freed []uintptr + fpFails bool + ) + BeforeEach(func() { + sNew, sFree, sAdd, sAddFP, sDim, sErr := CppSpeakerRegistryNew, CppSpeakerRegistryFree, CppSpeakerRegistryAddEmbedding, CppSpeakerRegistryAddEmbeddingFP, CppSpeakerDim, CppSpeakerRegistryLastError + sStrict, sFam, sID := CppSpeakerRegistrySetStrict, CppSpeakerEncoderFamily, CppSpeakerIdentity + restore = func() { + CppSpeakerRegistryNew, CppSpeakerRegistryFree, CppSpeakerRegistryAddEmbedding, CppSpeakerRegistryAddEmbeddingFP, CppSpeakerDim, CppSpeakerRegistryLastError = sNew, sFree, sAdd, sAddFP, sDim, sErr + CppSpeakerRegistrySetStrict, CppSpeakerEncoderFamily, CppSpeakerIdentity = sStrict, sFam, sID + } + pool = &diarizeCstrPool{} + plain, fp, strictSet, freed, fpFails = nil, nil, nil, nil, false + CppSpeakerDim = func(uintptr) int32 { return 2 } + CppSpeakerRegistryNew = func() uintptr { return 77 } + CppSpeakerRegistryFree = func(r uintptr) { freed = append(freed, r) } + CppSpeakerRegistryLastError = func(uintptr) string { return "stub error" } + CppSpeakerRegistryAddEmbedding = func(_ uintptr, name string, _ *float32, _ int32) int32 { + plain = append(plain, name) + return 0 + } + CppSpeakerRegistryAddEmbeddingFP = func(_ uintptr, name string, _ *float32, _ int32, family, weights string) int32 { + if fpFails { + return 1 + } + fp = append(fp, fpAdd{name, family, weights}) + return 0 + } + CppSpeakerRegistrySetStrict = func(_ uintptr, v int32) { strictSet = append(strictSet, v) } + CppSpeakerEncoderFamily = func(uintptr) uintptr { return pool.cstr(famECAPA) } + CppSpeakerIdentity = func(uintptr) uintptr { return pool.cstr(hashLoad) } + }) + AfterEach(func() { restore() }) + + v := func(id, family, weights string) *pb.KnownVoice { + return &pb.KnownVoice{Id: id, Name: "n" + id, Embedding: []float32{1, 0}, EncoderFamily: family, EncoderWeights: weights} + } + + It("stores the family and weights per voice and passes them to the registry", func() { + p := &ParakeetCpp{spkCtx: 5} + reg, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad), v("b", famECAPA, hashOther)}) + Expect(err).ToNot(HaveOccurred()) + Expect(reg).To(Equal(uintptr(77))) + Expect(fp).To(Equal([]fpAdd{{"a", famECAPA, hashLoad}, {"b", famECAPA, hashOther}})) + Expect(plain).To(BeEmpty()) + }) + + It("takes the family of the loaded encoder for a voice with the same weights and no family", func() { + p := &ParakeetCpp{spkCtx: 5} + _, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", "", hashLoad)}) + Expect(err).ToNot(HaveOccurred()) + Expect(fp).To(Equal([]fpAdd{{"a", famECAPA, hashLoad}})) + }) + + It("does not guess the family from another weights hash", func() { + p := &ParakeetCpp{spkCtx: 5} + _, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", "", hashOther)}) + Expect(err).ToNot(HaveOccurred()) + Expect(fp).To(BeEmpty()) + Expect(plain).To(Equal([]string{"a"})) + }) + + It("keeps the old path for voices without a fingerprint", func() { + p := &ParakeetCpp{spkCtx: 5} + reg, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", "", "")}) + Expect(err).ToNot(HaveOccurred()) + Expect(reg).To(Equal(uintptr(77))) + Expect(plain).To(Equal([]string{"a"})) + Expect(fp).To(BeEmpty()) + Expect(strictSet).To(BeEmpty()) + }) + + It("puts fingerprinted and unfingerprinted voices into one registry without a fingerprint", func() { + p := &ParakeetCpp{spkCtx: 5} + _, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad), v("b", "", ""), v("c", famCAMPP, hashOther)}) + Expect(err).ToNot(HaveOccurred()) + // The library refuses to mix them in one registry; the voice of another + // family can never match and is left out. + Expect(plain).To(ConsistOf("b", "a")) + Expect(fp).To(BeEmpty()) + }) + + It("leaves out voices of another family when a matching one exists", func() { + p := &ParakeetCpp{spkCtx: 5} + _, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad), v("c", famCAMPP, hashOther)}) + Expect(err).ToNot(HaveOccurred()) + Expect(fp).To(Equal([]fpAdd{{"a", famECAPA, hashLoad}})) + }) + + It("builds the registry from voices of another family when nothing else is usable, so the library refuses by name", func() { + p := &ParakeetCpp{spkCtx: 5} + reg, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("c", famCAMPP, hashOther)}) + Expect(err).ToNot(HaveOccurred()) + Expect(reg).To(Equal(uintptr(77))) + Expect(fp).To(Equal([]fpAdd{{"c", famCAMPP, hashOther}})) + }) + + It("frees the registry when the library refuses every voice", func() { + fpFails = true + p := &ParakeetCpp{spkCtx: 5} + reg, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad)}) + Expect(err).ToNot(HaveOccurred()) + Expect(reg).To(BeZero()) + Expect(freed).To(Equal([]uintptr{77})) + }) + + Describe("speaker_strict", func() { + It("drops unfingerprinted voices when a matching one exists", func() { + p := &ParakeetCpp{spkCtx: 5, speakerStrict: true} + _, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad), v("b", "", "")}) + Expect(err).ToNot(HaveOccurred()) + Expect(fp).To(Equal([]fpAdd{{"a", famECAPA, hashLoad}})) + Expect(plain).To(BeEmpty()) + }) + + It("builds a strict registry from unfingerprinted voices alone, so the library rejects it", func() { + p := &ParakeetCpp{spkCtx: 5, speakerStrict: true} + reg, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("b", "", "")}) + Expect(err).ToNot(HaveOccurred()) + Expect(reg).To(Equal(uintptr(77))) + Expect(plain).To(Equal([]string{"b"})) + Expect(strictSet).To(Equal([]int32{1})) + }) + + It("is not set on a registry that has a fingerprint", func() { + p := &ParakeetCpp{spkCtx: 5, speakerStrict: true} + _, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad)}) + Expect(err).ToNot(HaveOccurred()) + Expect(strictSet).To(BeEmpty()) + }) + }) + + It("treats every voice as unfingerprinted on a library without the fingerprint", func() { + CppSpeakerRegistryAddEmbeddingFP, CppSpeakerEncoderFamily = nil, nil + p := &ParakeetCpp{spkCtx: 5} + _, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad)}) + Expect(err).ToNot(HaveOccurred()) + Expect(plain).To(Equal([]string{"a"})) + }) + + It("reports the family of the encoder in Status next to its identity", func() { + sDim, sID := CppSpeakerDim, CppSpeakerIdentity + defer func() { CppSpeakerDim, CppSpeakerIdentity = sDim, sID }() + p := &ParakeetCpp{spkCtx: 5} + st, err := p.Status() + Expect(err).ToNot(HaveOccurred()) + Expect(st.GetSpeakerEncoder().GetIdentity()).To(Equal(hashLoad)) + Expect(st.GetSpeakerEncoder().GetFamily()).To(Equal(famECAPA)) + Expect(st.GetSpeakerEncoder().GetDimension()).To(Equal(int32(2))) + }) + + It("surfaces the library's family mismatch message from Diarize", func() { + rs := diarizeStubs() + defer rs() + sNamed, sPCM := CppDiarizeNamedPCMJSON, CppDiarizePCM + defer func() { CppDiarizeNamedPCMJSON, CppDiarizePCM = sNamed, sPCM }() + CppDiarizePCM = func(uintptr, *float32, int32, int32) uintptr { return 0 } + msg := fmt.Sprintf("registry encoder family %q differs from the speaker model's %q", famCAMPP, famECAPA) + CppDiarizeNamedPCMJSON = func(_, _, _ uintptr, _ *float32, _, _ int32, _, _ float32) uintptr { return 0 } + CppLastError = func(ctx uintptr) string { + if ctx == 2 { + return msg + } + return "" + } + CppSpeakerDim = func(uintptr) int32 { return 2 } + CppSpeakerRegistryNew = func() uintptr { return 77 } + CppSpeakerRegistryFree = func(uintptr) {} + p := &ParakeetCpp{diarCtx: 1, spkCtx: 2} + _, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(5), KnownVoices: []*pb.KnownVoice{v("c", famCAMPP, hashOther)}}) + Expect(err).To(HaveOccurred()) + Expect(err.Error()).To(ContainSubstring(famCAMPP)) + Expect(err.Error()).To(ContainSubstring(famECAPA)) + Expect(fp).To(Equal([]fpAdd{{"c", famCAMPP, hashOther}})) + }) +}) + +var _ = Describe("speaker_strict option", func() { + It("fails Load on a library without the strict switch", func() { + saved := CppSpeakerRegistrySetStrict + CppSpeakerRegistrySetStrict = nil + defer func() { CppSpeakerRegistrySetStrict = saved }() + f := newFakeLib().withModel("diar.gguf", modelKindDiarization).withModel("spk.gguf", modelKindSpeaker) + restore := f.install() + defer restore() + err := (&ParakeetCpp{}).Load(&pb.ModelOptions{ModelFile: "diar.gguf", Options: []string{"speaker_model:spk.gguf", "speaker_strict:true"}}) + Expect(err).To(HaveOccurred()) + }) + + It("rejects a value that is not a boolean", func() { + f := newFakeLib().withModel("diar.gguf", modelKindDiarization) + restore := f.install() + defer restore() + err := (&ParakeetCpp{}).Load(&pb.ModelOptions{ModelFile: "diar.gguf", Options: []string{"speaker_strict:maybe"}}) + Expect(err).To(MatchError(ContainSubstring("speaker_strict"))) + }) +}) diff --git a/backend/go/parakeet-cpp/speaker_registry.go b/backend/go/parakeet-cpp/speaker_registry.go index 2d56ebddf..89ce8ebf9 100644 --- a/backend/go/parakeet-cpp/speaker_registry.go +++ b/backend/go/parakeet-cpp/speaker_registry.go @@ -56,12 +56,61 @@ func parseSpeakerMargin(s string) (float32, error) { return nonZero(float32(v)), nil } +// encoderFingerprint is what the loaded speaker encoder reports about itself: +// the embedding space ("voicedetect:::") and the "sha256:" +// identity of its weights. Both are empty on a libparakeet.so without the +// fingerprint, which turns the checks off. +func (p *ParakeetCpp) encoderFingerprint() (family, weights string) { + if p.spkCtx == 0 || CppSpeakerRegistryAddEmbeddingFP == nil || CppSpeakerEncoderFamily == nil { + return "", "" + } + if ptr := CppSpeakerEncoderFamily(p.spkCtx); ptr != 0 { + family = goStringFromCPtr(ptr) + } + if CppSpeakerIdentity != nil { + if ptr := CppSpeakerIdentity(p.spkCtx); ptr != 0 { + weights = goStringFromCPtr(ptr) + } + } + return family, weights +} + +// voiceFamily is the encoder family of a registered voice, "" when unknown. A +// voice that only has a weights identity gets the family of the loaded encoder +// when the weights are the same file, because the same bytes are the same +// embedding space; a different hash proves nothing (another quantization of the +// encoder has the same family), so that voice stays unknown. +func voiceFamily(v *pb.KnownVoice, ownFamily, ownWeights string) string { + if f := v.GetEncoderFamily(); f != "" { + return f + } + if w := v.GetEncoderWeights(); w != "" && w == ownWeights { + return ownFamily + } + return "" +} + // buildSpeakerRegistryLocked makes a parakeet_speaker_registry from the registered voices // of one request or stream. Caller holds engineMu. It returns 0 (and no error) when there is // nothing to build: no speaker model loaded or no usable voices. A voice whose embedding size // differs from the speaker model's, or that the C side refuses, is skipped with a warning // (without its name: the log is not for the caller who may not see voice names) so one bad // voice cannot fail every request. The caller frees a non-zero result with freeSpeakerRegistry. +// +// Encoder fingerprint. The library keeps one registry per encoder and checks it against the +// speaker model before it names anyone, so the voices are sorted by what is known about the +// encoder that made them: +// +// - matching: the family is the one of the loaded encoder; +// - other: another family, which can never name a speaker here; +// - unfingerprinted: no family (registered before it was recorded, or by an encoder that +// cannot report one). The library refuses to mix these with fingerprinted voices in one +// registry, so when there are any, the request keeps the old behaviour: they and the +// matching voices go into one registry that has no fingerprint, and the encoder is +// unverified (logged). With speaker_strict they are dropped instead. +// +// With no usable voice but voices of another family, the registry is built from those, so the +// library refuses the request and its message names both families. func (p *ParakeetCpp) buildSpeakerRegistryLocked(voices []*pb.KnownVoice) (uintptr, error) { if p.spkCtx == 0 || CppSpeakerRegistryNew == nil || CppSpeakerRegistryAddEmbedding == nil || len(voices) == 0 { return 0, nil @@ -74,7 +123,10 @@ func (p *ParakeetCpp) buildSpeakerRegistryLocked(voices []*pb.KnownVoice) (uintp if reg == 0 { return 0, status.Error(codes.Internal, "parakeet-cpp: could not create a speaker registry") } - added, skipped := 0, 0 + ownFamily, ownWeights := p.encoderFingerprint() + skipped := 0 + var matching, other, plain []*pb.KnownVoice + family := map[*pb.KnownVoice]string{} for _, v := range voices { emb := v.GetEmbedding() if v.GetName() == "" || len(emb) == 0 { @@ -87,7 +139,67 @@ func (p *ParakeetCpp) buildSpeakerRegistryLocked(voices []*pb.KnownVoice) (uintp skipped++ continue } - if rc := CppSpeakerRegistryAddEmbedding(reg, voiceKey(v), &emb[0], int32(len(emb))); rc != 0 { + f := "" + if ownFamily != "" { + f = voiceFamily(v, ownFamily, ownWeights) + } + family[v] = f + switch { + case f == "": + plain = append(plain, v) + case f == ownFamily: + matching = append(matching, v) + default: + other = append(other, v) + } + } + + // use is the voices that go into the registry; fingerprinted says whether they carry one. + var use []*pb.KnownVoice + fingerprinted, strictPlain := false, false + switch { + case len(plain) > 0 && !p.speakerStrict: + use = append(plain, matching...) + if ownFamily != "" { + xlog.Warn("parakeet-cpp: registered voices without an encoder fingerprint are used unverified; register them again to record the encoder", + "unfingerprinted", len(plain)) + } + skipped += len(other) + case len(matching) > 0: + use, fingerprinted = matching, true + skipped += len(other) + len(plain) + case len(other) > 0: + // Nothing usable here: let the library refuse and say which families differ. + use, fingerprinted = other, true + skipped += len(plain) + case len(plain) > 0: + // speaker_strict: the library refuses a registry without a fingerprint. + use, strictPlain = plain, true + } + if len(use) == 0 { + if skipped > 0 { + xlog.Warn("parakeet-cpp: no registered voice is usable with this speaker model; speakers stay unnamed", "skipped", skipped) + } + CppSpeakerRegistryFree(reg) + return 0, nil + } + if p.speakerStrict && len(plain) > 0 && !strictPlain { + xlog.Warn("parakeet-cpp: speaker_strict: skipped registered voices without an encoder fingerprint", "skipped", len(plain)) + } + + if strictPlain && CppSpeakerRegistrySetStrict != nil { + CppSpeakerRegistrySetStrict(reg, 1) + } + added := 0 + for _, v := range use { + emb := v.GetEmbedding() + var rc int32 + if fingerprinted { + rc = CppSpeakerRegistryAddEmbeddingFP(reg, voiceKey(v), &emb[0], int32(len(emb)), family[v], v.GetEncoderWeights()) + } else { + rc = CppSpeakerRegistryAddEmbedding(reg, voiceKey(v), &emb[0], int32(len(emb))) + } + if rc != 0 { xlog.Warn("parakeet-cpp: skipped a registered voice the speaker registry refused", "error", CppSpeakerRegistryLastError(reg)) skipped++ continue diff --git a/backend/go/parakeet-cpp/vad.go b/backend/go/parakeet-cpp/vad.go index ea1c48c5d..a3c3f5228 100644 --- a/backend/go/parakeet-cpp/vad.go +++ b/backend/go/parakeet-cpp/vad.go @@ -30,10 +30,13 @@ var vadTuning = []struct { {"vad_min_speech", "min_speech", 0, math.Inf(1), false}, {"vad_speech_pad", "speech_pad", 0, math.Inf(1), false}, {"vad_max_segment", "max_segment", 0, math.Inf(1), true}, + // vad_trim shrinks each transcription segment to its speech plus this much. + // Unset keeps the library default (0.3 s); 0 keeps the whole cuts. + {"vad_trim", "trim", 0, math.Inf(1), false}, } // parseVADTuning reads the vad_threshold, vad_min_pause, vad_min_speech, -// vad_speech_pad and vad_max_segment model options and returns them as the JSON +// vad_speech_pad, vad_max_segment and vad_trim model options and returns them as the JSON // options object of the C-API, or "" when none is set. A value that does not // parse or is out of range fails the load. func parseVADTuning(opts *pb.ModelOptions) (string, error) { diff --git a/backend/go/parakeet-cpp/vad_rpc_test.go b/backend/go/parakeet-cpp/vad_rpc_test.go index d2cf827ab..e32164fda 100644 --- a/backend/go/parakeet-cpp/vad_rpc_test.go +++ b/backend/go/parakeet-cpp/vad_rpc_test.go @@ -169,6 +169,15 @@ var _ = Describe("VAD tuning options", func() { Expect(s).To(MatchJSON(`{"threshold":0.6,"min_pause":0.3,"min_speech":0.2,"speech_pad":0.05,"max_segment":20}`)) }) + It("maps vad_trim to the trim key and allows 0, which keeps the whole cuts", func() { + s, err := parseVADTuning(opts("vad_trim:0.5")) + Expect(err).ToNot(HaveOccurred()) + Expect(s).To(MatchJSON(`{"trim":0.5}`)) + s, err = parseVADTuning(opts("vad_trim:0")) + Expect(err).ToNot(HaveOccurred()) + Expect(s).To(MatchJSON(`{"trim":0}`)) + }) + It("allows a zero speech pad", func() { s, err := parseVADTuning(opts("vad_speech_pad:0")) Expect(err).ToNot(HaveOccurred()) @@ -184,6 +193,8 @@ var _ = Describe("VAD tuning options", func() { Entry("threshold zero", "vad_threshold:0", "out of range"), Entry("threshold above one", "vad_threshold:1.5", "out of range"), Entry("negative pad", "vad_speech_pad:-1", "out of range"), + Entry("negative trim", "vad_trim:-0.1", "out of range"), + Entry("trim not a number", "vad_trim:long", "is not a number"), Entry("NaN", "vad_min_pause:NaN", "is not a number"), ) }) @@ -266,6 +277,14 @@ var _ = Describe("vad_model", func() { Expect(calledWith).To(BeFalse()) }) + It("passes vad_trim:0 to the segmenter, so the old whole cuts stay available", func() { + p := &ParakeetCpp{ctxPtr: 7, vad: true, vadOptions: `{"trim":0}`} + _, err := p.transcribePathDoc("/x/long.wav") + Expect(err).ToNot(HaveOccurred()) + Expect(calledWith).To(BeTrue()) + Expect(gotOpts).To(Equal(`{"trim":0}`)) + }) + It("passes tuning to the head through _with (null Silero context)", func() { p := &ParakeetCpp{ctxPtr: 7, vad: true, vadOptions: `{"max_segment":20}`} _, err := p.transcribePathDoc("/x/long.wav") diff --git a/core/backend/diarization.go b/core/backend/diarization.go index a87356cd4..16c9fb4fd 100644 --- a/core/backend/diarization.go +++ b/core/backend/diarization.go @@ -44,7 +44,7 @@ type DiarizationRequest struct { func (r *DiarizationRequest) toProto(threads uint32, modelIdentity string) *proto.DiarizeRequest { known := make([]*proto.KnownVoice, 0, len(r.KnownVoices)) for _, v := range r.KnownVoices { - known = append(known, &proto.KnownVoice{Id: v.ID, Name: v.Name, Embedding: v.Embedding, Model: v.Model}) + known = append(known, &proto.KnownVoice{Id: v.ID, Name: v.Name, Embedding: v.Embedding, Model: v.Model, EncoderFamily: v.Family, EncoderWeights: v.Weights}) } return &proto.DiarizeRequest{ ModelIdentity: modelIdentity, @@ -210,7 +210,7 @@ func speakerEncoderFromBackend(ctx context.Context, m grpcPkg.Backend) (schema.S return schema.SpeakerEncoder{}, err } e := r.GetSpeakerEncoder() - trusted := schema.SpeakerEncoder{Identity: e.GetIdentity(), Dimension: int(e.GetDimension())} + trusted := schema.SpeakerEncoder{Identity: e.GetIdentity(), Dimension: int(e.GetDimension()), Family: e.GetFamily()} if err := (schema.SpeakerProfiles{Version: 1, Encoder: trusted}).Validate(trusted); err != nil { return schema.SpeakerEncoder{}, status.Error(codes.Unimplemented, "backend does not expose trusted speaker encoder metadata") } @@ -232,7 +232,10 @@ func decodeSpeakerProfiles(raw string, trusted schema.SpeakerEncoder) (*schema.S // Portable registrations require exact loaded identity and dimension. Legacy // candidates use the trusted dimension when available; older backends without -// metadata retain their native dimension check. No registry entry sets it. +// metadata retain their native dimension check. A portable voice with other +// weights is dropped here unless it carries an encoder family: the backend +// compares families, accepts another quantization of the same encoder with a +// warning, and refuses another encoder by name. func compatiblePortableVoices(ctx context.Context, m grpcPkg.Backend, voices []voicerecognition.KnownVoice) []voicerecognition.KnownVoice { if len(voices) == 0 { return voices @@ -243,7 +246,8 @@ func compatiblePortableVoices(ctx context.Context, m grpcPkg.Backend, voices []v if err == nil && len(v.Embedding) != trusted.Dimension { continue } - if strings.HasPrefix(v.Model, "sha256:") && (err != nil || v.Model != trusted.Identity || len(v.Embedding) != trusted.Dimension) { + if strings.HasPrefix(v.Model, "sha256:") && (err != nil || len(v.Embedding) != trusted.Dimension || + (v.Model != trusted.Identity && v.Family == "")) { continue } out = append(out, v) diff --git a/core/backend/diarization_profiles_test.go b/core/backend/diarization_profiles_test.go index d559385a3..d93d576ca 100644 --- a/core/backend/diarization_profiles_test.go +++ b/core/backend/diarization_profiles_test.go @@ -78,10 +78,47 @@ var _ = Describe("portable voice compatibility", func() { }) }) +var _ = Describe("portable voice encoder family", func() { + const fam, other = "voicedetect:ecapa_tdnn:ecapa:192", "voicedetect:campplus:campplus:192" + It("keeps a voice with another weights hash when it has a family, for the backend to judge", func() { + identity := "sha256:" + strings.Repeat("a", 64) + otherHash := "sha256:" + strings.Repeat("b", 64) + m := &portableStatusBackend{identity: identity} + voices := []voicerecognition.KnownVoice{ + {ID: "same-family", Model: otherHash, Weights: otherHash, Family: fam, Embedding: []float32{1, 0}}, + {ID: "other-family", Model: otherHash, Weights: otherHash, Family: other, Embedding: []float32{1, 0}}, + {ID: "no-family", Model: otherHash, Weights: otherHash, Embedding: []float32{1, 0}}, + {ID: "wrong-size", Model: otherHash, Weights: otherHash, Family: fam, Embedding: []float32{1, 0, 0}}, + } + got := compatiblePortableVoices(context.Background(), m, voices) + Expect(got).To(HaveLen(2)) + Expect(got[0].ID).To(Equal("same-family")) + Expect(got[1].ID).To(Equal("other-family")) + }) + It("sends the fingerprint to the backend, offline and live", func() { + v := voicerecognition.KnownVoice{ID: "a", Name: "Ada", Embedding: []float32{1, 0}, Model: "sha256:x", Family: "f", Weights: "sha256:x"} + offline := (&DiarizationRequest{KnownVoices: []voicerecognition.KnownVoice{v}}).toProto(2, "model") + Expect(offline.KnownVoices[0].EncoderFamily).To(Equal("f")) + Expect(offline.KnownVoices[0].EncoderWeights).To(Equal("sha256:x")) + var live liveOptions + WithKnownVoices([]voicerecognition.KnownVoice{v})(&live) + cfg := liveConfigProto("en", live) + Expect(cfg.KnownVoices[0].EncoderFamily).To(Equal("f")) + Expect(cfg.KnownVoices[0].EncoderWeights).To(Equal("sha256:x")) + }) + It("reads the family of the loaded encoder from the backend status", func() { + m := &portableStatusBackend{identity: "sha256:" + strings.Repeat("a", 64), family: fam} + trusted, err := speakerEncoderFromBackend(context.Background(), m) + Expect(err).NotTo(HaveOccurred()) + Expect(trusted.Family).To(Equal(fam)) + }) +}) + type portableStatusBackend struct { grpcPkg.Backend identity string dimension int32 + family string } func (m *portableStatusBackend) Status(context.Context) (*pb.StatusResponse, error) { @@ -89,7 +126,7 @@ func (m *portableStatusBackend) Status(context.Context) (*pb.StatusResponse, err if dim == 0 { dim = 2 } - return &pb.StatusResponse{SpeakerEncoder: &pb.SpeakerEncoder{Identity: m.identity, Dimension: dim}}, nil + return &pb.StatusResponse{SpeakerEncoder: &pb.SpeakerEncoder{Identity: m.identity, Dimension: dim, Family: m.family}}, nil } var _ = Describe("selection before portable compatibility", func() { diff --git a/core/backend/transcript_live.go b/core/backend/transcript_live.go index 7fcd27f82..01714bfd4 100644 --- a/core/backend/transcript_live.go +++ b/core/backend/transcript_live.go @@ -243,7 +243,7 @@ func WithKnownVoices(v []voicerecognition.KnownVoice) LiveOption { func liveConfigProto(language string, o liveOptions) *proto.TranscriptLiveConfig { cfg := &proto.TranscriptLiveConfig{Language: language, SampleRate: liveSampleRate} for _, v := range o.knownVoices { - cfg.KnownVoices = append(cfg.KnownVoices, &proto.KnownVoice{Id: v.ID, Name: v.Name, Embedding: v.Embedding, Model: v.Model}) + cfg.KnownVoices = append(cfg.KnownVoices, &proto.KnownVoice{Id: v.ID, Name: v.Name, Embedding: v.Embedding, Model: v.Model, EncoderFamily: v.Family, EncoderWeights: v.Weights}) } return cfg } diff --git a/core/http/endpoints/localai/voice_register.go b/core/http/endpoints/localai/voice_register.go index 9ae4d1785..d4a96dd42 100644 --- a/core/http/endpoints/localai/voice_register.go +++ b/core/http/endpoints/localai/voice_register.go @@ -34,7 +34,7 @@ func VoiceRegisterEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, } var embedding []float32 - var encoder string + var encoder, family string if input.SpeakerProfiles != nil { if input.Audio != "" || input.SpeakerSlot == nil { return echo.NewHTTPError(http.StatusBadRequest, "speaker_profiles requires speaker_slot and excludes audio") @@ -47,7 +47,7 @@ func VoiceRegisterEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, if err != nil { return echo.NewHTTPError(http.StatusBadRequest, err.Error()) } - embedding, encoder = selected.Embedding, trusted.Identity + embedding, encoder, family = selected.Embedding, trusted.Identity, trusted.Family } else { if input.SpeakerSlot != nil { return echo.NewHTTPError(http.StatusBadRequest, "speaker_slot requires speaker_profiles") @@ -63,7 +63,11 @@ func VoiceRegisterEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader, } embedding, encoder = res.GetEmbedding(), res.GetModel() } - stored, err := registry.Register(c.Request().Context(), embedding, voiceMetadata(input.Name, input.Labels, encoder)) + meta := voiceMetadata(input.Name, input.Labels, encoder) + // Only the portable route knows the family: it comes from the loaded + // encoder. A voice-detect embedding has none, so it stays unfingerprinted. + meta.EncoderFamily = family + stored, err := registry.Register(c.Request().Context(), embedding, meta) if err != nil { return err } diff --git a/core/schema/speaker_profiles.go b/core/schema/speaker_profiles.go index 20d77aa41..006456928 100644 --- a/core/schema/speaker_profiles.go +++ b/core/schema/speaker_profiles.go @@ -13,6 +13,9 @@ import ( type SpeakerEncoder struct { Identity string `json:"identity"` Dimension int `json:"dimension"` + // Family is the embedding space of the encoder. The server fills it from the + // loaded encoder; exported profiles do not carry it and it is not matched. + Family string `json:"family,omitempty"` } // SpeakerProfileInterval locates retained clean audio in the original recording, @@ -51,7 +54,7 @@ func (p SpeakerProfiles) Validate(trusted SpeakerEncoder) error { if !speakerEncoderIdentity.MatchString(trusted.Identity) || trusted.Dimension <= 0 { return fmt.Errorf("invalid trusted speaker encoder metadata") } - if p.Encoder != trusted { + if p.Encoder.Identity != trusted.Identity || p.Encoder.Dimension != trusted.Dimension { return fmt.Errorf("speaker profile encoder does not match loaded encoder") } seen := make(map[int]bool, len(p.Speakers)) diff --git a/core/schema/speaker_profiles_test.go b/core/schema/speaker_profiles_test.go index 4334af847..b675c15d5 100644 --- a/core/schema/speaker_profiles_test.go +++ b/core/schema/speaker_profiles_test.go @@ -31,6 +31,12 @@ var _ = Describe("Portable speaker profiles", func() { p.Encoder.Dimension = 3 Expect(p.Validate(trusted)).To(HaveOccurred()) }) + It("matches on identity and dimension, not on the family only the server knows", func() { + trusted.Family = "voicedetect:ecapa_tdnn:ecapa:192" + Expect(p.Validate(trusted)).To(Succeed()) + _, err := p.Select(3, trusted) + Expect(err).NotTo(HaveOccurred()) + }) It("round trips the backend JSON including unavailable profiles without embeddings", func() { raw := `{"version":1,"encoder":{"identity":"` + trusted.Identity + `","dimension":2},"speakers":[{"speaker":3,"clean_duration":3,"intervals":[{"start":0,"end":3}],"unavailable_reason":null,"embedding":[0.6,0.8]},{"speaker":4,"clean_duration":1,"intervals":[{"start":4,"end":5}],"unavailable_reason":"insufficient_clean_speech"}]}` Expect(json.Unmarshal([]byte(raw), &p)).To(Succeed()) diff --git a/core/services/voicerecognition/known_voices.go b/core/services/voicerecognition/known_voices.go index 0f4cf8fba..c01dfbd53 100644 --- a/core/services/voicerecognition/known_voices.go +++ b/core/services/voicerecognition/known_voices.go @@ -15,6 +15,11 @@ type KnownVoice struct { Name string Embedding []float32 Model string + // Family and Weights fingerprint the encoder that made the embedding, as + // far as it is known: the embedding space and the "sha256:" identity of the + // exact weights. Empty for a voice registered without them. + Family string + Weights string } // SpeakerModelFromOptions returns the value of a speaker_model: entry in @@ -62,7 +67,7 @@ func SelectKnownVoices(entries []Entry, speakerModelPath string) KnownVoiceSelec // Hash-tagged portable registrations are checked against the loaded // encoder by the backend, never against a filename or dimension alone. case strings.HasPrefix(e.Metadata.Model, "sha256:"), EncoderTag(e.Metadata.Model) == tag: - sel.Voices = append(sel.Voices, KnownVoice{ID: e.Metadata.ID, Name: e.Metadata.Name, Embedding: e.Embedding, Model: e.Metadata.Model}) + sel.Voices = append(sel.Voices, knownVoice(e)) default: sel.OtherEncoder++ } @@ -76,6 +81,14 @@ func SelectKnownVoices(entries []Entry, speakerModelPath string) KnownVoiceSelec return sel } +func knownVoice(e Entry) KnownVoice { + v := KnownVoice{ID: e.Metadata.ID, Name: e.Metadata.Name, Embedding: e.Embedding, Model: e.Metadata.Model, Family: e.Metadata.EncoderFamily} + if strings.HasPrefix(e.Metadata.Model, "sha256:") { + v.Weights = e.Metadata.Model + } + return v +} + // KnownVoicesFor lists the registry and selects the voices for a speaker model. func KnownVoicesFor(ctx context.Context, reg Registry, speakerModelPath string) (KnownVoiceSelection, error) { entries, err := reg.List(ctx) diff --git a/core/services/voicerecognition/known_voices_test.go b/core/services/voicerecognition/known_voices_test.go index fbe5510cf..6039a1a47 100644 --- a/core/services/voicerecognition/known_voices_test.go +++ b/core/services/voicerecognition/known_voices_test.go @@ -2,6 +2,7 @@ package voicerecognition_test import ( "context" + "encoding/json" "errors" "github.com/mudler/LocalAI/core/services/voicerecognition" @@ -117,6 +118,34 @@ func (r listRegistry) List(context.Context) ([]voicerecognition.Entry, error) { return r.entries, r.err } +var _ = Describe("encoder fingerprint of selected voices", func() { + const hash = "sha256:aaaa" + It("passes the family and takes the weights from a hash tag", func() { + e := entry("ada", hash, 1, 0) + e.Metadata.EncoderFamily = "voicedetect:ecapa_tdnn:ecapa:192" + sel := voicerecognition.SelectKnownVoices([]voicerecognition.Entry{e}, "spk.gguf") + Expect(sel.Voices).To(HaveLen(1)) + Expect(sel.Voices[0].Family).To(Equal("voicedetect:ecapa_tdnn:ecapa:192")) + Expect(sel.Voices[0].Weights).To(Equal(hash)) + }) + It("leaves a file-name tagged voice unfingerprinted", func() { + sel := voicerecognition.SelectKnownVoices([]voicerecognition.Entry{entry("ada", "spk.gguf", 1, 0)}, "spk.gguf") + Expect(sel.Voices[0].Family).To(BeEmpty()) + Expect(sel.Voices[0].Weights).To(BeEmpty()) + }) + It("loads a stored voice that has no family, and keeps the family when there is one", func() { + var old, fresh voicerecognition.Metadata + Expect(json.Unmarshal([]byte(`{"id":"1","name":"ada","registered_at":"2026-01-01T00:00:00Z","model":"spk.gguf"}`), &old)).To(Succeed()) + Expect(old.EncoderFamily).To(BeEmpty()) + raw, err := json.Marshal(voicerecognition.Metadata{ID: "2", Name: "ben", Model: hash, EncoderFamily: "f"}) + Expect(err).ToNot(HaveOccurred()) + Expect(json.Unmarshal(raw, &fresh)).To(Succeed()) + Expect(fresh.EncoderFamily).To(Equal("f")) + raw, _ = json.Marshal(old) + Expect(string(raw)).ToNot(ContainSubstring("encoder_family")) + }) +}) + var _ = Describe("KnownVoicesFor", func() { It("selects from the registry listing", func() { sel, err := voicerecognition.KnownVoicesFor(context.Background(), listRegistry{entries: []voicerecognition.Entry{entry("ada", "m.gguf", 1)}}, "m.gguf") diff --git a/core/services/voicerecognition/registry.go b/core/services/voicerecognition/registry.go index b76c0a16c..6e51272f2 100644 --- a/core/services/voicerecognition/registry.go +++ b/core/services/voicerecognition/registry.go @@ -58,6 +58,11 @@ type Metadata struct { // backend's model name, by default the GGUF file name). Empty for voices // registered before this field existed. Model string `json:"model,omitempty"` + // EncoderFamily is the embedding space of the encoder ("voicedetect:::"), + // recorded when the encoder reports it (portable enrollment from speaker + // profiles). Empty when unknown, and for voices registered before it existed. + // Model then holds the weights identity ("sha256:") for the same voices. + EncoderFamily string `json:"encoder_family,omitempty"` } // Match is a single result from Identify, ranked by similarity. diff --git a/docs/content/features/audio-to-text.md b/docs/content/features/audio-to-text.md index 4ecdbc109..98cf54a24 100644 --- a/docs/content/features/audio-to-text.md +++ b/docs/content/features/audio-to-text.md @@ -205,6 +205,7 @@ The same backend also serves the `/v1/audio/diarization` and `/v1/audio/classifi | `speaker_model:` | a model with a diarization model | names registered speakers (see [Voice Recognition]({{% relref "voice-recognition" %}}#naming-speakers-in-diarization-and-live-transcription)) | | `speaker_threshold:` | a model with `speaker_model` | distance (1 minus cosine similarity) under which a speaker is named, in (0, 2); default `0.5` | | `speaker_margin:` | a model with `speaker_model` | how much the best match must beat the runner-up, in [0, 1); default `0.05` | +| `speaker_strict:` | a model with `speaker_model` | do not use registered voices that have no encoder fingerprint (see [Voice Recognition]({{% relref "voice-recognition" %}}#encoder-fingerprint)); default `false` | With a `diarization_model` companion, `/v1/audio/transcriptions` labels each segment with its `speaker` (`"0"`, `"1"`, ... in order of first appearance) and splits segments where the speaker changes; with `timestamp_granularities[]=word` each word carries its speaker too. With `stream=true` the closing `transcript.text.done` event lists the segments with their speakers. Pass `-F diarize=false` to skip diarization for one request. The diarization GGUF can also be imported directly: `local-ai models import https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/nemotron-3-diarization-f16.gguf`. @@ -305,9 +306,28 @@ The segmenter options below apply to both `vad:true` and `vad_model`. Each is op | `vad_min_pause` | seconds | A silence this long separates two pieces | | `vad_min_speech` | seconds | Shorter speech runs are dropped | | `vad_max_segment` | seconds | Cap on the length of a piece (default 30) | +| `vad_trim` | seconds | Each piece shrinks to its first and last speech frame plus this much. Default `0.3`; `0` keeps the whole cuts, as before this option existed | `vad_speech_pad` (seconds) pads each region and only affects the [VAD endpoint]({{%relref "features/voice-activity-detection" %}}). `vad_model` needs a `libparakeet.so` that exports `parakeet_capi_transcribe_path_json_vad_with`; an older library fails the load with a message that names it. +### Dropping noise words (`guard_*`) + +A decode of noise or silence can contain words that no one said. An opt-in filter removes the words that stand alone or sit among low-confidence words, and the words that are only punctuation. It runs on the finished decode. It is off unless one of these options is set, and a bad value fails the load: + +| Option | Unit | Meaning | +|---|---|---| +| `guard_min_local_conf` | 0 to 1 | A word is dropped when the mean confidence of the words that start within `guard_local_radius` seconds of it, itself included, is below this. `0` is off. `0.5` is a good start; higher values also drop real words on some models | +| `guard_local_radius` | seconds, above 0 | The window of that mean. Default `5` | +| `guard_drop_punct_only` | `true` or `false` | Drop words that are only punctuation. A CTC model can emit a lone `.` on noise. Default `false` | + +```yaml +options: +- guard_min_local_conf:0.5 +- guard_drop_punct_only:true +``` + +The filter applies to offline transcription, and with `vad:true` or `vad_model` to each piece on its own. It bypasses dynamic batching, like `vad:true`. Streaming is not affected. Speech with confident words comes out the same as without the filter. The number of dropped words is written to the backend log at debug level; the transcription response has no field for it. The options need a `libparakeet.so` that exports `parakeet_capi_transcribe_path_json_with`; an older library fails the load with a message that names it. + ### Bundle GGUF files (several models in one file) A bundle is one GGUF file that holds several models, called components. Each component keeps its own licence. The backend opens the components it needs from the one file, so a single model YAML can serve transcription, VAD, diarization, speaker naming and sound events. A bundle needs a `libparakeet.so` from parakeet.cpp with bundle support (pin `781a973` or newer); the format is described in the [parakeet.cpp bundle documentation](https://github.com/mudler/parakeet.cpp/blob/master/docs/bundle.md). Single-model files and every existing option work as before. diff --git a/docs/content/features/voice-activity-detection.md b/docs/content/features/voice-activity-detection.md index b4bef8e53..814352704 100644 --- a/docs/content/features/voice-activity-detection.md +++ b/docs/content/features/voice-activity-detection.md @@ -149,6 +149,7 @@ All options are optional. An unset value keeps the default of the detector in us | `vad_min_pause` | seconds | `0.1` | `0.2` | A silence this long separates two segments; shorter gaps merge | | `vad_min_speech` | seconds | `0.25` | `0.1` | Shorter speech runs are dropped | | `vad_speech_pad` | seconds | `0.03` | `0` | Padding added around each segment | +| `vad_trim` | seconds | `0.3` | `0.3` | Only for transcription with `vad:true` or `vad_model`: each piece shrinks to its speech plus this much. `0` keeps the whole cuts. The endpoint ignores it | Option names differ from the Silero backend above (`min_silence_duration_ms` and `speech_pad_ms` are in milliseconds there). The same options tune transcription with `vad:true` or `vad_model`; see [audio to text]({{%relref "features/audio-to-text" %}}). Requests on one loaded model run one at a time. diff --git a/docs/content/features/voice-recognition.md b/docs/content/features/voice-recognition.md index d3c866ba0..f9ce777aa 100644 --- a/docs/content/features/voice-recognition.md +++ b/docs/content/features/voice-recognition.md @@ -223,6 +223,42 @@ backend skips a voice whose embedding size does not match the speaker model's, with a warning in the LocalAI log. Naming then falls back to the remaining voices, or to no names. +### Encoder fingerprint + +Two encoders can give embeddings of the same size (ECAPA and CAM++ both give +192 values), so a size match does not prove the voices and the `speaker_model:` +file share an embedding space. A voice enrolled from `speaker_profiles` (see +[Speaker Diarization]({{% relref "audio-diarization" %}})) is stored with the +encoder that made it: its **weights** (`sha256:` of the encoder file, kept in +the voice's `model` field as before) and its **family** +(`voicedetect:::`, read from the encoder GGUF metadata and +stored as `encoder_family`). The parakeet-cpp backend builds the registry with +that fingerprint, and libparakeet checks it against the loaded `speaker_model:` +before it names anyone: + +| Registered voices | Result | +|---|---| +| Same family, same weights | names are assigned | +| Same family, other weights (for example another quantization) | names are assigned, the library logs a warning | +| Another family, and no other usable voice | the request fails, and the error names both families | +| Another family, with usable voices of the right family | the other voices are left out, with a warning | +| No fingerprint | used as before, with a warning that the encoder is unverified | + +A voice with only a weights identity takes the family of the loaded encoder +when the weights are the same file. A voice with a different weights hash and +no family is dropped, as before. + +A voice registered from audio through the voice-detect backend has no +fingerprint: libvoicedetect reports no architecture or model name, so the +backend cannot tell the family, and only the file-name tag described above +applies. Such voices and fingerprinted voices cannot share one registry in the +library. When a request has any unfingerprinted voice, all of its voices are +used without the fingerprint check (the old behaviour). To get the check, enroll +every voice from `speaker_profiles`. With `speaker_strict:true` the backend +ignores unfingerprinted voices, and a request that has only those fails with +the library's message. The family is also reported in the internal backend +status next to the identity. + {{% notice warning %}} Do not set a `model_name:` option on the voice-detect model config. It replaces the default name, the voices are then tagged with it, and they no @@ -240,6 +276,7 @@ options). | `speaker_model:` | none | speaker encoder GGUF; needs a diarization model (the primary one, or `diarization_model:`) | | `speaker_threshold:` | `0.5` | largest distance (1 minus cosine similarity, the unit `/v1/voice/identify` reports) at which a speaker is named; must be in (0, 2) | | `speaker_margin:` | `0.05` | the best match must beat the runner-up by this much, otherwise the speaker stays unnamed; must be in [0, 1) | +| `speaker_strict:` | `false` | ignore registered voices that carry no [encoder fingerprint](#encoder-fingerprint); needs a libparakeet that exports `parakeet_capi_speaker_registry_set_strict` | parakeet.cpp's measured starting values for `speaker_threshold` are 0.5 for WeSpeaker ResNet34 and CAM++, and 0.3 for ECAPA. A lower value names fewer diff --git a/swagger/docs.go b/swagger/docs.go index cce42189e..271bbcf20 100644 --- a/swagger/docs.go +++ b/swagger/docs.go @@ -5281,6 +5281,10 @@ const docTemplate = `{ "dimension": { "type": "integer" }, + "family": { + "description": "embedding space of the encoder; empty when the backend cannot tell", + "type": "string" + }, "identity": { "description": "sha256 of loaded GGUF bytes", "type": "string" @@ -8207,6 +8211,10 @@ const docTemplate = `{ "dimension": { "type": "integer" }, + "family": { + "description": "Family is the embedding space of the encoder. The server fills it from the loaded encoder; exported profiles do not carry it and it is not matched.", + "type": "string" + }, "identity": { "type": "string" } diff --git a/swagger/swagger.json b/swagger/swagger.json index 6f1467a6a..cc74d5c7a 100644 --- a/swagger/swagger.json +++ b/swagger/swagger.json @@ -5278,6 +5278,10 @@ "dimension": { "type": "integer" }, + "family": { + "description": "embedding space of the encoder; empty when the backend cannot tell", + "type": "string" + }, "identity": { "description": "sha256 of loaded GGUF bytes", "type": "string" @@ -8204,6 +8208,10 @@ "dimension": { "type": "integer" }, + "family": { + "description": "Family is the embedding space of the encoder. The server fills it from the loaded encoder; exported profiles do not carry it and it is not matched.", + "type": "string" + }, "identity": { "type": "string" } diff --git a/swagger/swagger.yaml b/swagger/swagger.yaml index 84c48cd9a..70a585991 100644 --- a/swagger/swagger.yaml +++ b/swagger/swagger.yaml @@ -682,6 +682,9 @@ definitions: properties: dimension: type: integer + family: + description: embedding space of the encoder; empty when the backend cannot tell + type: string identity: description: sha256 of loaded GGUF bytes type: string @@ -2759,6 +2762,9 @@ definitions: properties: dimension: type: integer + family: + description: Family is the embedding space of the encoder. The server fills it from the loaded encoder; exported profiles do not carry it and it is not matched. + type: string identity: type: string type: object