mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-09 22:54:42 -04:00
feat(parakeet-cpp): encoder fingerprint for speaker naming, VAD trim and word filter options, pin bump (#12491)
* chore(parakeet-cpp): bump parakeet.cpp to 2de154c Brings in the speaker registry encoder fingerprint, the VAD segment trim and the opt-in word filter, a fix for a per-call thread count that stayed set on the process-wide backend after a Silero VAD pass, and bundle components loaded from memory. Assisted-by: Claude:claude-sonnet-5-5 [Claude Code] * feat(parakeet-cpp): encoder fingerprint for speaker naming, vad_trim and guard_* options Speaker naming. A registered voice now carries the encoder that made it: the embedding family (voicedetect:<arch>:<name>:<dim>) and the sha256 of the weights. The backend reports the family of the loaded speaker model in Status, voice enrollment from speaker_profiles stores it as encoder_family (old entries load without it), and the registry sent to parakeet.cpp is built with parakeet_capi_speaker_registry_add_embedding_fp. The library then refuses a registry of another encoder family and the error names both families; another quantization of the same family only warns. A voice with only a weights hash gets the loaded family when the hashes are equal. Voices without a fingerprint (registered from audio: libvoicedetect cannot report one) keep the file-name rule and are used with a warning. The library cannot mix them with fingerprinted voices in one registry, so a request that has any uses the old registry for all. speaker_strict:true drops them instead. A library without the symbols behaves as before. Transcription. vad_trim (seconds, 0 keeps the whole cuts) goes through the VAD options JSON, so it reaches /v1/vad and the segmenter. The guard_* options guard_min_local_conf, guard_local_radius and guard_drop_punct_only turn on the word filter through parakeet_capi_transcribe_path_json_with, or through the segmenter with vad:true. They are off by default, bad values fail the load, and a library without the symbol fails it with a clear message. The dropped word count is logged at debug level. Assisted-by: Claude:claude-sonnet-5-5 [Claude Code] --------- Co-authored-by: Ettore Di Giacinto <mudler@localai.io>
This commit is contained in:
1 parent
3d586cc3c5
commit
79a7631cc5
29 files changed
+876
-24
No files matched your search
@@ -851,6 +851,11 @@ message KnownVoice {
|
||||
string name = 1;
|
||||
repeated float embedding = 2;
|
||||
string model = 3;
|
||||
// Fingerprint of the encoder that made the embedding, when it is known: the
|
||||
// embedding space ("voicedetect:<arch>:<name>:<dim>") and the exact weights
|
||||
// ("sha256:<hex>"). Empty for voices registered before it was recorded.
|
||||
string encoder_family = 5;
|
||||
string encoder_weights = 6;
|
||||
}
|
||||
|
||||
message DiarizeResponse {
|
||||
@@ -913,6 +918,7 @@ message MemoryUsageData {
|
||||
message SpeakerEncoder {
|
||||
string identity = 1; // sha256 of loaded GGUF bytes
|
||||
int32 dimension = 2;
|
||||
string family = 3; // embedding space of the encoder; empty when the backend cannot tell
|
||||
}
|
||||
|
||||
message StatusResponse {
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# parakeet-cpp backend Makefile.
|
||||
#
|
||||
# Upstream pin lives below as PARAKEET_VERSION?=0cca477249ffb16c1623fb947d5bac0624961d41
|
||||
# Upstream pin lives below as PARAKEET_VERSION?=2de154c622830b62bfcfb92556dbb71bf0263ddd
|
||||
# (.github/bump_deps.sh) can find and update it - matches the
|
||||
# whisper.cpp / ds4 / vibevoice-cpp convention.
|
||||
#
|
||||
@@ -15,7 +15,7 @@
|
||||
# That's what the L0 smoke test uses. The default target below does the
|
||||
# proper clone-at-pin + cmake build so CI doesn't need a side-checkout.
|
||||
|
||||
PARAKEET_VERSION?=0cca477249ffb16c1623fb947d5bac0624961d41
|
||||
PARAKEET_VERSION?=2de154c622830b62bfcfb92556dbb71bf0263ddd
|
||||
PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp
|
||||
|
||||
GOCMD?=go
|
||||
|
||||
@@ -36,8 +36,12 @@ var (
|
||||
CppFree func(ctx uintptr)
|
||||
CppTranscribePath func(ctx uintptr, wavPath string, decoder int32) uintptr
|
||||
CppTranscribePathJSON func(ctx uintptr, wavPath string, decoder int32) uintptr
|
||||
CppFreeString func(s uintptr)
|
||||
CppLastError func(ctx uintptr) string
|
||||
// CppTranscribePathJSONWith is CppTranscribePathJSON with the optional word
|
||||
// filter (min_local_conf, local_radius, drop_punct_only as a JSON object).
|
||||
// nil on a libparakeet.so from before the filter.
|
||||
CppTranscribePathJSONWith func(ctx uintptr, wavPath string, decoder int32, optionsJSON string) uintptr
|
||||
CppFreeString func(s uintptr)
|
||||
CppLastError func(ctx uintptr) string
|
||||
|
||||
// Bundle GGUF (additive in the C-API, no ABI bump; see bundle.go). All three
|
||||
// are registered together and nil on an older libparakeet.so, where a
|
||||
@@ -151,6 +155,12 @@ var (
|
||||
CppSpeakerRegistryAddEmbedding func(reg uintptr, name string, emb *float32, dim int32) int32
|
||||
CppSpeakerRegistryLastError func(reg uintptr) string
|
||||
CppSceneStreamBeginSpeaker func(asr, diar, tagger, speaker, reg uintptr, o *cSceneOpts) uintptr
|
||||
// Encoder fingerprint (additive, ABI 10). Probed as a group; nil on a library
|
||||
// from before it. CppSpeakerEncoderFamily returns a borrowed char*, read it with
|
||||
// goStringFromCPtr and do not free it. A family or weights string "" means none.
|
||||
CppSpeakerRegistryAddEmbeddingFP func(reg uintptr, name string, emb *float32, dim int32, family, weights string) int32
|
||||
CppSpeakerRegistrySetStrict func(reg uintptr, strict int32)
|
||||
CppSpeakerEncoderFamily func(ctx uintptr) uintptr
|
||||
// CppDiarizeNamedPCMJSON takes two float32 arguments (acceptThreshold, margin), which
|
||||
// purego passes in floating-point registers. Not exercised without the real library.
|
||||
CppDiarizeNamedPCMJSON func(diar, speaker, reg uintptr, samples *float32, n, sampleRate int32, acceptThreshold, margin float32) uintptr
|
||||
@@ -201,6 +211,10 @@ type transcriptJSON struct {
|
||||
FrameSec float64 `json:"frame_sec"`
|
||||
Words []transcriptWord `json:"words"`
|
||||
Tokens []transcriptToken `json:"tokens"`
|
||||
// Guard is present only when the word filter ran (guard_* options).
|
||||
Guard *struct {
|
||||
DroppedWords int `json:"dropped_words"`
|
||||
} `json:"guard"`
|
||||
}
|
||||
|
||||
// streamFeedJSON mirrors the document returned by
|
||||
@@ -258,6 +272,9 @@ type ParakeetCpp struct {
|
||||
spkCtx uintptr
|
||||
speakerAccept float32
|
||||
speakerMargin float32
|
||||
// speakerStrict (speaker_strict:true) refuses registered voices that carry no
|
||||
// encoder fingerprint instead of using them unverified.
|
||||
speakerStrict bool
|
||||
// diarLatency is the PARAKEET_DIAR_LATENCY_* mode for diarization
|
||||
// streaming (diarization_latency: option, default "low"). Unused until
|
||||
// the diarization/scene streaming paths land.
|
||||
@@ -285,9 +302,13 @@ type ParakeetCpp struct {
|
||||
// RPC then falls back to the ASR context's own VAD head.
|
||||
vadCtx uintptr
|
||||
// vadOptions is the JSON object built from the vad_threshold, vad_min_pause,
|
||||
// vad_min_speech, vad_speech_pad and vad_max_segment model options ("" when
|
||||
// vad_min_speech, vad_speech_pad, vad_max_segment and vad_trim model options ("" when
|
||||
// none is set, so the library picks the defaults of the detector in use).
|
||||
vadOptions string
|
||||
// guardOptions is the JSON options object of the word filter (guard_*
|
||||
// model options), "" when it is off. Offline transcription then takes the
|
||||
// file-path route like vad:true, because the batched entry point has no filter.
|
||||
guardOptions string
|
||||
}
|
||||
|
||||
// Load is the LocalAI gRPC entry point for LoadModel: it calls
|
||||
@@ -310,6 +331,14 @@ func (p *ParakeetCpp) Load(opts *pb.ModelOptions) error {
|
||||
return err
|
||||
}
|
||||
p.vadOptions = vadOpts
|
||||
guardOpts, err := parseGuardOptions(opts)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
p.guardOptions = guardOpts
|
||||
if guardOpts != "" && p.vad && CppTranscribePathJSONVadWith == nil {
|
||||
return errors.New("parakeet-cpp: the guard_* options with vad:true need a libparakeet.so with parakeet_capi_transcribe_path_json_vad_with; rebuild the backend against a newer parakeet.cpp")
|
||||
}
|
||||
if optString(opts, "vad_model") != "" || optString(opts, "vad_component") != "" {
|
||||
if CppTranscribePathJSONVadWith == nil {
|
||||
return errors.New("parakeet-cpp: vad_model and vad_component need a libparakeet.so with parakeet_capi_transcribe_path_json_vad_with; rebuild the backend against a newer parakeet.cpp")
|
||||
@@ -505,7 +534,7 @@ func (p *ParakeetCpp) AudioTranscription(ctx context.Context, opts *pb.Transcrip
|
||||
// With vad:true the same file-path route is taken through the
|
||||
// VAD-segmented entry point, which cuts long audio at pauses. The batcher
|
||||
// has no VAD variant, so this path replaces it for offline requests.
|
||||
if p.bat == nil || p.vad {
|
||||
if p.bat == nil || p.vad || p.guardOptions != "" {
|
||||
converted, cleanup, err := convertToWavMono16k(opts.Dst)
|
||||
if err != nil {
|
||||
return pb.TranscriptResult{}, err
|
||||
@@ -515,7 +544,7 @@ func (p *ParakeetCpp) AudioTranscription(ctx context.Context, opts *pb.Transcrip
|
||||
if err != nil {
|
||||
return pb.TranscriptResult{}, err
|
||||
}
|
||||
if p.vad && p.wantSpeakers(opts.GetDiarize()) && len(doc.Words) > 0 {
|
||||
if (p.vad || p.guardOptions != "") && p.wantSpeakers(opts.GetDiarize()) && len(doc.Words) > 0 {
|
||||
pcm, _, err := decodeWavMono16k(converted)
|
||||
if err != nil {
|
||||
return pb.TranscriptResult{}, err
|
||||
@@ -575,13 +604,20 @@ func (p *ParakeetCpp) AudioTranscription(ctx context.Context, opts *pb.Transcrip
|
||||
func (p *ParakeetCpp) transcribePathDoc(path string) (transcriptJSON, error) {
|
||||
call, name := func() uintptr { return CppTranscribePathJSON(p.ctxPtr, path, 0) }, "transcribe_path_json"
|
||||
switch {
|
||||
case p.vad && (p.vadCtx != 0 || p.vadOptions != "") && CppTranscribePathJSONVadWith != nil:
|
||||
// An external Silero, or tuned segmenter options on the model's own head.
|
||||
case p.vad && (p.vadCtx != 0 || p.vadOptions != "" || p.guardOptions != "") && CppTranscribePathJSONVadWith != nil:
|
||||
// An external Silero, tuned segmenter options on the model's own head, or the
|
||||
// word filter: the segmenter takes the VAD keys and the filter keys in one object.
|
||||
opts, err := mergeJSONObjects(p.vadOptions, p.guardOptions)
|
||||
if err != nil {
|
||||
return transcriptJSON{}, fmt.Errorf("parakeet-cpp: build vad options: %w", err)
|
||||
}
|
||||
call, name = func() uintptr {
|
||||
return CppTranscribePathJSONVadWith(p.ctxPtr, p.vadCtx, path, 0, p.vadOptions)
|
||||
return CppTranscribePathJSONVadWith(p.ctxPtr, p.vadCtx, path, 0, opts)
|
||||
}, "transcribe_path_json_vad_with"
|
||||
case p.vad:
|
||||
call, name = func() uintptr { return CppTranscribePathJSONVad(p.ctxPtr, path, 0) }, "transcribe_path_json_vad"
|
||||
case p.guardOptions != "" && CppTranscribePathJSONWith != nil:
|
||||
call, name = func() uintptr { return CppTranscribePathJSONWith(p.ctxPtr, path, 0, p.guardOptions) }, "transcribe_path_json_with"
|
||||
}
|
||||
p.engineMu.Lock()
|
||||
cstr := call()
|
||||
@@ -599,6 +635,10 @@ func (p *ParakeetCpp) transcribePathDoc(path string) (transcriptJSON, error) {
|
||||
if err := json.Unmarshal([]byte(raw), &doc); err != nil {
|
||||
return transcriptJSON{}, fmt.Errorf("parakeet-cpp: decode transcript json: %w", err)
|
||||
}
|
||||
if doc.Guard != nil {
|
||||
// TranscriptResult has no field for it, so the count is only logged.
|
||||
xlog.Debug("parakeet-cpp: word filter", "dropped_words", doc.Guard.DroppedWords)
|
||||
}
|
||||
return doc, nil
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
"strconv"
|
||||
|
||||
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
|
||||
)
|
||||
|
||||
// The word filter of libparakeet (the "guard"): an opt-in pass over a finished
|
||||
// decode that drops words which stand alone or sit among low-confidence words,
|
||||
// as noise tends to give, and words that are only punctuation. It is off unless
|
||||
// one of these model options is set:
|
||||
//
|
||||
// guard_min_local_conf:0.5 0 to 1; 0 = off
|
||||
// guard_local_radius:5 seconds > 0; the library default is 5
|
||||
// guard_drop_punct_only:true
|
||||
//
|
||||
// The names carry the guard_ prefix because the bare library names
|
||||
// (min_local_conf, local_radius) say nothing next to vad_*, speaker_* and the
|
||||
// other options of this backend, and the library's result calls the section
|
||||
// "guard".
|
||||
|
||||
// parseGuardOptions reads the guard_* model options and returns the JSON
|
||||
// options object of parakeet_capi_transcribe_path_json_with (the keys the
|
||||
// library names min_local_conf, local_radius and drop_punct_only), or "" when
|
||||
// none is set. A value that does not parse or is out of range fails the load,
|
||||
// and so does the filter on a libparakeet.so that cannot run it.
|
||||
func parseGuardOptions(opts *pb.ModelOptions) (string, error) {
|
||||
obj := map[string]any{}
|
||||
if raw := optString(opts, "guard_min_local_conf"); raw != "" {
|
||||
v, err := strconv.ParseFloat(raw, 64)
|
||||
if err != nil || math.IsNaN(v) || math.IsInf(v, 0) {
|
||||
return "", fmt.Errorf("parakeet-cpp: option guard_min_local_conf: %q is not a number", raw)
|
||||
}
|
||||
if v < 0 || v > 1 {
|
||||
return "", fmt.Errorf("parakeet-cpp: option guard_min_local_conf: %v is out of range, want 0 to 1", v)
|
||||
}
|
||||
obj["min_local_conf"] = v
|
||||
}
|
||||
if raw := optString(opts, "guard_local_radius"); raw != "" {
|
||||
v, err := strconv.ParseFloat(raw, 64)
|
||||
if err != nil || math.IsNaN(v) || math.IsInf(v, 0) {
|
||||
return "", fmt.Errorf("parakeet-cpp: option guard_local_radius: %q is not a number", raw)
|
||||
}
|
||||
if v <= 0 {
|
||||
return "", fmt.Errorf("parakeet-cpp: option guard_local_radius: %v is out of range, want seconds > 0", v)
|
||||
}
|
||||
obj["local_radius"] = v
|
||||
}
|
||||
if optString(opts, "guard_drop_punct_only") != "" {
|
||||
b, err := optBool(opts, "guard_drop_punct_only", false)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
obj["drop_punct_only"] = b
|
||||
}
|
||||
if len(obj) == 0 {
|
||||
return "", nil
|
||||
}
|
||||
if CppTranscribePathJSONWith == nil {
|
||||
return "", errors.New("parakeet-cpp: the guard_* options need a libparakeet.so with parakeet_capi_transcribe_path_json_with; rebuild the backend against a newer parakeet.cpp")
|
||||
}
|
||||
b, err := json.Marshal(obj)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return string(b), nil
|
||||
}
|
||||
|
||||
// mergeJSONObjects joins two flat JSON objects ("" is an empty one). The
|
||||
// segmenter entry point takes the VAD keys and the guard keys in one object.
|
||||
func mergeJSONObjects(a, b string) (string, error) {
|
||||
if a == "" {
|
||||
return b, nil
|
||||
}
|
||||
if b == "" {
|
||||
return a, nil
|
||||
}
|
||||
m := map[string]json.RawMessage{}
|
||||
for _, s := range []string{a, b} {
|
||||
if err := json.Unmarshal([]byte(s), &m); err != nil {
|
||||
return "", err
|
||||
}
|
||||
}
|
||||
out, err := json.Marshal(m)
|
||||
return string(out), err
|
||||
}
|
||||
@@ -0,0 +1,145 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"path/filepath"
|
||||
|
||||
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
var _ = Describe("guard_* model options (word filter)", func() {
|
||||
var (
|
||||
savedPlain func(uintptr, string, int32) uintptr
|
||||
savedWithOpts func(uintptr, string, int32, string) uintptr
|
||||
savedVadWith func(ctx, vadCtx uintptr, p string, d int32, o string) uintptr
|
||||
savedFree func(uintptr)
|
||||
pool *diarizeCstrPool
|
||||
gotOpts string
|
||||
usedWith, usedVadWith bool
|
||||
)
|
||||
opts := func(o ...string) *pb.ModelOptions { return &pb.ModelOptions{Options: o} }
|
||||
|
||||
BeforeEach(func() {
|
||||
savedPlain, savedWithOpts, savedVadWith, savedFree = CppTranscribePathJSON, CppTranscribePathJSONWith, CppTranscribePathJSONVadWith, CppFreeString
|
||||
pool = &diarizeCstrPool{}
|
||||
usedWith, usedVadWith, gotOpts = false, false, ""
|
||||
CppFreeString = func(uintptr) {}
|
||||
CppTranscribePathJSON = func(uintptr, string, int32) uintptr {
|
||||
Fail("the plain entry point must not be used when the filter is on")
|
||||
return 0
|
||||
}
|
||||
CppTranscribePathJSONWith = func(_ uintptr, _ string, _ int32, o string) uintptr {
|
||||
usedWith, gotOpts = true, o
|
||||
return pool.cstr(`{"text":"hello.","frame_sec":0.08,"words":[{"w":"hello.","start":0.1,"end":0.4,"conf":0.9}],"tokens":[],"guard":{"dropped_words":2}}`)
|
||||
}
|
||||
CppTranscribePathJSONVadWith = func(_, _ uintptr, _ string, _ int32, o string) uintptr {
|
||||
usedVadWith, gotOpts = true, o
|
||||
return pool.cstr(`{"text":"hello.","frame_sec":0.08,"words":[],"tokens":[],"guard":{"dropped_words":0}}`)
|
||||
}
|
||||
})
|
||||
AfterEach(func() {
|
||||
CppTranscribePathJSON, CppTranscribePathJSONWith, CppTranscribePathJSONVadWith, CppFreeString = savedPlain, savedWithOpts, savedVadWith, savedFree
|
||||
})
|
||||
|
||||
It("is off when no option is set, and then needs no library symbol", func() {
|
||||
CppTranscribePathJSONWith = nil
|
||||
s, err := parseGuardOptions(opts("vad:true"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(s).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("maps the options to the library keys", func() {
|
||||
s, err := parseGuardOptions(opts("guard_min_local_conf:0.5", "guard_local_radius:3", "guard_drop_punct_only:true"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(s).To(MatchJSON(`{"min_local_conf":0.5,"local_radius":3,"drop_punct_only":true}`))
|
||||
})
|
||||
|
||||
It("keeps an explicit 0 (off) and false as given", func() {
|
||||
s, err := parseGuardOptions(opts("guard_min_local_conf:0", "guard_drop_punct_only:false"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(s).To(MatchJSON(`{"min_local_conf":0,"drop_punct_only":false}`))
|
||||
})
|
||||
|
||||
DescribeTable("rejects bad values at load",
|
||||
func(opt, msg string) {
|
||||
_, err := parseGuardOptions(opts(opt))
|
||||
Expect(err).To(MatchError(ContainSubstring(msg)))
|
||||
},
|
||||
Entry("confidence not a number", "guard_min_local_conf:high", "is not a number"),
|
||||
Entry("confidence above one", "guard_min_local_conf:1.5", "out of range"),
|
||||
Entry("negative confidence", "guard_min_local_conf:-0.1", "out of range"),
|
||||
Entry("radius zero", "guard_local_radius:0", "out of range"),
|
||||
Entry("radius NaN", "guard_local_radius:NaN", "is not a number"),
|
||||
Entry("bool typo", "guard_drop_punct_only:ture", "is not a boolean"),
|
||||
)
|
||||
|
||||
It("fails the load, naming the symbol, on a library without the filter", func() {
|
||||
CppTranscribePathJSONWith = nil
|
||||
_, err := parseGuardOptions(opts("guard_min_local_conf:0.5"))
|
||||
Expect(err).To(MatchError(ContainSubstring("parakeet_capi_transcribe_path_json_with")))
|
||||
})
|
||||
|
||||
It("fails Load for guard_* with vad:true when the segmenter entry point is missing", func() {
|
||||
CppTranscribePathJSONVadWith = nil
|
||||
f := newFakeLib().withModel("asr.gguf", modelKindASR)
|
||||
restore := f.install()
|
||||
defer restore()
|
||||
savedVad := CppTranscribePathJSONVad
|
||||
CppTranscribePathJSONVad = func(uintptr, string, int32) uintptr { return 0 }
|
||||
defer func() { CppTranscribePathJSONVad = savedVad }()
|
||||
err := (&ParakeetCpp{}).Load(&pb.ModelOptions{ModelFile: "asr.gguf", Options: []string{"vad:true", "guard_min_local_conf:0.5"}})
|
||||
Expect(err).To(MatchError(ContainSubstring("parakeet_capi_transcribe_path_json_vad_with")))
|
||||
})
|
||||
|
||||
It("fails Load on a bad value before any model is loaded", func() {
|
||||
f := newFakeLib().withModel("asr.gguf", modelKindASR)
|
||||
restore := f.install()
|
||||
defer restore()
|
||||
err := (&ParakeetCpp{}).Load(&pb.ModelOptions{ModelFile: "asr.gguf", Options: []string{"guard_min_local_conf:2"}})
|
||||
Expect(err).To(MatchError(ContainSubstring("guard_min_local_conf")))
|
||||
Expect(f.loadedPaths).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("routes offline transcription through the filter entry point, past the batcher", func() {
|
||||
wav := filepath.Join(GinkgoT().TempDir(), "a.wav")
|
||||
writeMono16kWav(wav, 16000)
|
||||
// A non-nil batcher: without the filter the request would go to it.
|
||||
p := &ParakeetCpp{ctxPtr: 7, bat: &batcher{}, guardOptions: `{"min_local_conf":0.5}`}
|
||||
res, err := p.AudioTranscription(context.Background(), &pb.TranscriptRequest{Dst: wav})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(usedWith).To(BeTrue())
|
||||
Expect(gotOpts).To(Equal(`{"min_local_conf":0.5}`))
|
||||
Expect(res.Text).To(Equal("hello."))
|
||||
})
|
||||
|
||||
It("sends the filter keys together with the VAD keys to the segmenter when vad is on", func() {
|
||||
p := &ParakeetCpp{ctxPtr: 7, vad: true, vadOptions: `{"trim":0}`, guardOptions: `{"min_local_conf":0.5}`}
|
||||
_, err := p.transcribePathDoc("/x/long.wav")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(usedVadWith).To(BeTrue())
|
||||
Expect(usedWith).To(BeFalse())
|
||||
Expect(gotOpts).To(MatchJSON(`{"trim":0,"min_local_conf":0.5}`))
|
||||
})
|
||||
|
||||
It("does not reach the filter entry point when the filter is off", func() {
|
||||
CppTranscribePathJSON = func(uintptr, string, int32) uintptr {
|
||||
return pool.cstr(`{"text":"plain.","frame_sec":0.08,"words":[],"tokens":[]}`)
|
||||
}
|
||||
p := &ParakeetCpp{ctxPtr: 7}
|
||||
doc, err := p.transcribePathDoc("/x/a.wav")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(doc.Text).To(Equal("plain."))
|
||||
Expect(usedWith).To(BeFalse())
|
||||
Expect(doc.Guard).To(BeNil())
|
||||
})
|
||||
|
||||
It("reads the dropped word count of the document", func() {
|
||||
p := &ParakeetCpp{ctxPtr: 7, guardOptions: `{"drop_punct_only":true}`}
|
||||
doc, err := p.transcribePathDoc("/x/a.wav")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(doc.Guard).ToNot(BeNil())
|
||||
Expect(doc.Guard.DroppedWords).To(Equal(2))
|
||||
})
|
||||
})
|
||||
@@ -163,6 +163,17 @@ func main() {
|
||||
purego.RegisterLibFunc(&CppDiarizeNamedPCMJSON, lib, "parakeet_capi_diarize_named_pcm_json")
|
||||
}
|
||||
|
||||
// Encoder fingerprint of the speaker registry (additive, no ABI bump).
|
||||
if sym, err := purego.Dlsym(lib, "parakeet_capi_speaker_registry_add_embedding_fp"); err == nil && sym != 0 {
|
||||
purego.RegisterLibFunc(&CppSpeakerRegistryAddEmbeddingFP, lib, "parakeet_capi_speaker_registry_add_embedding_fp")
|
||||
purego.RegisterLibFunc(&CppSpeakerRegistrySetStrict, lib, "parakeet_capi_speaker_registry_set_strict")
|
||||
purego.RegisterLibFunc(&CppSpeakerEncoderFamily, lib, "parakeet_capi_speaker_encoder_family")
|
||||
}
|
||||
// Word filter on transcription (additive, no ABI bump).
|
||||
if sym, err := purego.Dlsym(lib, "parakeet_capi_transcribe_path_json_with"); err == nil && sym != 0 {
|
||||
purego.RegisterLibFunc(&CppTranscribePathJSONWith, lib, "parakeet_capi_transcribe_path_json_with")
|
||||
}
|
||||
|
||||
for _, lf := range []LibFuncs{
|
||||
{&CppSpeakerIdentity, "parakeet_capi_speaker_identity"},
|
||||
{&CppSpeakerDim, "parakeet_capi_speaker_dim"},
|
||||
|
||||
@@ -17,6 +17,11 @@ func (p *ParakeetCpp) Status() (pb.StatusResponse, error) {
|
||||
dim := CppSpeakerDim(p.spkCtx)
|
||||
if identity != 0 && dim > 0 {
|
||||
result.SpeakerEncoder = &pb.SpeakerEncoder{Identity: goStringFromCPtr(identity), Dimension: dim}
|
||||
if CppSpeakerEncoderFamily != nil {
|
||||
if family := CppSpeakerEncoderFamily(p.spkCtx); family != 0 {
|
||||
result.SpeakerEncoder.Family = goStringFromCPtr(family)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return pb.StatusResponse{
|
||||
|
||||
@@ -171,6 +171,14 @@ func (p *ParakeetCpp) loadRoles(opts *pb.ModelOptions) error {
|
||||
return err
|
||||
}
|
||||
|
||||
strict, err := optBool(opts, "speaker_strict", false)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if strict && CppSpeakerRegistrySetStrict == nil {
|
||||
return errors.New("parakeet-cpp: speaker_strict needs a libparakeet.so with parakeet_capi_speaker_registry_set_strict; rebuild the backend against a newer parakeet.cpp")
|
||||
}
|
||||
|
||||
latency, err := parseDiarLatency(optString(opts, "diarization_latency"))
|
||||
if err != nil {
|
||||
return err
|
||||
@@ -311,6 +319,7 @@ func (p *ParakeetCpp) loadRoles(opts *pb.ModelOptions) error {
|
||||
return errors.New("parakeet-cpp: speaker_model needs a diarization model (the primary, diarization_model: or diar_component:)")
|
||||
}
|
||||
p.speakerAccept, p.speakerMargin = accept, margin
|
||||
p.speakerStrict = strict
|
||||
p.diarLatency = latency
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -120,6 +120,9 @@ func (p *ParakeetCpp) sceneBegin(voices []*pb.KnownVoice) sceneStreamHandle {
|
||||
opts.SpeakerMargin = p.speakerMargin
|
||||
s := CppSceneStreamBeginSpeaker(0, diar, tag, p.spkCtx, reg, &opts)
|
||||
if s == 0 {
|
||||
// The library also refuses a registry from another encoder (or without a
|
||||
// fingerprint under speaker_strict) here and says why on the speaker context.
|
||||
xlog.Warn("parakeet-cpp: could not start a live session with speaker names", "error", CppLastError(p.spkCtx))
|
||||
p.freeSpeakerRegistry(reg)
|
||||
return sceneStreamHandle{}
|
||||
}
|
||||
|
||||
@@ -123,6 +123,9 @@ var _ = Describe("scene stream with speaker names", func() {
|
||||
|
||||
It("frees the registry when the speaker begin fails, and degrades to no scene stream", func() {
|
||||
CppSceneStreamBeginSpeaker = func(asr, diar, tag, spk, reg uintptr, o *cSceneOpts) uintptr { return 0 }
|
||||
lastErr := CppLastError
|
||||
defer func() { CppLastError = lastErr }()
|
||||
CppLastError = func(uintptr) string { return "registry encoder family differs" }
|
||||
p := &ParakeetCpp{diarCtx: 1, spkCtx: 2}
|
||||
h := p.sceneBegin([]*pb.KnownVoice{{Name: "Ada", Embedding: []float32{1, 0}}})
|
||||
Expect(h.s).To(Equal(uintptr(0)))
|
||||
|
||||
@@ -0,0 +1,224 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
pb "github.com/mudler/LocalAI/pkg/grpc/proto"
|
||||
. "github.com/onsi/ginkgo/v2"
|
||||
. "github.com/onsi/gomega"
|
||||
)
|
||||
|
||||
type fpAdd struct{ key, family, weights string }
|
||||
|
||||
const (
|
||||
famECAPA = "voicedetect:ecapa_tdnn:speechbrain/spkrec-ecapa-voxceleb:192"
|
||||
famCAMPP = "voicedetect:campplus:3dspeaker/campplus:192"
|
||||
hashLoad = "sha256:aaaa"
|
||||
hashOther = "sha256:bbbb"
|
||||
)
|
||||
|
||||
// The fingerprint specs run against stubbed C entry points, like the registry
|
||||
// specs in speaker_registry_test.go.
|
||||
var _ = Describe("speaker registry encoder fingerprint", func() {
|
||||
var (
|
||||
restore func()
|
||||
pool *diarizeCstrPool
|
||||
plain []string
|
||||
fp []fpAdd
|
||||
strictSet []int32
|
||||
freed []uintptr
|
||||
fpFails bool
|
||||
)
|
||||
BeforeEach(func() {
|
||||
sNew, sFree, sAdd, sAddFP, sDim, sErr := CppSpeakerRegistryNew, CppSpeakerRegistryFree, CppSpeakerRegistryAddEmbedding, CppSpeakerRegistryAddEmbeddingFP, CppSpeakerDim, CppSpeakerRegistryLastError
|
||||
sStrict, sFam, sID := CppSpeakerRegistrySetStrict, CppSpeakerEncoderFamily, CppSpeakerIdentity
|
||||
restore = func() {
|
||||
CppSpeakerRegistryNew, CppSpeakerRegistryFree, CppSpeakerRegistryAddEmbedding, CppSpeakerRegistryAddEmbeddingFP, CppSpeakerDim, CppSpeakerRegistryLastError = sNew, sFree, sAdd, sAddFP, sDim, sErr
|
||||
CppSpeakerRegistrySetStrict, CppSpeakerEncoderFamily, CppSpeakerIdentity = sStrict, sFam, sID
|
||||
}
|
||||
pool = &diarizeCstrPool{}
|
||||
plain, fp, strictSet, freed, fpFails = nil, nil, nil, nil, false
|
||||
CppSpeakerDim = func(uintptr) int32 { return 2 }
|
||||
CppSpeakerRegistryNew = func() uintptr { return 77 }
|
||||
CppSpeakerRegistryFree = func(r uintptr) { freed = append(freed, r) }
|
||||
CppSpeakerRegistryLastError = func(uintptr) string { return "stub error" }
|
||||
CppSpeakerRegistryAddEmbedding = func(_ uintptr, name string, _ *float32, _ int32) int32 {
|
||||
plain = append(plain, name)
|
||||
return 0
|
||||
}
|
||||
CppSpeakerRegistryAddEmbeddingFP = func(_ uintptr, name string, _ *float32, _ int32, family, weights string) int32 {
|
||||
if fpFails {
|
||||
return 1
|
||||
}
|
||||
fp = append(fp, fpAdd{name, family, weights})
|
||||
return 0
|
||||
}
|
||||
CppSpeakerRegistrySetStrict = func(_ uintptr, v int32) { strictSet = append(strictSet, v) }
|
||||
CppSpeakerEncoderFamily = func(uintptr) uintptr { return pool.cstr(famECAPA) }
|
||||
CppSpeakerIdentity = func(uintptr) uintptr { return pool.cstr(hashLoad) }
|
||||
})
|
||||
AfterEach(func() { restore() })
|
||||
|
||||
v := func(id, family, weights string) *pb.KnownVoice {
|
||||
return &pb.KnownVoice{Id: id, Name: "n" + id, Embedding: []float32{1, 0}, EncoderFamily: family, EncoderWeights: weights}
|
||||
}
|
||||
|
||||
It("stores the family and weights per voice and passes them to the registry", func() {
|
||||
p := &ParakeetCpp{spkCtx: 5}
|
||||
reg, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad), v("b", famECAPA, hashOther)})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(reg).To(Equal(uintptr(77)))
|
||||
Expect(fp).To(Equal([]fpAdd{{"a", famECAPA, hashLoad}, {"b", famECAPA, hashOther}}))
|
||||
Expect(plain).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("takes the family of the loaded encoder for a voice with the same weights and no family", func() {
|
||||
p := &ParakeetCpp{spkCtx: 5}
|
||||
_, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", "", hashLoad)})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(fp).To(Equal([]fpAdd{{"a", famECAPA, hashLoad}}))
|
||||
})
|
||||
|
||||
It("does not guess the family from another weights hash", func() {
|
||||
p := &ParakeetCpp{spkCtx: 5}
|
||||
_, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", "", hashOther)})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(fp).To(BeEmpty())
|
||||
Expect(plain).To(Equal([]string{"a"}))
|
||||
})
|
||||
|
||||
It("keeps the old path for voices without a fingerprint", func() {
|
||||
p := &ParakeetCpp{spkCtx: 5}
|
||||
reg, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", "", "")})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(reg).To(Equal(uintptr(77)))
|
||||
Expect(plain).To(Equal([]string{"a"}))
|
||||
Expect(fp).To(BeEmpty())
|
||||
Expect(strictSet).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("puts fingerprinted and unfingerprinted voices into one registry without a fingerprint", func() {
|
||||
p := &ParakeetCpp{spkCtx: 5}
|
||||
_, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad), v("b", "", ""), v("c", famCAMPP, hashOther)})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
// The library refuses to mix them in one registry; the voice of another
|
||||
// family can never match and is left out.
|
||||
Expect(plain).To(ConsistOf("b", "a"))
|
||||
Expect(fp).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("leaves out voices of another family when a matching one exists", func() {
|
||||
p := &ParakeetCpp{spkCtx: 5}
|
||||
_, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad), v("c", famCAMPP, hashOther)})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(fp).To(Equal([]fpAdd{{"a", famECAPA, hashLoad}}))
|
||||
})
|
||||
|
||||
It("builds the registry from voices of another family when nothing else is usable, so the library refuses by name", func() {
|
||||
p := &ParakeetCpp{spkCtx: 5}
|
||||
reg, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("c", famCAMPP, hashOther)})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(reg).To(Equal(uintptr(77)))
|
||||
Expect(fp).To(Equal([]fpAdd{{"c", famCAMPP, hashOther}}))
|
||||
})
|
||||
|
||||
It("frees the registry when the library refuses every voice", func() {
|
||||
fpFails = true
|
||||
p := &ParakeetCpp{spkCtx: 5}
|
||||
reg, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad)})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(reg).To(BeZero())
|
||||
Expect(freed).To(Equal([]uintptr{77}))
|
||||
})
|
||||
|
||||
Describe("speaker_strict", func() {
|
||||
It("drops unfingerprinted voices when a matching one exists", func() {
|
||||
p := &ParakeetCpp{spkCtx: 5, speakerStrict: true}
|
||||
_, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad), v("b", "", "")})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(fp).To(Equal([]fpAdd{{"a", famECAPA, hashLoad}}))
|
||||
Expect(plain).To(BeEmpty())
|
||||
})
|
||||
|
||||
It("builds a strict registry from unfingerprinted voices alone, so the library rejects it", func() {
|
||||
p := &ParakeetCpp{spkCtx: 5, speakerStrict: true}
|
||||
reg, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("b", "", "")})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(reg).To(Equal(uintptr(77)))
|
||||
Expect(plain).To(Equal([]string{"b"}))
|
||||
Expect(strictSet).To(Equal([]int32{1}))
|
||||
})
|
||||
|
||||
It("is not set on a registry that has a fingerprint", func() {
|
||||
p := &ParakeetCpp{spkCtx: 5, speakerStrict: true}
|
||||
_, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad)})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(strictSet).To(BeEmpty())
|
||||
})
|
||||
})
|
||||
|
||||
It("treats every voice as unfingerprinted on a library without the fingerprint", func() {
|
||||
CppSpeakerRegistryAddEmbeddingFP, CppSpeakerEncoderFamily = nil, nil
|
||||
p := &ParakeetCpp{spkCtx: 5}
|
||||
_, err := p.buildSpeakerRegistry([]*pb.KnownVoice{v("a", famECAPA, hashLoad)})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(plain).To(Equal([]string{"a"}))
|
||||
})
|
||||
|
||||
It("reports the family of the encoder in Status next to its identity", func() {
|
||||
sDim, sID := CppSpeakerDim, CppSpeakerIdentity
|
||||
defer func() { CppSpeakerDim, CppSpeakerIdentity = sDim, sID }()
|
||||
p := &ParakeetCpp{spkCtx: 5}
|
||||
st, err := p.Status()
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(st.GetSpeakerEncoder().GetIdentity()).To(Equal(hashLoad))
|
||||
Expect(st.GetSpeakerEncoder().GetFamily()).To(Equal(famECAPA))
|
||||
Expect(st.GetSpeakerEncoder().GetDimension()).To(Equal(int32(2)))
|
||||
})
|
||||
|
||||
It("surfaces the library's family mismatch message from Diarize", func() {
|
||||
rs := diarizeStubs()
|
||||
defer rs()
|
||||
sNamed, sPCM := CppDiarizeNamedPCMJSON, CppDiarizePCM
|
||||
defer func() { CppDiarizeNamedPCMJSON, CppDiarizePCM = sNamed, sPCM }()
|
||||
CppDiarizePCM = func(uintptr, *float32, int32, int32) uintptr { return 0 }
|
||||
msg := fmt.Sprintf("registry encoder family %q differs from the speaker model's %q", famCAMPP, famECAPA)
|
||||
CppDiarizeNamedPCMJSON = func(_, _, _ uintptr, _ *float32, _, _ int32, _, _ float32) uintptr { return 0 }
|
||||
CppLastError = func(ctx uintptr) string {
|
||||
if ctx == 2 {
|
||||
return msg
|
||||
}
|
||||
return ""
|
||||
}
|
||||
CppSpeakerDim = func(uintptr) int32 { return 2 }
|
||||
CppSpeakerRegistryNew = func() uintptr { return 77 }
|
||||
CppSpeakerRegistryFree = func(uintptr) {}
|
||||
p := &ParakeetCpp{diarCtx: 1, spkCtx: 2}
|
||||
_, err := p.Diarize(&pb.DiarizeRequest{Dst: diarizeWav(5), KnownVoices: []*pb.KnownVoice{v("c", famCAMPP, hashOther)}})
|
||||
Expect(err).To(HaveOccurred())
|
||||
Expect(err.Error()).To(ContainSubstring(famCAMPP))
|
||||
Expect(err.Error()).To(ContainSubstring(famECAPA))
|
||||
Expect(fp).To(Equal([]fpAdd{{"c", famCAMPP, hashOther}}))
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("speaker_strict option", func() {
|
||||
It("fails Load on a library without the strict switch", func() {
|
||||
saved := CppSpeakerRegistrySetStrict
|
||||
CppSpeakerRegistrySetStrict = nil
|
||||
defer func() { CppSpeakerRegistrySetStrict = saved }()
|
||||
f := newFakeLib().withModel("diar.gguf", modelKindDiarization).withModel("spk.gguf", modelKindSpeaker)
|
||||
restore := f.install()
|
||||
defer restore()
|
||||
err := (&ParakeetCpp{}).Load(&pb.ModelOptions{ModelFile: "diar.gguf", Options: []string{"speaker_model:spk.gguf", "speaker_strict:true"}})
|
||||
Expect(err).To(HaveOccurred())
|
||||
})
|
||||
|
||||
It("rejects a value that is not a boolean", func() {
|
||||
f := newFakeLib().withModel("diar.gguf", modelKindDiarization)
|
||||
restore := f.install()
|
||||
defer restore()
|
||||
err := (&ParakeetCpp{}).Load(&pb.ModelOptions{ModelFile: "diar.gguf", Options: []string{"speaker_strict:maybe"}})
|
||||
Expect(err).To(MatchError(ContainSubstring("speaker_strict")))
|
||||
})
|
||||
})
|
||||
@@ -56,12 +56,61 @@ func parseSpeakerMargin(s string) (float32, error) {
|
||||
return nonZero(float32(v)), nil
|
||||
}
|
||||
|
||||
// encoderFingerprint is what the loaded speaker encoder reports about itself:
|
||||
// the embedding space ("voicedetect:<arch>:<name>:<dim>") and the "sha256:"
|
||||
// identity of its weights. Both are empty on a libparakeet.so without the
|
||||
// fingerprint, which turns the checks off.
|
||||
func (p *ParakeetCpp) encoderFingerprint() (family, weights string) {
|
||||
if p.spkCtx == 0 || CppSpeakerRegistryAddEmbeddingFP == nil || CppSpeakerEncoderFamily == nil {
|
||||
return "", ""
|
||||
}
|
||||
if ptr := CppSpeakerEncoderFamily(p.spkCtx); ptr != 0 {
|
||||
family = goStringFromCPtr(ptr)
|
||||
}
|
||||
if CppSpeakerIdentity != nil {
|
||||
if ptr := CppSpeakerIdentity(p.spkCtx); ptr != 0 {
|
||||
weights = goStringFromCPtr(ptr)
|
||||
}
|
||||
}
|
||||
return family, weights
|
||||
}
|
||||
|
||||
// voiceFamily is the encoder family of a registered voice, "" when unknown. A
|
||||
// voice that only has a weights identity gets the family of the loaded encoder
|
||||
// when the weights are the same file, because the same bytes are the same
|
||||
// embedding space; a different hash proves nothing (another quantization of the
|
||||
// encoder has the same family), so that voice stays unknown.
|
||||
func voiceFamily(v *pb.KnownVoice, ownFamily, ownWeights string) string {
|
||||
if f := v.GetEncoderFamily(); f != "" {
|
||||
return f
|
||||
}
|
||||
if w := v.GetEncoderWeights(); w != "" && w == ownWeights {
|
||||
return ownFamily
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// buildSpeakerRegistryLocked makes a parakeet_speaker_registry from the registered voices
|
||||
// of one request or stream. Caller holds engineMu. It returns 0 (and no error) when there is
|
||||
// nothing to build: no speaker model loaded or no usable voices. A voice whose embedding size
|
||||
// differs from the speaker model's, or that the C side refuses, is skipped with a warning
|
||||
// (without its name: the log is not for the caller who may not see voice names) so one bad
|
||||
// voice cannot fail every request. The caller frees a non-zero result with freeSpeakerRegistry.
|
||||
//
|
||||
// Encoder fingerprint. The library keeps one registry per encoder and checks it against the
|
||||
// speaker model before it names anyone, so the voices are sorted by what is known about the
|
||||
// encoder that made them:
|
||||
//
|
||||
// - matching: the family is the one of the loaded encoder;
|
||||
// - other: another family, which can never name a speaker here;
|
||||
// - unfingerprinted: no family (registered before it was recorded, or by an encoder that
|
||||
// cannot report one). The library refuses to mix these with fingerprinted voices in one
|
||||
// registry, so when there are any, the request keeps the old behaviour: they and the
|
||||
// matching voices go into one registry that has no fingerprint, and the encoder is
|
||||
// unverified (logged). With speaker_strict they are dropped instead.
|
||||
//
|
||||
// With no usable voice but voices of another family, the registry is built from those, so the
|
||||
// library refuses the request and its message names both families.
|
||||
func (p *ParakeetCpp) buildSpeakerRegistryLocked(voices []*pb.KnownVoice) (uintptr, error) {
|
||||
if p.spkCtx == 0 || CppSpeakerRegistryNew == nil || CppSpeakerRegistryAddEmbedding == nil || len(voices) == 0 {
|
||||
return 0, nil
|
||||
@@ -74,7 +123,10 @@ func (p *ParakeetCpp) buildSpeakerRegistryLocked(voices []*pb.KnownVoice) (uintp
|
||||
if reg == 0 {
|
||||
return 0, status.Error(codes.Internal, "parakeet-cpp: could not create a speaker registry")
|
||||
}
|
||||
added, skipped := 0, 0
|
||||
ownFamily, ownWeights := p.encoderFingerprint()
|
||||
skipped := 0
|
||||
var matching, other, plain []*pb.KnownVoice
|
||||
family := map[*pb.KnownVoice]string{}
|
||||
for _, v := range voices {
|
||||
emb := v.GetEmbedding()
|
||||
if v.GetName() == "" || len(emb) == 0 {
|
||||
@@ -87,7 +139,67 @@ func (p *ParakeetCpp) buildSpeakerRegistryLocked(voices []*pb.KnownVoice) (uintp
|
||||
skipped++
|
||||
continue
|
||||
}
|
||||
if rc := CppSpeakerRegistryAddEmbedding(reg, voiceKey(v), &emb[0], int32(len(emb))); rc != 0 {
|
||||
f := ""
|
||||
if ownFamily != "" {
|
||||
f = voiceFamily(v, ownFamily, ownWeights)
|
||||
}
|
||||
family[v] = f
|
||||
switch {
|
||||
case f == "":
|
||||
plain = append(plain, v)
|
||||
case f == ownFamily:
|
||||
matching = append(matching, v)
|
||||
default:
|
||||
other = append(other, v)
|
||||
}
|
||||
}
|
||||
|
||||
// use is the voices that go into the registry; fingerprinted says whether they carry one.
|
||||
var use []*pb.KnownVoice
|
||||
fingerprinted, strictPlain := false, false
|
||||
switch {
|
||||
case len(plain) > 0 && !p.speakerStrict:
|
||||
use = append(plain, matching...)
|
||||
if ownFamily != "" {
|
||||
xlog.Warn("parakeet-cpp: registered voices without an encoder fingerprint are used unverified; register them again to record the encoder",
|
||||
"unfingerprinted", len(plain))
|
||||
}
|
||||
skipped += len(other)
|
||||
case len(matching) > 0:
|
||||
use, fingerprinted = matching, true
|
||||
skipped += len(other) + len(plain)
|
||||
case len(other) > 0:
|
||||
// Nothing usable here: let the library refuse and say which families differ.
|
||||
use, fingerprinted = other, true
|
||||
skipped += len(plain)
|
||||
case len(plain) > 0:
|
||||
// speaker_strict: the library refuses a registry without a fingerprint.
|
||||
use, strictPlain = plain, true
|
||||
}
|
||||
if len(use) == 0 {
|
||||
if skipped > 0 {
|
||||
xlog.Warn("parakeet-cpp: no registered voice is usable with this speaker model; speakers stay unnamed", "skipped", skipped)
|
||||
}
|
||||
CppSpeakerRegistryFree(reg)
|
||||
return 0, nil
|
||||
}
|
||||
if p.speakerStrict && len(plain) > 0 && !strictPlain {
|
||||
xlog.Warn("parakeet-cpp: speaker_strict: skipped registered voices without an encoder fingerprint", "skipped", len(plain))
|
||||
}
|
||||
|
||||
if strictPlain && CppSpeakerRegistrySetStrict != nil {
|
||||
CppSpeakerRegistrySetStrict(reg, 1)
|
||||
}
|
||||
added := 0
|
||||
for _, v := range use {
|
||||
emb := v.GetEmbedding()
|
||||
var rc int32
|
||||
if fingerprinted {
|
||||
rc = CppSpeakerRegistryAddEmbeddingFP(reg, voiceKey(v), &emb[0], int32(len(emb)), family[v], v.GetEncoderWeights())
|
||||
} else {
|
||||
rc = CppSpeakerRegistryAddEmbedding(reg, voiceKey(v), &emb[0], int32(len(emb)))
|
||||
}
|
||||
if rc != 0 {
|
||||
xlog.Warn("parakeet-cpp: skipped a registered voice the speaker registry refused", "error", CppSpeakerRegistryLastError(reg))
|
||||
skipped++
|
||||
continue
|
||||
|
||||
@@ -30,10 +30,13 @@ var vadTuning = []struct {
|
||||
{"vad_min_speech", "min_speech", 0, math.Inf(1), false},
|
||||
{"vad_speech_pad", "speech_pad", 0, math.Inf(1), false},
|
||||
{"vad_max_segment", "max_segment", 0, math.Inf(1), true},
|
||||
// vad_trim shrinks each transcription segment to its speech plus this much.
|
||||
// Unset keeps the library default (0.3 s); 0 keeps the whole cuts.
|
||||
{"vad_trim", "trim", 0, math.Inf(1), false},
|
||||
}
|
||||
|
||||
// parseVADTuning reads the vad_threshold, vad_min_pause, vad_min_speech,
|
||||
// vad_speech_pad and vad_max_segment model options and returns them as the JSON
|
||||
// vad_speech_pad, vad_max_segment and vad_trim model options and returns them as the JSON
|
||||
// options object of the C-API, or "" when none is set. A value that does not
|
||||
// parse or is out of range fails the load.
|
||||
func parseVADTuning(opts *pb.ModelOptions) (string, error) {
|
||||
|
||||
@@ -169,6 +169,15 @@ var _ = Describe("VAD tuning options", func() {
|
||||
Expect(s).To(MatchJSON(`{"threshold":0.6,"min_pause":0.3,"min_speech":0.2,"speech_pad":0.05,"max_segment":20}`))
|
||||
})
|
||||
|
||||
It("maps vad_trim to the trim key and allows 0, which keeps the whole cuts", func() {
|
||||
s, err := parseVADTuning(opts("vad_trim:0.5"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(s).To(MatchJSON(`{"trim":0.5}`))
|
||||
s, err = parseVADTuning(opts("vad_trim:0"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(s).To(MatchJSON(`{"trim":0}`))
|
||||
})
|
||||
|
||||
It("allows a zero speech pad", func() {
|
||||
s, err := parseVADTuning(opts("vad_speech_pad:0"))
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
@@ -184,6 +193,8 @@ var _ = Describe("VAD tuning options", func() {
|
||||
Entry("threshold zero", "vad_threshold:0", "out of range"),
|
||||
Entry("threshold above one", "vad_threshold:1.5", "out of range"),
|
||||
Entry("negative pad", "vad_speech_pad:-1", "out of range"),
|
||||
Entry("negative trim", "vad_trim:-0.1", "out of range"),
|
||||
Entry("trim not a number", "vad_trim:long", "is not a number"),
|
||||
Entry("NaN", "vad_min_pause:NaN", "is not a number"),
|
||||
)
|
||||
})
|
||||
@@ -266,6 +277,14 @@ var _ = Describe("vad_model", func() {
|
||||
Expect(calledWith).To(BeFalse())
|
||||
})
|
||||
|
||||
It("passes vad_trim:0 to the segmenter, so the old whole cuts stay available", func() {
|
||||
p := &ParakeetCpp{ctxPtr: 7, vad: true, vadOptions: `{"trim":0}`}
|
||||
_, err := p.transcribePathDoc("/x/long.wav")
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(calledWith).To(BeTrue())
|
||||
Expect(gotOpts).To(Equal(`{"trim":0}`))
|
||||
})
|
||||
|
||||
It("passes tuning to the head through _with (null Silero context)", func() {
|
||||
p := &ParakeetCpp{ctxPtr: 7, vad: true, vadOptions: `{"max_segment":20}`}
|
||||
_, err := p.transcribePathDoc("/x/long.wav")
|
||||
|
||||
@@ -44,7 +44,7 @@ type DiarizationRequest struct {
|
||||
func (r *DiarizationRequest) toProto(threads uint32, modelIdentity string) *proto.DiarizeRequest {
|
||||
known := make([]*proto.KnownVoice, 0, len(r.KnownVoices))
|
||||
for _, v := range r.KnownVoices {
|
||||
known = append(known, &proto.KnownVoice{Id: v.ID, Name: v.Name, Embedding: v.Embedding, Model: v.Model})
|
||||
known = append(known, &proto.KnownVoice{Id: v.ID, Name: v.Name, Embedding: v.Embedding, Model: v.Model, EncoderFamily: v.Family, EncoderWeights: v.Weights})
|
||||
}
|
||||
return &proto.DiarizeRequest{
|
||||
ModelIdentity: modelIdentity,
|
||||
@@ -210,7 +210,7 @@ func speakerEncoderFromBackend(ctx context.Context, m grpcPkg.Backend) (schema.S
|
||||
return schema.SpeakerEncoder{}, err
|
||||
}
|
||||
e := r.GetSpeakerEncoder()
|
||||
trusted := schema.SpeakerEncoder{Identity: e.GetIdentity(), Dimension: int(e.GetDimension())}
|
||||
trusted := schema.SpeakerEncoder{Identity: e.GetIdentity(), Dimension: int(e.GetDimension()), Family: e.GetFamily()}
|
||||
if err := (schema.SpeakerProfiles{Version: 1, Encoder: trusted}).Validate(trusted); err != nil {
|
||||
return schema.SpeakerEncoder{}, status.Error(codes.Unimplemented, "backend does not expose trusted speaker encoder metadata")
|
||||
}
|
||||
@@ -232,7 +232,10 @@ func decodeSpeakerProfiles(raw string, trusted schema.SpeakerEncoder) (*schema.S
|
||||
|
||||
// Portable registrations require exact loaded identity and dimension. Legacy
|
||||
// candidates use the trusted dimension when available; older backends without
|
||||
// metadata retain their native dimension check. No registry entry sets it.
|
||||
// metadata retain their native dimension check. A portable voice with other
|
||||
// weights is dropped here unless it carries an encoder family: the backend
|
||||
// compares families, accepts another quantization of the same encoder with a
|
||||
// warning, and refuses another encoder by name.
|
||||
func compatiblePortableVoices(ctx context.Context, m grpcPkg.Backend, voices []voicerecognition.KnownVoice) []voicerecognition.KnownVoice {
|
||||
if len(voices) == 0 {
|
||||
return voices
|
||||
@@ -243,7 +246,8 @@ func compatiblePortableVoices(ctx context.Context, m grpcPkg.Backend, voices []v
|
||||
if err == nil && len(v.Embedding) != trusted.Dimension {
|
||||
continue
|
||||
}
|
||||
if strings.HasPrefix(v.Model, "sha256:") && (err != nil || v.Model != trusted.Identity || len(v.Embedding) != trusted.Dimension) {
|
||||
if strings.HasPrefix(v.Model, "sha256:") && (err != nil || len(v.Embedding) != trusted.Dimension ||
|
||||
(v.Model != trusted.Identity && v.Family == "")) {
|
||||
continue
|
||||
}
|
||||
out = append(out, v)
|
||||
|
||||
@@ -78,10 +78,47 @@ var _ = Describe("portable voice compatibility", func() {
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("portable voice encoder family", func() {
|
||||
const fam, other = "voicedetect:ecapa_tdnn:ecapa:192", "voicedetect:campplus:campplus:192"
|
||||
It("keeps a voice with another weights hash when it has a family, for the backend to judge", func() {
|
||||
identity := "sha256:" + strings.Repeat("a", 64)
|
||||
otherHash := "sha256:" + strings.Repeat("b", 64)
|
||||
m := &portableStatusBackend{identity: identity}
|
||||
voices := []voicerecognition.KnownVoice{
|
||||
{ID: "same-family", Model: otherHash, Weights: otherHash, Family: fam, Embedding: []float32{1, 0}},
|
||||
{ID: "other-family", Model: otherHash, Weights: otherHash, Family: other, Embedding: []float32{1, 0}},
|
||||
{ID: "no-family", Model: otherHash, Weights: otherHash, Embedding: []float32{1, 0}},
|
||||
{ID: "wrong-size", Model: otherHash, Weights: otherHash, Family: fam, Embedding: []float32{1, 0, 0}},
|
||||
}
|
||||
got := compatiblePortableVoices(context.Background(), m, voices)
|
||||
Expect(got).To(HaveLen(2))
|
||||
Expect(got[0].ID).To(Equal("same-family"))
|
||||
Expect(got[1].ID).To(Equal("other-family"))
|
||||
})
|
||||
It("sends the fingerprint to the backend, offline and live", func() {
|
||||
v := voicerecognition.KnownVoice{ID: "a", Name: "Ada", Embedding: []float32{1, 0}, Model: "sha256:x", Family: "f", Weights: "sha256:x"}
|
||||
offline := (&DiarizationRequest{KnownVoices: []voicerecognition.KnownVoice{v}}).toProto(2, "model")
|
||||
Expect(offline.KnownVoices[0].EncoderFamily).To(Equal("f"))
|
||||
Expect(offline.KnownVoices[0].EncoderWeights).To(Equal("sha256:x"))
|
||||
var live liveOptions
|
||||
WithKnownVoices([]voicerecognition.KnownVoice{v})(&live)
|
||||
cfg := liveConfigProto("en", live)
|
||||
Expect(cfg.KnownVoices[0].EncoderFamily).To(Equal("f"))
|
||||
Expect(cfg.KnownVoices[0].EncoderWeights).To(Equal("sha256:x"))
|
||||
})
|
||||
It("reads the family of the loaded encoder from the backend status", func() {
|
||||
m := &portableStatusBackend{identity: "sha256:" + strings.Repeat("a", 64), family: fam}
|
||||
trusted, err := speakerEncoderFromBackend(context.Background(), m)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
Expect(trusted.Family).To(Equal(fam))
|
||||
})
|
||||
})
|
||||
|
||||
type portableStatusBackend struct {
|
||||
grpcPkg.Backend
|
||||
identity string
|
||||
dimension int32
|
||||
family string
|
||||
}
|
||||
|
||||
func (m *portableStatusBackend) Status(context.Context) (*pb.StatusResponse, error) {
|
||||
@@ -89,7 +126,7 @@ func (m *portableStatusBackend) Status(context.Context) (*pb.StatusResponse, err
|
||||
if dim == 0 {
|
||||
dim = 2
|
||||
}
|
||||
return &pb.StatusResponse{SpeakerEncoder: &pb.SpeakerEncoder{Identity: m.identity, Dimension: dim}}, nil
|
||||
return &pb.StatusResponse{SpeakerEncoder: &pb.SpeakerEncoder{Identity: m.identity, Dimension: dim, Family: m.family}}, nil
|
||||
}
|
||||
|
||||
var _ = Describe("selection before portable compatibility", func() {
|
||||
|
||||
@@ -243,7 +243,7 @@ func WithKnownVoices(v []voicerecognition.KnownVoice) LiveOption {
|
||||
func liveConfigProto(language string, o liveOptions) *proto.TranscriptLiveConfig {
|
||||
cfg := &proto.TranscriptLiveConfig{Language: language, SampleRate: liveSampleRate}
|
||||
for _, v := range o.knownVoices {
|
||||
cfg.KnownVoices = append(cfg.KnownVoices, &proto.KnownVoice{Id: v.ID, Name: v.Name, Embedding: v.Embedding, Model: v.Model})
|
||||
cfg.KnownVoices = append(cfg.KnownVoices, &proto.KnownVoice{Id: v.ID, Name: v.Name, Embedding: v.Embedding, Model: v.Model, EncoderFamily: v.Family, EncoderWeights: v.Weights})
|
||||
}
|
||||
return cfg
|
||||
}
|
||||
|
||||
@@ -34,7 +34,7 @@ func VoiceRegisterEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader,
|
||||
}
|
||||
|
||||
var embedding []float32
|
||||
var encoder string
|
||||
var encoder, family string
|
||||
if input.SpeakerProfiles != nil {
|
||||
if input.Audio != "" || input.SpeakerSlot == nil {
|
||||
return echo.NewHTTPError(http.StatusBadRequest, "speaker_profiles requires speaker_slot and excludes audio")
|
||||
@@ -47,7 +47,7 @@ func VoiceRegisterEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader,
|
||||
if err != nil {
|
||||
return echo.NewHTTPError(http.StatusBadRequest, err.Error())
|
||||
}
|
||||
embedding, encoder = selected.Embedding, trusted.Identity
|
||||
embedding, encoder, family = selected.Embedding, trusted.Identity, trusted.Family
|
||||
} else {
|
||||
if input.SpeakerSlot != nil {
|
||||
return echo.NewHTTPError(http.StatusBadRequest, "speaker_slot requires speaker_profiles")
|
||||
@@ -63,7 +63,11 @@ func VoiceRegisterEndpoint(cl *config.ModelConfigLoader, ml *model.ModelLoader,
|
||||
}
|
||||
embedding, encoder = res.GetEmbedding(), res.GetModel()
|
||||
}
|
||||
stored, err := registry.Register(c.Request().Context(), embedding, voiceMetadata(input.Name, input.Labels, encoder))
|
||||
meta := voiceMetadata(input.Name, input.Labels, encoder)
|
||||
// Only the portable route knows the family: it comes from the loaded
|
||||
// encoder. A voice-detect embedding has none, so it stays unfingerprinted.
|
||||
meta.EncoderFamily = family
|
||||
stored, err := registry.Register(c.Request().Context(), embedding, meta)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -13,6 +13,9 @@ import (
|
||||
type SpeakerEncoder struct {
|
||||
Identity string `json:"identity"`
|
||||
Dimension int `json:"dimension"`
|
||||
// Family is the embedding space of the encoder. The server fills it from the
|
||||
// loaded encoder; exported profiles do not carry it and it is not matched.
|
||||
Family string `json:"family,omitempty"`
|
||||
}
|
||||
|
||||
// SpeakerProfileInterval locates retained clean audio in the original recording,
|
||||
@@ -51,7 +54,7 @@ func (p SpeakerProfiles) Validate(trusted SpeakerEncoder) error {
|
||||
if !speakerEncoderIdentity.MatchString(trusted.Identity) || trusted.Dimension <= 0 {
|
||||
return fmt.Errorf("invalid trusted speaker encoder metadata")
|
||||
}
|
||||
if p.Encoder != trusted {
|
||||
if p.Encoder.Identity != trusted.Identity || p.Encoder.Dimension != trusted.Dimension {
|
||||
return fmt.Errorf("speaker profile encoder does not match loaded encoder")
|
||||
}
|
||||
seen := make(map[int]bool, len(p.Speakers))
|
||||
|
||||
@@ -31,6 +31,12 @@ var _ = Describe("Portable speaker profiles", func() {
|
||||
p.Encoder.Dimension = 3
|
||||
Expect(p.Validate(trusted)).To(HaveOccurred())
|
||||
})
|
||||
It("matches on identity and dimension, not on the family only the server knows", func() {
|
||||
trusted.Family = "voicedetect:ecapa_tdnn:ecapa:192"
|
||||
Expect(p.Validate(trusted)).To(Succeed())
|
||||
_, err := p.Select(3, trusted)
|
||||
Expect(err).NotTo(HaveOccurred())
|
||||
})
|
||||
It("round trips the backend JSON including unavailable profiles without embeddings", func() {
|
||||
raw := `{"version":1,"encoder":{"identity":"` + trusted.Identity + `","dimension":2},"speakers":[{"speaker":3,"clean_duration":3,"intervals":[{"start":0,"end":3}],"unavailable_reason":null,"embedding":[0.6,0.8]},{"speaker":4,"clean_duration":1,"intervals":[{"start":4,"end":5}],"unavailable_reason":"insufficient_clean_speech"}]}`
|
||||
Expect(json.Unmarshal([]byte(raw), &p)).To(Succeed())
|
||||
|
||||
@@ -15,6 +15,11 @@ type KnownVoice struct {
|
||||
Name string
|
||||
Embedding []float32
|
||||
Model string
|
||||
// Family and Weights fingerprint the encoder that made the embedding, as
|
||||
// far as it is known: the embedding space and the "sha256:" identity of the
|
||||
// exact weights. Empty for a voice registered without them.
|
||||
Family string
|
||||
Weights string
|
||||
}
|
||||
|
||||
// SpeakerModelFromOptions returns the value of a speaker_model:<file> entry in
|
||||
@@ -62,7 +67,7 @@ func SelectKnownVoices(entries []Entry, speakerModelPath string) KnownVoiceSelec
|
||||
// Hash-tagged portable registrations are checked against the loaded
|
||||
// encoder by the backend, never against a filename or dimension alone.
|
||||
case strings.HasPrefix(e.Metadata.Model, "sha256:"), EncoderTag(e.Metadata.Model) == tag:
|
||||
sel.Voices = append(sel.Voices, KnownVoice{ID: e.Metadata.ID, Name: e.Metadata.Name, Embedding: e.Embedding, Model: e.Metadata.Model})
|
||||
sel.Voices = append(sel.Voices, knownVoice(e))
|
||||
default:
|
||||
sel.OtherEncoder++
|
||||
}
|
||||
@@ -76,6 +81,14 @@ func SelectKnownVoices(entries []Entry, speakerModelPath string) KnownVoiceSelec
|
||||
return sel
|
||||
}
|
||||
|
||||
func knownVoice(e Entry) KnownVoice {
|
||||
v := KnownVoice{ID: e.Metadata.ID, Name: e.Metadata.Name, Embedding: e.Embedding, Model: e.Metadata.Model, Family: e.Metadata.EncoderFamily}
|
||||
if strings.HasPrefix(e.Metadata.Model, "sha256:") {
|
||||
v.Weights = e.Metadata.Model
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
// KnownVoicesFor lists the registry and selects the voices for a speaker model.
|
||||
func KnownVoicesFor(ctx context.Context, reg Registry, speakerModelPath string) (KnownVoiceSelection, error) {
|
||||
entries, err := reg.List(ctx)
|
||||
|
||||
@@ -2,6 +2,7 @@ package voicerecognition_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
|
||||
"github.com/mudler/LocalAI/core/services/voicerecognition"
|
||||
@@ -117,6 +118,34 @@ func (r listRegistry) List(context.Context) ([]voicerecognition.Entry, error) {
|
||||
return r.entries, r.err
|
||||
}
|
||||
|
||||
var _ = Describe("encoder fingerprint of selected voices", func() {
|
||||
const hash = "sha256:aaaa"
|
||||
It("passes the family and takes the weights from a hash tag", func() {
|
||||
e := entry("ada", hash, 1, 0)
|
||||
e.Metadata.EncoderFamily = "voicedetect:ecapa_tdnn:ecapa:192"
|
||||
sel := voicerecognition.SelectKnownVoices([]voicerecognition.Entry{e}, "spk.gguf")
|
||||
Expect(sel.Voices).To(HaveLen(1))
|
||||
Expect(sel.Voices[0].Family).To(Equal("voicedetect:ecapa_tdnn:ecapa:192"))
|
||||
Expect(sel.Voices[0].Weights).To(Equal(hash))
|
||||
})
|
||||
It("leaves a file-name tagged voice unfingerprinted", func() {
|
||||
sel := voicerecognition.SelectKnownVoices([]voicerecognition.Entry{entry("ada", "spk.gguf", 1, 0)}, "spk.gguf")
|
||||
Expect(sel.Voices[0].Family).To(BeEmpty())
|
||||
Expect(sel.Voices[0].Weights).To(BeEmpty())
|
||||
})
|
||||
It("loads a stored voice that has no family, and keeps the family when there is one", func() {
|
||||
var old, fresh voicerecognition.Metadata
|
||||
Expect(json.Unmarshal([]byte(`{"id":"1","name":"ada","registered_at":"2026-01-01T00:00:00Z","model":"spk.gguf"}`), &old)).To(Succeed())
|
||||
Expect(old.EncoderFamily).To(BeEmpty())
|
||||
raw, err := json.Marshal(voicerecognition.Metadata{ID: "2", Name: "ben", Model: hash, EncoderFamily: "f"})
|
||||
Expect(err).ToNot(HaveOccurred())
|
||||
Expect(json.Unmarshal(raw, &fresh)).To(Succeed())
|
||||
Expect(fresh.EncoderFamily).To(Equal("f"))
|
||||
raw, _ = json.Marshal(old)
|
||||
Expect(string(raw)).ToNot(ContainSubstring("encoder_family"))
|
||||
})
|
||||
})
|
||||
|
||||
var _ = Describe("KnownVoicesFor", func() {
|
||||
It("selects from the registry listing", func() {
|
||||
sel, err := voicerecognition.KnownVoicesFor(context.Background(), listRegistry{entries: []voicerecognition.Entry{entry("ada", "m.gguf", 1)}}, "m.gguf")
|
||||
|
||||
@@ -58,6 +58,11 @@ type Metadata struct {
|
||||
// backend's model name, by default the GGUF file name). Empty for voices
|
||||
// registered before this field existed.
|
||||
Model string `json:"model,omitempty"`
|
||||
// EncoderFamily is the embedding space of the encoder ("voicedetect:<arch>:<name>:<dim>"),
|
||||
// recorded when the encoder reports it (portable enrollment from speaker
|
||||
// profiles). Empty when unknown, and for voices registered before it existed.
|
||||
// Model then holds the weights identity ("sha256:<hex>") for the same voices.
|
||||
EncoderFamily string `json:"encoder_family,omitempty"`
|
||||
}
|
||||
|
||||
// Match is a single result from Identify, ranked by similarity.
|
||||
|
||||
@@ -205,6 +205,7 @@ The same backend also serves the `/v1/audio/diarization` and `/v1/audio/classifi
|
||||
| `speaker_model:<path>` | a model with a diarization model | names registered speakers (see [Voice Recognition]({{% relref "voice-recognition" %}}#naming-speakers-in-diarization-and-live-transcription)) |
|
||||
| `speaker_threshold:<float>` | a model with `speaker_model` | distance (1 minus cosine similarity) under which a speaker is named, in (0, 2); default `0.5` |
|
||||
| `speaker_margin:<float>` | a model with `speaker_model` | how much the best match must beat the runner-up, in [0, 1); default `0.05` |
|
||||
| `speaker_strict:<bool>` | a model with `speaker_model` | do not use registered voices that have no encoder fingerprint (see [Voice Recognition]({{% relref "voice-recognition" %}}#encoder-fingerprint)); default `false` |
|
||||
|
||||
With a `diarization_model` companion, `/v1/audio/transcriptions` labels each segment with its `speaker` (`"0"`, `"1"`, ... in order of first appearance) and splits segments where the speaker changes; with `timestamp_granularities[]=word` each word carries its speaker too. With `stream=true` the closing `transcript.text.done` event lists the segments with their speakers. Pass `-F diarize=false` to skip diarization for one request. The diarization GGUF can also be imported directly: `local-ai models import https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/nemotron-3-diarization-f16.gguf`.
|
||||
|
||||
@@ -305,9 +306,28 @@ The segmenter options below apply to both `vad:true` and `vad_model`. Each is op
|
||||
| `vad_min_pause` | seconds | A silence this long separates two pieces |
|
||||
| `vad_min_speech` | seconds | Shorter speech runs are dropped |
|
||||
| `vad_max_segment` | seconds | Cap on the length of a piece (default 30) |
|
||||
| `vad_trim` | seconds | Each piece shrinks to its first and last speech frame plus this much. Default `0.3`; `0` keeps the whole cuts, as before this option existed |
|
||||
|
||||
`vad_speech_pad` (seconds) pads each region and only affects the [VAD endpoint]({{%relref "features/voice-activity-detection" %}}). `vad_model` needs a `libparakeet.so` that exports `parakeet_capi_transcribe_path_json_vad_with`; an older library fails the load with a message that names it.
|
||||
|
||||
### Dropping noise words (`guard_*`)
|
||||
|
||||
A decode of noise or silence can contain words that no one said. An opt-in filter removes the words that stand alone or sit among low-confidence words, and the words that are only punctuation. It runs on the finished decode. It is off unless one of these options is set, and a bad value fails the load:
|
||||
|
||||
| Option | Unit | Meaning |
|
||||
|---|---|---|
|
||||
| `guard_min_local_conf` | 0 to 1 | A word is dropped when the mean confidence of the words that start within `guard_local_radius` seconds of it, itself included, is below this. `0` is off. `0.5` is a good start; higher values also drop real words on some models |
|
||||
| `guard_local_radius` | seconds, above 0 | The window of that mean. Default `5` |
|
||||
| `guard_drop_punct_only` | `true` or `false` | Drop words that are only punctuation. A CTC model can emit a lone `.` on noise. Default `false` |
|
||||
|
||||
```yaml
|
||||
options:
|
||||
- guard_min_local_conf:0.5
|
||||
- guard_drop_punct_only:true
|
||||
```
|
||||
|
||||
The filter applies to offline transcription, and with `vad:true` or `vad_model` to each piece on its own. It bypasses dynamic batching, like `vad:true`. Streaming is not affected. Speech with confident words comes out the same as without the filter. The number of dropped words is written to the backend log at debug level; the transcription response has no field for it. The options need a `libparakeet.so` that exports `parakeet_capi_transcribe_path_json_with`; an older library fails the load with a message that names it.
|
||||
|
||||
### Bundle GGUF files (several models in one file)
|
||||
|
||||
A bundle is one GGUF file that holds several models, called components. Each component keeps its own licence. The backend opens the components it needs from the one file, so a single model YAML can serve transcription, VAD, diarization, speaker naming and sound events. A bundle needs a `libparakeet.so` from parakeet.cpp with bundle support (pin `781a973` or newer); the format is described in the [parakeet.cpp bundle documentation](https://github.com/mudler/parakeet.cpp/blob/master/docs/bundle.md). Single-model files and every existing option work as before.
|
||||
|
||||
@@ -149,6 +149,7 @@ All options are optional. An unset value keeps the default of the detector in us
|
||||
| `vad_min_pause` | seconds | `0.1` | `0.2` | A silence this long separates two segments; shorter gaps merge |
|
||||
| `vad_min_speech` | seconds | `0.25` | `0.1` | Shorter speech runs are dropped |
|
||||
| `vad_speech_pad` | seconds | `0.03` | `0` | Padding added around each segment |
|
||||
| `vad_trim` | seconds | `0.3` | `0.3` | Only for transcription with `vad:true` or `vad_model`: each piece shrinks to its speech plus this much. `0` keeps the whole cuts. The endpoint ignores it |
|
||||
|
||||
Option names differ from the Silero backend above (`min_silence_duration_ms` and `speech_pad_ms` are in milliseconds there). The same options tune transcription with `vad:true` or `vad_model`; see [audio to text]({{%relref "features/audio-to-text" %}}). Requests on one loaded model run one at a time.
|
||||
|
||||
|
||||
@@ -223,6 +223,42 @@ backend skips a voice whose embedding size does not match the speaker model's,
|
||||
with a warning in the LocalAI log. Naming then falls back to the remaining
|
||||
voices, or to no names.
|
||||
|
||||
### Encoder fingerprint
|
||||
|
||||
Two encoders can give embeddings of the same size (ECAPA and CAM++ both give
|
||||
192 values), so a size match does not prove the voices and the `speaker_model:`
|
||||
file share an embedding space. A voice enrolled from `speaker_profiles` (see
|
||||
[Speaker Diarization]({{% relref "audio-diarization" %}})) is stored with the
|
||||
encoder that made it: its **weights** (`sha256:` of the encoder file, kept in
|
||||
the voice's `model` field as before) and its **family**
|
||||
(`voicedetect:<arch>:<name>:<dim>`, read from the encoder GGUF metadata and
|
||||
stored as `encoder_family`). The parakeet-cpp backend builds the registry with
|
||||
that fingerprint, and libparakeet checks it against the loaded `speaker_model:`
|
||||
before it names anyone:
|
||||
|
||||
| Registered voices | Result |
|
||||
|---|---|
|
||||
| Same family, same weights | names are assigned |
|
||||
| Same family, other weights (for example another quantization) | names are assigned, the library logs a warning |
|
||||
| Another family, and no other usable voice | the request fails, and the error names both families |
|
||||
| Another family, with usable voices of the right family | the other voices are left out, with a warning |
|
||||
| No fingerprint | used as before, with a warning that the encoder is unverified |
|
||||
|
||||
A voice with only a weights identity takes the family of the loaded encoder
|
||||
when the weights are the same file. A voice with a different weights hash and
|
||||
no family is dropped, as before.
|
||||
|
||||
A voice registered from audio through the voice-detect backend has no
|
||||
fingerprint: libvoicedetect reports no architecture or model name, so the
|
||||
backend cannot tell the family, and only the file-name tag described above
|
||||
applies. Such voices and fingerprinted voices cannot share one registry in the
|
||||
library. When a request has any unfingerprinted voice, all of its voices are
|
||||
used without the fingerprint check (the old behaviour). To get the check, enroll
|
||||
every voice from `speaker_profiles`. With `speaker_strict:true` the backend
|
||||
ignores unfingerprinted voices, and a request that has only those fails with
|
||||
the library's message. The family is also reported in the internal backend
|
||||
status next to the identity.
|
||||
|
||||
{{% notice warning %}}
|
||||
Do not set a `model_name:` option on the voice-detect model config. It
|
||||
replaces the default name, the voices are then tagged with it, and they no
|
||||
@@ -240,6 +276,7 @@ options).
|
||||
| `speaker_model:<path>` | none | speaker encoder GGUF; needs a diarization model (the primary one, or `diarization_model:`) |
|
||||
| `speaker_threshold:<float>` | `0.5` | largest distance (1 minus cosine similarity, the unit `/v1/voice/identify` reports) at which a speaker is named; must be in (0, 2) |
|
||||
| `speaker_margin:<float>` | `0.05` | the best match must beat the runner-up by this much, otherwise the speaker stays unnamed; must be in [0, 1) |
|
||||
| `speaker_strict:<bool>` | `false` | ignore registered voices that carry no [encoder fingerprint](#encoder-fingerprint); needs a libparakeet that exports `parakeet_capi_speaker_registry_set_strict` |
|
||||
|
||||
parakeet.cpp's measured starting values for `speaker_threshold` are 0.5 for
|
||||
WeSpeaker ResNet34 and CAM++, and 0.3 for ECAPA. A lower value names fewer
|
||||
|
||||
@@ -5281,6 +5281,10 @@ const docTemplate = `{
|
||||
"dimension": {
|
||||
"type": "integer"
|
||||
},
|
||||
"family": {
|
||||
"description": "embedding space of the encoder; empty when the backend cannot tell",
|
||||
"type": "string"
|
||||
},
|
||||
"identity": {
|
||||
"description": "sha256 of loaded GGUF bytes",
|
||||
"type": "string"
|
||||
@@ -8207,6 +8211,10 @@ const docTemplate = `{
|
||||
"dimension": {
|
||||
"type": "integer"
|
||||
},
|
||||
"family": {
|
||||
"description": "Family is the embedding space of the encoder. The server fills it from the loaded encoder; exported profiles do not carry it and it is not matched.",
|
||||
"type": "string"
|
||||
},
|
||||
"identity": {
|
||||
"type": "string"
|
||||
}
|
||||
|
||||
@@ -5278,6 +5278,10 @@
|
||||
"dimension": {
|
||||
"type": "integer"
|
||||
},
|
||||
"family": {
|
||||
"description": "embedding space of the encoder; empty when the backend cannot tell",
|
||||
"type": "string"
|
||||
},
|
||||
"identity": {
|
||||
"description": "sha256 of loaded GGUF bytes",
|
||||
"type": "string"
|
||||
@@ -8204,6 +8208,10 @@
|
||||
"dimension": {
|
||||
"type": "integer"
|
||||
},
|
||||
"family": {
|
||||
"description": "Family is the embedding space of the encoder. The server fills it from the loaded encoder; exported profiles do not carry it and it is not matched.",
|
||||
"type": "string"
|
||||
},
|
||||
"identity": {
|
||||
"type": "string"
|
||||
}
|
||||
|
||||
@@ -682,6 +682,9 @@ definitions:
|
||||
properties:
|
||||
dimension:
|
||||
type: integer
|
||||
family:
|
||||
description: embedding space of the encoder; empty when the backend cannot tell
|
||||
type: string
|
||||
identity:
|
||||
description: sha256 of loaded GGUF bytes
|
||||
type: string
|
||||
@@ -2759,6 +2762,9 @@ definitions:
|
||||
properties:
|
||||
dimension:
|
||||
type: integer
|
||||
family:
|
||||
description: Family is the embedding space of the encoder. The server fills it from the loaded encoder; exported profiles do not carry it and it is not matched.
|
||||
type: string
|
||||
identity:
|
||||
type: string
|
||||
type: object
|
||||
|
||||
Reference in new issue
Block a user