diff --git a/backend/go/parakeet-cpp/Makefile b/backend/go/parakeet-cpp/Makefile index da0584a3a..9a4ce573b 100644 --- a/backend/go/parakeet-cpp/Makefile +++ b/backend/go/parakeet-cpp/Makefile @@ -1,6 +1,6 @@ # parakeet-cpp backend Makefile. # -# Upstream pin lives below as PARAKEET_VERSION?=6165e3de4b10fd736ec6ec49bf6c5c26558de0dd +# Upstream pin lives below as PARAKEET_VERSION?=e53a2539bd7fc3290951037696b953b34a9c4b9c # (.github/bump_deps.sh) can find and update it - matches the # whisper.cpp / ds4 / vibevoice-cpp convention. # @@ -15,7 +15,7 @@ # That's what the L0 smoke test uses. The default target below does the # proper clone-at-pin + cmake build so CI doesn't need a side-checkout. -PARAKEET_VERSION?=6165e3de4b10fd736ec6ec49bf6c5c26558de0dd +PARAKEET_VERSION?=e53a2539bd7fc3290951037696b953b34a9c4b9c PARAKEET_REPO?=https://github.com/mudler/parakeet.cpp GOCMD?=go diff --git a/core/gallery/parakeet_vad_entries_test.go b/core/gallery/parakeet_vad_entries_test.go index cdf24f725..dead1ff34 100644 --- a/core/gallery/parakeet_vad_entries_test.go +++ b/core/gallery/parakeet_vad_entries_test.go @@ -62,6 +62,27 @@ var _ = Describe("gallery/index.yaml parakeet-cpp VAD entries", func() { } }) + It("serves the VAD-only slices from files with the published checksums", func() { + entries := byName() + slices := map[string][2]string{ + "parakeet-cpp-vad-moondream-redux": {"parakeet-cpp/redux-vad.gguf", "588e1d6e2ee5b6cdfd9ec5ea98dc0993d5bea498d9cc4ec8d6077041eef8a34f"}, + "parakeet-cpp-vad-moondream-ultra": {"parakeet-cpp/ultra-vad-q8_0.gguf", "8b891a4435e97438104ca07c72530d0c5fe62b986baee48b2dd4e1500c1d4758"}, + } + for name, want := range slices { + e, ok := entries[name] + Expect(ok).To(BeTrue(), name) + Expect(e.Overrides["backend"]).To(Equal("parakeet-cpp"), name) + Expect(e.Overrides["known_usecases"]).To(ConsistOf("vad"), name) + Expect(e.License).To(Equal("cc-by-4.0"), name) + Expect(e.Overrides["parameters"]).To(HaveKeyWithValue("model", want[0]), name) + Expect(e.AdditionalFiles).To(HaveLen(1), name) + Expect(e.AdditionalFiles[0].Filename).To(Equal(want[0]), name) + Expect(e.AdditionalFiles[0].SHA256).To(Equal(want[1]), name) + Expect(e.AdditionalFiles[0].URI).To(HavePrefix("https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/"), name) + Expect(e.Variants).To(BeEmpty(), name) + } + }) + It("installs Silero from the parakeet-cpp-vad entry and declares no variants", func() { // Variant ranking prefers the larger build that fits, and these are // different detectors, so the entry must not offer a choice. diff --git a/docs/content/features/audio-to-text.md b/docs/content/features/audio-to-text.md index b5a363afd..08e5beed6 100644 --- a/docs/content/features/audio-to-text.md +++ b/docs/content/features/audio-to-text.md @@ -257,6 +257,17 @@ By default each request runs on its own. Raise `batch_max_size` (for example 4 t The packed Redux file stores the encoder as ternary weights (213 MB). It cannot load on a GPU backend and cannot stream. If a GPU build fails to load it, check the backend log for the library message and use the `redux-f16` or `redux-q8_0` entry instead. The weights are CC-BY-4.0: credit Moondream and NVIDIA. +#### VAD-only slices + +If you only need the VAD head, for the [VAD endpoint]({{%relref "features/voice-activity-detection" %}}) or to cut audio before transcription, the same repository has two small files with the head cut out of the full model. The weights are not retrained, and the files cannot transcribe: + +| Gallery entry | File | Size | Cut from | +|---|---|---|---| +| `parakeet-cpp-vad-moondream-redux` | `redux-vad.gguf` | 9.9 MB | Redux (213 MB packed to 1.4 GB) | +| `parakeet-cpp-vad-moondream-ultra` | `ultra-vad-q8_0.gguf` | 6.0 MB | Ultra Q8_0 | + +Measured by the parakeet.cpp author against loading a whole Redux or Ultra model: the files are 6 to 10 MB instead of 213 MB to 1.4 GB, load in a few milliseconds instead of 0.1 to 0.7 s, and use about 245 MiB peak memory for a 33 s clip instead of 0.6 to 1.6 GiB. The output is byte-identical to the full parent model, and the speed is the same as the parent's head. A transcription request on a slice fails with an error. The slices load only with a parakeet.cpp build that includes VAD-only GGUF support, so an older backend build fails to load them. The weights are CC-BY-4.0: credit Moondream and NVIDIA. + With `vad:true`, long audio is cut at pauses found by the model's VAD head into pieces of at most 30 seconds, and each piece is transcribed in turn. Word timestamps stay relative to the whole file. Audio of 30 seconds or less gives the same result as without the option. The gallery entries set it. Add it to your own model YAML like this: ```yaml diff --git a/docs/content/features/voice-activity-detection.md b/docs/content/features/voice-activity-detection.md index b868c7faa..75e7c358d 100644 --- a/docs/content/features/voice-activity-detection.md +++ b/docs/content/features/voice-activity-detection.md @@ -122,9 +122,10 @@ Reload the model (or restart LocalAI) after changing these options. The `parakeet-cpp` backend serves the same endpoint. It runs one of two detectors: - **Silero VAD** from a GGUF file (gallery entry `parakeet-cpp-silero-vad-f16`, 1.3 MB). One probability per 32 ms. -- **The VAD head** of a Moondream Ultra or Redux model (gallery entries `parakeet-cpp-vad-moondream-ultra-q8_0` and `parakeet-cpp-vad-moondream-redux-packed`). One probability per 80 ms. The packed Redux file runs on CPU only. +- **The VAD head** of a full Moondream Ultra or Redux model (gallery entries `parakeet-cpp-vad-moondream-ultra-q8_0` and `parakeet-cpp-vad-moondream-redux-packed`). One probability per 80 ms. The packed Redux file runs on CPU only. +- **A VAD-only slice** of that head (gallery entries `parakeet-cpp-vad-moondream-redux`, 9.9 MB, and `parakeet-cpp-vad-moondream-ultra`, 6.0 MB). The slice is cut out of the full model without retraining, so the segments are byte-identical to the full model's head, and the speed is the same. Compared with loading the whole model (213 MB to 1.4 GB), the file is 6 to 10 MB, loads in a few milliseconds instead of 0.1 to 0.7 s, and needs about 245 MiB of peak memory for a 33 s clip instead of 0.6 to 1.6 GiB. A slice cannot transcribe, and it needs a parakeet.cpp build with VAD-only GGUF support (pin e53a253 or newer). -The entry `parakeet-cpp-vad` installs Silero. The detectors differ and are not variants of one model, so install the entry of the VAD head by name if you want it. The request is the same as above: `audio` is 16 kHz mono float32 PCM, and the response lists `segments` with `start` and `end` in seconds. An ASR model that has no VAD head fails the request with `model has no VAD head`. +The entry `parakeet-cpp-vad` installs Silero. The detectors differ and are not variants of one model, so install the entry of the VAD head by name if you want it (`parakeet-cpp-vad-moondream-redux` or `parakeet-cpp-vad-moondream-ultra` for the small files). The request is the same as above: `audio` is 16 kHz mono float32 PCM, and the response lists `segments` with `start` and `end` in seconds. An ASR model that has no VAD head fails the request with `model has no VAD head`. ```yaml name: parakeet-vad diff --git a/gallery/index.yaml b/gallery/index.yaml index 1edf94d2d..29d68410d 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -54019,6 +54019,70 @@ - filename: parakeet-cpp/ultra-q8_0.gguf uri: https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/ultra-q8_0.gguf sha256: c2fb452a9df468a141012b01c8c168a25ce93f710897c7de6e353c6cc250986a +- name: parakeet-cpp-vad-moondream-redux + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/mudler/parakeet-cpp-gguf + - https://github.com/mudler/parakeet.cpp + - https://github.com/mudler/parakeet.cpp/blob/master/docs/vad.md + description: | + Voice activity detection with the VAD head of Moondream Redux cut out of the full model, as a small file (about 10 MB). It detects speech only and cannot transcribe: a transcription request fails with an error. The head is not retrained, so the output is the same as the full model's head, with a much smaller download and a faster load. + Needs a parakeet.cpp build that can load VAD-only GGUF files. Older backend builds fail to load the file. For the full model, install parakeet-cpp-vad-moondream-redux-packed. + Use it for the VAD endpoint. License CC-BY-4.0: credit Moondream and NVIDIA. + license: cc-by-4.0 + tags: + - parakeet + - parakeet-cpp + - moondream + - vad + - voice-activity-detection + - gguf + - ggml + - speech + - cpu + overrides: + backend: parakeet-cpp + known_usecases: + - vad + name: parakeet-cpp-vad-moondream-redux + parameters: + model: parakeet-cpp/redux-vad.gguf + files: + - filename: parakeet-cpp/redux-vad.gguf + uri: https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/redux-vad.gguf + sha256: 588e1d6e2ee5b6cdfd9ec5ea98dc0993d5bea498d9cc4ec8d6077041eef8a34f +- name: parakeet-cpp-vad-moondream-ultra + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/mudler/parakeet-cpp-gguf + - https://github.com/mudler/parakeet.cpp + - https://github.com/mudler/parakeet.cpp/blob/master/docs/vad.md + description: | + Voice activity detection with the VAD head of Moondream Ultra, Q8_0 cut out of the full model, as a small file (about 6 MB). It detects speech only and cannot transcribe: a transcription request fails with an error. The head is not retrained, so the output is the same as the full model's head, with a much smaller download and a faster load. + Needs a parakeet.cpp build that can load VAD-only GGUF files. Older backend builds fail to load the file. For the full model, install parakeet-cpp-vad-moondream-ultra-q8_0. + Use it for the VAD endpoint. License CC-BY-4.0: credit Moondream and NVIDIA. + license: cc-by-4.0 + tags: + - parakeet + - parakeet-cpp + - moondream + - vad + - voice-activity-detection + - gguf + - ggml + - speech + - cpu + overrides: + backend: parakeet-cpp + known_usecases: + - vad + name: parakeet-cpp-vad-moondream-ultra + parameters: + model: parakeet-cpp/ultra-vad-q8_0.gguf + files: + - filename: parakeet-cpp/ultra-vad-q8_0.gguf + uri: https://huggingface.co/mudler/parakeet-cpp-gguf/resolve/main/ultra-vad-q8_0.gguf + sha256: 8b891a4435e97438104ca07c72530d0c5fe62b986baee48b2dd4e1500c1d4758 - name: parakeet-cpp-vad url: github:mudler/LocalAI/gallery/virtual.yaml@master urls: