From 3c9047654e3d12aecd9467245902446912e56faf Mon Sep 17 00:00:00 2001 From: mudler <2420543+mudler@users.noreply.github.com> Date: Wed, 23 Sep 2026 12:58:03 +0000 Subject: [PATCH 1/2] chore(model gallery): :robot: add new models via gallery agent Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gallery/index.yaml | 46 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 46 insertions(+) diff --git a/gallery/index.yaml b/gallery/index.yaml index d5419f747..cceda1b77 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -1,4 +1,50 @@ --- +- name: "ternary-bonsai-2-27b" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf + description: | + # Qwen3.8-27B + + > [!Note] + > This repository contains model weights and configuration files for the post-trained model in the Hugging Face Transformers format. + > + > These artifacts are compatible with Hugging Face Transformers, vLLM, SGLang, TokenSpeed, etc. + + > [!Tip] + > For users seeking managed, scalable inference without infrastructure maintenance, the official Qwen API service is provided by Qwen Cloud. + > In particular, **Qwen3.8-27B** will be available as a hosted version with more production features, e.g., 1M context length by default, official built-in tools. For more information, please refer to the Qwen3.8-27B Overview. The service is coming soon. Stay tuned for updates. + + Following the widespread community adoption of the Qwen3.5 and Qwen3.6 series, we are pleased to introduce Qwen3.8, the most capable generation in the Qwen open-model family to date. + + ... + license: "apache-2.0" + tags: + - llm + - gguf + icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + mmproj: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf + options: + - use_jinja:true + parameters: + model: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-PTQ1_0.gguf + sha256: 53107f530aa52eb00912263ab1ee29bd199261c87cd7b4ad4ca1318c1fe33ee3 + uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-PTQ1_0.gguf + - filename: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf + sha256: 6807ede61d570bb86ba34b756a0fa109edc33668604de867c6ea6d8f1d631903 + uri: https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf/resolve/main/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf - name: "ornith-1.5-9b-uncensored" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: From 4a691099abeda53aeb16fdac1596926006c98f31 Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Sun, 27 Sep 2026 19:06:53 +0000 Subject: [PATCH 2/2] chore(gallery): serve ternary-bonsai-2-27b with the bonsai backend PTQ1_0 is a Prism-private GGUF type (GGML_TYPE_PTQ1_0 = 143 in the PrismML llama.cpp fork), so stock llama-cpp cannot load it. Switch to the bonsai backend like the existing ternary-bonsai-27b entries, and replace the scraped Qwen3.8 description and icon. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- gallery/index.yaml | 28 ++++++++++++---------------- 1 file changed, 12 insertions(+), 16 deletions(-) diff --git a/gallery/index.yaml b/gallery/index.yaml index cceda1b77..a348c60dc 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -3,34 +3,30 @@ url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: - https://huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf + - https://github.com/PrismML-Eng/llama.cpp description: | - # Qwen3.8-27B - - > [!Note] - > This repository contains model weights and configuration files for the post-trained model in the Hugging Face Transformers format. - > - > These artifacts are compatible with Hugging Face Transformers, vLLM, SGLang, TokenSpeed, etc. - - > [!Tip] - > For users seeking managed, scalable inference without infrastructure maintenance, the official Qwen API service is provided by Qwen Cloud. - > In particular, **Qwen3.8-27B** will be available as a hosted version with more production features, e.g., 1M context length by default, official built-in tools. For more information, please refer to the Qwen3.8-27B Overview. The service is coming soon. Stay tuned for updates. - - Following the widespread community adoption of the Qwen3.5 and Qwen3.6 series, we are pleased to introduce Qwen3.8, the most capable generation in the Qwen open-model family to date. - - ... + Ternary Bonsai 2 27B (PrismML) is a 27B-class reasoning model with ternary + transformer weights. This PTQ1_0 build packs the trits densely at 1.75 bits + per weight (5.95 GB) and includes the Q8_0 vision projector. PTQ1_0 is a + Prism-private GGUF type, so the entry uses the bonsai backend (PrismML's + llama.cpp fork) instead of stock llama.cpp. license: "apache-2.0" tags: - llm - gguf - icon: https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3.5/demo/CI_Demo/mathv-1327.jpg + - reasoning + - vision + - multimodal + icon: https://huggingface.co/prism-ml/Ternary-Bonsai-27B-gguf/resolve/main/assets/bonsai-logo.svg overrides: - backend: llama-cpp + backend: bonsai function: automatic_tool_parsing_fallback: true grammar: disable: true known_usecases: - chat + - vision mmproj: llama-cpp/mmproj/Ternary-Bonsai-2-27B-PTQ1_0/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf options: - use_jinja:true