From 92b8f1d8ede207e8a7c2862005eef89cbc443d56 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Sat, 26 Sep 2026 18:54:08 +0200 Subject: [PATCH] chore(gallery): add MiMo distill Qwen 9B variants (#12282) Add Q4_K_M and Q8_0 builds with the F16 vision projector and pinned artifact URLs. Document installation and explicit variant selection. Assisted-by: Codex:GPT-6 Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- docs/content/features/model-gallery.md | 9 +++ gallery/index.yaml | 86 ++++++++++++++++++++++++++ 2 files changed, 95 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index 7baa18087..dd48e73c7 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -39,6 +39,15 @@ Both views use the same model selection and store the view, search, filter, and selection in the URL. Installing from Explore does not move you away from the catalog; the entry updates in place when the operation finishes. +## MiMo-V2.6-Distill-Qwen-9B + +Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. +This MIT-licensed 9B Qwen3.5 fine-tune targets coding, agent tasks, and visual coding. +The gallery groups Q4_K_M and Q8_0 builds as variants; both include the F16 vision projector. +To select Q8_0 explicitly, run `local-ai models install mimo-v2.6-distill-qwen-9b --variant mimo-v2.6-distill-qwen-9b-q8`. +The configurations default to 32,768 context tokens and use the model's embedded chat template. +See the [model card](https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B) for training details. + ## Hemmingway-1 Install `hemmingway-1` for English text generation with llama.cpp. The gallery groups its Q4_K_M and Q8_0 builds as variants. diff --git a/gallery/index.yaml b/gallery/index.yaml index d5419f747..84733ea18 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -302,6 +302,92 @@ - filename: llama-cpp/mmproj/Qwopus3.8-27B-Flash-MTP-Q4_K_M/mmproj-F32.gguf uri: https://huggingface.co/Jackrong/Qwopus3.8-27B-Flash-GGUF/resolve/e146d61e88782677805b3b68ad3adf8674dde80d/mmproj-F32.gguf sha256: 52e6818e4d18eea010c50e5245eaa10a8cc3dcc30efea4ff60cbad8abf5669e1 +- name: mimo-v2.6-distill-qwen-9b + variants: + - model: mimo-v2.6-distill-qwen-9b-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B + - https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF + description: | + MiMo-V2.6-Distill-Qwen-9B is Xiaomi MiMo's 9B Qwen3.5 fine-tune for coding, agent tasks, and visual coding. + This Q4_K_M GGUF build uses llama.cpp with the model's embedded chat template and includes the F16 vision projector. + license: mit + tags: + - llm + - gguf + - cpu + - gpu + - coding + - vision + - multimodal + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf + options: + - use_jinja:true + template: + use_tokenizer_template: true + parameters: + model: MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf + files: + - filename: MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf + sha256: 4bca6f18c73f72270c7a20c2ea2bea581de8246e318714277120369d34048c81 + uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/MiMo-V2.6-Distill-Qwen-9B-Q4_K_M.gguf + - filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf + sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576 + uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf +- name: mimo-v2.6-distill-qwen-9b-q8 + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/XiaomiMiMo/MiMo-V2.6-Distill-Qwen-9B + - https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF + description: | + MiMo-V2.6-Distill-Qwen-9B is Xiaomi MiMo's 9B Qwen3.5 fine-tune for coding, agent tasks, and visual coding. + This Q8_0 GGUF build uses llama.cpp with the model's embedded chat template and includes the F16 vision projector. + license: mit + tags: + - llm + - gguf + - cpu + - gpu + - coding + - vision + - multimodal + last_checked: "2026-09-26" + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + - vision + mmproj: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf + options: + - use_jinja:true + template: + use_tokenizer_template: true + parameters: + model: MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf + files: + - filename: MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf + sha256: 2fad0aa11bb9e7aa491ff12f768954f9dd0a6e7d4ce4a897ca73ec420f3b90ae + uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/MiMo-V2.6-Distill-Qwen-9B-Q8_0.gguf + - filename: mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf + sha256: ff348f3180a63188aa7285db85f550fe38acb61dd013c599eb8bad08d2cc2576 + uri: https://huggingface.co/bartowski/MiMo-V2.6-Distill-Qwen-9B-GGUF/resolve/4371da10c84fb26da3592d4cf312d24aa82b7b65/mmproj-MiMo-V2.6-Distill-Qwen-9B-f16.gguf - name: hemmingway-1 variants: - model: hemmingway-1-q8