From dcddb641f0564ce3d328748581655b6b1b4a32d0 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> Date: Sun, 27 Sep 2026 08:05:46 +0000 Subject: [PATCH] chore(gallery): add Agention Qwen3.8 variants Add IQ4_XS and Q4_K_M GGUF builds with a BF16 vision projector. Pin verified artifacts and document installation and variant selection. Assisted-by: Codex:gpt-6 --- docs/content/features/model-gallery.md | 18 +++++ gallery/index.yaml | 98 ++++++++++++++++++++++++++ 2 files changed, 116 insertions(+) diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index dd48e73c7..bf83b6300 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -39,6 +39,24 @@ Both views use the same model selection and store the view, search, filter, and selection in the URL. Installing from Explore does not move you away from the catalog; the entry updates in place when the operation finishes. +## Qwen3.8-27B Agention Precision + +The gallery includes Agention Precision IQ4_XS and Q4_K_M GGUF builds of +Qwen3.8-27B for llama.cpp. Both include the BF16 vision projector for image +input and use a 32,768-token context by default. + +Install with automatic variant selection: + +```bash +local-ai models install qwen3.8-27b-agention-iq4-xs +``` + +To select a specific build, pass `--variant qwen3.8-27b-agention-iq4-xs` +or `--variant qwen3.8-27b-agention-q4-k-m` to the same command. +The files use standard llama.cpp quantization types and the Apache-2.0 license. +See the [publisher's model card](https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF) +for quantization details. These entries do not enable MTP speculative decoding. + ## MiMo-V2.6-Distill-Qwen-9B Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. diff --git a/gallery/index.yaml b/gallery/index.yaml index 84733ea18..0cab8c751 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -5247,6 +5247,104 @@ - filename: llama-cpp/mmproj/qwen3.8-27b-obliterated/mmproj-model-bf16.gguf uri: huggingface://OBLITERATUS/Qwen3.8-27B-OBLITERATED/mmproj-model-bf16.gguf sha256: e484e3b7e907ed0e0644c0de56c3f5929c7ad5c9c6cc84d35a9d8dc08d461545 +- name: qwen3.8-27b-agention-iq4-xs + url: github:mudler/LocalAI/gallery/virtual.yaml@master + variants: + - model: qwen3.8-27b-agention-q4-k-m + urls: + - https://huggingface.co/Qwen/Qwen3.8-27B + - https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF + license: apache-2.0 + description: | + Qwen3.8-27B with Agention Precision IQ4_XS quantization for llama.cpp. + This 27B reasoning model supports text and image input. The download + includes the BF16 vision projector and uses the embedded chat template. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - reasoning + - vision + - multimodal + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + - vision + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + options: + - use_jinja:true + parameters: + model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf + temperature: 1 + top_p: 0.95 + top_k: 20 + min_p: 0 + repeat_penalty: 1 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-IQ4_XS.gguf + sha256: 2074fd5c3c7f6540913c2f62ad02c50b3f7dde7880d18b3acb02432f2edcab67 + - filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf + sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53 +- name: qwen3.8-27b-agention-q4-k-m + url: github:mudler/LocalAI/gallery/virtual.yaml@master + urls: + - https://huggingface.co/Qwen/Qwen3.8-27B + - https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF + license: apache-2.0 + description: | + Qwen3.8-27B with Agention Precision Q4_K_M quantization for llama.cpp. + This 27B reasoning model supports text and image input. The download + includes the BF16 vision projector and uses the embedded chat template. + tags: + - llm + - gguf + - cpu + - gpu + - qwen + - reasoning + - vision + - multimodal + overrides: + backend: llama-cpp + context_size: 32768 + known_usecases: + - chat + - vision + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + options: + - use_jinja:true + parameters: + model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf + temperature: 1 + top_p: 0.95 + top_k: 20 + min_p: 0 + repeat_penalty: 1 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-Q4_K_M.gguf + sha256: c4c4b1d393b288205d6303c941c0c954d0ea57ef8e3228bca74187cc858e9d8e + - filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf + uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf + sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53 - &qwen3-8-27b name: "qwen3.8-27b-q4" variants: