mirror of
https://github.com/mudler/LocalAI.git
synced 2026-09-28 17:15:02 -04:00
chore(gallery): add Agention Qwen3.8 variants
Add IQ4_XS and Q4_K_M GGUF builds with a BF16 vision projector. Pin verified artifacts and document installation and variant selection. Assisted-by: Codex:gpt-6
This commit is contained in:
1 parent
92b8f1d8ed
commit
dcddb641f0
2 files changed
+116
No files matched your search
@@ -39,6 +39,24 @@ Both views use the same model selection and store the view, search, filter, and
|
||||
selection in the URL. Installing from Explore does not move you away from the
|
||||
catalog; the entry updates in place when the operation finishes.
|
||||
|
||||
## Qwen3.8-27B Agention Precision
|
||||
|
||||
The gallery includes Agention Precision IQ4_XS and Q4_K_M GGUF builds of
|
||||
Qwen3.8-27B for llama.cpp. Both include the BF16 vision projector for image
|
||||
input and use a 32,768-token context by default.
|
||||
|
||||
Install with automatic variant selection:
|
||||
|
||||
```bash
|
||||
local-ai models install qwen3.8-27b-agention-iq4-xs
|
||||
```
|
||||
|
||||
To select a specific build, pass `--variant qwen3.8-27b-agention-iq4-xs`
|
||||
or `--variant qwen3.8-27b-agention-q4-k-m` to the same command.
|
||||
The files use standard llama.cpp quantization types and the Apache-2.0 license.
|
||||
See the [publisher's model card](https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF)
|
||||
for quantization details. These entries do not enable MTP speculative decoding.
|
||||
|
||||
## MiMo-V2.6-Distill-Qwen-9B
|
||||
|
||||
Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp.
|
||||
|
||||
@@ -5247,6 +5247,104 @@
|
||||
- filename: llama-cpp/mmproj/qwen3.8-27b-obliterated/mmproj-model-bf16.gguf
|
||||
uri: huggingface://OBLITERATUS/Qwen3.8-27B-OBLITERATED/mmproj-model-bf16.gguf
|
||||
sha256: e484e3b7e907ed0e0644c0de56c3f5929c7ad5c9c6cc84d35a9d8dc08d461545
|
||||
- name: qwen3.8-27b-agention-iq4-xs
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
variants:
|
||||
- model: qwen3.8-27b-agention-q4-k-m
|
||||
urls:
|
||||
- https://huggingface.co/Qwen/Qwen3.8-27B
|
||||
- https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF
|
||||
license: apache-2.0
|
||||
description: |
|
||||
Qwen3.8-27B with Agention Precision IQ4_XS quantization for llama.cpp.
|
||||
This 27B reasoning model supports text and image input. The download
|
||||
includes the BF16 vision projector and uses the embedded chat template.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- qwen
|
||||
- reasoning
|
||||
- vision
|
||||
- multimodal
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf
|
||||
temperature: 1
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0
|
||||
repeat_penalty: 1
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-IQ4_XS.gguf
|
||||
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-IQ4_XS.gguf
|
||||
sha256: 2074fd5c3c7f6540913c2f62ad02c50b3f7dde7880d18b3acb02432f2edcab67
|
||||
- filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
|
||||
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf
|
||||
sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
|
||||
- name: qwen3.8-27b-agention-q4-k-m
|
||||
url: github:mudler/LocalAI/gallery/virtual.yaml@master
|
||||
urls:
|
||||
- https://huggingface.co/Qwen/Qwen3.8-27B
|
||||
- https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF
|
||||
license: apache-2.0
|
||||
description: |
|
||||
Qwen3.8-27B with Agention Precision Q4_K_M quantization for llama.cpp.
|
||||
This 27B reasoning model supports text and image input. The download
|
||||
includes the BF16 vision projector and uses the embedded chat template.
|
||||
tags:
|
||||
- llm
|
||||
- gguf
|
||||
- cpu
|
||||
- gpu
|
||||
- qwen
|
||||
- reasoning
|
||||
- vision
|
||||
- multimodal
|
||||
overrides:
|
||||
backend: llama-cpp
|
||||
context_size: 32768
|
||||
known_usecases:
|
||||
- chat
|
||||
- vision
|
||||
function:
|
||||
automatic_tool_parsing_fallback: true
|
||||
grammar:
|
||||
disable: true
|
||||
mmproj: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
|
||||
options:
|
||||
- use_jinja:true
|
||||
parameters:
|
||||
model: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf
|
||||
temperature: 1
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0
|
||||
repeat_penalty: 1
|
||||
template:
|
||||
use_tokenizer_template: true
|
||||
files:
|
||||
- filename: llama-cpp/models/qwen3.8-27b-agention/Qwen3.8-27B-AP-Q4_K_M.gguf
|
||||
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/Qwen3.8-27B-AP-Q4_K_M.gguf
|
||||
sha256: c4c4b1d393b288205d6303c941c0c954d0ea57ef8e3228bca74187cc858e9d8e
|
||||
- filename: llama-cpp/mmproj/qwen3.8-27b-agention/mmproj-BF16.gguf
|
||||
uri: https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF/resolve/17bf39b5fafab9e8ac379c78c207568d73da9a7b/mmproj-BF16.gguf
|
||||
sha256: 83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
|
||||
- &qwen3-8-27b
|
||||
name: "qwen3.8-27b-q4"
|
||||
variants:
|
||||
|
||||
Reference in new issue
Block a user