diff --git a/docs/content/features/model-gallery.md b/docs/content/features/model-gallery.md index 1d266a882..307dcbaa0 100644 --- a/docs/content/features/model-gallery.md +++ b/docs/content/features/model-gallery.md @@ -65,6 +65,21 @@ The files use standard llama.cpp quantization types and the Apache-2.0 license. See the [publisher's model card](https://huggingface.co/agentionai/Qwen3.8-27B-AP-GGUF) for quantization details. These entries do not enable MTP speculative decoding. +## Swift 1.5 Qwen3.8-27B GSQ-RCO + +Install `swift-1.5-qwen3.8-27b-gsq-rco` for text chat with llama.cpp. +The gallery groups IQ2_XS, IQ2_S, IQ3_XXS, and IQ3_S quantizations of this 27B reasoning and coding model. +To select IQ3_S explicitly, run: + +```bash +local-ai models install swift-1.5-qwen3.8-27b-gsq-rco --variant swift-1.5-qwen3.8-27b-gsq-rco-iq3-s +``` + +The configurations use the embedded chat template and default to 32,768 context tokens. +These builds support text chat only: the publisher has no verified vision projector for this release. +They use standard GGUF files without MTP decoding. +See the [model card](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF) and [Swift Open License v1.0](https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/blob/main/LICENSE) for usage terms. + ## MiMo-V2.6-Distill-Qwen-9B Install `mimo-v2.6-distill-qwen-9b` for text and image chat with llama.cpp. diff --git a/gallery/index.yaml b/gallery/index.yaml index f4cf195b8..331d4e35c 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -6017,6 +6017,178 @@ - filename: llama-cpp/mmproj/qwen3.8-27b-ridge/mmproj-Qwen3.8-27B-BF16.gguf uri: huggingface://empero-ai/Qwen3.8-27B-Ridge-GGUF/mmproj-Qwen3.8-27B-BF16.gguf sha256: 52228402ce4823f10705d901813cd43ced71859524cf2d8bf83305ad6b7dcbc2 +- name: "swift-1.5-qwen3.8-27b-gsq-rco" + variants: + - model: swift-1.5-qwen3.8-27b-gsq-rco-iq2-s + - model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs + - model: swift-1.5-qwen3.8-27b-gsq-rco-iq3-s + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ2_XS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf + sha256: 714c509c3fc496ea4abc409097658df7cd218bc966f78e1459fc1649758a9de8 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_XS.gguf +- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq2-s" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ2_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf + sha256: 08fac9876117b2cadb6b79fc7708d9612511c2fa31f3726f162e757870272455 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ2_S.gguf +- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-xxs" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ3_XXS GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf + sha256: 86969b8bde72e602bfb42deb83eb8bb3706c8f14250641f6444dd2355f934ac2 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_XXS.gguf +- name: "swift-1.5-qwen3.8-27b-gsq-rco-iq3-s" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27b + - https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF + license: "swift-open-license-1.0" + description: | + Swift 1.5 is a 27B Qwen3.8 fine-tune for reasoning, coding, and agent tasks. + This IQ3_S GGUF uses GSQ-RCO mixed-precision quantization with llama.cpp. + Text chat only; the publisher provides no verified vision projector for this release. + The weights use the Swift Open License v1.0. + tags: + - llm + - gguf + - cpu + - gpu + - reasoning + overrides: + backend: llama-cpp + context_size: 32768 + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + options: + - use_jinja:true + parameters: + min_p: 0 + model: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf + presence_penalty: 0 + repeat_penalty: 1 + temperature: 1 + top_k: 20 + top_p: 0.95 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/swift-1.5-qwen3.8-27b-gsq-rco/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf + sha256: 1333c6ea70ef348d4ac6d62732772e8ad6571ac5b3754c14ed54f1a0d904a786 + uri: https://huggingface.co/ukisai/Swift-1.5-Qwen3.8-27B-GSQ-RCO-GGUF/resolve/d74895bbe5db4bec1e0024e7cc87d59c02d7631a/Swift-1.5-Qwen3.8-27B-GSQ-RCO-IQ3_S.gguf - !!merge <<: *qwen3-8-27b name: "qwen3.8-27b-gsq-rco-iq2-xs" variants: []