From e17039fb30972e00287e7a5ad58c2fb550c7698b Mon Sep 17 00:00:00 2001 From: mudler <2420543+mudler@users.noreply.github.com> Date: Thu, 24 Sep 2026 00:49:43 +0000 Subject: [PATCH 1/2] chore(model gallery): :robot: add new models via gallery agent Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> --- gallery/index.yaml | 72 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 72 insertions(+) diff --git a/gallery/index.yaml b/gallery/index.yaml index d5419f747..929c337bb 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -1,4 +1,76 @@ --- +- name: "swift-qwen3.8-27b" + url: "github:mudler/LocalAI/gallery/virtual.yaml@master" + urls: + - https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF + description: | + Website  •  + Learn more  •  + GGUF  •  + Enterprise licensing + + # Swift-Qwen3.8-27B + + Swift-Qwen3.8-27B is UkisAI's reasoning-efficient derivative of Qwen3.8-27B, + using **58.3% fewer thinking tokens** while maintaining near-identical performance + (**<1% loss**) and as a result getting a **x1.95 speed-up** on several tasks. + + The prompt is a sample from LiveCodeBench v6 + + ## Training approach + + We built Swift by identifying reasoning-marker tokens that, in our analysis, trigger overthinking in Qwen’s + reasoning rollouts. We then fine-tuned Qwen by penalizing usage of those tokens while it reasons. + + Swift produces shorter reasoning traces. In our testing, we also observe fewer overthinking errors. + + For maximum gains, Swift also includes a transfer component derived from + BottleCap AI's ThinkingCap-Qwen3.6-27B. + + ## Evaluation scope + + > All results below compare the Qwen3.8-27B BF16 base with the same base plus the + > Swift adapter. + + ## Benchmarks + + ... + license: "other" + tags: + - llm + - gguf + - reasoning + overrides: + backend: llama-cpp + function: + automatic_tool_parsing_fallback: true + grammar: + disable: true + known_usecases: + - chat + mmproj: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf + options: + - use_jinja:true + - spec_type:draft-mtp + - spec_n_max:6 + - spec_p_min:0.75 + parameters: + min_p: 0 + model: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf + presence_penalty: 1.5 + repeat_penalty: 1 + temperature: 0.7 + top_k: 20 + top_p: 0.8 + template: + use_tokenizer_template: true + files: + - filename: llama-cpp/models/Swift-Qwen3.8-27B-Q4_K_M/Swift-Qwen3.8-27B-Q4_K_M.gguf + sha256: ad5811e291431bd0de1cec0c4004a5eac98daee9850882edac69a823209e88ab + uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/Swift-Qwen3.8-27B-Q4_K_M.gguf + - filename: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf + sha256: daa1116c9422fa390cc8688495da0e91781f92841dfc3b31a378ff252571745a + uri: https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF/resolve/main/mmproj-Swift-Qwen3.8-27B-F16.gguf - name: "ornith-1.5-9b-uncensored" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: From 7340970ae798a35e1e877e18f06c8a4791160d0e Mon Sep 17 00:00:00 2001 From: Ettore Di Giacinto Date: Sun, 27 Sep 2026 19:06:21 +0000 Subject: [PATCH 2/2] chore(gallery): tag swift-qwen3.8-27b as mtp and vision, fix license The entry enables spec_type:draft-mtp, so variant ranking needs the mtp tag. Replace the scraped model-card description, set the Swift Open License v1.0 and link the base model repo. Signed-off-by: Ettore Di Giacinto Assisted-by: Claude:claude-opus-5-5 [Claude Code] --- gallery/index.yaml | 42 ++++++++++-------------------------------- 1 file changed, 10 insertions(+), 32 deletions(-) diff --git a/gallery/index.yaml b/gallery/index.yaml index 929c337bb..eb9e5e15c 100644 --- a/gallery/index.yaml +++ b/gallery/index.yaml @@ -2,44 +2,21 @@ - name: "swift-qwen3.8-27b" url: "github:mudler/LocalAI/gallery/virtual.yaml@master" urls: + - https://huggingface.co/ukisai/Swift-Qwen3.8-27b - https://huggingface.co/ukisai/Swift-Qwen3.8-27B-GGUF description: | - Website  •  - Learn more  •  - GGUF  •  - Enterprise licensing - - # Swift-Qwen3.8-27B - - Swift-Qwen3.8-27B is UkisAI's reasoning-efficient derivative of Qwen3.8-27B, - using **58.3% fewer thinking tokens** while maintaining near-identical performance - (**<1% loss**) and as a result getting a **x1.95 speed-up** on several tasks. - - The prompt is a sample from LiveCodeBench v6 - - ## Training approach - - We built Swift by identifying reasoning-marker tokens that, in our analysis, trigger overthinking in Qwen’s - reasoning rollouts. We then fine-tuned Qwen by penalizing usage of those tokens while it reasons. - - Swift produces shorter reasoning traces. In our testing, we also observe fewer overthinking errors. - - For maximum gains, Swift also includes a transfer component derived from - BottleCap AI's ThinkingCap-Qwen3.6-27B. - - ## Evaluation scope - - > All results below compare the Qwen3.8-27B BF16 base with the same base plus the - > Swift adapter. - - ## Benchmarks - - ... - license: "other" + Swift-Qwen3.8-27B is UkisAI's reasoning-efficient fine-tune of Qwen3.8-27B. + The publisher reports 58.3% fewer thinking tokens with less than 1% quality loss. + This Q4_K_M GGUF includes the F16 vision projector and enables MTP speculative decoding. + The weights use the Swift Open License v1.0. + license: "swift-open-license-1.0" tags: - llm - gguf - reasoning + - vision + - multimodal + - mtp overrides: backend: llama-cpp function: @@ -48,6 +25,7 @@ disable: true known_usecases: - chat + - vision mmproj: llama-cpp/mmproj/Swift-Qwen3.8-27B-Q4_K_M/mmproj-Swift-Qwen3.8-27B-F16.gguf options: - use_jinja:true