From e87ac2d32db82d51c900c46d186a4472aace6800 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Thu, 8 Oct 2026 16:40:43 +0200 Subject: [PATCH] chore: :arrow_up: Update ggml-org/llama.cpp to `51ce9c11a6f2dfa895696c0048c4333e8953b728` (#12461) * :arrow_up: Update ggml-org/llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> * fix(llama-cpp): refresh decision patch contexts Preserve the upstream decision-order assignment when adding score logits. Refresh the TTS metadata context after upstream realigns its comments. Both patches apply to the new pin, and the gRPC source compiles. Assisted-by: Codex:gpt-6 --------- Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- backend/cpp/llama-cpp/Makefile | 2 +- .../0001-add-server-task-type-score.patch | 5 ++++- .../patches/0002-add-server-task-type-tts.patch | 16 ++++++++-------- 3 files changed, 13 insertions(+), 10 deletions(-) diff --git a/backend/cpp/llama-cpp/Makefile b/backend/cpp/llama-cpp/Makefile index 5cc250b66..869201031 100644 --- a/backend/cpp/llama-cpp/Makefile +++ b/backend/cpp/llama-cpp/Makefile @@ -1,5 +1,5 @@ -LLAMA_VERSION?=bed0a856606ee4a24a164066f73d2379447033f5 +LLAMA_VERSION?=51ce9c11a6f2dfa895696c0048c4333e8953b728 LLAMA_REPO?=https://github.com/ggerganov/llama.cpp CMAKE_ARGS?= diff --git a/backend/cpp/llama-cpp/patches/0001-add-server-task-type-score.patch b/backend/cpp/llama-cpp/patches/0001-add-server-task-type-score.patch index 0b992cded..f51f70ad7 100644 --- a/backend/cpp/llama-cpp/patches/0001-add-server-task-type-score.patch +++ b/backend/cpp/llama-cpp/patches/0001-add-server-task-type-score.patch @@ -367,7 +367,7 @@ index edb8e2d..9c88feb 100644 // make a checkpoint of the parts of the memory that cannot be rolled back. // checkpoints are created only if: -@@ -3670,13 +3944,46 @@ private: +@@ -3806,16 +4080,49 @@ private: // embedding requires all tokens in the batch to be output; // MTP also wants logits at every prompt position so the // streaming hook can mirror t_h_nextn into ctx_dft. @@ -384,6 +384,9 @@ index edb8e2d..9c88feb 100644 - /* output = */ slot.need_embd(), + /* output = */ slot.need_embd() || need_score_logit, /* is_prompt = */ true); + if (!slot.task->decision.order.empty()) { + batch.set_decision_order(batch.size() - 1, slot.task->decision.order[slot.prompt.n_tokens()]); + } slot.prompt.tokens.push_back(cur_tok); + // score tasks: break at the shared-prompt boundary so the checkpoint diff --git a/backend/cpp/llama-cpp/patches/0002-add-server-task-type-tts.patch b/backend/cpp/llama-cpp/patches/0002-add-server-task-type-tts.patch index 331e1a248..07cdb8901 100644 --- a/backend/cpp/llama-cpp/patches/0002-add-server-task-type-tts.patch +++ b/backend/cpp/llama-cpp/patches/0002-add-server-task-type-tts.patch @@ -580,14 +580,14 @@ index 9c88feb..064bb51 100644 } @@ -4678,6 +4835,8 @@ server_context_meta server_context::get_meta() const { - /* has_inp_image */ impl->chat_params.allow_image, - /* has_inp_audio */ impl->chat_params.allow_audio, - /* has_inp_video */ impl->chat_params.allow_video, -+ /* has_cap_chat */ impl->has_cap_chat(), -+ /* has_cap_tts */ impl->has_cap_tts(), - /* json_ui_settings */ impl->json_ui_settings, - /* slot_n_ctx */ impl->n_ctx_slot(), - /* pooling_type */ llama_pooling_type(impl->ctx_tgt), + /* has_inp_image */ impl->chat_params.allow_image, + /* has_inp_audio */ impl->chat_params.allow_audio, + /* has_inp_video */ impl->chat_params.allow_video, ++ /* has_cap_chat */ impl->has_cap_chat(), ++ /* has_cap_tts */ impl->has_cap_tts(), + /* json_ui_settings */ impl->json_ui_settings, + /* slot_n_ctx */ impl->n_ctx_slot(), + /* pooling_type */ llama_pooling_type(impl->ctx_tgt), @@ -4751,6 +4910,11 @@ std::unique_ptr server_routes::handle_completions_impl( res->set_req(&req); // will also set spipe if needed