chore: ⬆️ Update ggml-org/llama.cpp to 51ce9c11a6f2dfa895696c0048c4333e8953b728 (#12461)

* ⬆️ Update ggml-org/llama.cpp

Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>

* fix(llama-cpp): refresh decision patch contexts

Preserve the upstream decision-order assignment when adding score logits.
Refresh the TTS metadata context after upstream realigns its comments.
Both patches apply to the new pin, and the gRPC source compiles.

Assisted-by: Codex:gpt-6

---------

Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
Co-authored-by: mudler <2420543+mudler@users.noreply.github.com>
Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com>
This commit is contained in:
authored and GitHub committed 2026-10-08 16:40:43 +02:00
1 parent 6ad8cc0640
commit e87ac2d32d
3 files changed
+13 -10

No files matched your search

+1 -1
View File
@@ -1,5 +1,5 @@
LLAMA_VERSION?=bed0a856606ee4a24a164066f73d2379447033f5
LLAMA_VERSION?=51ce9c11a6f2dfa895696c0048c4333e8953b728
LLAMA_REPO?=https://github.com/ggerganov/llama.cpp
CMAKE_ARGS?=
@@ -367,7 +367,7 @@ index edb8e2d..9c88feb 100644
// make a checkpoint of the parts of the memory that cannot be rolled back.
// checkpoints are created only if:
@@ -3670,13 +3944,46 @@ private:
@@ -3806,16 +4080,49 @@ private:
// embedding requires all tokens in the batch to be output;
// MTP also wants logits at every prompt position so the
// streaming hook can mirror t_h_nextn into ctx_dft.
@@ -384,6 +384,9 @@ index edb8e2d..9c88feb 100644
- /* output = */ slot.need_embd(),
+ /* output = */ slot.need_embd() || need_score_logit,
/* is_prompt = */ true);
if (!slot.task->decision.order.empty()) {
batch.set_decision_order(batch.size() - 1, slot.task->decision.order[slot.prompt.n_tokens()]);
}
slot.prompt.tokens.push_back(cur_tok);
+ // score tasks: break at the shared-prompt boundary so the checkpoint
@@ -580,14 +580,14 @@ index 9c88feb..064bb51 100644
}
@@ -4678,6 +4835,8 @@ server_context_meta server_context::get_meta() const {
/* has_inp_image */ impl->chat_params.allow_image,
/* has_inp_audio */ impl->chat_params.allow_audio,
/* has_inp_video */ impl->chat_params.allow_video,
+ /* has_cap_chat */ impl->has_cap_chat(),
+ /* has_cap_tts */ impl->has_cap_tts(),
/* json_ui_settings */ impl->json_ui_settings,
/* slot_n_ctx */ impl->n_ctx_slot(),
/* pooling_type */ llama_pooling_type(impl->ctx_tgt),
/* has_inp_image */ impl->chat_params.allow_image,
/* has_inp_audio */ impl->chat_params.allow_audio,
/* has_inp_video */ impl->chat_params.allow_video,
+ /* has_cap_chat */ impl->has_cap_chat(),
+ /* has_cap_tts */ impl->has_cap_tts(),
/* json_ui_settings */ impl->json_ui_settings,
/* slot_n_ctx */ impl->n_ctx_slot(),
/* pooling_type */ llama_pooling_type(impl->ctx_tgt),
@@ -4751,6 +4910,11 @@ std::unique_ptr<server_res_generator> server_routes::handle_completions_impl(
res->set_req(&req); // will also set spipe if needed