From b9c975e71c4f60408e64affb6cb8178981cbf638 Mon Sep 17 00:00:00 2001 From: localai-org-maint-bot Date: Thu, 1 Oct 2026 08:40:58 +0200 Subject: [PATCH] chore: :arrow_up: Update ggml-org/llama.cpp to `a4d880fd5c7f88713ded6db9f0111893bd78afa6` (#12345) * :arrow_up: Update ggml-org/llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> * fix(llama-cpp): migrate score batches to the new API The pinned engine removes common_batch_add and the raw batch view. Use common_batch entries and llama_process for score suffix decoding. Read shared-prefix scores from the current common_batch view. Validation: reproduce both compiler errors on the original patch. The patched server context and complete grpc-server translation unit pass g++ -std=c++17 -fsyntax-only with generated protobuf headers. Assisted-by: Codex:gpt-6 --------- Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com> --- backend/cpp/llama-cpp/Makefile | 2 +- .../0001-add-server-task-type-score.patch | 20 +++++++++---------- 2 files changed, 10 insertions(+), 12 deletions(-) diff --git a/backend/cpp/llama-cpp/Makefile b/backend/cpp/llama-cpp/Makefile index af5529d48..19b3e711c 100644 --- a/backend/cpp/llama-cpp/Makefile +++ b/backend/cpp/llama-cpp/Makefile @@ -1,5 +1,5 @@ -LLAMA_VERSION?=4da6337767f973e2b4d0797e5b323d77d8565e4a +LLAMA_VERSION?=a4d880fd5c7f88713ded6db9f0111893bd78afa6 LLAMA_REPO?=https://github.com/ggerganov/llama.cpp CMAKE_ARGS?= diff --git a/backend/cpp/llama-cpp/patches/0001-add-server-task-type-score.patch b/backend/cpp/llama-cpp/patches/0001-add-server-task-type-score.patch index 253e8da5f..f2a34a149 100644 --- a/backend/cpp/llama-cpp/patches/0001-add-server-task-type-score.patch +++ b/backend/cpp/llama-cpp/patches/0001-add-server-task-type-score.patch @@ -97,7 +97,7 @@ index 3b5f6a1..d0e18e6 100644 json_schema = json(); task_prev = std::move(task); -@@ -2271,6 +2302,229 @@ private: +@@ -2271,6 +2302,227 @@ private: queue_results.send(std::move(res)); } @@ -121,7 +121,7 @@ index 3b5f6a1..d0e18e6 100644 + // region can straddle ubatch boundaries for long prompts, so this + // accumulates view by view instead of reading everything when the + // prompt completes. -+ void collect_score_logprobs(server_slot & slot, const llama_batch & batch) { ++ void collect_score_logprobs(server_slot & slot, const common_batch & batch) { + const int32_t n_prompt = slot.task->n_score_prompt; + const int32_t n_total = slot.task->n_tokens(); + const auto & suffixes = slot.task->score_suffixes; @@ -140,14 +140,14 @@ index 3b5f6a1..d0e18e6 100644 + + const int32_t n_vocab = llama_vocab_n_tokens(vocab); + -+ for (int32_t i = 0; i < batch.n_tokens; ++i) { -+ if (!batch.logits[i] || batch.seq_id[i][0] != slot.id) { ++ for (int32_t i = 0; i < batch.size(); ++i) { ++ if (!batch.tokens[i].output || batch.tokens[i].seq_id != slot.id) { + continue; + } + + // the output at position p predicts the task token at index p + 1; + // score tasks are text-only, so positions equal token indices -+ const int32_t target = batch.pos[i] + 1; ++ const int32_t target = batch.tokens[i].pos[0] + 1; + if (target < n_prompt || target > n_total) { + continue; + } @@ -249,7 +249,7 @@ index 3b5f6a1..d0e18e6 100644 + next++; + } + -+ llama_batch fb = llama_batch_init(n_tok, 0, 1); ++ common_batch fb(ctx_tgt); + + for (size_t k = 0; k < chunk.size(); ++k) { + const llama_seq_id seq = seq_base + (llama_seq_id) k; @@ -259,11 +259,11 @@ index 3b5f6a1..d0e18e6 100644 + llama_memory_seq_cp(mem, slot.id, seq, -1, -1); + + for (size_t j = 0; j < sfx.size(); ++j) { -+ common_batch_add(fb, sfx[j], pos0 + (llama_pos) j, { seq }, j + 1 < sfx.size()); ++ fb.add(sfx[j], pos0 + (llama_pos) j, seq, j + 1 < sfx.size()); + } + } + -+ const int ret = llama_decode(ctx_tgt, fb); ++ const int ret = llama_process(ctx_tgt, LLAMA_PROCESS_TYPE_DECODE, fb.get()); + + if (ret == 0) { + int32_t i = 0; @@ -290,8 +290,6 @@ index 3b5f6a1..d0e18e6 100644 + llama_memory_seq_rm(mem, seq_base + (llama_seq_id) k, -1, -1); + } + -+ llama_batch_free(fb); -+ + if (ret != 0) { + SLT_ERR(slot, "score suffix decode failed, ret = %d\n", ret); + return false; @@ -476,7 +474,7 @@ index 3b5f6a1..d0e18e6 100644 + // their outputs, not just the one holding the final token + if (slot.task && slot.task->type == SERVER_TASK_TYPE_SCORE && + (slot.state == SLOT_STATE_PROCESSING_PROMPT || slot.state == SLOT_STATE_DONE_PROMPT)) { -+ collect_score_logprobs(slot, batch_view); ++ collect_score_logprobs(slot, batch.view); + } + if (!is_inside_view(slot.i_batch)) {