diff --git a/backend/cpp/llama-cpp/Makefile b/backend/cpp/llama-cpp/Makefile index 5cc250b66..869201031 100644 --- a/backend/cpp/llama-cpp/Makefile +++ b/backend/cpp/llama-cpp/Makefile @@ -1,5 +1,5 @@ -LLAMA_VERSION?=bed0a856606ee4a24a164066f73d2379447033f5 +LLAMA_VERSION?=51ce9c11a6f2dfa895696c0048c4333e8953b728 LLAMA_REPO?=https://github.com/ggerganov/llama.cpp CMAKE_ARGS?= diff --git a/backend/cpp/llama-cpp/patches/0001-add-server-task-type-score.patch b/backend/cpp/llama-cpp/patches/0001-add-server-task-type-score.patch index 0b992cded..f51f70ad7 100644 --- a/backend/cpp/llama-cpp/patches/0001-add-server-task-type-score.patch +++ b/backend/cpp/llama-cpp/patches/0001-add-server-task-type-score.patch @@ -367,7 +367,7 @@ index edb8e2d..9c88feb 100644 // make a checkpoint of the parts of the memory that cannot be rolled back. // checkpoints are created only if: -@@ -3670,13 +3944,46 @@ private: +@@ -3806,16 +4080,49 @@ private: // embedding requires all tokens in the batch to be output; // MTP also wants logits at every prompt position so the // streaming hook can mirror t_h_nextn into ctx_dft. @@ -384,6 +384,9 @@ index edb8e2d..9c88feb 100644 - /* output = */ slot.need_embd(), + /* output = */ slot.need_embd() || need_score_logit, /* is_prompt = */ true); + if (!slot.task->decision.order.empty()) { + batch.set_decision_order(batch.size() - 1, slot.task->decision.order[slot.prompt.n_tokens()]); + } slot.prompt.tokens.push_back(cur_tok); + // score tasks: break at the shared-prompt boundary so the checkpoint diff --git a/backend/cpp/llama-cpp/patches/0002-add-server-task-type-tts.patch b/backend/cpp/llama-cpp/patches/0002-add-server-task-type-tts.patch index 331e1a248..07cdb8901 100644 --- a/backend/cpp/llama-cpp/patches/0002-add-server-task-type-tts.patch +++ b/backend/cpp/llama-cpp/patches/0002-add-server-task-type-tts.patch @@ -580,14 +580,14 @@ index 9c88feb..064bb51 100644 } @@ -4678,6 +4835,8 @@ server_context_meta server_context::get_meta() const { - /* has_inp_image */ impl->chat_params.allow_image, - /* has_inp_audio */ impl->chat_params.allow_audio, - /* has_inp_video */ impl->chat_params.allow_video, -+ /* has_cap_chat */ impl->has_cap_chat(), -+ /* has_cap_tts */ impl->has_cap_tts(), - /* json_ui_settings */ impl->json_ui_settings, - /* slot_n_ctx */ impl->n_ctx_slot(), - /* pooling_type */ llama_pooling_type(impl->ctx_tgt), + /* has_inp_image */ impl->chat_params.allow_image, + /* has_inp_audio */ impl->chat_params.allow_audio, + /* has_inp_video */ impl->chat_params.allow_video, ++ /* has_cap_chat */ impl->has_cap_chat(), ++ /* has_cap_tts */ impl->has_cap_tts(), + /* json_ui_settings */ impl->json_ui_settings, + /* slot_n_ctx */ impl->n_ctx_slot(), + /* pooling_type */ llama_pooling_type(impl->ctx_tgt), @@ -4751,6 +4910,11 @@ std::unique_ptr server_routes::handle_completions_impl( res->set_req(&req); // will also set spipe if needed