mirror of
https://github.com/mudler/LocalAI.git
synced 2026-10-04 20:14:43 -04:00
chore: ⬆️ Update ggml-org/llama.cpp to a4d880fd5c7f88713ded6db9f0111893bd78afa6 (#12345)
* ⬆️ Update ggml-org/llama.cpp Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> * fix(llama-cpp): migrate score batches to the new API The pinned engine removes common_batch_add and the raw batch view. Use common_batch entries and llama_process for score suffix decoding. Read shared-prefix scores from the current common_batch view. Validation: reproduce both compiler errors on the original patch. The patched server context and complete grpc-server translation unit pass g++ -std=c++17 -fsyntax-only with generated protobuf headers. Assisted-by: Codex:gpt-6 --------- Signed-off-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com> Co-authored-by: mudler <2420543+mudler@users.noreply.github.com> Co-authored-by: localai-org-maint-bot <306269227+localai-org-maint-bot@users.noreply.github.com>
This commit is contained in:
2 files changed
+10
-12
No files matched your search
@@ -1,5 +1,5 @@
|
||||
|
||||
LLAMA_VERSION?=4da6337767f973e2b4d0797e5b323d77d8565e4a
|
||||
LLAMA_VERSION?=a4d880fd5c7f88713ded6db9f0111893bd78afa6
|
||||
LLAMA_REPO?=https://github.com/ggerganov/llama.cpp
|
||||
|
||||
CMAKE_ARGS?=
|
||||
|
||||
@@ -97,7 +97,7 @@ index 3b5f6a1..d0e18e6 100644
|
||||
json_schema = json();
|
||||
|
||||
task_prev = std::move(task);
|
||||
@@ -2271,6 +2302,229 @@ private:
|
||||
@@ -2271,6 +2302,227 @@ private:
|
||||
queue_results.send(std::move(res));
|
||||
}
|
||||
|
||||
@@ -121,7 +121,7 @@ index 3b5f6a1..d0e18e6 100644
|
||||
+ // region can straddle ubatch boundaries for long prompts, so this
|
||||
+ // accumulates view by view instead of reading everything when the
|
||||
+ // prompt completes.
|
||||
+ void collect_score_logprobs(server_slot & slot, const llama_batch & batch) {
|
||||
+ void collect_score_logprobs(server_slot & slot, const common_batch & batch) {
|
||||
+ const int32_t n_prompt = slot.task->n_score_prompt;
|
||||
+ const int32_t n_total = slot.task->n_tokens();
|
||||
+ const auto & suffixes = slot.task->score_suffixes;
|
||||
@@ -140,14 +140,14 @@ index 3b5f6a1..d0e18e6 100644
|
||||
+
|
||||
+ const int32_t n_vocab = llama_vocab_n_tokens(vocab);
|
||||
+
|
||||
+ for (int32_t i = 0; i < batch.n_tokens; ++i) {
|
||||
+ if (!batch.logits[i] || batch.seq_id[i][0] != slot.id) {
|
||||
+ for (int32_t i = 0; i < batch.size(); ++i) {
|
||||
+ if (!batch.tokens[i].output || batch.tokens[i].seq_id != slot.id) {
|
||||
+ continue;
|
||||
+ }
|
||||
+
|
||||
+ // the output at position p predicts the task token at index p + 1;
|
||||
+ // score tasks are text-only, so positions equal token indices
|
||||
+ const int32_t target = batch.pos[i] + 1;
|
||||
+ const int32_t target = batch.tokens[i].pos[0] + 1;
|
||||
+ if (target < n_prompt || target > n_total) {
|
||||
+ continue;
|
||||
+ }
|
||||
@@ -249,7 +249,7 @@ index 3b5f6a1..d0e18e6 100644
|
||||
+ next++;
|
||||
+ }
|
||||
+
|
||||
+ llama_batch fb = llama_batch_init(n_tok, 0, 1);
|
||||
+ common_batch fb(ctx_tgt);
|
||||
+
|
||||
+ for (size_t k = 0; k < chunk.size(); ++k) {
|
||||
+ const llama_seq_id seq = seq_base + (llama_seq_id) k;
|
||||
@@ -259,11 +259,11 @@ index 3b5f6a1..d0e18e6 100644
|
||||
+ llama_memory_seq_cp(mem, slot.id, seq, -1, -1);
|
||||
+
|
||||
+ for (size_t j = 0; j < sfx.size(); ++j) {
|
||||
+ common_batch_add(fb, sfx[j], pos0 + (llama_pos) j, { seq }, j + 1 < sfx.size());
|
||||
+ fb.add(sfx[j], pos0 + (llama_pos) j, seq, j + 1 < sfx.size());
|
||||
+ }
|
||||
+ }
|
||||
+
|
||||
+ const int ret = llama_decode(ctx_tgt, fb);
|
||||
+ const int ret = llama_process(ctx_tgt, LLAMA_PROCESS_TYPE_DECODE, fb.get());
|
||||
+
|
||||
+ if (ret == 0) {
|
||||
+ int32_t i = 0;
|
||||
@@ -290,8 +290,6 @@ index 3b5f6a1..d0e18e6 100644
|
||||
+ llama_memory_seq_rm(mem, seq_base + (llama_seq_id) k, -1, -1);
|
||||
+ }
|
||||
+
|
||||
+ llama_batch_free(fb);
|
||||
+
|
||||
+ if (ret != 0) {
|
||||
+ SLT_ERR(slot, "score suffix decode failed, ret = %d\n", ret);
|
||||
+ return false;
|
||||
@@ -476,7 +474,7 @@ index 3b5f6a1..d0e18e6 100644
|
||||
+ // their outputs, not just the one holding the final token
|
||||
+ if (slot.task && slot.task->type == SERVER_TASK_TYPE_SCORE &&
|
||||
+ (slot.state == SLOT_STATE_PROCESSING_PROMPT || slot.state == SLOT_STATE_DONE_PROMPT)) {
|
||||
+ collect_score_logprobs(slot, batch_view);
|
||||
+ collect_score_logprobs(slot, batch.view);
|
||||
+ }
|
||||
+
|
||||
if (!is_inside_view(slot.i_batch)) {
|
||||
|
||||
Reference in new issue
Block a user