diff --git a/backend/cpp/llama-cpp/grpc-server.cpp b/backend/cpp/llama-cpp/grpc-server.cpp index 9a2240881..8b206612c 100644 --- a/backend/cpp/llama-cpp/grpc-server.cpp +++ b/backend/cpp/llama-cpp/grpc-server.cpp @@ -539,8 +539,10 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt // Initialize ctx_shift to false by default (can be overridden by options) params.ctx_shift = false; - // Initialize cache_ram_mib to -1 by default (no limit, can be overridden by options) - params.cache_ram_mib = -1; + // cache_ram_mib keeps llama.cpp's own default (8192 MiB) unless overridden by + // options. It used to be forced to -1 (no limit): since kv_unified and + // cache_idle_slots are on by default, every distinct prompt then leaves its + // slot state in host RAM and the backend grows without bound. // Initialize n_parallel to 1 by default (can be overridden by options) params.n_parallel = 1; // Initialize grpc_servers to empty (can be overridden by options) @@ -656,7 +658,7 @@ static void params_parse(server_context& /*ctx_server*/, const backend::ModelOpt try { params.cache_ram_mib = std::stoi(optval_str); } catch (const std::exception& e) { - // If conversion fails, keep default value (-1) + // If conversion fails, keep the default value } } } else if (!strcmp(optname, "parallel") || !strcmp(optname, "n_parallel")) { diff --git a/docs/content/features/text-generation.md b/docs/content/features/text-generation.md index c28bf9213..19dcf64b0 100644 --- a/docs/content/features/text-generation.md +++ b/docs/content/features/text-generation.md @@ -586,7 +586,7 @@ The `llama.cpp` backend supports additional configuration options that can be sp |--------|------|-------------|---------| | `use_jinja` or `jinja` | boolean | Enable Jinja2 template processing for chat templates. When enabled, the backend uses Jinja2-based chat templates from the model for formatting messages. | `use_jinja:true` | | `context_shift` | boolean | Enable context shifting, which allows the model to dynamically adjust context window usage. | `context_shift:true` | -| `cache_ram` | integer | Size budget in MiB for the **server-side prompt cache** (a host-RAM store of idle slot KV states that's reloaded on a prompt-prefix hit, see [upstream PR #16391](https://github.com/ggml-org/llama.cpp/pull/16391)). Default: `-1` (no limit). `0` disables the prompt cache entirely. Together with `kv_unified` and `cache_idle_slots` this is what makes a repeated system prompt skip prefill on subsequent calls. | `cache_ram:4096` | +| `cache_ram` | integer | Size budget in MiB for the **server-side prompt cache** (a host-RAM store of idle slot KV states that's reloaded on a prompt-prefix hit, see [upstream PR #16391](https://github.com/ggml-org/llama.cpp/pull/16391)). Default: `8192` MiB (llama.cpp default). `-1` removes the limit. `0` disables the prompt cache entirely. Together with `kv_unified` and `cache_idle_slots` this is what makes a repeated system prompt skip prefill on subsequent calls. | `cache_ram:4096` | | `parallel` or `n_parallel` | integer | Enable parallel request processing. When set to a value greater than 1, enables continuous batching for handling multiple requests concurrently. | `parallel:4` | | `grpc_servers` or `rpc_servers` | string | Comma-separated list of gRPC server addresses for distributed inference. Allows distributing workload across multiple llama.cpp workers. | `grpc_servers:localhost:50051,localhost:50052` | | `fit_params` or `fit` | boolean | Enable auto-adjustment of model/context parameters to fit available device memory. Default: `true`. | `fit_params:true` | @@ -642,7 +642,7 @@ Agents, coding assistants, and Anthropic/OpenAI-compatible CLIs typically resend | Setting | Default | Role | |---|---|---| -| `cache_ram:N` | `-1` (no limit) | Allocates the host-side prompt cache. `0` disables it. | +| `cache_ram:N` | `8192` (llama.cpp default) | Allocates the host-side prompt cache. `0` disables it. | | `kv_unified:true` | `true` | Single unified KV buffer (**prerequisite** for idle-slot saving). | | `cache_idle_slots:true` | `true` | Persists the idle slot's KV into the prompt cache on task switch. | @@ -657,6 +657,8 @@ options: Set `cache_ram:0` to opt out of the prompt cache entirely (saves host RAM at the cost of re-prefilling repeated prompts). +`cache_ram:-1` removes the limit. With idle-slot saving on, every distinct prompt then leaves its slot state in host RAM, so a workload with many different prompts (classification, ingestion) grows the backend by roughly the KV size of each prompt until the host runs out of memory. + #### Reference - [llama](https://github.com/ggerganov/llama.cpp)