From 9be1d52953e1bc117212b762578beeeaee9a5085 Mon Sep 17 00:00:00 2001 From: Colin Kealty <3266127+bartowski1182@users.noreply.github.com> Date: Thu, 4 Jun 2026 02:06:10 -0400 Subject: [PATCH 1/5] Add ctx-per-slot argument for unifid KV cache --- common/arg.cpp | 39 +++++++++++++++++++++++---------- common/common.h | 2 ++ tools/server/server-context.cpp | 14 ++++++++++++ tools/server/server.cpp | 24 ++++++++++++++++++++ 4 files changed, 68 insertions(+), 11 deletions(-) diff --git a/common/arg.cpp b/common/arg.cpp index 24d9734b934e..7fc3f72129c5 100644 --- a/common/arg.cpp +++ b/common/arg.cpp @@ -1285,17 +1285,34 @@ common_params_context common_params_parser_init(common_params & params, llama_ex } } ).set_env("LLAMA_ARG_CTX_SIZE")); - add_opt(common_arg( - {"-n", "--predict", "--n-predict"}, "N", - string_format( - ex == LLAMA_EXAMPLE_COMPLETION - ? "number of tokens to predict (default: %d, -1 = infinity, -2 = until context filled)" - : "number of tokens to predict (default: %d, -1 = infinity)", - params.n_predict), - [](common_params & params, int value) { - params.n_predict = value; - } - ).set_env("LLAMA_ARG_N_PREDICT")); + add_opt( + common_arg({ "--ctx-per-slot" }, "N", + "max context per parallel slot in unified KV mode (default: unset, behavior unchanged).\n" + "when set and -c/--ctx-size is not given, the shared KV pool is sized to n_parallel*N (fit-clamped)", + [](common_params & params, const std::string & value) { params.n_ctx_per_slot = std::stoi(value); }) + .set_env("LLAMA_ARG_CTX_PER_SLOT") + .set_examples({ LLAMA_EXAMPLE_SERVER })); + add_opt(common_arg({ "--ctx-pool-frac" }, "N", + string_format("fraction (0 < N <= 1) of the --ctx-per-slot pool to allocate (default: %.2f).\n" + "with unified KV, slots share the pool, so < 1.0 overcommits (allocate for the " + "expected, not worst, case)", + (double) params.ctx_pool_frac), + [](common_params & params, const std::string & value) { + params.ctx_pool_frac = std::stof(value); + if (params.ctx_pool_frac <= 0.0f || params.ctx_pool_frac > 1.0f) { + throw std::invalid_argument("error: --ctx-pool-frac must be in the range (0, 1]\n"); + } + }) + .set_env("LLAMA_ARG_CTX_POOL_FRAC") + .set_examples({ LLAMA_EXAMPLE_SERVER })); + add_opt(common_arg({ "-n", "--predict", "--n-predict" }, "N", + string_format( + ex == LLAMA_EXAMPLE_COMPLETION ? + "number of tokens to predict (default: %d, -1 = infinity, -2 = until context filled)" : + "number of tokens to predict (default: %d, -1 = infinity)", + params.n_predict), + [](common_params & params, int value) { params.n_predict = value; }) + .set_env("LLAMA_ARG_N_PREDICT")); add_opt(common_arg( {"-b", "--batch-size"}, "N", string_format("logical maximum batch size (default: %d)", params.n_batch), diff --git a/common/common.h b/common/common.h index dec90456afab..494b5cd4f63f 100644 --- a/common/common.h +++ b/common/common.h @@ -595,6 +595,8 @@ struct common_params { bool cache_idle_slots = true; // save and clear idle slots upon starting a new task int32_t n_ctx_checkpoints = 32; // max number of context checkpoints per slot int32_t checkpoint_every_nt = 8192; // make a checkpoint every n tokens during prefill + int32_t n_ctx_per_slot = 0; // max context per parallel slot; 0 = unset + float ctx_pool_frac = 1.0f; // fraction of the --ctx-per-slot pool to allocate int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc. std::string hostname = "127.0.0.1"; diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp index f517310266c0..1de71c73d00b 100644 --- a/tools/server/server-context.cpp +++ b/tools/server/server-context.cpp @@ -923,6 +923,20 @@ struct server_context_impl { const int n_ctx_train = llama_model_n_ctx_train(model_tgt); int n_ctx_slot = llama_n_ctx_seq(ctx_tgt); + if (params_base.n_ctx_per_slot > 0) { + if (n_ctx_slot > params_base.n_ctx_per_slot) { + SRV_INF("capping per-slot context (%d) to --ctx-per-slot (%d)\n", n_ctx_slot, + params_base.n_ctx_per_slot); + n_ctx_slot = params_base.n_ctx_per_slot; + } else if (params_base.n_ctx_per_slot > n_ctx_slot) { + // cap is above the per-slot pool capacity, so it can never bind + SRV_WRN( + "--ctx-per-slot (%d) exceeds the per-slot pool capacity (%d) - cap has no effect, " + "slots are limited to %d (raise the KV pool with -c, or unset -c to size it to " + "n_parallel*ctx_per_slot)\n", + params_base.n_ctx_per_slot, n_ctx_slot, n_ctx_slot); + } + } if (n_ctx_slot > n_ctx_train) { SRV_WRN("the slot context (%d) exceeds the training context of the model (%d) - capping\n", n_ctx_slot, n_ctx_train); n_ctx_slot = n_ctx_train; diff --git a/tools/server/server.cpp b/tools/server/server.cpp index 4d56d45e83cc..cce1f7373212 100644 --- a/tools/server/server.cpp +++ b/tools/server/server.cpp @@ -110,6 +110,30 @@ int llama_server(int argc, char ** argv) { params.kv_unified = true; } + // size the KV pool from --ctx-per-slot, unless the user pinned it with -c (-c 0 sets + // fit_params_min_ctx to UINT32_MAX) + const bool ctx_pool_auto_sized = + params.n_ctx_per_slot > 0 && params.n_ctx == 0 && (uint32_t) params.fit_params_min_ctx != UINT32_MAX; + + // ctx_pool_frac overcommits the pool; only meaningful with an auto-sized, unified pool + if (params.ctx_pool_frac != 1.0f) { + if (!ctx_pool_auto_sized) { + SRV_WRN("%s", "--ctx-pool-frac requires --ctx-per-slot and no explicit -c, disabling\n"); + params.ctx_pool_frac = 1.0f; + } else if (!params.kv_unified) { + SRV_WRN("%s", + "--ctx-pool-frac requires --kv-unified (non-unified KV partitions the pool " + "per slot, so the fraction would just shrink each slot), disabling\n"); + params.ctx_pool_frac = 1.0f; + } + } + + if (ctx_pool_auto_sized) { + params.n_ctx = (int32_t) (params.ctx_pool_frac * params.n_parallel * params.n_ctx_per_slot); + SRV_INF("--ctx-per-slot: sizing KV pool to ctx_pool_frac*n_parallel*ctx_per_slot = %.2f*%d*%d = %d\n", + params.ctx_pool_frac, params.n_parallel, params.n_ctx_per_slot, params.n_ctx); + } + // for consistency between server router mode and single-model mode, we set the same model name as alias if (params.model_alias.empty() && !params.model.name.empty()) { params.model_alias.insert(params.model.name); From b2d7dc68118e12311e8ac038bf143b7827a42161 Mon Sep 17 00:00:00 2001 From: Colin Kealty <3266127+bartowski1182@users.noreply.github.com> Date: Thu, 4 Jun 2026 09:41:17 -0400 Subject: [PATCH 2/5] Swap out ctx fractions for ctx pool slots --- common/arg.cpp | 24 ++++++++++++------------ common/common.h | 3 +-- tools/server/README.md | 2 ++ tools/server/server.cpp | 25 +++++++++++++++---------- 4 files changed, 30 insertions(+), 24 deletions(-) diff --git a/common/arg.cpp b/common/arg.cpp index de7ec2002114..7cf4c6920686 100644 --- a/common/arg.cpp +++ b/common/arg.cpp @@ -1279,23 +1279,23 @@ common_params_context common_params_parser_init(common_params & params, llama_ex ).set_env("LLAMA_ARG_CTX_SIZE")); add_opt( common_arg({ "--ctx-per-slot" }, "N", - "max context per parallel slot in unified KV mode (default: unset, behavior unchanged).\n" - "when set and -c/--ctx-size is not given, the shared KV pool is sized to n_parallel*N (fit-clamped)", + "context limit per parallel slot (default: unset, behavior unchanged).\n" + "when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N", [](common_params & params, const std::string & value) { params.n_ctx_per_slot = std::stoi(value); }) .set_env("LLAMA_ARG_CTX_PER_SLOT") .set_examples({ LLAMA_EXAMPLE_SERVER })); - add_opt(common_arg({ "--ctx-pool-frac" }, "N", - string_format("fraction (0 < N <= 1) of the --ctx-per-slot pool to allocate (default: %.2f).\n" - "with unified KV, slots share the pool, so < 1.0 overcommits (allocate for the " - "expected, not worst, case)", - (double) params.ctx_pool_frac), - [](common_params & params, const std::string & value) { - params.ctx_pool_frac = std::stof(value); - if (params.ctx_pool_frac <= 0.0f || params.ctx_pool_frac > 1.0f) { - throw std::invalid_argument("error: --ctx-pool-frac must be in the range (0, 1]\n"); + add_opt(common_arg({ "--ctx-pool-slots" }, "N", + "slots worth of context to provision in the shared --ctx-per-slot pool " + "(default: all n_parallel slots).\n" + "with unified KV, values below n_parallel overcommit the pool (allocate for the " + "expected, not worst, case)", + [](common_params & params, int value) { + if (value < 1) { + throw std::invalid_argument("error: --ctx-pool-slots must be >= 1\n"); } + params.ctx_pool_slots = value; }) - .set_env("LLAMA_ARG_CTX_POOL_FRAC") + .set_env("LLAMA_ARG_CTX_POOL_SLOTS") .set_examples({ LLAMA_EXAMPLE_SERVER })); add_opt(common_arg({ "-n", "--predict", "--n-predict" }, "N", string_format( diff --git a/common/common.h b/common/common.h index 0d63859b79cb..8763299da491 100644 --- a/common/common.h +++ b/common/common.h @@ -598,9 +598,8 @@ struct common_params { bool cache_prompt = true; // whether to enable prompt caching bool cache_idle_slots = true; // save and clear idle slots upon starting a new task int32_t n_ctx_checkpoints = 32; // max number of context checkpoints per slot - int32_t checkpoint_every_nt = 8192; // make a checkpoint every n tokens during prefill int32_t n_ctx_per_slot = 0; // max context per parallel slot; 0 = unset - float ctx_pool_frac = 1.0f; // fraction of the --ctx-per-slot pool to allocate + int32_t ctx_pool_slots = 0; // slots worth of context in the pool; 0 = all n_parallel slots int32_t checkpoint_min_step = 256; // minimum spacing between context checkpoints int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc. diff --git a/tools/server/README.md b/tools/server/README.md index f1eeec36aa0c..dad7d9d31845 100644 --- a/tools/server/README.md +++ b/tools/server/README.md @@ -48,6 +48,8 @@ For the full list of features, please refer to [server's changelog](https://gith | `--prio-batch N` | set process/thread priority : 0-normal, 1-medium, 2-high, 3-realtime (default: 0) | | `--poll-batch <0\|1>` | use polling to wait for work (default: same as --poll) | | `-c, --ctx-size N` | size of the prompt context (default: 0, 0 = loaded from model)
(env: LLAMA_ARG_CTX_SIZE) | +| `--ctx-per-slot N` | context limit per parallel slot (default: unset, behavior unchanged).
when set without `-c`/`--ctx-size`, the shared KV pool is sized to `n_parallel*N`
(env: LLAMA_ARG_CTX_PER_SLOT) | +| `--ctx-pool-slots N` | slots worth of context to provision in the shared `--ctx-per-slot` pool (default: all `n_parallel` slots).
with unified KV, values below `n_parallel` overcommit the pool (allocate for the expected, not worst, case)
(env: LLAMA_ARG_CTX_POOL_SLOTS) | | `-n, --predict, --n-predict N` | number of tokens to predict (default: -1, -1 = infinity)
(env: LLAMA_ARG_N_PREDICT) | | `-b, --batch-size N` | logical maximum batch size (default: 2048)
(env: LLAMA_ARG_BATCH) | | `-ub, --ubatch-size N` | physical maximum batch size (default: 512)
(env: LLAMA_ARG_UBATCH) | diff --git a/tools/server/server.cpp b/tools/server/server.cpp index c6a4f16f8cd3..2f4324e8d1bb 100644 --- a/tools/server/server.cpp +++ b/tools/server/server.cpp @@ -115,23 +115,28 @@ int llama_server(int argc, char ** argv) { const bool ctx_pool_auto_sized = params.n_ctx_per_slot > 0 && params.n_ctx == 0 && (uint32_t) params.fit_params_min_ctx != UINT32_MAX; - // ctx_pool_frac overcommits the pool; only meaningful with an auto-sized, unified pool - if (params.ctx_pool_frac != 1.0f) { + // ctx_pool_slots overcommits the pool; only meaningful with an auto-sized, unified pool + if (params.ctx_pool_slots > 0) { if (!ctx_pool_auto_sized) { - SRV_WRN("%s", "--ctx-pool-frac requires --ctx-per-slot and no explicit -c, disabling\n"); - params.ctx_pool_frac = 1.0f; + SRV_WRN("%s", "--ctx-pool-slots requires --ctx-per-slot and no explicit -c, disabling\n"); + params.ctx_pool_slots = 0; } else if (!params.kv_unified) { SRV_WRN("%s", - "--ctx-pool-frac requires --kv-unified (non-unified KV partitions the pool " - "per slot, so the fraction would just shrink each slot), disabling\n"); - params.ctx_pool_frac = 1.0f; + "--ctx-pool-slots requires --kv-unified (non-unified KV gives each slot its " + "own partition), disabling\n"); + params.ctx_pool_slots = 0; + } else if (params.ctx_pool_slots > params.n_parallel) { + SRV_WRN("--ctx-pool-slots (%d) exceeds n_parallel (%d), clamping\n", params.ctx_pool_slots, + params.n_parallel); + params.ctx_pool_slots = params.n_parallel; } } if (ctx_pool_auto_sized) { - params.n_ctx = (int32_t) (params.ctx_pool_frac * params.n_parallel * params.n_ctx_per_slot); - SRV_INF("--ctx-per-slot: sizing KV pool to ctx_pool_frac*n_parallel*ctx_per_slot = %.2f*%d*%d = %d\n", - params.ctx_pool_frac, params.n_parallel, params.n_ctx_per_slot, params.n_ctx); + const int32_t pool_slots = params.ctx_pool_slots > 0 ? params.ctx_pool_slots : params.n_parallel; + params.n_ctx = pool_slots * params.n_ctx_per_slot; + SRV_INF("--ctx-per-slot: sizing KV pool to pool_slots*ctx_per_slot = %d*%d = %d\n", pool_slots, + params.n_ctx_per_slot, params.n_ctx); } // for consistency between server router mode and single-model mode, we set the same model name as alias From 6386441f63eaafd5b5a7f495efe91370f4794425 Mon Sep 17 00:00:00 2001 From: Colin Kealty <3266127+bartowski1182@users.noreply.github.com> Date: Thu, 4 Jun 2026 09:55:24 -0400 Subject: [PATCH 3/5] Formatting cleanup --- common/arg.cpp | 59 +++++++++++++++++---------------- tools/server/README.md | 4 +-- tools/server/server-context.cpp | 6 ++-- tools/server/server.cpp | 11 +++--- 4 files changed, 42 insertions(+), 38 deletions(-) diff --git a/common/arg.cpp b/common/arg.cpp index 7cf4c6920686..0ff69423ced6 100644 --- a/common/arg.cpp +++ b/common/arg.cpp @@ -1277,34 +1277,37 @@ common_params_context common_params_parser_init(common_params & params, llama_ex } } ).set_env("LLAMA_ARG_CTX_SIZE")); - add_opt( - common_arg({ "--ctx-per-slot" }, "N", - "context limit per parallel slot (default: unset, behavior unchanged).\n" - "when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N", - [](common_params & params, const std::string & value) { params.n_ctx_per_slot = std::stoi(value); }) - .set_env("LLAMA_ARG_CTX_PER_SLOT") - .set_examples({ LLAMA_EXAMPLE_SERVER })); - add_opt(common_arg({ "--ctx-pool-slots" }, "N", - "slots worth of context to provision in the shared --ctx-per-slot pool " - "(default: all n_parallel slots).\n" - "with unified KV, values below n_parallel overcommit the pool (allocate for the " - "expected, not worst, case)", - [](common_params & params, int value) { - if (value < 1) { - throw std::invalid_argument("error: --ctx-pool-slots must be >= 1\n"); - } - params.ctx_pool_slots = value; - }) - .set_env("LLAMA_ARG_CTX_POOL_SLOTS") - .set_examples({ LLAMA_EXAMPLE_SERVER })); - add_opt(common_arg({ "-n", "--predict", "--n-predict" }, "N", - string_format( - ex == LLAMA_EXAMPLE_COMPLETION ? - "number of tokens to predict (default: %d, -1 = infinity, -2 = until context filled)" : - "number of tokens to predict (default: %d, -1 = infinity)", - params.n_predict), - [](common_params & params, int value) { params.n_predict = value; }) - .set_env("LLAMA_ARG_N_PREDICT")); + add_opt(common_arg( + { "--ctx-per-slot" }, "N", + "context limit per parallel slot (default: unset, behavior unchanged).\n" + "when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N", + [](common_params & params, const std::string & value) { + params.n_ctx_per_slot = std::stoi(value); + } + ).set_env("LLAMA_ARG_CTX_PER_SLOT").set_examples({ LLAMA_EXAMPLE_SERVER })); + add_opt(common_arg( + { "--ctx-pool-slots" }, "N", + "slots worth of context to provision in the shared --ctx-per-slot pool " + "(default: all n_parallel slots).\n" + "with unified KV, values below n_parallel overcommit the pool", + [](common_params & params, int value) { + if (value < 1) { + throw std::invalid_argument("error: --ctx-pool-slots must be >= 1\n"); + } + params.ctx_pool_slots = value; + } + ).set_env("LLAMA_ARG_CTX_POOL_SLOTS").set_examples({ LLAMA_EXAMPLE_SERVER })); + add_opt(common_arg( + {"-n", "--predict", "--n-predict"}, "N", + string_format( + ex == LLAMA_EXAMPLE_COMPLETION + ? "number of tokens to predict (default: %d, -1 = infinity, -2 = until context filled)" + : "number of tokens to predict (default: %d, -1 = infinity)", + params.n_predict), + [](common_params & params, int value) { + params.n_predict = value; + } + ).set_env("LLAMA_ARG_N_PREDICT")); add_opt(common_arg( {"-b", "--batch-size"}, "N", string_format("logical maximum batch size (default: %d)", params.n_batch), diff --git a/tools/server/README.md b/tools/server/README.md index dad7d9d31845..aa8ec2f348f2 100644 --- a/tools/server/README.md +++ b/tools/server/README.md @@ -48,8 +48,8 @@ For the full list of features, please refer to [server's changelog](https://gith | `--prio-batch N` | set process/thread priority : 0-normal, 1-medium, 2-high, 3-realtime (default: 0) | | `--poll-batch <0\|1>` | use polling to wait for work (default: same as --poll) | | `-c, --ctx-size N` | size of the prompt context (default: 0, 0 = loaded from model)
(env: LLAMA_ARG_CTX_SIZE) | -| `--ctx-per-slot N` | context limit per parallel slot (default: unset, behavior unchanged).
when set without `-c`/`--ctx-size`, the shared KV pool is sized to `n_parallel*N`
(env: LLAMA_ARG_CTX_PER_SLOT) | -| `--ctx-pool-slots N` | slots worth of context to provision in the shared `--ctx-per-slot` pool (default: all `n_parallel` slots).
with unified KV, values below `n_parallel` overcommit the pool (allocate for the expected, not worst, case)
(env: LLAMA_ARG_CTX_POOL_SLOTS) | +| `--ctx-per-slot N` | context limit per parallel slot (default: unset, behavior unchanged).
when set without `-c`/`--ctx-size`, the shared KV pool is sized to `n_parallel * N`
(env: LLAMA_ARG_CTX_PER_SLOT) | +| `--ctx-pool-slots N` | slots worth of context to provision in the shared `--ctx-per-slot` pool (default: all `n_parallel` slots).
with unified KV, values below `n_parallel` overcommit the pool
(env: LLAMA_ARG_CTX_POOL_SLOTS) | | `-n, --predict, --n-predict N` | number of tokens to predict (default: -1, -1 = infinity)
(env: LLAMA_ARG_N_PREDICT) | | `-b, --batch-size N` | logical maximum batch size (default: 2048)
(env: LLAMA_ARG_BATCH) | | `-ub, --ubatch-size N` | physical maximum batch size (default: 512)
(env: LLAMA_ARG_UBATCH) | diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp index 2db17cee1e60..9ee12637bd8a 100644 --- a/tools/server/server-context.cpp +++ b/tools/server/server-context.cpp @@ -1032,15 +1032,15 @@ struct server_context_impl { int n_ctx_slot = llama_n_ctx_seq(ctx_tgt); if (params_base.n_ctx_per_slot > 0) { if (n_ctx_slot > params_base.n_ctx_per_slot) { - SRV_INF("capping per-slot context (%d) to --ctx-per-slot (%d)\n", n_ctx_slot, - params_base.n_ctx_per_slot); + SRV_INF("capping per-slot context (%d) to --ctx-per-slot (%d)\n", + n_ctx_slot, params_base.n_ctx_per_slot); n_ctx_slot = params_base.n_ctx_per_slot; } else if (params_base.n_ctx_per_slot > n_ctx_slot) { // cap is above the per-slot pool capacity, so it can never bind SRV_WRN( "--ctx-per-slot (%d) exceeds the per-slot pool capacity (%d) - cap has no effect, " "slots are limited to %d (raise the KV pool with -c, or unset -c to size it to " - "n_parallel*ctx_per_slot)\n", + "n_parallel * ctx_per_slot)\n", params_base.n_ctx_per_slot, n_ctx_slot, n_ctx_slot); } } diff --git a/tools/server/server.cpp b/tools/server/server.cpp index 2f4324e8d1bb..acd07de1c213 100644 --- a/tools/server/server.cpp +++ b/tools/server/server.cpp @@ -110,10 +110,11 @@ int llama_server(int argc, char ** argv) { params.kv_unified = true; } - // size the KV pool from --ctx-per-slot, unless the user pinned it with -c (-c 0 sets - // fit_params_min_ctx to UINT32_MAX) - const bool ctx_pool_auto_sized = - params.n_ctx_per_slot > 0 && params.n_ctx == 0 && (uint32_t) params.fit_params_min_ctx != UINT32_MAX; + // size the KV pool from --ctx-per-slot, unless the user pinned it with -c + // or with -c 0 for max context + const bool ctx_pool_auto_sized = params.n_ctx_per_slot > 0 && + params.n_ctx == 0 && + (uint32_t) params.fit_params_min_ctx != UINT32_MAX; // ctx_pool_slots overcommits the pool; only meaningful with an auto-sized, unified pool if (params.ctx_pool_slots > 0) { @@ -135,7 +136,7 @@ int llama_server(int argc, char ** argv) { if (ctx_pool_auto_sized) { const int32_t pool_slots = params.ctx_pool_slots > 0 ? params.ctx_pool_slots : params.n_parallel; params.n_ctx = pool_slots * params.n_ctx_per_slot; - SRV_INF("--ctx-per-slot: sizing KV pool to pool_slots*ctx_per_slot = %d*%d = %d\n", pool_slots, + SRV_INF("--ctx-per-slot: sizing KV pool to pool_slots * ctx_per_slot = %d * %d = %d\n", pool_slots, params.n_ctx_per_slot, params.n_ctx); } From 291a0c4a956015bf082c901dbd518b1b65ff8c79 Mon Sep 17 00:00:00 2001 From: Colin Kealty <3266127+bartowski1182@users.noreply.github.com> Date: Fri, 12 Jun 2026 14:15:01 -0400 Subject: [PATCH 4/5] Remove ctx-pool-slots, make ctx-per-slot an int --- common/arg.cpp | 16 ++-------------- common/common.h | 1 - tools/server/README.md | 1 - tools/server/server.cpp | 22 ++-------------------- 4 files changed, 4 insertions(+), 36 deletions(-) diff --git a/common/arg.cpp b/common/arg.cpp index 0ff69423ced6..06f1da8fc2bd 100644 --- a/common/arg.cpp +++ b/common/arg.cpp @@ -1281,22 +1281,10 @@ common_params_context common_params_parser_init(common_params & params, llama_ex { "--ctx-per-slot" }, "N", "context limit per parallel slot (default: unset, behavior unchanged).\n" "when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N", - [](common_params & params, const std::string & value) { - params.n_ctx_per_slot = std::stoi(value); - } - ).set_env("LLAMA_ARG_CTX_PER_SLOT").set_examples({ LLAMA_EXAMPLE_SERVER })); - add_opt(common_arg( - { "--ctx-pool-slots" }, "N", - "slots worth of context to provision in the shared --ctx-per-slot pool " - "(default: all n_parallel slots).\n" - "with unified KV, values below n_parallel overcommit the pool", [](common_params & params, int value) { - if (value < 1) { - throw std::invalid_argument("error: --ctx-pool-slots must be >= 1\n"); - } - params.ctx_pool_slots = value; + params.n_ctx_per_slot = value; } - ).set_env("LLAMA_ARG_CTX_POOL_SLOTS").set_examples({ LLAMA_EXAMPLE_SERVER })); + ).set_env("LLAMA_ARG_CTX_PER_SLOT").set_examples({ LLAMA_EXAMPLE_SERVER })); add_opt(common_arg( {"-n", "--predict", "--n-predict"}, "N", string_format( diff --git a/common/common.h b/common/common.h index 8763299da491..223f8efca523 100644 --- a/common/common.h +++ b/common/common.h @@ -599,7 +599,6 @@ struct common_params { bool cache_idle_slots = true; // save and clear idle slots upon starting a new task int32_t n_ctx_checkpoints = 32; // max number of context checkpoints per slot int32_t n_ctx_per_slot = 0; // max context per parallel slot; 0 = unset - int32_t ctx_pool_slots = 0; // slots worth of context in the pool; 0 = all n_parallel slots int32_t checkpoint_min_step = 256; // minimum spacing between context checkpoints int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc. diff --git a/tools/server/README.md b/tools/server/README.md index aa8ec2f348f2..6c4aaa7066b3 100644 --- a/tools/server/README.md +++ b/tools/server/README.md @@ -49,7 +49,6 @@ For the full list of features, please refer to [server's changelog](https://gith | `--poll-batch <0\|1>` | use polling to wait for work (default: same as --poll) | | `-c, --ctx-size N` | size of the prompt context (default: 0, 0 = loaded from model)
(env: LLAMA_ARG_CTX_SIZE) | | `--ctx-per-slot N` | context limit per parallel slot (default: unset, behavior unchanged).
when set without `-c`/`--ctx-size`, the shared KV pool is sized to `n_parallel * N`
(env: LLAMA_ARG_CTX_PER_SLOT) | -| `--ctx-pool-slots N` | slots worth of context to provision in the shared `--ctx-per-slot` pool (default: all `n_parallel` slots).
with unified KV, values below `n_parallel` overcommit the pool
(env: LLAMA_ARG_CTX_POOL_SLOTS) | | `-n, --predict, --n-predict N` | number of tokens to predict (default: -1, -1 = infinity)
(env: LLAMA_ARG_N_PREDICT) | | `-b, --batch-size N` | logical maximum batch size (default: 2048)
(env: LLAMA_ARG_BATCH) | | `-ub, --ubatch-size N` | physical maximum batch size (default: 512)
(env: LLAMA_ARG_UBATCH) | diff --git a/tools/server/server.cpp b/tools/server/server.cpp index acd07de1c213..4e0a2cbb7d8c 100644 --- a/tools/server/server.cpp +++ b/tools/server/server.cpp @@ -116,27 +116,9 @@ int llama_server(int argc, char ** argv) { params.n_ctx == 0 && (uint32_t) params.fit_params_min_ctx != UINT32_MAX; - // ctx_pool_slots overcommits the pool; only meaningful with an auto-sized, unified pool - if (params.ctx_pool_slots > 0) { - if (!ctx_pool_auto_sized) { - SRV_WRN("%s", "--ctx-pool-slots requires --ctx-per-slot and no explicit -c, disabling\n"); - params.ctx_pool_slots = 0; - } else if (!params.kv_unified) { - SRV_WRN("%s", - "--ctx-pool-slots requires --kv-unified (non-unified KV gives each slot its " - "own partition), disabling\n"); - params.ctx_pool_slots = 0; - } else if (params.ctx_pool_slots > params.n_parallel) { - SRV_WRN("--ctx-pool-slots (%d) exceeds n_parallel (%d), clamping\n", params.ctx_pool_slots, - params.n_parallel); - params.ctx_pool_slots = params.n_parallel; - } - } - if (ctx_pool_auto_sized) { - const int32_t pool_slots = params.ctx_pool_slots > 0 ? params.ctx_pool_slots : params.n_parallel; - params.n_ctx = pool_slots * params.n_ctx_per_slot; - SRV_INF("--ctx-per-slot: sizing KV pool to pool_slots * ctx_per_slot = %d * %d = %d\n", pool_slots, + params.n_ctx = params.n_parallel * params.n_ctx_per_slot; + SRV_INF("--ctx-per-slot: sizing KV pool to n_parallel * ctx_per_slot = %d * %d = %d\n", params.n_parallel, params.n_ctx_per_slot, params.n_ctx); } From 185c169b078e56fec7819dc58bae23374f24382e Mon Sep 17 00:00:00 2001 From: Xuan Son Nguyen Date: Thu, 27 Aug 2026 21:44:29 +0200 Subject: [PATCH 5/5] refactor it --- common/arg.cpp | 6 ++-- common/common.h | 2 +- tools/server/README.md | 2 +- tools/server/server-context.cpp | 60 ++++++++++++++++++++------------- tools/server/server.cpp | 10 +++--- 5 files changed, 47 insertions(+), 33 deletions(-) diff --git a/common/arg.cpp b/common/arg.cpp index da405e81b3ad..e346863e51fd 100644 --- a/common/arg.cpp +++ b/common/arg.cpp @@ -1644,13 +1644,13 @@ common_params_context common_params_parser_init(common_params & params, llama_ex } ).set_env("LLAMA_ARG_CTX_SIZE")); add_opt(common_arg( - { "--ctx-per-slot" }, "N", + { "--kv-unified-per-slot" }, "N", "context limit per parallel slot (default: unset, behavior unchanged).\n" "when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N", [](common_params & params, int value) { - params.n_ctx_per_slot = value; + params.kv_unified_per_slot = value; } - ).set_env("LLAMA_ARG_CTX_PER_SLOT").set_examples({ LLAMA_EXAMPLE_SERVER })); + ).set_env("LLAMA_ARG_KV_UNIFIED_PER_SLOT").set_examples({ LLAMA_EXAMPLE_SERVER })); add_opt(common_arg( {"-n", "--predict", "--n-predict"}, "N", string_format( diff --git a/common/common.h b/common/common.h index 217452a821e8..a333f702ac1d 100644 --- a/common/common.h +++ b/common/common.h @@ -627,7 +627,7 @@ struct common_params { bool cache_prompt = true; // whether to enable prompt caching bool cache_idle_slots = true; // save and clear idle slots upon starting a new task int32_t n_ctx_checkpoints = 32; // max number of context checkpoints per slot - int32_t n_ctx_per_slot = 0; // max context per parallel slot; 0 = unset + int32_t kv_unified_per_slot = 0; // max context per parallel slot; 0 = unset int32_t checkpoint_min_step = 8192; // minimum spacing between context checkpoints int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc. diff --git a/tools/server/README.md b/tools/server/README.md index daabefbf803f..3c2228f34322 100644 --- a/tools/server/README.md +++ b/tools/server/README.md @@ -48,7 +48,6 @@ For the full list of features, please refer to [server's changelog](https://gith | `--prio-batch N` | set process/thread priority : 0-normal, 1-medium, 2-high, 3-realtime (default: 0) | | `--poll-batch <0\|1>` | use polling to wait for work (default: same as --poll) | | `-c, --ctx-size N` | size of the prompt context (default: 0, 0 = loaded from model)
(env: LLAMA_ARG_CTX_SIZE) | -| `--ctx-per-slot N` | context limit per parallel slot (default: unset, behavior unchanged).
when set without `-c`/`--ctx-size`, the shared KV pool is sized to `n_parallel * N`
(env: LLAMA_ARG_CTX_PER_SLOT) | | `-n, --predict, --n-predict N` | number of tokens to predict (default: -1, -1 = infinity)
(env: LLAMA_ARG_N_PREDICT) | | `-b, --batch-size N` | logical maximum batch size (default: 2048)
(env: LLAMA_ARG_BATCH) | | `-ub, --ubatch-size N` | physical maximum batch size (default: 512)
(env: LLAMA_ARG_UBATCH) | @@ -164,6 +163,7 @@ For the full list of features, please refer to [server's changelog](https://gith | -------- | ----------- | | `-lcs, --lookup-cache-static FNAME` | path to static lookup cache to use for lookup decoding (not updated by generation) | | `-lcd, --lookup-cache-dynamic FNAME` | path to dynamic lookup cache to use for lookup decoding (updated by generation) | +| `--kv-unified-per-slot N` | context limit per parallel slot (default: unset, behavior unchanged).
when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N
(env: LLAMA_ARG_KV_UNIFIED_PER_SLOT) | | `-ctxcp, --ctx-checkpoints, --swa-checkpoints N` | max number of context checkpoints to create per slot (default: 32)[(more info)](https://github.com/ggml-org/llama.cpp/pull/15293)
(env: LLAMA_ARG_CTX_CHECKPOINTS) | | `-cms, --checkpoint-min-step N` | minimum spacing between context checkpoints in tokens (default: 8192, 0 = no minimum)
(env: LLAMA_ARG_CHECKPOINT_MIN_SPACING_NT) | | `-cram, --cache-ram N` | set the maximum cache size in MiB (default: 8192, -1 - no limit, 0 - disable)[(more info)](https://github.com/ggml-org/llama.cpp/pull/16391)
(env: LLAMA_ARG_CACHE_RAM) | diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp index d66fe7548195..f5477356d61d 100644 --- a/tools/server/server-context.cpp +++ b/tools/server/server-context.cpp @@ -1208,24 +1208,31 @@ struct server_context_impl { const int n_ctx_train = llama_model_n_ctx_train(model_tgt); - int n_ctx_slot = llama_n_ctx_seq(ctx_tgt); - if (params_base.n_ctx_per_slot > 0) { - if (n_ctx_slot > params_base.n_ctx_per_slot) { - SRV_INF("capping per-slot context (%d) to --ctx-per-slot (%d)\n", - n_ctx_slot, params_base.n_ctx_per_slot); - n_ctx_slot = params_base.n_ctx_per_slot; - } else if (params_base.n_ctx_per_slot > n_ctx_slot) { - // cap is above the per-slot pool capacity, so it can never bind - SRV_WRN( - "--ctx-per-slot (%d) exceeds the per-slot pool capacity (%d) - cap has no effect, " - "slots are limited to %d (raise the KV pool with -c, or unset -c to size it to " - "n_parallel * ctx_per_slot)\n", - params_base.n_ctx_per_slot, n_ctx_slot, n_ctx_slot); - } - } - if (n_ctx_slot > n_ctx_train) { - SRV_WRN("the slot context (%d) exceeds the training context of the model (%d) - capping\n", n_ctx_slot, n_ctx_train); - n_ctx_slot = n_ctx_train; + { + // note: the capping itself is done in n_ctx_slot(), here we only report it + const int n_ctx_seq = llama_n_ctx_seq(ctx_tgt); + + if (params_base.kv_unified_per_slot > 0) { + if (n_ctx_seq > params_base.kv_unified_per_slot) { + SRV_INF("capping per-slot context (%d) to --kv-unified-per-slot (%d)\n", + n_ctx_seq, params_base.kv_unified_per_slot); + } else if (params_base.kv_unified_per_slot > n_ctx_seq) { + // cap is above the per-slot pool capacity, so it can never bind + SRV_WRN( + "--kv-unified-per-slot (%d) exceeds the per-slot pool capacity (%d) - cap has no effect, " + "slots are limited to %d (raise the KV pool with -c, or unset -c to size it to " + "n_parallel * kv_unified_per_slot)\n", + params_base.kv_unified_per_slot, n_ctx_seq, n_ctx_seq); + } + } + + const int n_ctx_capped = params_base.kv_unified_per_slot > 0 ? + std::min(n_ctx_seq, params_base.kv_unified_per_slot) : n_ctx_seq; + + if (n_ctx_capped > n_ctx_train) { + SRV_WRN("the slot context (%d) exceeds the training context of the model (%d) - capping\n", + n_ctx_capped, n_ctx_train); + } } slots.clear(); @@ -1241,7 +1248,7 @@ struct server_context_impl { // setup slots SRV_INF("initializing, n_slots = %d, n_ctx_slot = %d, kv_unified = '%s'\n", - params_base.n_parallel, n_ctx_slot, params_base.kv_unified ? "true" : "false"); + params_base.n_parallel, n_ctx_slot(), params_base.kv_unified ? "true" : "false"); // initialize slots for (int i = 0; i < params_base.n_parallel; i++) { @@ -1285,7 +1292,7 @@ struct server_context_impl { slot.ctx_dft = ctx_dft; slot.mem.init(ctx_tgt, ctx_dft); slot.spec = spec.get(); - slot.n_ctx = n_ctx_slot; + slot.n_ctx = n_ctx_slot(); slot.mctx = mctx; slot.prompt.tokens.has_mtmd = mctx != nullptr; @@ -3989,8 +3996,15 @@ struct server_context_impl { }); } - int get_slot_n_ctx() { - return slots.back().n_ctx; + // context size of a single slot, capped by --kv-unified-per-slot and by the training context of the model + int n_ctx_slot() const { + int res = llama_n_ctx_seq(ctx_tgt); + + if (params_base.kv_unified_per_slot > 0) { + res = std::min(res, params_base.kv_unified_per_slot); + } + + return std::min(res, llama_model_n_ctx_train(model_tgt)); } server_response_reader get_response_reader() { @@ -4156,7 +4170,7 @@ server_context_meta server_context::get_meta() const { /* has_inp_audio */ impl->chat_params.allow_audio, /* has_inp_video */ impl->chat_params.allow_video, /* json_ui_settings */ impl->json_ui_settings, - /* slot_n_ctx */ impl->get_slot_n_ctx(), + /* slot_n_ctx */ impl->n_ctx_slot(), /* pooling_type */ llama_pooling_type(impl->ctx_tgt), /* chat_params */ impl->chat_params, diff --git a/tools/server/server.cpp b/tools/server/server.cpp index c8038cbc0a5f..22378b38c5ef 100644 --- a/tools/server/server.cpp +++ b/tools/server/server.cpp @@ -157,16 +157,16 @@ int llama_server(common_params & params, int argc, char ** argv) { } } - // size the KV pool from --ctx-per-slot, unless the user pinned it with -c + // size the KV pool from --kv-unified-per-slot, unless the user pinned it with -c // or with -c 0 for max context - const bool ctx_pool_auto_sized = params.n_ctx_per_slot > 0 && + const bool ctx_pool_auto_sized = params.kv_unified_per_slot > 0 && params.n_ctx == 0 && (uint32_t) params.fit_params_min_ctx != UINT32_MAX; if (ctx_pool_auto_sized) { - params.n_ctx = params.n_parallel * params.n_ctx_per_slot; - SRV_INF("--ctx-per-slot: sizing KV pool to n_parallel * ctx_per_slot = %d * %d = %d\n", params.n_parallel, - params.n_ctx_per_slot, params.n_ctx); + params.n_ctx = params.n_parallel * params.kv_unified_per_slot; + SRV_INF("--kv-unified-per-slot: sizing KV pool to n_parallel * kv_unified_per_slot = %d * %d = %d\n", params.n_parallel, + params.kv_unified_per_slot, params.n_ctx); } // for consistency between server router mode and single-model mode, we set the same model name as alias