From 9be1d52953e1bc117212b762578beeeaee9a5085 Mon Sep 17 00:00:00 2001
From: Colin Kealty <3266127+bartowski1182@users.noreply.github.com>
Date: Thu, 4 Jun 2026 02:06:10 -0400
Subject: [PATCH 1/5] Add ctx-per-slot argument for unifid KV cache
---
common/arg.cpp | 39 +++++++++++++++++++++++----------
common/common.h | 2 ++
tools/server/server-context.cpp | 14 ++++++++++++
tools/server/server.cpp | 24 ++++++++++++++++++++
4 files changed, 68 insertions(+), 11 deletions(-)
diff --git a/common/arg.cpp b/common/arg.cpp
index 24d9734b934e..7fc3f72129c5 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -1285,17 +1285,34 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
}
}
).set_env("LLAMA_ARG_CTX_SIZE"));
- add_opt(common_arg(
- {"-n", "--predict", "--n-predict"}, "N",
- string_format(
- ex == LLAMA_EXAMPLE_COMPLETION
- ? "number of tokens to predict (default: %d, -1 = infinity, -2 = until context filled)"
- : "number of tokens to predict (default: %d, -1 = infinity)",
- params.n_predict),
- [](common_params & params, int value) {
- params.n_predict = value;
- }
- ).set_env("LLAMA_ARG_N_PREDICT"));
+ add_opt(
+ common_arg({ "--ctx-per-slot" }, "N",
+ "max context per parallel slot in unified KV mode (default: unset, behavior unchanged).\n"
+ "when set and -c/--ctx-size is not given, the shared KV pool is sized to n_parallel*N (fit-clamped)",
+ [](common_params & params, const std::string & value) { params.n_ctx_per_slot = std::stoi(value); })
+ .set_env("LLAMA_ARG_CTX_PER_SLOT")
+ .set_examples({ LLAMA_EXAMPLE_SERVER }));
+ add_opt(common_arg({ "--ctx-pool-frac" }, "N",
+ string_format("fraction (0 < N <= 1) of the --ctx-per-slot pool to allocate (default: %.2f).\n"
+ "with unified KV, slots share the pool, so < 1.0 overcommits (allocate for the "
+ "expected, not worst, case)",
+ (double) params.ctx_pool_frac),
+ [](common_params & params, const std::string & value) {
+ params.ctx_pool_frac = std::stof(value);
+ if (params.ctx_pool_frac <= 0.0f || params.ctx_pool_frac > 1.0f) {
+ throw std::invalid_argument("error: --ctx-pool-frac must be in the range (0, 1]\n");
+ }
+ })
+ .set_env("LLAMA_ARG_CTX_POOL_FRAC")
+ .set_examples({ LLAMA_EXAMPLE_SERVER }));
+ add_opt(common_arg({ "-n", "--predict", "--n-predict" }, "N",
+ string_format(
+ ex == LLAMA_EXAMPLE_COMPLETION ?
+ "number of tokens to predict (default: %d, -1 = infinity, -2 = until context filled)" :
+ "number of tokens to predict (default: %d, -1 = infinity)",
+ params.n_predict),
+ [](common_params & params, int value) { params.n_predict = value; })
+ .set_env("LLAMA_ARG_N_PREDICT"));
add_opt(common_arg(
{"-b", "--batch-size"}, "N",
string_format("logical maximum batch size (default: %d)", params.n_batch),
diff --git a/common/common.h b/common/common.h
index dec90456afab..494b5cd4f63f 100644
--- a/common/common.h
+++ b/common/common.h
@@ -595,6 +595,8 @@ struct common_params {
bool cache_idle_slots = true; // save and clear idle slots upon starting a new task
int32_t n_ctx_checkpoints = 32; // max number of context checkpoints per slot
int32_t checkpoint_every_nt = 8192; // make a checkpoint every n tokens during prefill
+ int32_t n_ctx_per_slot = 0; // max context per parallel slot; 0 = unset
+ float ctx_pool_frac = 1.0f; // fraction of the --ctx-per-slot pool to allocate
int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc.
std::string hostname = "127.0.0.1";
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
index f517310266c0..1de71c73d00b 100644
--- a/tools/server/server-context.cpp
+++ b/tools/server/server-context.cpp
@@ -923,6 +923,20 @@ struct server_context_impl {
const int n_ctx_train = llama_model_n_ctx_train(model_tgt);
int n_ctx_slot = llama_n_ctx_seq(ctx_tgt);
+ if (params_base.n_ctx_per_slot > 0) {
+ if (n_ctx_slot > params_base.n_ctx_per_slot) {
+ SRV_INF("capping per-slot context (%d) to --ctx-per-slot (%d)\n", n_ctx_slot,
+ params_base.n_ctx_per_slot);
+ n_ctx_slot = params_base.n_ctx_per_slot;
+ } else if (params_base.n_ctx_per_slot > n_ctx_slot) {
+ // cap is above the per-slot pool capacity, so it can never bind
+ SRV_WRN(
+ "--ctx-per-slot (%d) exceeds the per-slot pool capacity (%d) - cap has no effect, "
+ "slots are limited to %d (raise the KV pool with -c, or unset -c to size it to "
+ "n_parallel*ctx_per_slot)\n",
+ params_base.n_ctx_per_slot, n_ctx_slot, n_ctx_slot);
+ }
+ }
if (n_ctx_slot > n_ctx_train) {
SRV_WRN("the slot context (%d) exceeds the training context of the model (%d) - capping\n", n_ctx_slot, n_ctx_train);
n_ctx_slot = n_ctx_train;
diff --git a/tools/server/server.cpp b/tools/server/server.cpp
index 4d56d45e83cc..cce1f7373212 100644
--- a/tools/server/server.cpp
+++ b/tools/server/server.cpp
@@ -110,6 +110,30 @@ int llama_server(int argc, char ** argv) {
params.kv_unified = true;
}
+ // size the KV pool from --ctx-per-slot, unless the user pinned it with -c (-c 0 sets
+ // fit_params_min_ctx to UINT32_MAX)
+ const bool ctx_pool_auto_sized =
+ params.n_ctx_per_slot > 0 && params.n_ctx == 0 && (uint32_t) params.fit_params_min_ctx != UINT32_MAX;
+
+ // ctx_pool_frac overcommits the pool; only meaningful with an auto-sized, unified pool
+ if (params.ctx_pool_frac != 1.0f) {
+ if (!ctx_pool_auto_sized) {
+ SRV_WRN("%s", "--ctx-pool-frac requires --ctx-per-slot and no explicit -c, disabling\n");
+ params.ctx_pool_frac = 1.0f;
+ } else if (!params.kv_unified) {
+ SRV_WRN("%s",
+ "--ctx-pool-frac requires --kv-unified (non-unified KV partitions the pool "
+ "per slot, so the fraction would just shrink each slot), disabling\n");
+ params.ctx_pool_frac = 1.0f;
+ }
+ }
+
+ if (ctx_pool_auto_sized) {
+ params.n_ctx = (int32_t) (params.ctx_pool_frac * params.n_parallel * params.n_ctx_per_slot);
+ SRV_INF("--ctx-per-slot: sizing KV pool to ctx_pool_frac*n_parallel*ctx_per_slot = %.2f*%d*%d = %d\n",
+ params.ctx_pool_frac, params.n_parallel, params.n_ctx_per_slot, params.n_ctx);
+ }
+
// for consistency between server router mode and single-model mode, we set the same model name as alias
if (params.model_alias.empty() && !params.model.name.empty()) {
params.model_alias.insert(params.model.name);
From b2d7dc68118e12311e8ac038bf143b7827a42161 Mon Sep 17 00:00:00 2001
From: Colin Kealty <3266127+bartowski1182@users.noreply.github.com>
Date: Thu, 4 Jun 2026 09:41:17 -0400
Subject: [PATCH 2/5] Swap out ctx fractions for ctx pool slots
---
common/arg.cpp | 24 ++++++++++++------------
common/common.h | 3 +--
tools/server/README.md | 2 ++
tools/server/server.cpp | 25 +++++++++++++++----------
4 files changed, 30 insertions(+), 24 deletions(-)
diff --git a/common/arg.cpp b/common/arg.cpp
index de7ec2002114..7cf4c6920686 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -1279,23 +1279,23 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
).set_env("LLAMA_ARG_CTX_SIZE"));
add_opt(
common_arg({ "--ctx-per-slot" }, "N",
- "max context per parallel slot in unified KV mode (default: unset, behavior unchanged).\n"
- "when set and -c/--ctx-size is not given, the shared KV pool is sized to n_parallel*N (fit-clamped)",
+ "context limit per parallel slot (default: unset, behavior unchanged).\n"
+ "when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N",
[](common_params & params, const std::string & value) { params.n_ctx_per_slot = std::stoi(value); })
.set_env("LLAMA_ARG_CTX_PER_SLOT")
.set_examples({ LLAMA_EXAMPLE_SERVER }));
- add_opt(common_arg({ "--ctx-pool-frac" }, "N",
- string_format("fraction (0 < N <= 1) of the --ctx-per-slot pool to allocate (default: %.2f).\n"
- "with unified KV, slots share the pool, so < 1.0 overcommits (allocate for the "
- "expected, not worst, case)",
- (double) params.ctx_pool_frac),
- [](common_params & params, const std::string & value) {
- params.ctx_pool_frac = std::stof(value);
- if (params.ctx_pool_frac <= 0.0f || params.ctx_pool_frac > 1.0f) {
- throw std::invalid_argument("error: --ctx-pool-frac must be in the range (0, 1]\n");
+ add_opt(common_arg({ "--ctx-pool-slots" }, "N",
+ "slots worth of context to provision in the shared --ctx-per-slot pool "
+ "(default: all n_parallel slots).\n"
+ "with unified KV, values below n_parallel overcommit the pool (allocate for the "
+ "expected, not worst, case)",
+ [](common_params & params, int value) {
+ if (value < 1) {
+ throw std::invalid_argument("error: --ctx-pool-slots must be >= 1\n");
}
+ params.ctx_pool_slots = value;
})
- .set_env("LLAMA_ARG_CTX_POOL_FRAC")
+ .set_env("LLAMA_ARG_CTX_POOL_SLOTS")
.set_examples({ LLAMA_EXAMPLE_SERVER }));
add_opt(common_arg({ "-n", "--predict", "--n-predict" }, "N",
string_format(
diff --git a/common/common.h b/common/common.h
index 0d63859b79cb..8763299da491 100644
--- a/common/common.h
+++ b/common/common.h
@@ -598,9 +598,8 @@ struct common_params {
bool cache_prompt = true; // whether to enable prompt caching
bool cache_idle_slots = true; // save and clear idle slots upon starting a new task
int32_t n_ctx_checkpoints = 32; // max number of context checkpoints per slot
- int32_t checkpoint_every_nt = 8192; // make a checkpoint every n tokens during prefill
int32_t n_ctx_per_slot = 0; // max context per parallel slot; 0 = unset
- float ctx_pool_frac = 1.0f; // fraction of the --ctx-per-slot pool to allocate
+ int32_t ctx_pool_slots = 0; // slots worth of context in the pool; 0 = all n_parallel slots
int32_t checkpoint_min_step = 256; // minimum spacing between context checkpoints
int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc.
diff --git a/tools/server/README.md b/tools/server/README.md
index f1eeec36aa0c..dad7d9d31845 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -48,6 +48,8 @@ For the full list of features, please refer to [server's changelog](https://gith
| `--prio-batch N` | set process/thread priority : 0-normal, 1-medium, 2-high, 3-realtime (default: 0) |
| `--poll-batch <0\|1>` | use polling to wait for work (default: same as --poll) |
| `-c, --ctx-size N` | size of the prompt context (default: 0, 0 = loaded from model)
(env: LLAMA_ARG_CTX_SIZE) |
+| `--ctx-per-slot N` | context limit per parallel slot (default: unset, behavior unchanged).
when set without `-c`/`--ctx-size`, the shared KV pool is sized to `n_parallel*N`
(env: LLAMA_ARG_CTX_PER_SLOT) |
+| `--ctx-pool-slots N` | slots worth of context to provision in the shared `--ctx-per-slot` pool (default: all `n_parallel` slots).
with unified KV, values below `n_parallel` overcommit the pool (allocate for the expected, not worst, case)
(env: LLAMA_ARG_CTX_POOL_SLOTS) |
| `-n, --predict, --n-predict N` | number of tokens to predict (default: -1, -1 = infinity)
(env: LLAMA_ARG_N_PREDICT) |
| `-b, --batch-size N` | logical maximum batch size (default: 2048)
(env: LLAMA_ARG_BATCH) |
| `-ub, --ubatch-size N` | physical maximum batch size (default: 512)
(env: LLAMA_ARG_UBATCH) |
diff --git a/tools/server/server.cpp b/tools/server/server.cpp
index c6a4f16f8cd3..2f4324e8d1bb 100644
--- a/tools/server/server.cpp
+++ b/tools/server/server.cpp
@@ -115,23 +115,28 @@ int llama_server(int argc, char ** argv) {
const bool ctx_pool_auto_sized =
params.n_ctx_per_slot > 0 && params.n_ctx == 0 && (uint32_t) params.fit_params_min_ctx != UINT32_MAX;
- // ctx_pool_frac overcommits the pool; only meaningful with an auto-sized, unified pool
- if (params.ctx_pool_frac != 1.0f) {
+ // ctx_pool_slots overcommits the pool; only meaningful with an auto-sized, unified pool
+ if (params.ctx_pool_slots > 0) {
if (!ctx_pool_auto_sized) {
- SRV_WRN("%s", "--ctx-pool-frac requires --ctx-per-slot and no explicit -c, disabling\n");
- params.ctx_pool_frac = 1.0f;
+ SRV_WRN("%s", "--ctx-pool-slots requires --ctx-per-slot and no explicit -c, disabling\n");
+ params.ctx_pool_slots = 0;
} else if (!params.kv_unified) {
SRV_WRN("%s",
- "--ctx-pool-frac requires --kv-unified (non-unified KV partitions the pool "
- "per slot, so the fraction would just shrink each slot), disabling\n");
- params.ctx_pool_frac = 1.0f;
+ "--ctx-pool-slots requires --kv-unified (non-unified KV gives each slot its "
+ "own partition), disabling\n");
+ params.ctx_pool_slots = 0;
+ } else if (params.ctx_pool_slots > params.n_parallel) {
+ SRV_WRN("--ctx-pool-slots (%d) exceeds n_parallel (%d), clamping\n", params.ctx_pool_slots,
+ params.n_parallel);
+ params.ctx_pool_slots = params.n_parallel;
}
}
if (ctx_pool_auto_sized) {
- params.n_ctx = (int32_t) (params.ctx_pool_frac * params.n_parallel * params.n_ctx_per_slot);
- SRV_INF("--ctx-per-slot: sizing KV pool to ctx_pool_frac*n_parallel*ctx_per_slot = %.2f*%d*%d = %d\n",
- params.ctx_pool_frac, params.n_parallel, params.n_ctx_per_slot, params.n_ctx);
+ const int32_t pool_slots = params.ctx_pool_slots > 0 ? params.ctx_pool_slots : params.n_parallel;
+ params.n_ctx = pool_slots * params.n_ctx_per_slot;
+ SRV_INF("--ctx-per-slot: sizing KV pool to pool_slots*ctx_per_slot = %d*%d = %d\n", pool_slots,
+ params.n_ctx_per_slot, params.n_ctx);
}
// for consistency between server router mode and single-model mode, we set the same model name as alias
From 6386441f63eaafd5b5a7f495efe91370f4794425 Mon Sep 17 00:00:00 2001
From: Colin Kealty <3266127+bartowski1182@users.noreply.github.com>
Date: Thu, 4 Jun 2026 09:55:24 -0400
Subject: [PATCH 3/5] Formatting cleanup
---
common/arg.cpp | 59 +++++++++++++++++----------------
tools/server/README.md | 4 +--
tools/server/server-context.cpp | 6 ++--
tools/server/server.cpp | 11 +++---
4 files changed, 42 insertions(+), 38 deletions(-)
diff --git a/common/arg.cpp b/common/arg.cpp
index 7cf4c6920686..0ff69423ced6 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -1277,34 +1277,37 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
}
}
).set_env("LLAMA_ARG_CTX_SIZE"));
- add_opt(
- common_arg({ "--ctx-per-slot" }, "N",
- "context limit per parallel slot (default: unset, behavior unchanged).\n"
- "when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N",
- [](common_params & params, const std::string & value) { params.n_ctx_per_slot = std::stoi(value); })
- .set_env("LLAMA_ARG_CTX_PER_SLOT")
- .set_examples({ LLAMA_EXAMPLE_SERVER }));
- add_opt(common_arg({ "--ctx-pool-slots" }, "N",
- "slots worth of context to provision in the shared --ctx-per-slot pool "
- "(default: all n_parallel slots).\n"
- "with unified KV, values below n_parallel overcommit the pool (allocate for the "
- "expected, not worst, case)",
- [](common_params & params, int value) {
- if (value < 1) {
- throw std::invalid_argument("error: --ctx-pool-slots must be >= 1\n");
- }
- params.ctx_pool_slots = value;
- })
- .set_env("LLAMA_ARG_CTX_POOL_SLOTS")
- .set_examples({ LLAMA_EXAMPLE_SERVER }));
- add_opt(common_arg({ "-n", "--predict", "--n-predict" }, "N",
- string_format(
- ex == LLAMA_EXAMPLE_COMPLETION ?
- "number of tokens to predict (default: %d, -1 = infinity, -2 = until context filled)" :
- "number of tokens to predict (default: %d, -1 = infinity)",
- params.n_predict),
- [](common_params & params, int value) { params.n_predict = value; })
- .set_env("LLAMA_ARG_N_PREDICT"));
+ add_opt(common_arg(
+ { "--ctx-per-slot" }, "N",
+ "context limit per parallel slot (default: unset, behavior unchanged).\n"
+ "when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N",
+ [](common_params & params, const std::string & value) {
+ params.n_ctx_per_slot = std::stoi(value);
+ }
+ ).set_env("LLAMA_ARG_CTX_PER_SLOT").set_examples({ LLAMA_EXAMPLE_SERVER }));
+ add_opt(common_arg(
+ { "--ctx-pool-slots" }, "N",
+ "slots worth of context to provision in the shared --ctx-per-slot pool "
+ "(default: all n_parallel slots).\n"
+ "with unified KV, values below n_parallel overcommit the pool",
+ [](common_params & params, int value) {
+ if (value < 1) {
+ throw std::invalid_argument("error: --ctx-pool-slots must be >= 1\n");
+ }
+ params.ctx_pool_slots = value;
+ }
+ ).set_env("LLAMA_ARG_CTX_POOL_SLOTS").set_examples({ LLAMA_EXAMPLE_SERVER }));
+ add_opt(common_arg(
+ {"-n", "--predict", "--n-predict"}, "N",
+ string_format(
+ ex == LLAMA_EXAMPLE_COMPLETION
+ ? "number of tokens to predict (default: %d, -1 = infinity, -2 = until context filled)"
+ : "number of tokens to predict (default: %d, -1 = infinity)",
+ params.n_predict),
+ [](common_params & params, int value) {
+ params.n_predict = value;
+ }
+ ).set_env("LLAMA_ARG_N_PREDICT"));
add_opt(common_arg(
{"-b", "--batch-size"}, "N",
string_format("logical maximum batch size (default: %d)", params.n_batch),
diff --git a/tools/server/README.md b/tools/server/README.md
index dad7d9d31845..aa8ec2f348f2 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -48,8 +48,8 @@ For the full list of features, please refer to [server's changelog](https://gith
| `--prio-batch N` | set process/thread priority : 0-normal, 1-medium, 2-high, 3-realtime (default: 0) |
| `--poll-batch <0\|1>` | use polling to wait for work (default: same as --poll) |
| `-c, --ctx-size N` | size of the prompt context (default: 0, 0 = loaded from model)
(env: LLAMA_ARG_CTX_SIZE) |
-| `--ctx-per-slot N` | context limit per parallel slot (default: unset, behavior unchanged).
when set without `-c`/`--ctx-size`, the shared KV pool is sized to `n_parallel*N`
(env: LLAMA_ARG_CTX_PER_SLOT) |
-| `--ctx-pool-slots N` | slots worth of context to provision in the shared `--ctx-per-slot` pool (default: all `n_parallel` slots).
with unified KV, values below `n_parallel` overcommit the pool (allocate for the expected, not worst, case)
(env: LLAMA_ARG_CTX_POOL_SLOTS) |
+| `--ctx-per-slot N` | context limit per parallel slot (default: unset, behavior unchanged).
when set without `-c`/`--ctx-size`, the shared KV pool is sized to `n_parallel * N`
(env: LLAMA_ARG_CTX_PER_SLOT) |
+| `--ctx-pool-slots N` | slots worth of context to provision in the shared `--ctx-per-slot` pool (default: all `n_parallel` slots).
with unified KV, values below `n_parallel` overcommit the pool
(env: LLAMA_ARG_CTX_POOL_SLOTS) |
| `-n, --predict, --n-predict N` | number of tokens to predict (default: -1, -1 = infinity)
(env: LLAMA_ARG_N_PREDICT) |
| `-b, --batch-size N` | logical maximum batch size (default: 2048)
(env: LLAMA_ARG_BATCH) |
| `-ub, --ubatch-size N` | physical maximum batch size (default: 512)
(env: LLAMA_ARG_UBATCH) |
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
index 2db17cee1e60..9ee12637bd8a 100644
--- a/tools/server/server-context.cpp
+++ b/tools/server/server-context.cpp
@@ -1032,15 +1032,15 @@ struct server_context_impl {
int n_ctx_slot = llama_n_ctx_seq(ctx_tgt);
if (params_base.n_ctx_per_slot > 0) {
if (n_ctx_slot > params_base.n_ctx_per_slot) {
- SRV_INF("capping per-slot context (%d) to --ctx-per-slot (%d)\n", n_ctx_slot,
- params_base.n_ctx_per_slot);
+ SRV_INF("capping per-slot context (%d) to --ctx-per-slot (%d)\n",
+ n_ctx_slot, params_base.n_ctx_per_slot);
n_ctx_slot = params_base.n_ctx_per_slot;
} else if (params_base.n_ctx_per_slot > n_ctx_slot) {
// cap is above the per-slot pool capacity, so it can never bind
SRV_WRN(
"--ctx-per-slot (%d) exceeds the per-slot pool capacity (%d) - cap has no effect, "
"slots are limited to %d (raise the KV pool with -c, or unset -c to size it to "
- "n_parallel*ctx_per_slot)\n",
+ "n_parallel * ctx_per_slot)\n",
params_base.n_ctx_per_slot, n_ctx_slot, n_ctx_slot);
}
}
diff --git a/tools/server/server.cpp b/tools/server/server.cpp
index 2f4324e8d1bb..acd07de1c213 100644
--- a/tools/server/server.cpp
+++ b/tools/server/server.cpp
@@ -110,10 +110,11 @@ int llama_server(int argc, char ** argv) {
params.kv_unified = true;
}
- // size the KV pool from --ctx-per-slot, unless the user pinned it with -c (-c 0 sets
- // fit_params_min_ctx to UINT32_MAX)
- const bool ctx_pool_auto_sized =
- params.n_ctx_per_slot > 0 && params.n_ctx == 0 && (uint32_t) params.fit_params_min_ctx != UINT32_MAX;
+ // size the KV pool from --ctx-per-slot, unless the user pinned it with -c
+ // or with -c 0 for max context
+ const bool ctx_pool_auto_sized = params.n_ctx_per_slot > 0 &&
+ params.n_ctx == 0 &&
+ (uint32_t) params.fit_params_min_ctx != UINT32_MAX;
// ctx_pool_slots overcommits the pool; only meaningful with an auto-sized, unified pool
if (params.ctx_pool_slots > 0) {
@@ -135,7 +136,7 @@ int llama_server(int argc, char ** argv) {
if (ctx_pool_auto_sized) {
const int32_t pool_slots = params.ctx_pool_slots > 0 ? params.ctx_pool_slots : params.n_parallel;
params.n_ctx = pool_slots * params.n_ctx_per_slot;
- SRV_INF("--ctx-per-slot: sizing KV pool to pool_slots*ctx_per_slot = %d*%d = %d\n", pool_slots,
+ SRV_INF("--ctx-per-slot: sizing KV pool to pool_slots * ctx_per_slot = %d * %d = %d\n", pool_slots,
params.n_ctx_per_slot, params.n_ctx);
}
From 291a0c4a956015bf082c901dbd518b1b65ff8c79 Mon Sep 17 00:00:00 2001
From: Colin Kealty <3266127+bartowski1182@users.noreply.github.com>
Date: Fri, 12 Jun 2026 14:15:01 -0400
Subject: [PATCH 4/5] Remove ctx-pool-slots, make ctx-per-slot an int
---
common/arg.cpp | 16 ++--------------
common/common.h | 1 -
tools/server/README.md | 1 -
tools/server/server.cpp | 22 ++--------------------
4 files changed, 4 insertions(+), 36 deletions(-)
diff --git a/common/arg.cpp b/common/arg.cpp
index 0ff69423ced6..06f1da8fc2bd 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -1281,22 +1281,10 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
{ "--ctx-per-slot" }, "N",
"context limit per parallel slot (default: unset, behavior unchanged).\n"
"when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N",
- [](common_params & params, const std::string & value) {
- params.n_ctx_per_slot = std::stoi(value);
- }
- ).set_env("LLAMA_ARG_CTX_PER_SLOT").set_examples({ LLAMA_EXAMPLE_SERVER }));
- add_opt(common_arg(
- { "--ctx-pool-slots" }, "N",
- "slots worth of context to provision in the shared --ctx-per-slot pool "
- "(default: all n_parallel slots).\n"
- "with unified KV, values below n_parallel overcommit the pool",
[](common_params & params, int value) {
- if (value < 1) {
- throw std::invalid_argument("error: --ctx-pool-slots must be >= 1\n");
- }
- params.ctx_pool_slots = value;
+ params.n_ctx_per_slot = value;
}
- ).set_env("LLAMA_ARG_CTX_POOL_SLOTS").set_examples({ LLAMA_EXAMPLE_SERVER }));
+ ).set_env("LLAMA_ARG_CTX_PER_SLOT").set_examples({ LLAMA_EXAMPLE_SERVER }));
add_opt(common_arg(
{"-n", "--predict", "--n-predict"}, "N",
string_format(
diff --git a/common/common.h b/common/common.h
index 8763299da491..223f8efca523 100644
--- a/common/common.h
+++ b/common/common.h
@@ -599,7 +599,6 @@ struct common_params {
bool cache_idle_slots = true; // save and clear idle slots upon starting a new task
int32_t n_ctx_checkpoints = 32; // max number of context checkpoints per slot
int32_t n_ctx_per_slot = 0; // max context per parallel slot; 0 = unset
- int32_t ctx_pool_slots = 0; // slots worth of context in the pool; 0 = all n_parallel slots
int32_t checkpoint_min_step = 256; // minimum spacing between context checkpoints
int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc.
diff --git a/tools/server/README.md b/tools/server/README.md
index aa8ec2f348f2..6c4aaa7066b3 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -49,7 +49,6 @@ For the full list of features, please refer to [server's changelog](https://gith
| `--poll-batch <0\|1>` | use polling to wait for work (default: same as --poll) |
| `-c, --ctx-size N` | size of the prompt context (default: 0, 0 = loaded from model)
(env: LLAMA_ARG_CTX_SIZE) |
| `--ctx-per-slot N` | context limit per parallel slot (default: unset, behavior unchanged).
when set without `-c`/`--ctx-size`, the shared KV pool is sized to `n_parallel * N`
(env: LLAMA_ARG_CTX_PER_SLOT) |
-| `--ctx-pool-slots N` | slots worth of context to provision in the shared `--ctx-per-slot` pool (default: all `n_parallel` slots).
with unified KV, values below `n_parallel` overcommit the pool
(env: LLAMA_ARG_CTX_POOL_SLOTS) |
| `-n, --predict, --n-predict N` | number of tokens to predict (default: -1, -1 = infinity)
(env: LLAMA_ARG_N_PREDICT) |
| `-b, --batch-size N` | logical maximum batch size (default: 2048)
(env: LLAMA_ARG_BATCH) |
| `-ub, --ubatch-size N` | physical maximum batch size (default: 512)
(env: LLAMA_ARG_UBATCH) |
diff --git a/tools/server/server.cpp b/tools/server/server.cpp
index acd07de1c213..4e0a2cbb7d8c 100644
--- a/tools/server/server.cpp
+++ b/tools/server/server.cpp
@@ -116,27 +116,9 @@ int llama_server(int argc, char ** argv) {
params.n_ctx == 0 &&
(uint32_t) params.fit_params_min_ctx != UINT32_MAX;
- // ctx_pool_slots overcommits the pool; only meaningful with an auto-sized, unified pool
- if (params.ctx_pool_slots > 0) {
- if (!ctx_pool_auto_sized) {
- SRV_WRN("%s", "--ctx-pool-slots requires --ctx-per-slot and no explicit -c, disabling\n");
- params.ctx_pool_slots = 0;
- } else if (!params.kv_unified) {
- SRV_WRN("%s",
- "--ctx-pool-slots requires --kv-unified (non-unified KV gives each slot its "
- "own partition), disabling\n");
- params.ctx_pool_slots = 0;
- } else if (params.ctx_pool_slots > params.n_parallel) {
- SRV_WRN("--ctx-pool-slots (%d) exceeds n_parallel (%d), clamping\n", params.ctx_pool_slots,
- params.n_parallel);
- params.ctx_pool_slots = params.n_parallel;
- }
- }
-
if (ctx_pool_auto_sized) {
- const int32_t pool_slots = params.ctx_pool_slots > 0 ? params.ctx_pool_slots : params.n_parallel;
- params.n_ctx = pool_slots * params.n_ctx_per_slot;
- SRV_INF("--ctx-per-slot: sizing KV pool to pool_slots * ctx_per_slot = %d * %d = %d\n", pool_slots,
+ params.n_ctx = params.n_parallel * params.n_ctx_per_slot;
+ SRV_INF("--ctx-per-slot: sizing KV pool to n_parallel * ctx_per_slot = %d * %d = %d\n", params.n_parallel,
params.n_ctx_per_slot, params.n_ctx);
}
From 185c169b078e56fec7819dc58bae23374f24382e Mon Sep 17 00:00:00 2001
From: Xuan Son Nguyen
Date: Thu, 27 Aug 2026 21:44:29 +0200
Subject: [PATCH 5/5] refactor it
---
common/arg.cpp | 6 ++--
common/common.h | 2 +-
tools/server/README.md | 2 +-
tools/server/server-context.cpp | 60 ++++++++++++++++++++-------------
tools/server/server.cpp | 10 +++---
5 files changed, 47 insertions(+), 33 deletions(-)
diff --git a/common/arg.cpp b/common/arg.cpp
index da405e81b3ad..e346863e51fd 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -1644,13 +1644,13 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
}
).set_env("LLAMA_ARG_CTX_SIZE"));
add_opt(common_arg(
- { "--ctx-per-slot" }, "N",
+ { "--kv-unified-per-slot" }, "N",
"context limit per parallel slot (default: unset, behavior unchanged).\n"
"when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N",
[](common_params & params, int value) {
- params.n_ctx_per_slot = value;
+ params.kv_unified_per_slot = value;
}
- ).set_env("LLAMA_ARG_CTX_PER_SLOT").set_examples({ LLAMA_EXAMPLE_SERVER }));
+ ).set_env("LLAMA_ARG_KV_UNIFIED_PER_SLOT").set_examples({ LLAMA_EXAMPLE_SERVER }));
add_opt(common_arg(
{"-n", "--predict", "--n-predict"}, "N",
string_format(
diff --git a/common/common.h b/common/common.h
index 217452a821e8..a333f702ac1d 100644
--- a/common/common.h
+++ b/common/common.h
@@ -627,7 +627,7 @@ struct common_params {
bool cache_prompt = true; // whether to enable prompt caching
bool cache_idle_slots = true; // save and clear idle slots upon starting a new task
int32_t n_ctx_checkpoints = 32; // max number of context checkpoints per slot
- int32_t n_ctx_per_slot = 0; // max context per parallel slot; 0 = unset
+ int32_t kv_unified_per_slot = 0; // max context per parallel slot; 0 = unset
int32_t checkpoint_min_step = 8192; // minimum spacing between context checkpoints
int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc.
diff --git a/tools/server/README.md b/tools/server/README.md
index daabefbf803f..3c2228f34322 100644
--- a/tools/server/README.md
+++ b/tools/server/README.md
@@ -48,7 +48,6 @@ For the full list of features, please refer to [server's changelog](https://gith
| `--prio-batch N` | set process/thread priority : 0-normal, 1-medium, 2-high, 3-realtime (default: 0) |
| `--poll-batch <0\|1>` | use polling to wait for work (default: same as --poll) |
| `-c, --ctx-size N` | size of the prompt context (default: 0, 0 = loaded from model)
(env: LLAMA_ARG_CTX_SIZE) |
-| `--ctx-per-slot N` | context limit per parallel slot (default: unset, behavior unchanged).
when set without `-c`/`--ctx-size`, the shared KV pool is sized to `n_parallel * N`
(env: LLAMA_ARG_CTX_PER_SLOT) |
| `-n, --predict, --n-predict N` | number of tokens to predict (default: -1, -1 = infinity)
(env: LLAMA_ARG_N_PREDICT) |
| `-b, --batch-size N` | logical maximum batch size (default: 2048)
(env: LLAMA_ARG_BATCH) |
| `-ub, --ubatch-size N` | physical maximum batch size (default: 512)
(env: LLAMA_ARG_UBATCH) |
@@ -164,6 +163,7 @@ For the full list of features, please refer to [server's changelog](https://gith
| -------- | ----------- |
| `-lcs, --lookup-cache-static FNAME` | path to static lookup cache to use for lookup decoding (not updated by generation) |
| `-lcd, --lookup-cache-dynamic FNAME` | path to dynamic lookup cache to use for lookup decoding (updated by generation) |
+| `--kv-unified-per-slot N` | context limit per parallel slot (default: unset, behavior unchanged).
when set without -c/--ctx-size, the shared KV pool is sized to n_parallel*N
(env: LLAMA_ARG_KV_UNIFIED_PER_SLOT) |
| `-ctxcp, --ctx-checkpoints, --swa-checkpoints N` | max number of context checkpoints to create per slot (default: 32)[(more info)](https://github.com/ggml-org/llama.cpp/pull/15293)
(env: LLAMA_ARG_CTX_CHECKPOINTS) |
| `-cms, --checkpoint-min-step N` | minimum spacing between context checkpoints in tokens (default: 8192, 0 = no minimum)
(env: LLAMA_ARG_CHECKPOINT_MIN_SPACING_NT) |
| `-cram, --cache-ram N` | set the maximum cache size in MiB (default: 8192, -1 - no limit, 0 - disable)[(more info)](https://github.com/ggml-org/llama.cpp/pull/16391)
(env: LLAMA_ARG_CACHE_RAM) |
diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp
index d66fe7548195..f5477356d61d 100644
--- a/tools/server/server-context.cpp
+++ b/tools/server/server-context.cpp
@@ -1208,24 +1208,31 @@ struct server_context_impl {
const int n_ctx_train = llama_model_n_ctx_train(model_tgt);
- int n_ctx_slot = llama_n_ctx_seq(ctx_tgt);
- if (params_base.n_ctx_per_slot > 0) {
- if (n_ctx_slot > params_base.n_ctx_per_slot) {
- SRV_INF("capping per-slot context (%d) to --ctx-per-slot (%d)\n",
- n_ctx_slot, params_base.n_ctx_per_slot);
- n_ctx_slot = params_base.n_ctx_per_slot;
- } else if (params_base.n_ctx_per_slot > n_ctx_slot) {
- // cap is above the per-slot pool capacity, so it can never bind
- SRV_WRN(
- "--ctx-per-slot (%d) exceeds the per-slot pool capacity (%d) - cap has no effect, "
- "slots are limited to %d (raise the KV pool with -c, or unset -c to size it to "
- "n_parallel * ctx_per_slot)\n",
- params_base.n_ctx_per_slot, n_ctx_slot, n_ctx_slot);
- }
- }
- if (n_ctx_slot > n_ctx_train) {
- SRV_WRN("the slot context (%d) exceeds the training context of the model (%d) - capping\n", n_ctx_slot, n_ctx_train);
- n_ctx_slot = n_ctx_train;
+ {
+ // note: the capping itself is done in n_ctx_slot(), here we only report it
+ const int n_ctx_seq = llama_n_ctx_seq(ctx_tgt);
+
+ if (params_base.kv_unified_per_slot > 0) {
+ if (n_ctx_seq > params_base.kv_unified_per_slot) {
+ SRV_INF("capping per-slot context (%d) to --kv-unified-per-slot (%d)\n",
+ n_ctx_seq, params_base.kv_unified_per_slot);
+ } else if (params_base.kv_unified_per_slot > n_ctx_seq) {
+ // cap is above the per-slot pool capacity, so it can never bind
+ SRV_WRN(
+ "--kv-unified-per-slot (%d) exceeds the per-slot pool capacity (%d) - cap has no effect, "
+ "slots are limited to %d (raise the KV pool with -c, or unset -c to size it to "
+ "n_parallel * kv_unified_per_slot)\n",
+ params_base.kv_unified_per_slot, n_ctx_seq, n_ctx_seq);
+ }
+ }
+
+ const int n_ctx_capped = params_base.kv_unified_per_slot > 0 ?
+ std::min(n_ctx_seq, params_base.kv_unified_per_slot) : n_ctx_seq;
+
+ if (n_ctx_capped > n_ctx_train) {
+ SRV_WRN("the slot context (%d) exceeds the training context of the model (%d) - capping\n",
+ n_ctx_capped, n_ctx_train);
+ }
}
slots.clear();
@@ -1241,7 +1248,7 @@ struct server_context_impl {
// setup slots
SRV_INF("initializing, n_slots = %d, n_ctx_slot = %d, kv_unified = '%s'\n",
- params_base.n_parallel, n_ctx_slot, params_base.kv_unified ? "true" : "false");
+ params_base.n_parallel, n_ctx_slot(), params_base.kv_unified ? "true" : "false");
// initialize slots
for (int i = 0; i < params_base.n_parallel; i++) {
@@ -1285,7 +1292,7 @@ struct server_context_impl {
slot.ctx_dft = ctx_dft;
slot.mem.init(ctx_tgt, ctx_dft);
slot.spec = spec.get();
- slot.n_ctx = n_ctx_slot;
+ slot.n_ctx = n_ctx_slot();
slot.mctx = mctx;
slot.prompt.tokens.has_mtmd = mctx != nullptr;
@@ -3989,8 +3996,15 @@ struct server_context_impl {
});
}
- int get_slot_n_ctx() {
- return slots.back().n_ctx;
+ // context size of a single slot, capped by --kv-unified-per-slot and by the training context of the model
+ int n_ctx_slot() const {
+ int res = llama_n_ctx_seq(ctx_tgt);
+
+ if (params_base.kv_unified_per_slot > 0) {
+ res = std::min(res, params_base.kv_unified_per_slot);
+ }
+
+ return std::min(res, llama_model_n_ctx_train(model_tgt));
}
server_response_reader get_response_reader() {
@@ -4156,7 +4170,7 @@ server_context_meta server_context::get_meta() const {
/* has_inp_audio */ impl->chat_params.allow_audio,
/* has_inp_video */ impl->chat_params.allow_video,
/* json_ui_settings */ impl->json_ui_settings,
- /* slot_n_ctx */ impl->get_slot_n_ctx(),
+ /* slot_n_ctx */ impl->n_ctx_slot(),
/* pooling_type */ llama_pooling_type(impl->ctx_tgt),
/* chat_params */ impl->chat_params,
diff --git a/tools/server/server.cpp b/tools/server/server.cpp
index c8038cbc0a5f..22378b38c5ef 100644
--- a/tools/server/server.cpp
+++ b/tools/server/server.cpp
@@ -157,16 +157,16 @@ int llama_server(common_params & params, int argc, char ** argv) {
}
}
- // size the KV pool from --ctx-per-slot, unless the user pinned it with -c
+ // size the KV pool from --kv-unified-per-slot, unless the user pinned it with -c
// or with -c 0 for max context
- const bool ctx_pool_auto_sized = params.n_ctx_per_slot > 0 &&
+ const bool ctx_pool_auto_sized = params.kv_unified_per_slot > 0 &&
params.n_ctx == 0 &&
(uint32_t) params.fit_params_min_ctx != UINT32_MAX;
if (ctx_pool_auto_sized) {
- params.n_ctx = params.n_parallel * params.n_ctx_per_slot;
- SRV_INF("--ctx-per-slot: sizing KV pool to n_parallel * ctx_per_slot = %d * %d = %d\n", params.n_parallel,
- params.n_ctx_per_slot, params.n_ctx);
+ params.n_ctx = params.n_parallel * params.kv_unified_per_slot;
+ SRV_INF("--kv-unified-per-slot: sizing KV pool to n_parallel * kv_unified_per_slot = %d * %d = %d\n", params.n_parallel,
+ params.kv_unified_per_slot, params.n_ctx);
}
// for consistency between server router mode and single-model mode, we set the same model name as alias