From: Yongyue Sun Date: Thu, 4 Jun 2026 13:09:01 +0000 (+0800) Subject: server: avoid unnecessary checkpoint restore when new tokens are present (#24110) X-Git-Tag: upstream/0.0.10438~929 X-Git-Url: https://git.djapps.eu/?a=commitdiff_plain;h=6f3a9f3dee3c27545371044a3a38005721ac8a8e;p=pkg%2Fggml%2Fsources%2Fllama.cpp server: avoid unnecessary checkpoint restore when new tokens are present (#24110) * server: avoid unnecessary checkpoint restore when new tokens are present The pos_min_thold calculation unconditionally subtracts 1 to ensure at least one token is evaluated for logits when no new tokens exist. However, when the request contains new tokens beyond the cached prefix, this -1 is overly conservative and may trigger an unnecessary checkpoint restore. Conditionally apply the -1 only when n_past >= task.n_tokens() (no new tokens), avoiding redundant KV state restoration when there is actual work to do. * cont : add ref --------- Co-authored-by: Georgi Gerganov --- diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp index 28b1158c7..28f738c3f 100644 --- a/tools/server/server-context.cpp +++ b/tools/server/server-context.cpp @@ -2782,8 +2782,11 @@ private: llama_pos pos_next = slot.prompt.tokens.pos_next(n_past); + // ref: https://github.com/ggml-org/llama.cpp/pull/24110 + const bool has_new_tokens = (n_past < slot.task->n_tokens()); + // the largest pos_min required for a checkpoint to be useful - const auto pos_min_thold = std::max(0, pos_next - n_swa - 1); + const auto pos_min_thold = std::max(0, pos_next - n_swa - (has_new_tokens ? 0 : 1)); if (n_past > 0 && n_past <= slot.prompt.n_tokens()) { const auto pos_min = llama_memory_seq_pos_min(llama_get_memory(ctx_tgt), slot.id);