From: Oliver Simons Date: Tue, 4 Aug 2026 18:28:55 +0000 (+0200) Subject: sampler : remove "full-context windows" from history-based samplers (#26524) X-Git-Tag: upstream/0.0.10438~165 X-Git-Url: https://git.djapps.eu/?a=commitdiff_plain;h=a6aa6f5450eaad18b3c86631b5c3fff330f5a46e;p=pkg%2Fggml%2Fsources%2Fllama.cpp sampler : remove "full-context windows" from history-based samplers (#26524) * Resolve -1 to 1024 instead of ctx-len for samplers Because of backend-sampling we initialize samplers before the complete llama_context is there. Therefore, we cannot infer the resolved context length yet at the time we construct the samplers. * Shared default of 64 for history-based samplers, remove context_size --- diff --git a/common/arg.cpp b/common/arg.cpp index 86af0ba10..da4087474 100644 --- a/common/arg.cpp +++ b/common/arg.cpp @@ -2008,9 +2008,9 @@ common_params_context common_params_parser_init(common_params & params, llama_ex ).set_sampling()); add_opt(common_arg( {"--repeat-last-n"}, "N", - string_format("last n tokens to consider for penalize (default: %d, 0 = disabled, -1 = ctx_size)", params.sampling.penalty_last_n), + string_format("last n tokens to consider for penalize (default: %d, 0 = disabled)", params.sampling.penalty_last_n), [](common_params & params, int value) { - if (value < -1) { + if (value < 0) { throw std::runtime_error(string_format("error: invalid repeat-last-n = %d\n", value)); } params.sampling.penalty_last_n = value; @@ -2081,9 +2081,9 @@ common_params_context common_params_parser_init(common_params & params, llama_ex ).set_sampling()); add_opt(common_arg( {"--dry-penalty-last-n"}, "N", - string_format("set DRY penalty for the last n tokens (default: %d, 0 = disable, -1 = context size)", params.sampling.dry_penalty_last_n), + string_format("set DRY penalty for the last n tokens (default: %d, 0 = disable)", params.sampling.dry_penalty_last_n), [](common_params & params, int value) { - if (value < -1) { + if (value < 0) { throw std::runtime_error(string_format("error: invalid dry-penalty-last-n = %d\n", value)); } params.sampling.dry_penalty_last_n = value; diff --git a/common/common.cpp b/common/common.cpp index d9ce57551..ffe3e7761 100644 --- a/common/common.cpp +++ b/common/common.cpp @@ -1302,23 +1302,12 @@ common_init_result::common_init_result(common_params & params, bool model_only) params.sampling.logit_bias_eog.begin(), params.sampling.logit_bias_eog.end()); } - //if (params.sampling.penalty_last_n == -1) { - // LOG_TRC("%s: setting penalty_last_n to ctx_size = %d\n", __func__, llama_n_ctx(lctx)); - // params.sampling.penalty_last_n = llama_n_ctx(lctx); - //} - - //if (params.sampling.dry_penalty_last_n == -1) { - // LOG_TRC("%s: setting dry_penalty_last_n to ctx_size = %d\n", __func__, llama_n_ctx(lctx)); - // params.sampling.dry_penalty_last_n = llama_n_ctx(lctx); - //} - // init the backend samplers as part of the context creation pimpl->samplers.resize(cparams.n_seq_max); pimpl->samplers_seq_config.resize(cparams.n_seq_max); - const int32_t n_ctx = cparams.n_ctx > 0 ? (int32_t) cparams.n_ctx : llama_model_n_ctx_train(model); for (int i = 0; i < (int) cparams.n_seq_max; ++i) { - pimpl->samplers[i].reset(common_sampler_init(model, params.sampling, n_ctx)); + pimpl->samplers[i].reset(common_sampler_init(model, params.sampling)); pimpl->samplers_seq_config[i] = { i, common_sampler_get(pimpl->samplers[i].get()) }; } diff --git a/common/common.h b/common/common.h index 3444aa157..2e15ec3f8 100644 --- a/common/common.h +++ b/common/common.h @@ -235,14 +235,14 @@ struct common_params_sampling { float temp = 0.80f; // <= 0.0 to sample greedily, 0.0 to not output probabilities float dynatemp_range = 0.00f; // 0.0 = disabled float dynatemp_exponent = 1.00f; // controls how entropy maps to temperature in dynamic temperature sampler - int32_t penalty_last_n = 64; // last n tokens to penalize (0 = disable penalty, -1 = context size) + int32_t penalty_last_n = 64; // last n tokens to penalize (0 = disable penalty) float penalty_repeat = 1.00f; // 1.0 = disabled float penalty_freq = 0.00f; // 0.0 = disabled float penalty_present = 0.00f; // 0.0 = disabled float dry_multiplier = 0.0f; // 0.0 = disabled; DRY repetition penalty for tokens extending repetition: float dry_base = 1.75f; // 0.0 = disabled; multiplier * base ^ (length of sequence before token - allowed length) int32_t dry_allowed_length = 2; // tokens extending repetitions beyond this receive penalty - int32_t dry_penalty_last_n = -1; // how many tokens to scan for repetitions (0 = disable penalty, -1 = context size) + int32_t dry_penalty_last_n = 64; // how many tokens to scan for repetitions (0 = disable penalty) float adaptive_target = -1.0f; // select tokens near this probability (valid range 0.0 to 1.0; negative = disabled) float adaptive_decay = 0.90f; // EMA decay for adaptation; history ≈ 1/(1-decay) tokens (0.0 - 0.99) int32_t mirostat = 0; // 0 = disabled, 1 = mirostat, 2 = mirostat 2.0 diff --git a/common/sampling.cpp b/common/sampling.cpp index ba5504ed0..ec9c885dd 100644 --- a/common/sampling.cpp +++ b/common/sampling.cpp @@ -186,8 +186,7 @@ std::string common_params_sampling::print() const { struct common_sampler * common_sampler_init( const struct llama_model * model, - struct common_params_sampling & params, - int32_t n_ctx) { + struct common_params_sampling & params) { if (!std::isfinite(params.penalty_repeat) || params.penalty_repeat <= 0.0f || !std::isfinite(1.0f/params.penalty_repeat)) { @@ -199,10 +198,6 @@ struct common_sampler * common_sampler_init( if (!std::isfinite(params.penalty_present)) { throw std::invalid_argument("penalty_present must be finite"); } - if (params.penalty_last_n == -1) { - params.penalty_last_n = n_ctx > 0 ? n_ctx : llama_model_n_ctx_train(model); - } - const llama_vocab * vocab = llama_model_get_vocab(model); llama_sampler_chain_params lparams = llama_sampler_chain_default_params(); @@ -355,7 +350,7 @@ struct common_sampler * common_sampler_init( for (const auto & str : params.dry_sequence_breakers) { c_breakers.push_back(str.c_str()); } - samplers.push_back(llama_sampler_init_dry(vocab, llama_model_n_ctx_train(model), params.dry_multiplier, params.dry_base, params.dry_allowed_length, params.dry_penalty_last_n, c_breakers.data(), c_breakers.size())); + samplers.push_back(llama_sampler_init_dry(vocab, params.dry_multiplier, params.dry_base, params.dry_allowed_length, params.dry_penalty_last_n, c_breakers.data(), c_breakers.size())); } break; case COMMON_SAMPLER_TYPE_TOP_K: diff --git a/common/sampling.h b/common/sampling.h index 91e2cea78..cb90d4ae7 100644 --- a/common/sampling.h +++ b/common/sampling.h @@ -39,8 +39,7 @@ struct common_sampler; // note: can mutate params in some cases struct common_sampler * common_sampler_init( const struct llama_model * model, - struct common_params_sampling & params, - int32_t n_ctx = 0); + struct common_params_sampling & params); void common_sampler_free(struct common_sampler * gsmpl); diff --git a/include/llama.h b/include/llama.h index fb2ca38ce..a14498925 100644 --- a/include/llama.h +++ b/include/llama.h @@ -1425,7 +1425,7 @@ extern "C" { /// NOTE: Avoid using on the full vocabulary as searching for repeated tokens can become slow. For example, apply top-k or top-p sampling first. LLAMA_API struct llama_sampler * llama_sampler_init_penalties( int32_t n_vocab, - int32_t penalty_last_n, // last n tokens to penalize (0 = disable penalty, -1 = context size) + int32_t penalty_last_n, // last n tokens to penalize (0 = disable penalty) float penalty_repeat, // must be > 0.0, 1.0 = disabled float penalty_freq, // must be finite, 0.0 = disabled float penalty_present); // must be finite, 0.0 = disabled @@ -1433,11 +1433,10 @@ extern "C" { /// @details DRY sampler, designed by p-e-w, as described in: https://github.com/oobabooga/text-generation-webui/pull/5677, porting Koboldcpp implementation authored by pi6am: https://github.com/LostRuins/koboldcpp/pull/982 LLAMA_API struct llama_sampler * llama_sampler_init_dry( const struct llama_vocab * vocab, - int32_t n_ctx_train, float dry_multiplier, float dry_base, int32_t dry_allowed_length, - int32_t dry_penalty_last_n, + int32_t dry_penalty_last_n, // last n tokens to penalize (0 = disable penalty) const char ** seq_breakers, size_t num_breakers); diff --git a/src/llama-sampler.cpp b/src/llama-sampler.cpp index 6cf2d27cf..e550fbe4a 100644 --- a/src/llama-sampler.cpp +++ b/src/llama-sampler.cpp @@ -3078,8 +3078,6 @@ struct llama_sampler * llama_sampler_init_top_n_sigma(float n) { // DRY struct llama_sampler_dry { - int32_t total_context_size; - const float dry_multiplier; const float dry_base; const int32_t dry_allowed_length; @@ -3155,8 +3153,7 @@ static void llama_sampler_dry_apply(struct llama_sampler * smpl, llama_token_dat return; } - int32_t effective_dry_penalty_last_n = (ctx->dry_penalty_last_n == -1) ? ctx->total_context_size : std::max(ctx->dry_penalty_last_n, 0); - int last_n_repeat = std::min(std::min((int)ctx->last_tokens.size(), effective_dry_penalty_last_n), ctx->total_context_size); + int last_n_repeat = std::min((int) ctx->last_tokens.size(), ctx->dry_penalty_last_n); if (last_n_repeat <= ctx->dry_allowed_length) { return; @@ -3369,7 +3366,7 @@ static struct llama_sampler * llama_sampler_dry_clone(const struct llama_sampler llama_vocab dummy_vocab; // dummy vocab is passed because it is only needed for raw sequence breaker processing, which we have already done and will simply be copying - auto * result = llama_sampler_init_dry(&dummy_vocab, ctx->total_context_size, ctx->dry_multiplier, ctx->dry_base, ctx->dry_allowed_length, ctx->dry_penalty_last_n, NULL, 0); + auto * result = llama_sampler_init_dry(&dummy_vocab, ctx->dry_multiplier, ctx->dry_base, ctx->dry_allowed_length, ctx->dry_penalty_last_n, NULL, 0); // Copy the state, including the processed breakers { @@ -3400,8 +3397,8 @@ static struct llama_sampler_i llama_sampler_dry_i = { /* .backend_set_input = */ nullptr, }; -struct llama_sampler * llama_sampler_init_dry(const struct llama_vocab * vocab, int32_t n_ctx_train, float dry_multiplier, float dry_base, int32_t dry_allowed_length, int32_t dry_penalty_last_n, const char** seq_breakers, size_t num_breakers) { - int32_t effective_dry_penalty_last_n = (dry_penalty_last_n == -1) ? n_ctx_train : std::max(dry_penalty_last_n, 0); +struct llama_sampler * llama_sampler_init_dry(const struct llama_vocab * vocab, float dry_multiplier, float dry_base, int32_t dry_allowed_length, int32_t dry_penalty_last_n, const char** seq_breakers, size_t num_breakers) { + dry_penalty_last_n = std::max(dry_penalty_last_n, 0); std::unordered_multimap> processed_breakers; const int MAX_CHAR_LEN = 40; const int MAX_SEQ_LEN = 20; @@ -3438,23 +3435,22 @@ struct llama_sampler * llama_sampler_init_dry(const struct llama_vocab * vocab, return llama_sampler_init( /* .iface = */ &llama_sampler_dry_i, /* .ctx = */ new llama_sampler_dry { - /* .total_context_size = */ n_ctx_train, /* .dry_multiplier = */ dry_multiplier, /* .dry_base = */ dry_base, /* .dry_allowed_length = */ dry_allowed_length, /* .dry_penalty_last_n = */ dry_penalty_last_n, /* .dry_processed_breakers = */ std::move(processed_breakers), - /* .dry_repeat_count = */ dry_enabled ? std::vector(effective_dry_penalty_last_n, 0) : std::vector{}, + /* .dry_repeat_count = */ dry_enabled ? std::vector(dry_penalty_last_n, 0) : std::vector{}, /* .dry_max_token_repeat = */ {}, - /* .last_tokens = */ dry_enabled ? ring_buffer(effective_dry_penalty_last_n) : ring_buffer(0), + /* .last_tokens = */ dry_enabled ? ring_buffer(dry_penalty_last_n) : ring_buffer(0), } ); } // wrapper for test-sampling.cpp -struct llama_sampler * llama_sampler_init_dry_testing(int32_t context_size, float dry_multiplier, float dry_base, int32_t dry_allowed_length, int32_t dry_penalty_last_n, const std::vector>& seq_breakers) { +struct llama_sampler * llama_sampler_init_dry_testing(float dry_multiplier, float dry_base, int32_t dry_allowed_length, int32_t dry_penalty_last_n, const std::vector>& seq_breakers) { llama_vocab dummy_vocab; - auto * result = llama_sampler_init_dry(&dummy_vocab, context_size, dry_multiplier, dry_base, dry_allowed_length, dry_penalty_last_n, NULL, 0); + auto * result = llama_sampler_init_dry(&dummy_vocab, dry_multiplier, dry_base, dry_allowed_length, dry_penalty_last_n, NULL, 0); auto * ctx = (llama_sampler_dry *) result->ctx; // Process the token-based sequence breakers diff --git a/src/llama-sampler.h b/src/llama-sampler.h index b9bfc20d2..929207514 100644 --- a/src/llama-sampler.h +++ b/src/llama-sampler.h @@ -34,7 +34,6 @@ struct llama_sampler_chain { }; struct llama_sampler * llama_sampler_init_dry_testing( - int32_t context_size, float dry_multiplier, float dry_base, int32_t dry_allowed_length, diff --git a/tests/test-arg-parser.cpp b/tests/test-arg-parser.cpp index fd5adb740..50db29727 100644 --- a/tests/test-arg-parser.cpp +++ b/tests/test-arg-parser.cpp @@ -101,6 +101,14 @@ static void test(void) { { common_params penalty_params; + assert(penalty_params.sampling.penalty_last_n == 64); + assert(penalty_params.sampling.dry_penalty_last_n == 64); + + argv = {"binary_name", "--repeat-last-n", "-1"}; + assert(false == common_params_parse(argv.size(), list_str_to_char(argv).data(), penalty_params, LLAMA_EXAMPLE_COMMON)); + + argv = {"binary_name", "--dry-penalty-last-n", "-1"}; + assert(false == common_params_parse(argv.size(), list_str_to_char(argv).data(), penalty_params, LLAMA_EXAMPLE_COMMON)); argv = {"binary_name", "--repeat-penalty", "0"}; assert(false == common_params_parse(argv.size(), list_str_to_char(argv).data(), penalty_params, LLAMA_EXAMPLE_COMMON)); diff --git a/tests/test-sampling.cpp b/tests/test-sampling.cpp index 297f76015..df1eb1a20 100644 --- a/tests/test-sampling.cpp +++ b/tests/test-sampling.cpp @@ -10,7 +10,7 @@ #include #include -extern struct llama_sampler * llama_sampler_init_dry_testing(int32_t context_size, float dry_multiplier, float dry_base, int32_t dry_allowed_length, int32_t dry_penalty_last_n, const std::vector>& seq_breakers); +extern struct llama_sampler * llama_sampler_init_dry_testing(float dry_multiplier, float dry_base, int32_t dry_allowed_length, int32_t dry_penalty_last_n, const std::vector>& seq_breakers); static void dump(const llama_token_data_array * cur_p) { for (size_t i = 0; i < cur_p->size; i++) { @@ -168,7 +168,7 @@ static void test_dry( sampler_tester tester(probs, expected_probs); - auto * sampler = llama_sampler_init_dry_testing(1024, dry_multiplier, dry_base, dry_allowed_length, dry_penalty_last_n, seq_breakers); + auto * sampler = llama_sampler_init_dry_testing(dry_multiplier, dry_base, dry_allowed_length, dry_penalty_last_n, seq_breakers); for (size_t i = 0; i < last_tokens.size(); i++) { llama_sampler_accept(sampler, last_tokens[i]); diff --git a/tools/cli/README.md b/tools/cli/README.md index bcddd0570..4d86ce7c0 100644 --- a/tools/cli/README.md +++ b/tools/cli/README.md @@ -116,14 +116,14 @@ | `--xtc-probability N` | xtc probability (default: 0.00, 0.0 = disabled) | | `--xtc-threshold N` | xtc threshold (default: 0.10, 1.0 = disabled) | | `--typical, --typical-p N` | locally typical sampling, parameter p (default: 1.00, 1.0 = disabled) | -| `--repeat-last-n N` | last n tokens to consider for penalize (default: 64, 0 = disabled, -1 = ctx_size) | +| `--repeat-last-n N` | last n tokens to consider for penalize (default: 64, 0 = disabled) | | `--repeat-penalty N` | penalize repeat sequence of tokens (default: 1.00, 1.0 = disabled) | | `--presence-penalty N` | repeat alpha presence penalty (default: 0.00, 0.0 = disabled) | | `--frequency-penalty N` | repeat alpha frequency penalty (default: 0.00, 0.0 = disabled) | | `--dry-multiplier N` | set DRY sampling multiplier (default: 0.00, 0.0 = disabled) | | `--dry-base N` | set DRY sampling base value (default: 1.75) | | `--dry-allowed-length N` | set allowed length for DRY sampling (default: 2) | -| `--dry-penalty-last-n N` | set DRY penalty for the last n tokens (default: -1, 0 = disable, -1 = context size) | +| `--dry-penalty-last-n N` | set DRY penalty for the last n tokens (default: 64, 0 = disable) | | `--dry-sequence-breaker STRING` | add sequence breaker for DRY sampling, clearing out default breakers ('\n', ':', '"', '*') in the process; use "none" to not use any sequence breakers | | `--adaptive-target N` | adaptive-p: select tokens near this probability (valid range 0.0 to 1.0; negative = disabled) (default: -1.00)
[(more info)](https://github.com/ggml-org/llama.cpp/pull/17927) | | `--adaptive-decay N` | adaptive-p: decay rate for target adaptation over time. lower values are more reactive, higher values are more stable.
(valid range 0.0 to 0.99) (default: 0.90) | diff --git a/tools/completion/README.md b/tools/completion/README.md index bce71d68d..2abe7aaa2 100644 --- a/tools/completion/README.md +++ b/tools/completion/README.md @@ -199,14 +199,14 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1 | `--xtc-probability N` | xtc probability (default: 0.00, 0.0 = disabled) | | `--xtc-threshold N` | xtc threshold (default: 0.10, 1.0 = disabled) | | `--typical, --typical-p N` | locally typical sampling, parameter p (default: 1.00, 1.0 = disabled) | -| `--repeat-last-n N` | last n tokens to consider for penalize (default: 64, 0 = disabled, -1 = ctx_size) | +| `--repeat-last-n N` | last n tokens to consider for penalize (default: 64, 0 = disabled) | | `--repeat-penalty N` | penalize repeat sequence of tokens (default: 1.00, 1.0 = disabled) | | `--presence-penalty N` | repeat alpha presence penalty (default: 0.00, 0.0 = disabled) | | `--frequency-penalty N` | repeat alpha frequency penalty (default: 0.00, 0.0 = disabled) | | `--dry-multiplier N` | set DRY sampling multiplier (default: 0.00, 0.0 = disabled) | | `--dry-base N` | set DRY sampling base value (default: 1.75) | | `--dry-allowed-length N` | set allowed length for DRY sampling (default: 2) | -| `--dry-penalty-last-n N` | set DRY penalty for the last n tokens (default: -1, 0 = disable, -1 = context size) | +| `--dry-penalty-last-n N` | set DRY penalty for the last n tokens (default: 64, 0 = disable) | | `--dry-sequence-breaker STRING` | add sequence breaker for DRY sampling, clearing out default breakers ('\n', ':', '"', '*') in the process; use "none" to not use any sequence breakers | | `--adaptive-target N` | adaptive-p: select tokens near this probability (valid range 0.0 to 1.0; negative = disabled) (default: -1.00)
[(more info)](https://github.com/ggml-org/llama.cpp/pull/17927) | | `--adaptive-decay N` | adaptive-p: decay rate for target adaptation over time. lower values are more reactive, higher values are more stable.
(valid range 0.0 to 0.99) (default: 0.90) | @@ -388,11 +388,11 @@ Example usage: `--temp 0` ### Repeat Penalty - `--repeat-penalty N`: Control the repetition of token sequences in the generated text default: 1.0, 1.0 = disabled). -- `--repeat-last-n N`: Last n tokens to consider for penalizing repetition (default: 64, 0 = disabled, -1 = ctx-size). +- `--repeat-last-n N`: Last n tokens to consider for penalizing repetition (default: 64, 0 = disabled). The `repeat-penalty` option helps prevent the model from generating repetitive or monotonous text. A higher value (e.g., 1.5) will penalize repetitions more strongly, while a lower value (e.g., 0.9) will be more lenient. The default value is 1. -The `repeat-last-n` option controls the number of tokens in the history to consider for penalizing repetition. A larger value will look further back in the generated text to prevent repetitions, while a smaller value will only consider recent tokens. A value of 0 disables the penalty, and a value of -1 sets the number of tokens considered equal to the context size (`ctx-size`). +The `repeat-last-n` option controls the number of tokens in the history to consider for penalizing repetition. A larger value will look further back in the generated text to prevent repetitions, while a smaller value will only consider recent tokens. A value of 0 disables the penalty. ### DRY Repetition Penalty @@ -401,7 +401,7 @@ DRY (Don't Repeat Yourself) sampling is an effective technique for reducing repe - `--dry-multiplier N`: Set the DRY sampling multiplier (default: 0.0, 0.0 = disabled). - `--dry-base N`: Set the DRY sampling base value (default: 1.75). - `--dry-allowed-length N`: Set the allowed length for DRY sampling (default: 2). -- `--dry-penalty-last-n N`: Set DRY penalty for the last n tokens (default: -1, 0 = disable, -1 = context size). +- `--dry-penalty-last-n N`: Set DRY penalty for the last n tokens (default: 64, 0 = disable). - `--dry-sequence-breaker STRING`: Add a sequence breaker for DRY sampling. Can be used more than once to add multiple sequence breakers. Using this clears out the default breakers, which consist of: `['\n', ':', '"', '*']`. If the string `"none"` is supplied, no sequence breakers are used. The `dry-multiplier` option controls the strength of the DRY sampling effect. A value of 0.0 disables DRY sampling, while higher values increase its influence. A typical recommended value is 0.8. @@ -410,13 +410,13 @@ The `dry-base` option sets the base value for the exponential penalty calculatio The `dry-allowed-length` option sets the maximum length of repeated sequences that will not be penalized. Repetitions shorter than or equal to this length are not penalized, allowing for natural repetitions of short phrases or common words. -The `dry-penalty-last-n` option controls how many recent tokens to consider when applying the DRY penalty. A value of -1 considers the entire context. Use a positive value to limit the consideration to a specific number of recent tokens. +The `dry-penalty-last-n` option controls how many recent tokens to consider when applying the DRY penalty. A value of 0 disables the penalty. Use a positive value to limit the consideration to a specific number of recent tokens. The `dry-sequence-breaker` option adds a single sequence breaker and can be used more than once to specify multiple sequence breakers. Sequence breakers interrupt sequence matching and break the input into parts where matching can be applied. DRY sampling provides more nuanced control over text generation, particularly for reducing long-range repetitions and maintaining global coherence. -Example usage: `--dry-multiplier 0.8 --dry-base 1.75 --dry-allowed-length 2 --dry-penalty-last-n -1 --dry-sequence-breaker "—" --dry-sequence-breaker "##"` +Example usage: `--dry-multiplier 0.8 --dry-base 1.75 --dry-allowed-length 2 --dry-penalty-last-n 64 --dry-sequence-breaker "—" --dry-sequence-breaker "##"` ### Top-K Sampling diff --git a/tools/mtmd/tests/test-deepseek-ocr.py b/tools/mtmd/tests/test-deepseek-ocr.py index 8a9640550..f1edaebd8 100644 --- a/tools/mtmd/tests/test-deepseek-ocr.py +++ b/tools/mtmd/tests/test-deepseek-ocr.py @@ -215,7 +215,7 @@ def run_mtmd_cli(spec: "ModelSpec", model_path, mmproj_path, image_path, bin_pat "--dry-multiplier", "0.8", "--dry-base", "1.75", "--dry-allowed-length", "2", - "--dry-penalty-last-n", "-1", + "--dry-penalty-last-n", "64", "--dry-sequence-breaker", "none", ] if spec.n_ctx is not None: diff --git a/tools/server/README.md b/tools/server/README.md index a0956f9e6..c8e7fcdcf 100644 --- a/tools/server/README.md +++ b/tools/server/README.md @@ -133,14 +133,14 @@ For the full list of features, please refer to [server's changelog](https://gith | `--xtc-probability N` | xtc probability (default: 0.00, 0.0 = disabled) | | `--xtc-threshold N` | xtc threshold (default: 0.10, 1.0 = disabled) | | `--typical, --typical-p N` | locally typical sampling, parameter p (default: 1.00, 1.0 = disabled) | -| `--repeat-last-n N` | last n tokens to consider for penalize (default: 64, 0 = disabled, -1 = ctx_size) | +| `--repeat-last-n N` | last n tokens to consider for penalize (default: 64, 0 = disabled) | | `--repeat-penalty N` | penalize repeat sequence of tokens (default: 1.00, 1.0 = disabled) | | `--presence-penalty N` | repeat alpha presence penalty (default: 0.00, 0.0 = disabled) | | `--frequency-penalty N` | repeat alpha frequency penalty (default: 0.00, 0.0 = disabled) | | `--dry-multiplier N` | set DRY sampling multiplier (default: 0.00, 0.0 = disabled) | | `--dry-base N` | set DRY sampling base value (default: 1.75) | | `--dry-allowed-length N` | set allowed length for DRY sampling (default: 2) | -| `--dry-penalty-last-n N` | set DRY penalty for the last n tokens (default: -1, 0 = disable, -1 = context size) | +| `--dry-penalty-last-n N` | set DRY penalty for the last n tokens (default: 64, 0 = disable) | | `--dry-sequence-breaker STRING` | add sequence breaker for DRY sampling, clearing out default breakers ('\n', ':', '"', '*') in the process; use "none" to not use any sequence breakers | | `--adaptive-target N` | adaptive-p: select tokens near this probability (valid range 0.0 to 1.0; negative = disabled) (default: -1.00)
[(more info)](https://github.com/ggml-org/llama.cpp/pull/17927) | | `--adaptive-decay N` | adaptive-p: decay rate for target adaptation over time. lower values are more reactive, higher values are more stable.
(valid range 0.0 to 0.99) (default: 0.90) | @@ -476,7 +476,7 @@ These words will not be included in the completion, so make sure to add them to `repeat_penalty`: Control the repetition of token sequences in the generated text. Default: `1.1` -`repeat_last_n`: Last n tokens to consider for penalizing repetition. Default: `64`, where `0` is disabled and `-1` is ctx-size. +`repeat_last_n`: Last n tokens to consider for penalizing repetition. Default: `64`, where `0` is disabled. `presence_penalty`: Repeat alpha presence penalty. Default: `0.0`, which is disabled. @@ -488,7 +488,7 @@ These words will not be included in the completion, so make sure to add them to `dry_allowed_length`: Tokens that extend repetition beyond this receive exponentially increasing penalty: multiplier * base ^ (length of repeating sequence before token - allowed length). Default: `2` -`dry_penalty_last_n`: How many tokens to scan for repetitions. Default: `-1`, where `0` is disabled and `-1` is context size. +`dry_penalty_last_n`: How many tokens to scan for repetitions. Default: `64`, where `0` is disabled. `dry_sequence_breakers`: Specify an array of sequence breakers for DRY sampling. Only a JSON array of strings is accepted. Default: `['\n', ':', '"', '*']` @@ -796,7 +796,7 @@ By default, it is read-only. To make POST request to change global properties, y "dry_multiplier": 0.0, "dry_base": 1.75, "dry_allowed_length": 2, - "dry_penalty_last_n": -1, + "dry_penalty_last_n": 64, "dry_sequence_breakers": [ "\n", ":", diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp index 5d2798cc1..380e62af6 100644 --- a/tools/server/server-context.cpp +++ b/tools/server/server-context.cpp @@ -1807,8 +1807,7 @@ private: // initialize samplers if (task.need_sampling()) { try { - slot.smpl.reset(common_sampler_init( - model_tgt, task.params.sampling, (int32_t) llama_n_ctx(ctx_tgt))); + slot.smpl.reset(common_sampler_init(model_tgt, task.params.sampling)); } catch (std::exception & e) { std::string err_msg = std::string("Failed to initialize samplers: ") + e.what(); send_error(task, err_msg, ERROR_TYPE_INVALID_REQUEST); @@ -4148,7 +4147,6 @@ std::unique_ptr server_routes::handle_completions_impl( task.params = server_schema::eval_llama_cmpl_schema( ctx_server.vocab, params, - meta->slot_n_ctx, meta->logit_bias_eog, data); diff --git a/tools/server/server-schema.cpp b/tools/server/server-schema.cpp index 674d3ba33..5cef3908f 100644 --- a/tools/server/server-schema.cpp +++ b/tools/server/server-schema.cpp @@ -124,8 +124,8 @@ std::vector> make_llama_cmpl_schema(const common_params & ->set_desc("Dynamic temperature exponent, controls how entropy maps to temperature")); add((new field_num("repeat_last_n", params.sampling.penalty_last_n)) - ->set_hard_limits(-1, INT32_MAX) - ->set_desc("Last n tokens to consider for penalizing repetition (0 = disabled, -1 = ctx-size)")); + ->set_hard_limits(0, INT32_MAX) + ->set_desc("Last n tokens to consider for penalizing repetition (0 = disabled)")); add((new field_num("repeat_penalty", params.sampling.penalty_repeat)) ->set_desc("Control the repetition of token sequences in the generated text (1.0 = disabled)")); @@ -151,8 +151,8 @@ std::vector> make_llama_cmpl_schema(const common_params & ->set_desc("Tokens that extend repetition beyond this length receive exponentially increasing penalty: multiplier * base ^ (sequence_length - allowed_length)")); add((new field_num("dry_penalty_last_n", params.sampling.dry_penalty_last_n)) - ->set_hard_limits(-1, INT32_MAX) - ->set_desc("How many tokens to scan for repetitions (0 = disabled, -1 = context size)")); + ->set_hard_limits(0, INT32_MAX) + ->set_desc("How many tokens to scan for repetitions (0 = disabled)")); add((new field_num("mirostat", params.sampling.mirostat)) ->set_limits(0, 2) @@ -515,7 +515,6 @@ std::vector> make_llama_cmpl_schema(const common_params & task_params eval_llama_cmpl_schema( const llama_vocab * vocab, const common_params & params_base, - const int n_ctx_slot, const std::vector & logit_bias_eog, const json & data) { task_params params; @@ -549,15 +548,6 @@ task_params eval_llama_cmpl_schema( // post-processing { - if (params.sampling.penalty_last_n == -1) { - // note: should be the slot's context and not the full context, but it's ok - params.sampling.penalty_last_n = n_ctx_slot; - } - - if (params.sampling.dry_penalty_last_n == -1) { - params.sampling.dry_penalty_last_n = n_ctx_slot; - } - // if "reasoning_format" is not provided, its handler will not be called, we will need to handle it here auto reasoning_format = params.chat_parser_params.reasoning_format; params.chat_parser_params.reasoning_in_content = params.stream && (reasoning_format == COMMON_REASONING_FORMAT_DEEPSEEK_LEGACY); diff --git a/tools/server/server-schema.h b/tools/server/server-schema.h index 08cf427dc..d0a81431b 100644 --- a/tools/server/server-schema.h +++ b/tools/server/server-schema.h @@ -98,7 +98,6 @@ std::vector> make_llama_cmpl_schema( task_params eval_llama_cmpl_schema( const llama_vocab * vocab, const common_params & params_base, - const int n_ctx_slot, const std::vector & logit_bias_eog, const json & data); diff --git a/tools/ui/src/lib/services/parameter-sync.service.spec.ts b/tools/ui/src/lib/services/parameter-sync.service.spec.ts index 2e7fcd521..1cf1624df 100644 --- a/tools/ui/src/lib/services/parameter-sync.service.spec.ts +++ b/tools/ui/src/lib/services/parameter-sync.service.spec.ts @@ -30,7 +30,7 @@ describe('ParameterSyncService', () => { dry_multiplier: 0.0, dry_base: 1.75, dry_allowed_length: 2, - dry_penalty_last_n: -1, + dry_penalty_last_n: 64, mirostat: 0, mirostat_tau: 5.0, mirostat_eta: 0.1, @@ -96,7 +96,7 @@ describe('ParameterSyncService', () => { dry_multiplier: 0.0, dry_base: 1.75, dry_allowed_length: 2, - dry_penalty_last_n: -1, + dry_penalty_last_n: 64, mirostat: 0, mirostat_tau: 5.0, mirostat_eta: 0.1,