From 748eca633add709e1245a3a04064f06215da1246 Mon Sep 17 00:00:00 2001 From: Oliver Simons Date: Mon, 3 Aug 2026 16:13:07 +0200 Subject: [PATCH] Resolve -1 to 1024 instead of ctx-len for samplers Because of backend-sampling we initialize samplers before the complete llama_context is there. Therefore, we cannot infer the resolved context length yet at the time we construct the samplers. --- common/arg.cpp | 4 +-- common/common.cpp | 13 +------- common/common.h | 4 +-- common/sampling.cpp | 7 +---- common/sampling.h | 3 +- include/llama.h | 4 +-- src/llama-sampler.cpp | 11 +++---- tests/test-backend-sampler.cpp | 3 ++ tests/test-sampling.cpp | 54 ++++++++++++++++++++++++++++++++- tools/cli/README.md | 4 +-- tools/completion/README.md | 4 +-- tools/server/README.md | 4 +-- tools/server/server-context.cpp | 4 +-- tools/server/server-schema.cpp | 12 +------- tools/server/server-schema.h | 1 - 15 files changed, 78 insertions(+), 54 deletions(-) diff --git a/common/arg.cpp b/common/arg.cpp index b75f4f05f0..491a00c6bc 100644 --- a/common/arg.cpp +++ b/common/arg.cpp @@ -2023,7 +2023,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex ).set_sampling()); add_opt(common_arg( {"--repeat-last-n"}, "N", - string_format("last n tokens to consider for penalize (default: %d, 0 = disabled, -1 = ctx_size)", params.sampling.penalty_last_n), + string_format("last n tokens to consider for penalize (default: %d, 0 = disabled, -1 = 1024)", params.sampling.penalty_last_n), [](common_params & params, int value) { if (value < -1) { throw std::runtime_error(string_format("error: invalid repeat-last-n = %d\n", value)); @@ -2096,7 +2096,7 @@ common_params_context common_params_parser_init(common_params & params, llama_ex ).set_sampling()); add_opt(common_arg( {"--dry-penalty-last-n"}, "N", - string_format("set DRY penalty for the last n tokens (default: %d, 0 = disable, -1 = context size)", params.sampling.dry_penalty_last_n), + string_format("set DRY penalty for the last n tokens (default: %d, 0 = disable, -1 = 1024)", params.sampling.dry_penalty_last_n), [](common_params & params, int value) { if (value < -1) { throw std::runtime_error(string_format("error: invalid dry-penalty-last-n = %d\n", value)); diff --git a/common/common.cpp b/common/common.cpp index d9ce575516..ffe3e7761b 100644 --- a/common/common.cpp +++ b/common/common.cpp @@ -1302,23 +1302,12 @@ common_init_result::common_init_result(common_params & params, bool model_only) params.sampling.logit_bias_eog.begin(), params.sampling.logit_bias_eog.end()); } - //if (params.sampling.penalty_last_n == -1) { - // LOG_TRC("%s: setting penalty_last_n to ctx_size = %d\n", __func__, llama_n_ctx(lctx)); - // params.sampling.penalty_last_n = llama_n_ctx(lctx); - //} - - //if (params.sampling.dry_penalty_last_n == -1) { - // LOG_TRC("%s: setting dry_penalty_last_n to ctx_size = %d\n", __func__, llama_n_ctx(lctx)); - // params.sampling.dry_penalty_last_n = llama_n_ctx(lctx); - //} - // init the backend samplers as part of the context creation pimpl->samplers.resize(cparams.n_seq_max); pimpl->samplers_seq_config.resize(cparams.n_seq_max); - const int32_t n_ctx = cparams.n_ctx > 0 ? (int32_t) cparams.n_ctx : llama_model_n_ctx_train(model); for (int i = 0; i < (int) cparams.n_seq_max; ++i) { - pimpl->samplers[i].reset(common_sampler_init(model, params.sampling, n_ctx)); + pimpl->samplers[i].reset(common_sampler_init(model, params.sampling)); pimpl->samplers_seq_config[i] = { i, common_sampler_get(pimpl->samplers[i].get()) }; } diff --git a/common/common.h b/common/common.h index 78d0877566..4d959451ea 100644 --- a/common/common.h +++ b/common/common.h @@ -235,14 +235,14 @@ struct common_params_sampling { float temp = 0.80f; // <= 0.0 to sample greedily, 0.0 to not output probabilities float dynatemp_range = 0.00f; // 0.0 = disabled float dynatemp_exponent = 1.00f; // controls how entropy maps to temperature in dynamic temperature sampler - int32_t penalty_last_n = 64; // last n tokens to penalize (0 = disable penalty, -1 = context size) + int32_t penalty_last_n = 64; // last n tokens to penalize (0 = disable penalty, -1 = 1024) float penalty_repeat = 1.00f; // 1.0 = disabled float penalty_freq = 0.00f; // 0.0 = disabled float penalty_present = 0.00f; // 0.0 = disabled float dry_multiplier = 0.0f; // 0.0 = disabled; DRY repetition penalty for tokens extending repetition: float dry_base = 1.75f; // 0.0 = disabled; multiplier * base ^ (length of sequence before token - allowed length) int32_t dry_allowed_length = 2; // tokens extending repetitions beyond this receive penalty - int32_t dry_penalty_last_n = -1; // how many tokens to scan for repetitions (0 = disable penalty, -1 = context size) + int32_t dry_penalty_last_n = -1; // how many tokens to scan for repetitions (0 = disable penalty, -1 = 1024) float adaptive_target = -1.0f; // select tokens near this probability (valid range 0.0 to 1.0; negative = disabled) float adaptive_decay = 0.90f; // EMA decay for adaptation; history ≈ 1/(1-decay) tokens (0.0 - 0.99) int32_t mirostat = 0; // 0 = disabled, 1 = mirostat, 2 = mirostat 2.0 diff --git a/common/sampling.cpp b/common/sampling.cpp index ba5504ed01..eda3e00d76 100644 --- a/common/sampling.cpp +++ b/common/sampling.cpp @@ -186,8 +186,7 @@ std::string common_params_sampling::print() const { struct common_sampler * common_sampler_init( const struct llama_model * model, - struct common_params_sampling & params, - int32_t n_ctx) { + struct common_params_sampling & params) { if (!std::isfinite(params.penalty_repeat) || params.penalty_repeat <= 0.0f || !std::isfinite(1.0f/params.penalty_repeat)) { @@ -199,10 +198,6 @@ struct common_sampler * common_sampler_init( if (!std::isfinite(params.penalty_present)) { throw std::invalid_argument("penalty_present must be finite"); } - if (params.penalty_last_n == -1) { - params.penalty_last_n = n_ctx > 0 ? n_ctx : llama_model_n_ctx_train(model); - } - const llama_vocab * vocab = llama_model_get_vocab(model); llama_sampler_chain_params lparams = llama_sampler_chain_default_params(); diff --git a/common/sampling.h b/common/sampling.h index 91e2cea787..cb90d4ae7a 100644 --- a/common/sampling.h +++ b/common/sampling.h @@ -39,8 +39,7 @@ struct common_sampler; // note: can mutate params in some cases struct common_sampler * common_sampler_init( const struct llama_model * model, - struct common_params_sampling & params, - int32_t n_ctx = 0); + struct common_params_sampling & params); void common_sampler_free(struct common_sampler * gsmpl); diff --git a/include/llama.h b/include/llama.h index fb2ca38cee..d44601d1d0 100644 --- a/include/llama.h +++ b/include/llama.h @@ -1425,7 +1425,7 @@ extern "C" { /// NOTE: Avoid using on the full vocabulary as searching for repeated tokens can become slow. For example, apply top-k or top-p sampling first. LLAMA_API struct llama_sampler * llama_sampler_init_penalties( int32_t n_vocab, - int32_t penalty_last_n, // last n tokens to penalize (0 = disable penalty, -1 = context size) + int32_t penalty_last_n, // last n tokens to penalize (0 = disable penalty, -1 = 1024) float penalty_repeat, // must be > 0.0, 1.0 = disabled float penalty_freq, // must be finite, 0.0 = disabled float penalty_present); // must be finite, 0.0 = disabled @@ -1437,7 +1437,7 @@ extern "C" { float dry_multiplier, float dry_base, int32_t dry_allowed_length, - int32_t dry_penalty_last_n, + int32_t dry_penalty_last_n, // last n tokens to penalize (0 = disable penalty, -1 = 1024) const char ** seq_breakers, size_t num_breakers); diff --git a/src/llama-sampler.cpp b/src/llama-sampler.cpp index 6cf2d27cf9..5c4126a88c 100644 --- a/src/llama-sampler.cpp +++ b/src/llama-sampler.cpp @@ -2971,7 +2971,7 @@ struct llama_sampler * llama_sampler_init_penalties( float penalty_repeat, float penalty_freq, float penalty_present) { - penalty_last_n = std::max(penalty_last_n, 0); + penalty_last_n = penalty_last_n == -1 ? 1024 : std::max(penalty_last_n, 0); if (llama_sampler_penalties::is_disabled( penalty_last_n, penalty_repeat, penalty_freq, penalty_present)) { @@ -3155,8 +3155,7 @@ static void llama_sampler_dry_apply(struct llama_sampler * smpl, llama_token_dat return; } - int32_t effective_dry_penalty_last_n = (ctx->dry_penalty_last_n == -1) ? ctx->total_context_size : std::max(ctx->dry_penalty_last_n, 0); - int last_n_repeat = std::min(std::min((int)ctx->last_tokens.size(), effective_dry_penalty_last_n), ctx->total_context_size); + int last_n_repeat = std::min(std::min((int)ctx->last_tokens.size(), ctx->dry_penalty_last_n), ctx->total_context_size); if (last_n_repeat <= ctx->dry_allowed_length) { return; @@ -3401,7 +3400,7 @@ static struct llama_sampler_i llama_sampler_dry_i = { }; struct llama_sampler * llama_sampler_init_dry(const struct llama_vocab * vocab, int32_t n_ctx_train, float dry_multiplier, float dry_base, int32_t dry_allowed_length, int32_t dry_penalty_last_n, const char** seq_breakers, size_t num_breakers) { - int32_t effective_dry_penalty_last_n = (dry_penalty_last_n == -1) ? n_ctx_train : std::max(dry_penalty_last_n, 0); + dry_penalty_last_n = dry_penalty_last_n == -1 ? 1024 : std::max(dry_penalty_last_n, 0); std::unordered_multimap> processed_breakers; const int MAX_CHAR_LEN = 40; const int MAX_SEQ_LEN = 20; @@ -3444,9 +3443,9 @@ struct llama_sampler * llama_sampler_init_dry(const struct llama_vocab * vocab, /* .dry_allowed_length = */ dry_allowed_length, /* .dry_penalty_last_n = */ dry_penalty_last_n, /* .dry_processed_breakers = */ std::move(processed_breakers), - /* .dry_repeat_count = */ dry_enabled ? std::vector(effective_dry_penalty_last_n, 0) : std::vector{}, + /* .dry_repeat_count = */ dry_enabled ? std::vector(dry_penalty_last_n, 0) : std::vector{}, /* .dry_max_token_repeat = */ {}, - /* .last_tokens = */ dry_enabled ? ring_buffer(effective_dry_penalty_last_n) : ring_buffer(0), + /* .last_tokens = */ dry_enabled ? ring_buffer(dry_penalty_last_n) : ring_buffer(0), } ); } diff --git a/tests/test-backend-sampler.cpp b/tests/test-backend-sampler.cpp index 1165f46f0c..11b3c12328 100644 --- a/tests/test-backend-sampler.cpp +++ b/tests/test-backend-sampler.cpp @@ -1242,6 +1242,9 @@ static void test_backend_penalties_sampling(const test_params & params) { printf("Testing backend penalties (repeat + freq + presence)\n"); compare_penalties_logits(params, 64, 1.1f, 0.5f, 0.25f, "Hello Hello world"); + printf("Testing backend penalties with penalty_last_n = -1\n"); + compare_penalties_logits(params, -1, 1.1f, 0.5f, 0.25f, "Hello Hello world"); + printf("Testing backend penalties with penalty_last_n > 64\n"); const auto * vocab = llama_model_get_vocab(params.model.get()); std::vector tokens(8); diff --git a/tests/test-sampling.cpp b/tests/test-sampling.cpp index 297f760157..cbbb74512f 100644 --- a/tests/test-sampling.cpp +++ b/tests/test-sampling.cpp @@ -158,6 +158,26 @@ static void test_penalties( tester.check(); } +static void test_penalties_last_n() { + const int32_t n_vocab = 1025; + std::vector data; + data.reserve(n_vocab); + + auto * sampler = llama_sampler_init_penalties(n_vocab, -1, 2.0f, 0.0f, 0.0f); + for (llama_token token = 0; token < n_vocab; ++token) { + data.push_back({ token, 1.0f, 0.0f }); + llama_sampler_accept(sampler, token); + } + + llama_token_data_array cur_p = { data.data(), data.size(), -1, false }; + llama_sampler_apply(sampler, &cur_p); + llama_sampler_free(sampler); + + GGML_ASSERT(data[0].logit == 1.0f); + GGML_ASSERT(data[1].logit == 0.5f); + GGML_ASSERT(data[1024].logit == 0.5f); +} + static void test_dry( const std::vector & probs, const std::vector & last_tokens, const std::vector & expected_probs, float dry_multiplier, float dry_base, @@ -181,6 +201,37 @@ static void test_dry( tester.check(); } +static void test_dry_last_n() { + const int32_t n_vocab = 1027; + std::vector last_tokens = { 0, 1, 2 }; + for (llama_token token = 3; token < n_vocab; ++token) { + last_tokens.push_back(token); + } + last_tokens.push_back(0); + last_tokens.push_back(1); + + const auto apply_dry = [&](int32_t dry_penalty_last_n) { + std::vector data; + data.reserve(n_vocab); + for (llama_token token = 0; token < n_vocab; ++token) { + data.push_back({ token, 0.0f, 0.0f }); + } + + auto * sampler = llama_sampler_init_dry_testing(2048, 1.0f, 2.0f, 1, dry_penalty_last_n, {}); + for (llama_token token : last_tokens) { + llama_sampler_accept(sampler, token); + } + + llama_token_data_array cur_p = { data.data(), data.size(), -1, false }; + llama_sampler_apply(sampler, &cur_p); + llama_sampler_free(sampler); + return data[2].logit; + }; + + GGML_ASSERT(apply_dry(-1) == 0.0f); + GGML_ASSERT(apply_dry(2048) < 0.0f); +} + static void test_top_n_sigma(const std::vector & probs, const std::vector & probs_expected, int n) { sampler_tester tester(probs, probs_expected); @@ -352,13 +403,14 @@ int main(void) { test_penalties({0.2f, 0.2f, 0.2f, 0.2f, 0.2f}, {0}, {0.000011f, 0.249997f, 0.249997f, 0.249997f, 0.249997f}, 1.0f, 5.0f, 5.0f); test_penalties({0.2f, 0.2f, 0.2f, 0.2f, 0.2f}, {0, 1, 2}, {0.000023f, 0.000023f, 0.000023f, 0.499966f, 0.499966f}, 1.0f, 5.0f, 5.0f); test_penalties({0.2f, 0.2f, 0.2f, 0.2f, 0.2f}, {0, 1, 2, 0, 0}, {0.000000f, 0.000023f, 0.000023f, 0.499977f, 0.499977f}, 1.0f, 5.0f, 5.0f); - + test_penalties_last_n(); test_dry({0.25f, 0.25f, 0.25f, 0.25f}, {0, 1}, {0.25f, 0.25f, 0.25f, 0.25f}, 1.0f, 1.1f, 2, 4, {}); test_dry({0.25f, 0.25f, 0.25f, 0.25f}, {0, 1, 2, 0, 1}, {0.296923f, 0.296923f, 0.109232f, 0.296923f}, 1.0f, 1.1f, 2, 5, {}); test_dry({0.2f, 0.2f, 0.2f, 0.2f, 0.2f}, {0, 1, 3, 4, 0, 1}, {0.2f, 0.2f, 0.2f, 0.2f, 0.2f}, 1.0f, 1.1f, 2, 6, {{3}}); test_dry({0.2f, 0.2f, 0.2f, 0.2f, 0.2f}, {0, 1, 2, 0, 1}, {0.241818f, 0.241818f, 0.032727f, 0.241818f, 0.241818f}, 2.0f, 1.1f, 2, 5, {}); test_dry({0.2f, 0.2f, 0.2f, 0.2f, 0.2f}, {0, 1, 2, 3, 4, 0, 1}, {0.2f, 0.2f, 0.2f, 0.2f, 0.2f}, 1.0f, 1.1f, 4, 7, {}); + test_dry_last_n(); test_top_n_sigma({0.1f, 0.2f, 0.3f, 0.4f}, {0.0f, 0.0f, 0.428571f, 0.571429f}, 1.00f); test_top_n_sigma({0.1f, 0.2f, 0.3f, 0.4f}, {0.1f, 0.2f, 0.3f, 0.4f}, 0.00f); // top_n_sigma == 0 now represents a no-op rather than greedy decoding as of PR#13345 diff --git a/tools/cli/README.md b/tools/cli/README.md index bcddd05702..035a659d77 100644 --- a/tools/cli/README.md +++ b/tools/cli/README.md @@ -116,14 +116,14 @@ | `--xtc-probability N` | xtc probability (default: 0.00, 0.0 = disabled) | | `--xtc-threshold N` | xtc threshold (default: 0.10, 1.0 = disabled) | | `--typical, --typical-p N` | locally typical sampling, parameter p (default: 1.00, 1.0 = disabled) | -| `--repeat-last-n N` | last n tokens to consider for penalize (default: 64, 0 = disabled, -1 = ctx_size) | +| `--repeat-last-n N` | last n tokens to consider for penalize (default: 64, 0 = disabled, -1 = 1024) | | `--repeat-penalty N` | penalize repeat sequence of tokens (default: 1.00, 1.0 = disabled) | | `--presence-penalty N` | repeat alpha presence penalty (default: 0.00, 0.0 = disabled) | | `--frequency-penalty N` | repeat alpha frequency penalty (default: 0.00, 0.0 = disabled) | | `--dry-multiplier N` | set DRY sampling multiplier (default: 0.00, 0.0 = disabled) | | `--dry-base N` | set DRY sampling base value (default: 1.75) | | `--dry-allowed-length N` | set allowed length for DRY sampling (default: 2) | -| `--dry-penalty-last-n N` | set DRY penalty for the last n tokens (default: -1, 0 = disable, -1 = context size) | +| `--dry-penalty-last-n N` | set DRY penalty for the last n tokens (default: -1, 0 = disable, -1 = 1024) | | `--dry-sequence-breaker STRING` | add sequence breaker for DRY sampling, clearing out default breakers ('\n', ':', '"', '*') in the process; use "none" to not use any sequence breakers | | `--adaptive-target N` | adaptive-p: select tokens near this probability (valid range 0.0 to 1.0; negative = disabled) (default: -1.00)
[(more info)](https://github.com/ggml-org/llama.cpp/pull/17927) | | `--adaptive-decay N` | adaptive-p: decay rate for target adaptation over time. lower values are more reactive, higher values are more stable.
(valid range 0.0 to 0.99) (default: 0.90) | diff --git a/tools/completion/README.md b/tools/completion/README.md index bce71d68d9..015f66c392 100644 --- a/tools/completion/README.md +++ b/tools/completion/README.md @@ -199,14 +199,14 @@ llama-completion.exe -m models\gemma-1.1-7b-it.Q4_K_M.gguf --ignore-eos -n -1 | `--xtc-probability N` | xtc probability (default: 0.00, 0.0 = disabled) | | `--xtc-threshold N` | xtc threshold (default: 0.10, 1.0 = disabled) | | `--typical, --typical-p N` | locally typical sampling, parameter p (default: 1.00, 1.0 = disabled) | -| `--repeat-last-n N` | last n tokens to consider for penalize (default: 64, 0 = disabled, -1 = ctx_size) | +| `--repeat-last-n N` | last n tokens to consider for penalize (default: 64, 0 = disabled, -1 = 1024) | | `--repeat-penalty N` | penalize repeat sequence of tokens (default: 1.00, 1.0 = disabled) | | `--presence-penalty N` | repeat alpha presence penalty (default: 0.00, 0.0 = disabled) | | `--frequency-penalty N` | repeat alpha frequency penalty (default: 0.00, 0.0 = disabled) | | `--dry-multiplier N` | set DRY sampling multiplier (default: 0.00, 0.0 = disabled) | | `--dry-base N` | set DRY sampling base value (default: 1.75) | | `--dry-allowed-length N` | set allowed length for DRY sampling (default: 2) | -| `--dry-penalty-last-n N` | set DRY penalty for the last n tokens (default: -1, 0 = disable, -1 = context size) | +| `--dry-penalty-last-n N` | set DRY penalty for the last n tokens (default: -1, 0 = disable, -1 = 1024) | | `--dry-sequence-breaker STRING` | add sequence breaker for DRY sampling, clearing out default breakers ('\n', ':', '"', '*') in the process; use "none" to not use any sequence breakers | | `--adaptive-target N` | adaptive-p: select tokens near this probability (valid range 0.0 to 1.0; negative = disabled) (default: -1.00)
[(more info)](https://github.com/ggml-org/llama.cpp/pull/17927) | | `--adaptive-decay N` | adaptive-p: decay rate for target adaptation over time. lower values are more reactive, higher values are more stable.
(valid range 0.0 to 0.99) (default: 0.90) | diff --git a/tools/server/README.md b/tools/server/README.md index a0956f9e65..65d989cbef 100644 --- a/tools/server/README.md +++ b/tools/server/README.md @@ -133,14 +133,14 @@ For the full list of features, please refer to [server's changelog](https://gith | `--xtc-probability N` | xtc probability (default: 0.00, 0.0 = disabled) | | `--xtc-threshold N` | xtc threshold (default: 0.10, 1.0 = disabled) | | `--typical, --typical-p N` | locally typical sampling, parameter p (default: 1.00, 1.0 = disabled) | -| `--repeat-last-n N` | last n tokens to consider for penalize (default: 64, 0 = disabled, -1 = ctx_size) | +| `--repeat-last-n N` | last n tokens to consider for penalize (default: 64, 0 = disabled, -1 = 1024) | | `--repeat-penalty N` | penalize repeat sequence of tokens (default: 1.00, 1.0 = disabled) | | `--presence-penalty N` | repeat alpha presence penalty (default: 0.00, 0.0 = disabled) | | `--frequency-penalty N` | repeat alpha frequency penalty (default: 0.00, 0.0 = disabled) | | `--dry-multiplier N` | set DRY sampling multiplier (default: 0.00, 0.0 = disabled) | | `--dry-base N` | set DRY sampling base value (default: 1.75) | | `--dry-allowed-length N` | set allowed length for DRY sampling (default: 2) | -| `--dry-penalty-last-n N` | set DRY penalty for the last n tokens (default: -1, 0 = disable, -1 = context size) | +| `--dry-penalty-last-n N` | set DRY penalty for the last n tokens (default: -1, 0 = disable, -1 = 1024) | | `--dry-sequence-breaker STRING` | add sequence breaker for DRY sampling, clearing out default breakers ('\n', ':', '"', '*') in the process; use "none" to not use any sequence breakers | | `--adaptive-target N` | adaptive-p: select tokens near this probability (valid range 0.0 to 1.0; negative = disabled) (default: -1.00)
[(more info)](https://github.com/ggml-org/llama.cpp/pull/17927) | | `--adaptive-decay N` | adaptive-p: decay rate for target adaptation over time. lower values are more reactive, higher values are more stable.
(valid range 0.0 to 0.99) (default: 0.90) | diff --git a/tools/server/server-context.cpp b/tools/server/server-context.cpp index 5d2798cc14..380e62af67 100644 --- a/tools/server/server-context.cpp +++ b/tools/server/server-context.cpp @@ -1807,8 +1807,7 @@ private: // initialize samplers if (task.need_sampling()) { try { - slot.smpl.reset(common_sampler_init( - model_tgt, task.params.sampling, (int32_t) llama_n_ctx(ctx_tgt))); + slot.smpl.reset(common_sampler_init(model_tgt, task.params.sampling)); } catch (std::exception & e) { std::string err_msg = std::string("Failed to initialize samplers: ") + e.what(); send_error(task, err_msg, ERROR_TYPE_INVALID_REQUEST); @@ -4148,7 +4147,6 @@ std::unique_ptr server_routes::handle_completions_impl( task.params = server_schema::eval_llama_cmpl_schema( ctx_server.vocab, params, - meta->slot_n_ctx, meta->logit_bias_eog, data); diff --git a/tools/server/server-schema.cpp b/tools/server/server-schema.cpp index 674d3ba337..589cb188be 100644 --- a/tools/server/server-schema.cpp +++ b/tools/server/server-schema.cpp @@ -152,7 +152,7 @@ std::vector> make_llama_cmpl_schema(const common_params & add((new field_num("dry_penalty_last_n", params.sampling.dry_penalty_last_n)) ->set_hard_limits(-1, INT32_MAX) - ->set_desc("How many tokens to scan for repetitions (0 = disabled, -1 = context size)")); + ->set_desc("How many tokens to scan for repetitions (0 = disabled, -1 = 1024)")); add((new field_num("mirostat", params.sampling.mirostat)) ->set_limits(0, 2) @@ -515,7 +515,6 @@ std::vector> make_llama_cmpl_schema(const common_params & task_params eval_llama_cmpl_schema( const llama_vocab * vocab, const common_params & params_base, - const int n_ctx_slot, const std::vector & logit_bias_eog, const json & data) { task_params params; @@ -549,15 +548,6 @@ task_params eval_llama_cmpl_schema( // post-processing { - if (params.sampling.penalty_last_n == -1) { - // note: should be the slot's context and not the full context, but it's ok - params.sampling.penalty_last_n = n_ctx_slot; - } - - if (params.sampling.dry_penalty_last_n == -1) { - params.sampling.dry_penalty_last_n = n_ctx_slot; - } - // if "reasoning_format" is not provided, its handler will not be called, we will need to handle it here auto reasoning_format = params.chat_parser_params.reasoning_format; params.chat_parser_params.reasoning_in_content = params.stream && (reasoning_format == COMMON_REASONING_FORMAT_DEEPSEEK_LEGACY); diff --git a/tools/server/server-schema.h b/tools/server/server-schema.h index 08cf427dc9..d0a81431bc 100644 --- a/tools/server/server-schema.h +++ b/tools/server/server-schema.h @@ -98,7 +98,6 @@ std::vector> make_llama_cmpl_schema( task_params eval_llama_cmpl_schema( const llama_vocab * vocab, const common_params & params_base, - const int n_ctx_slot, const std::vector & logit_bias_eog, const json & data);