mirror of
https://github.com/LostRuins/koboldcpp.git
synced 2026-09-19 01:05:09 +02:00
Merge commit '1c3c9674de4d455f1e571bed808252af54932767' into concedo_experimental
# Conflicts: # .github/workflows/build-apple.yml # .github/workflows/build-vulkan.yml # .github/workflows/release.yml # docs/ops.md # docs/ops/Vulkan.csv # examples/gen-docs/gen-docs.cpp # ggml/CMakeLists.txt # ggml/src/ggml-opencl/ggml-opencl.cpp # ggml/src/ggml-sycl/concat.cpp # ggml/src/ggml-sycl/fattn-onednn.cpp # ggml/src/ggml-sycl/fattn.cpp # models/templates/deepseek-ai-DeepSeek-V4.jinja # scripts/sync-ggml.last # scripts/sync_vendor.py # src/CMakeLists.txt # src/llama-model-loader.cpp # tests/CMakeLists.txt # tests/test-arg-parser.cpp # tests/test-backend-sampler.cpp # tests/test-chat.cpp # tests/test-sampling.cpp # tools/server/README.md
This commit is contained in:
+4
-3
@@ -1427,10 +1427,11 @@ extern "C" {
|
||||
|
||||
/// NOTE: Avoid using on the full vocabulary as searching for repeated tokens can become slow. For example, apply top-k or top-p sampling first.
|
||||
LLAMA_API struct llama_sampler * llama_sampler_init_penalties(
|
||||
int32_t n_vocab,
|
||||
int32_t penalty_last_n, // last n tokens to penalize (0 = disable penalty, -1 = context size)
|
||||
float penalty_repeat, // 1.0 = disabled
|
||||
float penalty_freq, // 0.0 = disabled
|
||||
float penalty_present); // 0.0 = disabled
|
||||
float penalty_repeat, // must be > 0.0, 1.0 = disabled
|
||||
float penalty_freq, // must be finite, 0.0 = disabled
|
||||
float penalty_present); // must be finite, 0.0 = disabled
|
||||
|
||||
/// @details DRY sampler, designed by p-e-w, as described in: https://github.com/oobabooga/text-generation-webui/pull/5677, porting Koboldcpp implementation authored by pi6am: https://github.com/LostRuins/koboldcpp/pull/982
|
||||
LLAMA_API struct llama_sampler * llama_sampler_init_dry(
|
||||
|
||||
Reference in New Issue
Block a user