diff --git a/common/common.h b/common/common.h index 38cc595697..dbd4a08062 100644 --- a/common/common.h +++ b/common/common.h @@ -175,7 +175,6 @@ enum common_speculative_type { COMMON_SPECULATIVE_TYPE_COUNT // number of types, unknown type }; - // sampling parameters struct common_params_sampling { uint32_t seed = LLAMA_DEFAULT_SEED; // the seed used to initialize llama_sampler @@ -276,6 +275,7 @@ struct common_params_speculative { // draftless: common_speculative_type draftless_type = COMMON_SPECULATIVE_TYPE_NONE; // type of speculative decoding without a draft model + uint16_t spec_ngram_size_n = 12; // ngram size for lookup uint16_t spec_ngram_size_m = 48; // mgram size for speculative tokens uint16_t spec_ngram_check_rate = 1; // check rate for ngram lookup diff --git a/common/ngram-map.h b/common/ngram-map.h index 0738d69052..fd4564bf00 100644 --- a/common/ngram-map.h +++ b/common/ngram-map.h @@ -12,7 +12,6 @@ #include "llama.h" -#include #include // n-gram simple @@ -29,7 +28,7 @@ struct common_ngram_simple_config { struct common_ngram_simple_state { common_ngram_simple_config config; - size_t idx_last_check = 0; // index of last check in context history (mutable) + size_t idx_last_check = 0; // index of last check in context history (mutable) common_ngram_simple_state(const common_ngram_simple_config & config) : config(config) {} diff --git a/common/speculative.cpp b/common/speculative.cpp index 06fd9fbf8e..7092ea27d9 100644 --- a/common/speculative.cpp +++ b/common/speculative.cpp @@ -1,10 +1,5 @@ #include "speculative.h" -#include -#include -#include -#include - #include "ggml.h" #include "llama.h" #include "log.h" @@ -13,6 +8,11 @@ #include "ngram-map.h" #include "sampling.h" +#include +#include +#include +#include + #define SPEC_VOCAB_MAX_SIZE_DIFFERENCE 128 #define SPEC_VOCAB_CHECK_START_TOKEN_ID 5 @@ -44,13 +44,71 @@ struct common_speculative_config { const common_params_speculative & p = common_params_speculative{}) : type(t), params(p) {} }; +static bool common_speculative_are_compatible( + const struct llama_model * model_tgt, + const struct llama_model * model_dft) { + const struct llama_vocab * vocab_tgt = llama_model_get_vocab(model_tgt); + const struct llama_vocab * vocab_dft = llama_model_get_vocab(model_dft); + + const bool vocab_type_tgt = llama_vocab_type(vocab_tgt); + LOG_DBG("%s: vocab_type tgt: %d\n", __func__, vocab_type_tgt); + + const bool vocab_type_dft = llama_vocab_type(vocab_dft); + LOG_DBG("%s: vocab_type dft: %d\n", __func__, vocab_type_dft); + + if (vocab_type_tgt != vocab_type_dft) { + LOG_DBG("%s: draft model vocab type must match target model to use speculation but ", __func__); + LOG_DBG("vocab_type_dft = %d while vocab_type_tgt = %d\n", vocab_type_dft, vocab_type_tgt); + return false; + } + + if ( + llama_vocab_get_add_bos(vocab_tgt) != llama_vocab_get_add_bos(vocab_dft) || + llama_vocab_get_add_eos(vocab_tgt) != llama_vocab_get_add_eos(vocab_dft) || + llama_vocab_bos(vocab_tgt) != llama_vocab_bos(vocab_dft) || + llama_vocab_eos(vocab_tgt) != llama_vocab_eos(vocab_dft) + ) { + LOG_DBG("%s: draft model special tokens must match target model to use speculation\n", __func__); + return false; + } + + { + const int n_vocab_tgt = llama_vocab_n_tokens(vocab_tgt); + const int n_vocab_dft = llama_vocab_n_tokens(vocab_dft); + const int vocab_diff = n_vocab_tgt > n_vocab_dft + ? n_vocab_tgt - n_vocab_dft + : n_vocab_dft - n_vocab_tgt; + + if (vocab_diff > SPEC_VOCAB_MAX_SIZE_DIFFERENCE) { + LOG_DBG("%s: draft model vocab must closely match target model to use speculation but ", __func__); + LOG_DBG("target vocab size %d does not match draft vocab size %d - difference %d, max allowed %d\n", + n_vocab_tgt, llama_vocab_n_tokens(vocab_dft), vocab_diff, SPEC_VOCAB_MAX_SIZE_DIFFERENCE); + return false; + } + + for (int i = SPEC_VOCAB_CHECK_START_TOKEN_ID; i < std::min(n_vocab_tgt, n_vocab_dft); ++i) { + const char * token_text_tgt = llama_vocab_get_text(vocab_tgt, i); + const char * token_text_dft = llama_vocab_get_text(vocab_dft, i); + + if (std::strcmp(token_text_tgt, token_text_dft) != 0) { + LOG_DBG("%s: draft model vocab must match target model to use speculation but ", __func__); + LOG_DBG("token %d content differs - target '%s', draft '%s'\n", i, + common_token_to_piece(vocab_tgt, i).c_str(), + common_token_to_piece(vocab_dft, i).c_str()); + return false; + } + } + } + + return true; +} + // state of an implementation of speculative decoding // // each implementation has a unique type and a state that is implementation-specific // in a subclass of common_speculative_state struct common_speculative_state { const enum common_speculative_type type; - const bool gen_perf = true; // whether to generate performance stats. size_t drafts_call_count = 0; // number of times this implementation was called. size_t drafts_generated_count = 0; // number of times a draft or part was generated by this implementation. @@ -58,7 +116,10 @@ struct common_speculative_state { size_t drafts_generated_tokens = 0; // number of tokens generated by this implementation. size_t drafts_accepted_tokens = 0; // number of tokens accepted by the target model. - int64_t gen_duration_ms = 0; // total time spent in this implementation in milliseconds. + // TODO: track performance of most recent calls + const bool gen_perf = true; // whether to generate performance stats. + + int64_t gen_duration_us = 0; // total time spent in this implementation in milliseconds. virtual ~common_speculative_state() = default; @@ -408,65 +469,6 @@ void common_speculative_free(struct common_speculative * spec) { delete spec; } -bool common_speculative_are_compatible( - const struct llama_model * model_tgt, - const struct llama_model * model_dft) { - const struct llama_vocab * vocab_tgt = llama_model_get_vocab(model_tgt); - const struct llama_vocab * vocab_dft = llama_model_get_vocab(model_dft); - - const bool vocab_type_tgt = llama_vocab_type(vocab_tgt); - LOG_DBG("%s: vocab_type tgt: %d\n", __func__, vocab_type_tgt); - - const bool vocab_type_dft = llama_vocab_type(vocab_dft); - LOG_DBG("%s: vocab_type dft: %d\n", __func__, vocab_type_dft); - - if (vocab_type_tgt != vocab_type_dft) { - LOG_DBG("%s: draft model vocab type must match target model to use speculation but ", __func__); - LOG_DBG("vocab_type_dft = %d while vocab_type_tgt = %d\n", vocab_type_dft, vocab_type_tgt); - return false; - } - - if ( - llama_vocab_get_add_bos(vocab_tgt) != llama_vocab_get_add_bos(vocab_dft) || - llama_vocab_get_add_eos(vocab_tgt) != llama_vocab_get_add_eos(vocab_dft) || - llama_vocab_bos(vocab_tgt) != llama_vocab_bos(vocab_dft) || - llama_vocab_eos(vocab_tgt) != llama_vocab_eos(vocab_dft) - ) { - LOG_DBG("%s: draft model special tokens must match target model to use speculation\n", __func__); - return false; - } - - { - const int n_vocab_tgt = llama_vocab_n_tokens(vocab_tgt); - const int n_vocab_dft = llama_vocab_n_tokens(vocab_dft); - const int vocab_diff = n_vocab_tgt > n_vocab_dft - ? n_vocab_tgt - n_vocab_dft - : n_vocab_dft - n_vocab_tgt; - - if (vocab_diff > SPEC_VOCAB_MAX_SIZE_DIFFERENCE) { - LOG_DBG("%s: draft model vocab must closely match target model to use speculation but ", __func__); - LOG_DBG("target vocab size %d does not match draft vocab size %d - difference %d, max allowed %d\n", - n_vocab_tgt, llama_vocab_n_tokens(vocab_dft), vocab_diff, SPEC_VOCAB_MAX_SIZE_DIFFERENCE); - return false; - } - - for (int i = SPEC_VOCAB_CHECK_START_TOKEN_ID; i < std::min(n_vocab_tgt, n_vocab_dft); ++i) { - const char * token_text_tgt = llama_vocab_get_text(vocab_tgt, i); - const char * token_text_dft = llama_vocab_get_text(vocab_dft, i); - - if (std::strcmp(token_text_tgt, token_text_dft) != 0) { - LOG_DBG("%s: draft model vocab must match target model to use speculation but ", __func__); - LOG_DBG("token %d content differs - target '%s', draft '%s'\n", i, - common_token_to_piece(vocab_tgt, i).c_str(), - common_token_to_piece(vocab_dft, i).c_str()); - return false; - } - } - } - - return true; -} - static std::string replace_to_dft( struct common_speculative_state_draft * spec, const std::string & input) { @@ -740,7 +742,7 @@ llama_tokens common_speculative_gen_draft( // TODO: avoid dynamic casts for (auto & impl : spec->impls) { impl->drafts_call_count++; - const int64_t t_start_ms = impl->gen_perf ? ggml_time_ms() : 0; + const int64_t t_start_us = impl->gen_perf ? ggml_time_us() : 0; switch (impl->type) { case COMMON_SPECULATIVE_TYPE_NONE: @@ -805,8 +807,8 @@ llama_tokens common_speculative_gen_draft( } } - const int64_t t_now_ms = impl->gen_perf ? ggml_time_ms() : 0; - impl->gen_duration_ms += t_now_ms - t_start_ms; // accumulate duration for this implementation + const int64_t t_now_us = impl->gen_perf ? ggml_time_us() : 0; + impl->gen_duration_us += t_now_us - t_start_us; // accumulate duration for this implementation if (!result.empty()) { LOG_DBG("%s: called impl %s, hist size = %zu, call_count = %zu, gen = %zu\n", __func__, @@ -855,10 +857,15 @@ void common_speculative_print_stats(const struct common_speculative * spec) { } for (const auto & impl : spec->impls) { - // std::string via impl->gen_duration_ms (if >0, else "") - std::string performance = impl->gen_perf - ? ", dur = " + std::to_string(impl->gen_duration_ms) + " ms" - : ""; + std::string str_perf; + if (impl->gen_perf) { + std::ostringstream oss; + oss << std::fixed << std::setprecision(3) << impl->gen_duration_us / 1000.0; + str_perf = ", dur = " + oss.str() + " ms"; + } else { + str_perf = ""; + } + LOG_INF("statistics %s: #calls = %zu, #gen drafts = %zu, #acc drafts = %zu, #gen tokens = %zu, #acc tokens = %zu%s\n", common_speculative_type_to_str(impl->type).c_str(), impl->drafts_call_count, @@ -866,6 +873,6 @@ void common_speculative_print_stats(const struct common_speculative * spec) { impl->drafts_accepted_count, impl->drafts_generated_tokens, impl->drafts_accepted_tokens, - performance.c_str()); + str_perf.c_str()); } } diff --git a/common/speculative.h b/common/speculative.h index 870c9e0f99..423c62883f 100644 --- a/common/speculative.h +++ b/common/speculative.h @@ -28,10 +28,6 @@ struct common_speculative * common_speculative_init( void common_speculative_free(struct common_speculative * spec); -bool common_speculative_are_compatible( - const struct llama_model * model_tgt, - const struct llama_model * model_dft); - // sample up to n_draft tokens and add them to the batch using the draft model llama_tokens common_speculative_gen_draft( struct common_speculative * spec,