mirror of
https://github.com/LostRuins/koboldcpp.git
synced 2026-09-19 09:15:18 +02:00
Merge commit '533b18257b7da879a5de39f5d6437041e0e42c39' into concedo_experimental
# Conflicts: # CMakeLists.txt # examples/gguf-hash/CMakeLists.txt # examples/gguf-hash/gguf-hash.cpp # scripts/sync_vendor.py # tools/mtmd/CMakeLists.txt # vendor/hash/rotate-bits/rotate-bits.h # vendor/hash/sha1/sha1.c # vendor/hash/sha1/sha1.h # vendor/hash/sha256/sha256.c # vendor/hash/sha256/sha256.h # vendor/hash/xxhash/xxhash.c # vendor/hash/xxhash/xxhash.h
This commit is contained in:
+17
-4
@@ -2664,17 +2664,24 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
|
||||
const uint32_t n_tokens = gguf_get_arr_n(ctx, token_idx);
|
||||
|
||||
const float * scores = nullptr;
|
||||
const int * iscores = nullptr;
|
||||
const int score_idx = gguf_find_key(ctx, kv(LLM_KV_TOKENIZER_SCORES).c_str());
|
||||
if (score_idx != -1) {
|
||||
if (gguf_get_kv_type(ctx, score_idx) != GGUF_TYPE_ARRAY ||
|
||||
gguf_get_arr_type(ctx, score_idx) != GGUF_TYPE_FLOAT32) {
|
||||
const gguf_type kv_type = gguf_get_kv_type(ctx, score_idx);
|
||||
const gguf_type arr_type = kv_type == GGUF_TYPE_ARRAY ? gguf_get_arr_type(ctx, score_idx) : GGUF_TYPE_COUNT;
|
||||
if (arr_type != GGUF_TYPE_INT32 &&
|
||||
arr_type != GGUF_TYPE_FLOAT32) {
|
||||
throw std::runtime_error(format("invalid gguf type for %s", kv(LLM_KV_TOKENIZER_SCORES).c_str()));
|
||||
}
|
||||
const uint32_t n_scores = gguf_get_arr_n(ctx, score_idx);
|
||||
if (n_scores < n_tokens) {
|
||||
throw std::runtime_error("Index out of array bounds for scores (" + std::to_string(n_scores) + " < " + std::to_string(n_tokens) + ")\n");
|
||||
}
|
||||
scores = (const float * ) gguf_get_arr_data(ctx, score_idx);
|
||||
if (arr_type == GGUF_TYPE_INT32) {
|
||||
iscores = (const int *) gguf_get_arr_data(ctx, score_idx);
|
||||
} else {
|
||||
scores = (const float * ) gguf_get_arr_data(ctx, score_idx);
|
||||
}
|
||||
}
|
||||
|
||||
const int * toktypes = nullptr;
|
||||
@@ -2708,7 +2715,13 @@ void llama_vocab::impl::load(llama_model_loader & ml, const LLM_KV & kv) {
|
||||
|
||||
auto & token_data = id_to_token[i];
|
||||
token_data.text = std::move(word);
|
||||
token_data.score = scores ? scores[i] : 0.0f;
|
||||
if (scores) {
|
||||
token_data.score = scores[i];
|
||||
} else if (iscores) {
|
||||
token_data.score = static_cast<float>(iscores[i]);
|
||||
} else {
|
||||
token_data.score = 0.0f;
|
||||
}
|
||||
token_data.attr = LLAMA_TOKEN_ATTR_NORMAL;
|
||||
|
||||
if (toktypes) { //TODO: remove, required until per token attributes are available from GGUF file
|
||||
|
||||
Reference in New Issue
Block a user