mirror of
https://github.com/ggml-org/llama.cpp.git
synced 2026-08-27 07:31:24 +02:00
7221e24f57
* feat(convert): Add conversion for GraniteSWAForCausalLM Branch: GraniteSWAForCausalLM AI-usage: full (Bob, OpenCode + Qwen3.6-35b) Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * feat(llama): Add granite_swa support Branch: GraniteSWAForCausalLM AI-usage: full (Bob, OpenCode + Qwen3.6-35b) Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * feat(conversion): Add conversion infra for rope_pattern array NOTE: There is other work also targeting this, so this may be removed depending on merge order. Branch: GraniteSWAForCausalLM AI-usage: full (Bob) Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * fix(conversion): Fix SWA pattern logic and support for non-rope layers Branch: GraniteSWAForCausalLM AI-usage: full (Bob) Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * feat(conversion): Add support for GraniteMoeSWA Branch: GraniteSWAForCausalLM AI-usage: full (Bob) Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * feat: Add llama_hparams::has_rope and arch constants NOTE: This shadows the work done for Granite Speech https://github.com/ggml-org/llama.cpp/pull/25107 Branch: GraniteSWAForCausalLM AI-usage: full (Bob) Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * feat: Add support for per-layer rope determination Branch: GraniteSWAForCausalLM AI-usage: full (Bob) Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * style: Fix failing flake8 for extra newlines Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * test: Write out SLIDING_WINDOW_PATTERN in llama-model-saver Branch: GraniteSWAForCausalLM AI-usage: full (OpenCode + Qwen3.6-35b) Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * fix(convert): Fix missing registration for GraniteMoeSWAForCausalLM Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * fix: Load MoE params as optional Branch: GraniteSWAForCausalLM AI-usage: draft (OpenCode + Qwen3.6-35b) Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * feat: Handle MoE params in conversion branch: GraniteSWAForCausalLM AI-usage: full (OpenCode + Qwen3.6-35b) Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * style: Remove unnecessary newline AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * fix: Remove unnecessary tensor additions to GRANITE architecture Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * fix: Correctly handle naming for ffn gate inp Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * fix: Always default hparams.rope_pattern to 1s This isn't strictly necessary, but it will allow other models to rely on hparams.has_rope(il) without needting to prepopulate. Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * feat: Move to has_rope for all granite model architectures Now that we have a proper hparam for this, it's better to use it and not require a hacky fallback in the hparam method itself. Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * feat: No hacky rope_finetuned fallback in has_rope Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * fix: Fully remove rope hparam filling in granitemoe There are no granitemoe models that use NoPE (it's not actually used in the layer building below), so this was just dead code. Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * fix: Save out rope_pattern in model-saver Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * fix: Set hparams.rope_finetuned for round trip Since the value is _read_ from rope_finetuned, we need to persist it when the model is saved with the saver. Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * fix: Code review cleanup Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co> Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co> * refactor: Keep gate/up fused for MoE path Branch: GraniteSWAForCausalLM AI-usage: full (Claude + Sonnet 5) Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * fix: Skip GRANITE_SWA in model saver https://github.com/ggml-org/llama.cpp/pull/25505#discussion_r3773175651 Keeping is_swa_impl in the saver can break other models. Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * add sliding window pattern for model in test * style: Fix indentation Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * fix: Fix \r\n Thanks Claude! Branch: GraniteSWAForCausalLM AI-usage: none Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * feat: Keep shared expert fused Branch: GraniteSWAForCausalLM AI-usage: full (Claude + Sonnet 5) Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> * style: More indentation fixes Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co> --------- Signed-off-by: Gabe Goodhart <ghart@us.ibm.com> Co-authored-by: Sigbjørn Skjæret <sigbjorn.skjaeret@huggingface.co>
308 lines
8.0 KiB
C++
308 lines
8.0 KiB
C++
#include "llama-hparams.h"
|
|
|
|
#include "ggml.h"
|
|
|
|
#include <algorithm>
|
|
#include <cassert>
|
|
|
|
void llama_hparams::set_swa_pattern(uint32_t n_pattern, bool dense_first) {
|
|
if (dense_first) {
|
|
for (uint32_t il = 0; il < n_layer(); ++il) {
|
|
is_swa_impl[il] = n_pattern == 0 || (il % n_pattern != 0);
|
|
}
|
|
} else {
|
|
for (uint32_t il = 0; il < n_layer(); ++il) {
|
|
is_swa_impl[il] = n_pattern == 0 || (il % n_pattern < (n_pattern - 1));
|
|
}
|
|
}
|
|
|
|
for (uint32_t il = n_layer(); il < n_layer_all; ++il) {
|
|
is_swa_impl[il] = false;
|
|
}
|
|
}
|
|
|
|
void llama_hparams::set_recr_pattern(uint32_t n_pattern, bool dense_first) {
|
|
if (dense_first) {
|
|
for (uint32_t il = 0; il < n_layer(); ++il) {
|
|
is_recr_impl[il] = n_pattern == 0 || (il % n_pattern != 0);
|
|
}
|
|
} else {
|
|
for (uint32_t il = 0; il < n_layer(); ++il) {
|
|
is_recr_impl[il] = n_pattern == 0 || (il % n_pattern < (n_pattern - 1));
|
|
}
|
|
}
|
|
|
|
for (uint32_t il = n_layer(); il < n_layer_all; ++il) {
|
|
is_recr_impl[il] = false;
|
|
}
|
|
}
|
|
|
|
bool llama_hparams::is_swa_any() const {
|
|
for (uint32_t il = 0; il < n_layer_all; ++il) {
|
|
if (is_swa_impl[il]) {
|
|
return true;
|
|
}
|
|
}
|
|
|
|
return false;
|
|
}
|
|
|
|
uint32_t llama_hparams::n_head(uint32_t il) const {
|
|
if (il < n_layer_all) {
|
|
return n_head_arr[il];
|
|
}
|
|
|
|
GGML_ABORT("fatal error");
|
|
}
|
|
|
|
uint32_t llama_hparams::n_head_kv(uint32_t il) const {
|
|
if (il < n_layer_all) {
|
|
return n_head_kv_arr[il];
|
|
}
|
|
|
|
GGML_ABORT("fatal error");
|
|
}
|
|
|
|
uint32_t llama_hparams::n_ff(uint32_t il) const {
|
|
if (il < n_layer_all) {
|
|
return n_ff_arr[il];
|
|
}
|
|
|
|
GGML_ABORT("fatal error");
|
|
}
|
|
|
|
uint32_t llama_hparams::n_gqa(uint32_t il) const {
|
|
const uint32_t n_head = this->n_head(il);
|
|
const uint32_t n_head_kv = this->n_head_kv(il);
|
|
|
|
if (n_head_kv == 0) {
|
|
return 0;
|
|
}
|
|
|
|
return n_head/n_head_kv;
|
|
}
|
|
|
|
uint32_t llama_hparams::n_rot(uint32_t il) const {
|
|
if (il < n_layer_all) {
|
|
return is_swa(il) ? n_rot_swa : n_rot_full;
|
|
}
|
|
|
|
GGML_ABORT("fatal error");
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_inp() const {
|
|
if (n_embd_inp_impl > 0) {
|
|
return n_embd_inp_impl;
|
|
}
|
|
|
|
uint32_t n_embd_inp = n_embd;
|
|
|
|
if (n_deepstack_layers > 0) {
|
|
n_embd_inp += n_embd * n_deepstack_layers;
|
|
}
|
|
|
|
return n_embd_inp;
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_inp_enc() const {
|
|
return n_embd_inp_enc_impl > 0 ? n_embd_inp_enc_impl : n_embd_inp();
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_out() const {
|
|
return n_embd_out_impl > 0 ? n_embd_out_impl : n_embd;
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_head_k(uint32_t il) const {
|
|
if (il < n_layer_all) {
|
|
return is_swa(il) ? n_embd_head_k_swa : n_embd_head_k_full;
|
|
}
|
|
|
|
GGML_ABORT("fatal error");
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_head_v(uint32_t il) const {
|
|
if (il < n_layer_all) {
|
|
return is_swa(il) ? n_embd_head_v_swa : n_embd_head_v_full;
|
|
}
|
|
|
|
GGML_ABORT("fatal error");
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_k_gqa(uint32_t il) const {
|
|
const uint32_t n_head_kv = this->n_head_kv(il);
|
|
|
|
return n_embd_head_k(il) * n_head_kv;
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_v_gqa(uint32_t il) const {
|
|
const uint32_t n_head_kv = this->n_head_kv(il);
|
|
|
|
return n_embd_head_v(il) * n_head_kv;
|
|
}
|
|
|
|
bool llama_hparams::is_n_embd_k_gqa_variable() const {
|
|
const uint32_t val = n_embd_k_gqa();
|
|
for (uint32_t il = 0; il < n_layer_all; ++il) {
|
|
if (val != n_embd_k_gqa(il)) {
|
|
return true;
|
|
}
|
|
}
|
|
|
|
return false;
|
|
}
|
|
|
|
bool llama_hparams::is_n_embd_v_gqa_variable() const {
|
|
const uint32_t val = n_embd_v_gqa();
|
|
for (uint32_t il = 0; il < n_layer_all; ++il) {
|
|
if (val != n_embd_v_gqa(il)) {
|
|
return true;
|
|
}
|
|
}
|
|
|
|
return false;
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_k_gqa_max() const {
|
|
uint32_t val = n_embd_k_gqa();
|
|
for (uint32_t il = 0; il < n_layer_all; ++il) {
|
|
val = std::max(val, n_embd_k_gqa(il));
|
|
}
|
|
|
|
return val;
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_v_gqa_max() const {
|
|
uint32_t val = n_embd_v_gqa();
|
|
for (uint32_t il = 0; il < n_layer_all; ++il) {
|
|
val = std::max(val, n_embd_v_gqa(il));
|
|
}
|
|
|
|
return val;
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_r() const {
|
|
if (wkv_head_size != 0) {
|
|
// for RWKV models
|
|
return token_shift_count * n_embd;
|
|
}
|
|
|
|
if (n_shortconv_l_cache != 0) {
|
|
// for LFM2 models
|
|
return n_embd * (n_shortconv_l_cache - 1);
|
|
}
|
|
|
|
if (n_embd_head_kda != 0) {
|
|
// for Kimi KDA layers
|
|
// Conv state for Q, K, V: 3 * (d_conv - 1) * n_head * head_dim
|
|
const uint32_t d_inner = n_head() * n_embd_head_kda; // 32 * 128 = 4096
|
|
return 3 * (ssm_d_conv > 0 ? ssm_d_conv - 1 : 3) * d_inner;
|
|
}
|
|
|
|
// TODO: maybe support other convolution strides than 1
|
|
// NOTE: since the first column of the conv_state is shifted out each time, it's not actually needed
|
|
// Corresponds to Mamba's conv_states size
|
|
return (ssm_d_conv > 0 ? ssm_d_conv - 1 : 0) * (ssm_d_inner + 2*ssm_n_group*ssm_d_state);
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_s() const {
|
|
if (wkv_head_size != 0) {
|
|
// corresponds to RWKV's wkv_states size
|
|
return n_embd * wkv_head_size;
|
|
}
|
|
|
|
if (n_embd_head_kda != 0) {
|
|
// for Kimi KDA layers
|
|
// Full recurrent state: head_dim * head_dim * n_head
|
|
// h tensor shape for delta attention: [head_dim, head_dim, n_head]
|
|
return n_embd_head_kda * n_embd_head_kda * n_head(); // 128 * 128 * 32 = 524288
|
|
}
|
|
|
|
if (n_embd_head_la != 0) {
|
|
// for MiniMax-Text-01 linear attention layers
|
|
// Full recurrent state: head_dim * head_dim * n_head
|
|
// tensor shape for linear attention: [head_dim, head_dim, n_head]
|
|
return n_embd_head_la * n_embd_head_la * n_head(); // 128 * 128 * 64 = 1048576
|
|
}
|
|
|
|
// corresponds to Mamba's ssm_states size
|
|
return ssm_d_state * ssm_d_inner;
|
|
}
|
|
|
|
bool llama_hparams::is_recr(uint32_t il) const {
|
|
if (il < n_layer_all) {
|
|
return is_recr_impl[il];
|
|
}
|
|
|
|
GGML_ABORT("%s: il (%u) out of bounds (n_layer_all: %u)\n", __func__, il, n_layer_all);
|
|
}
|
|
|
|
uint32_t llama_hparams::n_pos_per_embd() const {
|
|
return rope_type == LLAMA_ROPE_TYPE_MROPE || rope_type == LLAMA_ROPE_TYPE_IMROPE ? 4 : 1;
|
|
}
|
|
|
|
bool llama_hparams::is_swa(uint32_t il) const {
|
|
if (il < n_layer_all) {
|
|
return is_swa_impl[il];
|
|
}
|
|
|
|
GGML_ABORT("%s: il (%u) out of bounds (n_layer_all: %u)\n", __func__, il, n_layer_all);
|
|
}
|
|
|
|
bool llama_hparams::is_mla() const {
|
|
assert((n_embd_head_k_mla_impl == 0 && n_embd_head_v_mla_impl == 0) ||
|
|
(n_embd_head_k_mla_impl != 0 && n_embd_head_v_mla_impl != 0));
|
|
|
|
return n_embd_head_k_mla_impl != 0 && n_embd_head_v_mla_impl != 0;
|
|
}
|
|
|
|
bool llama_hparams::is_indexer_full(uint32_t il) const {
|
|
if (il < n_layer()) {
|
|
return is_indexer_full_impl[il];
|
|
}
|
|
|
|
GGML_ABORT("%s: il (%u) out of bounds (n_layer: %u)\n", __func__, il, n_layer());
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_head_k_mla() const {
|
|
return is_mla() ? n_embd_head_k_mla_impl : n_embd_head_k();
|
|
}
|
|
|
|
uint32_t llama_hparams::n_embd_head_v_mla() const {
|
|
return is_mla() ? n_embd_head_v_mla_impl : n_embd_head_v();
|
|
}
|
|
|
|
bool llama_hparams::has_kv(uint32_t il) const {
|
|
if (n_layer_kv_from_start >= 0) {
|
|
if (il < (uint32_t) n_layer_kv_from_start) {
|
|
return true;
|
|
}
|
|
|
|
return false;
|
|
}
|
|
|
|
// by default, all layers have kv
|
|
return true;
|
|
}
|
|
|
|
bool llama_hparams::has_rope(uint32_t il) const {
|
|
// the router layer stores adapter routing signal, not positional info,
|
|
// so it must not be RoPE-shifted
|
|
if (router_layer >= 0 && (int32_t) il == router_layer) {
|
|
return false;
|
|
}
|
|
|
|
if (il < n_layer_all) {
|
|
return rope_pattern[il] != 0;
|
|
}
|
|
|
|
GGML_ABORT("%s: il (%u) out of bounds (n_layer_all: %u)\n", __func__, il, n_layer_all);
|
|
}
|
|
|
|
uint32_t llama_hparams::n_layer() const {
|
|
return n_layer_all - n_layer_nextn;
|
|
}
|
|
|
|
bool llama_hparams::use_mrope() const {
|
|
return rope_sections[0] > 0 && rope_sections[1] > 0;
|
|
}
|