diff --git a/common/arg.cpp b/common/arg.cpp
index b78b74b8c..d9e0bedf1 100644
--- a/common/arg.cpp
+++ b/common/arg.cpp
@@ -3455,7 +3455,8 @@ common_params_context common_params_parser_init(common_params & params, llama_ex
{"--reasoning-format"}, "FORMAT",
"controls whether thought tags are allowed and/or extracted from the response, and in which format they're returned; one of:\n"
"- none: leaves thoughts unparsed in `message.content`\n"
- "- deepseek: puts thoughts in `message.reasoning_content` (except in streaming mode, which behaves as `none`)\n"
+ "- deepseek: puts thoughts in `message.reasoning_content`\n"
+ "- deepseek-legacy: keeps `` tags in `message.content` while also populating `message.reasoning_content`\n"
"(default: auto)",
[](common_params & params, const std::string & value) {
params.reasoning_format = common_reasoning_format_from_name(value);
diff --git a/common/chat-parser.cpp b/common/chat-parser.cpp
index b3362519a..7365782e7 100644
--- a/common/chat-parser.cpp
+++ b/common/chat-parser.cpp
@@ -3,9 +3,12 @@
#include "log.h"
#include "regex-partial.h"
+#include
+#include
#include
#include
#include
+#include
#include
using json = nlohmann::ordered_json;
@@ -166,6 +169,27 @@ void common_chat_msg_parser::consume_literal(const std::string & literal) {
}
bool common_chat_msg_parser::try_parse_reasoning(const std::string & start_think, const std::string & end_think) {
+ std::string pending_reasoning_prefix;
+
+ if (syntax_.reasoning_format == COMMON_REASONING_FORMAT_NONE) {
+ return false;
+ }
+
+ auto set_reasoning_prefix = [&](size_t prefix_pos) {
+ if (!syntax_.thinking_forced_open || syntax_.reasoning_in_content) {
+ return;
+ }
+ if (prefix_pos + start_think.size() > input_.size()) {
+ pending_reasoning_prefix.clear();
+ return;
+ }
+ // Capture the exact literal that opened the reasoning section so we can
+ // surface it back to callers. This ensures formats that force the
+ // reasoning tag open (e.g. DeepSeek R1) retain their original prefix
+ // instead of dropping it during parsing.
+ pending_reasoning_prefix = input_.substr(prefix_pos, start_think.size());
+ };
+
auto handle_reasoning = [&](const std::string & reasoning, bool closed) {
auto stripped_reasoning = string_strip(reasoning);
if (stripped_reasoning.empty()) {
@@ -178,28 +202,116 @@ bool common_chat_msg_parser::try_parse_reasoning(const std::string & start_think
add_content(syntax_.reasoning_format == COMMON_REASONING_FORMAT_DEEPSEEK ? "" : end_think);
}
} else {
+ if (!pending_reasoning_prefix.empty()) {
+ add_reasoning_content(pending_reasoning_prefix);
+ pending_reasoning_prefix.clear();
+ }
add_reasoning_content(stripped_reasoning);
}
};
- if (syntax_.reasoning_format != COMMON_REASONING_FORMAT_NONE) {
- if (syntax_.thinking_forced_open || try_consume_literal(start_think)) {
- if (auto res = try_find_literal(end_think)) {
- handle_reasoning(res->prelude, /* closed */ true);
- consume_spaces();
- return true;
- }
- auto rest = consume_rest();
+
+ const size_t saved_pos = pos_;
+ const size_t saved_content_size = result_.content.size();
+ const size_t saved_reasoning_size = result_.reasoning_content.size();
+
+ auto restore_state = [&]() {
+ move_to(saved_pos);
+ result_.content.resize(saved_content_size);
+ result_.reasoning_content.resize(saved_reasoning_size);
+ };
+
+ // Allow leading whitespace to be preserved as content when reasoning is present at the start
+ size_t cursor = pos_;
+ size_t whitespace_end = cursor;
+ while (whitespace_end < input_.size() && std::isspace(static_cast(input_[whitespace_end]))) {
+ ++whitespace_end;
+ }
+
+ if (whitespace_end >= input_.size()) {
+ restore_state();
+ if (syntax_.thinking_forced_open) {
+ auto rest = input_.substr(saved_pos);
if (!rest.empty()) {
handle_reasoning(rest, /* closed */ !is_partial());
}
- // Allow unclosed thinking tags, for now (https://github.com/ggml-org/llama.cpp/issues/13812, https://github.com/ggml-org/llama.cpp/issues/13877)
- // if (!syntax_.thinking_forced_open) {
- // throw common_chat_msg_partial_exception(end_think);
- // }
+ move_to(input_.size());
return true;
}
+ return false;
+ }
+
+ cursor = whitespace_end;
+ const size_t remaining = input_.size() - cursor;
+ const size_t start_prefix = std::min(start_think.size(), remaining);
+ const bool has_start_tag = input_.compare(cursor, start_prefix, start_think, 0, start_prefix) == 0;
+
+ if (has_start_tag && start_prefix < start_think.size()) {
+ move_to(input_.size());
+ return true;
+ }
+
+ if (has_start_tag) {
+ if (whitespace_end > pos_) {
+ add_content(input_.substr(pos_, whitespace_end - pos_));
+ }
+ set_reasoning_prefix(cursor);
+ cursor += start_think.size();
+ } else if (syntax_.thinking_forced_open) {
+ cursor = whitespace_end;
+ } else {
+ restore_state();
+ return false;
+ }
+ while (true) {
+ if (cursor >= input_.size()) {
+ move_to(input_.size());
+ return true;
+ }
+
+ size_t end_pos = input_.find(end_think, cursor);
+ if (end_pos == std::string::npos) {
+ std::string_view remaining_view(input_.data() + cursor, input_.size() - cursor);
+ size_t partial_off = string_find_partial_stop(remaining_view, end_think);
+ size_t reasoning_end = partial_off == std::string::npos ? input_.size() : cursor + partial_off;
+ if (reasoning_end > cursor) {
+ handle_reasoning(input_.substr(cursor, reasoning_end - cursor), /* closed */ partial_off == std::string::npos && !is_partial());
+ }
+ move_to(input_.size());
+ return true;
+ }
+
+ if (end_pos > cursor) {
+ handle_reasoning(input_.substr(cursor, end_pos - cursor), /* closed */ true);
+ } else {
+ handle_reasoning("", /* closed */ true);
+ }
+
+ cursor = end_pos + end_think.size();
+
+ while (cursor < input_.size() && std::isspace(static_cast(input_[cursor]))) {
+ ++cursor;
+ }
+
+ const size_t next_remaining = input_.size() - cursor;
+ if (next_remaining == 0) {
+ move_to(cursor);
+ return true;
+ }
+
+ const size_t next_prefix = std::min(start_think.size(), next_remaining);
+ if (input_.compare(cursor, next_prefix, start_think, 0, next_prefix) == 0) {
+ if (next_prefix < start_think.size()) {
+ move_to(input_.size());
+ return true;
+ }
+ set_reasoning_prefix(cursor);
+ cursor += start_think.size();
+ continue;
+ }
+
+ move_to(cursor);
+ return true;
}
- return false;
}
std::string common_chat_msg_parser::consume_rest() {
diff --git a/common/chat.cpp b/common/chat.cpp
index 76e645bae..9c9426625 100644
--- a/common/chat.cpp
+++ b/common/chat.cpp
@@ -1408,6 +1408,8 @@ static common_chat_params common_chat_params_init_apertus(const common_chat_temp
return data;
}
static void common_chat_parse_llama_3_1(common_chat_msg_parser & builder, bool with_builtin_tools = false) {
+ builder.try_parse_reasoning("", "");
+
if (!builder.syntax().parse_tool_calls) {
builder.add_content(builder.consume_rest());
return;
@@ -2862,6 +2864,7 @@ common_chat_params common_chat_templates_apply(
}
static void common_chat_parse_content_only(common_chat_msg_parser & builder) {
+ builder.try_parse_reasoning("", "");
builder.add_content(builder.consume_rest());
}
diff --git a/common/common.h b/common/common.h
index 01d5ed0e3..8554cbabf 100644
--- a/common/common.h
+++ b/common/common.h
@@ -429,7 +429,7 @@ struct common_params {
std::string chat_template = ""; // NOLINT
bool use_jinja = false; // NOLINT
bool enable_chat_template = true;
- common_reasoning_format reasoning_format = COMMON_REASONING_FORMAT_AUTO;
+ common_reasoning_format reasoning_format = COMMON_REASONING_FORMAT_DEEPSEEK;
int reasoning_budget = -1;
bool prefill_assistant = true; // if true, any trailing assistant message will be prefilled into the response
diff --git a/convert_hf_to_gguf.py b/convert_hf_to_gguf.py
index 15edb59f0..b11eb8e35 100755
--- a/convert_hf_to_gguf.py
+++ b/convert_hf_to_gguf.py
@@ -96,13 +96,15 @@ class ModelBase:
# Mistral format specifics
is_mistral_format: bool = False
disable_mistral_community_chat_template: bool = False
+ sentence_transformers_dense_modules: bool = False
def __init__(self, dir_model: Path, ftype: gguf.LlamaFileType, fname_out: Path, *, is_big_endian: bool = False,
use_temp_file: bool = False, eager: bool = False,
metadata_override: Path | None = None, model_name: str | None = None,
split_max_tensors: int = 0, split_max_size: int = 0, dry_run: bool = False,
small_first_shard: bool = False, hparams: dict[str, Any] | None = None, remote_hf_model_id: str | None = None,
- disable_mistral_community_chat_template: bool = False):
+ disable_mistral_community_chat_template: bool = False,
+ sentence_transformers_dense_modules: bool = False):
if type(self) is ModelBase or \
type(self) is TextModel or \
type(self) is MmprojModel:
@@ -117,6 +119,7 @@ class ModelBase:
self.lazy = not eager or (remote_hf_model_id is not None)
self.dry_run = dry_run
self.remote_hf_model_id = remote_hf_model_id
+ self.sentence_transformers_dense_modules = sentence_transformers_dense_modules
if remote_hf_model_id is not None:
self.is_safetensors = True
@@ -5274,6 +5277,53 @@ class Gemma3Model(TextModel):
@ModelBase.register("Gemma3TextModel")
class EmbeddingGemma(Gemma3Model):
model_arch = gguf.MODEL_ARCH.GEMMA_EMBEDDING
+ module_paths = []
+ dense_features_dims = {}
+
+ def __init__(self, *args, **kwargs):
+ super().__init__(*args, **kwargs)
+ if self.sentence_transformers_dense_modules:
+ # read modules.json to determine if model has Dense layers
+ modules_file = self.dir_model / "modules.json"
+ if modules_file.is_file():
+ with open(modules_file, encoding="utf-8") as modules_json_file:
+ mods = json.load(modules_json_file)
+ for mod in mods:
+ if mod["type"] == "sentence_transformers.models.Dense":
+ mod_path = mod["path"]
+ # check if model.safetensors file for Dense layer exists
+ model_tensors_file = self.dir_model / mod_path / "model.safetensors"
+ if model_tensors_file.is_file():
+ self.module_paths.append(mod_path)
+ # read config.json of the Dense layer to get in/out features
+ mod_conf_file = self.dir_model / mod_path / "config.json"
+ if mod_conf_file.is_file():
+ with open(mod_conf_file, encoding="utf-8") as mod_conf_json_file:
+ mod_conf = json.load(mod_conf_json_file)
+ # hparams dense_2_feat_out and dense_3_feat_in are required when loading model's dense weights
+ prefix = self._get_dense_prefix(mod_path)
+ if mod_conf["in_features"] is not None and mod_conf["out_features"] is not None:
+ self.dense_features_dims[prefix] = (mod_conf["in_features"], mod_conf["out_features"])
+
+ def generate_extra_tensors(self) -> Iterable[tuple[str, Tensor]]:
+ from safetensors.torch import load_file
+ module_paths = list(self.module_paths)
+ for i, module_path in enumerate(module_paths):
+ tensors_file = self.dir_model / module_path / "model.safetensors"
+ local_tensors = load_file(tensors_file)
+ tensor_name = self._get_dense_prefix(module_path)
+ for name, local_tensor in local_tensors.items():
+ if not name.endswith(".weight"):
+ continue
+ orig_name = name.replace("linear", tensor_name)
+ name = self.map_tensor_name(orig_name)
+ yield name, local_tensor.clone()
+
+ @staticmethod
+ def _get_dense_prefix(module_path) -> str:
+ """Get the tensor name prefix for the Dense layer from module path."""
+ tensor_name = "dense_2" if module_path == "2_Dense" else "dense_3"
+ return tensor_name
def set_gguf_parameters(self):
super().set_gguf_parameters()
@@ -5290,6 +5340,10 @@ class EmbeddingGemma(Gemma3Model):
logger.info(f"Using original sliding_window from config: {orig_sliding_window} "
f"instead of {self.hparams['sliding_window']}")
self.gguf_writer.add_sliding_window(orig_sliding_window)
+ if self.sentence_transformers_dense_modules:
+ for dense, dims in self.dense_features_dims.items():
+ logger.info(f"Setting dense layer {dense} in/out features to {dims}")
+ self.gguf_writer.add_dense_features_dims(dense, dims[0], dims[1])
self._try_set_pooling_type()
@@ -9340,6 +9394,13 @@ def parse_args() -> argparse.Namespace:
)
)
+ parser.add_argument(
+ "--sentence-transformers-dense-modules", action="store_true",
+ help=("Whether to include sentence-transformers dense modules."
+ "It can be used for sentence-transformers models, like google/embeddinggemma-300m"
+ "Default these modules are not included.")
+ )
+
args = parser.parse_args()
if not args.print_supported_models and args.model is None:
parser.error("the following arguments are required: model")
@@ -9402,9 +9463,13 @@ def main() -> None:
if args.remote:
hf_repo_id = args.model
from huggingface_hub import snapshot_download
+ allowed_patterns = ["LICENSE", "*.json", "*.md", "*.txt", "tokenizer.model"]
+ if args.sentence_transformers_dense_modules:
+ # include sentence-transformers dense modules safetensors files
+ allowed_patterns.append("*.safetensors")
local_dir = snapshot_download(
repo_id=hf_repo_id,
- allow_patterns=["LICENSE", "*.json", "*.md", "*.txt", "tokenizer.model"])
+ allow_patterns=allowed_patterns)
dir_model = Path(local_dir)
logger.info(f"Downloaded config and tokenizer to {local_dir}")
else:
@@ -9472,7 +9537,8 @@ def main() -> None:
split_max_tensors=args.split_max_tensors,
split_max_size=split_str_to_n_bytes(args.split_max_size), dry_run=args.dry_run,
small_first_shard=args.no_tensor_first_split,
- remote_hf_model_id=hf_repo_id, disable_mistral_community_chat_template=disable_mistral_community_chat_template
+ remote_hf_model_id=hf_repo_id, disable_mistral_community_chat_template=disable_mistral_community_chat_template,
+ sentence_transformers_dense_modules=args.sentence_transformers_dense_modules
)
if args.vocab_only:
diff --git a/ggml/src/ggml-cuda/ggml-cuda.cu b/ggml/src/ggml-cuda/ggml-cuda.cu
index 48cb9eeed..5c5d00514 100644
--- a/ggml/src/ggml-cuda/ggml-cuda.cu
+++ b/ggml/src/ggml-cuda/ggml-cuda.cu
@@ -234,7 +234,7 @@ static ggml_cuda_device_info ggml_cuda_init() {
info.default_tensor_split[id] = total_vram;
total_vram += prop.totalGlobalMem;
- info.devices[id].integrated = prop.integrated;
+ info.devices[id].integrated = false; // Temporarily disabled due to issues with corrupted output (e.g. #15034)
info.devices[id].nsm = prop.multiProcessorCount;
info.devices[id].smpb = prop.sharedMemPerBlock;
info.devices[id].warp_size = prop.warpSize;
diff --git a/gguf-py/gguf/constants.py b/gguf-py/gguf/constants.py
index 9c99b90fa..f5e5fba80 100644
--- a/gguf-py/gguf/constants.py
+++ b/gguf-py/gguf/constants.py
@@ -128,6 +128,8 @@ class Keys:
ALTUP_ACTIVE_IDX = "{arch}.altup.active_idx"
ALTUP_NUM_INPUTS = "{arch}.altup.num_inputs"
EMBD_LENGTH_PER_LAYER_INP = "{arch}.embedding_length_per_layer_input"
+ DENSE_FEAT_IN_SIZE = "{arch}.{dense}_feat_in"
+ DENSE_FEAT_OUT_SIZE = "{arch}.{dense}_feat_out"
class Attention:
HEAD_COUNT = "{arch}.attention.head_count"
@@ -433,6 +435,8 @@ class MODEL_TENSOR(IntEnum):
TOKEN_TYPES = auto()
POS_EMBD = auto()
OUTPUT = auto()
+ DENSE_2_OUT = auto() # embeddinggemma 2_Dense
+ DENSE_3_OUT = auto() # embeddinggemma 3_Dense
OUTPUT_NORM = auto()
ROPE_FREQS = auto()
ROPE_FACTORS_LONG = auto()
@@ -777,6 +781,8 @@ TENSOR_NAMES: dict[MODEL_TENSOR, str] = {
MODEL_TENSOR.POS_EMBD: "position_embd",
MODEL_TENSOR.OUTPUT_NORM: "output_norm",
MODEL_TENSOR.OUTPUT: "output",
+ MODEL_TENSOR.DENSE_2_OUT: "dense_2", # embeddinggemma 2_Dense
+ MODEL_TENSOR.DENSE_3_OUT: "dense_3", # embeddinggemma 2_Dense
MODEL_TENSOR.ROPE_FREQS: "rope_freqs",
MODEL_TENSOR.ROPE_FACTORS_LONG: "rope_factors_long",
MODEL_TENSOR.ROPE_FACTORS_SHORT: "rope_factors_short",
@@ -1759,6 +1765,8 @@ MODEL_TENSORS: dict[MODEL_ARCH, list[MODEL_TENSOR]] = {
MODEL_ARCH.GEMMA_EMBEDDING: [
MODEL_TENSOR.TOKEN_EMBD,
MODEL_TENSOR.OUTPUT,
+ MODEL_TENSOR.DENSE_2_OUT,
+ MODEL_TENSOR.DENSE_3_OUT,
MODEL_TENSOR.OUTPUT_NORM,
MODEL_TENSOR.ATTN_Q,
MODEL_TENSOR.ATTN_Q_NORM,
diff --git a/gguf-py/gguf/gguf_writer.py b/gguf-py/gguf/gguf_writer.py
index dfe4bfd49..306679e21 100644
--- a/gguf-py/gguf/gguf_writer.py
+++ b/gguf-py/gguf/gguf_writer.py
@@ -730,6 +730,10 @@ class GGUFWriter:
def add_sliding_window_pattern(self, value: Sequence[bool]) -> None:
self.add_array(Keys.Attention.SLIDING_WINDOW_PATTERN.format(arch=self.arch), value)
+ def add_dense_features_dims(self, dense:str, in_f:int, out_f:int) -> None:
+ self.add_uint32(Keys.LLM.DENSE_FEAT_IN_SIZE.format(arch=self.arch, dense=dense), in_f)
+ self.add_uint32(Keys.LLM.DENSE_FEAT_OUT_SIZE.format(arch=self.arch, dense=dense), out_f)
+
def add_logit_scale(self, value: float) -> None:
self.add_float32(Keys.LLM.LOGIT_SCALE.format(arch=self.arch), value)
diff --git a/gguf-py/gguf/tensor_mapping.py b/gguf-py/gguf/tensor_mapping.py
index 3e9a2dd8f..c05aa6cc4 100644
--- a/gguf-py/gguf/tensor_mapping.py
+++ b/gguf-py/gguf/tensor_mapping.py
@@ -76,7 +76,12 @@ class TensorNameMap:
"lm_head", # llama4
"model.transformer.ff_out", # llada
),
-
+ MODEL_TENSOR.DENSE_2_OUT: (
+ "dense_2_out", # embeddinggemma
+ ),
+ MODEL_TENSOR.DENSE_3_OUT: (
+ "dense_3_out", # embeddinggemma
+ ),
# Output norm
MODEL_TENSOR.OUTPUT_NORM: (
"gpt_neox.final_layer_norm", # gptneox
diff --git a/src/llama-arch.cpp b/src/llama-arch.cpp
index 45f0d0e2c..869e4dccf 100644
--- a/src/llama-arch.cpp
+++ b/src/llama-arch.cpp
@@ -219,6 +219,11 @@ static const std::map LLM_KV_NAMES = {
{ LLM_KV_CLASSIFIER_OUTPUT_LABELS, "%s.classifier.output_labels" },
{ LLM_KV_SHORTCONV_L_CACHE, "%s.shortconv.l_cache" },
+ // sentence-transformers dense modules feature dims
+ { LLM_KV_DENSE_2_FEAT_IN, "%s.dense_2_feat_in" },
+ { LLM_KV_DENSE_2_FEAT_OUT, "%s.dense_2_feat_out" },
+ { LLM_KV_DENSE_3_FEAT_IN, "%s.dense_3_feat_in" },
+ { LLM_KV_DENSE_3_FEAT_OUT, "%s.dense_3_feat_out" },
{ LLM_KV_TOKENIZER_MODEL, "tokenizer.ggml.model" },
{ LLM_KV_TOKENIZER_PRE, "tokenizer.ggml.pre" },
@@ -1071,6 +1076,8 @@ static const std::map> LLM_TENSOR_N
{ LLM_TENSOR_TOKEN_EMBD, "token_embd" },
{ LLM_TENSOR_OUTPUT_NORM, "output_norm" },
{ LLM_TENSOR_OUTPUT, "output" },
+ { LLM_TENSOR_DENSE_2_OUT, "dense_2" },
+ { LLM_TENSOR_DENSE_3_OUT, "dense_3" },
{ LLM_TENSOR_ATTN_NORM, "blk.%d.attn_norm" },
{ LLM_TENSOR_ATTN_Q, "blk.%d.attn_q" },
{ LLM_TENSOR_ATTN_Q_NORM, "blk.%d.attn_q_norm" },
@@ -2281,6 +2288,8 @@ static const std::map LLM_TENSOR_INFOS = {
{LLM_TENSOR_OUTPUT, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
{LLM_TENSOR_CLS, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
{LLM_TENSOR_CLS_OUT, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}},
+ {LLM_TENSOR_DENSE_2_OUT, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}}, // Dense layer output
+ {LLM_TENSOR_DENSE_3_OUT, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL_MAT}}, // Dense layer output
{LLM_TENSOR_OUTPUT_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
{LLM_TENSOR_DEC_OUTPUT_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
{LLM_TENSOR_ENC_OUTPUT_NORM, {LLM_TENSOR_LAYER_OUTPUT, GGML_OP_MUL}},
diff --git a/src/llama-arch.h b/src/llama-arch.h
index 507fe5f37..c3ae71655 100644
--- a/src/llama-arch.h
+++ b/src/llama-arch.h
@@ -271,6 +271,12 @@ enum llm_kv {
LLM_KV_TOKENIZER_PREFIX_ID,
LLM_KV_TOKENIZER_SUFFIX_ID,
LLM_KV_TOKENIZER_MIDDLE_ID,
+
+ // sentence-transformers dense layers in and out features
+ LLM_KV_DENSE_2_FEAT_IN,
+ LLM_KV_DENSE_2_FEAT_OUT,
+ LLM_KV_DENSE_3_FEAT_IN,
+ LLM_KV_DENSE_3_FEAT_OUT,
};
enum llm_tensor {
@@ -278,6 +284,8 @@ enum llm_tensor {
LLM_TENSOR_TOKEN_EMBD_NORM,
LLM_TENSOR_TOKEN_TYPES,
LLM_TENSOR_POS_EMBD,
+ LLM_TENSOR_DENSE_2_OUT,
+ LLM_TENSOR_DENSE_3_OUT,
LLM_TENSOR_OUTPUT,
LLM_TENSOR_OUTPUT_NORM,
LLM_TENSOR_ROPE_FREQS,
diff --git a/src/llama-context.cpp b/src/llama-context.cpp
index 2778e278f..311a5a9ae 100644
--- a/src/llama-context.cpp
+++ b/src/llama-context.cpp
@@ -2346,6 +2346,12 @@ llama_context * llama_init_from_model(
return nullptr;
}
+ if (params.pooling_type != model->hparams.pooling_type) {
+ //user-specified pooling-type is different from the model default
+ LLAMA_LOG_WARN("%s: model default pooling_type is [%d], but [%d] was specified\n", __func__,
+ model->hparams.pooling_type, params.pooling_type);
+ }
+
try {
auto * ctx = new llama_context(*model, params);
return ctx;
diff --git a/src/llama-graph.cpp b/src/llama-graph.cpp
index 90cd885a6..a24853c63 100644
--- a/src/llama-graph.cpp
+++ b/src/llama-graph.cpp
@@ -1853,6 +1853,23 @@ llm_graph_input_mem_hybrid * llm_graph_context::build_inp_mem_hybrid() const {
return (llm_graph_input_mem_hybrid *) res->add_input(std::move(inp));
}
+void llm_graph_context::build_dense_out(
+ ggml_tensor * dense_2,
+ ggml_tensor * dense_3) const {
+ if (!cparams.embeddings || dense_2 == nullptr || dense_3 == nullptr) {
+ return;
+ }
+ ggml_tensor * cur = res->t_embd_pooled != nullptr ? res->t_embd_pooled : res->t_embd;
+ GGML_ASSERT(cur != nullptr && "missing t_embd_pooled/t_embd");
+
+ cur = ggml_mul_mat(ctx0, dense_2, cur);
+ cur = ggml_mul_mat(ctx0, dense_3, cur);
+ cb(cur, "result_embd_pooled", -1);
+ res->t_embd_pooled = cur;
+ ggml_build_forward_expand(gf, cur);
+}
+
+
void llm_graph_context::build_pooling(
ggml_tensor * cls,
ggml_tensor * cls_b,
diff --git a/src/llama-graph.h b/src/llama-graph.h
index 34b984afe..dc84b7942 100644
--- a/src/llama-graph.h
+++ b/src/llama-graph.h
@@ -814,6 +814,14 @@ struct llm_graph_context {
ggml_tensor * cls_b,
ggml_tensor * cls_out,
ggml_tensor * cls_out_b) const;
+
+ //
+ // dense (out)
+ //
+
+ void build_dense_out(
+ ggml_tensor * dense_2,
+ ggml_tensor * dense_3) const;
};
// TODO: better name
diff --git a/src/llama-hparams.h b/src/llama-hparams.h
index f29b23eef..4e7f73ec2 100644
--- a/src/llama-hparams.h
+++ b/src/llama-hparams.h
@@ -169,6 +169,12 @@ struct llama_hparams {
uint32_t laurel_rank = 64;
uint32_t n_embd_altup = 256;
+ // needed for sentence-transformers dense layers
+ uint32_t dense_2_feat_in = 0; // in_features of the 2_Dense
+ uint32_t dense_2_feat_out = 0; // out_features of the 2_Dense
+ uint32_t dense_3_feat_in = 0; // in_features of the 3_Dense
+ uint32_t dense_3_feat_out = 0; // out_features of the 3_Dense
+
// xIELU
std::array xielu_alpha_n;
std::array xielu_alpha_p;
diff --git a/src/llama-model.cpp b/src/llama-model.cpp
index a229d148a..8357e3d81 100644
--- a/src/llama-model.cpp
+++ b/src/llama-model.cpp
@@ -1223,12 +1223,21 @@ void llama_model::load_hparams(llama_model_loader & ml) {
hparams.set_swa_pattern(6);
hparams.causal_attn = false; // embeddings do not use causal attention
- hparams.rope_freq_base_train_swa = 10000.0f;
+ hparams.rope_freq_base_train_swa = 10000.0f;
hparams.rope_freq_scale_train_swa = 1.0f;
- ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa);
+ ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa);
ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);
- ml.get_key(LLM_KV_POOLING_TYPE, hparams.pooling_type);
+ ml.get_key(LLM_KV_POOLING_TYPE, hparams.pooling_type);
+
+ //applied only if model converted with --sentence-transformers-dense-modules
+ ml.get_key(LLM_KV_DENSE_2_FEAT_IN, hparams.dense_2_feat_in, false);
+ ml.get_key(LLM_KV_DENSE_2_FEAT_OUT, hparams.dense_2_feat_out, false);
+ ml.get_key(LLM_KV_DENSE_3_FEAT_IN, hparams.dense_3_feat_in, false);
+ ml.get_key(LLM_KV_DENSE_3_FEAT_OUT, hparams.dense_3_feat_out, false);
+
+ GGML_ASSERT((hparams.dense_2_feat_in == 0 || hparams.dense_2_feat_in == hparams.n_embd) && "dense_2_feat_in must be equal to n_embd");
+ GGML_ASSERT((hparams.dense_3_feat_out == 0 || hparams.dense_3_feat_out == hparams.n_embd) && "dense_3_feat_out must be equal to n_embd");
switch (hparams.n_layer) {
case 24: type = LLM_TYPE_0_3B; break;
@@ -3744,6 +3753,11 @@ bool llama_model::load_tensors(llama_model_loader & ml) {
output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED);
}
+ // Dense linear weights
+ dense_2_out_layers = create_tensor(tn(LLM_TENSOR_DENSE_2_OUT, "weight"), {n_embd, hparams.dense_2_feat_out}, TENSOR_NOT_REQUIRED);
+ dense_3_out_layers = create_tensor(tn(LLM_TENSOR_DENSE_3_OUT, "weight"), {hparams.dense_3_feat_in, n_embd}, TENSOR_NOT_REQUIRED);
+
+
for (int i = 0; i < n_layer; ++i) {
auto & layer = layers[i];
@@ -19955,6 +19969,12 @@ ggml_cgraph * llama_model::build_graph(const llm_graph_params & params) const {
// add on pooling layer
llm->build_pooling(cls, cls_b, cls_out, cls_out_b);
+ // if the gguf model was converted with --sentence-transformers-dense-modules
+ // there will be two additional dense projection layers
+ // dense linear projections are applied after pooling
+ // TODO: move reranking logic here and generalize
+ llm->build_dense_out(dense_2_out_layers, dense_3_out_layers);
+
return llm->res->get_gf();
}
diff --git a/src/llama-model.h b/src/llama-model.h
index 20b59d952..7f48662f2 100644
--- a/src/llama-model.h
+++ b/src/llama-model.h
@@ -438,6 +438,12 @@ struct llama_model {
std::vector layers;
+ //Dense linear projections for SentenceTransformers models like embeddinggemma
+ // For Sentence Transformers models structure see
+ // https://sbert.net/docs/sentence_transformer/usage/custom_models.html#structure-of-sentence-transformer-models
+ struct ggml_tensor * dense_2_out_layers = nullptr;
+ struct ggml_tensor * dense_3_out_layers = nullptr;
+
llama_model_params params;
// gguf metadata
diff --git a/tools/server/public/index.html.gz b/tools/server/public/index.html.gz
index 8d57b4a16..550df72e9 100644
Binary files a/tools/server/public/index.html.gz and b/tools/server/public/index.html.gz differ
diff --git a/tools/server/webui/src/lib/components/app/chat/ChatMessages/ChatMessage.svelte b/tools/server/webui/src/lib/components/app/chat/ChatMessages/ChatMessage.svelte
index c923bf9e0..fed0cf712 100644
--- a/tools/server/webui/src/lib/components/app/chat/ChatMessages/ChatMessage.svelte
+++ b/tools/server/webui/src/lib/components/app/chat/ChatMessages/ChatMessage.svelte
@@ -1,7 +1,6 @@
{
+ const { updateConfig } = await import('$lib/stores/settings.svelte');
+ updateConfig('disableReasoningFormat', false);
+ }}
/>
{
+ const { updateConfig } = await import('$lib/stores/settings.svelte');
+ updateConfig('disableReasoningFormat', false);
+ }}
/>
{
+ const { updateConfig } = await import('$lib/stores/settings.svelte');
+ updateConfig('disableReasoningFormat', false);
+ }}
+/>
+
+ {
+ const { updateConfig } = await import('$lib/stores/settings.svelte');
+ updateConfig('disableReasoningFormat', true);
+ }}
+/>
+
+ {
+ const { updateConfig } = await import('$lib/stores/settings.svelte');
+ updateConfig('disableReasoningFormat', false);
// Phase 1: Stream reasoning content in chunks
let reasoningText =
'I need to think about this carefully. Let me break down the problem:\n\n1. The user is asking for help with something complex\n2. I should provide a thorough and helpful response\n3. I need to consider multiple approaches\n4. The best solution would be to explain step by step\n\nThis approach will ensure clarity and understanding.';
@@ -187,126 +192,16 @@
message: processingMessage
}}
play={async () => {
+ const { updateConfig } = await import('$lib/stores/settings.svelte');
+ updateConfig('disableReasoningFormat', false);
// Import the chat store to simulate loading state
const { chatStore } = await import('$lib/stores/chat.svelte');
-
+
// Set loading state to true to trigger the processing UI
chatStore.isLoading = true;
-
+
// Simulate the processing state hook behavior
// This will show the "Generating..." text and parameter details
- await new Promise(resolve => setTimeout(resolve, 100));
+ await new Promise((resolve) => setTimeout(resolve, 100));
}}
/>
-
-
-
-
-
- {
- // Phase 1: Stream reasoning content
- const thinkingContent =
- 'Let me work through this problem systematically:\n\n1. First, I need to understand what the user is asking\n2. Then I should consider different approaches\n3. I need to evaluate the pros and cons\n4. Finally, I should provide a clear recommendation\n\nThis step-by-step approach will ensure accuracy.';
-
- let currentContent = '\n';
- streamingThinkMessage.content = currentContent;
-
- for (let i = 0; i < thinkingContent.length; i++) {
- currentContent += thinkingContent[i];
- streamingThinkMessage.content = currentContent;
- await new Promise((resolve) => setTimeout(resolve, 5));
- }
-
- // Close the thinking block
- currentContent += '\n\n\n';
- streamingThinkMessage.content = currentContent;
- await new Promise((resolve) => setTimeout(resolve, 200));
-
- // Phase 2: Stream main response content
- const responseContent =
- "Based on my analysis above, here's the solution:\n\n**Key Points:**\n- The approach should be systematic\n- We need to consider all factors\n- Implementation should be step-by-step\n\nThis ensures the best possible outcome.";
-
- for (let i = 0; i < responseContent.length; i++) {
- currentContent += responseContent[i];
- streamingThinkMessage.content = currentContent;
- await new Promise((resolve) => setTimeout(resolve, 10));
- }
-
- streamingThinkMessage.timestamp = Date.now();
- }}
->
-
-
-
-
-
- {
- // Phase 1: Stream [THINK] reasoning content
- const thinkingContent =
- 'Using the DeepSeek format now:\n\n- This demonstrates the [THINK] bracket format\n- Should parse identically to <think> tags\n- The UI should display this in the thinking section\n- Main content should be separate\n\nBoth formats provide the same functionality.';
-
- let currentContent = '[THINK]\n';
- streamingBracketMessage.content = currentContent;
-
- for (let i = 0; i < thinkingContent.length; i++) {
- currentContent += thinkingContent[i];
- streamingBracketMessage.content = currentContent;
- await new Promise((resolve) => setTimeout(resolve, 5));
- }
-
- // Close the thinking block
- currentContent += '\n[/THINK]\n\n';
- streamingBracketMessage.content = currentContent;
- await new Promise((resolve) => setTimeout(resolve, 200));
-
- // Phase 2: Stream main response content
- const responseContent =
- "Here's my response after using the [THINK] format:\n\n**Observations:**\n- Both <think> and [THINK] formats work seamlessly\n- The parsing logic handles both cases\n- UI display is consistent across formats\n\nThis demonstrates the enhanced thinking content support.";
-
- for (let i = 0; i < responseContent.length; i++) {
- currentContent += responseContent[i];
- streamingBracketMessage.content = currentContent;
- await new Promise((resolve) => setTimeout(resolve, 10));
- }
-
- streamingBracketMessage.timestamp = Date.now();
- }}
->
-
-
-
-