mirror of
https://github.com/LostRuins/koboldcpp.git
synced 2026-09-18 00:35:04 +02:00
sd: merge upstream int8 convrot and fp8 scaled at master-841-6b3edaa (#2452)
Gated behind disabled macros KCPP_MAINLINE_INT8_CONVROT and KCPP_MAINLINE_FP8_SCALED .
This commit is contained in:
@@ -136,10 +136,13 @@ enum sd_type_t {
|
||||
// SD_TYPE_IQ4_NL_4_4 = 36,
|
||||
// SD_TYPE_IQ4_NL_4_8 = 37,
|
||||
// SD_TYPE_IQ4_NL_8_8 = 38,
|
||||
SD_TYPE_MXFP4 = 39, // MXFP4 (1 block)
|
||||
SD_TYPE_NVFP4 = 40, // NVFP4 (4 blocks, E4M3 scale)
|
||||
SD_TYPE_Q1_0 = 41,
|
||||
SD_TYPE_COUNT = 42,
|
||||
SD_TYPE_MXFP4 = 39, // MXFP4 (1 block)
|
||||
SD_TYPE_NVFP4 = 40, // NVFP4 (4 blocks, E4M3 scale)
|
||||
SD_TYPE_Q1_0 = 41,
|
||||
SD_TYPE_Q2_0 = 42,
|
||||
SD_TYPE_F8_E4M3 = 43,
|
||||
SD_TYPE_F8_E5M2 = 44,
|
||||
SD_TYPE_COUNT = 45,
|
||||
};
|
||||
|
||||
enum sd_log_level_t {
|
||||
|
||||
@@ -31,6 +31,16 @@
|
||||
#include "ggml-backend.h"
|
||||
#include "ggml.h"
|
||||
|
||||
// kcpp sidestep int8 convrot support
|
||||
#ifndef KCPP_MAINLINE_INT8_CONVROT
|
||||
#define KCPP_MAINLINE_INT8_CONVROT 0
|
||||
#endif
|
||||
|
||||
// kcpp sidestep fp8 scaled support
|
||||
#ifndef KCPP_MAINLINE_FP8_SCALED
|
||||
#define KCPP_MAINLINE_FP8_SCALED 0
|
||||
#endif
|
||||
|
||||
#include "core/tensor.hpp"
|
||||
#include "model.h"
|
||||
|
||||
@@ -1039,6 +1049,40 @@ __STATIC_INLINE__ ggml_tensor* ggml_ext_linear(ggml_context* ctx,
|
||||
return x;
|
||||
}
|
||||
|
||||
#if KCPP_MAINLINE_INT8_CONVROT
|
||||
__STATIC_INLINE__ ggml_tensor* ggml_ext_linear_i8_tensorwise(ggml_context* ctx,
|
||||
ggml_tensor* x,
|
||||
ggml_tensor* w,
|
||||
ggml_tensor* weight_scale,
|
||||
ggml_tensor* b,
|
||||
int convrot_group_size,
|
||||
float scale = 1.f) {
|
||||
GGML_ASSERT(x->type == GGML_TYPE_F32 || (x->type == GGML_TYPE_I8 && scale == 1.f));
|
||||
if (scale != 1.f) {
|
||||
x = ggml_ext_scale(ctx, x, scale);
|
||||
}
|
||||
|
||||
ggml_tensor* fused_bias = scale == 1.f ? b : nullptr;
|
||||
if (x->ne[2] * x->ne[3] > 1024) {
|
||||
int64_t ne2 = x->ne[2];
|
||||
int64_t ne3 = x->ne[3];
|
||||
x = ggml_reshape_2d(ctx, x, x->ne[0], x->ne[1] * x->ne[2] * x->ne[3]);
|
||||
x = ggml_mul_mat_i8_tensorwise(ctx, w, x, weight_scale, fused_bias, convrot_group_size);
|
||||
x = ggml_reshape_4d(ctx, x, x->ne[0], x->ne[1] / ne2 / ne3, ne2, ne3);
|
||||
} else {
|
||||
x = ggml_mul_mat_i8_tensorwise(ctx, w, x, weight_scale, fused_bias, convrot_group_size);
|
||||
}
|
||||
|
||||
if (scale != 1.f) {
|
||||
x = ggml_ext_scale(ctx, x, 1.f / scale);
|
||||
if (b != nullptr) {
|
||||
x = ggml_add_inplace(ctx, x, b);
|
||||
}
|
||||
}
|
||||
return x;
|
||||
}
|
||||
#endif
|
||||
|
||||
__STATIC_INLINE__ ggml_tensor* ggml_ext_pad_ext(ggml_context* ctx,
|
||||
ggml_backend_t backend,
|
||||
ggml_tensor* x,
|
||||
@@ -1679,6 +1723,13 @@ struct WeightAdapter {
|
||||
ggml_tensor* b,
|
||||
const std::string& prefix,
|
||||
ForwardParams forward_params) = 0;
|
||||
virtual ggml_tensor* add_lora_to_output(ggml_context* ctx,
|
||||
ggml_backend_t backend,
|
||||
ggml_tensor* x,
|
||||
ggml_tensor* w,
|
||||
ggml_tensor* output,
|
||||
const std::string& prefix,
|
||||
ForwardParams forward_params) = 0;
|
||||
virtual size_t get_extra_graph_size() = 0;
|
||||
};
|
||||
|
||||
@@ -1696,6 +1747,7 @@ struct GGMLRunnerContext {
|
||||
std::function<ggml_tensor*(const std::string&)> get_cache_tensor;
|
||||
std::function<void(const std::string&, ggml_tensor*)> cache_tensor;
|
||||
std::function<void(ggml_tensor*, const void*)> set_backend_tensor_data;
|
||||
std::map<std::pair<ggml_tensor*, int>, ggml_tensor*> int8_convrot_cache;
|
||||
|
||||
void capture_tensor(const std::string& name, ggml_tensor* tensor) {
|
||||
if (debug_tensors == nullptr || tensor == nullptr) {
|
||||
@@ -1754,7 +1806,8 @@ protected:
|
||||
|
||||
std::vector<ggml_backend_t> extra_runtime_backends; // borrowed (SDBackendManager-owned)
|
||||
ggml_backend_sched_t sched = nullptr; // owned
|
||||
ggml_backend_t cpu_fallback_backend = nullptr; // owned, sched requires a trailing CPU backend
|
||||
size_t sched_graph_capacity = 0;
|
||||
ggml_backend_t cpu_fallback_backend = nullptr; // owned, sched requires a trailing CPU backend
|
||||
bool multi_device_eval_callback_warned = false;
|
||||
|
||||
std::shared_ptr<WeightAdapter> weight_adapter = nullptr;
|
||||
@@ -2040,9 +2093,19 @@ protected:
|
||||
// Pass explicit buffer types: synthesized defaults can make CUDA devices
|
||||
// report supporting each other's buffers and skip a required copy.
|
||||
bool ensure_sched(ggml_cgraph* gf) {
|
||||
if (sched != nullptr) {
|
||||
const size_t required_graph_size = gf != nullptr
|
||||
? std::max<size_t>(1,
|
||||
(size_t)ggml_graph_n_nodes(gf) +
|
||||
sd::ggml_graph_cut::leaf_count(gf))
|
||||
: 1;
|
||||
if (sched != nullptr && sched_graph_capacity >= required_graph_size) {
|
||||
return true;
|
||||
}
|
||||
if (sched != nullptr) {
|
||||
ggml_backend_sched_free(sched);
|
||||
sched = nullptr;
|
||||
sched_graph_capacity = 0;
|
||||
}
|
||||
std::vector<ggml_backend_t> backends;
|
||||
backends.reserve(extra_runtime_backends.size() + 2);
|
||||
backends.push_back(runtime_backend);
|
||||
@@ -2070,20 +2133,17 @@ protected:
|
||||
bufts.push_back(buft);
|
||||
}
|
||||
|
||||
size_t graph_size = MAX_GRAPH_SIZE;
|
||||
if (gf != nullptr) {
|
||||
graph_size = std::max<size_t>(graph_size, (size_t)ggml_graph_n_nodes(gf));
|
||||
}
|
||||
sched = ggml_backend_sched_new(backends.data(),
|
||||
bufts.data(),
|
||||
(int)backends.size(),
|
||||
graph_size,
|
||||
required_graph_size,
|
||||
/*parallel=*/false,
|
||||
/*op_offload=*/false);
|
||||
if (sched == nullptr) {
|
||||
LOG_ERROR("%s: failed to create backend sched", get_desc().c_str());
|
||||
return false;
|
||||
}
|
||||
sched_graph_capacity = required_graph_size;
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -3033,7 +3093,8 @@ public:
|
||||
}
|
||||
if (sched != nullptr) {
|
||||
ggml_backend_sched_free(sched);
|
||||
sched = nullptr;
|
||||
sched = nullptr;
|
||||
sched_graph_capacity = 0;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3358,15 +3419,18 @@ protected:
|
||||
bool bias;
|
||||
bool force_f32;
|
||||
bool force_prec_f32;
|
||||
bool allow_weight_scale;
|
||||
bool has_weight_scale = false;
|
||||
bool has_weight_scale = false;
|
||||
bool int8_convrot = false;
|
||||
int int8_convrot_group_size = 0;
|
||||
float scale;
|
||||
std::string prefix;
|
||||
|
||||
void init_params(ggml_context* ctx, const String2TensorStorage& tensor_storage_map = {}, const std::string prefix = "") override {
|
||||
this->prefix = prefix;
|
||||
has_weight_scale = false;
|
||||
enum ggml_type wtype = get_type(prefix + "weight", tensor_storage_map, GGML_TYPE_F32);
|
||||
this->prefix = prefix;
|
||||
has_weight_scale = false;
|
||||
int8_convrot = false;
|
||||
int8_convrot_group_size = 0;
|
||||
enum ggml_type wtype = get_type(prefix + "weight", tensor_storage_map, GGML_TYPE_F32);
|
||||
if (in_features % ggml_blck_size(wtype) != 0 || force_f32) {
|
||||
wtype = GGML_TYPE_F32;
|
||||
}
|
||||
@@ -3375,26 +3439,35 @@ protected:
|
||||
enum ggml_type wtype = GGML_TYPE_F32;
|
||||
params["bias"] = ggml_new_tensor_1d(ctx, wtype, out_features);
|
||||
}
|
||||
if (allow_weight_scale && tensor_storage_map.find(prefix + "weight_scale") != tensor_storage_map.end()) {
|
||||
params["weight_scale"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, out_features);
|
||||
auto weight_storage = tensor_storage_map.find(prefix + "weight");
|
||||
const bool is_int8_tensorwise = weight_storage != tensor_storage_map.end() && weight_storage->second.is_int8_tensorwise;
|
||||
auto weight_scale_storage = tensor_storage_map.find(prefix + "weight_scale");
|
||||
if (weight_scale_storage != tensor_storage_map.end()) {
|
||||
const int64_t scale_nelements = weight_scale_storage->second.nelements();
|
||||
GGML_ASSERT(scale_nelements == 1 || scale_nelements == out_features);
|
||||
params["weight_scale"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, scale_nelements);
|
||||
has_weight_scale = true;
|
||||
}
|
||||
if (is_int8_tensorwise) {
|
||||
GGML_ASSERT(wtype == GGML_TYPE_I8);
|
||||
GGML_ASSERT(has_weight_scale);
|
||||
int8_convrot = weight_storage->second.int8_convrot;
|
||||
int8_convrot_group_size = weight_storage->second.int8_convrot_group_size;
|
||||
}
|
||||
}
|
||||
|
||||
public:
|
||||
Linear(int64_t in_features,
|
||||
int64_t out_features,
|
||||
bool bias = true,
|
||||
bool force_f32 = false,
|
||||
bool force_prec_f32 = false,
|
||||
float scale = 1.f,
|
||||
bool allow_weight_scale = false)
|
||||
bool bias = true,
|
||||
bool force_f32 = false,
|
||||
bool force_prec_f32 = false,
|
||||
float scale = 1.f)
|
||||
: in_features(in_features),
|
||||
out_features(out_features),
|
||||
bias(bias),
|
||||
force_f32(force_f32),
|
||||
force_prec_f32(force_prec_f32),
|
||||
allow_weight_scale(allow_weight_scale),
|
||||
scale(scale) {}
|
||||
|
||||
void set_scale(float scale_) {
|
||||
@@ -3406,13 +3479,98 @@ public:
|
||||
}
|
||||
|
||||
ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) override {
|
||||
ggml_tensor* w = params["weight"];
|
||||
ggml_tensor* w = params["weight"];
|
||||
ggml_tensor* weight_scale = has_weight_scale ? params["weight_scale"] : nullptr;
|
||||
#if KCPP_MAINLINE_FP8_SCALED
|
||||
if (w->type == GGML_TYPE_F8_E4M3 || w->type == GGML_TYPE_F8_E5M2) {
|
||||
bool supports_fp8_matmul = false;
|
||||
if (ctx->backend != nullptr) {
|
||||
ggml_tensor* fp8_matmul = ggml_mul_mat(ctx->ggml_ctx, w, x);
|
||||
if (force_prec_f32) {
|
||||
ggml_mul_mat_set_prec(fp8_matmul, GGML_PREC_F32);
|
||||
}
|
||||
supports_fp8_matmul = ggml_backend_supports_op(ctx->backend, fp8_matmul);
|
||||
}
|
||||
if (!supports_fp8_matmul) {
|
||||
w = ggml_cast(ctx->ggml_ctx, w, GGML_TYPE_BF16);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
ggml_tensor* b = nullptr;
|
||||
if (bias) {
|
||||
b = params["bias"];
|
||||
}
|
||||
ggml_tensor* linear_bias = has_weight_scale ? nullptr : b;
|
||||
ggml_tensor* out = nullptr;
|
||||
#if KCPP_MAINLINE_INT8_CONVROT
|
||||
if (w->type == GGML_TYPE_I8) {
|
||||
if (x->type != GGML_TYPE_F32) {
|
||||
x = ggml_ext_cast_f32(ctx->ggml_ctx, ctx->backend, x);
|
||||
}
|
||||
if (!ggml_is_contiguous(x)) {
|
||||
x = ggml_cont(ctx->ggml_ctx, x);
|
||||
}
|
||||
ggml_tensor* lora_input = x;
|
||||
if (ctx->weight_adapter && b != nullptr) {
|
||||
b = ctx->weight_adapter->patch_weight(ctx->ggml_ctx, ctx->backend, b, prefix + "bias");
|
||||
}
|
||||
if (int8_convrot && scale == 1.f) {
|
||||
const auto cache_key = std::make_pair(x, int8_convrot_group_size);
|
||||
auto cached = ctx->int8_convrot_cache.find(cache_key);
|
||||
if (cached == ctx->int8_convrot_cache.end()) {
|
||||
x = ggml_quantize_i8_convrot(ctx->ggml_ctx, x, int8_convrot_group_size);
|
||||
ctx->int8_convrot_cache.emplace(cache_key, x);
|
||||
} else {
|
||||
x = cached->second;
|
||||
}
|
||||
}
|
||||
out = ggml_ext_linear_i8_tensorwise(ctx->ggml_ctx,
|
||||
x,
|
||||
w,
|
||||
weight_scale,
|
||||
b,
|
||||
int8_convrot ? int8_convrot_group_size : 0,
|
||||
scale);
|
||||
if (ctx->weight_adapter) {
|
||||
WeightAdapter::ForwardParams forward_params;
|
||||
forward_params.op_type = WeightAdapter::ForwardParams::op_type_t::OP_LINEAR;
|
||||
forward_params.linear.force_prec_f32 = force_prec_f32;
|
||||
forward_params.linear.scale = scale;
|
||||
out = ctx->weight_adapter->add_lora_to_output(ctx->ggml_ctx,
|
||||
ctx->backend,
|
||||
lora_input,
|
||||
w,
|
||||
out,
|
||||
prefix,
|
||||
forward_params);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
#endif // kcpp
|
||||
if (has_weight_scale) {
|
||||
out = ggml_ext_linear(ctx->ggml_ctx, x, w, nullptr, force_prec_f32, scale);
|
||||
out = ggml_mul(ctx->ggml_ctx, out, weight_scale);
|
||||
if (ctx->weight_adapter) {
|
||||
WeightAdapter::ForwardParams forward_params;
|
||||
forward_params.op_type = WeightAdapter::ForwardParams::op_type_t::OP_LINEAR;
|
||||
forward_params.linear.force_prec_f32 = force_prec_f32;
|
||||
forward_params.linear.scale = scale;
|
||||
out = ctx->weight_adapter->add_lora_to_output(ctx->ggml_ctx,
|
||||
ctx->backend,
|
||||
x,
|
||||
w,
|
||||
out,
|
||||
prefix,
|
||||
forward_params);
|
||||
if (b != nullptr) {
|
||||
b = ctx->weight_adapter->patch_weight(ctx->ggml_ctx, ctx->backend, b, prefix + "bias");
|
||||
}
|
||||
}
|
||||
if (b != nullptr) {
|
||||
out = ggml_add_inplace(ctx->ggml_ctx, out, b);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
if (ctx->weight_adapter) {
|
||||
WeightAdapter::ForwardParams forward_params;
|
||||
forward_params.op_type = WeightAdapter::ForwardParams::op_type_t::OP_LINEAR;
|
||||
@@ -3422,12 +3580,6 @@ public:
|
||||
} else {
|
||||
out = ggml_ext_linear(ctx->ggml_ctx, x, w, linear_bias, force_prec_f32, scale);
|
||||
}
|
||||
if (has_weight_scale) {
|
||||
out = ggml_mul(ctx->ggml_ctx, out, params["weight_scale"]);
|
||||
if (b != nullptr) {
|
||||
out = ggml_add_inplace(ctx->ggml_ctx, out, b);
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
};
|
||||
|
||||
@@ -1072,6 +1072,34 @@ public:
|
||||
return out;
|
||||
}
|
||||
|
||||
ggml_tensor* add_lora_to_output(ggml_context* ctx,
|
||||
ggml_backend_t backend,
|
||||
ggml_tensor* x,
|
||||
ggml_tensor* w,
|
||||
ggml_tensor* output,
|
||||
const std::string& prefix,
|
||||
WeightAdapter::ForwardParams forward_params) override {
|
||||
for (auto& lora_model : lora_models) {
|
||||
ggml_tensor* weight_diff = lora_model->get_weight_diff(prefix + "weight", backend, ctx, w, false);
|
||||
if (weight_diff != nullptr) {
|
||||
GGML_ASSERT(forward_params.op_type == ForwardParams::op_type_t::OP_LINEAR);
|
||||
ggml_tensor* out_diff = ggml_ext_linear(ctx,
|
||||
x,
|
||||
weight_diff,
|
||||
nullptr,
|
||||
forward_params.linear.force_prec_f32,
|
||||
forward_params.linear.scale);
|
||||
output = ggml_add_inplace(ctx, output, out_diff);
|
||||
}
|
||||
|
||||
ggml_tensor* out_diff = lora_model->get_out_diff(ctx, backend, x, w, forward_params, prefix + "weight");
|
||||
if (out_diff != nullptr) {
|
||||
output = ggml_add_inplace(ctx, output, out_diff);
|
||||
}
|
||||
}
|
||||
return output;
|
||||
}
|
||||
|
||||
size_t get_extra_graph_size() override {
|
||||
size_t lora_tensor_num = 0;
|
||||
for (auto& lora_model : lora_models) {
|
||||
|
||||
@@ -142,7 +142,7 @@ namespace Ideogram4 {
|
||||
__STATIC_INLINE__ std::shared_ptr<Linear> make_linear(int64_t in_features,
|
||||
int64_t out_features,
|
||||
bool bias = true) {
|
||||
return std::make_shared<Linear>(in_features, out_features, bias, false, false, 1.f, true);
|
||||
return std::make_shared<Linear>(in_features, out_features, bias);
|
||||
}
|
||||
|
||||
__STATIC_INLINE__ std::vector<float> gen_ideogram4_pe(int grid_h,
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
#include <fstream>
|
||||
#include <ostream>
|
||||
#include <string>
|
||||
#include <unordered_map>
|
||||
#include <unordered_set>
|
||||
#include <vector>
|
||||
|
||||
@@ -86,20 +87,73 @@ static ggml_type safetensors_dtype_to_ggml_type(const std::string& dtype) {
|
||||
ttype = GGML_TYPE_F32;
|
||||
} else if (dtype == "F64") {
|
||||
ttype = GGML_TYPE_F32;
|
||||
#if KCPP_MAINLINE_FP8_SCALED
|
||||
} else if (dtype == "F8_E4M3") {
|
||||
ttype = GGML_TYPE_F8_E4M3;
|
||||
} else if (dtype == "F8_E5M2") {
|
||||
ttype = GGML_TYPE_F8_E5M2;
|
||||
#else
|
||||
} else if (dtype == "F8_E4M3") {
|
||||
ttype = GGML_TYPE_F16;
|
||||
} else if (dtype == "F8_E5M2") {
|
||||
ttype = GGML_TYPE_F16;
|
||||
#endif
|
||||
#if !KCPP_MAINLINE_INT8_CONVROT
|
||||
} else if (dtype == "I8") {
|
||||
ttype = GGML_TYPE_F16;
|
||||
#endif
|
||||
} else if (dtype == "I32") {
|
||||
ttype = GGML_TYPE_I32;
|
||||
} else if (dtype == "I64") {
|
||||
ttype = GGML_TYPE_I32;
|
||||
#if KCPP_MAINLINE_INT8_CONVROT
|
||||
} else if (dtype == "I8") {
|
||||
ttype = GGML_TYPE_I8;
|
||||
#endif
|
||||
}
|
||||
return ttype;
|
||||
}
|
||||
|
||||
struct ComfyQuantConfig {
|
||||
std::string format;
|
||||
bool convrot = false;
|
||||
int group_size = 0;
|
||||
};
|
||||
|
||||
static bool read_comfy_quant_config(std::ifstream& file,
|
||||
const std::string& file_path,
|
||||
const std::string& tensor_name,
|
||||
size_t offset,
|
||||
size_t size,
|
||||
ComfyQuantConfig& config,
|
||||
std::string* error) {
|
||||
static constexpr size_t MAX_COMFY_QUANT_CONFIG_SIZE = 64 * 1024;
|
||||
if (size == 0 || size > MAX_COMFY_QUANT_CONFIG_SIZE) {
|
||||
set_error(error, "invalid ComfyUI quantization metadata tensor '" + tensor_name + "' in '" + file_path + "'");
|
||||
return false;
|
||||
}
|
||||
|
||||
std::vector<char> data(size + 1, '\0');
|
||||
file.clear();
|
||||
file.seekg((std::streamoff)offset, std::ios::beg);
|
||||
file.read(data.data(), (std::streamsize)size);
|
||||
if (!file) {
|
||||
set_error(error, "read ComfyUI quantization metadata tensor failed: '" + tensor_name + "'");
|
||||
return false;
|
||||
}
|
||||
|
||||
try {
|
||||
const nlohmann::json json = nlohmann::json::parse(data.data(), data.data() + size);
|
||||
config.format = json.value("format", "");
|
||||
config.convrot = json.value("convrot", false);
|
||||
config.group_size = json.value("convrot_groupsize", 0);
|
||||
} catch (const std::exception&) {
|
||||
set_error(error, "parsing ComfyUI quantization metadata tensor failed: '" + tensor_name + "'");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
// https://huggingface.co/docs/safetensors/index
|
||||
bool read_safetensors_file(const std::string& file_path,
|
||||
std::vector<TensorStorage>& tensor_storages,
|
||||
@@ -166,8 +220,37 @@ bool read_safetensors_file(const std::string& file_path,
|
||||
}
|
||||
}
|
||||
|
||||
std::unordered_map<std::string, ComfyQuantConfig> comfy_quant_configs;
|
||||
for (const auto& item : header_.items()) {
|
||||
const std::string& name = item.key();
|
||||
if (name == "__metadata__" || !ends_with(name, ".comfy_quant")) {
|
||||
continue;
|
||||
}
|
||||
|
||||
const nlohmann::json& tensor_info = item.value();
|
||||
if (tensor_info.value("dtype", "") != "U8") {
|
||||
set_error(error, "invalid dtype for ComfyUI quantization metadata tensor '" + name + "'");
|
||||
return false;
|
||||
}
|
||||
const size_t begin = tensor_info["data_offsets"][0].get<size_t>();
|
||||
const size_t end = tensor_info["data_offsets"][1].get<size_t>();
|
||||
if (begin > end || end > file_size_ - data_start) {
|
||||
set_error(error, "data offsets out of bounds for tensor '" + name + "'");
|
||||
return false;
|
||||
}
|
||||
|
||||
ComfyQuantConfig config;
|
||||
if (!read_comfy_quant_config(file, file_path, name, data_start + begin, end - begin, config, error)) {
|
||||
return false;
|
||||
}
|
||||
const std::string module_name = name.substr(0, name.size() - std::string(".comfy_quant").size());
|
||||
comfy_quant_configs.emplace(module_name, std::move(config));
|
||||
}
|
||||
|
||||
tensor_storages.clear();
|
||||
#if !KCPP_MAINLINE_INT8_CONVROT
|
||||
auto quant_layers = kcpp_safetensors_quant::read_quantization_metadata(header_);
|
||||
#endif
|
||||
for (auto& item : header_.items()) {
|
||||
std::string name = item.key();
|
||||
nlohmann::json tensor_info = item.value();
|
||||
@@ -177,9 +260,11 @@ bool read_safetensors_file(const std::string& file_path,
|
||||
continue;
|
||||
}
|
||||
|
||||
#if !KCPP_MAINLINE_INT8_CONVROT
|
||||
if (kcpp_safetensors_quant::should_skip_side_tensor(header_, name, quant_layers)) {
|
||||
continue;
|
||||
}
|
||||
#endif
|
||||
|
||||
std::string dtype = tensor_info["dtype"];
|
||||
nlohmann::json shape = tensor_info["shape"];
|
||||
@@ -228,10 +313,48 @@ bool read_safetensors_file(const std::string& file_path,
|
||||
TensorStorage tensor_storage(name, type, ne, n_dims, 0, data_start + begin);
|
||||
tensor_storage.reverse_ne();
|
||||
|
||||
#if KCPP_MAINLINE_INT8_CONVROT
|
||||
if (ends_with(name, ".weight")) {
|
||||
const std::string module_name = name.substr(0, name.size() - std::string(".weight").size());
|
||||
auto config = comfy_quant_configs.find(module_name);
|
||||
if (config != comfy_quant_configs.end() && config->second.format == "int8_tensorwise") {
|
||||
if (type != GGML_TYPE_I8) {
|
||||
set_error(error, "ComfyUI int8_tensorwise weight is not I8: '" + name + "'");
|
||||
return false;
|
||||
}
|
||||
if (config->second.convrot) {
|
||||
int group_size_remainder = config->second.group_size;
|
||||
while (group_size_remainder > 1 && group_size_remainder % 4 == 0) {
|
||||
group_size_remainder /= 4;
|
||||
}
|
||||
if (group_size_remainder != 1 || tensor_storage.ne[0] % config->second.group_size != 0) {
|
||||
set_error(error, "invalid ComfyUI convrot group size for tensor '" + name + "'");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
tensor_storage.is_int8_tensorwise = true;
|
||||
tensor_storage.int8_convrot = config->second.convrot;
|
||||
tensor_storage.int8_convrot_group_size = config->second.group_size;
|
||||
}
|
||||
} else if (ends_with(name, ".weight_scale")) {
|
||||
const std::string module_name = name.substr(0, name.size() - std::string(".weight_scale").size());
|
||||
auto config = comfy_quant_configs.find(module_name);
|
||||
if (config != comfy_quant_configs.end() && config->second.format == "int8_tensorwise" &&
|
||||
tensor_storage.n_dims == 2 && tensor_storage.ne[0] == 1) {
|
||||
tensor_storage.ne[0] = tensor_storage.ne[1];
|
||||
tensor_storage.ne[1] = 1;
|
||||
tensor_storage.n_dims = 1;
|
||||
}
|
||||
}
|
||||
#endif // kcpp
|
||||
|
||||
size_t tensor_data_size = end - begin;
|
||||
|
||||
bool tensor_size_ok;
|
||||
if (dtype == "I8") {
|
||||
#if KCPP_MAINLINE_INT8_CONVROT
|
||||
tensor_size_ok = (tensor_storage.nbytes() == tensor_data_size);
|
||||
#else
|
||||
if (!kcpp_safetensors_quant::fill_i8_tensorwise_storage(header_,
|
||||
file,
|
||||
quant_layers,
|
||||
@@ -244,6 +367,15 @@ bool read_safetensors_file(const std::string& file_path,
|
||||
return false;
|
||||
}
|
||||
tensor_size_ok = true;
|
||||
#endif
|
||||
#if KCPP_MAINLINE_FP8_SCALED
|
||||
} else if (dtype == "F8_E4M3") {
|
||||
tensor_storage.is_f8_e4m3 = true;
|
||||
tensor_size_ok = (tensor_storage.nbytes() == tensor_data_size);
|
||||
} else if (dtype == "F8_E5M2") {
|
||||
tensor_storage.is_f8_e5m2 = true;
|
||||
tensor_size_ok = (tensor_storage.nbytes() == tensor_data_size);
|
||||
#else
|
||||
} else if (dtype == "F8_E4M3") {
|
||||
tensor_storage.is_f8_e4m3 = true;
|
||||
// f8 -> f16
|
||||
@@ -252,6 +384,7 @@ bool read_safetensors_file(const std::string& file_path,
|
||||
tensor_storage.is_f8_e5m2 = true;
|
||||
// f8 -> f16
|
||||
tensor_size_ok = (tensor_storage.nbytes() == tensor_data_size * 2);
|
||||
#endif
|
||||
} else if (dtype == "F64") {
|
||||
tensor_storage.is_f64 = true;
|
||||
// f64 -> f32
|
||||
|
||||
@@ -20,14 +20,17 @@ struct TensorStorageExt;
|
||||
|
||||
struct TensorStorage {
|
||||
std::string name;
|
||||
ggml_type type = GGML_TYPE_F32;
|
||||
ggml_type expected_type = GGML_TYPE_COUNT;
|
||||
bool is_f8_e4m3 = false;
|
||||
bool is_f8_e5m2 = false;
|
||||
bool is_f64 = false;
|
||||
bool is_i64 = false;
|
||||
int64_t ne[SD_MAX_DIMS] = {1, 1, 1, 1, 1};
|
||||
int n_dims = 0;
|
||||
ggml_type type = GGML_TYPE_F32;
|
||||
ggml_type expected_type = GGML_TYPE_COUNT;
|
||||
bool is_f8_e4m3 = false;
|
||||
bool is_f8_e5m2 = false;
|
||||
bool is_f64 = false;
|
||||
bool is_i64 = false;
|
||||
bool is_int8_tensorwise = false;
|
||||
bool int8_convrot = false;
|
||||
int int8_convrot_group_size = 0;
|
||||
int64_t ne[SD_MAX_DIMS] = {1, 1, 1, 1, 1};
|
||||
int n_dims = 0;
|
||||
std::shared_ptr<kcpp_safetensors_quant::TensorStorageExt> kcpp_ext;
|
||||
|
||||
std::string storage_key;
|
||||
@@ -57,12 +60,14 @@ struct TensorStorage {
|
||||
}
|
||||
|
||||
int64_t nbytes_to_read() const {
|
||||
if (is_f8_e4m3 || is_f8_e5m2) {
|
||||
if (is_f64 || is_i64) {
|
||||
return nbytes() * 2;
|
||||
#if !KCPP_MAINLINE_FP8_SCALED
|
||||
} else if (is_f8_e4m3 || is_f8_e5m2) {
|
||||
return nbytes() / 2;
|
||||
#endif
|
||||
} else if (kcpp_ext) {
|
||||
return nelements();
|
||||
} else if (is_f64 || is_i64) {
|
||||
return nbytes() * 2;
|
||||
} else {
|
||||
return nbytes();
|
||||
}
|
||||
|
||||
@@ -79,20 +79,7 @@ bool is_unused_tensor(const std::string& name) {
|
||||
return false;
|
||||
}
|
||||
|
||||
std::string kcpp_fix_wrong_img_tensor_name(const std::string& name) //kcpp function that fixes common wrong tensor names
|
||||
{
|
||||
if (starts_with(name, "text_encoders.qwen25_7b.transformer.model.")) {
|
||||
return "text_encoders.llm.model." + name.substr(strlen("text_encoders.qwen25_7b.transformer.model."));
|
||||
}
|
||||
if (starts_with(name, "text_encoders.qwen25_7b.transformer.visual.")) {
|
||||
return "text_encoders.llm.visual." + name.substr(strlen("text_encoders.qwen25_7b.transformer.visual."));
|
||||
}
|
||||
if (starts_with(name, "text_encoders.umt5xxl.")) {
|
||||
return "text_encoders.t5xxl." + name.substr(strlen("text_encoders.umt5xxl."));
|
||||
}
|
||||
return name;
|
||||
}
|
||||
|
||||
#if !KCPP_MAINLINE_FP8_SCALED
|
||||
uint16_t f8_e4m3_to_f16(uint8_t f8) {
|
||||
// do we need to support uz?
|
||||
|
||||
@@ -152,6 +139,21 @@ void f8_e5m2_to_f16_vec(uint8_t* src, uint16_t* dst, int64_t n) {
|
||||
dst[i] = f8_e5m2_to_f16(src[i]);
|
||||
}
|
||||
}
|
||||
#endif
|
||||
|
||||
std::string kcpp_fix_wrong_img_tensor_name(const std::string& name) //kcpp function that fixes common wrong tensor names
|
||||
{
|
||||
if (starts_with(name, "text_encoders.qwen25_7b.transformer.model.")) {
|
||||
return "text_encoders.llm.model." + name.substr(strlen("text_encoders.qwen25_7b.transformer.model."));
|
||||
}
|
||||
if (starts_with(name, "text_encoders.qwen25_7b.transformer.visual.")) {
|
||||
return "text_encoders.llm.visual." + name.substr(strlen("text_encoders.qwen25_7b.transformer.visual."));
|
||||
}
|
||||
if (starts_with(name, "text_encoders.umt5xxl.")) {
|
||||
return "text_encoders.t5xxl." + name.substr(strlen("text_encoders.umt5xxl."));
|
||||
}
|
||||
return name;
|
||||
}
|
||||
|
||||
void f64_to_f32_vec(double* src, float* dst, int64_t n) {
|
||||
// support inplace op
|
||||
@@ -958,10 +960,12 @@ std::vector<MmapTensorStore> ModelLoader::mmap_tensors(std::map<std::string, ggm
|
||||
if (dst_tensor == nullptr)
|
||||
continue;
|
||||
|
||||
if (tensor_storage.is_f8_e4m3 ||
|
||||
tensor_storage.is_f8_e5m2 ||
|
||||
tensor_storage.is_f64 ||
|
||||
if (tensor_storage.is_f64 ||
|
||||
tensor_storage.is_i64 ||
|
||||
#if !KCPP_MAINLINE_FP8_SCALED
|
||||
tensor_storage.is_f8_e4m3 ||
|
||||
tensor_storage.is_f8_e5m2 ||
|
||||
#endif
|
||||
tensor_storage.kcpp_ext ||
|
||||
tensor_storage.type != dst_tensor->type) {
|
||||
continue;
|
||||
@@ -1272,11 +1276,15 @@ bool ModelLoader::load_tensors(on_new_tensor_cb_t on_new_tensor_cb,
|
||||
return;
|
||||
}
|
||||
bytes_processed.fetch_add(scale_nbytes);
|
||||
} else if (tensor_storage.is_f8_e4m3) {
|
||||
} else
|
||||
#if !KCPP_MAINLINE_FP8_SCALED
|
||||
if (tensor_storage.is_f8_e4m3) {
|
||||
f8_e4m3_to_f16_vec((uint8_t*)read_buf, (uint16_t*)target_buf, tensor_storage.nelements());
|
||||
} else if (tensor_storage.is_f8_e5m2) {
|
||||
f8_e5m2_to_f16_vec((uint8_t*)read_buf, (uint16_t*)target_buf, tensor_storage.nelements());
|
||||
} else if (tensor_storage.is_f64) {
|
||||
} else
|
||||
#endif
|
||||
if (tensor_storage.is_f64) {
|
||||
f64_to_f32_vec((double*)read_buf, (float*)target_buf, tensor_storage.nelements());
|
||||
} else if (tensor_storage.is_i64) {
|
||||
i64_to_i32_vec((int64_t*)read_buf, (int32_t*)target_buf, tensor_storage.nelements());
|
||||
@@ -1565,6 +1573,9 @@ bool ModelLoader::load_tensors(std::map<std::string, ggml_tensor*>& tensors,
|
||||
|
||||
bool ModelLoader::tensor_should_be_converted(const TensorStorage& tensor_storage, ggml_type type) {
|
||||
const std::string& name = tensor_storage.name;
|
||||
if (tensor_storage.is_int8_tensorwise) {
|
||||
return false;
|
||||
}
|
||||
if (type != GGML_TYPE_COUNT) {
|
||||
if (ggml_is_quantized(type) && tensor_storage.ne[0] % ggml_blck_size(type) != 0) {
|
||||
// Pass, do not convert
|
||||
|
||||
@@ -1571,6 +1571,11 @@ std::string convert_tensor_name(std::string name, SDVersion version) {
|
||||
}
|
||||
}
|
||||
|
||||
static const std::vector<std::pair<std::string, std::string>> generic_name_map = {
|
||||
{".scale_weight", ".weight_scale"},
|
||||
};
|
||||
replace_with_name_map(name, generic_name_map);
|
||||
|
||||
if (is_lora) {
|
||||
name = "lora." + name;
|
||||
}
|
||||
|
||||
@@ -1218,8 +1218,16 @@ public:
|
||||
|
||||
LOG_DEBUG("ggml tensor size = %d bytes", (int)sizeof(ggml_tensor));
|
||||
|
||||
bool have_int8_tensorwise = false;
|
||||
for (const auto& [_, tensor_storage] : model_loader.get_tensor_storage_map()) {
|
||||
if (tensor_storage.is_int8_tensorwise) {
|
||||
have_int8_tensorwise = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (sd_ctx_params->lora_apply_mode == LORA_APPLY_AUTO) {
|
||||
bool have_quantized_weight = false;
|
||||
bool have_quantized_weight = have_int8_tensorwise;
|
||||
for (const auto& [type, _] : wtype_stat) {
|
||||
if (ggml_is_quantized(type)) {
|
||||
have_quantized_weight = true;
|
||||
@@ -1235,12 +1243,19 @@ public:
|
||||
apply_lora_immediately = true;
|
||||
}
|
||||
} else if (sd_ctx_params->lora_apply_mode == LORA_APPLY_IMMEDIATELY) {
|
||||
if (row_split_active()) {
|
||||
if (have_int8_tensorwise) {
|
||||
LOG_WARN(
|
||||
"INT8 tensorwise weights do not support the immediately LoRA apply mode; "
|
||||
"using at_runtime instead");
|
||||
apply_lora_immediately = false;
|
||||
} else if (row_split_active()) {
|
||||
LOG_WARN(
|
||||
"row-split tensors do not support the immediately LoRA apply mode; "
|
||||
"LoRAs will not be applied to them (use --lora-apply-mode at_runtime)");
|
||||
apply_lora_immediately = false;
|
||||
} else {
|
||||
apply_lora_immediately = true;
|
||||
}
|
||||
apply_lora_immediately = true;
|
||||
} else {
|
||||
apply_lora_immediately = false;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user