From f40fff30b2ce94e7fce62b6a77ad521c6b37cc3b Mon Sep 17 00:00:00 2001 From: aegioscy Date: Mon, 28 Sep 2026 05:44:46 +0000 Subject: [PATCH 1/5] feat(convrot): integrate MiniMax H3 on 2026-08-11 --- ggml | 2 +- src/core/ggml_extend.hpp | 242 ++++++++++++++- src/model/adapter/lora.hpp | 16 + src/model/diffusion/minimax_h3.hpp | 4 +- src/model/te/llm.hpp | 4 +- src/model_io/safetensors_io.cpp | 298 +++++++++++++++--- src/model_io/tensor_storage.h | 43 +++ src/model_loader.cpp | 267 ++++++++++++++++- src/model_loader.h | 5 + src/model_manager.cpp | 138 +++++++++ src/stable-diffusion.cpp | 10 +- tests/CMakeLists.txt | 6 + tests/test-safetensors-convrot.cpp | 448 ++++++++++++++++++++++++++++ tests/test_safetensors_metadata.cpp | 3 +- 14 files changed, 1437 insertions(+), 49 deletions(-) create mode 100644 tests/test-safetensors-convrot.cpp diff --git a/ggml b/ggml index 7d9ce11cd4..9a7d2b36e9 160000 --- a/ggml +++ b/ggml @@ -1 +1 @@ -Subproject commit 7d9ce11cd47f338b361a00e866ffe7c224abedff +Subproject commit 9a7d2b36e96a198c1c67c013cafcde4e4c2c60eb diff --git a/src/core/ggml_extend.hpp b/src/core/ggml_extend.hpp index ed99e4977b..898af09003 100644 --- a/src/core/ggml_extend.hpp +++ b/src/core/ggml_extend.hpp @@ -20,6 +20,7 @@ #include #include #include +#include #include #include #include @@ -43,6 +44,153 @@ #define EPS 1e-05f +// Construct only the operation metadata needed for the normal backend +// supports_op query. This stays private to the loader policy: backend +// capabilities are expressed through the existing ggml interface, not a new +// public ConvRot-specific API. +inline bool ggml_backend_supports_convrot_op(ggml_backend_t backend) { + if (backend == nullptr) { + return false; + } + std::vector storage(4 * ggml_tensor_overhead() + 1024); + ggml_init_params params = { + /*.mem_size =*/storage.size(), + /*.mem_buffer =*/storage.data(), + /*.no_alloc =*/true, + }; + ggml_context* ctx = ggml_init(params); + if (ctx == nullptr) { + return false; + } + ggml_tensor* activations = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, 256, 1); + ggml_tensor* weights = ggml_new_tensor_2d(ctx, GGML_TYPE_I8, 256, 1); + ggml_tensor* scales = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, 1); + ggml_tensor* op = ggml_mul_mat_convrot(ctx, activations, weights, scales, 256); + const bool supported = ggml_backend_supports_op(backend, op); + ggml_free(ctx); + return supported; +} + +inline bool ggml_backend_supports_convrot_rotation_op(ggml_backend_t backend) { + if (backend == nullptr) { + return false; + } + std::vector storage(3 * ggml_tensor_overhead() + 1024); + ggml_init_params params = { + /*.mem_size =*/storage.size(), + /*.mem_buffer =*/storage.data(), + /*.no_alloc =*/true, + }; + ggml_context* ctx = ggml_init(params); + if (ctx == nullptr) { + return false; + } + ggml_tensor* activations = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, 256, 1); + ggml_tensor* op = ggml_convrot(ctx, activations, 256); + const bool supported = ggml_backend_supports_op(backend, op); + ggml_free(ctx); + return supported; +} + +enum class ConvRotExecutionPath { + NATIVE, + DENSE_H256, + ROTATION_OP, + COMPAT, +}; + +inline ConvRotExecutionPath select_convrot_execution_path(const char* backend_name, + bool rotation_op_supported, + const char* mode) { + if (mode != nullptr && std::strcmp(mode, "compat") == 0) { + return ConvRotExecutionPath::COMPAT; + } + if (mode != nullptr && std::strcmp(mode, "native") == 0) { + return ConvRotExecutionPath::NATIVE; + } + if (mode != nullptr && std::strcmp(mode, "dense") == 0) { + return ConvRotExecutionPath::DENSE_H256; + } + if (mode != nullptr && std::strcmp(mode, "op") == 0) { + return rotation_op_supported ? ConvRotExecutionPath::ROTATION_OP + : ConvRotExecutionPath::NATIVE; + } + if (mode != nullptr && std::strcmp(mode, "q8") != 0 && std::strcmp(mode, "auto") != 0) { + throw std::runtime_error("invalid SD_CONVROT_MODE; expected 'auto', 'native', 'q8', 'dense', 'op', or 'compat'"); + } + + // The CUDA radix-4 transform is mathematically correct, but the H3 text + // encoder amplifies its rounding difference enough to change the scene. + // CUDA therefore keeps the numerically stable dense transform. Other + // backends use the standalone transform when they implement it. + if (backend_name != nullptr && std::strncmp(backend_name, "CUDA", 4) == 0) { + return ConvRotExecutionPath::DENSE_H256; + } + return rotation_op_supported ? ConvRotExecutionPath::ROTATION_OP + : ConvRotExecutionPath::NATIVE; +} + +// Select the compact representation before any model parameter tensor is +// created. SD_CONVROT_MODE can override the automatic per-backend policy. +inline String2TensorStorage select_convrot_tensor_storage(ggml_backend_t backend, + const String2TensorStorage& source, + const std::string& component, + const std::string& prefix = "") { + // OrderedMap's default copy also copies its iterator index; rebuild it so + // this independent policy view owns a valid index into its own list. + String2TensorStorage selected; + bool has_convrot = false; + for (const auto& [name, storage] : source) { + selected.insert({name, storage}); + has_convrot = has_convrot || + (name.rfind(prefix, 0) == 0 && storage.is_comfy_int8_convrot_weight()); + } + if (!has_convrot) { + return selected; + } + + const char* mode = std::getenv("SD_CONVROT_MODE"); + const char* backend_name = backend != nullptr ? ggml_backend_name(backend) : "unknown"; + const bool rotation_op_supported = ggml_backend_supports_convrot_rotation_op(backend); + const ConvRotExecutionPath path = select_convrot_execution_path(backend_name, rotation_op_supported, mode); + if (path == ConvRotExecutionPath::COMPAT) { + LOG_INFO("ConvRot: using explicitly selected F16 compatibility path for %s on backend %s", + component.c_str(), backend_name); + return selected; + } + if (path == ConvRotExecutionPath::DENSE_H256 || path == ConvRotExecutionPath::ROTATION_OP) { + for (auto& [name, storage] : selected) { + if (name.rfind(prefix, 0) == 0 && storage.is_comfy_int8_convrot_weight()) { + storage.comfy_int8_native_enabled = true; + storage.comfy_int8_q8_decomp_enabled = true; + storage.comfy_int8_convrot_op_enabled = path == ConvRotExecutionPath::ROTATION_OP; + } + } + LOG_INFO("ConvRot: selected Q8_0 %s activation transform for %s on backend %s", + path == ConvRotExecutionPath::ROTATION_OP ? "standalone" : "dense-H256", + component.c_str(), backend_name); + return selected; + } + if (mode != nullptr && std::strcmp(mode, "op") == 0 && !rotation_op_supported) { + LOG_WARN("ConvRot: standalone rotation is unavailable on backend %s; falling back to the native operation for %s", + backend_name, component.c_str()); + } + if (!ggml_backend_supports_convrot_op(backend)) { + throw std::runtime_error("ConvRot native support is required for " + component + + " but backend '" + backend_name + + "' lacks the 256-wide I8/F32 ConvRot operation; use a capable backend or set " + "SD_CONVROT_MODE=compat to select the F16 compatibility path"); + } + for (auto& [name, storage] : selected) { + if (name.rfind(prefix, 0) == 0 && storage.is_comfy_int8_convrot_weight()) { + storage.comfy_int8_native_enabled = true; + } + } + LOG_INFO("ConvRot: selected native compact I8/F32 path for %s on backend %s", + component.c_str(), backend_name); + return selected; +} + #ifndef __STATIC_INLINE__ #define __STATIC_INLINE__ static inline #endif @@ -1695,6 +1843,15 @@ struct WeightAdapter { ggml_tensor* b, const std::string& prefix, ForwardParams forward_params) = 0; + // Return only the adapter's output-space contribution. Native operations + // such as compact ConvRot own their base-weight arithmetic and therefore + // cannot use forward_with_lora() without recomputing an incompatible base. + virtual ggml_tensor* lora_output_delta(ggml_context* ctx, + ggml_backend_t backend, + ggml_tensor* x, + ggml_tensor* w, + const std::string& prefix, + ForwardParams forward_params) = 0; virtual size_t get_extra_graph_size() = 0; }; @@ -3973,12 +4130,51 @@ class Linear : public UnaryBlock { bool force_prec_f32; bool allow_weight_scale; bool has_weight_scale = false; + // This is distinct from `weight_scale`: the latter is a regular + // post-linear model parameter, while ConvRot's F32 vector is a private + // sidecar input to GGML_OP_MUL_MAT_CONVROT. + bool has_convrot_weight = false; + bool use_convrot_f16_compat = false; + bool use_convrot_q8_decomp = false; + bool use_convrot_rotation_op = false; + bool use_convrot_fast_h256 = false; float scale; std::string prefix; void init_params(ggml_context* ctx, const String2TensorStorage& tensor_storage_map = {}, const std::string prefix = "") override { this->prefix = prefix; has_weight_scale = false; + has_convrot_weight = false; + use_convrot_f16_compat = false; + use_convrot_q8_decomp = false; + use_convrot_rotation_op = false; + use_convrot_fast_h256 = false; + const auto storage_it = tensor_storage_map.find(prefix + "weight"); + if (storage_it != tensor_storage_map.end() && storage_it->second.is_comfy_int8_convrot_weight() && + storage_it->second.comfy_int8_native_enabled) { + has_convrot_weight = true; + if (storage_it->second.comfy_int8_q8_decomp_enabled) { + params["weight"] = ggml_new_tensor_2d(ctx, GGML_TYPE_Q8_0, in_features, out_features); + use_convrot_q8_decomp = true; + use_convrot_rotation_op = storage_it->second.comfy_int8_convrot_op_enabled; + if (!use_convrot_rotation_op) { + params["weight.convrot_h256"] = ggml_new_tensor_2d(ctx, GGML_TYPE_F32, 256, 256); + const char* h256_mode = std::getenv("SD_CONVROT_H256_MODE"); + use_convrot_fast_h256 = h256_mode != nullptr && std::strcmp(h256_mode, "fast") == 0; + if (h256_mode != nullptr && !use_convrot_fast_h256 && std::strcmp(h256_mode, "dense") != 0) { + throw std::runtime_error("invalid SD_CONVROT_H256_MODE; expected 'dense' or 'fast'"); + } + } + } else { + params["weight"] = ggml_new_tensor_2d(ctx, GGML_TYPE_I8, in_features, out_features); + params["weight.convrot_scale"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, out_features); + use_convrot_f16_compat = storage_it->second.name.rfind("text_encoders.llm.", 0) == 0; + } + if (bias) { + params["bias"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, out_features); + } + return; + } enum ggml_type wtype = get_type(prefix + "weight", tensor_storage_map, GGML_TYPE_F32); if (in_features % ggml_blck_size(wtype) != 0 || force_f32) { wtype = GGML_TYPE_F32; @@ -4026,7 +4222,51 @@ class Linear : public UnaryBlock { } ggml_tensor* linear_bias = has_weight_scale ? nullptr : b; ggml_tensor* out = nullptr; - if (ctx->weight_adapter) { + if (has_convrot_weight) { + if (use_convrot_q8_decomp) { + if (use_convrot_rotation_op) { + ggml_tensor* rotated = ggml_convrot(ctx->ggml_ctx, x, 256); + ggml_set_name(rotated, (prefix + "trace.convrot.post_h256").c_str()); + out = ggml_mul_mat(ctx->ggml_ctx, w, rotated); + } else { + // H256 is symmetric: (H*w_row).x == w_row.(H*x). Rotate + // each 256-wide block before the stock Q8_0 matmul. + ggml_tensor* contiguous = ggml_is_contiguous(x) ? x : ggml_cont(ctx->ggml_ctx, x); + ggml_tensor* blocks = ggml_reshape_2d(ctx->ggml_ctx, contiguous, 256, + ggml_nelements(contiguous) / 256); + ggml_set_name(blocks, (prefix + "trace.convrot.pre_h256").c_str()); + ggml_tensor* rotated = ggml_mul_mat(ctx->ggml_ctx, params["weight.convrot_h256"], blocks); + ggml_set_name(rotated, (prefix + "trace.convrot.post_h256").c_str()); + if (use_convrot_fast_h256) { + ggml_mul_mat_set_hint(rotated, GGML_HINT_SRC0_IS_CONVROT_H256); + } + rotated = ggml_reshape_4d(ctx->ggml_ctx, rotated, x->ne[0], x->ne[1], x->ne[2], x->ne[3]); + out = ggml_mul_mat(ctx->ggml_ctx, w, rotated); + } + ggml_set_name(out, (prefix + "trace.convrot.post_q8_gemm").c_str()); + } else { + // ConvRot weights and their tensor-wise scales remain compact + // at rest. The operator owns the scale semantics. + out = ggml_mul_mat_convrot(ctx->ggml_ctx, x, w, params["weight.convrot_scale"], 256); + ggml_set_name(out, (prefix + "trace.convrot.native_out").c_str()); + if (use_convrot_f16_compat) { + ggml_mul_mat_convrot_set_f16_compat(out, true); + } + } + if (b != nullptr) { + out = ggml_add_inplace(ctx->ggml_ctx, out, b); + } + if (ctx->weight_adapter) { + WeightAdapter::ForwardParams forward_params; + forward_params.op_type = WeightAdapter::ForwardParams::op_type_t::OP_LINEAR; + forward_params.linear.force_prec_f32 = force_prec_f32; + forward_params.linear.scale = scale; + if (ggml_tensor* delta = ctx->weight_adapter->lora_output_delta( + ctx->ggml_ctx, ctx->backend, x, w, prefix, forward_params)) { + out = ggml_add_inplace(ctx->ggml_ctx, out, delta); + } + } + } else if (ctx->weight_adapter) { WeightAdapter::ForwardParams forward_params; forward_params.op_type = WeightAdapter::ForwardParams::op_type_t::OP_LINEAR; forward_params.linear.force_prec_f32 = force_prec_f32; diff --git a/src/model/adapter/lora.hpp b/src/model/adapter/lora.hpp index 9c19e46996..b21c7e61b0 100644 --- a/src/model/adapter/lora.hpp +++ b/src/model/adapter/lora.hpp @@ -1305,6 +1305,22 @@ struct MultiLoraAdapter : public WeightAdapter { return out; } + ggml_tensor* lora_output_delta(ggml_context* ctx, + ggml_backend_t backend, + ggml_tensor* x, + ggml_tensor* w, + const std::string& prefix, + WeightAdapter::ForwardParams forward_params) override { + ggml_tensor* delta = nullptr; + for (auto& lora_model : lora_models) { + ggml_tensor* current = lora_model->get_out_diff(ctx, backend, x, w, forward_params, prefix + "weight"); + if (current != nullptr) { + delta = delta == nullptr ? current : ggml_add_inplace(ctx, delta, current); + } + } + return delta; + } + size_t get_extra_graph_size() override { size_t lora_tensor_num = 0; for (auto& lora_model : lora_models) { diff --git a/src/model/diffusion/minimax_h3.hpp b/src/model/diffusion/minimax_h3.hpp index e5a5a9a9b4..f2f18d1a42 100644 --- a/src/model/diffusion/minimax_h3.hpp +++ b/src/model/diffusion/minimax_h3.hpp @@ -986,7 +986,9 @@ namespace MiniMaxH3 { : DiffusionModelRunner(backend, prefix, weight_manager), config(Config::detect_from_weights(tensors, prefix)), model(config) { - model.init(params_ctx, tensors, prefix); + model.init(params_ctx, + select_convrot_tensor_storage(backend, tensors, "MiniMax-H3 diffusion model", prefix), + prefix); } std::string get_desc() override { diff --git a/src/model/te/llm.hpp b/src/model/te/llm.hpp index f4dfa9f763..0615546b39 100644 --- a/src/model/te/llm.hpp +++ b/src/model/te/llm.hpp @@ -1771,7 +1771,9 @@ namespace LLM { } } model = LLM(config, enable_vision, config.llama_cpp_style); - model.init(params_ctx, tensor_storage_map, prefix); + model.init(params_ctx, + select_convrot_tensor_storage(backend, tensor_storage_map, "LLM", prefix), + prefix); } std::string get_desc() override { diff --git a/src/model_io/safetensors_io.cpp b/src/model_io/safetensors_io.cpp index f63e4ded9b..fcff6ed085 100644 --- a/src/model_io/safetensors_io.cpp +++ b/src/model_io/safetensors_io.cpp @@ -6,7 +6,9 @@ #include #include #include +#include #include +#include #include #include #include @@ -94,10 +96,217 @@ static ggml_type safetensors_dtype_to_ggml_type(const std::string& dtype) { ttype = GGML_TYPE_I32; } else if (dtype == "I64") { ttype = GGML_TYPE_I32; + } else if (dtype == "I8") { + ttype = GGML_TYPE_I8; } return ttype; } +struct SafetensorsTensorInfo { + std::string dtype; + std::vector shape; + uint64_t begin = 0; + uint64_t end = 0; +}; + +struct ComfyInt8Info { + bool convrot = false; + uint32_t group_size = 0; + TensorStorageSidecar scale; +}; + +static bool read_safetensors_tensor_info(const nlohmann::json& value, + const std::string& name, + uint64_t data_start, + uint64_t data_size, + bool metadata_only, + SafetensorsTensorInfo* result, + std::string* error) { + try { + if (!value.is_object() || !value.contains("dtype") || !value["dtype"].is_string() || + !value.contains("shape") || !value["shape"].is_array() || + !value.contains("data_offsets") || !value["data_offsets"].is_array() || + value["data_offsets"].size() != 2) { + set_error(error, "invalid safetensors descriptor for tensor '" + name + "'"); + return false; + } + + SafetensorsTensorInfo info; + info.dtype = value["dtype"].get(); + if (value["shape"].size() > SD_MAX_DIMS) { + set_error(error, "too many dimensions for tensor '" + name + "'"); + return false; + } + uint64_t elements = 1; + for (const auto& dimension : value["shape"]) { + if (!dimension.is_number_unsigned()) { + set_error(error, "invalid dimension for tensor '" + name + "'"); + return false; + } + const uint64_t size = dimension.get(); + if (size > static_cast(std::numeric_limits::max()) || + (size != 0 && elements > static_cast(std::numeric_limits::max()) / size)) { + set_error(error, "invalid dimension for tensor '" + name + "'"); + return false; + } + elements *= size; + info.shape.push_back(static_cast(size)); + } + const auto& offsets = value["data_offsets"]; + if (!offsets[0].is_number_unsigned() || !offsets[1].is_number_unsigned()) { + set_error(error, "invalid data offsets for tensor '" + name + "'"); + return false; + } + info.begin = offsets[0].get(); + info.end = offsets[1].get(); + if (info.begin > info.end || info.end > std::numeric_limits::max() - data_start || + (!metadata_only && info.end > data_size)) { + set_error(error, "data offsets out of bounds for tensor '" + name + "'"); + return false; + } + *result = std::move(info); + return true; + } catch (const std::exception&) { + set_error(error, "invalid safetensors descriptor for tensor '" + name + "'"); + return false; + } +} + +static bool is_power_of_four(uint64_t value) { + if (value < 4) { + return false; + } + while (value % 4 == 0) { + value /= 4; + } + return value == 1; +} + +static bool read_comfy_int8_metadata(std::ifstream& file, + const std::map& tensors, + uint64_t data_start, + uint64_t data_size, + std::map* result, + std::set* scale_tensor_names, + std::string* error) { + result->clear(); + scale_tensor_names->clear(); + constexpr const char* marker_suffix = ".comfy_quant"; + constexpr size_t marker_suffix_len = 12; + + for (const auto& [marker_name, marker] : tensors) { + if (!ends_with(marker_name, marker_suffix)) { + continue; + } + if (marker.dtype != "U8") { + set_error(error, "ComfyUI quantization marker '" + marker_name + "' must use U8 storage"); + return false; + } + if (marker.end - marker.begin > 4096) { + set_error(error, "ComfyUI quantization marker '" + marker_name + "' is too large"); + return false; + } + if (marker.end > data_size || data_start + marker.begin > static_cast(std::numeric_limits::max())) { + set_error(error, "ComfyUI quantization marker '" + marker_name + "' bytes are missing"); + return false; + } + uint64_t marker_elements = 1; + for (int64_t dimension : marker.shape) { + if (dimension <= 0 || marker_elements > std::numeric_limits::max() / + static_cast(dimension)) { + set_error(error, "ComfyUI quantization marker '" + marker_name + "' shape overflows"); + return false; + } + marker_elements *= static_cast(dimension); + } + if (marker_elements != marker.end - marker.begin) { + set_error(error, "ComfyUI quantization marker '" + marker_name + "' has an invalid byte length"); + return false; + } + + std::string marker_json(static_cast(marker.end - marker.begin), '\0'); + file.clear(); + file.seekg(static_cast(data_start + marker.begin)); + file.read(marker_json.data(), static_cast(marker_json.size())); + if (!file) { + set_error(error, "failed to read ComfyUI quantization marker '" + marker_name + "'"); + return false; + } + + // Reject malformed markers explicitly, including builds where JSON + // parsing exceptions are disabled. + nlohmann::json config = nlohmann::json::parse(marker_json, nullptr, false); + if (config.is_discarded()) { + set_error(error, "invalid JSON in ComfyUI quantization marker '" + marker_name + "'"); + return false; + } + if (!config.is_object() || !config.contains("format") || !config["format"].is_string()) { + set_error(error, "invalid ComfyUI quantization marker '" + marker_name + "'"); + return false; + } + if (config["format"].get() != "int8_tensorwise") { + continue; + } + + const std::string base = marker_name.substr(0, marker_name.size() - marker_suffix_len); + const auto weight_it = tensors.find(base + ".weight"); + const auto scale_it = tensors.find(base + ".weight_scale"); + if (weight_it == tensors.end() || scale_it == tensors.end()) { + set_error(error, "ComfyUI Int8 marker '" + marker_name + "' is missing its weight or weight_scale tensor"); + return false; + } + const SafetensorsTensorInfo& weight = weight_it->second; + const SafetensorsTensorInfo& scale = scale_it->second; + if (weight.dtype != "I8" || weight.shape.size() != 2 || scale.dtype != "F32" || + scale.shape.size() != 2 || scale.shape[0] != weight.shape[0] || scale.shape[1] != 1) { + set_error(error, "ComfyUI Int8 marker '" + marker_name + "' has incompatible weight and scale tensors"); + return false; + } + const uint64_t output_rows = static_cast(weight.shape[0]); + if (output_rows > std::numeric_limits::max() / sizeof(float) || + scale.end - scale.begin != output_rows * sizeof(float)) { + set_error(error, "ComfyUI Int8 marker '" + marker_name + "' has incompatible weight and scale tensors"); + return false; + } + + ComfyInt8Info info; + if (config.contains("convrot")) { + if (!config["convrot"].is_boolean()) { + set_error(error, "ComfyUI Int8 marker '" + marker_name + "' has a non-boolean convrot field"); + return false; + } + info.convrot = config["convrot"].get(); + } + if (info.convrot) { + if (!config.contains("convrot_groupsize") || !config["convrot_groupsize"].is_number_unsigned()) { + set_error(error, "ComfyUI Int8 marker '" + marker_name + "' has no valid ConvRot group size"); + return false; + } + uint64_t group_size = config["convrot_groupsize"].get(); + if (group_size != 256 || !is_power_of_four(group_size) || + static_cast(weight.shape[1]) % group_size != 0) { + set_error(error, "ComfyUI Int8 marker '" + marker_name + "' uses an unsupported ConvRot group size"); + return false; + } + info.group_size = static_cast(group_size); + } else if (config.contains("convrot_groupsize")) { + set_error(error, "ComfyUI Int8 marker '" + marker_name + "' declares a group size without ConvRot"); + return false; + } + info.scale.name = base + ".weight_scale"; + info.scale.type = GGML_TYPE_F32; + info.scale.n_dims = 2; + // TensorStorage uses GGML's column-first shape convention. + info.scale.ne[0] = 1; + info.scale.ne[1] = weight.shape[0]; + info.scale.offset = data_start + scale.begin; + info.scale.nbytes = scale.end - scale.begin; + result->emplace(weight_it->first, info); + scale_tensor_names->emplace(scale_it->first); + } + return true; +} + // https://huggingface.co/docs/safetensors/index bool read_safetensors_file(const std::string& file_path, std::vector& tensor_storages, @@ -164,39 +373,43 @@ bool read_safetensors_file(const std::string& file_path, } } - tensor_storages.clear(); - for (auto& item : header_.items()) { - std::string name = item.key(); - nlohmann::json tensor_info = item.value(); - // LOG_DEBUG("%s %s\n", name.c_str(), tensor_info.dump().c_str()); - - if (name == "__metadata__") { + const uint64_t data_size = file_size_ - data_start; + std::map tensor_infos; + for (const auto& item : header_.items()) { + if (item.key() == "__metadata__") { continue; } + SafetensorsTensorInfo info; + if (!read_safetensors_tensor_info(item.value(), item.key(), data_start, data_size, + sd_get_metadata_only_read(), &info, error)) { + return false; + } + tensor_infos.emplace(item.key(), std::move(info)); + } + std::map comfy_int8_tensors; + std::set comfy_int8_scale_tensors; + if (!read_comfy_int8_metadata(file, + tensor_infos, + data_start, + data_size, + &comfy_int8_tensors, + &comfy_int8_scale_tensors, + error)) { + return false; + } + + tensor_storages.clear(); + for (const auto& [name, tensor_info] : tensor_infos) { + // LOG_DEBUG("%s %s\n", name.c_str(), tensor_info.dump().c_str()); - std::string dtype = tensor_info["dtype"]; - nlohmann::json shape = tensor_info["shape"]; + const std::string& dtype = tensor_info.dtype; - if (dtype == "U8") { + if (dtype == "U8" || comfy_int8_scale_tensors.find(name) != comfy_int8_scale_tensors.end()) { continue; } - const auto& offsets = tensor_info["data_offsets"]; - const auto valid_offset = [](const nlohmann::json& value) { - return value.is_number_unsigned() ? value.get() <= std::numeric_limits::max() - : value.is_number_integer() && value.get() >= 0; - }; - if (!offsets.is_array() || offsets.size() != 2 || !valid_offset(offsets[0]) || !valid_offset(offsets[1])) { - set_error(error, "invalid data offsets for tensor '" + name + "'"); - return false; - } - size_t begin = offsets[0].get(); - size_t end = offsets[1].get(); - if (begin > end || end > std::numeric_limits::max() - data_start || - (!sd_get_metadata_only_read() && end > file_size_ - data_start)) { - set_error(error, "data offsets out of bounds for tensor '" + name + "'"); - return false; - } + const uint64_t begin = tensor_info.begin; + const uint64_t end = tensor_info.end; ggml_type type = safetensors_dtype_to_ggml_type(dtype); if (type == GGML_TYPE_COUNT) { @@ -204,12 +417,7 @@ bool read_safetensors_file(const std::string& file_path, return false; } - if (!shape.is_array() || shape.size() > SD_MAX_DIMS) { - set_error(error, "invalid tensor '" + name + "'"); - return false; - } - - int n_dims = (int)shape.size(); + int n_dims = static_cast(tensor_info.shape.size()); int64_t ne[SD_MAX_DIMS] = {1, 1, 1, 1, 1}; // Bound intermediate products too, including F64/I64 conversion and zero-sized tensors. const uint64_t max_nelements = @@ -217,11 +425,7 @@ bool read_safetensors_file(const std::string& file_path, (2 * ggml_type_size(type)); uint64_t nelements = 1; for (int i = 0; i < n_dims; i++) { - if (!shape[i].is_number_unsigned()) { - set_error(error, "invalid shape for tensor '" + name + "'"); - return false; - } - const uint64_t dim = shape[i].get(); + const uint64_t dim = static_cast(tensor_info.shape[i]); if (dim > max_nelements || (dim != 0 && nelements > max_nelements / dim)) { set_error(error, "invalid shape for tensor '" + name + "'"); return false; @@ -231,6 +435,10 @@ bool read_safetensors_file(const std::string& file_path, } if (n_dims == 5) { + if (ne[1] == 0 || ne[0] > std::numeric_limits::max() / ne[1]) { + set_error(error, "tensor dimensions overflow for '" + name + "'"); + return false; + } n_dims = 4; ne[0] = ne[0] * ne[1]; ne[1] = ne[2]; @@ -246,7 +454,7 @@ bool read_safetensors_file(const std::string& file_path, TensorStorage tensor_storage(name, type, ne, n_dims, 0, data_start + begin); tensor_storage.reverse_ne(); - size_t tensor_data_size = end - begin; + uint64_t tensor_data_size = end - begin; bool tensor_size_ok; if (dtype == "F8_E4M3") { @@ -273,6 +481,20 @@ bool read_safetensors_file(const std::string& file_path, return false; } + auto comfy_int8 = comfy_int8_tensors.find(name); + if (dtype == "I8") { + if (comfy_int8 == comfy_int8_tensors.end()) { + set_error(error, "unsupported Int8 safetensors tensor '" + name + "' without a ComfyUI Int8 marker"); + return false; + } + tensor_storage.is_comfy_int8_tensorwise = true; + tensor_storage.comfy_int8_convrot = comfy_int8->second.convrot; + tensor_storage.comfy_int8_group_size = comfy_int8->second.group_size; + tensor_storage.comfy_int8_scale = comfy_int8->second.scale; + // The runtime reconstructs an F16 matrix before backend upload. + tensor_storage.expected_type = GGML_TYPE_F16; + } + tensor_storages.push_back(tensor_storage); // LOG_DEBUG("%s %s", tensor_storage.to_string().c_str(), dtype.c_str()); diff --git a/src/model_io/tensor_storage.h b/src/model_io/tensor_storage.h index 5c977f516f..118412a050 100644 --- a/src/model_io/tensor_storage.h +++ b/src/model_io/tensor_storage.h @@ -13,6 +13,23 @@ #define SD_MAX_DIMS 5 +// A safetensors sidecar is deliberately not added to the model parameter map. +// It remains addressable from its owning tensor so compound on-disk formats can +// upload both buffers without exposing implementation metadata as a second +// model parameter. +struct TensorStorageSidecar { + std::string name; + ggml_type type = GGML_TYPE_COUNT; + int64_t ne[SD_MAX_DIMS] = {1, 1, 1, 1, 1}; + int n_dims = 0; + uint64_t offset = 0; + uint64_t nbytes = 0; + + bool valid() const { + return !name.empty() && type != GGML_TYPE_COUNT && n_dims > 0 && n_dims <= SD_MAX_DIMS && nbytes > 0; + } +}; + struct TensorStorage { std::string name; ggml_type type = GGML_TYPE_F32; @@ -22,6 +39,23 @@ struct TensorStorage { bool is_f8_e5m2 = false; bool is_f64 = false; bool is_i64 = false; + // ComfyUI TensorWiseINT8 stores the I8 weight and its per-output-row F32 + // scale separately. ConvRot metadata is carried by a U8 JSON side tensor. + // Keep the scale as an associated raw sidecar: it must not be mistaken for + // a model-level `weight_scale` parameter, but native operators need it. + bool is_comfy_int8_tensorwise = false; + bool comfy_int8_convrot = false; + // Set by the runner's backend policy before parameters are constructed. + // False selects the verified F16 compatibility reconstruction. + bool comfy_int8_native_enabled = false; + // Repack the raw row-scaled I8 data as Q8_0 and move the symmetric H256 + // transform to the activations before an ordinary ggml_mul_mat. + bool comfy_int8_q8_decomp_enabled = false; + // Use GGML_OP_CONVROT for the activation transform instead of materializing + // and multiplying by a dense H256 matrix. + bool comfy_int8_convrot_op_enabled = false; + uint32_t comfy_int8_group_size = 0; + TensorStorageSidecar comfy_int8_scale; int64_t ne[SD_MAX_DIMS] = {1, 1, 1, 1, 1}; int n_dims = 0; @@ -61,6 +95,15 @@ struct TensorStorage { } } + bool has_comfy_int8_scale() const { + return is_comfy_int8_tensorwise && comfy_int8_scale.valid(); + } + + bool is_comfy_int8_convrot_weight() const { + return has_comfy_int8_scale() && comfy_int8_convrot && comfy_int8_group_size == 256 && n_dims == 2 && + ne[0] > 0 && ne[1] > 0 && ne[0] % static_cast(comfy_int8_group_size) == 0; + } + void unsqueeze() { if (n_dims == 2) { n_dims = 4; diff --git a/src/model_loader.cpp b/src/model_loader.cpp index a70ffefd3e..1330fd4416 100644 --- a/src/model_loader.cpp +++ b/src/model_loader.cpp @@ -2,6 +2,7 @@ #include #include #include +#include #include #include #include @@ -154,6 +155,85 @@ void i64_to_i32_vec(int64_t* src, int32_t* dst, int64_t n) { } } +static void apply_regular_hadamard_4(float* values, size_t stride) { + const float a = values[0 * stride]; + const float b = values[1 * stride]; + const float c = values[2 * stride]; + const float d = values[3 * stride]; + values[0 * stride] = (a + b + c - d) * 0.5f; + values[1 * stride] = (a + b - c + d) * 0.5f; + values[2 * stride] = (a - b + c + d) * 0.5f; + values[3 * stride] = (-a + b + c + d) * 0.5f; +} + +static bool dequantize_comfy_int8_tensorwise(const TensorStorage& tensor_storage, + const int8_t* quantized, + const float* scales, + void* dst, + ggml_type dst_type, + std::string* error) { + if (!tensor_storage.is_comfy_int8_tensorwise || tensor_storage.n_dims != 2 || + tensor_storage.ne[0] <= 0 || tensor_storage.ne[1] <= 0) { + *error = "invalid ComfyUI Int8 tensor metadata"; + return false; + } + if (dst_type != GGML_TYPE_F16 && dst_type != GGML_TYPE_F32) { + *error = "ComfyUI Int8 compatibility loading requires an F16 or F32 destination"; + return false; + } + + const size_t columns = static_cast(tensor_storage.ne[0]); + const size_t rows = static_cast(tensor_storage.ne[1]); + if (rows > std::numeric_limits::max() / columns || !tensor_storage.has_comfy_int8_scale() || + tensor_storage.comfy_int8_scale.type != GGML_TYPE_F32 || tensor_storage.comfy_int8_scale.n_dims != 2 || + tensor_storage.comfy_int8_scale.ne[0] != 1 || tensor_storage.comfy_int8_scale.ne[1] != static_cast(rows) || + tensor_storage.comfy_int8_scale.nbytes != rows * sizeof(float)) { + *error = "invalid ComfyUI Int8 tensor dimensions or scale size"; + return false; + } + const size_t group_size = tensor_storage.comfy_int8_convrot + ? static_cast(tensor_storage.comfy_int8_group_size) + : std::min(columns, 4096); + if (group_size == 0 || (tensor_storage.comfy_int8_convrot && columns % group_size != 0)) { + *error = "invalid ComfyUI Int8 ConvRot group size"; + return false; + } + + std::vector values(group_size); + for (size_t row = 0; row < rows; ++row) { + const float scale = scales[row]; + if (!std::isfinite(scale) || scale <= 0.f) { + *error = "ComfyUI Int8 tensor has a non-positive or non-finite scale"; + return false; + } + for (size_t column = 0; column < columns; column += group_size) { + const size_t chunk_size = std::min(group_size, columns - column); + const size_t offset = row * columns + column; + for (size_t i = 0; i < chunk_size; ++i) { + values[i] = static_cast(quantized[offset + i]) * scale; + } + if (tensor_storage.comfy_int8_convrot) { + for (size_t stride = 1; stride < group_size; stride *= 4) { + const size_t block = stride * 4; + for (size_t base = 0; base < group_size; base += block) { + for (size_t i = 0; i < stride; ++i) { + apply_regular_hadamard_4(values.data() + base + i, stride); + } + } + } + } + if (dst_type == GGML_TYPE_F16) { + auto* output = static_cast(dst) + offset; + ggml_fp32_to_fp16_row(values.data(), output, static_cast(chunk_size)); + } else { + auto* output = static_cast(dst) + offset; + memcpy(output, values.data(), chunk_size * sizeof(float)); + } + } + } + return true; +} + void convert_tensor(void* src, ggml_type src_type, void* dst, @@ -1240,10 +1320,42 @@ bool ModelLoader::load_tensors(on_new_tensor_cb_t on_new_tensor_cb, return true; }; + auto read_comfy_int8_scales = [&](char* buf, size_t n) -> bool { + if (zip != nullptr) { + LOG_ERROR("ComfyUI Int8 tensor '%s' cannot be stored in a zip file", tensor_storage.name.c_str()); + return false; + } + if (mmapped) { + if (!mmapped->copy_data(buf, n, tensor_storage.comfy_int8_scale.offset)) { + LOG_ERROR("read ComfyUI Int8 scales failed: '%s'", file_path.c_str()); + return false; + } + } else { + file.clear(); + file.seekg(static_cast(tensor_storage.comfy_int8_scale.offset)); + file.read(buf, static_cast(n)); + if (!file) { + LOG_ERROR("read ComfyUI Int8 scales failed: '%s'", file_path.c_str()); + return false; + } + } + return true; + }; + char* read_buf = nullptr; char* target_buf = nullptr; char* convert_buf = nullptr; - if (dst_tensor->buffer == nullptr || ggml_backend_buffer_is_host(dst_tensor->buffer)) { + const bool is_comfy_int8 = tensor_storage.is_comfy_int8_tensorwise; + if (is_comfy_int8) { + read_buffer.resize(nbytes_to_read); + read_buf = reinterpret_cast(read_buffer.data()); + if (dst_tensor->buffer == nullptr || ggml_backend_buffer_is_host(dst_tensor->buffer)) { + target_buf = reinterpret_cast(dst_tensor->data); + } else { + convert_buffer.resize(ggml_nbytes(dst_tensor)); + target_buf = reinterpret_cast(convert_buffer.data()); + } + } else if (dst_tensor->buffer == nullptr || ggml_backend_buffer_is_host(dst_tensor->buffer)) { if (tensor_storage.type == dst_tensor->type) { GGML_ASSERT(ggml_nbytes(dst_tensor) == tensor_storage.nbytes()); if (tensor_storage.is_f64 || tensor_storage.is_i64) { @@ -1279,7 +1391,27 @@ bool ModelLoader::load_tensors(on_new_tensor_cb_t on_new_tensor_cb, read_time_ms.fetch_add(t1 - t0); t0 = ggml_time_ms(); - if (tensor_storage.is_f8_e4m3) { + if (is_comfy_int8) { + std::vector scale_buffer(tensor_storage.comfy_int8_scale.nbytes / sizeof(float)); + if (!read_comfy_int8_scales(reinterpret_cast(scale_buffer.data()), tensor_storage.comfy_int8_scale.nbytes)) { + failed = true; + break; + } + std::string dequantization_error; + if (!dequantize_comfy_int8_tensorwise(tensor_storage, + reinterpret_cast(read_buf), + scale_buffer.data(), + target_buf, + dst_tensor->type, + &dequantization_error)) { + LOG_ERROR("ComfyUI Int8 tensor '%s' cannot be reconstructed: %s", + tensor_storage.name.c_str(), + dequantization_error.c_str()); + failed = true; + break; + } + convert_buf = target_buf; + } else if (tensor_storage.is_f8_e4m3) { f8_e4m3_to_f16_vec((uint8_t*)read_buf, (uint16_t*)target_buf, tensor_storage.nelements()); } else if (tensor_storage.is_f8_e5m2) { f8_e5m2_to_f16_vec((uint8_t*)read_buf, (uint16_t*)target_buf, tensor_storage.nelements()); @@ -1288,7 +1420,7 @@ bool ModelLoader::load_tensors(on_new_tensor_cb_t on_new_tensor_cb, } else if (tensor_storage.is_i64) { i64_to_i32_vec((int64_t*)read_buf, (int32_t*)target_buf, tensor_storage.nelements()); } - if (tensor_storage.type != dst_tensor->type) { + if (!is_comfy_int8 && tensor_storage.type != dst_tensor->type) { if (convert_buf == nullptr) { LOG_ERROR("read tensor data failed: too less memory for conversion"); failed = true; @@ -1303,7 +1435,7 @@ bool ModelLoader::load_tensors(on_new_tensor_cb_t on_new_tensor_cb, tensor_storage.nelements() / tensor_storage.ne[0], tensor_storage.ne[0], std::move(imatrix)); - } else { + } else if (!is_comfy_int8) { convert_buf = read_buf; } t1 = ggml_time_ms(); @@ -1319,7 +1451,8 @@ bool ModelLoader::load_tensors(on_new_tensor_cb_t on_new_tensor_cb, copy_to_backend_time_ms.fetch_add(t1 - t0); } - bytes_processed.fetch_add((uint64_t)nbytes_to_read); + bytes_processed.fetch_add((uint64_t)nbytes_to_read + + (is_comfy_int8 ? tensor_storage.comfy_int8_scale.nbytes : 0)); } if (zip != nullptr) { zip_close(zip); @@ -1427,6 +1560,130 @@ bool ModelLoader::load_tensor(const TensorStorage& tensor_storage, ggml_tensor* return true; } +bool ModelLoader::load_comfy_int8_tensorwise(const TensorStorage& tensor_storage, + ggml_tensor* dst_weight, + ggml_tensor* dst_scale) { + if (!tensor_storage.is_comfy_int8_convrot_weight()) { + LOG_ERROR("native ComfyUI Int8 load requested for invalid ConvRot tensor '%s'", tensor_storage.name.c_str()); + return false; + } + const bool q8_decomp = dst_weight != nullptr && dst_weight->type == GGML_TYPE_Q8_0; + if (dst_weight == nullptr || dst_weight->data == nullptr || + (!q8_decomp && (dst_scale == nullptr || dst_scale->data == nullptr))) { + LOG_ERROR("native ComfyUI Int8 load has null destination for '%s'", tensor_storage.name.c_str()); + return false; + } + const bool weight_shape_ok = ggml_n_dims(dst_weight) == 2 && dst_weight->ne[0] == tensor_storage.ne[0] && + dst_weight->ne[1] == tensor_storage.ne[1]; + const bool weight_storage_ok = q8_decomp + ? ggml_nbytes(dst_weight) == ggml_row_size(GGML_TYPE_Q8_0, tensor_storage.ne[0]) * static_cast(tensor_storage.ne[1]) + : dst_weight->type == GGML_TYPE_I8 && ggml_nbytes(dst_weight) == static_cast(tensor_storage.nbytes()); + if (!weight_shape_ok || !weight_storage_ok) { + LOG_ERROR("native ComfyUI Int8 weight destination is incompatible for '%s'", tensor_storage.name.c_str()); + return false; + } + const int64_t output_rows = tensor_storage.ne[1]; + const bool scale_shape_ok = q8_decomp || + (dst_scale->type == GGML_TYPE_F32 && + ((ggml_n_dims(dst_scale) == 1 && dst_scale->ne[0] == output_rows) || + (ggml_n_dims(dst_scale) == 2 && dst_scale->ne[0] == 1 && dst_scale->ne[1] == output_rows))); + if (!q8_decomp && (!scale_shape_ok || ggml_nbytes(dst_scale) != tensor_storage.comfy_int8_scale.nbytes)) { + LOG_ERROR("native ComfyUI Int8 scale destination is incompatible for '%s'", tensor_storage.name.c_str()); + return false; + } + if (tensor_storage.file_index >= file_paths_.size()) { + LOG_ERROR("native ComfyUI Int8 source file is unavailable for '%s'", tensor_storage.name.c_str()); + return false; + } + if (tensor_storage.index_in_zip >= 0) { + LOG_ERROR("native ComfyUI Int8 tensor '%s' cannot be loaded from a zip container", tensor_storage.name.c_str()); + return false; + } + + const size_t weight_nbytes = static_cast(tensor_storage.nbytes()); + const size_t scale_nbytes = static_cast(tensor_storage.comfy_int8_scale.nbytes); + std::vector weights(weight_nbytes); + std::vector scales(scale_nbytes); + std::ifstream file(file_paths_[tensor_storage.file_index], std::ios::binary); + if (!file.is_open()) { + LOG_ERROR("failed to open native ComfyUI Int8 source '%s'", file_paths_[tensor_storage.file_index].c_str()); + return false; + } + const auto read_at = [&](uint64_t offset, uint8_t* dst, size_t n, const char* what) -> bool { + if (offset > static_cast(std::numeric_limits::max())) { + LOG_ERROR("native ComfyUI Int8 %s offset overflows for '%s'", what, tensor_storage.name.c_str()); + return false; + } + file.clear(); + file.seekg(static_cast(offset)); + file.read(reinterpret_cast(dst), static_cast(n)); + if (!file) { + LOG_ERROR("failed to read native ComfyUI Int8 %s for '%s'", what, tensor_storage.name.c_str()); + return false; + } + return true; + }; + if (!read_at(tensor_storage.offset, weights.data(), weights.size(), "weight") || + !read_at(tensor_storage.comfy_int8_scale.offset, scales.data(), scales.size(), "scale")) { + return false; + } + + // The compact native path does not dequantize on the host, so validate the + // sidecar before uploading it. Otherwise a malformed zero/NaN scale + // reaches the backend without the compatibility loader's validation. + const size_t scale_count = static_cast(output_rows); + std::vector scale_values(scale_count); + for (size_t row = 0; row < scale_count; ++row) { + memcpy(&scale_values[row], scales.data() + row * sizeof(float), sizeof(float)); + if (!std::isfinite(scale_values[row]) || scale_values[row] <= 0.f) { + LOG_ERROR("native ComfyUI Int8 tensor '%s' has a non-positive or non-finite scale", tensor_storage.name.c_str()); + return false; + } + } + + const auto upload = [](ggml_tensor* dst, const void* src, size_t n) { + if (dst->buffer != nullptr && !ggml_backend_buffer_is_host(dst->buffer)) { + ggml_backend_tensor_set(dst, src, 0, n); + } else { + memcpy(dst->data, src, n); + } + }; + if (q8_decomp) { + size_t subnormal_rows = 0; + double max_scale_relative_error = 0.0; + for (size_t row = 0; row < scale_count; ++row) { + const float original = scale_values[row]; + const float packed = ggml_fp16_to_fp32(ggml_fp32_to_fp16(original)); + if (!std::isfinite(packed) || packed <= 0.0f) { + LOG_ERROR("ConvRot Q8_0 scale for '%s' row %zu is outside the finite positive F16 range; use SD_CONVROT_MODE=native", + tensor_storage.name.c_str(), row); + return false; + } + subnormal_rows += original < 0x1p-14f; + max_scale_relative_error = std::max(max_scale_relative_error, + double(std::fabs(packed - original)) / original); + } + if (subnormal_rows > 0) { + LOG_WARN("ConvRot Q8_0 '%s': %zu/%zu scales are F16 subnormal; max scale relative error %.3g", + tensor_storage.name.c_str(), subnormal_rows, scale_count, max_scale_relative_error); + } + const size_t packed_size = ggml_convrot_repack_q8_0(nullptr, nullptr, + tensor_storage.ne[0], tensor_storage.ne[1], nullptr); + if (packed_size != ggml_nbytes(dst_weight)) { + LOG_ERROR("unexpected Q8_0 size while repacking '%s'", tensor_storage.name.c_str()); + return false; + } + std::vector packed(packed_size); + ggml_convrot_repack_q8_0(reinterpret_cast(weights.data()), scale_values.data(), + tensor_storage.ne[0], tensor_storage.ne[1], packed.data()); + upload(dst_weight, packed.data(), packed.size()); + } else { + upload(dst_weight, weights.data(), weights.size()); + upload(dst_scale, scales.data(), scales.size()); + } + return true; +} + bool ModelLoader::load_float_tensor(const std::string& name, std::vector& data, int n_threads, diff --git a/src/model_loader.h b/src/model_loader.h index f7ebcf3ef1..730da6b5d5 100644 --- a/src/model_loader.h +++ b/src/model_loader.h @@ -83,6 +83,11 @@ class ModelLoader { int n_threads = 0, bool use_mmap = false); bool load_tensor(const TensorStorage& tensor_storage, ggml_tensor* dst_tensor); + // Upload the raw I8 weight and its associated F32 scale sidecar, or repack + // it into Q8_0 when dst_weight requests the decomposition representation. + bool load_comfy_int8_tensorwise(const TensorStorage& tensor_storage, + ggml_tensor* dst_weight, + ggml_tensor* dst_scale); std::vector get_tensor_names() const { std::vector names; diff --git a/src/model_manager.cpp b/src/model_manager.cpp index 825af29b95..52fd59c0b6 100644 --- a/src/model_manager.cpp +++ b/src/model_manager.cpp @@ -37,6 +37,48 @@ static std::string lora_id(const ModelManager::LoraSpec& lora) { return lora.is_high_noise ? "|high_noise|" + lora.path : lora.path; } +// `weight.convrot_scale` is an internal parameter name owned by a native +// ConvRot Linear. It maps to the scale sidecar associated with the preceding +// safetensors `weight`, never to an ordinary model `weight_scale` tensor. +static bool convrot_scale_weight_name(const std::string& name, std::string* weight_name) { + static constexpr const char* suffix = ".convrot_scale"; + static constexpr size_t suffix_len = 14; + if (!ends_with(name, suffix) || name.size() == suffix_len) { + return false; + } + if (weight_name != nullptr) { + *weight_name = name.substr(0, name.size() - suffix_len); + } + return true; +} + +static bool convrot_h256_weight_name(const std::string& name, std::string* weight_name) { + static constexpr const char* suffix = ".convrot_h256"; + static constexpr size_t suffix_len = 13; + if (!ends_with(name, suffix) || name.size() == suffix_len) { + return false; + } + if (weight_name != nullptr) { + *weight_name = name.substr(0, name.size() - suffix_len); + } + return true; +} + +static void convrot_h256(float* values) { + for (size_t stride = 1; stride < 256; stride *= 4) { + for (size_t base = 0; base < 256; base += 4 * stride) { + for (size_t i = 0; i < stride; ++i) { + float* v = values + base + i; + const float a = v[0], b = v[stride], c = v[2 * stride], d = v[3 * stride]; + v[0] = (a + b + c - d) * 0.5f; + v[stride] = (a + b - c + d) * 0.5f; + v[2 * stride] = (a - b + c + d) * 0.5f; + v[3 * stride] = (-a + b + c + d) * 0.5f; + } + } + } +} + static bool backend_supports_host_buffer(ggml_backend_t backend) { if (backend == nullptr) { return false; @@ -680,6 +722,30 @@ bool ModelManager::validate_tensor(const TensorState& state) const { } const auto& tensor_storage_map = model_loader_.get_tensor_storage_map(); + std::string convrot_weight_name; + if (convrot_h256_weight_name(state.name, &convrot_weight_name)) { + const auto weight_it = tensor_storage_map.find(convrot_weight_name); + if (weight_it == tensor_storage_map.end() || !weight_it->second.is_comfy_int8_convrot_weight() || + state.tensor->type != GGML_TYPE_F32 || state.tensor->ne[0] != 256 || state.tensor->ne[1] != 256) { + LOG_ERROR("%s ConvRot H256 parameter '%s' is invalid", state.desc.c_str(), state.name.c_str()); + return false; + } + return true; + } + if (convrot_scale_weight_name(state.name, &convrot_weight_name)) { + const auto weight_it = tensor_storage_map.find(convrot_weight_name); + if (weight_it == tensor_storage_map.end() || !weight_it->second.is_comfy_int8_convrot_weight()) { + LOG_ERROR("%s ConvRot scale '%s' has no associated marked weight", state.desc.c_str(), state.name.c_str()); + return false; + } + const TensorStorage& weight = weight_it->second; + if (state.tensor->type != GGML_TYPE_F32 || ggml_n_dims(state.tensor) != 1 || + state.tensor->ne[0] != weight.ne[1] || ggml_nbytes(state.tensor) != weight.comfy_int8_scale.nbytes) { + LOG_ERROR("%s ConvRot scale '%s' has incompatible shape or type", state.desc.c_str(), state.name.c_str()); + return false; + } + return true; + } auto ts_it = tensor_storage_map.find(state.name); if (ts_it == tensor_storage_map.end()) { LOG_ERROR("%s tensor '%s' not in model metadata", state.desc.c_str(), state.name.c_str()); @@ -747,6 +813,18 @@ bool ModelManager::can_mmap_storage(const TensorState& state) const { if (state.compute_backend == nullptr || state.params_backend == nullptr) { return false; } + std::string convrot_weight_name; + if (convrot_h256_weight_name(state.name, &convrot_weight_name)) { + return false; + } + if (convrot_scale_weight_name(state.name, &convrot_weight_name)) { + return false; + } + const auto storage_it = model_loader_.get_tensor_storage_map().find(state.name); + if (storage_it != model_loader_.get_tensor_storage_map().end() && storage_it->second.is_comfy_int8_tensorwise) { + // The I8 bytes must be uploaded together with their sidecar scale. + return false; + } return sd_backend_is_cpu(state.compute_backend) || sd_backend_is_cpu(state.params_backend) || backend_supports_host_buffer(state.compute_backend); @@ -863,6 +941,32 @@ bool ModelManager::load_tensors(const std::vector& states) { std::set loaded_names; std::mutex loaded_names_mutex; + // H256 is a small internal parameter, not a tensor in the model file. + // Initialize it after parameter buffers are allocated and exclude it from + // the loader's target-name filter. + for (const auto& pair : states_by_name) { + std::string weight_name; + if (!convrot_h256_weight_name(pair.first, &weight_name)) { + continue; + } + ggml_tensor* h = pair.second != nullptr ? pair.second->tensor : nullptr; + if (h == nullptr || h->type != GGML_TYPE_F32 || h->ne[0] != 256 || h->ne[1] != 256) { + LOG_ERROR("invalid internal ConvRot H256 tensor '%s'", pair.first.c_str()); + return false; + } + std::vector matrix(256 * 256, 0.0f); + for (size_t row = 0; row < 256; ++row) { + matrix[row * 256 + row] = 1.0f; + convrot_h256(matrix.data() + row * 256); + } + if (h->buffer != nullptr && !ggml_backend_buffer_is_host(h->buffer)) { + ggml_backend_tensor_set(h, matrix.data(), 0, matrix.size() * sizeof(float)); + } else { + memcpy(h->data, matrix.data(), matrix.size() * sizeof(float)); + } + loaded_names.insert(pair.first); + target_tensor_names.erase(pair.first); + } auto on_new_tensor_cb = [&](const TensorStorage& tensor_storage, ggml_tensor** dst_tensor) -> bool { const std::string& name = tensor_storage.name; *dst_tensor = nullptr; @@ -878,6 +982,40 @@ bool ModelManager::load_tensors(const std::vector& states) { return false; } + if (tensor_storage.is_comfy_int8_convrot_weight() && state->tensor->type == GGML_TYPE_Q8_0) { + if (!model_loader_.load_comfy_int8_tensorwise(tensor_storage, state->tensor, nullptr)) { + return false; + } + std::lock_guard lock(loaded_names_mutex); + loaded_names.insert(name); + return true; + } + + // Only the native Linear registers an I8 destination plus its private + // sidecar parameter. The explicit compatibility policy deliberately + // registers an F16 destination and must continue through load_tensor, + // where the marked bytes are reconstructed to F16. + if (tensor_storage.is_comfy_int8_convrot_weight() && state->tensor->type == GGML_TYPE_I8) { + const std::string scale_name = name + ".convrot_scale"; + const auto scale_it = states_by_name.find(scale_name); + if (scale_it == states_by_name.end() || scale_it->second == nullptr || + scale_it->second->tensor == nullptr) { + LOG_ERROR("native ConvRot tensor '%s' is missing its dedicated scale parameter", name.c_str()); + return false; + } + if (!model_loader_.load_comfy_int8_tensorwise(tensor_storage, state->tensor, scale_it->second->tensor)) { + return false; + } + { + std::lock_guard lock(loaded_names_mutex); + loaded_names.insert(name); + loaded_names.insert(scale_name); + } + // The raw uploader has already initialized both tensors. Do not + // let ModelLoader select its F16 compatibility reconstruction. + return true; + } + if (state->tensor->ne[0] != tensor_storage.ne[0] || state->tensor->ne[1] != tensor_storage.ne[1] || state->tensor->ne[2] != tensor_storage.ne[2] || diff --git a/src/stable-diffusion.cpp b/src/stable-diffusion.cpp index 2ec7069573..fdfe272026 100644 --- a/src/stable-diffusion.cpp +++ b/src/stable-diffusion.cpp @@ -3995,7 +3995,15 @@ sd_ctx_t* new_sd_ctx(const sd_ctx_params_t* sd_ctx_params) { return nullptr; } - if (!sd_ctx->sd->init(sd_ctx_params)) { + bool initialized = false; + try { + initialized = sd_ctx->sd->init(sd_ctx_params); + } catch (const std::exception& error) { + LOG_ERROR("failed to initialize Stable Diffusion context: %s", error.what()); + } catch (...) { + LOG_ERROR("failed to initialize Stable Diffusion context: unknown exception"); + } + if (!initialized) { delete sd_ctx->sd; sd_ctx->sd = nullptr; free(sd_ctx); diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 11050ef9e9..0c068e106c 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -61,3 +61,9 @@ add_executable(test-vae-name-conversion test-vae-name-conversion.cpp) target_include_directories(test-vae-name-conversion PRIVATE "${PROJECT_SOURCE_DIR}/src") target_link_libraries(test-vae-name-conversion PRIVATE stable-diffusion ${CMAKE_THREAD_LIBS_INIT}) add_test(NAME test-vae-name-conversion COMMAND test-vae-name-conversion) + +add_executable(test-safetensors-convrot test-safetensors-convrot.cpp) +target_include_directories(test-safetensors-convrot PRIVATE + "${PROJECT_SOURCE_DIR}/src") +target_link_libraries(test-safetensors-convrot PRIVATE stable-diffusion zip ${CMAKE_THREAD_LIBS_INIT}) +add_test(NAME test-safetensors-convrot COMMAND test-safetensors-convrot) diff --git a/tests/test-safetensors-convrot.cpp b/tests/test-safetensors-convrot.cpp new file mode 100644 index 0000000000..8cb17374b1 --- /dev/null +++ b/tests/test-safetensors-convrot.cpp @@ -0,0 +1,448 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "core/ggml_extend.hpp" +#include "core/util.h" +#include "model_io/binary_io.h" +#include "model_io/safetensors_io.h" +#include "model_loader.h" + +namespace { + + std::string make_header(const std::string& marker, size_t columns = 256) { + const size_t weight_bytes = 4 * columns; + return "{\"layer.weight\":{\"dtype\":\"I8\",\"shape\":[4," + std::to_string(columns) + + "],\"data_offsets\":[0," + std::to_string(weight_bytes) + "]}," + "\"layer.weight_scale\":{\"dtype\":\"F32\",\"shape\":[4,1],\"data_offsets\":[" + + std::to_string(weight_bytes) + "," + std::to_string(weight_bytes + 16) + "]}," + "\"layer.comfy_quant\":{\"dtype\":\"U8\",\"shape\":[" + + std::to_string(marker.size()) + "],\"data_offsets\":[" + + std::to_string(weight_bytes + 16) + "," + + std::to_string(weight_bytes + 16 + marker.size()) + "]}}"; + } + + void write_fixture(const std::filesystem::path& path, const std::string& marker, float scale_value = 0.5f, + size_t columns = 256) { + const std::string header = make_header(marker, columns); + std::vector weights(4 * columns, 0); + weights[0] = 2; + const float scales[] = {scale_value, scale_value, scale_value, scale_value}; + + std::ofstream file(path, std::ios::binary | std::ios::trunc); + GGML_ASSERT(file.is_open()); + model_io::write_u64(file, header.size()); + file.write(header.data(), static_cast(header.size())); + file.write(reinterpret_cast(weights.data()), static_cast(weights.size())); + file.write(reinterpret_cast(scales), sizeof(scales)); + file.write(marker.data(), static_cast(marker.size())); + GGML_ASSERT(file.good()); + } + + void write_markerless_fixture(const std::filesystem::path& path) { + const std::string header = + "{\"layer.weight\":{\"dtype\":\"I8\",\"shape\":[4,256],\"data_offsets\":[0,1024]}," + "\"layer.weight_scale\":{\"dtype\":\"F32\",\"shape\":[4,1],\"data_offsets\":[1024,1040]}}"; + const std::vector weights(1024, 0); + const float scales[] = {0.5f, 0.5f, 0.5f, 0.5f}; + std::ofstream file(path, std::ios::binary | std::ios::trunc); + GGML_ASSERT(file.is_open()); + model_io::write_u64(file, header.size()); + file.write(header.data(), static_cast(header.size())); + file.write(reinterpret_cast(weights.data()), static_cast(weights.size())); + file.write(reinterpret_cast(scales), sizeof(scales)); + GGML_ASSERT(file.good()); + } + + void write_marker_first_fixture(const std::filesystem::path& path, const std::string& marker, bool payload) { + const size_t marker_end = marker.size(); + const std::string header = + "{\"layer.comfy_quant\":{\"dtype\":\"U8\",\"shape\":[" + std::to_string(marker_end) + + "],\"data_offsets\":[0," + std::to_string(marker_end) + + "]}," + "\"layer.weight\":{\"dtype\":\"I8\",\"shape\":[4,256],\"data_offsets\":[" + + std::to_string(marker_end) + "," + std::to_string(marker_end + 1024) + + "]}," + "\"layer.weight_scale\":{\"dtype\":\"F32\",\"shape\":[4,1],\"data_offsets\":[" + + std::to_string(marker_end + 1024) + "," + std::to_string(marker_end + 1040) + "]}}"; + std::ofstream file(path, std::ios::binary | std::ios::trunc); + GGML_ASSERT(file.is_open()); + model_io::write_u64(file, header.size()); + file.write(header.data(), static_cast(header.size())); + file.write(marker.data(), static_cast(marker.size())); + if (payload) { + const std::vector weights(1024, 0); + const float scales[] = {0.5f, 0.5f, 0.5f, 0.5f}; + file.write(reinterpret_cast(weights.data()), static_cast(weights.size())); + file.write(reinterpret_cast(scales), sizeof(scales)); + } + GGML_ASSERT(file.good()); + } + + const TensorStorage& find_tensor(const ModelLoader& loader, const std::string& name) { + const auto& tensors = loader.get_tensor_storage_map(); + const auto it = tensors.find(name); + GGML_ASSERT(it != tensors.end()); + return it->second; + } + + int set_test_environment(const char* name, const char* value) { +#ifdef _WIN32 + return _putenv_s(name, value); +#else + return setenv(name, value, 1); +#endif + } + + int unset_test_environment(const char* name) { +#ifdef _WIN32 + return _putenv_s(name, ""); +#else + return unsetenv(name); +#endif + } + + ggml_backend_t init_cpu_backend() { + ggml_backend_load_all(); + return ggml_backend_init_by_type(GGML_BACKEND_DEVICE_TYPE_CPU, nullptr); + } + + class InspectableLinear : public Linear { + public: + using Linear::Linear; + bool fast_h256_enabled() const { return use_convrot_fast_h256; } + bool rotation_op_enabled() const { return use_convrot_rotation_op; } + }; + +} // namespace + +int main() { + GGML_ASSERT(select_convrot_execution_path("CUDA0", true, nullptr) == ConvRotExecutionPath::DENSE_H256); + GGML_ASSERT(select_convrot_execution_path("Vulkan0", true, nullptr) == ConvRotExecutionPath::ROTATION_OP); + GGML_ASSERT(select_convrot_execution_path("MTL0", true, nullptr) == ConvRotExecutionPath::ROTATION_OP); + GGML_ASSERT(select_convrot_execution_path("CPU", true, nullptr) == ConvRotExecutionPath::ROTATION_OP); + GGML_ASSERT(select_convrot_execution_path("unknown", false, nullptr) == ConvRotExecutionPath::NATIVE); + GGML_ASSERT(select_convrot_execution_path("CUDA0", true, "native") == ConvRotExecutionPath::NATIVE); + GGML_ASSERT(select_convrot_execution_path("CUDA0", true, "dense") == ConvRotExecutionPath::DENSE_H256); + GGML_ASSERT(select_convrot_execution_path("Vulkan0", true, "op") == ConvRotExecutionPath::ROTATION_OP); + GGML_ASSERT(select_convrot_execution_path("Vulkan0", false, "op") == ConvRotExecutionPath::NATIVE); + GGML_ASSERT(select_convrot_execution_path("CUDA0", true, "compat") == ConvRotExecutionPath::COMPAT); + bool invalid_mode_rejected = false; + try { + (void)select_convrot_execution_path("CUDA0", true, "invalid"); + } catch (const std::runtime_error&) { + invalid_mode_rejected = true; + } + GGML_ASSERT(invalid_mode_rejected); + + const std::filesystem::path path = std::filesystem::temp_directory_path() / + "stable-diffusion-convrot-test.safetensors"; + const std::string marker = + "{\"format\":\"int8_tensorwise\",\"convrot\":true,\"convrot_groupsize\":256}"; + write_fixture(path, marker); + + const std::filesystem::path sparse_path = std::filesystem::temp_directory_path() / + "stable-diffusion-convrot-sparse-test.safetensors"; + write_marker_first_fixture(sparse_path, marker, true); + std::vector full_tensors; + GGML_ASSERT(read_safetensors_file(sparse_path.string(), full_tensors)); + write_marker_first_fixture(sparse_path, marker, false); + std::vector metadata_tensors; + std::string sparse_error; + GGML_ASSERT(!read_safetensors_file(sparse_path.string(), metadata_tensors, &sparse_error)); + { + SDMetadataOnlyReadScope scope; + GGML_ASSERT(read_safetensors_file(sparse_path.string(), metadata_tensors, &sparse_error)); + GGML_ASSERT(metadata_tensors.size() == full_tensors.size()); + GGML_ASSERT(metadata_tensors[0].to_string() == full_tensors[0].to_string()); + GGML_ASSERT(metadata_tensors[0].is_comfy_int8_convrot_weight()); + } + write_marker_first_fixture(sparse_path, marker, true); + std::filesystem::resize_file(sparse_path, std::filesystem::file_size(sparse_path) - 1); + GGML_ASSERT(!read_safetensors_file(sparse_path.string(), metadata_tensors, &sparse_error)); + // Restore the header and remove only the marker bytes. + write_marker_first_fixture(sparse_path, marker, false); + std::filesystem::resize_file(sparse_path, std::filesystem::file_size(sparse_path) - marker.size()); + { + SDMetadataOnlyReadScope scope; + GGML_ASSERT(!read_safetensors_file(sparse_path.string(), metadata_tensors, &sparse_error)); + GGML_ASSERT(sparse_error.find("marker") != std::string::npos); + } + std::error_code sparse_remove_error; + std::filesystem::remove(sparse_path, sparse_remove_error); + + ModelLoader loader; + GGML_ASSERT(loader.init_from_file(path.string())); + const TensorStorage& weight = find_tensor(loader, "layer.weight"); + GGML_ASSERT(weight.type == GGML_TYPE_I8); + GGML_ASSERT(weight.expected_type == GGML_TYPE_F16); + GGML_ASSERT(weight.is_comfy_int8_tensorwise); + GGML_ASSERT(weight.comfy_int8_convrot); + GGML_ASSERT(weight.comfy_int8_group_size == 256); + GGML_ASSERT(weight.has_comfy_int8_scale()); + GGML_ASSERT(weight.comfy_int8_scale.name == "layer.weight_scale"); + GGML_ASSERT(weight.comfy_int8_scale.type == GGML_TYPE_F32); + GGML_ASSERT(weight.comfy_int8_scale.n_dims == 2); + GGML_ASSERT(weight.comfy_int8_scale.ne[0] == 1); + GGML_ASSERT(weight.comfy_int8_scale.ne[1] == 4); + GGML_ASSERT(weight.comfy_int8_scale.nbytes == 4 * sizeof(float)); + GGML_ASSERT(weight.is_comfy_int8_convrot_weight()); + GGML_ASSERT(loader.get_tensor_storage_map().find("layer.weight_scale") == loader.get_tensor_storage_map().end()); + + ggml_init_params params = {4096, nullptr, false}; + ggml_context* ctx = ggml_init(params); + GGML_ASSERT(ctx != nullptr); + ggml_tensor* decoded = ggml_new_tensor_2d(ctx, GGML_TYPE_F16, 256, 4); + GGML_ASSERT(loader.load_tensor(weight, decoded)); + + std::vector values(4 * 256); + ggml_fp16_to_fp32_row(static_cast(decoded->data), values.data(), values.size()); + float energy = 0.f; + for (float value : values) { + energy += value * value; + } + // The normalized regular Hadamard matrix is orthogonal: ConvRot inversion + // preserves the squared norm of the dequantized first row. + GGML_ASSERT(std::fabs(energy - 1.f) < 0.002f); + ggml_free(ctx); + + ggml_init_params native_params = {ggml_tensor_overhead() * 2 + 1024 + 4 * sizeof(float) + 4096, nullptr, false}; + ggml_context* native_ctx = ggml_init(native_params); + GGML_ASSERT(native_ctx != nullptr); + ggml_tensor* raw_weight = ggml_new_tensor_2d(native_ctx, GGML_TYPE_I8, 256, 4); + ggml_tensor* raw_scale = ggml_new_tensor_1d(native_ctx, GGML_TYPE_F32, 4); + GGML_ASSERT(loader.load_comfy_int8_tensorwise(weight, raw_weight, raw_scale)); + GGML_ASSERT(static_cast(raw_weight->data)[0] == 2); + GGML_ASSERT(static_cast(raw_weight->data)[1] == 0); + GGML_ASSERT(static_cast(raw_scale->data)[0] == 0.5f); + GGML_ASSERT(static_cast(raw_scale->data)[3] == 0.5f); + ggml_free(native_ctx); + + const size_t q8_bytes = ggml_row_size(GGML_TYPE_Q8_0, 256) * 4; + ggml_init_params q8_params = {ggml_tensor_overhead() + q8_bytes + 4096, nullptr, false}; + ggml_context* q8_ctx = ggml_init(q8_params); + GGML_ASSERT(q8_ctx != nullptr); + ggml_tensor* q8_weight = ggml_new_tensor_2d(q8_ctx, GGML_TYPE_Q8_0, 256, 4); + GGML_ASSERT(loader.load_comfy_int8_tensorwise(weight, q8_weight, nullptr)); + ggml_fp16_t packed_scale; + memcpy(&packed_scale, q8_weight->data, sizeof(packed_scale)); + GGML_ASSERT(ggml_fp16_to_fp32(packed_scale) == 0.5f); + GGML_ASSERT(static_cast(q8_weight->data)[sizeof(packed_scale)] == 2); + GGML_ASSERT(static_cast(q8_weight->data)[sizeof(packed_scale) + 1] == 0); + ggml_free(q8_ctx); + + // FP16-subnormal scales are reported but remain usable; scales that would + // underflow to zero must not silently produce an all-zero Q8_0 row. + for (float scale : {1e-6f, 1e-9f}) { + write_fixture(path, marker, scale); + ModelLoader range_loader; + GGML_ASSERT(range_loader.init_from_file(path.string())); + const TensorStorage& range_weight = find_tensor(range_loader, "layer.weight"); + ggml_init_params range_params = {ggml_tensor_overhead() + q8_bytes + 4096, nullptr, false}; + ggml_context* range_ctx = ggml_init(range_params); + GGML_ASSERT(range_ctx != nullptr); + ggml_tensor* range_q8 = ggml_new_tensor_2d(range_ctx, GGML_TYPE_Q8_0, 256, 4); + GGML_ASSERT(range_loader.load_comfy_int8_tensorwise(range_weight, range_q8, nullptr) == (scale == 1e-6f)); + if (scale == 1e-6f) { + ggml_fp16_t range_half; + memcpy(&range_half, range_q8->data, sizeof(range_half)); + GGML_ASSERT(ggml_fp16_to_fp32(range_half) > 0.0f); + } + ggml_free(range_ctx); + } + write_fixture(path, marker); + + ggml_backend_t cpu_backend = init_cpu_backend(); + GGML_ASSERT(cpu_backend != nullptr); + GGML_ASSERT(set_test_environment("SD_CONVROT_MODE", "native") == 0); + const auto native_selection = select_convrot_tensor_storage(cpu_backend, + loader.get_tensor_storage_map(), + "ConvRot loader test"); + GGML_ASSERT(native_selection.at("layer.weight").comfy_int8_native_enabled); + GGML_ASSERT(!native_selection.at("layer.weight").comfy_int8_q8_decomp_enabled); + GGML_ASSERT(ggml_backend_supports_convrot_op(cpu_backend)); + GGML_ASSERT(unset_test_environment("SD_CONVROT_MODE") == 0); + + const auto automatic_selection = select_convrot_tensor_storage(cpu_backend, + loader.get_tensor_storage_map(), + "ConvRot loader test"); + GGML_ASSERT(automatic_selection.at("layer.weight").comfy_int8_q8_decomp_enabled); + GGML_ASSERT(automatic_selection.at("layer.weight").comfy_int8_convrot_op_enabled); + + GGML_ASSERT(set_test_environment("SD_CONVROT_MODE", "q8") == 0); + const auto q8_selection = select_convrot_tensor_storage(cpu_backend, + loader.get_tensor_storage_map(), + "ConvRot loader test"); + GGML_ASSERT(q8_selection.at("layer.weight").comfy_int8_native_enabled); + GGML_ASSERT(q8_selection.at("layer.weight").comfy_int8_q8_decomp_enabled); + GGML_ASSERT(q8_selection.at("layer.weight").comfy_int8_convrot_op_enabled); + GGML_ASSERT(unset_test_environment("SD_CONVROT_MODE") == 0); + + ggml_init_params op_graph_params = {1024 * 1024, nullptr, true}; + ggml_context* op_graph_ctx = ggml_init(op_graph_params); + GGML_ASSERT(op_graph_ctx != nullptr); + InspectableLinear op_linear(256, 4, false); + op_linear.init(op_graph_ctx, automatic_selection, "layer"); + GGML_ASSERT(op_linear.rotation_op_enabled()); + ggml_tensor* op_input = ggml_new_tensor_2d(op_graph_ctx, GGML_TYPE_F32, 256, 2); + GGMLRunnerContext op_runner_ctx; + op_runner_ctx.ggml_ctx = op_graph_ctx; + ggml_tensor* op_output = op_linear.forward(&op_runner_ctx, op_input); + GGML_ASSERT(op_output != nullptr && op_output->op == GGML_OP_MUL_MAT); + GGML_ASSERT(op_output->src[1] != nullptr && op_output->src[1]->op == GGML_OP_CONVROT); + ggml_free(op_graph_ctx); + + GGML_ASSERT(set_test_environment("SD_CONVROT_MODE", "dense") == 0); + const auto dense_selection = select_convrot_tensor_storage(cpu_backend, + loader.get_tensor_storage_map(), + "ConvRot loader test"); + GGML_ASSERT(dense_selection.at("layer.weight").comfy_int8_q8_decomp_enabled); + GGML_ASSERT(!dense_selection.at("layer.weight").comfy_int8_convrot_op_enabled); + GGML_ASSERT(unset_test_environment("SD_CONVROT_MODE") == 0); + + auto h256_hint_for_mode = [&](const char* mode) { + if (mode == nullptr) { + GGML_ASSERT(unset_test_environment("SD_CONVROT_H256_MODE") == 0); + } else { + GGML_ASSERT(set_test_environment("SD_CONVROT_H256_MODE", mode) == 0); + } + ggml_init_params graph_params = {1024 * 1024, nullptr, true}; + ggml_context* graph_ctx = ggml_init(graph_params); + GGML_ASSERT(graph_ctx != nullptr); + InspectableLinear linear(256, 4, false); + linear.init(graph_ctx, dense_selection, "layer"); + GGML_ASSERT(linear.fast_h256_enabled() == (mode != nullptr && strcmp(mode, "fast") == 0)); + ggml_tensor* input = ggml_new_tensor_2d(graph_ctx, GGML_TYPE_F32, 256, 2); + GGMLRunnerContext runner_ctx; + runner_ctx.ggml_ctx = graph_ctx; + ggml_tensor* output = linear.forward(&runner_ctx, input); + GGML_ASSERT(output != nullptr && output->op == GGML_OP_MUL_MAT); + ggml_tensor* rotated = output->src[1]; + if (rotated->op != GGML_OP_MUL_MAT) { + rotated = rotated->src[0]; + } + GGML_ASSERT(rotated != nullptr && rotated->op == GGML_OP_MUL_MAT); + int32_t hint = GGML_HINT_NONE; + memcpy(&hint, rotated->op_params + sizeof(int32_t), sizeof(hint)); + ggml_free(graph_ctx); + return hint; + }; + GGML_ASSERT(h256_hint_for_mode(nullptr) == GGML_HINT_NONE); + GGML_ASSERT(h256_hint_for_mode("dense") == GGML_HINT_NONE); + // The accessor assertion in h256_hint_for_mode verifies that "fast" + // selects the opt-in path. The backend-specific hint itself is covered by + // test-mul-mat-convrot-h256-hint. + (void)h256_hint_for_mode("fast"); + GGML_ASSERT(unset_test_environment("SD_CONVROT_H256_MODE") == 0); + + // A runner for another component must not inherit this component's + // ConvRot requirement when both live in the shared storage map. + const auto other_component_selection = select_convrot_tensor_storage(nullptr, + loader.get_tensor_storage_map(), + "unrelated component", + "other_component."); + GGML_ASSERT(!other_component_selection.at("layer.weight").comfy_int8_native_enabled); + + bool unsupported_backend_rejected = false; + try { + (void)select_convrot_tensor_storage(nullptr, loader.get_tensor_storage_map(), "ConvRot loader test"); + } catch (const std::runtime_error&) { + unsupported_backend_rejected = true; + } + GGML_ASSERT(unsupported_backend_rejected); + + GGML_ASSERT(set_test_environment("SD_CONVROT_MODE", "compat") == 0); + const auto compatibility_selection = select_convrot_tensor_storage(cpu_backend, + loader.get_tensor_storage_map(), + "ConvRot loader test"); + GGML_ASSERT(!compatibility_selection.at("layer.weight").comfy_int8_native_enabled); + GGML_ASSERT(unset_test_environment("SD_CONVROT_MODE") == 0); + ggml_backend_free(cpu_backend); + + // Tensorwise Int8 without ConvRot can have a final row chunk shorter than + // the loader's 4096-element conversion buffer. + write_fixture(path, "{\"format\":\"int8_tensorwise\"}", 0.5f, 5000); + ModelLoader wide_loader; + GGML_ASSERT(wide_loader.init_from_file(path.string())); + const TensorStorage& wide_weight = find_tensor(wide_loader, "layer.weight"); + ggml_init_params wide_params = {ggml_tensor_overhead() + 4 * 5000 * sizeof(ggml_fp16_t) + 4096, + nullptr, false}; + ggml_context* wide_ctx = ggml_init(wide_params); + GGML_ASSERT(wide_ctx != nullptr); + ggml_tensor* wide_decoded = ggml_new_tensor_2d(wide_ctx, GGML_TYPE_F16, 5000, 4); + GGML_ASSERT(wide_loader.load_tensor(wide_weight, wide_decoded)); + GGML_ASSERT(ggml_fp16_to_fp32(static_cast(wide_decoded->data)[0]) == 1.0f); + GGML_ASSERT(ggml_fp16_to_fp32(static_cast(wide_decoded->data)[4999]) == 0.0f); + ggml_free(wide_ctx); + + write_fixture(path, marker, std::numeric_limits::quiet_NaN()); + ModelLoader invalid_scale_loader; + GGML_ASSERT(invalid_scale_loader.init_from_file(path.string())); + const TensorStorage& invalid_scale_weight = find_tensor(invalid_scale_loader, "layer.weight"); + ggml_init_params invalid_scale_params = {ggml_tensor_overhead() * 2 + 1024 + 4 * sizeof(float) + 4096, nullptr, false}; + ggml_context* invalid_scale_ctx = ggml_init(invalid_scale_params); + GGML_ASSERT(invalid_scale_ctx != nullptr); + ggml_tensor* invalid_raw_weight = ggml_new_tensor_2d(invalid_scale_ctx, GGML_TYPE_I8, 256, 4); + ggml_tensor* invalid_raw_scale = ggml_new_tensor_1d(invalid_scale_ctx, GGML_TYPE_F32, 4); + GGML_ASSERT(!invalid_scale_loader.load_comfy_int8_tensorwise(invalid_scale_weight, invalid_raw_weight, invalid_raw_scale)); + ggml_free(invalid_scale_ctx); + + const std::string unsupported_group = + "{\"format\":\"int8_tensorwise\",\"convrot\":true,\"convrot_groupsize\":16}"; + write_fixture(path, unsupported_group); + ModelLoader invalid_loader; + GGML_ASSERT(!invalid_loader.init_from_file(path.string())); + + write_fixture(path, "not-json"); + ModelLoader malformed_loader; + GGML_ASSERT(!malformed_loader.init_from_file(path.string())); + + write_markerless_fixture(path); + std::vector markerless_tensors; + std::string markerless_error; + GGML_ASSERT(!read_safetensors_file(path.string(), markerless_tensors, &markerless_error)); + GGML_ASSERT(markerless_error.find("without a ComfyUI Int8 marker") != std::string::npos); + + write_fixture(path, ""); + ModelLoader empty_marker_loader; + GGML_ASSERT(!empty_marker_loader.init_from_file(path.string())); + + if (const char* real_model_path = std::getenv("CONVROT_MODEL_PATH")) { + ModelLoader real_loader; + GGML_ASSERT(real_loader.init_from_file(real_model_path)); + const TensorStorage* real_weight = nullptr; + for (const auto& [_, tensor] : real_loader.get_tensor_storage_map()) { + if (tensor.is_comfy_int8_tensorwise) { + real_weight = &tensor; + break; + } + } + GGML_ASSERT(real_weight != nullptr); + const size_t elements = static_cast(real_weight->ne[0]) * + static_cast(real_weight->ne[1]); + ggml_init_params real_params = {ggml_tensor_overhead() + elements * sizeof(ggml_fp16_t) + 4096, + nullptr, + false}; + ggml_context* real_ctx = ggml_init(real_params); + GGML_ASSERT(real_ctx != nullptr); + ggml_tensor* real_decoded = ggml_new_tensor_2d(real_ctx, + GGML_TYPE_F16, + real_weight->ne[0], + real_weight->ne[1]); + GGML_ASSERT(real_loader.load_tensor(*real_weight, real_decoded)); + const float first_value = ggml_fp16_to_fp32(static_cast(real_decoded->data)[0]); + GGML_ASSERT(std::isfinite(first_value)); + ggml_free(real_ctx); + } + + std::error_code ec; + std::filesystem::remove(path, ec); + return 0; +} diff --git a/tests/test_safetensors_metadata.cpp b/tests/test_safetensors_metadata.cpp index 3782ab4714..3218260e7b 100644 --- a/tests/test_safetensors_metadata.cpp +++ b/tests/test_safetensors_metadata.cpp @@ -231,7 +231,8 @@ namespace safetensors_metadata_test { bool caught = false; try { SDMetadataOnlyReadScope scope; - read_file(path); + GGML_ASSERT(!read_file(path)); + throw std::runtime_error("metadata scope cleanup"); } catch (const std::exception&) { caught = true; } From f8fa0eef63e6942bdaa1a6c3d8ff4961505d79ea Mon Sep 17 00:00:00 2001 From: aegioscy Date: Tue, 29 Sep 2026 09:01:32 +0000 Subject: [PATCH 2/5] fix(convrot): address PR 44 review findings --- src/core/ggml_extend.hpp | 16 ++++++-- src/model/adapter/lora.hpp | 6 +++ src/model_loader.cpp | 4 ++ src/model_manager.cpp | 12 ++++++ src/model_manager.h | 1 + src/stable-diffusion.cpp | 6 +++ tests/test-lora-validation.cpp | 31 +++++++++++++++ tests/test-safetensors-convrot.cpp | 62 ++++++++++++++++++++++++++++++ 8 files changed, 135 insertions(+), 3 deletions(-) diff --git a/src/core/ggml_extend.hpp b/src/core/ggml_extend.hpp index 898af09003..1d02a65bf1 100644 --- a/src/core/ggml_extend.hpp +++ b/src/core/ggml_extend.hpp @@ -4150,7 +4150,7 @@ class Linear : public UnaryBlock { use_convrot_rotation_op = false; use_convrot_fast_h256 = false; const auto storage_it = tensor_storage_map.find(prefix + "weight"); - if (storage_it != tensor_storage_map.end() && storage_it->second.is_comfy_int8_convrot_weight() && + if (storage_it != tensor_storage_map.end() && storage_it->second.is_comfy_int8_convrot_weight() && !force_f32 && storage_it->second.comfy_int8_native_enabled) { has_convrot_weight = true; if (storage_it->second.comfy_int8_q8_decomp_enabled) { @@ -4224,14 +4224,15 @@ class Linear : public UnaryBlock { ggml_tensor* out = nullptr; if (has_convrot_weight) { if (use_convrot_q8_decomp) { + ggml_tensor* scaled_x = scale != 1.f ? ggml_ext_scale(ctx->ggml_ctx, x, scale) : x; if (use_convrot_rotation_op) { - ggml_tensor* rotated = ggml_convrot(ctx->ggml_ctx, x, 256); + ggml_tensor* rotated = ggml_convrot(ctx->ggml_ctx, scaled_x, 256); ggml_set_name(rotated, (prefix + "trace.convrot.post_h256").c_str()); out = ggml_mul_mat(ctx->ggml_ctx, w, rotated); } else { // H256 is symmetric: (H*w_row).x == w_row.(H*x). Rotate // each 256-wide block before the stock Q8_0 matmul. - ggml_tensor* contiguous = ggml_is_contiguous(x) ? x : ggml_cont(ctx->ggml_ctx, x); + ggml_tensor* contiguous = ggml_is_contiguous(scaled_x) ? scaled_x : ggml_cont(ctx->ggml_ctx, scaled_x); ggml_tensor* blocks = ggml_reshape_2d(ctx->ggml_ctx, contiguous, 256, ggml_nelements(contiguous) / 256); ggml_set_name(blocks, (prefix + "trace.convrot.pre_h256").c_str()); @@ -4243,7 +4244,13 @@ class Linear : public UnaryBlock { rotated = ggml_reshape_4d(ctx->ggml_ctx, rotated, x->ne[0], x->ne[1], x->ne[2], x->ne[3]); out = ggml_mul_mat(ctx->ggml_ctx, w, rotated); } + if (force_prec_f32) { + ggml_mul_mat_set_prec(out, GGML_PREC_F32); + } ggml_set_name(out, (prefix + "trace.convrot.post_q8_gemm").c_str()); + if (scale != 1.f) { + out = ggml_ext_scale(ctx->ggml_ctx, out, 1.f / scale); + } } else { // ConvRot weights and their tensor-wise scales remain compact // at rest. The operator owns the scale semantics. @@ -4254,6 +4261,9 @@ class Linear : public UnaryBlock { } } if (b != nullptr) { + if (ctx->weight_adapter) { + b = ctx->weight_adapter->patch_weight(ctx->ggml_ctx, ctx->backend, b, prefix + "bias"); + } out = ggml_add_inplace(ctx->ggml_ctx, out, b); } if (ctx->weight_adapter) { diff --git a/src/model/adapter/lora.hpp b/src/model/adapter/lora.hpp index b21c7e61b0..fd6790d2b0 100644 --- a/src/model/adapter/lora.hpp +++ b/src/model/adapter/lora.hpp @@ -1313,6 +1313,12 @@ struct MultiLoraAdapter : public WeightAdapter { WeightAdapter::ForwardParams forward_params) override { ggml_tensor* delta = nullptr; for (auto& lora_model : lora_models) { + if (ggml_tensor* diff = lora_model->get_weight_diff(prefix + "weight", backend, ctx, w, false)) { + ggml_tensor* current = ggml_ext_linear(ctx, x, diff, nullptr, + forward_params.linear.force_prec_f32, + forward_params.linear.scale); + delta = delta == nullptr ? current : ggml_add_inplace(ctx, delta, current); + } ggml_tensor* current = lora_model->get_out_diff(ctx, backend, x, w, forward_params, prefix + "weight"); if (current != nullptr) { delta = delta == nullptr ? current : ggml_add_inplace(ctx, delta, current); diff --git a/src/model_loader.cpp b/src/model_loader.cpp index 1330fd4416..058b5e9397 100644 --- a/src/model_loader.cpp +++ b/src/model_loader.cpp @@ -963,6 +963,10 @@ void ModelLoader::set_wtype_override(ggml_type wtype, std::string tensor_type_ru if (!tensor_should_be_converted(tensor_storage, dst_type)) { continue; } + if (tensor_storage.is_comfy_int8_convrot_weight() && ggml_is_quantized(dst_type)) { + LOG_WARN("ignoring quantized weight-type override for ConvRot tensor '%s'", name.c_str()); + continue; + } tensor_storage.expected_type = dst_type; } } diff --git a/src/model_manager.cpp b/src/model_manager.cpp index 52fd59c0b6..6c273b0495 100644 --- a/src/model_manager.cpp +++ b/src/model_manager.cpp @@ -590,6 +590,18 @@ bool ModelManager::apply_loras_to_params(const std::vector& states state->applied_lora_epoch = current_lora_epoch_; continue; } + const auto& storage_map = model_loader_.get_tensor_storage_map(); + const auto storage_it = storage_map.find(state->name); + if (storage_it != storage_map.end() && storage_it->second.is_comfy_int8_convrot_weight() && + (state->tensor->type == GGML_TYPE_I8 || state->tensor->type == GGML_TYPE_Q8_0)) { + if (!warned_convrot_lora_skip_) { + LOG_WARN("model manager skipping immediate LoRA application to ConvRot I8/Q8_0 weights " + "(use --lora-apply-mode at_runtime)"); + warned_convrot_lora_skip_ = true; + } + state->applied_lora_epoch = current_lora_epoch_; + continue; + } if (state->tensor->data == nullptr) { LOG_ERROR("model manager lora target tensor '%s' is not prepared", state->name.c_str()); return false; diff --git a/src/model_manager.h b/src/model_manager.h index 3dc6338645..0a75f18075 100644 --- a/src/model_manager.h +++ b/src/model_manager.h @@ -68,6 +68,7 @@ class ModelManager : public RunnerWeightManager { std::vector> compute_staging_blocks_; std::map split_buffer_types_; bool warned_split_lora_skip_ = false; + bool warned_convrot_lora_skip_ = false; std::set common_ignore_tensors_; std::vector loras_; SDVersion lora_version_ = VERSION_COUNT; diff --git a/src/stable-diffusion.cpp b/src/stable-diffusion.cpp index fdfe272026..3ff526069c 100644 --- a/src/stable-diffusion.cpp +++ b/src/stable-diffusion.cpp @@ -1023,6 +1023,12 @@ class StableDiffusionGGML { break; } } + for (const auto& [name, storage] : model_loader.get_tensor_storage_map()) { + if (storage.is_comfy_int8_convrot_weight()) { + have_quantized_weight = true; + break; + } + } // Avoid full-model LoRA merge buffers on constrained setups. const bool params_offloaded = params_backend_for(SDBackendModule::DIFFUSION) != backend_for(SDBackendModule::DIFFUSION); const bool streaming_constrained = stream_layers || params_offloaded; diff --git a/tests/test-lora-validation.cpp b/tests/test-lora-validation.cpp index 1233f29c4e..7cb0b92167 100644 --- a/tests/test-lora-validation.cpp +++ b/tests/test-lora-validation.cpp @@ -100,6 +100,37 @@ int main() { ggml_new_tensor_2d(ctx, GGML_TYPE_F32, 2, 6); GGML_ASSERT(loha.count_compatible_model_tensors(model_tensors) == 1); } + { + ggml_init_params graph_init = {1024 * 1024, nullptr, true}; + ggml_context* graph_ctx = ggml_init(graph_init); + GGML_ASSERT(graph_ctx != nullptr); + ggml_tensor* one = ggml_new_tensor_1d(graph_ctx, GGML_TYPE_F32, 1); + ggml_set_name(one, "ggml_runner_build_in_tensor:one"); + ggml_tensor* zero = ggml_new_tensor_1d(graph_ctx, GGML_TYPE_I32, 1); + ggml_set_name(zero, "ggml_runner_build_in_tensor:zero_int"); + ggml_tensor* weight = ggml_new_tensor_2d(graph_ctx, GGML_TYPE_F32, 4, 6); + ggml_tensor* input = ggml_new_tensor_2d(graph_ctx, GGML_TYPE_F32, 4, 2); + for (const char* kind : {"diff", "loha"}) { + auto model = std::make_shared(kind, cpu, cpu, missing_path); + const std::string prefix = "lora.model.diffusion_model.test.weight."; + if (std::string(kind) == "diff") { + model->lora_tensors[prefix + "diff"] = ggml_new_tensor_2d(graph_ctx, GGML_TYPE_F32, 4, 6); + } else { + model->lora_tensors[prefix + "hada_w1_b"] = ggml_new_tensor_2d(graph_ctx, GGML_TYPE_F32, 4, 2); + model->lora_tensors[prefix + "hada_w1_a"] = ggml_new_tensor_2d(graph_ctx, GGML_TYPE_F32, 2, 6); + model->lora_tensors[prefix + "hada_w2_b"] = ggml_new_tensor_2d(graph_ctx, GGML_TYPE_F32, 4, 2); + model->lora_tensors[prefix + "hada_w2_a"] = ggml_new_tensor_2d(graph_ctx, GGML_TYPE_F32, 2, 6); + } + MultiLoraAdapter adapter({model}); + WeightAdapter::ForwardParams forward_params; + forward_params.op_type = WeightAdapter::ForwardParams::op_type_t::OP_LINEAR; + ggml_tensor* delta = adapter.lora_output_delta(graph_ctx, cpu, input, weight, + "model.diffusion_model.test.", forward_params); + GGML_ASSERT(delta != nullptr && delta->ne[0] == 6 && delta->ne[1] == 2); + GGML_ASSERT(!model->applied_lora_tensors.empty()); + } + ggml_free(graph_ctx); + } { LoraModel lokr("lokr", cpu, cpu, missing_path); lokr.lora_tensors diff --git a/tests/test-safetensors-convrot.cpp b/tests/test-safetensors-convrot.cpp index 8cb17374b1..edea2dcecb 100644 --- a/tests/test-safetensors-convrot.cpp +++ b/tests/test-safetensors-convrot.cpp @@ -119,6 +119,7 @@ namespace { using Linear::Linear; bool fast_h256_enabled() const { return use_convrot_fast_h256; } bool rotation_op_enabled() const { return use_convrot_rotation_op; } + ggml_type weight_type() const { return params.at("weight")->type; } }; } // namespace @@ -195,6 +196,8 @@ int main() { GGML_ASSERT(weight.comfy_int8_scale.nbytes == 4 * sizeof(float)); GGML_ASSERT(weight.is_comfy_int8_convrot_weight()); GGML_ASSERT(loader.get_tensor_storage_map().find("layer.weight_scale") == loader.get_tensor_storage_map().end()); + loader.set_wtype_override(GGML_TYPE_Q8_0, ""); + GGML_ASSERT(find_tensor(loader, "layer.weight").expected_type == GGML_TYPE_F16); ggml_init_params params = {4096, nullptr, false}; ggml_context* ctx = ggml_init(params); @@ -299,6 +302,34 @@ int main() { GGML_ASSERT(op_output->src[1] != nullptr && op_output->src[1]->op == GGML_OP_CONVROT); ggml_free(op_graph_ctx); + ggml_init_params precise_graph_params = {1024 * 1024, nullptr, true}; + ggml_context* precise_graph_ctx = ggml_init(precise_graph_params); + GGML_ASSERT(precise_graph_ctx != nullptr); + InspectableLinear precise_linear(256, 4, false, false, true, 1.f / 128.f); + precise_linear.init(precise_graph_ctx, automatic_selection, "layer"); + ggml_tensor* precise_input = ggml_new_tensor_2d(precise_graph_ctx, GGML_TYPE_F32, 256, 2); + GGMLRunnerContext precise_runner_ctx; + precise_runner_ctx.ggml_ctx = precise_graph_ctx; + ggml_tensor* precise_output = precise_linear.forward(&precise_runner_ctx, precise_input); + GGML_ASSERT(precise_output->op == GGML_OP_SCALE); + ggml_tensor* precise_matmul = precise_output->src[0]; + GGML_ASSERT(precise_matmul != nullptr && precise_matmul->op == GGML_OP_MUL_MAT); + int32_t precision = GGML_PREC_DEFAULT; + memcpy(&precision, precise_matmul->op_params, sizeof(precision)); + GGML_ASSERT(precision == GGML_PREC_F32); + GGML_ASSERT(precise_matmul->src[1]->op == GGML_OP_CONVROT); + GGML_ASSERT(precise_matmul->src[1]->src[0]->op == GGML_OP_SCALE); + GGML_ASSERT(precise_matmul->src[1]->src[0]->src[0] == precise_input); + ggml_free(precise_graph_ctx); + + ggml_init_params forced_graph_params = {1024 * 1024, nullptr, true}; + ggml_context* forced_graph_ctx = ggml_init(forced_graph_params); + GGML_ASSERT(forced_graph_ctx != nullptr); + InspectableLinear forced_linear(256, 4, false, true); + forced_linear.init(forced_graph_ctx, automatic_selection, "layer"); + GGML_ASSERT(forced_linear.weight_type() == GGML_TYPE_F32); + ggml_free(forced_graph_ctx); + GGML_ASSERT(set_test_environment("SD_CONVROT_MODE", "dense") == 0); const auto dense_selection = select_convrot_tensor_storage(cpu_backend, loader.get_tensor_storage_map(), @@ -307,6 +338,30 @@ int main() { GGML_ASSERT(!dense_selection.at("layer.weight").comfy_int8_convrot_op_enabled); GGML_ASSERT(unset_test_environment("SD_CONVROT_MODE") == 0); + ggml_init_params dense_precise_params = {1024 * 1024, nullptr, true}; + ggml_context* dense_precise_ctx = ggml_init(dense_precise_params); + GGML_ASSERT(dense_precise_ctx != nullptr); + InspectableLinear dense_precise_linear(256, 4, false, false, true, 1.f / 128.f); + dense_precise_linear.init(dense_precise_ctx, dense_selection, "layer"); + ggml_tensor* dense_precise_input = ggml_new_tensor_2d(dense_precise_ctx, GGML_TYPE_F32, 256, 2); + GGMLRunnerContext dense_precise_runner; + dense_precise_runner.ggml_ctx = dense_precise_ctx; + ggml_tensor* dense_precise_output = dense_precise_linear.forward(&dense_precise_runner, dense_precise_input); + GGML_ASSERT(dense_precise_output->op == GGML_OP_SCALE); + ggml_tensor* dense_precise_matmul = dense_precise_output->src[0]; + GGML_ASSERT(dense_precise_matmul->op == GGML_OP_MUL_MAT); + int32_t dense_precision = GGML_PREC_DEFAULT; + memcpy(&dense_precision, dense_precise_matmul->op_params, sizeof(dense_precision)); + GGML_ASSERT(dense_precision == GGML_PREC_F32); + ggml_tensor* dense_rotated = dense_precise_matmul->src[1]; + GGML_ASSERT(dense_rotated->op == GGML_OP_RESHAPE); + GGML_ASSERT(dense_rotated->src[0]->op == GGML_OP_MUL_MAT); + ggml_tensor* dense_blocks = dense_rotated->src[0]->src[1]; + GGML_ASSERT(dense_blocks->op == GGML_OP_RESHAPE); + GGML_ASSERT(dense_blocks->src[0]->op == GGML_OP_SCALE); + GGML_ASSERT(dense_blocks->src[0]->src[0] == dense_precise_input); + ggml_free(dense_precise_ctx); + auto h256_hint_for_mode = [&](const char* mode) { if (mode == nullptr) { GGML_ASSERT(unset_test_environment("SD_CONVROT_H256_MODE") == 0); @@ -363,6 +418,13 @@ int main() { loader.get_tensor_storage_map(), "ConvRot loader test"); GGML_ASSERT(!compatibility_selection.at("layer.weight").comfy_int8_native_enabled); + ggml_init_params compat_graph_params = {1024 * 1024, nullptr, true}; + ggml_context* compat_graph_ctx = ggml_init(compat_graph_params); + GGML_ASSERT(compat_graph_ctx != nullptr); + InspectableLinear compat_linear(256, 4, false); + compat_linear.init(compat_graph_ctx, compatibility_selection, "layer"); + GGML_ASSERT(compat_linear.weight_type() == GGML_TYPE_F16); + ggml_free(compat_graph_ctx); GGML_ASSERT(unset_test_environment("SD_CONVROT_MODE") == 0); ggml_backend_free(cpu_backend); From 6dee045b3642685c5e208ee88d6d3df3c891a7e5 Mon Sep 17 00:00:00 2001 From: aegioscy Date: Tue, 29 Sep 2026 11:25:54 +0000 Subject: [PATCH 3/5] fix(convrot): address tensorwise override and value review --- src/model_loader.cpp | 4 +- tests/test-safetensors-convrot.cpp | 82 ++++++++++++++++++++++++++++++ 2 files changed, 84 insertions(+), 2 deletions(-) diff --git a/src/model_loader.cpp b/src/model_loader.cpp index 058b5e9397..0f289bf4cc 100644 --- a/src/model_loader.cpp +++ b/src/model_loader.cpp @@ -963,8 +963,8 @@ void ModelLoader::set_wtype_override(ggml_type wtype, std::string tensor_type_ru if (!tensor_should_be_converted(tensor_storage, dst_type)) { continue; } - if (tensor_storage.is_comfy_int8_convrot_weight() && ggml_is_quantized(dst_type)) { - LOG_WARN("ignoring quantized weight-type override for ConvRot tensor '%s'", name.c_str()); + if (tensor_storage.is_comfy_int8_tensorwise && ggml_is_quantized(dst_type)) { + LOG_WARN("ignoring quantized weight-type override for ComfyUI Int8 tensor '%s'", name.c_str()); continue; } tensor_storage.expected_type = dst_type; diff --git a/tests/test-safetensors-convrot.cpp b/tests/test-safetensors-convrot.cpp index edea2dcecb..6f9cedc286 100644 --- a/tests/test-safetensors-convrot.cpp +++ b/tests/test-safetensors-convrot.cpp @@ -11,6 +11,7 @@ #include "core/ggml_extend.hpp" #include "core/util.h" +#include "ggml-cpu.h" #include "model_io/binary_io.h" #include "model_io/safetensors_io.h" #include "model_loader.h" @@ -120,6 +121,7 @@ namespace { bool fast_h256_enabled() const { return use_convrot_fast_h256; } bool rotation_op_enabled() const { return use_convrot_rotation_op; } ggml_type weight_type() const { return params.at("weight")->type; } + ggml_tensor* parameter(const std::string& name) const { return params.at(name); } }; } // namespace @@ -425,6 +427,70 @@ int main() { compat_linear.init(compat_graph_ctx, compatibility_selection, "layer"); GGML_ASSERT(compat_linear.weight_type() == GGML_TYPE_F16); ggml_free(compat_graph_ctx); + + const std::vector bias_values = {0.25f, -0.5f, 1.0f, -1.25f}; + std::vector input_values(256 * 2); + for (size_t i = 0; i < input_values.size(); ++i) { + input_values[i] = 0.5f + static_cast(static_cast((i * 17) % 29) - 14) * 0.125f; + } + auto run_linear = [&](const String2TensorStorage& selection) { + ggml_init_params exec_params = {8 * 1024 * 1024, nullptr, false}; + ggml_context* exec_ctx = ggml_init(exec_params); + GGML_ASSERT(exec_ctx != nullptr); + InspectableLinear linear(256, 4, true, false, true, 1.f / 128.f); + linear.init(exec_ctx, selection, "layer"); + ggml_tensor* exec_weight = linear.parameter("weight"); + if (exec_weight->type == GGML_TYPE_Q8_0) { + GGML_ASSERT(loader.load_comfy_int8_tensorwise(weight, exec_weight, nullptr)); + if (!linear.rotation_op_enabled()) { + ggml_tensor* h256 = linear.parameter("weight.convrot_h256"); + std::vector matrix(256 * 256, 0.f); + for (size_t row = 0; row < 256; ++row) { + matrix[row * 256 + row] = 1.f; + float* values = matrix.data() + row * 256; + for (size_t stride = 1; stride < 256; stride *= 4) { + for (size_t base = 0; base < 256; base += 4 * stride) { + for (size_t i = 0; i < stride; ++i) { + float* v = values + base + i; + const float a = v[0], b = v[stride], c = v[2 * stride], d = v[3 * stride]; + v[0] = (a + b + c - d) * 0.5f; + v[stride] = (a + b - c + d) * 0.5f; + v[2 * stride] = (a - b + c + d) * 0.5f; + v[3 * stride] = (-a + b + c + d) * 0.5f; + } + } + } + } + memcpy(h256->data, matrix.data(), matrix.size() * sizeof(float)); + } + } else { + GGML_ASSERT(exec_weight->type == GGML_TYPE_F16); + GGML_ASSERT(loader.load_tensor(weight, exec_weight)); + } + memcpy(linear.parameter("bias")->data, bias_values.data(), bias_values.size() * sizeof(float)); + ggml_tensor* exec_input = ggml_new_tensor_2d(exec_ctx, GGML_TYPE_F32, 256, 2); + memcpy(exec_input->data, input_values.data(), input_values.size() * sizeof(float)); + GGMLRunnerContext exec_runner; + exec_runner.ggml_ctx = exec_ctx; + ggml_tensor* exec_output = linear.forward(&exec_runner, exec_input); + ggml_cgraph* graph = ggml_new_graph(exec_ctx); + ggml_build_forward_expand(graph, exec_output); + GGML_ASSERT(ggml_graph_compute_with_ctx(exec_ctx, graph, 1) == GGML_STATUS_SUCCESS); + std::vector values(8); + memcpy(values.data(), exec_output->data, values.size() * sizeof(float)); + ggml_free(exec_ctx); + return values; + }; + const auto reference_values = run_linear(compatibility_selection); + GGML_ASSERT(std::fabs(reference_values[0] - bias_values[0]) > 0.1f); + GGML_ASSERT(std::fabs(reference_values[4] - bias_values[0]) > 0.1f); + for (const auto& selection : {automatic_selection, dense_selection}) { + const auto q8_values = run_linear(selection); + for (size_t i = 0; i < q8_values.size(); ++i) { + GGML_ASSERT(std::isfinite(q8_values[i])); + GGML_ASSERT(std::fabs(q8_values[i] - reference_values[i]) < 0.05f); + } + } GGML_ASSERT(unset_test_environment("SD_CONVROT_MODE") == 0); ggml_backend_free(cpu_backend); @@ -444,6 +510,22 @@ int main() { GGML_ASSERT(ggml_fp16_to_fp32(static_cast(wide_decoded->data)[4999]) == 0.0f); ggml_free(wide_ctx); + write_fixture(path, "{\"format\":\"int8_tensorwise\"}", 0.5f, 256); + ModelLoader plain_loader; + GGML_ASSERT(plain_loader.init_from_file(path.string())); + plain_loader.set_wtype_override(GGML_TYPE_Q8_0, ""); + const TensorStorage& plain_weight = find_tensor(plain_loader, "layer.weight"); + GGML_ASSERT(plain_weight.is_comfy_int8_tensorwise && !plain_weight.comfy_int8_convrot); + GGML_ASSERT(plain_weight.expected_type == GGML_TYPE_F16); + ggml_init_params plain_params = {ggml_tensor_overhead() + 4 * 256 * sizeof(ggml_fp16_t) + 4096, + nullptr, false}; + ggml_context* plain_ctx = ggml_init(plain_params); + GGML_ASSERT(plain_ctx != nullptr); + ggml_tensor* plain_decoded = ggml_new_tensor_2d(plain_ctx, GGML_TYPE_F16, 256, 4); + GGML_ASSERT(plain_loader.load_tensor(plain_weight, plain_decoded)); + GGML_ASSERT(ggml_fp16_to_fp32(static_cast(plain_decoded->data)[0]) == 1.f); + ggml_free(plain_ctx); + write_fixture(path, marker, std::numeric_limits::quiet_NaN()); ModelLoader invalid_scale_loader; GGML_ASSERT(invalid_scale_loader.init_from_file(path.string())); From a69aade311f522de758790f060ec3a2bca367bc1 Mon Sep 17 00:00:00 2001 From: aegioscy Date: Tue, 29 Sep 2026 12:00:53 +0000 Subject: [PATCH 4/5] test(convrot): compare Q8 paths with F32 reference --- tests/test-safetensors-convrot.cpp | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/tests/test-safetensors-convrot.cpp b/tests/test-safetensors-convrot.cpp index 6f9cedc286..b0c536dea2 100644 --- a/tests/test-safetensors-convrot.cpp +++ b/tests/test-safetensors-convrot.cpp @@ -464,7 +464,7 @@ int main() { memcpy(h256->data, matrix.data(), matrix.size() * sizeof(float)); } } else { - GGML_ASSERT(exec_weight->type == GGML_TYPE_F16); + GGML_ASSERT(exec_weight->type == GGML_TYPE_F16 || exec_weight->type == GGML_TYPE_F32); GGML_ASSERT(loader.load_tensor(weight, exec_weight)); } memcpy(linear.parameter("bias")->data, bias_values.data(), bias_values.size() * sizeof(float)); @@ -481,7 +481,9 @@ int main() { ggml_free(exec_ctx); return values; }; - const auto reference_values = run_linear(compatibility_selection); + auto f32_compat_selection = compatibility_selection; + f32_compat_selection.at("layer.weight").expected_type = GGML_TYPE_F32; + const auto reference_values = run_linear(f32_compat_selection); GGML_ASSERT(std::fabs(reference_values[0] - bias_values[0]) > 0.1f); GGML_ASSERT(std::fabs(reference_values[4] - bias_values[0]) > 0.1f); for (const auto& selection : {automatic_selection, dense_selection}) { From a3c6fcb197ce1028d51c7dda156c321848553412 Mon Sep 17 00:00:00 2001 From: aegioscy Date: Tue, 29 Sep 2026 15:31:47 +0000 Subject: [PATCH 5/5] test(convrot): use backend graph API for dynamic CPU --- tests/test-safetensors-convrot.cpp | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/tests/test-safetensors-convrot.cpp b/tests/test-safetensors-convrot.cpp index b0c536dea2..ac4d95b420 100644 --- a/tests/test-safetensors-convrot.cpp +++ b/tests/test-safetensors-convrot.cpp @@ -11,7 +11,6 @@ #include "core/ggml_extend.hpp" #include "core/util.h" -#include "ggml-cpu.h" #include "model_io/binary_io.h" #include "model_io/safetensors_io.h" #include "model_loader.h" @@ -475,7 +474,7 @@ int main() { ggml_tensor* exec_output = linear.forward(&exec_runner, exec_input); ggml_cgraph* graph = ggml_new_graph(exec_ctx); ggml_build_forward_expand(graph, exec_output); - GGML_ASSERT(ggml_graph_compute_with_ctx(exec_ctx, graph, 1) == GGML_STATUS_SUCCESS); + GGML_ASSERT(ggml_backend_graph_compute(cpu_backend, graph) == GGML_STATUS_SUCCESS); std::vector values(8); memcpy(values.data(), exec_output->data, values.size() * sizeof(float)); ggml_free(exec_ctx);