Commit f89d9b1 for stable-diffusion.cpp
commit f89d9b13d730eabeede7314ce49dacd18d3c90c2
Author: leejet <leejet714@gmail.com>
Date: Sat Oct 10 01:39:49 2026 +0800
feat: add Iris-3B text-to-image support (#2117)
diff --git a/README.md b/README.md
index 2305974..832ee33 100644
--- a/README.md
+++ b/README.md
@@ -67,6 +67,7 @@ API and command-line option may change frequently.***
- [LLaDA-Image](./docs/llada_image.md)
- [Ming-Image Design](./docs/ming_image.md)
- [PixArt](./docs/pixart.md)
+ - [Iris-3B](./docs/iris.md)
- [Image Edit Models](./docs/edit.md)
- [FLUX.1-Kontext-dev](./docs/kontext.md)
- [Qwen Image Edit series](./docs/qwen_image_edit.md)
diff --git a/assets/iris/example.png b/assets/iris/example.png
new file mode 100644
index 0000000..af3a4fd
Binary files /dev/null and b/assets/iris/example.png differ
diff --git a/docs/iris.md b/docs/iris.md
new file mode 100644
index 0000000..940fd1e
--- /dev/null
+++ b/docs/iris.md
@@ -0,0 +1,24 @@
+# How to Use
+
+Iris-3B generates images directly in pixel space and uses Qwen3-VL-4B-Instruct as the text encoder. No VAE is required.
+
+## Download weights
+
+- Download Iris-3B
+ - safetensors: https://huggingface.co/speridlabs/iris-3b/tree/main (`model.safetensors` in the root directory)
+- Download Qwen3-VL-4B-Instruct
+ - safetensors: https://huggingface.co/Comfy-Org/Krea-2/tree/main/text_encoders
+ - gguf: https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct-GGUF/tree/main
+
+## Examples
+
+```
+.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\iris-3b.safetensors --llm ..\models\text_encoders\Qwen3-VL-4B-Instruct-Q4_K_M.gguf -p "a lovely cat" --cfg-scale 3 -H 1024 -W 1024 --diffusion-fa
+```
+
+<img width="256" alt="iris-3b example" src="../assets/iris/example.png" />
+
+## Notes
+
+- Width and height must be multiples of 16. Do not pass `--vae`.
+- Captions are limited to 300 tokens including the assistant-turn suffix. Positive and negative prompts use the same template.
diff --git a/src/conditioning/conditioner.hpp b/src/conditioning/conditioner.hpp
index c1aad85..0062c4e 100644
--- a/src/conditioning/conditioner.hpp
+++ b/src/conditioning/conditioner.hpp
@@ -2012,7 +2012,8 @@ struct LLMEmbedder : public Conditioner {
sd_version_is_sefi_image(version) ||
sd_version_is_krea2(version) ||
sd_version_is_minimax_h3(version) ||
- sd_version_is_mage_flow(version)) {
+ sd_version_is_mage_flow(version) ||
+ version == VERSION_IRIS) {
arch = LLM::LLMArch::QWEN3_VL;
} else if (sd_version_is_z_image(version) || sd_version_is_z_image_l2p(version) || version == VERSION_OVIS_IMAGE || version == VERSION_FLUX2_KLEIN) {
arch = LLM::LLMArch::QWEN3;
@@ -2138,9 +2139,10 @@ struct LLMEmbedder : public Conditioner {
std::tuple<std::vector<int>, std::vector<float>, std::vector<float>> tokenize(std::string text,
const std::pair<int, int>& attn_range,
- size_t min_length = 0,
- size_t max_length = 100000000,
- bool spell_quotes = false) {
+ size_t min_length = 0,
+ size_t max_length = 100000000,
+ bool spell_quotes = false,
+ const std::string& suffix = "") {
std::vector<std::pair<std::string, float>> parsed_attention;
if (attn_range.first >= 0 && attn_range.second > 0) {
if (attn_range.first > 0) {
@@ -2185,6 +2187,20 @@ struct LLMEmbedder : public Conditioner {
weights.insert(weights.end(), curr_tokens.size(), curr_weight);
}
+ if (!suffix.empty()) {
+ std::vector<int> suffix_tokens;
+ if (!tokenizer->encode(suffix, suffix_tokens) || suffix_tokens.size() >= max_length) {
+ return {};
+ }
+ // Reserve the assistant-turn marker before truncating the caption.
+ if (tokens.size() > max_length - suffix_tokens.size()) {
+ tokens.resize(max_length - suffix_tokens.size());
+ weights.resize(tokens.size());
+ }
+ tokens.insert(tokens.end(), suffix_tokens.begin(), suffix_tokens.end());
+ weights.insert(weights.end(), suffix_tokens.size(), 1.f);
+ }
+
std::vector<float> mask;
tokenizer->pad_tokens(tokens, &weights, &mask, min_length, max_length);
@@ -2207,8 +2223,10 @@ struct LLMEmbedder : public Conditioner {
bool spell_quotes = false,
int max_length = 100000000,
const LLM::DeepStackImageEmbeds& deepstack_image_embeds = {},
- const std::vector<LLM::ImageGrid>& image_grids = {}) {
- auto tokens_weights_mask = tokenize(prompt, prompt_attn_range, min_length, max_length, spell_quotes);
+ const std::vector<LLM::ImageGrid>& image_grids = {},
+ const std::string& prompt_suffix = "",
+ sd::Tensor<float>* output_mask = nullptr) {
+ auto tokens_weights_mask = tokenize(prompt, prompt_attn_range, min_length, max_length, spell_quotes, prompt_suffix);
auto& tokens = std::get<0>(tokens_weights_mask);
auto& weights = std::get<1>(tokens_weights_mask);
auto& mask = std::get<2>(tokens_weights_mask);
@@ -2271,6 +2289,12 @@ struct LLMEmbedder : public Conditioner {
1);
}
+ if (output_mask != nullptr) {
+ *output_mask = sd::Tensor<float>::zeros({new_hidden_states.shape()[1]});
+ for (size_t i = prompt_template_encode_start_idx; i < mask.size(); ++i) {
+ output_mask->data()[i - prompt_template_encode_start_idx] = mask[i];
+ }
+ }
return new_hidden_states;
}
@@ -2337,6 +2361,8 @@ struct LLMEmbedder : public Conditioner {
int hidden_states_min_length = 0; // zero pad hidden_states
bool spell_quotes = false;
std::set<int> out_layers;
+ std::string prompt_suffix;
+ sd::Tensor<float> text_mask;
int64_t t0 = ggml_time_ms();
RefImageResizeMode resize_mode = conditioner_params.ref_image_params.vlm_resize_mode;
@@ -3086,6 +3112,23 @@ struct LLMEmbedder : public Conditioner {
SDCondition result;
result.c_crossattn = std::move(hidden_states);
return result;
+ } else if (version == VERSION_IRIS) {
+ prompt =
+ "<|im_start|>system\n"
+ "Describe the image by detailing the color, shape, size, texture, quantity, text, spatial "
+ "relationships of the objects and background:<|im_end|>\n<|im_start|>user\n";
+ auto prefix_tokens = std::get<0>(tokenize(prompt, {0, 0}));
+ if (prefix_tokens.empty()) {
+ return {};
+ }
+ prompt_template_encode_start_idx = static_cast<int>(prefix_tokens.size());
+ hidden_states_min_length = 300;
+ max_length = prompt_template_encode_start_idx + hidden_states_min_length;
+ out_layers = {2, 5, 8, 11, 14, 17, 20, 23, 26, 29, 32, 35};
+ prompt_attn_range.first = static_cast<int>(prompt.size());
+ prompt += conditioner_params.text;
+ prompt_attn_range.second = static_cast<int>(prompt.size());
+ prompt_suffix = "<|im_end|>\n<|im_start|>assistant\n";
} else {
GGML_ABORT("unknown version %d", version);
}
@@ -3101,7 +3144,9 @@ struct LLMEmbedder : public Conditioner {
spell_quotes,
max_length,
deepstack_image_embeds,
- image_grids);
+ image_grids,
+ prompt_suffix,
+ version == VERSION_IRIS ? &text_mask : nullptr);
if (hidden_states.empty()) {
return {};
}
@@ -3166,6 +3211,7 @@ struct LLMEmbedder : public Conditioner {
SDCondition result;
result.c_crossattn = std::move(hidden_states);
result.extra_c_crossattns = std::move(extra_hidden_states_vec);
+ result.c_vector = std::move(text_mask);
if (version == VERSION_QWEN_IMAGE_2_1) {
auto slots = sd::Tensor<int32_t>::zeros({result.c_crossattn.shape()[1]});
for (size_t i = 0; i < image_embeds.size(); ++i) {
diff --git a/src/model.h b/src/model.h
index 19a8213..0d291ce 100644
--- a/src/model.h
+++ b/src/model.h
@@ -65,6 +65,7 @@ enum SDVersion {
VERSION_PIXART,
VERSION_MING_IMAGE,
VERSION_Z_IMAGE_L2P,
+ VERSION_IRIS,
VERSION_COUNT,
};
@@ -334,7 +335,8 @@ static inline bool sd_version_is_dit(SDVersion version) {
sd_version_is_krea2(version) ||
sd_version_is_mage_flow(version) ||
sd_version_is_sensenova_u1(version) ||
- sd_version_is_pixart(version)) {
+ sd_version_is_pixart(version) ||
+ version == VERSION_IRIS) {
return true;
}
return false;
diff --git a/src/model/diffusion/iris.hpp b/src/model/diffusion/iris.hpp
new file mode 100644
index 0000000..b90fbe9
--- /dev/null
+++ b/src/model/diffusion/iris.hpp
@@ -0,0 +1,450 @@
+#ifndef __SD_MODEL_DIFFUSION_IRIS_HPP__
+#define __SD_MODEL_DIFFUSION_IRIS_HPP__
+
+#include <array>
+#include <stdexcept>
+#include "model/diffusion/pid.hpp"
+
+namespace Iris {
+ struct IrisConfig {
+ int64_t hidden_size = 2560;
+ int64_t depth = 24;
+ int64_t dual_depth = 8;
+ int64_t head_dim = 128;
+ int64_t num_heads = 20;
+ int64_t kv_heads = 5;
+ int64_t mlp_dim = 6826;
+ int64_t patch_size = 16;
+ int64_t text_dim = 2560;
+ int64_t text_len = 300;
+ int64_t text_layers = 12;
+ int64_t layer_heads = 32;
+ int64_t layer_mlp_dim = 3328;
+ int64_t pixel_dim = 16;
+ int64_t pixel_attn_dim = 1280;
+ int64_t pixel_heads = 10;
+ int64_t pixel_depth = 4;
+
+ static IrisConfig detect_from_weights(const String2TensorStorage& weights, const std::string& prefix) {
+ IrisConfig config;
+ auto find = [&](const std::string& name) -> const TensorStorage& {
+ auto it = weights.find(prefix + "." + name);
+ if (it == weights.end()) {
+ throw std::runtime_error("Iris-3B: missing weight " + prefix + "." + name);
+ }
+ return it->second;
+ };
+ config.hidden_size = find("s_embedder.proj.weight").ne[1];
+ config.patch_size = static_cast<int64_t>(std::sqrt(find("s_embedder.proj.weight").ne[0] / 3));
+ config.head_dim = find("blocks.0.attn.q_norm_x.weight").ne[0];
+ config.num_heads = config.hidden_size / config.head_dim;
+ config.kv_heads = find("blocks.0.attn.k_proj_x.weight").ne[1] / config.head_dim;
+ config.mlp_dim = find("blocks.0.mlp_x.w1.weight").ne[1];
+ config.text_dim = find("y_embedder.refiner.proj.weight").ne[0];
+ config.text_len = find("y_pos_embedding").ne[1];
+ config.text_layers = find("y_embedder.layer_pool.weight").ne[0];
+ config.layer_mlp_dim = find("y_embedder.layer_blocks.0.mlp.0.weight").ne[1];
+ config.pixel_dim = find("pixel_embedder.proj.weight").ne[1];
+ config.pixel_attn_dim = find("pixel_blocks.0.compress.weight").ne[1];
+ config.pixel_heads = config.pixel_attn_dim / find("pixel_blocks.0.attn.q_norm.weight").ne[0];
+ config.depth = config.dual_depth = config.pixel_depth = 0;
+ for (const auto& [name, tensor] : weights) {
+ if (!starts_with(name, prefix + ".")) {
+ continue;
+ }
+ auto parts = split_string(name.substr(prefix.size() + 1), '.');
+ if (parts.size() > 2 && parts[0] == "blocks") {
+ int64_t index = std::stoi(parts[1]);
+ config.depth = std::max(config.depth, index + 1);
+ if (parts[2] == "adaln_img") {
+ config.dual_depth = std::max(config.dual_depth, index + 1);
+ }
+ } else if (parts.size() > 2 && parts[0] == "pixel_blocks") {
+ config.pixel_depth = std::max(config.pixel_depth, int64_t(std::stoi(parts[1]) + 1));
+ }
+ }
+ if (config.text_layers != 12 || config.text_dim != 2560 || config.text_len != 300 ||
+ find("modulation_cores.adaln_img.weight").ne[1] != 6 * config.hidden_size ||
+ find("pixel_blocks.0.adaln.weight").ne[1] != 4 * config.pixel_dim * config.patch_size * config.patch_size) {
+ throw std::runtime_error("Iris-3B: unsupported checkpoint configuration");
+ }
+ LOG_VERBOSE("iris: hidden_size = %" PRId64 ", depth = %" PRId64 ", dual_depth = %" PRId64 ", heads = %" PRId64 "/%" PRId64 ", pixel_depth = %" PRId64,
+ config.hidden_size, config.depth, config.dual_depth, config.num_heads, config.kv_heads, config.pixel_depth);
+ return config;
+ }
+ };
+
+ struct SelfAttention : public GGMLBlock {
+ int64_t dim;
+ int64_t heads;
+ bool qk_norm;
+
+ SelfAttention(int64_t dim, int64_t heads, bool qk_norm = true)
+ : dim(dim), heads(heads), qk_norm(qk_norm) {
+ blocks["qkv"] = std::make_shared<Linear>(dim, 3 * dim, false);
+ blocks["proj"] = std::make_shared<Linear>(dim, dim, true);
+ if (qk_norm) {
+ blocks["q_norm"] = std::make_shared<RMSNorm>(dim / heads, 1e-6f);
+ blocks["k_norm"] = std::make_shared<RMSNorm>(dim / heads, 1e-6f);
+ }
+ }
+
+ ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* pe = nullptr, ggml_tensor* mask = nullptr) {
+ auto c = ctx->ggml_ctx;
+ auto qkv = split_qkv(c, std::dynamic_pointer_cast<Linear>(blocks["qkv"])->forward(ctx, x));
+ auto q = ggml_reshape_4d(c, qkv[0], dim / heads, heads, x->ne[1], x->ne[2]);
+ auto k = ggml_reshape_4d(c, qkv[1], dim / heads, heads, x->ne[1], x->ne[2]);
+ auto v = ggml_reshape_4d(c, qkv[2], dim / heads, heads, x->ne[1], x->ne[2]);
+ if (qk_norm) {
+ q = std::dynamic_pointer_cast<RMSNorm>(blocks["q_norm"])->forward(ctx, q);
+ k = std::dynamic_pointer_cast<RMSNorm>(blocks["k_norm"])->forward(ctx, k);
+ }
+ if (pe) {
+ x = Rope::attention(ctx, q, k, v, pe, mask);
+ } else {
+ q = ggml_reshape_3d(c, q, dim, x->ne[1], x->ne[2]);
+ k = ggml_reshape_3d(c, k, dim, x->ne[1], x->ne[2]);
+ v = ggml_reshape_3d(c, v, dim, x->ne[1], x->ne[2]);
+ x = ggml_ext_attention_ext(ctx, q, k, v, heads, mask, false, ctx->flash_attn_enabled);
+ }
+ return std::dynamic_pointer_cast<Linear>(blocks["proj"])->forward(ctx, x);
+ }
+ };
+
+ struct TextBlock : public GGMLBlock {
+ bool layerwise;
+
+ TextBlock(int64_t dim, int64_t heads, int64_t mlp_dim, bool layerwise)
+ : layerwise(layerwise) {
+ blocks["norm1"] = std::make_shared<RMSNorm>(dim, 1e-6f);
+ blocks["norm2"] = std::make_shared<RMSNorm>(dim, 1e-6f);
+ blocks["attn"] = std::make_shared<SelfAttention>(dim, heads, !layerwise);
+ if (layerwise) {
+ blocks["mlp.0"] = std::make_shared<Linear>(dim, mlp_dim, true);
+ blocks["mlp.2"] = std::make_shared<Linear>(mlp_dim, dim, true);
+ } else {
+ blocks["mlp"] = std::make_shared<Pid::FeedForward>(dim, mlp_dim);
+ }
+ }
+
+ ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* mask = nullptr) {
+ auto h = std::dynamic_pointer_cast<RMSNorm>(blocks["norm1"])->forward(ctx, x);
+ h = std::dynamic_pointer_cast<SelfAttention>(blocks["attn"])->forward(ctx, h, nullptr, mask);
+ x = ggml_add(ctx->ggml_ctx, x, h);
+ h = std::dynamic_pointer_cast<RMSNorm>(blocks["norm2"])->forward(ctx, x);
+ if (layerwise) {
+ h = std::dynamic_pointer_cast<Linear>(blocks["mlp.0"])->forward(ctx, h);
+ h = ggml_silu(ctx->ggml_ctx, h);
+ h = std::dynamic_pointer_cast<Linear>(blocks["mlp.2"])->forward(ctx, h);
+ } else {
+ h = std::dynamic_pointer_cast<Pid::FeedForward>(blocks["mlp"])->forward(ctx, h);
+ }
+ return ggml_add(ctx->ggml_ctx, x, h);
+ }
+ };
+
+ struct TextEmbedder : public GGMLBlock {
+ IrisConfig config;
+
+ TextEmbedder(const IrisConfig& config)
+ : config(config) {
+ for (int i = 0; i < 2; ++i) {
+ blocks["layer_blocks." + std::to_string(i)] = std::make_shared<TextBlock>(config.text_dim, config.layer_heads, config.layer_mlp_dim, true);
+ blocks["refiner.blocks." + std::to_string(i)] = std::make_shared<TextBlock>(config.hidden_size, config.num_heads, config.mlp_dim, false);
+ }
+ blocks["layer_pool"] = std::make_shared<Linear>(config.text_layers, 1, true);
+ blocks["refiner.proj"] = std::make_shared<Linear>(config.text_dim, config.hidden_size, true);
+ blocks["refiner.norm"] = std::make_shared<RMSNorm>(config.hidden_size, 1e-6f);
+ }
+
+ ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* mask) {
+ auto c = ctx->ggml_ctx;
+ int64_t tokens = x->ne[1];
+ int64_t batch = x->ne[2];
+ // The encoder concatenates selected layers within each token's feature dimension.
+ x = ggml_reshape_3d(c, x, config.text_dim, config.text_layers, tokens * batch);
+ for (int i = 0; i < 2; ++i) {
+ x = std::dynamic_pointer_cast<TextBlock>(blocks["layer_blocks." + std::to_string(i)])->forward(ctx, x);
+ }
+ x = ggml_cont(c, ggml_permute(c, x, 1, 0, 2, 3));
+ x = std::dynamic_pointer_cast<Linear>(blocks["layer_pool"])->forward(ctx, x);
+ x = ggml_reshape_3d(c, x, config.text_dim, tokens, batch);
+ x = std::dynamic_pointer_cast<Linear>(blocks["refiner.proj"])->forward(ctx, x);
+ for (int i = 0; i < 2; ++i) {
+ x = std::dynamic_pointer_cast<TextBlock>(blocks["refiner.blocks." + std::to_string(i)])->forward(ctx, x, mask);
+ }
+ return std::dynamic_pointer_cast<RMSNorm>(blocks["refiner.norm"])->forward(ctx, x);
+ }
+ };
+
+ struct TrunkBlock : public GGMLBlock {
+ IrisConfig config;
+ bool dual;
+
+ TrunkBlock(const IrisConfig& config, bool dual)
+ : config(config), dual(dual) {
+ int64_t d = config.hidden_size;
+ for (const std::string stream : dual ? std::vector<std::string>{"x", "y"} : std::vector<std::string>{""}) {
+ std::string suffix = stream.empty() ? "" : "_" + stream;
+ std::string attn = dual ? "attn." : "";
+ blocks[attn + "q_proj" + suffix] = std::make_shared<Linear>(d, d, false);
+ blocks[attn + "k_proj" + suffix] = std::make_shared<Linear>(d, config.kv_heads * config.head_dim, false);
+ blocks[attn + "v_proj" + suffix] = std::make_shared<Linear>(d, config.kv_heads * config.head_dim, false);
+ blocks[attn + "q_norm" + suffix] = std::make_shared<RMSNorm>(config.head_dim, 1e-6f);
+ blocks[attn + "k_norm" + suffix] = std::make_shared<RMSNorm>(config.head_dim, 1e-6f);
+ blocks[dual ? "attn.proj" + suffix : "attn_proj"] = std::make_shared<Linear>(d, d, true);
+ blocks["attn_gate" + suffix] = std::make_shared<Linear>(d, d, false);
+ blocks["norm" + suffix + "1"] = std::make_shared<RMSNorm>(d, 1e-6f);
+ blocks["norm" + suffix + "2"] = std::make_shared<RMSNorm>(d, 1e-6f);
+ blocks["attn_post_norm" + suffix] = std::make_shared<RMSNorm>(d, 1e-6f);
+ blocks["mlp_post_norm" + suffix] = std::make_shared<RMSNorm>(d, 1e-6f);
+ blocks["mlp" + suffix] = std::make_shared<Pid::FeedForward>(d, config.mlp_dim);
+ }
+ }
+
+ void init_params(ggml_context* ctx, const String2TensorStorage& = {}, const std::string = "") override {
+ if (dual) {
+ params["adaln_img.bias"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, 6 * config.hidden_size);
+ params["adaln_txt.bias"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, 6 * config.hidden_size);
+ } else {
+ params["adaln.bias"] = ggml_new_tensor_1d(ctx, GGML_TYPE_F32, 6 * config.hidden_size);
+ }
+ }
+
+ ggml_tensor* norm(GGMLRunnerContext* ctx, const std::string& name, ggml_tensor* x) {
+ return std::dynamic_pointer_cast<RMSNorm>(blocks[name])->forward(ctx, x);
+ }
+
+ ggml_tensor* linear(GGMLRunnerContext* ctx, const std::string& name, ggml_tensor* x) {
+ return std::dynamic_pointer_cast<Linear>(blocks[name])->forward(ctx, x);
+ }
+
+ std::array<ggml_tensor*, 3> project(GGMLRunnerContext* ctx, ggml_tensor* h, const std::string& suffix) {
+ auto c = ctx->ggml_ctx;
+ std::string a = dual ? "attn." : "";
+ auto q = linear(ctx, a + "q_proj" + suffix, h);
+ auto k = linear(ctx, a + "k_proj" + suffix, h);
+ auto v = linear(ctx, a + "v_proj" + suffix, h);
+ q = ggml_reshape_4d(c, q, config.head_dim, config.num_heads, h->ne[1], h->ne[2]);
+ k = ggml_reshape_4d(c, k, config.head_dim, config.kv_heads, h->ne[1], h->ne[2]);
+ v = ggml_reshape_4d(c, v, config.head_dim, config.kv_heads, h->ne[1], h->ne[2]);
+ return {norm(ctx, a + "q_norm" + suffix, q), norm(ctx, a + "k_norm" + suffix, k), v};
+ }
+
+ ggml_tensor* finish(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* h, ggml_tensor* out, const std::vector<ggml_tensor*>& mods, const std::string& suffix) {
+ auto c = ctx->ggml_ctx;
+ out = ggml_mul(c, out, ggml_sigmoid(c, linear(ctx, "attn_gate" + suffix, h)));
+ out = linear(ctx, dual ? "attn.proj" + suffix : "attn_proj", out);
+ out = norm(ctx, "attn_post_norm" + suffix, out);
+ x = ggml_add(c, x, ggml_mul(c, out, mods[2]));
+ h = Pid::apply_adaln(c, norm(ctx, "norm" + suffix + "2", x), mods[3], mods[4]);
+ h = std::dynamic_pointer_cast<Pid::FeedForward>(blocks["mlp" + suffix])->forward(ctx, h);
+ h = norm(ctx, "mlp_post_norm" + suffix, h);
+ return ggml_add(c, x, ggml_mul(c, h, mods[5]));
+ }
+
+ std::pair<ggml_tensor*, ggml_tensor*> forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* y, ggml_tensor* img_mod, ggml_tensor* txt_mod, ggml_tensor* pe) {
+ auto c = ctx->ggml_ctx;
+ int64_t nt = y->ne[1];
+ auto mx = ggml_ext_chunk(c, ggml_add(c, img_mod, params[dual ? "adaln_img.bias" : "adaln.bias"]), 6, 0);
+ if (!dual) {
+ auto tokens = ggml_concat(c, y, x, 1);
+ auto h = Pid::apply_adaln(c, norm(ctx, "norm1", tokens), mx[0], mx[1]);
+ auto qkv = project(ctx, h, "");
+ auto out = Rope::attention(ctx, qkv[0], qkv[1], qkv[2], pe, nullptr);
+ tokens = finish(ctx, tokens, h, out, mx, "");
+ return {ggml_ext_slice(c, tokens, 1, nt, tokens->ne[1]), ggml_ext_slice(c, tokens, 1, 0, nt)};
+ }
+ auto my = ggml_ext_chunk(c, ggml_add(c, txt_mod, params["adaln_txt.bias"]), 6, 0);
+ auto hx = Pid::apply_adaln(c, norm(ctx, "norm_x1", x), mx[0], mx[1]);
+ auto hy = Pid::apply_adaln(c, norm(ctx, "norm_y1", y), my[0], my[1]);
+ auto qx = project(ctx, hx, "_x");
+ auto qy = project(ctx, hy, "_y");
+ auto out = Rope::attention(ctx, ggml_concat(c, qy[0], qx[0], 2), ggml_concat(c, qy[1], qx[1], 2),
+ ggml_concat(c, qy[2], qx[2], 2), pe, nullptr);
+ x = finish(ctx, x, hx, ggml_ext_slice(c, out, 1, nt, out->ne[1]), mx, "_x");
+ y = finish(ctx, y, hy, ggml_ext_slice(c, out, 1, 0, nt), my, "_y");
+ return {x, y};
+ }
+ };
+
+ struct PixelBlock : public GGMLBlock {
+ IrisConfig config;
+
+ PixelBlock(const IrisConfig& config)
+ : config(config) {
+ auto d = config.pixel_dim;
+ auto flat = d * config.patch_size * config.patch_size;
+ blocks["norm1"] = std::make_shared<RMSNorm>(d, 1e-6f);
+ blocks["norm2"] = std::make_shared<RMSNorm>(d, 1e-6f);
+ blocks["compress"] = std::make_shared<Linear>(flat, config.pixel_attn_dim, true);
+ blocks["expand"] = std::make_shared<Linear>(config.pixel_attn_dim, flat, true);
+ blocks["adaln"] = std::make_shared<Linear>(config.hidden_size, 4 * flat, true);
+ blocks["attn"] = std::make_shared<SelfAttention>(config.pixel_attn_dim, config.pixel_heads);
+ blocks["mlp.fc1"] = std::make_shared<Linear>(d, d * 4, true);
+ blocks["mlp.fc2"] = std::make_shared<Linear>(d * 4, d, true);
+ }
+
+ ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x, ggml_tensor* cond, ggml_tensor* pe) {
+ auto c = ctx->ggml_ctx;
+ auto d = config.pixel_dim;
+ auto p2 = config.patch_size * config.patch_size;
+ auto patches = cond->ne[1];
+ auto batch = cond->ne[2];
+ auto mods = std::dynamic_pointer_cast<Linear>(blocks["adaln"])->forward(ctx, cond);
+ mods = ggml_reshape_3d(c, mods, 4 * d, p2, patches * batch);
+ auto m = ggml_ext_chunk(c, mods, 4, 0);
+ auto h = std::dynamic_pointer_cast<RMSNorm>(blocks["norm1"])->forward(ctx, x);
+ h = ggml_reshape_3d(c, h, d * p2, patches, batch);
+ h = std::dynamic_pointer_cast<Linear>(blocks["compress"])->forward(ctx, h);
+ h = std::dynamic_pointer_cast<SelfAttention>(blocks["attn"])->forward(ctx, h, pe);
+ h = std::dynamic_pointer_cast<Linear>(blocks["expand"])->forward(ctx, h);
+ h = ggml_reshape_3d(c, h, d, p2, patches * batch);
+ x = ggml_add(c, x, Pid::apply_adaln(c, h, m[1], m[0]));
+ h = std::dynamic_pointer_cast<RMSNorm>(blocks["norm2"])->forward(ctx, x);
+ h = std::dynamic_pointer_cast<Linear>(blocks["mlp.fc1"])->forward(ctx, h);
+ h = ggml_gelu_erf(c, h);
+ h = std::dynamic_pointer_cast<Linear>(blocks["mlp.fc2"])->forward(ctx, h);
+ return ggml_add(c, x, Pid::apply_adaln(c, h, m[3], m[2]));
+ }
+ };
+
+ struct IrisModel : public GGMLBlock {
+ IrisConfig config;
+
+ IrisModel(const IrisConfig& config)
+ : config(config) {
+ blocks["s_embedder"] = std::make_shared<Pid::PatchTokenEmbedder>(3 * config.patch_size * config.patch_size, config.hidden_size);
+ blocks["t_embedder"] = std::make_shared<Pid::PixelDiTTimestepEmbedder>(config.hidden_size);
+ blocks["y_embedder"] = std::make_shared<TextEmbedder>(config);
+ blocks["modulation_cores.adaln_img"] = std::make_shared<Linear>(config.hidden_size, 6 * config.hidden_size, true);
+ blocks["modulation_cores.adaln_txt"] = std::make_shared<Linear>(config.hidden_size, 6 * config.hidden_size, true);
+ for (int64_t i = 0; i < config.depth; ++i) {
+ blocks["blocks." + std::to_string(i)] = std::make_shared<TrunkBlock>(config, i < config.dual_depth);
+ }
+ blocks["pixel_embedder"] = std::make_shared<Pid::PixelTokenEmbedder>(3, config.pixel_dim);
+ for (int64_t i = 0; i < config.pixel_depth; ++i) {
+ blocks["pixel_blocks." + std::to_string(i)] = std::make_shared<PixelBlock>(config);
+ }
+ blocks["final_layer"] = std::make_shared<Pid::FinalLayer>(config.pixel_dim, 3);
+ }
+
+ void init_params(ggml_context* ctx, const String2TensorStorage& = {}, const std::string = "") override {
+ params["y_pos_embedding"] = ggml_new_tensor_3d(ctx, GGML_TYPE_F32, config.hidden_size, config.text_len, 1);
+ }
+
+ ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* image, ggml_tensor* t, ggml_tensor* text, ggml_tensor* mask, ggml_tensor* joint_pe, ggml_tensor* pixel_pe, ggml_tensor* pixel_pos) {
+ auto c = ctx->ggml_ctx;
+ int p = static_cast<int>(config.patch_size);
+ auto x = DiT::patchify(c, image, p, p, true);
+ x = std::dynamic_pointer_cast<Pid::PatchTokenEmbedder>(blocks["s_embedder"])->forward(ctx, x);
+ auto te = std::dynamic_pointer_cast<Pid::PixelDiTTimestepEmbedder>(blocks["t_embedder"])->forward(ctx, t);
+ te = ggml_reshape_3d(c, te, config.hidden_size, 1, image->ne[3]);
+ auto cond = ggml_silu(c, te);
+ auto im = std::dynamic_pointer_cast<Linear>(blocks["modulation_cores.adaln_img"])->forward(ctx, cond);
+ auto tm = std::dynamic_pointer_cast<Linear>(blocks["modulation_cores.adaln_txt"])->forward(ctx, cond);
+ auto y = std::dynamic_pointer_cast<TextEmbedder>(blocks["y_embedder"])->forward(ctx, text, mask);
+ y = ggml_add(c, y, params["y_pos_embedding"]);
+ for (int64_t i = 0; i < config.depth; ++i) {
+ auto block = std::dynamic_pointer_cast<TrunkBlock>(blocks["blocks." + std::to_string(i)]);
+ std::tie(x, y) = block->forward(ctx, x, y, im, tm, joint_pe);
+ sd::ggml_graph_cut::mark_graph_cut(x, "iris.blocks." + std::to_string(i), "x");
+ sd::ggml_graph_cut::mark_graph_cut(y, "iris.blocks." + std::to_string(i), "y");
+ }
+ cond = ggml_silu(c, ggml_add(c, x, te));
+ x = std::dynamic_pointer_cast<Pid::PixelTokenEmbedder>(blocks["pixel_embedder"])->forward(ctx, image, p, pixel_pos);
+ for (int64_t i = 0; i < config.pixel_depth; ++i) {
+ x = std::dynamic_pointer_cast<PixelBlock>(blocks["pixel_blocks." + std::to_string(i)])->forward(ctx, x, cond, pixel_pe);
+ sd::ggml_graph_cut::mark_graph_cut(x, "iris.pixel_blocks." + std::to_string(i), "x");
+ }
+ x = std::dynamic_pointer_cast<Pid::FinalLayer>(blocks["final_layer"])->forward(ctx, x);
+ x = ggml_reshape_3d(c, x, 3 * p * p, cond->ne[1], image->ne[3]);
+ return DiT::unpatchify(c, x, image->ne[1] / p, image->ne[0] / p, p, p, false);
+ }
+ };
+
+ inline std::vector<float> image_rope(int64_t height, int64_t width, int64_t dim) {
+ std::vector<float> pe;
+ pe.reserve(height * width * dim * 2);
+ float step = 16.f / static_cast<float>(std::max<int64_t>(std::max(height, width) - 1, 1));
+ for (int64_t y = 0; y < height; ++y) {
+ for (int64_t x = 0; x < width; ++x) {
+ for (int64_t i = 0; i < dim / 4; ++i) {
+ float frequency = std::pow(10000.f, -4.f * static_cast<float>(i) / static_cast<float>(dim));
+ for (int64_t pos : {x, y}) {
+ float angle = static_cast<float>(pos) * step * frequency;
+ pe.insert(pe.end(), {std::cos(angle), -std::sin(angle), std::sin(angle), std::cos(angle)});
+ }
+ }
+ }
+ }
+ return pe;
+ }
+
+ struct IrisRunner : public DiffusionModelRunner {
+ IrisConfig config;
+ IrisModel model;
+ std::vector<float> joint_pe_data;
+ std::vector<float> pixel_pe_data;
+ std::vector<float> pixel_pos_data;
+ std::vector<float> text_mask_data;
+
+ IrisRunner(ggml_backend_t backend, const String2TensorStorage& weights, const std::string& prefix, std::shared_ptr<RunnerWeightManager> weight_manager = nullptr)
+ : DiffusionModelRunner(backend, prefix, weight_manager),
+ config(IrisConfig::detect_from_weights(weights, prefix)),
+ model(config) {
+ model.init(params_ctx, weights, prefix);
+ }
+
+ std::string get_desc() override { return "Iris-3B"; }
+
+ void get_param_tensors(std::map<std::string, ggml_tensor*>& tensors, const std::string& prefix) override {
+ model.get_param_tensors(tensors, prefix);
+ }
+
+ sd::Tensor<float> compute(int n_threads, const DiffusionParams& inputs) override {
+ if (!inputs.x || !inputs.timesteps || !inputs.context || !inputs.y || inputs.y->empty()) {
+ LOG_ERROR("Iris-3B requires text features and their padding mask");
+ return {};
+ }
+ auto get_graph = [&]() {
+ auto gf = new_graph_custom(196608);
+ auto x = make_input(*inputs.x);
+ auto t = make_input(*inputs.timesteps);
+ auto text = make_input(*inputs.context);
+ int h = static_cast<int>(x->ne[1]);
+ int w = static_cast<int>(x->ne[0]);
+ int nt = static_cast<int>(config.text_len);
+ GGML_ASSERT(h % config.patch_size == 0 && w % config.patch_size == 0);
+ GGML_ASSERT(text->ne[0] == config.text_dim * config.text_layers && text->ne[1] == nt);
+ GGML_ASSERT(inputs.y->numel() == nt && x->ne[3] == 1);
+ joint_pe_data = Pid::make_rope_1d(nt, static_cast<int>(config.head_dim), 10000.f);
+ auto img = image_rope(h / config.patch_size, w / config.patch_size, config.head_dim);
+ joint_pe_data.insert(joint_pe_data.end(), img.begin(), img.end());
+ pixel_pe_data = image_rope(h / config.patch_size, w / config.patch_size, config.pixel_attn_dim / config.pixel_heads);
+ pixel_pos_data = Pid::make_pixel_abs_pos(h, w, static_cast<int>(config.pixel_dim));
+ text_mask_data.resize(nt * nt);
+ for (int q = 0; q < nt; ++q) {
+ for (int k = 0; k < nt; ++k) {
+ text_mask_data[q * nt + k] = inputs.y->values()[k] != 0.f || q == k ? 0.f : -INFINITY;
+ }
+ }
+ auto pe = ggml_new_tensor_4d(compute_ctx, GGML_TYPE_F32, 2, 2, config.head_dim / 2, nt + img.size() / (2 * config.head_dim));
+ auto pp = ggml_new_tensor_4d(compute_ctx, GGML_TYPE_F32, 2, 2, config.pixel_attn_dim / config.pixel_heads / 2, (h / config.patch_size) * (w / config.patch_size));
+ auto pos = ggml_new_tensor_3d(compute_ctx, GGML_TYPE_F32, config.pixel_dim, h * w, 1);
+ auto mask = ggml_new_tensor_2d(compute_ctx, GGML_TYPE_F32, nt, nt);
+ set_backend_tensor_data(pe, joint_pe_data.data());
+ set_backend_tensor_data(pp, pixel_pe_data.data());
+ set_backend_tensor_data(pos, pixel_pos_data.data());
+ set_backend_tensor_data(mask, text_mask_data.data());
+ auto ctx = get_context();
+ auto out = model.forward(&ctx, x, t, text, mask, pe, pp, pos);
+ ggml_build_forward_expand(gf, out);
+ return gf;
+ };
+ return restore_trailing_singleton_dims(GGMLRunner::compute(get_graph, n_threads, false), inputs.x->dim());
+ }
+ };
+} // namespace Iris
+
+#endif // __SD_MODEL_DIFFUSION_IRIS_HPP__
diff --git a/src/model/vae/vae.hpp b/src/model/vae/vae.hpp
index d5ac6f5..3c66965 100644
--- a/src/model/vae/vae.hpp
+++ b/src/model/vae/vae.hpp
@@ -176,11 +176,20 @@ public:
int scale_factor = 8;
if (version == VERSION_LTXAV) {
scale_factor = 32;
- } else if (version == VERSION_WAN2_2_TI2V || version == VERSION_QWEN_IMAGE_2_1 || sd_version_is_hunyuan_video(version) || sd_version_is_mage_flow(version) || sd_version_is_minimax_h3(version)) {
+ } else if (version == VERSION_WAN2_2_TI2V ||
+ version == VERSION_QWEN_IMAGE_2_1 ||
+ sd_version_is_hunyuan_video(version) ||
+ sd_version_is_mage_flow(version) ||
+ sd_version_is_minimax_h3(version)) {
scale_factor = 16;
} else if (sd_version_uses_flux2_vae(version)) {
scale_factor = 16;
- } else if (version == VERSION_CHROMA_RADIANCE || version == VERSION_HIDREAM_O1 || sd_version_is_minit2i(version) || sd_version_is_sensenova_u1(version) || sd_version_is_z_image_l2p(version)) {
+ } else if (version == VERSION_CHROMA_RADIANCE ||
+ version == VERSION_HIDREAM_O1 ||
+ sd_version_is_minit2i(version) ||
+ sd_version_is_sensenova_u1(version) ||
+ sd_version_is_z_image_l2p(version) ||
+ version == VERSION_IRIS) {
scale_factor = 1;
}
return scale_factor;
diff --git a/src/model_loader.cpp b/src/model_loader.cpp
index 4c16065..f14044d 100644
--- a/src/model_loader.cpp
+++ b/src/model_loader.cpp
@@ -556,6 +556,10 @@ SDVersion ModelLoader::get_sd_version() const {
tensor_storage_map.find("model.diffusion_model.audio_patch_proj.weight") != tensor_storage_map.end()) {
return VERSION_MINIMAX_H3;
}
+ if (name == "model.diffusion_model.y_embedder.layer_pool.weight" &&
+ tensor_storage_map.count("model.diffusion_model.modulation_cores.adaln_img.weight")) {
+ return VERSION_IRIS;
+ }
if (tensor_storage.name.find("model.diffusion_model.blocks.0.cross_attn.norm_k.weight") != std::string::npos) {
is_wan = true;
}
diff --git a/src/pipeline/diffusion_engine.cpp b/src/pipeline/diffusion_engine.cpp
index f9785aa..68129bb 100644
--- a/src/pipeline/diffusion_engine.cpp
+++ b/src/pipeline/diffusion_engine.cpp
@@ -108,6 +108,7 @@ const char* model_version_to_str[] = {
"PixArt",
"Ming-Image",
"Z-Image L2P",
+ "Iris-3B",
};
static_assert(VERSION_COUNT == sizeof(model_version_to_str) / sizeof(model_version_to_str[0]),
@@ -1382,7 +1383,8 @@ bool StableDiffusionGGML::build_denoiser() {
sd_version_is_llada_image(version) ||
sd_version_is_boogu_image(version) ||
sd_version_is_pid(version) ||
- sd_version_is_ideogram4(version)) {
+ sd_version_is_ideogram4(version) ||
+ version == VERSION_IRIS) {
pred_type = FLOW_PRED;
if (sd_version_is_wan(version)) {
default_flow_shift = 5.f;
@@ -1390,7 +1392,7 @@ bool StableDiffusionGGML::build_denoiser() {
default_flow_shift = 7.f;
} else if (sd_version_is_minimax_h3(version)) {
default_flow_shift = 12.f;
- } else if (sd_version_is_ernie_image(version)) {
+ } else if (sd_version_is_ernie_image(version) || version == VERSION_IRIS) {
default_flow_shift = 4.f;
} else if (sd_version_is_pid(version)) {
default_flow_shift = 1.5f;
@@ -2765,7 +2767,7 @@ int StableDiffusionGGML::get_diffusion_model_down_factor() {
if (sd_version_is_dit(version)) {
if (sd_version_is_sensenova_u1(version)) {
down_factor = 32;
- } else if (sd_version_is_z_image_l2p(version)) {
+ } else if (sd_version_is_z_image_l2p(version) || version == VERSION_IRIS) {
down_factor = 16;
} else if (version == VERSION_QWEN_IMAGE_2_1 || version == VERSION_MING_IMAGE || sd_version_is_wan(version) || sd_version_is_lingbot_video(version) || sd_version_is_minimax_h3(version) || sd_version_is_pixart(version)) {
down_factor = 2;
@@ -2799,7 +2801,7 @@ int StableDiffusionGGML::get_latent_channel() {
latent_channel = 3;
} else if (sd_version_is_z_image_l2p(version)) {
latent_channel = 3;
- } else if (sd_version_is_pid(version)) {
+ } else if (sd_version_is_pid(version) || version == VERSION_IRIS) {
latent_channel = 3;
} else if (sd_version_is_sefi_image(version)) {
latent_channel = 144;
@@ -2900,7 +2902,7 @@ sd::Tensor<float> StableDiffusionGGML::encode_first_stage(const sd::Tensor<float
}
sd::Tensor<float> StableDiffusionGGML::decode_first_stage(const sd::Tensor<float>& x, bool decode_video) {
- if (sd_version_is_pid(version) || sd_version_is_minit2i(version)) {
+ if (sd_version_is_pid(version) || sd_version_is_minit2i(version) || version == VERSION_IRIS) {
return sd::ops::clamp((x + 1.f) * 0.5f, 0.0f, 1.0f);
}
auto latents = first_stage_model->diffusion_to_vae_latents(x);
diff --git a/src/pipeline/model_builders.cpp b/src/pipeline/model_builders.cpp
index 3b64c8b..96b5a97 100644
--- a/src/pipeline/model_builders.cpp
+++ b/src/pipeline/model_builders.cpp
@@ -18,6 +18,7 @@
#include "model/diffusion/hidream_o1.hpp"
#include "model/diffusion/hunyuan.hpp"
#include "model/diffusion/ideogram4.hpp"
+#include "model/diffusion/iris.hpp"
#include "model/diffusion/krea2.hpp"
#include "model/diffusion/lens.hpp"
#include "model/diffusion/lingbot_video.hpp"
@@ -456,6 +457,11 @@ namespace sd::model_builders {
tensor_storage_map,
"model.diffusion_model",
weight_manager);
+ } else if (version == VERSION_IRIS) {
+ result.conditioner = std::make_shared<LLMEmbedder>(ctx.backends.runtime_backend(SDBackendModule::TE),
+ tensor_storage_map, version, "", false, weight_manager, tokenizers);
+ result.diffusion = std::make_shared<Iris::IrisRunner>(ctx.backends.runtime_backend(SDBackendModule::DIFFUSION),
+ tensor_storage_map, "model.diffusion_model", weight_manager);
} else { // SD1.x SD2.x SDXL
std::map<std::string, std::string> embbeding_map;
for (uint32_t i = 0; i < sd_ctx_params->embedding_count; i++) {
@@ -626,7 +632,12 @@ namespace sd::model_builders {
}
};
- if (version == VERSION_CHROMA_RADIANCE || version == VERSION_HIDREAM_O1 || sd_version_is_minit2i(version) || sd_version_is_sensenova_u1(version) || sd_version_is_z_image_l2p(version)) {
+ if (version == VERSION_CHROMA_RADIANCE ||
+ version == VERSION_HIDREAM_O1 ||
+ sd_version_is_minit2i(version) ||
+ sd_version_is_sensenova_u1(version) ||
+ sd_version_is_z_image_l2p(version) ||
+ version == VERSION_IRIS) {
LOG_INFO("using FakeVAE");
result.vae = std::make_shared<FakeVAE>(version,
ctx.backends.runtime_backend(SDBackendModule::VAE),