Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -67,6 +67,7 @@ API and command-line option may change frequently.***
- [LLaDA-Image](./docs/llada_image.md)
- [Ming-Image Design](./docs/ming_image.md)
- [PixArt](./docs/pixart.md)
- [Iris-3B](./docs/iris.md)
- [Image Edit Models](./docs/edit.md)
- [FLUX.1-Kontext-dev](./docs/kontext.md)
- [Qwen Image Edit series](./docs/qwen_image_edit.md)
Expand Down
Binary file added assets/iris/example.png
Loading
Sorry, something went wrong. Reload?
Sorry, we cannot display this file.
Sorry, this file is invalid so it cannot be displayed.
24 changes: 24 additions & 0 deletions docs/iris.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,24 @@
# How to Use

Iris-3B generates images directly in pixel space and uses Qwen3-VL-4B-Instruct as the text encoder. No VAE is required.

## Download weights

- Download Iris-3B
- safetensors: https://huggingface.co/speridlabs/iris-3b/tree/main (`model.safetensors` in the root directory)
- Download Qwen3-VL-4B-Instruct
- safetensors: https://huggingface.co/Comfy-Org/Krea-2/tree/main/text_encoders
- gguf: https://huggingface.co/Qwen/Qwen3-VL-4B-Instruct-GGUF/tree/main

## Examples

```
.\bin\Release\sd-cli.exe --diffusion-model ..\models\diffusion_models\iris-3b.safetensors --llm ..\models\text_encoders\Qwen3-VL-4B-Instruct-Q4_K_M.gguf -p "a lovely cat" --cfg-scale 3 -H 1024 -W 1024 --diffusion-fa
```

<img width="256" alt="iris-3b example" src="../assets/iris/example.png" />

## Notes

- Width and height must be multiples of 16. Do not pass `--vae`.
- Captions are limited to 300 tokens including the assistant-turn suffix. Positive and negative prompts use the same template.
60 changes: 53 additions & 7 deletions src/conditioning/conditioner.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -2012,7 +2012,8 @@ struct LLMEmbedder : public Conditioner {
sd_version_is_sefi_image(version) ||
sd_version_is_krea2(version) ||
sd_version_is_minimax_h3(version) ||
sd_version_is_mage_flow(version)) {
sd_version_is_mage_flow(version) ||
version == VERSION_IRIS) {
arch = LLM::LLMArch::QWEN3_VL;
} else if (sd_version_is_z_image(version) || sd_version_is_z_image_l2p(version) || version == VERSION_OVIS_IMAGE || version == VERSION_FLUX2_KLEIN) {
arch = LLM::LLMArch::QWEN3;
Expand Down Expand Up @@ -2138,9 +2139,10 @@ struct LLMEmbedder : public Conditioner {

std::tuple<std::vector<int>, std::vector<float>, std::vector<float>> tokenize(std::string text,
const std::pair<int, int>& attn_range,
size_t min_length = 0,
size_t max_length = 100000000,
bool spell_quotes = false) {
size_t min_length = 0,
size_t max_length = 100000000,
bool spell_quotes = false,
const std::string& suffix = "") {
std::vector<std::pair<std::string, float>> parsed_attention;
if (attn_range.first >= 0 && attn_range.second > 0) {
if (attn_range.first > 0) {
Expand Down Expand Up @@ -2185,6 +2187,20 @@ struct LLMEmbedder : public Conditioner {
weights.insert(weights.end(), curr_tokens.size(), curr_weight);
}

if (!suffix.empty()) {
std::vector<int> suffix_tokens;
if (!tokenizer->encode(suffix, suffix_tokens) || suffix_tokens.size() >= max_length) {
return {};
}
// Reserve the assistant-turn marker before truncating the caption.
if (tokens.size() > max_length - suffix_tokens.size()) {
tokens.resize(max_length - suffix_tokens.size());
weights.resize(tokens.size());
}
tokens.insert(tokens.end(), suffix_tokens.begin(), suffix_tokens.end());
weights.insert(weights.end(), suffix_tokens.size(), 1.f);
}

Comment thread
leejet marked this conversation as resolved.
std::vector<float> mask;
tokenizer->pad_tokens(tokens, &weights, &mask, min_length, max_length);

Expand All @@ -2207,8 +2223,10 @@ struct LLMEmbedder : public Conditioner {
bool spell_quotes = false,
int max_length = 100000000,
const LLM::DeepStackImageEmbeds& deepstack_image_embeds = {},
const std::vector<LLM::ImageGrid>& image_grids = {}) {
auto tokens_weights_mask = tokenize(prompt, prompt_attn_range, min_length, max_length, spell_quotes);
const std::vector<LLM::ImageGrid>& image_grids = {},
const std::string& prompt_suffix = "",
sd::Tensor<float>* output_mask = nullptr) {
auto tokens_weights_mask = tokenize(prompt, prompt_attn_range, min_length, max_length, spell_quotes, prompt_suffix);
auto& tokens = std::get<0>(tokens_weights_mask);
auto& weights = std::get<1>(tokens_weights_mask);
auto& mask = std::get<2>(tokens_weights_mask);
Expand Down Expand Up @@ -2271,6 +2289,12 @@ struct LLMEmbedder : public Conditioner {
1);
}

if (output_mask != nullptr) {
*output_mask = sd::Tensor<float>::zeros({new_hidden_states.shape()[1]});
for (size_t i = prompt_template_encode_start_idx; i < mask.size(); ++i) {
output_mask->data()[i - prompt_template_encode_start_idx] = mask[i];
}
}
return new_hidden_states;
}

Expand Down Expand Up @@ -2337,6 +2361,8 @@ struct LLMEmbedder : public Conditioner {
int hidden_states_min_length = 0; // zero pad hidden_states
bool spell_quotes = false;
std::set<int> out_layers;
std::string prompt_suffix;
sd::Tensor<float> text_mask;

int64_t t0 = ggml_time_ms();
RefImageResizeMode resize_mode = conditioner_params.ref_image_params.vlm_resize_mode;
Expand Down Expand Up @@ -3086,6 +3112,23 @@ struct LLMEmbedder : public Conditioner {
SDCondition result;
result.c_crossattn = std::move(hidden_states);
return result;
} else if (version == VERSION_IRIS) {
prompt =
"<|im_start|>system\n"
"Describe the image by detailing the color, shape, size, texture, quantity, text, spatial "
"relationships of the objects and background:<|im_end|>\n<|im_start|>user\n";
auto prefix_tokens = std::get<0>(tokenize(prompt, {0, 0}));
if (prefix_tokens.empty()) {
return {};
}
prompt_template_encode_start_idx = static_cast<int>(prefix_tokens.size());
hidden_states_min_length = 300;
max_length = prompt_template_encode_start_idx + hidden_states_min_length;
out_layers = {2, 5, 8, 11, 14, 17, 20, 23, 26, 29, 32, 35};
prompt_attn_range.first = static_cast<int>(prompt.size());
prompt += conditioner_params.text;
prompt_attn_range.second = static_cast<int>(prompt.size());
prompt_suffix = "<|im_end|>\n<|im_start|>assistant\n";
} else {
GGML_ABORT("unknown version %d", version);
}
Expand All @@ -3101,7 +3144,9 @@ struct LLMEmbedder : public Conditioner {
spell_quotes,
max_length,
deepstack_image_embeds,
image_grids);
image_grids,
prompt_suffix,
version == VERSION_IRIS ? &text_mask : nullptr);
if (hidden_states.empty()) {
return {};
}
Expand Down Expand Up @@ -3166,6 +3211,7 @@ struct LLMEmbedder : public Conditioner {
SDCondition result;
result.c_crossattn = std::move(hidden_states);
result.extra_c_crossattns = std::move(extra_hidden_states_vec);
result.c_vector = std::move(text_mask);
if (version == VERSION_QWEN_IMAGE_2_1) {
auto slots = sd::Tensor<int32_t>::zeros({result.c_crossattn.shape()[1]});
for (size_t i = 0; i < image_embeds.size(); ++i) {
Expand Down
4 changes: 3 additions & 1 deletion src/model.h
Original file line number Diff line number Diff line change
Expand Up @@ -65,6 +65,7 @@ enum SDVersion {
VERSION_PIXART,
VERSION_MING_IMAGE,
VERSION_Z_IMAGE_L2P,
VERSION_IRIS,
VERSION_COUNT,
};

Expand Down Expand Up @@ -334,7 +335,8 @@ static inline bool sd_version_is_dit(SDVersion version) {
sd_version_is_krea2(version) ||
sd_version_is_mage_flow(version) ||
sd_version_is_sensenova_u1(version) ||
sd_version_is_pixart(version)) {
sd_version_is_pixart(version) ||
version == VERSION_IRIS) {
return true;
}
return false;
Expand Down
Loading
Loading