Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions docs/qwen_image_2.1.md
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,14 @@ Pass the reference image with `-r` and describe the edit in `-p`. Vision weights

For multiple reference images, repeat `-r` in the desired order, for example `-r first.png -r second.png`.

### Prefix cache

By default, the first denoising call for each fixed condition saves the text and reference-image keys and values from every transformer layer. Later calls only compute the target-image tokens. Positive and negative conditions use separate caches, which are released when sampling ends.

The cache uses FP32 on all attention backends. For the default 32-layer model, a prefix of 4096 tokens takes about 4 GiB per condition, in addition to weights and working buffers. The runner accounts for the cache when checking the memory budget. If a cached execution runs out of memory, it releases the prefix caches, disables caching for the rest of that sampling run, and retries the full sequence once. Per-step conditioning extensions currently use the full-sequence path.

Disable this optimization with `--model-args qwen_image_2_1_prefix_cache=false`. It reuses step-independent activations; numerical results can still differ slightly because the matrix sizes change.

### Alpha channel

This model supports alpha channel output. As the model determines whether to output a regular image or with transparency through the prompt, according to [official recommendation](https://github.com/QwenLM/Qwen-Image-2.1#transparent-image-generation-rgba), use the following prompt format for better results:
Expand Down
2 changes: 1 addition & 1 deletion examples/common/common.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -518,7 +518,7 @@ ArgOptions SDContextParams::get_options() {
{"",
"--model-args",
"extra model args, key=value list. Supports chroma_use_dit_mask, chroma_use_t5_mask, "
"chroma_t5_mask_pad, qwen_image_zero_cond_t",
"chroma_t5_mask_pad, qwen_image_zero_cond_t, qwen_image_2_1_prefix_cache",
(int)',',
&model_args},
{"",
Expand Down
18 changes: 14 additions & 4 deletions src/core/ggml_runner.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -644,6 +644,10 @@ std::optional<sd::Tensor<float>> GGMLRunner::compute(get_graph_cb_t get_graph,
std::optional<sd::Tensor<float>> output;
try {
output = execute_graph(graph, n_threads, no_return, read_outputs);
} catch (const std::bad_alloc&) {
last_compute_status_ = GGML_STATUS_ALLOC_FAILED;
LOG_ERROR("%s graph allocation failed", get_desc().c_str());
return std::nullopt;
} catch (const std::exception& error) {
last_compute_status_ = GGML_STATUS_FAILED;
LOG_ERROR("%s graph execution failed on %s: %s", get_desc().c_str(),
Expand Down Expand Up @@ -964,10 +968,16 @@ std::optional<Tensor<float>> GGMLRunner::execute_graph(ggml_cgraph* graph, int n
}
LOG_DEBUG("%s executing segment %zu/%zu: %s", get_desc().c_str(),
index + 1, plan.segments.size(), segment.group_name.c_str());
if (!execute_segment(segment_graph, n_threads) ||
!cache_.capture(segment_graph) ||
!cut_cache_.capture(graph, segment, get_desc().c_str())) {
return fail_segment("execution or output caching");
if (!execute_segment(segment_graph, n_threads)) {
return fail_segment("execution");
}
auto cache_status = cache_.capture(segment_graph);
if (cache_status == GGML_STATUS_SUCCESS) {
cache_status = cut_cache_.capture(graph, segment, get_desc().c_str());
}
if (cache_status != GGML_STATUS_SUCCESS) {
last_compute_status_ = cache_status;
return fail_segment("output caching");
}
sync_runtime_residency();
if (last) {
Expand Down
30 changes: 18 additions & 12 deletions src/core/runner_cache.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -26,10 +26,13 @@ namespace sd {

std::unique_ptr<CachedTensor> CachedTensor::copy(ggml_backend_t backend,
const std::string& name,
ggml_tensor* source) {
ggml_tensor* source,
ggml_status& status) {
status = GGML_STATUS_FAILED;
if (ggml_graph_cut::tensor_buffer(source) == nullptr) {
return nullptr;
}
status = GGML_STATUS_ALLOC_FAILED;
auto entry = std::make_unique<CachedTensor>();
entry->context = ggml_init({2 * ggml_tensor_overhead(), nullptr, true});
if (entry->context == nullptr) {
Expand All @@ -50,6 +53,7 @@ namespace sd {
} else {
ggml_backend_tensor_copy(source, entry->tensor);
}
status = GGML_STATUS_SUCCESS;
return entry;
}

Expand Down Expand Up @@ -106,24 +110,25 @@ namespace sd {
return pending > SIZE_MAX - committed ? SIZE_MAX : committed + pending;
}

bool RunnerCache::capture(ggml_cgraph* graph) {
ggml_status RunnerCache::capture(ggml_cgraph* graph) {
if (outputs_.empty()) {
return true;
return GGML_STATUS_SUCCESS;
}
const auto tensors = cache_graph_tensors(graph);
for (const auto& output : outputs_) {
if (pending_.count(output.first) || !tensors.count(output.second)) {
continue;
}
GGML_ASSERT(ggml_is_contiguous(output.second));
auto entry = CachedTensor::copy(backend_, output.first, output.second);
ggml_status status;
auto entry = CachedTensor::copy(backend_, output.first, output.second, status);
if (entry == nullptr) {
return false;
return status;
}
pending_[output.first] = std::move(entry);
}
ggml_backend_synchronize(backend_);
return true;
return GGML_STATUS_SUCCESS;
}

void RunnerCache::graph_end(bool success) {
Expand Down Expand Up @@ -180,9 +185,9 @@ namespace sd {
}
}

bool GraphCutTensorCache::capture(ggml_cgraph* graph,
const ggml_graph_cut::Segment& segment,
const char* log_desc) {
ggml_status GraphCutTensorCache::capture(ggml_cgraph* graph,
const ggml_graph_cut::Segment& segment,
const char* log_desc) {
size_t copied_bytes = 0;
size_t copied_count = 0;
for (int index : segment.output_node_indices) {
Expand All @@ -191,10 +196,11 @@ namespace sd {
!segment.future_cut_names.count(output->name)) {
continue;
}
auto entry = CachedTensor::copy(backend_, output->name, ggml_graph_cut::cache_source_tensor(output));
ggml_status status;
auto entry = CachedTensor::copy(backend_, output->name, ggml_graph_cut::cache_source_tensor(output), status);
if (entry == nullptr) {
LOG_ERROR("%s failed to capture graph cut tensor: %s", log_desc, output->name);
return false;
return status;
}
const size_t size = ggml_backend_buffer_get_size(entry->buffer);
copied_bytes = size > SIZE_MAX - copied_bytes ? SIZE_MAX : copied_bytes + size;
Expand All @@ -206,6 +212,6 @@ namespace sd {
LOG_DEBUG("%s graph cut cache added %6.2f MB (%zu tensors)",
log_desc, copied_bytes / (1024.f * 1024.f), copied_count);
}
return true;
return GGML_STATUS_SUCCESS;
}
}
8 changes: 5 additions & 3 deletions src/core/runner_cache.h
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,8 @@ namespace sd {
~CachedTensor();
static std::unique_ptr<CachedTensor> copy(ggml_backend_t backend,
const std::string& name,
ggml_tensor* source);
ggml_tensor* source,
ggml_status& status);
};
using CachedTensors = std::map<std::string, std::unique_ptr<CachedTensor>>;

Expand All @@ -41,7 +42,8 @@ namespace sd {
const std::map<std::string, ggml_tensor*>& outputs() const { return outputs_; }
size_t pending_bytes(ggml_cgraph* graph) const;
size_t resident_bytes(ggml_backend_dev_t device) const;
bool capture(ggml_cgraph* graph);
bool empty() const { return committed_.empty(); }
ggml_status capture(ggml_cgraph* graph);
void graph_end(bool success);
void clear();
};
Expand All @@ -57,7 +59,7 @@ namespace sd {
size_t resident_bytes(ggml_backend_dev_t device) const;
size_t estimate_output_bytes(ggml_cgraph* graph,
const ggml_graph_cut::Segment& segment) const;
bool capture(ggml_cgraph* graph, const ggml_graph_cut::Segment& segment, const char* log_desc);
ggml_status capture(ggml_cgraph* graph, const ggml_graph_cut::Segment& segment, const char* log_desc);
void prune(const std::unordered_set<std::string>& keep_names);
void clear() { tensors_.clear(); }
};
Expand Down
2 changes: 2 additions & 0 deletions src/model/diffusion/model.hpp
Original file line number Diff line number Diff line change
Expand Up @@ -71,6 +71,8 @@ struct AnimaDiffusionExtra {

struct QwenImage21DiffusionExtra {
const sd::Tensor<int32_t>* image_slots = nullptr;
// Nonzero IDs identify immutable prefix inputs within one sampling run.
uint64_t prefix_id = 0;
};

struct WanDiffusionExtra {
Expand Down
Loading
Loading