From 56ba2f14ea44617e94aac75e4926294bdd279064 Mon Sep 17 00:00:00 2001 From: assouan <750048+assouan@users.noreply.github.com> Date: Sat, 22 Aug 2026 01:03:11 +0200 Subject: [PATCH 1/4] fix: cache MiniMax-H3 refined context during sampling Run condition projection and both token-refiner blocks once per conditioning context, then reuse the refined output across denoising steps. Keep entries distinct by condition and active weight adapter. Split each refiner block into its own streaming segment and clear the sampling-scoped cache when sampling finishes. --- src/model/diffusion/minimax_h3.hpp | 145 ++++++++++++++++++++++++++++- src/model/diffusion/model.hpp | 7 ++ src/stable-diffusion.cpp | 21 +++-- 3 files changed, 165 insertions(+), 8 deletions(-) diff --git a/src/model/diffusion/minimax_h3.hpp b/src/model/diffusion/minimax_h3.hpp index d0683166df..143db3e8dc 100644 --- a/src/model/diffusion/minimax_h3.hpp +++ b/src/model/diffusion/minimax_h3.hpp @@ -3,6 +3,7 @@ #include #include +#include #include #include #include @@ -267,11 +268,18 @@ namespace MiniMaxH3 { } ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) { + auto final_norm = std::dynamic_pointer_cast(blocks["final_norm"]); for (int64_t i = 0; i < num_layers; ++i) { auto block = std::dynamic_pointer_cast(blocks["blocks." + std::to_string(i)]); x = block->forward(ctx, x); + if (i + 1 == num_layers) { + x = final_norm->forward(ctx, x); + } + sd::ggml_graph_cut::mark_graph_cut(x, + "minimax_h3.token_refiner.blocks." + std::to_string(i), + "hidden_states"); } - return std::dynamic_pointer_cast(blocks["final_norm"])->forward(ctx, x); + return num_layers == 0 ? final_norm->forward(ctx, x) : x; } }; @@ -952,6 +960,34 @@ namespace MiniMaxH3 { } struct MiniMaxH3Runner : public DiffusionModelRunner { + struct RefinedContextCacheEntry { + const void* condition_identity = nullptr; + std::shared_ptr weight_adapter = nullptr; + ggml_context* refined_ctx = nullptr; + ggml_backend_buffer_t refined_buffer = nullptr; + ggml_tensor* refined = nullptr; + + ~RefinedContextCacheEntry() { + if (refined_buffer != nullptr) { + ggml_backend_buffer_free(refined_buffer); + } + if (refined_ctx != nullptr) { + ggml_free(refined_ctx); + } + } + + RefinedContextCacheEntry() = default; + RefinedContextCacheEntry(const RefinedContextCacheEntry&) = delete; + RefinedContextCacheEntry& operator=(const RefinedContextCacheEntry&) = delete; + + bool matches(const void* identity, + const std::shared_ptr& adapter) const { + return condition_identity == identity && weight_adapter == adapter; + } + }; + + static constexpr size_t REFINED_CONTEXT_CACHE_CAPACITY = 4; + Config config; MiniMaxH3Transformer3DModel model; sd::Tensor video_input_cache; @@ -961,6 +997,7 @@ namespace MiniMaxH3 { sd::Tensor curve_index_input_cache; sd::Tensor curve_upper_index_input_cache; sd::Tensor curve_fraction_input_cache; + std::vector> refined_context_cache; MiniMaxH3Runner(ggml_backend_t backend, const String2TensorStorage& tensors, @@ -981,6 +1018,92 @@ namespace MiniMaxH3 { model.get_param_tensors(tensors, prefix); } + std::unique_ptr create_refined_context_cache_entry( + const sd::Tensor& context, + const void* condition_identity) { + auto entry = std::make_unique(); + entry->condition_identity = condition_identity; + entry->weight_adapter = weight_adapter; + + auto refined_shape = context.shape(); + refined_shape[0] = config.hidden_size; + ggml_init_params params; + params.mem_size = ggml_tensor_overhead(); + params.mem_buffer = nullptr; + params.no_alloc = true; + entry->refined_ctx = ggml_init(params); + GGML_ASSERT(entry->refined_ctx != nullptr); + entry->refined = ggml_new_tensor(entry->refined_ctx, + GGML_TYPE_F32, + static_cast(refined_shape.size()), + refined_shape.data()); + ggml_set_name(entry->refined, "minimax_h3.refined_context"); + entry->refined_buffer = ggml_backend_alloc_ctx_tensors(entry->refined_ctx, + runtime_backend); + GGML_ASSERT(entry->refined_buffer != nullptr); + ggml_backend_buffer_set_usage(entry->refined_buffer, + GGML_BACKEND_BUFFER_USAGE_WEIGHTS); + return entry; + } + + ggml_cgraph* build_context_refinement_graph(const sd::Tensor& context, + ggml_tensor* refined_output) { + GGML_ASSERT(!context.empty() && context.shape()[0] == config.text_dim); + GGML_ASSERT(refined_output != nullptr && refined_output->ne[0] == config.hidden_size); + auto context_input = make_input(context); + auto runner_ctx = get_context(); + auto refined = model.refine_context(&runner_ctx, context_input); + // Refinement graph buffers are transient; persist only their final output. + auto output = ggml_cpy(runner_ctx.ggml_ctx, refined, refined_output); + auto graph = new_graph_custom(H3_GRAPH_SIZE); + ggml_build_forward_expand(graph, output); + return graph; + } + + ggml_tensor* get_refined_context(const sd::Tensor& context, + const void* condition_identity, + int n_threads) { + GGML_ASSERT(!context.empty()); + GGML_ASSERT(condition_identity != nullptr); + GGML_ASSERT(context.shape()[0] == config.text_dim || + context.shape()[0] == config.hidden_size); + + for (const auto& entry : refined_context_cache) { + if (entry->matches(condition_identity, weight_adapter)) { + return entry->refined; + } + } + + auto entry = create_refined_context_cache_entry(context, condition_identity); + if (context.shape()[0] == config.hidden_size) { + ggml_backend_tensor_set(entry->refined, + context.data(), + 0, + ggml_nbytes(entry->refined)); + ggml_backend_synchronize(runtime_backend); + } else { + auto get_graph = [&]() { + return build_context_refinement_graph(context, entry->refined); + }; + auto result = GGMLRunner::compute(get_graph, + n_threads, + false, + true, + true, + true); + if (!result.has_value()) { + return nullptr; + } + } + + auto refined = entry->refined; + if (refined_context_cache.size() == REFINED_CONTEXT_CACHE_CAPACITY) { + refined_context_cache.erase(refined_context_cache.begin()); + } + refined_context_cache.push_back(std::move(entry)); + return refined; + } + std::pair, sd::Tensor> split_av_latents(const sd::Tensor& packed, int audio_length) const { GGML_ASSERT(packed.dim() == 4 || packed.dim() == 5); @@ -1026,6 +1149,7 @@ namespace MiniMaxH3 { ggml_cgraph* build_graph(const sd::Tensor& packed, const sd::Tensor& timestep, const sd::Tensor& context_tensor, + ggml_tensor* refined_context, const std::vector>& condition_videos, const std::vector>& condition_audios, const sd::Tensor& text_tags, @@ -1039,10 +1163,12 @@ namespace MiniMaxH3 { audio_input_cache = std::move(split.second); GGML_ASSERT(!audio_input_cache.empty()); GGML_ASSERT(!context_tensor.empty()); + GGML_ASSERT(refined_context != nullptr && + refined_context->ne[0] == config.hidden_size); auto video = make_input(video_input_cache); auto audio = make_input(audio_input_cache); - auto context = make_input(context_tensor); + auto context = refined_context; std::vector condition_inputs; condition_inputs.reserve(condition_videos.size()); for (const auto& condition : condition_videos) { @@ -1151,10 +1277,20 @@ namespace MiniMaxH3 { ? empty_reference_blocks : *extra->reference_blocks; const sd::Tensor empty_int; + const void* condition_identity = params.condition_identity != nullptr + ? params.condition_identity + : params.context; + auto context = get_refined_context(*params.context, + condition_identity, + n_threads); + if (context == nullptr) { + return {}; + } auto get_graph = [&]() { return build_graph(*params.x, *params.timesteps, *params.context, + context, conditions, audio_conditions, extra->text_token_tags == nullptr ? empty_int : *extra->text_token_tags, @@ -1171,6 +1307,11 @@ namespace MiniMaxH3 { false), params.x->dim()); } + + protected: + void on_sampling_done() override { + refined_context_cache.clear(); + } }; } // namespace MiniMaxH3 diff --git a/src/model/diffusion/model.hpp b/src/model/diffusion/model.hpp index 070ca53d4e..f8fde4df0d 100644 --- a/src/model/diffusion/model.hpp +++ b/src/model/diffusion/model.hpp @@ -137,6 +137,7 @@ struct DiffusionParams { const sd::Tensor* x = nullptr; const sd::Tensor* timesteps = nullptr; const sd::Tensor* context = nullptr; + const void* condition_identity = nullptr; const sd::Tensor* c_concat = nullptr; const sd::Tensor* y = nullptr; const std::vector>* ref_latents = nullptr; @@ -160,6 +161,7 @@ static inline const sd::Tensor& tensor_or_empty(const sd::Tensor* tensor) struct DiffusionModelRunner : public GGMLRunner { protected: std::string prefix; + virtual void on_sampling_done() {} public: DiffusionModelRunner(ggml_backend_t backend, @@ -171,6 +173,11 @@ struct DiffusionModelRunner : public GGMLRunner { virtual sd::Tensor compute(int n_threads, const DiffusionParams& diffusion_params) = 0; + void sampling_done() { + runner_done(); + on_sampling_done(); + } + void get_param_tensors(std::map& tensors) { get_param_tensors(tensors, prefix); } diff --git a/src/stable-diffusion.cpp b/src/stable-diffusion.cpp index 109a2a483a..cf517131e1 100644 --- a/src/stable-diffusion.cpp +++ b/src/stable-diffusion.cpp @@ -2531,6 +2531,16 @@ class StableDiffusionGGML { float frame_rate, const sd_cache_params_t* cache_params, const sd::Tensor& video_positions = {}) { + struct SamplingDoneOnExit { + DiffusionModelRunner* runner = nullptr; + ~SamplingDoneOnExit() { + if (runner != nullptr) { + runner->sampling_done(); + } + } + }; + SamplingDoneOnExit sample_diffusion_runner_done{work_diffusion_model.get()}; + struct RunnerDoneOnExit { GGMLRunner* runner = nullptr; ~RunnerDoneOnExit() { @@ -2539,8 +2549,6 @@ class StableDiffusionGGML { } } }; - RunnerDoneOnExit sample_diffusion_runner_done{work_diffusion_model.get()}; - RunnerDoneOnExit sample_control_runner_done{!control_image.empty() && control_net != nullptr ? control_net.get() : nullptr}; std::vector skip_layers(guidance.slg.layers, guidance.slg.layers + guidance.slg.layer_count); @@ -2712,10 +2720,11 @@ class StableDiffusionGGML { const std::vector* local_skip_layers = nullptr, const std::vector>* ref_latents_override = nullptr, bool use_uncond_ip = false) -> sd::Tensor { - diffusion_params.context = condition.c_crossattn.empty() ? nullptr : &condition.c_crossattn; - diffusion_params.c_concat = c_concat_override != nullptr ? c_concat_override : (condition.c_concat.empty() ? nullptr : &condition.c_concat); - diffusion_params.y = condition.c_vector.empty() ? nullptr : &condition.c_vector; - diffusion_params.ref_latents = ref_latents_override != nullptr ? ref_latents_override : (condition.c_ref_images.empty() ? &ref_latents : &condition.c_ref_images); + diffusion_params.context = condition.c_crossattn.empty() ? nullptr : &condition.c_crossattn; + diffusion_params.condition_identity = &condition; + diffusion_params.c_concat = c_concat_override != nullptr ? c_concat_override : (condition.c_concat.empty() ? nullptr : &condition.c_concat); + diffusion_params.y = condition.c_vector.empty() ? nullptr : &condition.c_vector; + diffusion_params.ref_latents = ref_latents_override != nullptr ? ref_latents_override : (condition.c_ref_images.empty() ? &ref_latents : &condition.c_ref_images); if (sd_version_is_unet(version)) { int nvf = -1; From 34a3146c6948f4cd0039af9f1ed427c7308947ef Mon Sep 17 00:00:00 2001 From: assouan <750048+assouan@users.noreply.github.com> Date: Sat, 22 Aug 2026 02:48:45 +0200 Subject: [PATCH 2/4] fix: harden MiniMax-H3 refined context streaming Split condition projection and token refiners into sequential streamed segments while folding the persistent copy into the final refiner segment. Tie cached refined contexts to the source tensor identity, storage, shape, and active weight adapter. --- src/model/diffusion/minimax_h3.hpp | 70 ++++++++++++++++++++---------- src/model/diffusion/model.hpp | 2 +- src/stable-diffusion.cpp | 10 ++--- 3 files changed, 52 insertions(+), 30 deletions(-) diff --git a/src/model/diffusion/minimax_h3.hpp b/src/model/diffusion/minimax_h3.hpp index 143db3e8dc..c3b9c06902 100644 --- a/src/model/diffusion/minimax_h3.hpp +++ b/src/model/diffusion/minimax_h3.hpp @@ -267,17 +267,22 @@ namespace MiniMaxH3 { blocks["final_norm"] = std::make_shared(config.hidden_size, config.final_norm_eps); } - ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) { + ggml_tensor* forward(GGMLRunnerContext* ctx, + ggml_tensor* x, + bool cut_after_last = true) { auto final_norm = std::dynamic_pointer_cast(blocks["final_norm"]); for (int64_t i = 0; i < num_layers; ++i) { auto block = std::dynamic_pointer_cast(blocks["blocks." + std::to_string(i)]); x = block->forward(ctx, x); - if (i + 1 == num_layers) { + const bool is_last = i + 1 == num_layers; + if (is_last) { x = final_norm->forward(ctx, x); } - sd::ggml_graph_cut::mark_graph_cut(x, - "minimax_h3.token_refiner.blocks." + std::to_string(i), - "hidden_states"); + if (!is_last || cut_after_last) { + sd::ggml_graph_cut::mark_graph_cut(x, + "minimax_h3.token_refiner.blocks." + std::to_string(i), + "hidden_states"); + } } return num_layers == 0 ? final_norm->forward(ctx, x) : x; } @@ -534,14 +539,20 @@ namespace MiniMaxH3 { } } - ggml_tensor* refine_context(GGMLRunnerContext* ctx, ggml_tensor* context) { + ggml_tensor* refine_context(GGMLRunnerContext* ctx, + ggml_tensor* context, + bool cut_after_last_refiner = true) { if (context->ne[0] == config.hidden_size) { return context; } GGML_ASSERT(context->ne[0] == config.text_dim); auto condition_proj = std::dynamic_pointer_cast(blocks["condition_proj"]); auto token_refiner = std::dynamic_pointer_cast(blocks["token_refiner"]); - return token_refiner->forward(ctx, condition_proj->forward(ctx, context)); + auto projected = condition_proj->forward(ctx, context); + sd::ggml_graph_cut::mark_graph_cut(projected, + "minimax_h3.condition_proj", + "hidden_states"); + return token_refiner->forward(ctx, projected, cut_after_last_refiner); } ggml_tensor* time_embedding(GGMLRunnerContext* ctx, @@ -961,7 +972,10 @@ namespace MiniMaxH3 { struct MiniMaxH3Runner : public DiffusionModelRunner { struct RefinedContextCacheEntry { - const void* condition_identity = nullptr; + const void* context_cache_identity = nullptr; + const sd::Tensor* source_context = nullptr; + const float* source_data = nullptr; + std::vector source_shape; std::shared_ptr weight_adapter = nullptr; ggml_context* refined_ctx = nullptr; ggml_backend_buffer_t refined_buffer = nullptr; @@ -981,8 +995,13 @@ namespace MiniMaxH3 { RefinedContextCacheEntry& operator=(const RefinedContextCacheEntry&) = delete; bool matches(const void* identity, + const sd::Tensor& context, const std::shared_ptr& adapter) const { - return condition_identity == identity && weight_adapter == adapter; + return context_cache_identity == identity && + weight_adapter == adapter && + source_context == &context && + source_data == context.data() && + source_shape == context.shape(); } }; @@ -1020,10 +1039,13 @@ namespace MiniMaxH3 { std::unique_ptr create_refined_context_cache_entry( const sd::Tensor& context, - const void* condition_identity) { - auto entry = std::make_unique(); - entry->condition_identity = condition_identity; - entry->weight_adapter = weight_adapter; + const void* context_cache_identity) { + auto entry = std::make_unique(); + entry->context_cache_identity = context_cache_identity; + entry->source_context = &context; + entry->source_data = context.data(); + entry->source_shape = context.shape(); + entry->weight_adapter = weight_adapter; auto refined_shape = context.shape(); refined_shape[0] = config.hidden_size; @@ -1052,7 +1074,7 @@ namespace MiniMaxH3 { GGML_ASSERT(refined_output != nullptr && refined_output->ne[0] == config.hidden_size); auto context_input = make_input(context); auto runner_ctx = get_context(); - auto refined = model.refine_context(&runner_ctx, context_input); + auto refined = model.refine_context(&runner_ctx, context_input, false); // Refinement graph buffers are transient; persist only their final output. auto output = ggml_cpy(runner_ctx.ggml_ctx, refined, refined_output); auto graph = new_graph_custom(H3_GRAPH_SIZE); @@ -1061,20 +1083,20 @@ namespace MiniMaxH3 { } ggml_tensor* get_refined_context(const sd::Tensor& context, - const void* condition_identity, + const void* context_cache_identity, int n_threads) { GGML_ASSERT(!context.empty()); - GGML_ASSERT(condition_identity != nullptr); + GGML_ASSERT(context_cache_identity != nullptr); GGML_ASSERT(context.shape()[0] == config.text_dim || context.shape()[0] == config.hidden_size); for (const auto& entry : refined_context_cache) { - if (entry->matches(condition_identity, weight_adapter)) { + if (entry->matches(context_cache_identity, context, weight_adapter)) { return entry->refined; } } - auto entry = create_refined_context_cache_entry(context, condition_identity); + auto entry = create_refined_context_cache_entry(context, context_cache_identity); if (context.shape()[0] == config.hidden_size) { ggml_backend_tensor_set(entry->refined, context.data(), @@ -1277,12 +1299,12 @@ namespace MiniMaxH3 { ? empty_reference_blocks : *extra->reference_blocks; const sd::Tensor empty_int; - const void* condition_identity = params.condition_identity != nullptr - ? params.condition_identity - : params.context; - auto context = get_refined_context(*params.context, - condition_identity, - n_threads); + const void* context_cache_identity = params.context_cache_identity != nullptr + ? params.context_cache_identity + : params.context; + auto context = get_refined_context(*params.context, + context_cache_identity, + n_threads); if (context == nullptr) { return {}; } diff --git a/src/model/diffusion/model.hpp b/src/model/diffusion/model.hpp index f8fde4df0d..f895e5eae9 100644 --- a/src/model/diffusion/model.hpp +++ b/src/model/diffusion/model.hpp @@ -137,7 +137,7 @@ struct DiffusionParams { const sd::Tensor* x = nullptr; const sd::Tensor* timesteps = nullptr; const sd::Tensor* context = nullptr; - const void* condition_identity = nullptr; + const void* context_cache_identity = nullptr; const sd::Tensor* c_concat = nullptr; const sd::Tensor* y = nullptr; const std::vector>* ref_latents = nullptr; diff --git a/src/stable-diffusion.cpp b/src/stable-diffusion.cpp index cf517131e1..c7587b0e32 100644 --- a/src/stable-diffusion.cpp +++ b/src/stable-diffusion.cpp @@ -2720,11 +2720,11 @@ class StableDiffusionGGML { const std::vector* local_skip_layers = nullptr, const std::vector>* ref_latents_override = nullptr, bool use_uncond_ip = false) -> sd::Tensor { - diffusion_params.context = condition.c_crossattn.empty() ? nullptr : &condition.c_crossattn; - diffusion_params.condition_identity = &condition; - diffusion_params.c_concat = c_concat_override != nullptr ? c_concat_override : (condition.c_concat.empty() ? nullptr : &condition.c_concat); - diffusion_params.y = condition.c_vector.empty() ? nullptr : &condition.c_vector; - diffusion_params.ref_latents = ref_latents_override != nullptr ? ref_latents_override : (condition.c_ref_images.empty() ? &ref_latents : &condition.c_ref_images); + diffusion_params.context = condition.c_crossattn.empty() ? nullptr : &condition.c_crossattn; + diffusion_params.context_cache_identity = &condition; + diffusion_params.c_concat = c_concat_override != nullptr ? c_concat_override : (condition.c_concat.empty() ? nullptr : &condition.c_concat); + diffusion_params.y = condition.c_vector.empty() ? nullptr : &condition.c_vector; + diffusion_params.ref_latents = ref_latents_override != nullptr ? ref_latents_override : (condition.c_ref_images.empty() ? &ref_latents : &condition.c_ref_images); if (sd_version_is_unet(version)) { int nvf = -1; From 3587885b45b2e8a5e41c827c0e65fa753df880e7 Mon Sep 17 00:00:00 2001 From: leejet Date: Wed, 23 Sep 2026 23:46:38 +0800 Subject: [PATCH 3/4] fix: use RunnerCache for MiniMax-H3 context caching --- docs/minimax_h3.md | 14 ++ src/model/diffusion/minimax_h3.hpp | 233 ++++++++--------------------- src/model/diffusion/model.hpp | 9 +- src/pipeline/diffusion_engine.cpp | 34 ++--- src/pipeline/model_builders.cpp | 3 +- 5 files changed, 98 insertions(+), 195 deletions(-) diff --git a/docs/minimax_h3.md b/docs/minimax_h3.md index 82d0e2ffb4..9b584aeacd 100644 --- a/docs/minimax_h3.md +++ b/docs/minimax_h3.md @@ -94,3 +94,17 @@ frame rate and optional soundtrack; non-24-fps inputs are resampled internally. - MiniMax-H3 runs at 24 fps; another requested value is overridden. - The default video flow shift is 12. The audio stream is mapped internally to its shift of 3, so the regular samplers can operate on the packed AV latent. + +## Text conditioning cache + +The first denoising call for each fixed condition caches the output of +`condition_proj` and `token_refiner`. Later calls reuse it; positive and negative +conditions have separate entries. The cache uses the runner's memory budget and +is released when sampling ends. If cached execution runs out of memory, it clears +the caches, disables caching for the rest of that sampling run, and retries once +without caching. Per-step conditioning extensions use the uncached path. + +Disable this optimization with `--model-args minimax_h3_context_cache=false`. +Only preprocessing of the text conditioning is cached. Reference projections, +RoPE, and the timestep-dependent transformer backbone are recomputed on every +call, including its bidirectional attention keys and values. diff --git a/src/model/diffusion/minimax_h3.hpp b/src/model/diffusion/minimax_h3.hpp index ad88767cbf..36d9e6afbe 100644 --- a/src/model/diffusion/minimax_h3.hpp +++ b/src/model/diffusion/minimax_h3.hpp @@ -262,8 +262,7 @@ namespace MiniMaxH3 { } ggml_tensor* forward(GGMLRunnerContext* ctx, - ggml_tensor* x, - bool cut_after_last = true) { + ggml_tensor* x) { auto final_norm = std::dynamic_pointer_cast(blocks["final_norm"]); for (int64_t i = 0; i < num_layers; ++i) { auto block = std::dynamic_pointer_cast(blocks["blocks." + std::to_string(i)]); @@ -272,11 +271,9 @@ namespace MiniMaxH3 { if (is_last) { x = final_norm->forward(ctx, x); } - if (!is_last || cut_after_last) { - sd::ggml_graph_cut::mark_graph_cut(x, - "minimax_h3.token_refiner.blocks." + std::to_string(i), - "hidden_states"); - } + sd::ggml_graph_cut::mark_graph_cut(x, + "minimax_h3.token_refiner.blocks." + std::to_string(i), + "hidden_states"); } return num_layers == 0 ? final_norm->forward(ctx, x) : x; } @@ -534,8 +531,7 @@ namespace MiniMaxH3 { } ggml_tensor* refine_context(GGMLRunnerContext* ctx, - ggml_tensor* context, - bool cut_after_last_refiner = true) { + ggml_tensor* context) { if (context->ne[0] == config.hidden_size) { return context; } @@ -546,7 +542,7 @@ namespace MiniMaxH3 { sd::ggml_graph_cut::mark_graph_cut(projected, "minimax_h3.condition_proj", "hidden_states"); - return token_refiner->forward(ctx, projected, cut_after_last_refiner); + return token_refiner->forward(ctx, projected); } ggml_tensor* time_embedding(GGMLRunnerContext* ctx, @@ -964,42 +960,6 @@ namespace MiniMaxH3 { } struct MiniMaxH3Runner : public DiffusionModelRunner { - struct RefinedContextCacheEntry { - const void* context_cache_identity = nullptr; - const sd::Tensor* source_context = nullptr; - const float* source_data = nullptr; - std::vector source_shape; - std::shared_ptr weight_adapter = nullptr; - ggml_context* refined_ctx = nullptr; - ggml_backend_buffer_t refined_buffer = nullptr; - ggml_tensor* refined = nullptr; - - ~RefinedContextCacheEntry() { - if (refined_buffer != nullptr) { - ggml_backend_buffer_free(refined_buffer); - } - if (refined_ctx != nullptr) { - ggml_free(refined_ctx); - } - } - - RefinedContextCacheEntry() = default; - RefinedContextCacheEntry(const RefinedContextCacheEntry&) = delete; - RefinedContextCacheEntry& operator=(const RefinedContextCacheEntry&) = delete; - - bool matches(const void* identity, - const sd::Tensor& context, - const std::shared_ptr& adapter) const { - return context_cache_identity == identity && - weight_adapter == adapter && - source_context == &context && - source_data == context.data() && - source_shape == context.shape(); - } - }; - - static constexpr size_t REFINED_CONTEXT_CACHE_CAPACITY = 4; - Config config; MiniMaxH3Transformer3DModel model; sd::Tensor video_input_cache; @@ -1009,15 +969,22 @@ namespace MiniMaxH3 { sd::Tensor curve_index_input_cache; sd::Tensor curve_upper_index_input_cache; sd::Tensor curve_fraction_input_cache; - std::vector> refined_context_cache; + bool context_cache_enabled = true; + bool context_cache_disabled = false; MiniMaxH3Runner(ggml_backend_t backend, const String2TensorStorage& tensors, const std::string& prefix = "model.diffusion_model", - std::shared_ptr weight_manager = nullptr) + std::shared_ptr weight_manager = nullptr, + const char* model_args = nullptr) : DiffusionModelRunner(backend, prefix, weight_manager), config(Config::detect_from_weights(tensors, prefix)), model(config) { + for (const auto& [key, value] : parse_key_value_args(model_args, "model arg")) { + if (key == "minimax_h3_context_cache" && !parse_strict_bool(value, context_cache_enabled)) { + LOG_WARN("ignoring invalid MiniMax-H3 model arg '%s=%s'", key.c_str(), value.c_str()); + } + } model.init(params_ctx, tensors, prefix); } @@ -1030,93 +997,6 @@ namespace MiniMaxH3 { model.get_param_tensors(tensors, prefix); } - std::unique_ptr create_refined_context_cache_entry( - const sd::Tensor& context, - const void* context_cache_identity) { - auto entry = std::make_unique(); - entry->context_cache_identity = context_cache_identity; - entry->source_context = &context; - entry->source_data = context.data(); - entry->source_shape = context.shape(); - entry->weight_adapter = weight_adapter; - - auto refined_shape = context.shape(); - refined_shape[0] = config.hidden_size; - ggml_init_params params; - params.mem_size = ggml_tensor_overhead(); - params.mem_buffer = nullptr; - params.no_alloc = true; - entry->refined_ctx = ggml_init(params); - GGML_ASSERT(entry->refined_ctx != nullptr); - entry->refined = ggml_new_tensor(entry->refined_ctx, - GGML_TYPE_F32, - static_cast(refined_shape.size()), - refined_shape.data()); - ggml_set_name(entry->refined, "minimax_h3.refined_context"); - entry->refined_buffer = ggml_backend_alloc_ctx_tensors(entry->refined_ctx, - runtime_backend); - GGML_ASSERT(entry->refined_buffer != nullptr); - ggml_backend_buffer_set_usage(entry->refined_buffer, - GGML_BACKEND_BUFFER_USAGE_WEIGHTS); - return entry; - } - - ggml_cgraph* build_context_refinement_graph(const sd::Tensor& context, - ggml_tensor* refined_output) { - GGML_ASSERT(!context.empty() && context.shape()[0] == config.text_dim); - GGML_ASSERT(refined_output != nullptr && refined_output->ne[0] == config.hidden_size); - auto context_input = make_input(context); - auto runner_ctx = get_context(); - auto refined = model.refine_context(&runner_ctx, context_input, false); - // Refinement graph buffers are transient; persist only their final output. - auto output = ggml_cpy(runner_ctx.ggml_ctx, refined, refined_output); - auto graph = new_graph_custom(H3_GRAPH_SIZE); - ggml_build_forward_expand(graph, output); - return graph; - } - - ggml_tensor* get_refined_context(const sd::Tensor& context, - const void* context_cache_identity, - int n_threads) { - GGML_ASSERT(!context.empty()); - GGML_ASSERT(context_cache_identity != nullptr); - GGML_ASSERT(context.shape()[0] == config.text_dim || - context.shape()[0] == config.hidden_size); - - for (const auto& entry : refined_context_cache) { - if (entry->matches(context_cache_identity, context, weight_adapter)) { - return entry->refined; - } - } - - auto entry = create_refined_context_cache_entry(context, context_cache_identity); - if (context.shape()[0] == config.hidden_size) { - ggml_backend_tensor_set(entry->refined, - context.data(), - 0, - ggml_nbytes(entry->refined)); - ggml_backend_synchronize(runtime_backend); - } else { - auto get_graph = [&]() { - return build_context_refinement_graph(context, entry->refined); - }; - auto result = GGMLRunner::compute(get_graph, - n_threads, - false, - true); - if (!result.has_value()) { - return nullptr; - } - } - - auto refined = entry->refined; - if (refined_context_cache.size() == REFINED_CONTEXT_CACHE_CAPACITY) { - refined_context_cache.erase(refined_context_cache.begin()); - } - refined_context_cache.push_back(std::move(entry)); - return refined; - } - std::pair, sd::Tensor> split_av_latents(const sd::Tensor& packed, int audio_length) const { GGML_ASSERT(packed.dim() == 4 || packed.dim() == 5); @@ -1162,7 +1042,7 @@ namespace MiniMaxH3 { ggml_cgraph* build_graph(const sd::Tensor& packed, const sd::Tensor& timestep, const sd::Tensor& context_tensor, - ggml_tensor* refined_context, + const std::string& context_cache_name, const std::vector>& condition_videos, const std::vector>& condition_audios, const sd::Tensor& text_tags, @@ -1176,12 +1056,17 @@ namespace MiniMaxH3 { audio_input_cache = std::move(split.second); GGML_ASSERT(!audio_input_cache.empty()); GGML_ASSERT(!context_tensor.empty()); - GGML_ASSERT(refined_context != nullptr && - refined_context->ne[0] == config.hidden_size); auto video = make_input(video_input_cache); auto audio_carrier = make_input(audio_input_cache); - auto context = refined_context; + auto context = context_cache_name.empty() ? nullptr : get_cache_tensor_by_name(context_cache_name); + if (context == nullptr) { + auto ctx = get_context(); + context = model.refine_context(&ctx, make_input(context_tensor)); + if (!context_cache_name.empty()) { + cache(context_cache_name, context); + } + } std::vector condition_inputs; condition_inputs.reserve(condition_videos.size()); for (const auto& condition : condition_videos) { @@ -1289,6 +1174,7 @@ namespace MiniMaxH3 { sd::Tensor compute(int n_threads, const DiffusionParams& params) override { GGML_ASSERT(params.x != nullptr && params.timesteps != nullptr && params.context != nullptr); + GGML_ASSERT(!params.context->empty()); const auto* extra = diffusion_extra_as(params); static const std::vector> empty_conditions; static const std::vector empty_reference_blocks; @@ -1300,38 +1186,47 @@ namespace MiniMaxH3 { ? empty_reference_blocks : *extra->reference_blocks; const sd::Tensor empty_int; - const void* context_cache_identity = params.context_cache_identity != nullptr - ? params.context_cache_identity - : params.context; - auto context = get_refined_context(*params.context, - context_cache_identity, - n_threads); - if (context == nullptr) { - return {}; + if (!runner_started()) { + context_cache_disabled = false; + } + std::string cache_name; + if (context_cache_enabled && !context_cache_disabled && extra->context_id != 0 && + params.context->shape()[0] != config.hidden_size) { + cache_name = "minimax_h3.context." + std::to_string(extra->context_id); } - auto get_graph = [&]() { - return build_graph(*params.x, - *params.timesteps, - *params.context, - context, - conditions, - audio_conditions, - extra->text_token_tags == nullptr ? empty_int : *extra->text_token_tags, - extra->keyframe_indices == nullptr ? empty_int : *extra->keyframe_indices, - reference_blocks, - extra->audio_length, - extra->video_sigma_shift, - extra->audio_sigma_shift); + auto run = [&](const std::string& active_cache_name) { + auto get_graph = [&]() { + return build_graph(*params.x, + *params.timesteps, + *params.context, + active_cache_name, + conditions, + audio_conditions, + extra->text_token_tags == nullptr ? empty_int : *extra->text_token_tags, + extra->keyframe_indices == nullptr ? empty_int : *extra->keyframe_indices, + reference_blocks, + extra->audio_length, + extra->video_sigma_shift, + extra->audio_sigma_shift); + }; + return restore_trailing_singleton_dims(GGMLRunner::compute(get_graph, n_threads, false), + params.x->dim()); }; - return restore_trailing_singleton_dims(GGMLRunner::compute(get_graph, - n_threads, - false), - params.x->dim()); - } - - protected: - void on_sampling_done() override { - refined_context_cache.clear(); + auto result = run(cache_name); + if (result.empty() && last_compute_status() == GGML_STATUS_ALLOC_FAILED && + (!cache_name.empty() || !cache_.empty())) { + // The failed graph has ended before persistent inputs are released. + free_cache_ctx_and_buffer(); + context_cache_disabled = true; + LOG_WARN("MiniMax-H3: insufficient memory for context caching; retrying without it for this sampling run"); + return run(""); + } + if (!result.empty() && !cache_name.empty() && get_cache_tensor_by_name(cache_name) == nullptr) { + free_cache_ctx_and_buffer(); + context_cache_disabled = true; + LOG_WARN("MiniMax-H3: missing refined context cache; disabling it for this sampling run"); + } + return result; } }; diff --git a/src/model/diffusion/model.hpp b/src/model/diffusion/model.hpp index b3d744efa7..08d9486192 100644 --- a/src/model/diffusion/model.hpp +++ b/src/model/diffusion/model.hpp @@ -119,6 +119,8 @@ struct MiniMaxH3DiffusionExtra { int audio_length = 0; float video_sigma_shift = 12.f; float audio_sigma_shift = 3.f; + // Nonzero IDs identify immutable text conditioning and weights within one sampling run. + uint64_t context_id = 0; }; struct MiniT2IDiffusionExtra { @@ -160,7 +162,6 @@ struct DiffusionParams { const sd::Tensor* x = nullptr; const sd::Tensor* timesteps = nullptr; const sd::Tensor* context = nullptr; - const void* context_cache_identity = nullptr; const sd::Tensor* c_concat = nullptr; const sd::Tensor* y = nullptr; const std::vector>* ref_latents = nullptr; @@ -184,7 +185,6 @@ static inline const sd::Tensor& tensor_or_empty(const sd::Tensor* tensor) struct DiffusionModelRunner : public GGMLRunner { protected: std::string prefix; - virtual void on_sampling_done() {} public: DiffusionModelRunner(ggml_backend_t backend, @@ -196,11 +196,6 @@ struct DiffusionModelRunner : public GGMLRunner { virtual sd::Tensor compute(int n_threads, const DiffusionParams& diffusion_params) = 0; - void sampling_done() { - runner_end(); - on_sampling_done(); - } - void get_param_tensors(std::map& tensors) { get_param_tensors(tensors, prefix); } diff --git a/src/pipeline/diffusion_engine.cpp b/src/pipeline/diffusion_engine.cpp index db3dd2326a..8e4bb995eb 100644 --- a/src/pipeline/diffusion_engine.cpp +++ b/src/pipeline/diffusion_engine.cpp @@ -2254,24 +2254,17 @@ sd::Tensor StableDiffusionGGML::sample(const std::shared_ptrsampling_done(); - } - } - }; - SamplingDoneOnExit sample_diffusion_runner_done{work_diffusion_model.get()}; + RunnerEndOnExit sample_diffusion_runner_end{work_diffusion_model.get()}; // These inputs are immutable for this sampling run. Extensions may replace or // modify them per step, so those paths need an explicit stability contract first. - const bool cache_qwen_prefix = version == VERSION_QWEN_IMAGE_2_1 && - std::none_of(generation_extensions.begin(), generation_extensions.end(), - [](const auto& extension) { return extension->is_enabled(); }); + const bool cache_condition_inputs = (version == VERSION_QWEN_IMAGE_2_1 || sd_version_is_minimax_h3(version)) && + std::none_of(generation_extensions.begin(), generation_extensions.end(), + [](const auto& extension) { return extension->is_enabled(); }); using QwenPrefixInputs = std::tuple*, const sd::Tensor*, const std::vector>*>; std::vector qwen_prefix_inputs; + std::vector*> minimax_context_inputs; RunnerEndOnExit sample_control_runner_end{!control_image.empty() && control_net != nullptr ? control_net.get() : nullptr}; @@ -2460,11 +2453,10 @@ sd::Tensor StableDiffusionGGML::sample(const std::shared_ptr* local_skip_layers = nullptr, const std::vector>* ref_latents_override = nullptr, bool use_uncond_ip = false) -> sd::Tensor { - diffusion_params.context = condition.c_crossattn.empty() ? nullptr : &condition.c_crossattn; - diffusion_params.context_cache_identity = &condition; - diffusion_params.c_concat = c_concat_override != nullptr ? c_concat_override : (condition.c_concat.empty() ? nullptr : &condition.c_concat); - diffusion_params.y = condition.c_vector.empty() ? nullptr : &condition.c_vector; - diffusion_params.ref_latents = ref_latents_override != nullptr ? ref_latents_override : (condition.c_ref_images.empty() ? &ref_latents : &condition.c_ref_images); + diffusion_params.context = condition.c_crossattn.empty() ? nullptr : &condition.c_crossattn; + diffusion_params.c_concat = c_concat_override != nullptr ? c_concat_override : (condition.c_concat.empty() ? nullptr : &condition.c_concat); + diffusion_params.y = condition.c_vector.empty() ? nullptr : &condition.c_vector; + diffusion_params.ref_latents = ref_latents_override != nullptr ? ref_latents_override : (condition.c_ref_images.empty() ? &ref_latents : &condition.c_ref_images); if (sd_version_is_unet(version)) { int nvf = -1; @@ -2543,7 +2535,7 @@ sd::Tensor StableDiffusionGGML::sample(const std::shared_ptrbefore_diffusion(diffusion_params, step); } - if (cache_qwen_prefix) { + if (cache_condition_inputs) { auto* extra = std::get_if(&diffusion_params.extra); if (extra != nullptr) { auto key = std::make_tuple(diffusion_params.context, extra->image_slots, @@ -2553,6 +2545,12 @@ sd::Tensor StableDiffusionGGML::sample(const std::shared_ptr(&diffusion_params.extra)) { + auto entry = std::find(minimax_context_inputs.begin(), minimax_context_inputs.end(), diffusion_params.context); + minimax_extra->context_id = static_cast(entry - minimax_context_inputs.begin()) + 1; + if (entry == minimax_context_inputs.end()) { + minimax_context_inputs.push_back(diffusion_params.context); + } } } auto output_opt = work_diffusion_model->compute(n_threads, diffusion_params); diff --git a/src/pipeline/model_builders.cpp b/src/pipeline/model_builders.cpp index 9b8819a05f..958a7a753a 100644 --- a/src/pipeline/model_builders.cpp +++ b/src/pipeline/model_builders.cpp @@ -205,7 +205,8 @@ namespace sd::model_builders { result.diffusion = std::make_shared(ctx.backends.runtime_backend(SDBackendModule::DIFFUSION), tensor_storage_map, "model.diffusion_model", - weight_manager); + weight_manager, + sd_ctx_params->model_args); } else if (sd_version_is_hunyuan_video(version)) { result.conditioner = std::make_shared(ctx.backends.runtime_backend(SDBackendModule::TE), tensor_storage_map, From 2858438c3b72fdafbe97f3ea9656f17c2ca448d2 Mon Sep 17 00:00:00 2001 From: leejet Date: Wed, 23 Sep 2026 23:59:24 +0800 Subject: [PATCH 4/4] revert token refiner cache --- docs/minimax_h3.md | 14 ---- src/model/diffusion/minimax_h3.hpp | 95 +++++++------------------- src/model/diffusion/model.hpp | 2 - src/model/diffusion/qwen_image_2_1.hpp | 14 ++-- src/pipeline/diffusion_engine.cpp | 15 ++-- src/pipeline/model_builders.cpp | 3 +- 6 files changed, 35 insertions(+), 108 deletions(-) diff --git a/docs/minimax_h3.md b/docs/minimax_h3.md index 9b584aeacd..82d0e2ffb4 100644 --- a/docs/minimax_h3.md +++ b/docs/minimax_h3.md @@ -94,17 +94,3 @@ frame rate and optional soundtrack; non-24-fps inputs are resampled internally. - MiniMax-H3 runs at 24 fps; another requested value is overridden. - The default video flow shift is 12. The audio stream is mapped internally to its shift of 3, so the regular samplers can operate on the packed AV latent. - -## Text conditioning cache - -The first denoising call for each fixed condition caches the output of -`condition_proj` and `token_refiner`. Later calls reuse it; positive and negative -conditions have separate entries. The cache uses the runner's memory budget and -is released when sampling ends. If cached execution runs out of memory, it clears -the caches, disables caching for the rest of that sampling run, and retries once -without caching. Per-step conditioning extensions use the uncached path. - -Disable this optimization with `--model-args minimax_h3_context_cache=false`. -Only preprocessing of the text conditioning is cached. Reference projections, -RoPE, and the timestep-dependent transformer backbone are recomputed on every -call, including its bidirectional attention keys and values. diff --git a/src/model/diffusion/minimax_h3.hpp b/src/model/diffusion/minimax_h3.hpp index 36d9e6afbe..ed7c5ac77c 100644 --- a/src/model/diffusion/minimax_h3.hpp +++ b/src/model/diffusion/minimax_h3.hpp @@ -4,7 +4,6 @@ #include #include #include -#include #include #include #include @@ -261,21 +260,15 @@ namespace MiniMaxH3 { blocks["final_norm"] = std::make_shared(config.hidden_size, config.final_norm_eps); } - ggml_tensor* forward(GGMLRunnerContext* ctx, - ggml_tensor* x) { - auto final_norm = std::dynamic_pointer_cast(blocks["final_norm"]); + ggml_tensor* forward(GGMLRunnerContext* ctx, ggml_tensor* x) { for (int64_t i = 0; i < num_layers; ++i) { - auto block = std::dynamic_pointer_cast(blocks["blocks." + std::to_string(i)]); - x = block->forward(ctx, x); - const bool is_last = i + 1 == num_layers; - if (is_last) { - x = final_norm->forward(ctx, x); - } + auto block = std::dynamic_pointer_cast(blocks["blocks." + std::to_string(i)]); + x = block->forward(ctx, x); sd::ggml_graph_cut::mark_graph_cut(x, "minimax_h3.token_refiner.blocks." + std::to_string(i), "hidden_states"); } - return num_layers == 0 ? final_norm->forward(ctx, x) : x; + return std::dynamic_pointer_cast(blocks["final_norm"])->forward(ctx, x); } }; @@ -530,8 +523,7 @@ namespace MiniMaxH3 { } } - ggml_tensor* refine_context(GGMLRunnerContext* ctx, - ggml_tensor* context) { + ggml_tensor* refine_context(GGMLRunnerContext* ctx, ggml_tensor* context) { if (context->ne[0] == config.hidden_size) { return context; } @@ -969,22 +961,14 @@ namespace MiniMaxH3 { sd::Tensor curve_index_input_cache; sd::Tensor curve_upper_index_input_cache; sd::Tensor curve_fraction_input_cache; - bool context_cache_enabled = true; - bool context_cache_disabled = false; MiniMaxH3Runner(ggml_backend_t backend, const String2TensorStorage& tensors, const std::string& prefix = "model.diffusion_model", - std::shared_ptr weight_manager = nullptr, - const char* model_args = nullptr) + std::shared_ptr weight_manager = nullptr) : DiffusionModelRunner(backend, prefix, weight_manager), config(Config::detect_from_weights(tensors, prefix)), model(config) { - for (const auto& [key, value] : parse_key_value_args(model_args, "model arg")) { - if (key == "minimax_h3_context_cache" && !parse_strict_bool(value, context_cache_enabled)) { - LOG_WARN("ignoring invalid MiniMax-H3 model arg '%s=%s'", key.c_str(), value.c_str()); - } - } model.init(params_ctx, tensors, prefix); } @@ -1042,7 +1026,6 @@ namespace MiniMaxH3 { ggml_cgraph* build_graph(const sd::Tensor& packed, const sd::Tensor& timestep, const sd::Tensor& context_tensor, - const std::string& context_cache_name, const std::vector>& condition_videos, const std::vector>& condition_audios, const sd::Tensor& text_tags, @@ -1059,14 +1042,7 @@ namespace MiniMaxH3 { auto video = make_input(video_input_cache); auto audio_carrier = make_input(audio_input_cache); - auto context = context_cache_name.empty() ? nullptr : get_cache_tensor_by_name(context_cache_name); - if (context == nullptr) { - auto ctx = get_context(); - context = model.refine_context(&ctx, make_input(context_tensor)); - if (!context_cache_name.empty()) { - cache(context_cache_name, context); - } - } + auto context = make_input(context_tensor); std::vector condition_inputs; condition_inputs.reserve(condition_videos.size()); for (const auto& condition : condition_videos) { @@ -1174,7 +1150,6 @@ namespace MiniMaxH3 { sd::Tensor compute(int n_threads, const DiffusionParams& params) override { GGML_ASSERT(params.x != nullptr && params.timesteps != nullptr && params.context != nullptr); - GGML_ASSERT(!params.context->empty()); const auto* extra = diffusion_extra_as(params); static const std::vector> empty_conditions; static const std::vector empty_reference_blocks; @@ -1186,47 +1161,23 @@ namespace MiniMaxH3 { ? empty_reference_blocks : *extra->reference_blocks; const sd::Tensor empty_int; - if (!runner_started()) { - context_cache_disabled = false; - } - std::string cache_name; - if (context_cache_enabled && !context_cache_disabled && extra->context_id != 0 && - params.context->shape()[0] != config.hidden_size) { - cache_name = "minimax_h3.context." + std::to_string(extra->context_id); - } - auto run = [&](const std::string& active_cache_name) { - auto get_graph = [&]() { - return build_graph(*params.x, - *params.timesteps, - *params.context, - active_cache_name, - conditions, - audio_conditions, - extra->text_token_tags == nullptr ? empty_int : *extra->text_token_tags, - extra->keyframe_indices == nullptr ? empty_int : *extra->keyframe_indices, - reference_blocks, - extra->audio_length, - extra->video_sigma_shift, - extra->audio_sigma_shift); - }; - return restore_trailing_singleton_dims(GGMLRunner::compute(get_graph, n_threads, false), - params.x->dim()); + auto get_graph = [&]() { + return build_graph(*params.x, + *params.timesteps, + *params.context, + conditions, + audio_conditions, + extra->text_token_tags == nullptr ? empty_int : *extra->text_token_tags, + extra->keyframe_indices == nullptr ? empty_int : *extra->keyframe_indices, + reference_blocks, + extra->audio_length, + extra->video_sigma_shift, + extra->audio_sigma_shift); }; - auto result = run(cache_name); - if (result.empty() && last_compute_status() == GGML_STATUS_ALLOC_FAILED && - (!cache_name.empty() || !cache_.empty())) { - // The failed graph has ended before persistent inputs are released. - free_cache_ctx_and_buffer(); - context_cache_disabled = true; - LOG_WARN("MiniMax-H3: insufficient memory for context caching; retrying without it for this sampling run"); - return run(""); - } - if (!result.empty() && !cache_name.empty() && get_cache_tensor_by_name(cache_name) == nullptr) { - free_cache_ctx_and_buffer(); - context_cache_disabled = true; - LOG_WARN("MiniMax-H3: missing refined context cache; disabling it for this sampling run"); - } - return result; + return restore_trailing_singleton_dims(GGMLRunner::compute(get_graph, + n_threads, + false), + params.x->dim()); } }; diff --git a/src/model/diffusion/model.hpp b/src/model/diffusion/model.hpp index 08d9486192..f82fc36f65 100644 --- a/src/model/diffusion/model.hpp +++ b/src/model/diffusion/model.hpp @@ -119,8 +119,6 @@ struct MiniMaxH3DiffusionExtra { int audio_length = 0; float video_sigma_shift = 12.f; float audio_sigma_shift = 3.f; - // Nonzero IDs identify immutable text conditioning and weights within one sampling run. - uint64_t context_id = 0; }; struct MiniT2IDiffusionExtra { diff --git a/src/model/diffusion/qwen_image_2_1.hpp b/src/model/diffusion/qwen_image_2_1.hpp index 1364de139b..37483125cc 100644 --- a/src/model/diffusion/qwen_image_2_1.hpp +++ b/src/model/diffusion/qwen_image_2_1.hpp @@ -178,13 +178,13 @@ namespace Qwen { auto h = std::dynamic_pointer_cast(blocks[name])->forward(ctx, x); return ggml_reshape_4d(ctx->ggml_ctx, h, dim_head, heads, x->ne[1], x->ne[2]); }; - auto q = project("to_q"); - auto k = project("to_k"); - auto v = project("to_v"); - q = std::dynamic_pointer_cast(blocks["norm_q"])->forward(ctx, q); - k = std::dynamic_pointer_cast(blocks["norm_k"])->forward(ctx, k); - q = Rope::apply_rope(ctx->ggml_ctx, q, pe); - k = Rope::apply_rope(ctx->ggml_ctx, k, pe); + auto q = project("to_q"); + auto k = project("to_k"); + auto v = project("to_v"); + q = std::dynamic_pointer_cast(blocks["norm_q"])->forward(ctx, q); + k = std::dynamic_pointer_cast(blocks["norm_k"])->forward(ctx, k); + q = Rope::apply_rope(ctx->ggml_ctx, q, pe); + k = Rope::apply_rope(ctx->ggml_ctx, k, pe); if (cache.mode == QwenImage21PrefixCache::Mode::STORE) { auto persist = [&](ggml_tensor* tensor, int axis, const char* name) { auto part = ggml_ext_slice(ctx->ggml_ctx, tensor, axis, 0, cache.prefix_length); diff --git a/src/pipeline/diffusion_engine.cpp b/src/pipeline/diffusion_engine.cpp index 8e4bb995eb..0220939571 100644 --- a/src/pipeline/diffusion_engine.cpp +++ b/src/pipeline/diffusion_engine.cpp @@ -2258,13 +2258,12 @@ sd::Tensor StableDiffusionGGML::sample(const std::shared_ptris_enabled(); }); + const bool cache_qwen_prefix = version == VERSION_QWEN_IMAGE_2_1 && + std::none_of(generation_extensions.begin(), generation_extensions.end(), + [](const auto& extension) { return extension->is_enabled(); }); using QwenPrefixInputs = std::tuple*, const sd::Tensor*, const std::vector>*>; std::vector qwen_prefix_inputs; - std::vector*> minimax_context_inputs; RunnerEndOnExit sample_control_runner_end{!control_image.empty() && control_net != nullptr ? control_net.get() : nullptr}; @@ -2535,7 +2534,7 @@ sd::Tensor StableDiffusionGGML::sample(const std::shared_ptrbefore_diffusion(diffusion_params, step); } - if (cache_condition_inputs) { + if (cache_qwen_prefix) { auto* extra = std::get_if(&diffusion_params.extra); if (extra != nullptr) { auto key = std::make_tuple(diffusion_params.context, extra->image_slots, @@ -2545,12 +2544,6 @@ sd::Tensor StableDiffusionGGML::sample(const std::shared_ptr(&diffusion_params.extra)) { - auto entry = std::find(minimax_context_inputs.begin(), minimax_context_inputs.end(), diffusion_params.context); - minimax_extra->context_id = static_cast(entry - minimax_context_inputs.begin()) + 1; - if (entry == minimax_context_inputs.end()) { - minimax_context_inputs.push_back(diffusion_params.context); - } } } auto output_opt = work_diffusion_model->compute(n_threads, diffusion_params); diff --git a/src/pipeline/model_builders.cpp b/src/pipeline/model_builders.cpp index 958a7a753a..9b8819a05f 100644 --- a/src/pipeline/model_builders.cpp +++ b/src/pipeline/model_builders.cpp @@ -205,8 +205,7 @@ namespace sd::model_builders { result.diffusion = std::make_shared(ctx.backends.runtime_backend(SDBackendModule::DIFFUSION), tensor_storage_map, "model.diffusion_model", - weight_manager, - sd_ctx_params->model_args); + weight_manager); } else if (sd_version_is_hunyuan_video(version)) { result.conditioner = std::make_shared(ctx.backends.runtime_backend(SDBackendModule::TE), tensor_storage_map,