Rework img/video reference in Minimax H3 to more closely match the comfy ui implementation.

This commit is contained in:
Jaret Burkett
2026-08-15 07:13:54 -06:00
parent f1faa7725b
commit 127d6f626d
9 changed files with 209 additions and 51 deletions

View File

@@ -2309,10 +2309,15 @@ class TextEmbeddingCachingMixin:
self.sd.set_device_state_preset('cache_text_encoder')
did_move = True
if file_item.encode_control_in_text_embeddings and file_item.control_path is not None:
control_video_paths = getattr(file_item, 'control_video_paths', None) or []
if file_item.encode_control_in_text_embeddings and (
file_item.control_path is not None or len(control_video_paths) > 0
):
ctrl_img_list = []
control_path_list = file_item.control_path
if not isinstance(file_item.control_path, list):
if control_path_list is None:
control_path_list = []
elif not isinstance(control_path_list, list):
control_path_list = [control_path_list]
for i in range(len(control_path_list)):
try:
@@ -2328,6 +2333,14 @@ class TextEmbeddingCachingMixin:
except Exception as e:
print_acc(f"Error: {e}")
print_acc(f"Error loading control image: {control_path_list[i]}")
# control VIDEOS ride into the presentation by path (models
# with supports_video_control_images turn them into
# timestamped vision blocks); images first, then videos.
# The model needs the dataset config to treat the clip
# exactly like its latent rows (frame count / trim)
ctrl_img_list.extend(control_video_paths)
if len(control_video_paths) > 0:
self.sd._ref_video_dataset_config = self.dataset_config
if len(ctrl_img_list) == 0:
ctrl_img = None

View File

@@ -283,6 +283,13 @@ class BaseModel:
divisibility = divisibility * 2
return divisibility
def prepare_sample_prompt_context(self, gen_config):
"""Optional hook called right before a sample prompt is encoded, with
the sample's GenerateImageConfig, for models whose control conditioning
in the text embeds depends on sample settings (e.g. a video reference's
length capped at the sample's frame count)."""
return None
def get_frame_count_snapper(self):
"""Optional hook for video models whose VAE accepts frame counts on a
grid other than the default ``temporal_compression * n + 1``.
@@ -605,6 +612,7 @@ class BaseModel:
else:
ctrl_img = ctrl_img_list[0] if len(ctrl_img_list) > 0 else None
# encode the prompt ourselves so we can do fun stuff with embeddings
self.prepare_sample_prompt_context(gen_config)
if isinstance(self.adapter, CustomAdapter):
self.adapter.is_unconditional_run = False
conditional_embeds = self.encode_prompt(

View File

@@ -215,6 +215,9 @@ class StableDiffusion:
# set true for models that encode control image into text embeddings
self.encode_control_in_text_embeddings = False
# control files may be VIDEOS (paths exposed on the batch as
# control_video_paths_list); see minimax_h3 ref2va
self.supports_video_control_images = False
# control images will come in as a list for encoding some things if true
self.has_multiple_control_images = False
# do not resize control images
@@ -292,7 +295,20 @@ class StableDiffusion:
if self.is_flux or self.is_v3:
divisibility = divisibility * 2
return divisibility * 2 # todo remove this
def get_frame_count_snapper(self):
"""Optional hook for video models whose VAE accepts frame counts on a
grid other than the default ``temporal_compression * n + 1``. Return a
MODULE-LEVEL function ``(num_frames) -> int`` (picklable — file items
travel into dataloader workers) that snaps a frame count DOWN to a
valid count, or None for the default auto_frame_count math."""
return None
def prepare_sample_prompt_context(self, gen_config):
"""Optional hook called right before a sample prompt is encoded, with
the sample's GenerateImageConfig, for models whose control conditioning
in the text embeds depends on sample settings."""
return None
def load_model(self):
if self.is_loaded: