Rework img/video reference in Minimax H3 to more closely match the comfy ui implementation.
This commit is contained in:
@@ -2309,10 +2309,15 @@ class TextEmbeddingCachingMixin:
|
||||
self.sd.set_device_state_preset('cache_text_encoder')
|
||||
did_move = True
|
||||
|
||||
if file_item.encode_control_in_text_embeddings and file_item.control_path is not None:
|
||||
control_video_paths = getattr(file_item, 'control_video_paths', None) or []
|
||||
if file_item.encode_control_in_text_embeddings and (
|
||||
file_item.control_path is not None or len(control_video_paths) > 0
|
||||
):
|
||||
ctrl_img_list = []
|
||||
control_path_list = file_item.control_path
|
||||
if not isinstance(file_item.control_path, list):
|
||||
if control_path_list is None:
|
||||
control_path_list = []
|
||||
elif not isinstance(control_path_list, list):
|
||||
control_path_list = [control_path_list]
|
||||
for i in range(len(control_path_list)):
|
||||
try:
|
||||
@@ -2328,6 +2333,14 @@ class TextEmbeddingCachingMixin:
|
||||
except Exception as e:
|
||||
print_acc(f"Error: {e}")
|
||||
print_acc(f"Error loading control image: {control_path_list[i]}")
|
||||
# control VIDEOS ride into the presentation by path (models
|
||||
# with supports_video_control_images turn them into
|
||||
# timestamped vision blocks); images first, then videos.
|
||||
# The model needs the dataset config to treat the clip
|
||||
# exactly like its latent rows (frame count / trim)
|
||||
ctrl_img_list.extend(control_video_paths)
|
||||
if len(control_video_paths) > 0:
|
||||
self.sd._ref_video_dataset_config = self.dataset_config
|
||||
|
||||
if len(ctrl_img_list) == 0:
|
||||
ctrl_img = None
|
||||
|
||||
@@ -283,6 +283,13 @@ class BaseModel:
|
||||
divisibility = divisibility * 2
|
||||
return divisibility
|
||||
|
||||
def prepare_sample_prompt_context(self, gen_config):
|
||||
"""Optional hook called right before a sample prompt is encoded, with
|
||||
the sample's GenerateImageConfig, for models whose control conditioning
|
||||
in the text embeds depends on sample settings (e.g. a video reference's
|
||||
length capped at the sample's frame count)."""
|
||||
return None
|
||||
|
||||
def get_frame_count_snapper(self):
|
||||
"""Optional hook for video models whose VAE accepts frame counts on a
|
||||
grid other than the default ``temporal_compression * n + 1``.
|
||||
@@ -605,6 +612,7 @@ class BaseModel:
|
||||
else:
|
||||
ctrl_img = ctrl_img_list[0] if len(ctrl_img_list) > 0 else None
|
||||
# encode the prompt ourselves so we can do fun stuff with embeddings
|
||||
self.prepare_sample_prompt_context(gen_config)
|
||||
if isinstance(self.adapter, CustomAdapter):
|
||||
self.adapter.is_unconditional_run = False
|
||||
conditional_embeds = self.encode_prompt(
|
||||
|
||||
@@ -215,6 +215,9 @@ class StableDiffusion:
|
||||
|
||||
# set true for models that encode control image into text embeddings
|
||||
self.encode_control_in_text_embeddings = False
|
||||
# control files may be VIDEOS (paths exposed on the batch as
|
||||
# control_video_paths_list); see minimax_h3 ref2va
|
||||
self.supports_video_control_images = False
|
||||
# control images will come in as a list for encoding some things if true
|
||||
self.has_multiple_control_images = False
|
||||
# do not resize control images
|
||||
@@ -292,7 +295,20 @@ class StableDiffusion:
|
||||
if self.is_flux or self.is_v3:
|
||||
divisibility = divisibility * 2
|
||||
return divisibility * 2 # todo remove this
|
||||
|
||||
|
||||
def get_frame_count_snapper(self):
|
||||
"""Optional hook for video models whose VAE accepts frame counts on a
|
||||
grid other than the default ``temporal_compression * n + 1``. Return a
|
||||
MODULE-LEVEL function ``(num_frames) -> int`` (picklable — file items
|
||||
travel into dataloader workers) that snaps a frame count DOWN to a
|
||||
valid count, or None for the default auto_frame_count math."""
|
||||
return None
|
||||
|
||||
def prepare_sample_prompt_context(self, gen_config):
|
||||
"""Optional hook called right before a sample prompt is encoded, with
|
||||
the sample's GenerateImageConfig, for models whose control conditioning
|
||||
in the text embeds depends on sample settings."""
|
||||
return None
|
||||
|
||||
def load_model(self):
|
||||
if self.is_loaded:
|
||||
|
||||
Reference in New Issue
Block a user