Fix issue with double sampling qwen vl video frames for hidream h3 for reference videos

This commit is contained in:
Jaret Burkett
2026-08-17 15:39:07 -06:00
parent 42dfe9c661
commit afd1d92722
3 changed files with 14 additions and 2 deletions

View File

@@ -160,8 +160,15 @@ def encode_minimax_h3_prompt(
pixel_values = vision["pixel_values"]
image_grid_thw = vision["image_grid_thw"]
if videos:
import numpy as np
# frames are already sampled at 2 fps: do_sample_frames=False keeps
# the video processor from resampling them (official diffusers
# ref2va encoder passes the same flag)
vids = processor.video_processor(
videos=[v.frames for v in videos], return_tensors="pt"
videos=[np.stack([np.asarray(f.convert("RGB")) for f in v.frames]) for v in videos],
do_sample_frames=False,
return_tensors="pt",
)
pixel_values_videos = vids["pixel_values_videos"]
video_grid_thw = vids["video_grid_thw"]

View File

@@ -2136,6 +2136,11 @@ class TextEmbeddingFileItemDTOMixin:
item["control_path"] = self.control_path
if self.encode_control_in_text_embeddings and getattr(self, 'control_video_paths', None):
item["control_videos"] = sorted(self.control_video_paths)
# v2: reference-video vision blocks are no longer resampled by the
# processor (do_sample_frames=False); older video-ref embeds are
# misaligned with their presentation. Only items WITH control
# videos carry this key, so no other cache is touched.
item["control_videos_version"] = 2
# first-frame vision conditioning changes the embedding content -> new cache key
elif (
getattr(self, "encode_first_frame_in_text_embeddings", False)

View File

@@ -1 +1 @@
VERSION = "0.12.24"
VERSION = "0.12.25"