Add support for audo frame count so datasets can have varrying length videos. Varous ltx 2.3 VAE optimizations such as removing tiling articacts, and doing frame split encoding to reduce vram on encoding/decoding.
This commit is contained in:
@@ -389,7 +389,7 @@ class AiToolkitDataset(LatentCachingMixin, ControlCachingMixin, CLIPCachingMixin
|
||||
self.dataset_config = dataset_config
|
||||
# update bucket divisibility
|
||||
self.dataset_config.bucket_tolerance = sd.get_bucket_divisibility()
|
||||
self.is_video = dataset_config.num_frames > 1
|
||||
self.is_video = dataset_config.num_frames > 1 or dataset_config.auto_frame_count
|
||||
super().__init__()
|
||||
folder_path = dataset_config.folder_path
|
||||
self.dataset_path = dataset_config.dataset_path
|
||||
@@ -505,6 +505,13 @@ class AiToolkitDataset(LatentCachingMixin, ControlCachingMixin, CLIPCachingMixin
|
||||
latent_space_version = 'sdxl'
|
||||
else:
|
||||
latent_space_version = self.sd.model_config.arch if self.sd is not None else "sd1"
|
||||
|
||||
temporal_compression = 8
|
||||
if self.sd is not None:
|
||||
if hasattr(self.sd.vae.config, 'scale_factor_temporal'):
|
||||
temporal_compression = self.sd.vae.config.scale_factor_temporal
|
||||
if hasattr(self.sd.unet.config, 'temporal_compression_ratio'):
|
||||
temporal_compression = self.sd.unet.config.temporal_compression_ratio
|
||||
|
||||
bad_count = 0
|
||||
for file in tqdm(file_list):
|
||||
@@ -520,6 +527,7 @@ class AiToolkitDataset(LatentCachingMixin, ControlCachingMixin, CLIPCachingMixin
|
||||
text_embedding_space_version=self.sd.model_config.arch if self.sd else "sd1",
|
||||
te_padding_side=self.sd.te_padding_side if self.sd else "right",
|
||||
latent_space_version=latent_space_version,
|
||||
temporal_compression=temporal_compression,
|
||||
)
|
||||
self.file_list.append(file_item)
|
||||
except Exception as e:
|
||||
|
||||
Reference in New Issue
Block a user