mirror of
https://github.com/storytold/FineTrainers-Conditioning.git
synced 2026-10-09 00:09:45 +00:00
Make a fix to the video tensors.
This commit is contained in:
@@ -198,7 +198,7 @@ class LTXConditionedPipeline(LTXPipeline):
|
||||
|
||||
img_ref_latent = condition_latents_prepare.prepare_latents_for_conditioning(
|
||||
vae=self.vae,
|
||||
image_or_video=pose_video,
|
||||
image_or_video=img_ref_video,
|
||||
patch_size=self.transformer.config.patch_size,
|
||||
patch_size_t=self.transformer.config.patch_size_t,
|
||||
device=device,
|
||||
|
||||
@@ -1359,6 +1359,8 @@ class Trainer:
|
||||
frames = frames[: max_num_frames].float()
|
||||
frames = frames.permute(0, 3, 1, 2).contiguous()
|
||||
frames = torch.stack([frame for frame in frames], dim=0)
|
||||
# add singleton dimension for downstream inference since no preprocessing took place
|
||||
frames = frames.unsqueeze(0)
|
||||
return frames
|
||||
|
||||
@staticmethod
|
||||
@@ -1381,4 +1383,6 @@ class Trainer:
|
||||
# nearest_res = self._find_nearest_resolution(frames.shape[2], frames.shape[3])
|
||||
# frames_resized = torch.stack([frame for frame in frames], dim=0)
|
||||
frames = torch.stack([frame for frame in frames], dim=0)
|
||||
# adding singleton dimension for downstream inference.
|
||||
frames = frames.unsqueeze(0)
|
||||
return frames
|
||||
|
||||
Reference in New Issue
Block a user