Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
54 changes: 50 additions & 4 deletions src/diffusers/modular_pipelines/ltx2/before_denoise.py
Original file line number Diff line number Diff line change
Expand Up @@ -752,6 +752,15 @@ def inputs(self) -> list[InputParam]:
InputParam(
"frame_rate", type_hint=float, default=24.0, description="Frames per second of the generated video."
),
InputParam(
"conditioning_frame_rate",
type_hint=float,
description=(
"Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults "
"to `frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture "
"rate differs from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`."
),
),
InputParam("audio_num_frames", type_hint=int, required=True),
InputParam.template("num_images_per_prompt", name="num_videos_per_prompt"),
InputParam(
Expand Down Expand Up @@ -790,7 +799,12 @@ def __call__(self, components, state: PipelineState) -> tuple[LTX2ModularPipelin
latent_num_frames = (block_state.num_frames - 1) // components.vae_temporal_compression_ratio + 1

block_state.video_coords = components.transformer.rope.prepare_video_coords(
batch_size, latent_num_frames, latent_height, latent_width, device, fps=block_state.frame_rate
batch_size,
latent_num_frames,
latent_height,
latent_width,
device,
fps=block_state.conditioning_frame_rate or block_state.frame_rate,
)
block_state.audio_coords = components.transformer.audio_rope.prepare_audio_coords(
batch_size, block_state.audio_num_frames, device
Expand Down Expand Up @@ -863,6 +877,15 @@ def inputs(self) -> list[InputParam]:
InputParam(
"frame_rate", type_hint=float, default=24.0, description="Frames per second of the generated video."
),
InputParam(
"conditioning_frame_rate",
type_hint=float,
description=(
"Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults "
"to `frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture "
"rate differs from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`."
),
),
InputParam(
"noise_scale",
type_hint=float,
Expand Down Expand Up @@ -1022,7 +1045,7 @@ def __call__(self, components, state: PipelineState) -> tuple[LTX2ModularPipelin
keyframe_latent_width=kf_latent_width,
pixel_frame_idx=(latent_idx - 1) * frame_scale_factor + 1,
num_pixel_frames=num_pixel_frames,
fps=block_state.frame_rate,
fps=block_state.conditioning_frame_rate or block_state.frame_rate,
patch_size=spatial_patch,
patch_size_t=temporal_patch,
scale_factors=scale_factors,
Expand Down Expand Up @@ -1159,6 +1182,15 @@ def inputs(self) -> list[InputParam]:
InputParam(
"frame_rate", type_hint=float, default=24.0, description="Frames per second of the generated video."
),
InputParam(
"conditioning_frame_rate",
type_hint=float,
description=(
"Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults "
"to `frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture "
"rate differs from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`."
),
),
InputParam(
"noise_scale",
type_hint=float,
Expand Down Expand Up @@ -1317,7 +1349,7 @@ def __call__(self, components, state: PipelineState) -> tuple[LTX2ModularPipelin
keyframe_latent_width=kf_latent_width,
pixel_frame_idx=(latent_idx - 1) * frame_scale_factor + 1,
num_pixel_frames=num_pixel_frames,
fps=block_state.frame_rate,
fps=block_state.conditioning_frame_rate or block_state.frame_rate,
patch_size=spatial_patch,
patch_size_t=temporal_patch,
scale_factors=scale_factors,
Expand Down Expand Up @@ -1706,6 +1738,15 @@ def inputs(self) -> list[InputParam]:
InputParam(
"frame_rate", type_hint=float, default=24.0, description="Frames per second of the generated video."
),
InputParam(
"conditioning_frame_rate",
type_hint=float,
description=(
"Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults "
"to `frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture "
"rate differs from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`."
),
),
InputParam("audio_num_frames", type_hint=int, required=True),
InputParam(
"appended_coords",
Expand Down Expand Up @@ -1750,7 +1791,12 @@ def __call__(self, components, state: PipelineState) -> tuple[LTX2ModularPipelin
latent_num_frames = (block_state.num_frames - 1) // components.vae_temporal_compression_ratio + 1

video_coords = components.transformer.rope.prepare_video_coords(
batch_size, latent_num_frames, latent_height, latent_width, device, fps=block_state.frame_rate
batch_size,
latent_num_frames,
latent_height,
latent_width,
device,
fps=block_state.conditioning_frame_rate or block_state.frame_rate,
)
block_state.video_coords = torch.cat([video_coords, block_state.appended_coords.to(video_coords.dtype)], dim=2)
block_state.audio_coords = components.transformer.audio_rope.prepare_audio_coords(
Expand Down
11 changes: 10 additions & 1 deletion src/diffusers/modular_pipelines/ltx2/denoise.py
Original file line number Diff line number Diff line change
Expand Up @@ -319,6 +319,15 @@ def inputs(self) -> list[InputParam]:
InputParam(
"frame_rate", type_hint=float, default=24.0, description="Frames per second of the generated video."
),
InputParam(
"conditioning_frame_rate",
type_hint=float,
description=(
"Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults "
"to `frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture "
"rate differs from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`."
),
),
InputParam(
"use_cross_timestep",
type_hint=bool,
Expand Down Expand Up @@ -361,7 +370,7 @@ def __call__(
num_frames=latent_num_frames,
height=latent_height,
width=latent_width,
fps=block_state.frame_rate,
fps=block_state.conditioning_frame_rate or block_state.frame_rate,
use_cross_timestep=block_state.use_cross_timestep,
attention_kwargs=block_state.attention_kwargs,
perturbation_mask=None,
Expand Down
11 changes: 10 additions & 1 deletion src/diffusers/modular_pipelines/ltx2/encoders.py
Original file line number Diff line number Diff line change
Expand Up @@ -1197,6 +1197,15 @@ def inputs(self) -> list[InputParam]:
InputParam(
"frame_rate", type_hint=float, default=24.0, description="Frames per second of the generated video."
),
InputParam(
"conditioning_frame_rate",
type_hint=float,
description=(
"Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults "
"to `frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture "
"rate differs from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`."
),
),
InputParam.template("generator"),
]

Expand Down Expand Up @@ -1285,7 +1294,7 @@ def __call__(self, components, state: PipelineState) -> tuple[LTX2ModularPipelin
height=ref_latent_height,
width=ref_latent_width,
device=device,
fps=block_state.frame_rate,
fps=block_state.conditioning_frame_rate or block_state.frame_rate,
)
if downscale_factor != 1:
ref_coords[:, 1, :, :] = ref_coords[:, 1, :, :] * downscale_factor
Expand Down
44 changes: 44 additions & 0 deletions src/diffusers/modular_pipelines/ltx2/modular_blocks_ltx2.py
Original file line number Diff line number Diff line change
Expand Up @@ -322,6 +322,10 @@ class LTX2CoreDenoiseStep(SequentialPipelineBlocks):
Frames per second of the generated video.
audio_latents (`Tensor`, *optional*):
Optional pre-encoded audio latents; random noise is used when not provided.
conditioning_frame_rate (`float`, *optional*):
Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults to
`frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture rate differs
from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`.
dtype (`dtype`):
The dtype the model inputs are cast to.
**denoiser_input_fields (`None`, *optional*):
Expand Down Expand Up @@ -424,6 +428,10 @@ class LTX2Image2VideoCoreDenoiseStep(SequentialPipelineBlocks):
Frames per second of the generated video.
audio_latents (`Tensor`, *optional*):
Optional pre-encoded audio latents; random noise is used when not provided.
conditioning_frame_rate (`float`, *optional*):
Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults to
`frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture rate differs
from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`.
dtype (`dtype`):
The dtype the model inputs are cast to.
**denoiser_input_fields (`None`, *optional*):
Expand Down Expand Up @@ -520,6 +528,10 @@ class LTX2ConditionCoreDenoiseStep(SequentialPipelineBlocks):
`LTX2AutoDurationStep`).
frame_rate (`float`, *optional*, defaults to 24.0):
Frames per second of the generated video.
conditioning_frame_rate (`float`, *optional*):
Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults to
`frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture rate differs
from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`.
noise_scale (`float`, *optional*):
Initial noise level for the un-conditioned tokens. `None` (default) resolves to `sigmas[0]` when custom
`sigmas` are supplied, else 1.0.
Expand Down Expand Up @@ -676,6 +688,10 @@ class LTX2AutoReferenceEncoderStep(ConditionalPipelineBlocks):
`LTX2AutoDurationStep`).
frame_rate (`float`, *optional*, defaults to 24.0):
Frames per second of the generated video.
conditioning_frame_rate (`float`, *optional*):
Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults to
`frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture rate differs
from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`.
generator (`Generator`, *optional*):
Torch generator for deterministic generation.

Expand Down Expand Up @@ -831,6 +847,10 @@ class LTX2InContextCoreDenoiseStep(SequentialPipelineBlocks):
`LTX2AutoDurationStep`).
frame_rate (`float`, *optional*, defaults to 24.0):
Frames per second of the generated video.
conditioning_frame_rate (`float`, *optional*):
Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults to
`frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture rate differs
from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`.
noise_scale (`float`, *optional*):
Initial noise level for the un-conditioned tokens. `None` (default) resolves to `sigmas[0]` when custom
`sigmas` are supplied, else 1.0.
Expand Down Expand Up @@ -962,6 +982,10 @@ class LTX2AutoCoreDenoiseStep(ConditionalPipelineBlocks):
`LTX2AutoDurationStep`).
frame_rate (`float`, *optional*, defaults to 24.0):
Frames per second of the generated video.
conditioning_frame_rate (`float`, *optional*):
Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults to
`frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture rate differs
from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`.
noise_scale (`float`, *optional*):
Initial noise level for the un-conditioned tokens. `None` (default) resolves to `sigmas[0]` when custom
`sigmas` are supplied, else 1.0.
Expand Down Expand Up @@ -1297,6 +1321,10 @@ class LTX2Blocks(SequentialPipelineBlocks):
which keeps the provided latents.
audio_latents (`Tensor`, *optional*):
Optional pre-encoded audio latents; random noise is used when not provided.
conditioning_frame_rate (`float`, *optional*):
Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults to
`frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture rate differs
from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`.
**denoiser_input_fields (`None`, *optional*):
conditional model inputs for the denoiser: e.g. prompt_embeds, negative_prompt_embeds, etc.
use_cross_timestep (`bool`, *optional*, defaults to True):
Expand Down Expand Up @@ -1412,6 +1440,10 @@ class LTX2ImageToVideoBlocks(SequentialPipelineBlocks):
VAE-encoded reference-image latents used for image-to-video conditioning.
audio_latents (`Tensor`, *optional*):
Optional pre-encoded audio latents; random noise is used when not provided.
conditioning_frame_rate (`float`, *optional*):
Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults to
`frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture rate differs
from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`.
**denoiser_input_fields (`None`, *optional*):
conditional model inputs for the denoiser: e.g. prompt_embeds, negative_prompt_embeds, etc.
use_cross_timestep (`bool`, *optional*, defaults to True):
Expand Down Expand Up @@ -1511,6 +1543,10 @@ class LTX2ConditionBlocks(SequentialPipelineBlocks):
The number of images to generate per prompt.
latents (`Tensor`, *optional*):
Pre-generated noisy latents for image generation.
conditioning_frame_rate (`float`, *optional*):
Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults to
`frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture rate differs
from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`.
noise_scale (`float`, *optional*):
Initial noise level for the un-conditioned tokens. `None` (default) resolves to `sigmas[0]` when custom
`sigmas` are supplied, else 1.0.
Expand Down Expand Up @@ -1633,6 +1669,10 @@ class LTX2InContextBlocks(SequentialPipelineBlocks):
`conditioning_attention_strength`.
frame_rate (`float`, *optional*, defaults to 24.0):
Frames per second of the generated video.
conditioning_frame_rate (`float`, *optional*):
Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults to
`frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture rate differs
from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`.
num_videos_per_prompt (`int`, *optional*, defaults to 1):
The number of images to generate per prompt.
reference_latents (`Tensor`, *optional*):
Expand Down Expand Up @@ -1794,6 +1834,10 @@ class LTX2AutoBlocks(SequentialPipelineBlocks):
Optional pixel-space mask of shape (1, 1, F, H, W) with values in [0, 1] giving spatially varying
attention strength. Downsampled to the reference's latent grid and multiplied by
`conditioning_attention_strength`.
conditioning_frame_rate (`float`, *optional*):
Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults to
`frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture rate differs
from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`.
num_videos_per_prompt (`int`, *optional*, defaults to 1):
The number of images to generate per prompt.
condition_latents (`list`, *optional*):
Expand Down
4 changes: 4 additions & 0 deletions src/diffusers/modular_pipelines/ltx2/modular_blocks_ltx25.py
Original file line number Diff line number Diff line change
Expand Up @@ -290,6 +290,10 @@ class LTX25AutoBlocks(SequentialPipelineBlocks):
Optional pixel-space mask of shape (1, 1, F, H, W) with values in [0, 1] giving spatially varying
attention strength. Downsampled to the reference's latent grid and multiplied by
`conditioning_attention_strength`.
conditioning_frame_rate (`float`, *optional*):
Frame rate the model is conditioned on (the time axis of the positional embeddings). Defaults to
`frame_rate`. Set it apart from `frame_rate` for adapters trained on footage whose capture rate differs
from its playback rate, e.g. slow-motion LoRAs: `frame_rate / speed`.
num_videos_per_prompt (`int`, *optional*, defaults to 1):
The number of images to generate per prompt.
condition_latents (`list`, *optional*):
Expand Down
Loading
Loading