Skip to content

vllm_omni.diffusion.models.ltx2.ltx2_denoise

Shared denoise execution primitives for LTX pipelines.

LTXDenoiseStep module-attribute

LTXDenoiseStep = Callable[
    [int, torch.Tensor, LTXAVState], LTXAVState
]

LTX_ANCESTRAL_ETA module-attribute

LTX_ANCESTRAL_ETA = 1.0

LTX_ANCESTRAL_NOISE_SEED_OFFSET module-attribute

LTX_ANCESTRAL_NOISE_SEED_OFFSET = 10000

LTX_ANCESTRAL_S_NOISE module-attribute

LTX_ANCESTRAL_S_NOISE = 1.0

LTX_OFFICIAL_DEFAULT_SEED module-attribute

LTX_OFFICIAL_DEFAULT_SEED = 10

LTXDenoiseContext dataclass

Mutable AV state and positional metadata for a denoise phase.

audio_attention_mask class-attribute instance-attribute

audio_attention_mask: Tensor | None = None

audio_coords instance-attribute

audio_coords: Tensor

audio_latents instance-attribute

audio_latents: Tensor

conditioning_mask class-attribute instance-attribute

conditioning_mask: Tensor | None = None

conditioning_mask_for_model class-attribute instance-attribute

conditioning_mask_for_model: Tensor | None = None

latents instance-attribute

latents: Tensor

video_coords instance-attribute

video_coords: Tensor

LTXDenoiseExecutor

Run the one shared LTX denoise loop.

Prediction and scheduler math remain injectable so structural refactors do not change the existing LTX2/LTX2.3 numerical paths. Guidance will replace that step policy independently.

run staticmethod

run(
    pipeline: LTXDenoisePipeline,
    state: LTXAVState,
    timesteps: Iterable[Tensor],
    step: LTXDenoiseStep,
) -> LTXAVState

LTXDenoisePipeline

Bases: Protocol

Pipeline state required by :class:LTXDenoiseExecutor.

interrupt property

interrupt: bool

progress_bar

progress_bar(iterable=None, total=None)

LTXForwardContext dataclass

Immutable metadata and schedulers for one LTX denoise phase.

attention_kwargs instance-attribute

attention_kwargs: dict[str, Any] | None

audio_scheduler instance-attribute

audio_scheduler: Any

batch_size property

batch_size: int

device instance-attribute

device: device

guidance_parallel_ready instance-attribute

guidance_parallel_ready: bool

latent_height instance-attribute

latent_height: int

latent_mel_bins instance-attribute

latent_mel_bins: int

latent_num_frames instance-attribute

latent_num_frames: int

latent_width instance-attribute

latent_width: int

num_videos_per_prompt property

num_videos_per_prompt: int

original_audio_num_frames instance-attribute

original_audio_num_frames: int

padded_audio_num_frames instance-attribute

padded_audio_num_frames: int

prompt_context instance-attribute

prompt_context: LTXPromptContext

req instance-attribute

request_inputs instance-attribute

request_inputs: LTXRequestInputs

sampler instance-attribute

sampler: str

timesteps instance-attribute

timesteps: Tensor

video_audio_step_adapter instance-attribute

video_audio_step_adapter: Any

LTXPhaseExecutor

Prepare and execute one LTX phase without owning model modules.

run staticmethod

run(
    pipeline: Any,
    req: DiffusionRequestBatch,
    request_inputs: LTXRequestInputs,
    *,
    noise_scale: float,
    sigmas: list[float] | None,
    timesteps: list[int] | None,
    attention_kwargs: dict[str, Any] | None,
    phase_recipe: LTXPhaseRecipe,
    image: Any | None = None,
    prompt_context: LTXPromptContext | None = None,
) -> LTXPhaseResult

LTXPhaseResult dataclass

Denoised AV latents and the context used to produce them.

audio instance-attribute

audio: Tensor

audio_for_next_phase class-attribute instance-attribute

audio_for_next_phase: Tensor | None = None

forward_context instance-attribute

forward_context: LTXForwardContext

video instance-attribute

video: Tensor

LTXVideoAudioStepAdapter

Expose the shared LTX Euler update through the distributed scheduler API.

step

step(
    noise_pred,
    t,
    latents,
    return_dict=False,
    generator=None,
)

build_transformer_kwargs

build_transformer_kwargs(
    pipeline: Any,
    forward_ctx: LTXForwardContext,
    denoise_ctx: LTXDenoiseContext,
    *,
    hidden_states: Tensor,
    audio_hidden_states: Tensor,
    encoder_hidden_states: Tensor,
    audio_encoder_hidden_states: Tensor,
    encoder_attention_mask: Tensor | None,
    audio_encoder_attention_mask: Tensor | None,
    ts: Tensor,
    attention_kwargs: dict[str, Any] | None = None,
) -> dict[str, Any]

calculate_shift

calculate_shift(
    image_seq_len: int,
    base_seq_len: int = 256,
    max_seq_len: int = 4096,
    base_shift: float = 0.5,
    max_shift: float = 1.15,
) -> float

prepare_rope_coords_stage

prepare_rope_coords_stage(
    pipeline: Any,
    forward_ctx: LTXForwardContext,
    latents: Tensor,
    audio_latents: Tensor,
) -> tuple[Tensor, Tensor]

prepare_scheduler_stage

prepare_scheduler_stage(
    pipeline: Any,
    request_inputs: LTXRequestInputs,
    *,
    device: device,
    sigmas: list[float] | None,
    timesteps: list[int] | None,
    latent_num_frames: int,
    latent_height: int,
    latent_width: int,
    use_official_sigma_schedule: bool,
    image_conditioned: bool = False,
    sampler: str = "euler",
    generator: Generator | list[Generator] | None = None,
    conditioning_mask: Tensor | None = None,
) -> tuple[Any, Any, Tensor]

step_denoised_latents

step_denoised_latents(
    pipeline: Any,
    forward_ctx: LTXForwardContext,
    denoise_ctx: LTXDenoiseContext,
    noise_pred_video: Tensor,
    noise_pred_audio: Tensor,
    timestep: Tensor,
) -> tuple[Tensor, Tensor]