Skip to content

vllm_omni.diffusion.models.minimax_h3.pipeline_minimax_h3

vLLM-Omni pipeline for MiniMax H3 FL2VA and Ref2VA partitions.

MINIMAX_H3_AUDIO_REF_COND_TIMESTEP module-attribute

MINIMAX_H3_AUDIO_REF_COND_TIMESTEP = 1.0

MINIMAX_H3_AUDIO_SAMPLE_RATE module-attribute

MINIMAX_H3_AUDIO_SAMPLE_RATE = 32000

MINIMAX_H3_DIFFUSION_DOWNLOAD_PATTERNS module-attribute

MINIMAX_H3_DIFFUSION_DOWNLOAD_PATTERNS = {
    "fl2va": [
        "FL2VA/model_index.json",
        "FL2VA/transformer/**",
        "FL2VA/video_vae/**",
        "FL2VA/audio_vae/**",
    ],
    "ref2va": [
        "Ref2VA/model_index.json",
        "Ref2VA/transformer/**",
        "Ref2VA/video_vae/**",
        "Ref2VA/audio_vae/**",
    ],
    "combined": [
        "FL2VA/model_index.json",
        "FL2VA/transformer/**",
        "FL2VA/video_vae/**",
        "FL2VA/audio_vae/**",
        "Ref2VA/model_index.json",
        "Ref2VA/transformer/**",
    ],
}

MINIMAX_H3_DOWNLOAD_PATTERNS module-attribute

MINIMAX_H3_DOWNLOAD_PATTERNS = [
    "FL2VA/**",
    "Ref2VA/model_index.json",
    "Ref2VA/transformer/**",
]

MINIMAX_H3_FPS module-attribute

MINIMAX_H3_FPS = 24

MINIMAX_H3_IMGVID_COND_TIMESTEP module-attribute

MINIMAX_H3_IMGVID_COND_TIMESTEP = 0.999

MINIMAX_H3_TASK_DOWNLOAD_PATTERNS module-attribute

MINIMAX_H3_TASK_DOWNLOAD_PATTERNS = {
    "fl2va": ["FL2VA/**"],
    "ref2va": ["Ref2VA/**"],
}

logger module-attribute

logger = init_logger(__name__)

MiniMaxH3Pipeline

Bases: Module, DenoiseProgressMixin, ProgressBarMixin, DiffusionPipelineProfilerMixin, SupportImageInput, SupportAudioInput, SupportAudioOutput, SupportsComponentDiscovery

CFG-distilled joint video/audio generation for MiniMax H3.

audio_vae instance-attribute

audio_vae = MiniMaxH3AudioVAE(
    os.path.join(vae_model_path, "audio_vae"),
    device=self.device,
    load_device=component_load_device,
    decode_only=not self.load_vae_encoder,
    trust_remote_code=od_config.trust_remote_code,
)

default_audio_shift instance-attribute

default_audio_shift = float(shifts.get('audio', 3.0))

default_video_shift instance-attribute

default_video_shift = float(shifts.get('video', 12.0))

device instance-attribute

device = get_local_device()

dummy_run_num_frames class-attribute

dummy_run_num_frames: int = 0

load_text_encoder instance-attribute

load_text_encoder = od_config.model_loaded.get(
    "text_encoder", True
)

load_vae_encoder instance-attribute

load_vae_encoder = od_config.model_loaded.get(
    "vae_encoder", True
)

lora_is_fused property

lora_is_fused: bool

True when --lora-path was consumed as a load-time weight fusion.

od_config instance-attribute

od_config = od_config

parallel_config instance-attribute

parallel_config = od_config.parallel_config

partition instance-attribute

partition = _minimax_h3_partition_for_task(
    getattr(od_config, "task_type", None),
    str(od_config.model),
)

processor instance-attribute

processor = Qwen3VLProcessor.from_pretrained(
    str(model_path),
    subfolder="processor",
    local_files_only=os.path.isdir(model_path),
)

supported_tasks instance-attribute

supported_tasks = frozenset(supported_tasks)

supports_step_execution class-attribute

supports_step_execution: bool = True

text_encoder instance-attribute

text_encoder = MiniMaxH3Qwen3VLEncoder(
    os.path.join(model_path, "text_encoder"),
    device=self.device,
    load_model=rank < text_encoder_tp_size,
    encoder_group=self.text_encoder_group,
    quant_config=_resolve_minimax_h3_text_encoder_quant_config(
        od_config.quantization_config
    ),
)

text_encoder_group instance-attribute

text_encoder_group = self._build_text_encoder_group(
    text_encoder_tp_size
)

text_encoder_tp_size instance-attribute

text_encoder_tp_size = text_encoder_tp_size

tokenizer instance-attribute

tokenizer = Qwen2TokenizerFast.from_pretrained(
    str(model_path),
    subfolder="tokenizer",
    local_files_only=os.path.isdir(model_path),
)

transformer instance-attribute

transformer = MiniMaxH3DiTModel(
    od_config,
    quant_config=transformer_quant_config,
    diffusers_weights=modular,
)

transformers_ref instance-attribute

transformers_ref = MiniMaxH3DiTModel(
    od_config,
    quant_config=transformer_quant_config,
    diffusers_weights=modular,
)

vae instance-attribute

vae = self.video_vae

video_vae instance-attribute

video_vae = MiniMaxH3VideoVAE(
    os.path.join(vae_model_path, "video_vae"),
    device=self.device,
    load_device=component_load_device,
    decode_only=not self.load_vae_encoder,
    trust_remote_code=od_config.trust_remote_code,
)

weights_sources instance-attribute

weights_sources = [
    DiffusersPipelineLoader.ComponentSource(
        model_or_path=str(model_path),
        subfolder="transformer",
        revision=od_config.revision,
        prefix="transformer.",
        fall_back_to_pt=False,
    )
]

adopt_cache_dit_backend

adopt_cache_dit_backend(backend: CacheDiTBackend) -> None

Adopt runner-installed generic Cache-DiT for request transitions.

decode

decode(
    video_latent: Tensor,
    audio_latent: Tensor,
    *,
    height: int,
    width: int,
) -> tuple[Tensor, Tensor]

decode_to_mp4

decode_to_mp4(
    video_latent: Tensor,
    audio_latent: Tensor,
    *,
    height: int,
    width: int,
    max_pending: int = 2,
    batch_frames: int = 17,
    video_codec_options: dict[str, str] | None = None,
) -> bytes

Decode and encode one output on the worker without full-video materialization.

Audio is decoded first so the incremental mux session can attach its audio stream before temporal video chunks arrive. Everything after a chunk is committed -- crop, quantization, transfer, encoding -- is the shared consumer's job; this method only supplies what is specific to H3: the audio waveform, the requested-size crop, and the fixed rate.

Every rank of a distributed VAE group drives the temporal collectives, but only the output owner receives chunks, so a peer rank returns empty bytes -- the pre-encoded counterpart of the empty tensor the full decode leaves there.

denoise_step

denoise_step(
    input_batch: InputBatch,
    *,
    states: Sequence[StepRequestState] | None = None,
    **kwargs: Any,
) -> Tensor | None

Run one denoise forward covering every request in the batch.

Requests are concatenated into a single packed sequence that keeps one attention document each, so the whole batch costs one DiT forward. Backends that ignore cu_seqlens cannot express that isolation, so they fall back to one forward per request.

diffuse

diffuse(
    *,
    task: str,
    text_embeddings: Tensor,
    text_tags: Tensor,
    seed: int,
    latent_t: int,
    latent_h: int,
    latent_w: int,
    audio_t: int,
    num_frames: int,
    num_steps: int,
    video_shift: float,
    audio_shift: float,
    base_schedule: Sequence[float] | None,
    visual_condition: Tensor | None,
    visual_condition_shape: tuple[int, int, int] | None,
    audio_condition: Tensor | None,
    ref_audio_t: int | None,
    ref_blocks: list[dict[str, Any]] | None = None,
    visual_condition_shapes: list[tuple[int, int, int]]
    | None = None,
    audio_condition_lengths: list[int] | None = None,
    keyframe_frame_indices: list[int] | None = None,
    pad_seq_len: int | None = None,
    locked_audio_rows: Tensor | None = None,
    temporal_offset: float = 0.0,
    media_time_origin: int | None = None,
    video_edit_clean_rows: Tensor | None = None,
    video_edit_mask_rows: Tensor | None = None,
    video_edit_restore_mask_rows: Tensor | None = None,
    audio_edit_clean_rows: Tensor | None = None,
    audio_edit_mask_rows: Tensor | None = None,
    audio_edit_restore_mask_rows: Tensor | None = None,
) -> tuple[Tensor, Tensor]

disable_omni_model_cpu_offload

disable_omni_model_cpu_offload() -> None

enable_omni_model_cpu_offload

enable_omni_model_cpu_offload(
    *,
    device: device,
    pin_memory: bool,
    use_hsdp: bool,
    offload_components: frozenset[str] | None = None,
) -> None

encode_prompt

encode_prompt(
    prepared: PreparedEncoderInputs | None,
) -> tuple[Tensor, Tensor]

forward

is_cache_dit_enabled

is_cache_dit_enabled() -> bool

Return the request-scoped Cache-DiT installation state.

load_weights

load_weights(
    weights: Iterable[tuple[str, Tensor]],
) -> set[str]

post_decode

post_decode(
    state: StepRequestState, **kwargs: Any
) -> DiffusionOutput

Unpack the denoised rows and run the joint video/audio VAE decode.

prepare_encode

prepare_encode(
    state: StepRequestState, **kwargs: Any
) -> StepRequestState

Run every request-level stage once and seed the per-request step state.

step_scheduler

step_scheduler(
    state: StepRequestState,
    noise_pred: Tensor,
    **kwargs: Any,
) -> None

Apply one Euler-eta0 update to this request's video and audio rows.

get_minimax_h3_post_process_func

get_minimax_h3_post_process_func(
    od_config: OmniDiffusionConfig,
)

resolve_minimax_h3_diffusion_model_path

resolve_minimax_h3_diffusion_model_path(
    model: str, revision: str | None, task_type: str | None
) -> str

Resolve a repository root or Hub ID to its startup partition.