Skip to content

vllm_omni.model_executor.models.minimax_h3.pipeline

Two-stage MiniMax H3 topology.

MINIMAX_H3_PIPELINE module-attribute

MINIMAX_H3_PIPELINE = PipelineConfig(
    model_type="minimax_h3_disaggregated",
    default_deploy_config_name="minimax_h3_disaggregated.yaml",
    stage_cli_aliases={
        "text_encoder_tp_size": (0, "tensor_parallel_size")
    },
    model_arch="MiniMaxH3Encoder",
    stages=(
        StagePipelineConfig(
            stage_id=0,
            model_stage="encoder",
            execution_type=StageExecutionType.LLM_AR,
            input_sources=(),
            owns_tokenizer=True,
            requires_multimodal_data=True,
            model_arch="MiniMaxH3Encoder",
            engine_output_type="latent",
            prompt_transform_func=f"{_PROCESSOR}.prepare_encoder_prompt",
            custom_process_next_stage_input_func=f"{_PROCESSOR}.encoder2diffusion_full_payload",
            sampling_constraints={
                "max_tokens": 1,
                "temperature": 0.0,
                "detokenize": False,
            },
            model_path_resolver=f"{_CHECKPOINT}.resolve_minimax_h3_model_root",
        ),
        StagePipelineConfig(
            stage_id=1,
            model_stage="dit",
            execution_type=StageExecutionType.DIFFUSION,
            input_sources=(0,),
            final_output=True,
            final_output_type="video",
            requires_multimodal_data=False,
            model_arch="MiniMaxH3Pipeline",
            custom_process_input_func=f"{_PROCESSOR}.encoder2diffusion",
            stage_input_payload_keys=("encoder_output",),
            omni_kv_config={"need_recv_cache": False},
            model_path_resolver=f"{_DIFFUSION_PIPELINE}.resolve_minimax_h3_diffusion_model_path",
            inline_diffusion=True,
        ),
    ),
)