vllm_omni.model_executor.models.minimax_h3.pipeline ¶
Two-stage MiniMax H3 topology.
MINIMAX_H3_PIPELINE module-attribute ¶
MINIMAX_H3_PIPELINE = PipelineConfig(
model_type="minimax_h3_disaggregated",
default_deploy_config_name="minimax_h3_disaggregated.yaml",
stage_cli_aliases={
"text_encoder_tp_size": (0, "tensor_parallel_size")
},
model_arch="MiniMaxH3Encoder",
stages=(
StagePipelineConfig(
stage_id=0,
model_stage="encoder",
execution_type=StageExecutionType.LLM_AR,
input_sources=(),
owns_tokenizer=True,
requires_multimodal_data=True,
model_arch="MiniMaxH3Encoder",
engine_output_type="latent",
prompt_transform_func=f"{_PROCESSOR}.prepare_encoder_prompt",
custom_process_next_stage_input_func=f"{_PROCESSOR}.encoder2diffusion_full_payload",
sampling_constraints={
"max_tokens": 1,
"temperature": 0.0,
"detokenize": False,
},
model_path_resolver=f"{_CHECKPOINT}.resolve_minimax_h3_model_root",
),
StagePipelineConfig(
stage_id=1,
model_stage="dit",
execution_type=StageExecutionType.DIFFUSION,
input_sources=(0,),
final_output=True,
final_output_type="video",
requires_multimodal_data=False,
model_arch="MiniMaxH3Pipeline",
custom_process_input_func=f"{_PROCESSOR}.encoder2diffusion",
stage_input_payload_keys=("encoder_output",),
omni_kv_config={"need_recv_cache": False},
model_path_resolver=f"{_DIFFUSION_PIPELINE}.resolve_minimax_h3_diffusion_model_path",
inline_diffusion=True,
),
),
)