Skip to content

vllm_omni.diffusion.models.lingbot_video

Modules:

Name Description
image_condition
lingbot_video_transformer
pipeline_lingbot_video
request_utils

LingBotGenerationMode

Bases: str, Enum

T2I class-attribute instance-attribute

T2I = 't2i'

T2V class-attribute instance-attribute

T2V = 't2v'

TI2V class-attribute instance-attribute

TI2V = 'ti2v'

LingBotImageCondition dataclass

clean_latent instance-attribute

clean_latent: Tensor

vlm_image instance-attribute

vlm_image: Image

LingBotRequestConfig dataclass

fps instance-attribute

fps: int

guidance_scale instance-attribute

guidance_scale: float

height instance-attribute

height: int

input_image instance-attribute

input_image: Image | None

mode instance-attribute

negative_prompt instance-attribute

negative_prompt: str

num_frames instance-attribute

num_frames: int

num_inference_steps instance-attribute

num_inference_steps: int

output_type instance-attribute

output_type: str

prompt instance-attribute

prompt: str

shift instance-attribute

shift: float

width instance-attribute

width: int

LingBotVideoPipeline

Bases: Module, SupportImageInput, ProgressBarMixin, SupportsComponentDiscovery

Native vLLM-Omni entry for LingBot-Video checkpoints.

The in-tree transformer supports both dense MLP blocks and routed MoE blocks. Fused expert kernels and the optional refiner/ transformer are not loaded or executed. t_thresh only selects a low-noise sigma schedule for the primary transformer; it does not enable automatic refiner orchestration.

default_image_negative_prompt instance-attribute

default_image_negative_prompt = (
    DEFAULT_NEGATIVE_PROMPT_IMAGE
)

default_negative_prompt instance-attribute

default_negative_prompt = DEFAULT_NEGATIVE_PROMPT

device instance-attribute

device = get_local_device()

hidden_state_skip_layer instance-attribute

hidden_state_skip_layer = HIDDEN_STATE_SKIP_LAYER

img_prompt_template instance-attribute

img_prompt_template = IMG_PROMPT_TEMPLATE

max_outputs_per_prompt class-attribute

max_outputs_per_prompt: int = 1

od_config instance-attribute

od_config = od_config

processor instance-attribute

processor = Qwen3VLProcessor.from_pretrained(
    model,
    subfolder=processor_subfolder,
    local_files_only=local_files_only,
)

prompt_template instance-attribute

prompt_template = PROMPT_TEMPLATE

scheduler instance-attribute

scheduler = FlowUniPCMultistepScheduler.from_pretrained(
    model,
    subfolder=scheduler_subfolder,
    local_files_only=local_files_only,
)

supports_step_execution class-attribute

supports_step_execution: bool = False

text_encoder instance-attribute

text_encoder = (
    Qwen3VLForConditionalGeneration.from_pretrained(
        model,
        subfolder=text_encoder_subfolder,
        **text_encoder_kwargs,
    ).to(self.device)
)

token_length instance-attribute

token_length = TOKEN_LENGTH

transformer instance-attribute

transformer = (
    LingBotVideoTransformer3DModel.from_pretrained(
        model,
        subfolder=transformer_subfolder,
        torch_dtype=transformer_dtype,
        local_files_only=local_files_only,
    ).to(self.device)
)

vae instance-attribute

vae = AutoencoderKLWan.from_pretrained(
    model,
    subfolder=vae_subfolder,
    torch_dtype=vae_dtype,
    local_files_only=local_files_only,
).to(self.device)

vae_scale_factor_spatial instance-attribute

vae_scale_factor_spatial = 8

vae_scale_factor_temporal instance-attribute

vae_scale_factor_temporal = 4

apply_text_to_template staticmethod

apply_text_to_template(
    text: str, template: str = PROMPT_TEMPLATE
) -> str

check_inputs staticmethod

check_inputs(
    height: int, width: int, num_frames: int
) -> None

encode_prompt

encode_prompt(
    prompt: str | list[str],
    *,
    images: Any | None = None,
    device: str | device | None = None,
) -> tuple[Tensor, Tensor]

forward

load_weights

load_weights(
    weights: Iterable[tuple[str, Tensor]],
) -> set[str]

prepare_latents

prepare_latents(
    num_frames: int,
    height: int,
    width: int,
    generator: Generator | None,
    latents: Tensor | None,
    device: device,
) -> Tensor

prepare_ti2v_image_condition

prepare_ti2v_image_condition(
    image: Image,
    *,
    height: int,
    width: int,
    generator: Generator | None = None,
) -> LingBotImageCondition

to

to(*args, **kwargs)

LingBotVideoTransformer3DModel

Bases: ModelMixin, ConfigMixin

blocks instance-attribute

blocks = nn.ModuleList(
    [
        LingBotVideoBlock(
            hidden_size=hidden_size,
            num_attention_heads=num_attention_heads,
            intermediate_size=intermediate_size,
            norm_eps=norm_eps,
            qkv_bias=qkv_bias,
            out_bias=out_bias,
            num_experts=num_experts,
            num_experts_per_tok=num_experts_per_tok,
            moe_intermediate_size=moe_intermediate_size,
            decoder_sparse_step=decoder_sparse_step,
            mlp_only_layers=mlp_only_layers,
            n_shared_experts=n_shared_experts,
            score_func=score_func,
            norm_topk_prob=norm_topk_prob,
            n_group=n_group,
            topk_group=topk_group,
            routed_scaling_factor=routed_scaling_factor,
            layer_idx=i,
        )
        for i in range(depth)
    ]
)

norm_out instance-attribute

norm_out = nn.LayerNorm(
    hidden_size, elementwise_affine=False, eps=norm_eps
)

norm_out_modulation instance-attribute

norm_out_modulation = nn.Sequential(
    nn.SiLU(), nn.Linear(hidden_size, 2 * hidden_size)
)

patch_embedder instance-attribute

patch_embedder = nn.Linear(
    in_channels * math.prod(patch_size),
    hidden_size,
    bias=patch_embed_bias,
)

proj_out instance-attribute

proj_out = nn.Linear(
    hidden_size, math.prod(patch_size) * out_channels
)

rope instance-attribute

rope = LingBotVideoRotaryEmbedding(
    axes_dims, axes_lens, rope_theta
)

text_embedder instance-attribute

text_embedder = LingBotVideoTextEmbedder(
    text_dim, hidden_size
)

time_embedder instance-attribute

time_embedder = TimestepEmbedding(
    freq_dim,
    hidden_size,
    act_fn="silu",
    sample_proj_bias=timestep_mlp_bias,
)

time_modulation instance-attribute

time_modulation = nn.Sequential(
    nn.SiLU(), nn.Linear(hidden_size, 6 * hidden_size)
)

time_proj instance-attribute

time_proj = Timesteps(
    freq_dim, flip_sin_to_cos=True, downscale_freq_shift=0
)

forward

forward(
    hidden_states: Tensor,
    timestep: Tensor,
    encoder_hidden_states: Tensor,
    encoder_attention_mask: Tensor | None = None,
    return_dict: bool = True,
)

to

to(*args, **kwargs)

apply_clean_prefix

apply_clean_prefix(
    latents: Tensor, clean_latent: Tensor
) -> Tensor

caption_from_lingbot_prompt

caption_from_lingbot_prompt(value: Any) -> str

geometry_align_image

geometry_align_image(
    image: Image, height: int, width: int
) -> Tensor

get_lingbot_video_post_process_func

get_lingbot_video_post_process_func(
    od_config: OmniDiffusionConfig,
)

get_lingbot_video_pre_process_func

get_lingbot_video_pre_process_func(
    od_config: OmniDiffusionConfig,
)

normalize_lingbot_num_frames

normalize_lingbot_num_frames(value: Any) -> int

Round a video length up to LingBot's causal VAE 4n+1 grid.

normalize_lingbot_request

normalize_lingbot_request(
    request: Any,
    *,
    default_negative_prompt: str,
    default_image_negative_prompt: str,
    default_height: int = 480,
    default_width: int = 480,
    default_num_frames: int = 81,
    default_fps: int = 24,
    default_num_inference_steps: int = 40,
    default_guidance_scale: float = 6.0,
    default_shift: float = 3.0,
    default_output_type: str = "pt",
) -> LingBotRequestConfig

prepare_ti2v_image_condition

prepare_ti2v_image_condition(
    image: Image,
    *,
    height: int,
    width: int,
    vae: Module,
    vision_patch_size: int,
    device: device,
    generator: Generator | None = None,
) -> LingBotImageCondition

resolve_lingbot_num_frames

resolve_lingbot_num_frames(duration: Any, fps: Any) -> int

resolve_lingbot_output_dimensions

resolve_lingbot_output_dimensions(
    *,
    sampling_width: Any = None,
    sampling_height: Any = None,
    prompt_fields: Mapping[str, Any] | None = None,
    extra_fields: Mapping[str, Any] | None = None,
    default_width: int = 480,
    default_height: int = 480,
) -> tuple[int, int]

Resolve the (width, height) that LingBot will use for generation.

resolve_lingbot_size

resolve_lingbot_size(
    *,
    width: Any = None,
    height: Any = None,
    size: Any = None,
    resolution: Any = None,
    ratio: Any = None,
) -> tuple[int, int]