Skip to content

vllm_omni.diffusion.models.lingbot_video.pipeline_lingbot_video

DEFAULT_NEGATIVE_PROMPT module-attribute

DEFAULT_NEGATIVE_PROMPT = '{"universal_negative": {"visual_quality": ["low quality", "worst quality", "blurry", "pixelated", "jpeg artifacts", "low resolution", "unstable color", "color flicker", "underexposed", "overexposed", "invisible subject", "subject hidden in darkness"], "artistic_style": ["painting", "illustration", "drawing", "cartoon", "3d render", "cgi", "sketch", "digital art"], "composition_and_content": ["text", "watermark", "signature", "logo", "subtitles", "pillarboxed", "side bars", "portrait image in landscape frame"], "temporal_and_motion_stability": ["flickering", "jittery", "motion blur", "temporal inconsistency", "warping", "morphing", "incoherent motion", "unnatural movement", "static object with sudden jump", "frame-to-frame inconsistency"], "material_and_structure": ["plastic-like glass", "unrealistic texture", "deformed bottle", "liquid freezing improperly", "distorted reflections"]}}'

DEFAULT_NEGATIVE_PROMPT_IMAGE module-attribute

DEFAULT_NEGATIVE_PROMPT_IMAGE = '{"universal_negative": {"visual_quality": ["low quality", "worst quality", "blurry", "pixelated", "jpeg artifacts", "low resolution", "underexposed", "overexposed", "invisible subject", "subject hidden in darkness"], "artistic_style": ["painting", "illustration", "drawing", "cartoon", "3d render", "cgi", "sketch", "digital art"], "composition_and_content": ["text", "watermark", "signature", "logo", "pillarboxed", "side bars", "portrait image in landscape frame"], "material_and_structure": ["plastic-like glass", "unrealistic texture", "deformed bottle", "distorted reflections"]}}'

HIDDEN_STATE_SKIP_LAYER module-attribute

HIDDEN_STATE_SKIP_LAYER = 0

IMG_PROMPT_TEMPLATE module-attribute

IMG_PROMPT_TEMPLATE = (
    "<|vision_start|><|image_pad|><|vision_end|>"
)

LOW_NOISE_TAIL_V1_DEFAULT_STEPS module-attribute

LOW_NOISE_TAIL_V1_DEFAULT_STEPS = 2

PROMPT_TEMPLATE module-attribute

PROMPT_TEMPLATE = '<|im_start|>system\nGiven a user input that may include a text prompt alone, a text prompt with an image reference, or a text prompt with a video reference or a video reference alone, generate an "Enhanced prompt" that provides detailed visual descriptions suitable for video generation. Evaluate the level of detail in the user\'s input: if it is simple, enrich it by adding specifics about colors, shapes, sizes, textures, lighting, motion dynamics, camera movement, temporal progression, and spatial relationships to create vivid, concrete, and temporally coherent scenes to create vivid and concrete scenes. Please generate only the enhanced description for the prompt below and avoid including any additional commentary or evaluations:<|im_end|>\n<|im_start|>user\n{}<|im_end|>\n<|im_start|>assistant\n'

TOKEN_LENGTH module-attribute

TOKEN_LENGTH = 37698

LingBotVideoPipeline

Bases: Module, SupportImageInput, ProgressBarMixin, SupportsComponentDiscovery

Native vLLM-Omni entry for LingBot-Video checkpoints.

The in-tree transformer supports both dense MLP blocks and routed MoE blocks. Fused expert kernels and the optional refiner/ transformer are not loaded or executed. t_thresh only selects a low-noise sigma schedule for the primary transformer; it does not enable automatic refiner orchestration.

default_image_negative_prompt instance-attribute

default_image_negative_prompt = (
    DEFAULT_NEGATIVE_PROMPT_IMAGE
)

default_negative_prompt instance-attribute

default_negative_prompt = DEFAULT_NEGATIVE_PROMPT

device instance-attribute

device = get_local_device()

hidden_state_skip_layer instance-attribute

hidden_state_skip_layer = HIDDEN_STATE_SKIP_LAYER

img_prompt_template instance-attribute

img_prompt_template = IMG_PROMPT_TEMPLATE

max_outputs_per_prompt class-attribute

max_outputs_per_prompt: int = 1

od_config instance-attribute

od_config = od_config

processor instance-attribute

processor = Qwen3VLProcessor.from_pretrained(
    model,
    subfolder=processor_subfolder,
    local_files_only=local_files_only,
)

prompt_template instance-attribute

prompt_template = PROMPT_TEMPLATE

scheduler instance-attribute

scheduler = FlowUniPCMultistepScheduler.from_pretrained(
    model,
    subfolder=scheduler_subfolder,
    local_files_only=local_files_only,
)

supports_step_execution class-attribute

supports_step_execution: bool = False

text_encoder instance-attribute

text_encoder = (
    Qwen3VLForConditionalGeneration.from_pretrained(
        model,
        subfolder=text_encoder_subfolder,
        **text_encoder_kwargs,
    ).to(self.device)
)

token_length instance-attribute

token_length = TOKEN_LENGTH

transformer instance-attribute

transformer = (
    LingBotVideoTransformer3DModel.from_pretrained(
        model,
        subfolder=transformer_subfolder,
        torch_dtype=transformer_dtype,
        local_files_only=local_files_only,
    ).to(self.device)
)

vae instance-attribute

vae = AutoencoderKLWan.from_pretrained(
    model,
    subfolder=vae_subfolder,
    torch_dtype=vae_dtype,
    local_files_only=local_files_only,
).to(self.device)

vae_scale_factor_spatial instance-attribute

vae_scale_factor_spatial = 8

vae_scale_factor_temporal instance-attribute

vae_scale_factor_temporal = 4

apply_text_to_template staticmethod

apply_text_to_template(
    text: str, template: str = PROMPT_TEMPLATE
) -> str

check_inputs staticmethod

check_inputs(
    height: int, width: int, num_frames: int
) -> None

encode_prompt

encode_prompt(
    prompt: str | list[str],
    *,
    images: Any | None = None,
    device: str | device | None = None,
) -> tuple[Tensor, Tensor]

forward

load_weights

load_weights(
    weights: Iterable[tuple[str, Tensor]],
) -> set[str]

prepare_latents

prepare_latents(
    num_frames: int,
    height: int,
    width: int,
    generator: Generator | None,
    latents: Tensor | None,
    device: device,
) -> Tensor

prepare_ti2v_image_condition

prepare_ti2v_image_condition(
    image: Image,
    *,
    height: int,
    width: int,
    generator: Generator | None = None,
) -> LingBotImageCondition

to

to(*args, **kwargs)

get_lingbot_video_post_process_func

get_lingbot_video_post_process_func(
    od_config: OmniDiffusionConfig,
)

get_lingbot_video_pre_process_func

get_lingbot_video_pre_process_func(
    od_config: OmniDiffusionConfig,
)