Skip to content

vllm_omni.diffusion.models.lingbot_video.image_condition

IMAGE_MAX_TOKEN_NUM module-attribute

IMAGE_MAX_TOKEN_NUM = 16384

IMAGE_MIN_TOKEN_NUM module-attribute

IMAGE_MIN_TOKEN_NUM = 4

MAX_IMAGE_RATIO module-attribute

MAX_IMAGE_RATIO = 200

SPATIAL_MERGE_SIZE module-attribute

SPATIAL_MERGE_SIZE = 2

LingBotImageCondition dataclass

clean_latent instance-attribute

clean_latent: Tensor

vlm_image instance-attribute

vlm_image: Image

apply_clean_prefix

apply_clean_prefix(
    latents: Tensor, clean_latent: Tensor
) -> Tensor

encode_clean_image_latent

encode_clean_image_latent(
    pixel: Tensor,
    *,
    vae: Module,
    generator: Generator | None = None,
) -> Tensor

geometry_align_image

geometry_align_image(
    image: Image, height: int, width: int
) -> Tensor

pixel_tensor_to_pil

pixel_tensor_to_pil(pixel: Tensor) -> Image

prepare_ti2v_image_condition

prepare_ti2v_image_condition(
    image: Image,
    *,
    height: int,
    width: int,
    vae: Module,
    vision_patch_size: int,
    device: device,
    generator: Generator | None = None,
) -> LingBotImageCondition

smart_resize

smart_resize(
    height: int,
    width: int,
    factor: int,
    min_pixels: int | None = None,
    max_pixels: int | None = None,
) -> tuple[int, int]

vlm_image_from_aligned

vlm_image_from_aligned(
    pixel: Tensor, *, vision_patch_size: int
) -> Image