Skip to content

vllm_omni.diffusion.models.lingbot_video.request_utils

LINGBOT_RESOLUTION_PRESETS module-attribute

LINGBOT_RESOLUTION_PRESETS: dict[
    str, dict[str, tuple[int, int]]
] = {
    "192p": {
        "1:1": (192, 192),
        "9:16": (192, 320),
        "16:9": (320, 192),
        "3:4": (192, 256),
        "4:3": (256, 192),
    },
    "480p": {
        "1:1": (480, 480),
        "9:16": (480, 832),
        "16:9": (832, 480),
        "3:4": (480, 640),
        "4:3": (640, 480),
    },
    "720p": {
        "1:1": (736, 736),
        "9:16": (736, 1280),
        "16:9": (1280, 736),
        "3:4": (736, 960),
        "4:3": (960, 736),
    },
    "1080p": {
        "1:1": (1088, 1088),
        "9:16": (1088, 1920),
        "16:9": (1920, 1088),
        "3:4": (1088, 1440),
        "4:3": (1440, 1088),
    },
    "2k": {
        "1:1": (1440, 1440),
        "9:16": (1440, 2560),
        "16:9": (2560, 1440),
        "3:4": (1440, 1920),
        "4:3": (1920, 1440),
    },
    "4k": {
        "1:1": (2176, 2176),
        "9:16": (2176, 3840),
        "16:9": (3840, 2176),
        "3:4": (2176, 2880),
        "4:3": (2880, 2176),
    },
}

LINGBOT_RUNTIME_PROMPT_FIELDS module-attribute

LINGBOT_RUNTIME_PROMPT_FIELDS = frozenset(
    {
        "duration",
        "seconds",
        "fps",
        "height",
        "width",
        "size",
        "num_frames",
        "resolution",
        "ratio",
    }
)

logger module-attribute

logger = init_logger(__name__)

LingBotGenerationMode

Bases: str, Enum

T2I class-attribute instance-attribute

T2I = 't2i'

T2V class-attribute instance-attribute

T2V = 't2v'

TI2V class-attribute instance-attribute

TI2V = 'ti2v'

LingBotRequestConfig dataclass

fps instance-attribute

fps: int

guidance_scale instance-attribute

guidance_scale: float

height instance-attribute

height: int

input_image instance-attribute

input_image: Image | None

mode instance-attribute

negative_prompt instance-attribute

negative_prompt: str

num_frames instance-attribute

num_frames: int

num_inference_steps instance-attribute

num_inference_steps: int

output_type instance-attribute

output_type: str

prompt instance-attribute

prompt: str

shift instance-attribute

shift: float

width instance-attribute

width: int

caption_from_lingbot_prompt

caption_from_lingbot_prompt(value: Any) -> str

normalize_lingbot_num_frames

normalize_lingbot_num_frames(value: Any) -> int

Round a video length up to LingBot's causal VAE 4n+1 grid.

normalize_lingbot_request

normalize_lingbot_request(
    request: Any,
    *,
    default_negative_prompt: str,
    default_image_negative_prompt: str,
    default_height: int = 480,
    default_width: int = 480,
    default_num_frames: int = 81,
    default_fps: int = 24,
    default_num_inference_steps: int = 40,
    default_guidance_scale: float = 6.0,
    default_shift: float = 3.0,
    default_output_type: str = "pt",
) -> LingBotRequestConfig

resolve_lingbot_mode

resolve_lingbot_mode(
    prompt_obj: Any,
) -> LingBotGenerationMode

resolve_lingbot_num_frames

resolve_lingbot_num_frames(duration: Any, fps: Any) -> int

resolve_lingbot_output_dimensions

resolve_lingbot_output_dimensions(
    *,
    sampling_width: Any = None,
    sampling_height: Any = None,
    prompt_fields: Mapping[str, Any] | None = None,
    extra_fields: Mapping[str, Any] | None = None,
    default_width: int = 480,
    default_height: int = 480,
) -> tuple[int, int]

Resolve the (width, height) that LingBot will use for generation.

resolve_lingbot_size

resolve_lingbot_size(
    *,
    width: Any = None,
    height: Any = None,
    size: Any = None,
    resolution: Any = None,
    ratio: Any = None,
) -> tuple[int, int]