vllm_omni.diffusion.models.lingbot_video.request_utils ¶
LINGBOT_RESOLUTION_PRESETS module-attribute ¶
LINGBOT_RESOLUTION_PRESETS: dict[
str, dict[str, tuple[int, int]]
] = {
"192p": {
"1:1": (192, 192),
"9:16": (192, 320),
"16:9": (320, 192),
"3:4": (192, 256),
"4:3": (256, 192),
},
"480p": {
"1:1": (480, 480),
"9:16": (480, 832),
"16:9": (832, 480),
"3:4": (480, 640),
"4:3": (640, 480),
},
"720p": {
"1:1": (736, 736),
"9:16": (736, 1280),
"16:9": (1280, 736),
"3:4": (736, 960),
"4:3": (960, 736),
},
"1080p": {
"1:1": (1088, 1088),
"9:16": (1088, 1920),
"16:9": (1920, 1088),
"3:4": (1088, 1440),
"4:3": (1440, 1088),
},
"2k": {
"1:1": (1440, 1440),
"9:16": (1440, 2560),
"16:9": (2560, 1440),
"3:4": (1440, 1920),
"4:3": (1920, 1440),
},
"4k": {
"1:1": (2176, 2176),
"9:16": (2176, 3840),
"16:9": (3840, 2176),
"3:4": (2176, 2880),
"4:3": (2880, 2176),
},
}
LINGBOT_RUNTIME_PROMPT_FIELDS module-attribute ¶
LINGBOT_RUNTIME_PROMPT_FIELDS = frozenset(
{
"duration",
"seconds",
"fps",
"height",
"width",
"size",
"num_frames",
"resolution",
"ratio",
}
)
LingBotGenerationMode ¶
LingBotRequestConfig dataclass ¶
normalize_lingbot_num_frames ¶
Round a video length up to LingBot's causal VAE 4n+1 grid.
normalize_lingbot_request ¶
normalize_lingbot_request(
request: Any,
*,
default_negative_prompt: str,
default_image_negative_prompt: str,
default_height: int = 480,
default_width: int = 480,
default_num_frames: int = 81,
default_fps: int = 24,
default_num_inference_steps: int = 40,
default_guidance_scale: float = 6.0,
default_shift: float = 3.0,
default_output_type: str = "pt",
) -> LingBotRequestConfig
resolve_lingbot_output_dimensions ¶
resolve_lingbot_output_dimensions(
*,
sampling_width: Any = None,
sampling_height: Any = None,
prompt_fields: Mapping[str, Any] | None = None,
extra_fields: Mapping[str, Any] | None = None,
default_width: int = 480,
default_height: int = 480,
) -> tuple[int, int]
Resolve the (width, height) that LingBot will use for generation.