vllm_omni.diffusion.models.cosmos3.utils ¶
COSMOS3_DEFAULT_CONDITION_FRAME_INDEXES_VISION module-attribute ¶
COSMOS3_DEFAULT_CONDITION_VIDEO_KEEP module-attribute ¶
ROBOLAB_CONCAT_VIEW_DESCRIPTION module-attribute ¶
ROBOLAB_CONCAT_VIEW_DESCRIPTION = "The top row is from the wrist-mounted camera. The bottom row contains two horizontally concatenated third-person perspective views of the scene from opposite sides, with the robot visible."
VIDEO_RES_SIZE_INFO module-attribute ¶
VIDEO_RES_SIZE_INFO: dict[
str, dict[str, tuple[int, int]]
] = {
"256": {
"1,1": (256, 256),
"4,3": (320, 256),
"3,4": (256, 320),
"16,9": (320, 192),
"9,16": (192, 320),
},
"480": {
"1,1": (640, 640),
"4,3": (736, 544),
"3,4": (544, 736),
"16,9": (832, 480),
"9,16": (480, 832),
},
"704": {
"1,1": (960, 960),
"4,3": (1088, 832),
"3,4": (832, 1088),
"16,9": (1280, 704),
"9,16": (704, 1280),
},
"720": {
"1,1": (960, 960),
"4,3": (1104, 832),
"3,4": (832, 1104),
"16,9": (1280, 720),
"9,16": (720, 1280),
},
}
RoboLabActionPostprocessInputs dataclass ¶
RoboLabPolicyInputs dataclass ¶
build_robolab_unipc_scheduler ¶
condition_pixel_frame_count ¶
condition_pixel_frame_count(
condition_frame_indexes_vision: Iterable[int],
temporal_compression: int = COSMOS3_VAE_TEMPORAL_COMPRESSION,
) -> int
ensure_2d_float_array ¶
extract_robolab_prompt_image ¶
lazy_action_transform_pipeline ¶
make_robolab_action_postprocess_inputs ¶
make_robolab_action_postprocess_inputs(
inputs: RoboLabPolicyInputs,
) -> RoboLabActionPostprocessInputs
next_robolab_seed ¶
normalize_condition_frame_indexes_vision ¶
Normalize Cosmos3 vision-conditioning latent frame indexes.
postprocess_robolab_action ¶
postprocess_robolab_action(
action: Tensor, inputs: RoboLabActionPostprocessInputs
) -> ndarray
preflight_cosmos3_action_framework_imports ¶
Validate every Cosmos Framework symbol used by action policy serving.