vllm_omni.diffusion.models.cosmos3.action ¶
Action-token helpers for Cosmos3 action generation.
These helpers cover action modes that use action tokens as an auxiliary output stream: policy, forward_dynamics, and inverse_dynamics. The pipeline returns predicted actions through the diffusion output payload/metadata envelope rather than as a video frame stream.
ACTION_MODES module-attribute ¶
ACTION_MODES = {
ACTION_MODE_POLICY,
ACTION_MODE_FORWARD_DYNAMICS,
ACTION_MODE_INVERSE_DYNAMICS,
}
EMBODIMENT_TO_DOMAIN_ID module-attribute ¶
EMBODIMENT_TO_DOMAIN_ID: dict[str, int] = {
"no_action": 0,
"av": 1,
"camera_pose": 2,
"hand_pose": 3,
"pusht": 4,
"libero": 5,
"umi": 6,
"bridge_orig_lerobot": 7,
"droid_lerobot": 8,
"robomind-franka": 8,
"embodiment_b": 9,
"galbot": 9,
"robomind-franka-dual": 12,
"robomind-ur": 13,
"agibotworld": 15,
"embodiment_c_gripper": 15,
"embodiment_c_gripper_ext": 15,
"agibot_gear_gripper": 15,
"agibot_gear_gripper_ext": 15,
"xdof_yam": 16,
"molmoact2_yam": 16,
"abc_yam": 16,
"fractal": 20,
"drawanything": 21,
"behavior1k_lerobot": 22,
"maniparena": 23,
}
action_start_frame_offset ¶
build_action_condition_mask ¶
build_action_condition_mask(
mode: str,
action_length: int,
*,
device: device,
dtype: dtype,
) -> Tensor
build_vision_condition_mask ¶
build_vision_condition_mask(
mode: str,
video_length: int,
temporal_compression_factor: int,
*,
device: device,
dtype: dtype,
) -> Tensor