Skip to content

vllm_omni.platforms.npu.platform

logger module-attribute

logger = init_logger(__name__)

NPUOmniPlatform

Bases: OmniPlatform, NPUPlatform

NPU/Ascend implementation of OmniPlatform.

Inherits all NPU-specific implementations from vllm-ascend's NPUPlatform, and adds Omni-specific interfaces from OmniPlatform.

dist_backend class-attribute instance-attribute

dist_backend: str = 'hccl'

build_diffusion_kv_attn_metadata classmethod

build_diffusion_kv_attn_metadata(
    **kwargs: Any,
) -> dict[str, Any]

Build the Ascend metadata required by the native NPU backend.

configure_diffusion_vllm_config classmethod

configure_diffusion_vllm_config(
    vllm_config: Any, od_config: Any
) -> None

Use the block geometry required by Ascend's native paged kernel.

create_autocast_context classmethod

create_autocast_context(
    *, device_type, dtype, enabled=True
)

get_default_stage_config_path classmethod

get_default_stage_config_path() -> str

get_device_count classmethod

get_device_count() -> int

get_device_memory classmethod

get_device_memory(
    device: device | None = None,
) -> tuple[int, int]

get_device_total_memory classmethod

get_device_total_memory(device_id: int = 0) -> int

get_device_version classmethod

get_device_version() -> str | None

get_diffusion_attn_backend_cls classmethod

get_diffusion_attn_backend_cls(
    selected_backend: str | None,
    head_size: int,
    allow_trtllm_default: bool = True,
) -> str

get_diffusion_kv_block_tables_cls classmethod

get_diffusion_kv_block_tables_cls() -> type

get_diffusion_packed_modules_mapping classmethod

get_diffusion_packed_modules_mapping(
    model_class: type[Module],
) -> dict[str, list[str]] | None

get_diffusion_paged_kv_attn_backend classmethod

get_diffusion_paged_kv_attn_backend(
    attn_backend: type, *, ulysses_degree: int
) -> type

Keep strict Ulysses paged FIA out of vLLM's PCP implementation.

get_free_memory classmethod

get_free_memory(device: device | None = None) -> int

get_graph_wrapper_cls classmethod

get_graph_wrapper_cls() -> type

get_omni_ar_worker_cls classmethod

get_omni_ar_worker_cls() -> str

get_omni_generation_worker_cls classmethod

get_omni_generation_worker_cls() -> str

get_profiler_cls classmethod

get_profiler_cls() -> str

get_torch_device classmethod

get_torch_device(local_rank: int | None = None) -> device

init_diffusion_model_runner_runtime classmethod

init_diffusion_model_runner_runtime(
    vllm_config: Any, od_config: Any, device: device
) -> None

init_diffusion_worker_vllm_config classmethod

init_diffusion_worker_vllm_config(vllm_config: Any) -> None

memory_reserved classmethod

memory_reserved(device: device | int | None = None) -> int

prepare_diffusion_op_runtime classmethod

prepare_diffusion_op_runtime(
    op_name: str, **kwargs: Any
) -> None

record_device_event classmethod

record_device_event() -> Event | None

Record a NPU event on the default stream to mark tensor readiness.

On NPU/Ascend with HCCL, distributed communication may use internal streams not visible to the default stream. Synchronize the default stream first so that HCCL results are written back before we record the event, ensuring d2h_stream.wait_event() captures the complete output data.

register_additional_diffusion_fused_moe_hooks classmethod

register_additional_diffusion_fused_moe_hooks(
    moe_runner: Any,
) -> None

requires_diffusion_paged_kv_prewrite classmethod

requires_diffusion_paged_kv_prewrite() -> bool

Write the full K/V span once before piecewise FIA segments.

reset_diffusion_fused_moe_forward_context classmethod

reset_diffusion_fused_moe_forward_context() -> None

set_device classmethod

set_device(device: device) -> None

set_forward_context classmethod

set_forward_context(
    attn_metadata,
    vllm_config,
    *,
    cudagraph_runtime_mode,
    batch_descriptor,
)

supports_diffusion_dense_flash_attention classmethod

supports_diffusion_dense_flash_attention() -> bool

Return whether MindIE-SD is installed for dense NPU FlashAttention.

supports_torch_inductor classmethod

supports_torch_inductor() -> bool

synchronize classmethod

synchronize() -> None