Skip to content

vllm_omni.model_executor.models.audex.pipeline

Audex (Nemotron-Labs-Audex-2B) TTS pipeline topology.

Stage 0: Thinker — ChatML text prompt → tokens (LLM AR). Stage 1: Code2Wav — streaming causal speech decoder → 16 kHz waveform.

Users pass the HF repo ROOT (its manifest config.json carries model_type: nemotron_labs_audex); per-stage checkpoints resolve through model_subdir/tokenizer_subdir. The decoder subfolder has no tokenizer files, so stage 1's tokenizer also points at the thinker checkpoint.

AUDEX_AUDIOGEN_END_TOKEN_ID module-attribute

AUDEX_AUDIOGEN_END_TOKEN_ID = 131074

AUDEX_S2S_PIPELINE module-attribute

AUDEX_S2S_PIPELINE = PipelineConfig(
    model_type="audex_s2s",
    default_deploy_config_name="audex_s2s.yaml",
    stages=(
        StagePipelineConfig(
            stage_id=0,
            model_stage="audex_omni",
            execution_type=StageExecutionType.LLM_AR,
            input_sources=(),
            final_output=True,
            final_output_type="text",
            owns_tokenizer=True,
            requires_multimodal_data=True,
            engine_output_type="latent",
            model_subdir="checkpoint_folder_full",
            tokenizer_subdir="checkpoint_folder_full",
            prompt_expand_func=f"{_PROC}.expand_cfg_prompts",
            async_chunk_process_next_stage_input_func=f"{_PROC}.thinker2code2wav_async_chunk",
            custom_process_next_stage_input_func=f"{_PROC}.thinker2code2wav_full_payload",
            sampling_constraints={
                "detokenize": True,
                "stop_token_ids": [
                    AUDEX_SPEECHGEN_END_TOKEN_ID
                ],
            },
        ),
        StagePipelineConfig(
            stage_id=1,
            model_stage="audex_code2wav",
            execution_type=StageExecutionType.LLM_GENERATION,
            input_sources=(0,),
            final_output=True,
            final_output_type="audio",
            engine_output_type="audio",
            model_arch="AudexCode2Wav",
            model_subdir="audex_causal_speech_decoder",
            tokenizer_subdir="checkpoint_folder_audiogen",
            sync_process_input_func=f"{_PROC}.thinker2code2wav_token_only",
            sampling_constraints={"detokenize": True},
            requires_full_payload_input=True,
        ),
    ),
)

AUDEX_SPEECHGEN_END_TOKEN_ID module-attribute

AUDEX_SPEECHGEN_END_TOKEN_ID = 131076

AUDEX_THINKER_ONLY_PIPELINE module-attribute

AUDEX_THINKER_ONLY_PIPELINE = PipelineConfig(
    model_type="audex_thinker_only",
    default_deploy_config_name="audex_thinker_only.yaml",
    stages=(
        StagePipelineConfig(
            stage_id=0,
            model_stage="audex_omni",
            execution_type=StageExecutionType.LLM_AR,
            input_sources=(),
            final_output=True,
            final_output_type="text",
            owns_tokenizer=True,
            requires_multimodal_data=True,
            engine_output_type="text",
            model_subdir="checkpoint_folder_full",
            tokenizer_subdir="checkpoint_folder_full",
            sampling_constraints={"detokenize": True},
        ),
    ),
)

AUDEX_TTA_PIPELINE module-attribute

AUDEX_TTA_PIPELINE = PipelineConfig(
    model_type="audex_tta",
    default_deploy_config_name="audex_tta.yaml",
    stages=(
        StagePipelineConfig(
            stage_id=0,
            model_stage="audex_tta_thinker",
            execution_type=StageExecutionType.LLM_AR,
            input_sources=(),
            owns_tokenizer=True,
            engine_output_type="latent",
            model_subdir="checkpoint_folder_audiogen",
            tokenizer_subdir="checkpoint_folder_audiogen",
            prompt_expand_func=f"{_PROC}.expand_cfg_prompts",
            custom_process_next_stage_input_func=f"{_PROC}.thinker2xcodec_full_payload",
            sampling_constraints={
                "detokenize": False,
                "stop_token_ids": [
                    AUDEX_AUDIOGEN_END_TOKEN_ID
                ],
            },
        ),
        StagePipelineConfig(
            stage_id=1,
            model_stage="audex_xcodec",
            execution_type=StageExecutionType.LLM_GENERATION,
            input_sources=(0,),
            final_output=True,
            final_output_type="audio",
            engine_output_type="audio",
            model_arch="AudexXCodec1",
            sync_process_input_func=f"{_PROC}.thinker2code2wav_token_only",
            sampling_constraints={"detokenize": True},
            requires_full_payload_input=True,
        ),
    ),
)

AUDEX_TTS_PIPELINE module-attribute

AUDEX_TTS_PIPELINE = PipelineConfig(
    model_type="audex_tts",
    default_deploy_config_name="audex_tts.yaml",
    stages=(
        StagePipelineConfig(
            stage_id=0,
            model_stage="audex_thinker",
            execution_type=StageExecutionType.LLM_AR,
            input_sources=(),
            owns_tokenizer=True,
            engine_output_type="latent",
            model_subdir="checkpoint_folder_audiogen",
            tokenizer_subdir="checkpoint_folder_audiogen",
            prompt_expand_func=f"{_PROC}.expand_cfg_prompts",
            async_chunk_process_next_stage_input_func=f"{_PROC}.thinker2code2wav_async_chunk",
            custom_process_next_stage_input_func=f"{_PROC}.thinker2code2wav_full_payload",
            sampling_constraints={
                "detokenize": False,
                "stop_token_ids": [
                    AUDEX_SPEECHGEN_END_TOKEN_ID
                ],
            },
        ),
        StagePipelineConfig(
            stage_id=1,
            model_stage="audex_code2wav",
            execution_type=StageExecutionType.LLM_GENERATION,
            input_sources=(0,),
            final_output=True,
            final_output_type="audio",
            engine_output_type="audio",
            model_arch="AudexCode2Wav",
            model_subdir="audex_causal_speech_decoder",
            tokenizer_subdir="checkpoint_folder_audiogen",
            sync_process_input_func=f"{_PROC}.thinker2code2wav_token_only",
            sampling_constraints={"detokenize": True},
            requires_full_payload_input=True,
        ),
    ),
)