vllm_omni.model_executor.models.audex.pipeline ¶
Audex (Nemotron-Labs-Audex-2B) TTS pipeline topology.
Stage 0: Thinker — ChatML text prompt →
Users pass the HF repo ROOT (its manifest config.json carries model_type: nemotron_labs_audex); per-stage checkpoints resolve through model_subdir/tokenizer_subdir. The decoder subfolder has no tokenizer files, so stage 1's tokenizer also points at the thinker checkpoint.
AUDEX_S2S_PIPELINE module-attribute ¶
AUDEX_S2S_PIPELINE = PipelineConfig(
model_type="audex_s2s",
default_deploy_config_name="audex_s2s.yaml",
stages=(
StagePipelineConfig(
stage_id=0,
model_stage="audex_omni",
execution_type=StageExecutionType.LLM_AR,
input_sources=(),
final_output=True,
final_output_type="text",
owns_tokenizer=True,
requires_multimodal_data=True,
engine_output_type="latent",
model_subdir="checkpoint_folder_full",
tokenizer_subdir="checkpoint_folder_full",
prompt_expand_func=f"{_PROC}.expand_cfg_prompts",
async_chunk_process_next_stage_input_func=f"{_PROC}.thinker2code2wav_async_chunk",
custom_process_next_stage_input_func=f"{_PROC}.thinker2code2wav_full_payload",
sampling_constraints={
"detokenize": True,
"stop_token_ids": [
AUDEX_SPEECHGEN_END_TOKEN_ID
],
},
),
StagePipelineConfig(
stage_id=1,
model_stage="audex_code2wav",
execution_type=StageExecutionType.LLM_GENERATION,
input_sources=(0,),
final_output=True,
final_output_type="audio",
engine_output_type="audio",
model_arch="AudexCode2Wav",
model_subdir="audex_causal_speech_decoder",
tokenizer_subdir="checkpoint_folder_audiogen",
sync_process_input_func=f"{_PROC}.thinker2code2wav_token_only",
sampling_constraints={"detokenize": True},
requires_full_payload_input=True,
),
),
)
AUDEX_THINKER_ONLY_PIPELINE module-attribute ¶
AUDEX_THINKER_ONLY_PIPELINE = PipelineConfig(
model_type="audex_thinker_only",
default_deploy_config_name="audex_thinker_only.yaml",
stages=(
StagePipelineConfig(
stage_id=0,
model_stage="audex_omni",
execution_type=StageExecutionType.LLM_AR,
input_sources=(),
final_output=True,
final_output_type="text",
owns_tokenizer=True,
requires_multimodal_data=True,
engine_output_type="text",
model_subdir="checkpoint_folder_full",
tokenizer_subdir="checkpoint_folder_full",
sampling_constraints={"detokenize": True},
),
),
)
AUDEX_TTA_PIPELINE module-attribute ¶
AUDEX_TTA_PIPELINE = PipelineConfig(
model_type="audex_tta",
default_deploy_config_name="audex_tta.yaml",
stages=(
StagePipelineConfig(
stage_id=0,
model_stage="audex_tta_thinker",
execution_type=StageExecutionType.LLM_AR,
input_sources=(),
owns_tokenizer=True,
engine_output_type="latent",
model_subdir="checkpoint_folder_audiogen",
tokenizer_subdir="checkpoint_folder_audiogen",
prompt_expand_func=f"{_PROC}.expand_cfg_prompts",
custom_process_next_stage_input_func=f"{_PROC}.thinker2xcodec_full_payload",
sampling_constraints={
"detokenize": False,
"stop_token_ids": [
AUDEX_AUDIOGEN_END_TOKEN_ID
],
},
),
StagePipelineConfig(
stage_id=1,
model_stage="audex_xcodec",
execution_type=StageExecutionType.LLM_GENERATION,
input_sources=(0,),
final_output=True,
final_output_type="audio",
engine_output_type="audio",
model_arch="AudexXCodec1",
sync_process_input_func=f"{_PROC}.thinker2code2wav_token_only",
sampling_constraints={"detokenize": True},
requires_full_payload_input=True,
),
),
)
AUDEX_TTS_PIPELINE module-attribute ¶
AUDEX_TTS_PIPELINE = PipelineConfig(
model_type="audex_tts",
default_deploy_config_name="audex_tts.yaml",
stages=(
StagePipelineConfig(
stage_id=0,
model_stage="audex_thinker",
execution_type=StageExecutionType.LLM_AR,
input_sources=(),
owns_tokenizer=True,
engine_output_type="latent",
model_subdir="checkpoint_folder_audiogen",
tokenizer_subdir="checkpoint_folder_audiogen",
prompt_expand_func=f"{_PROC}.expand_cfg_prompts",
async_chunk_process_next_stage_input_func=f"{_PROC}.thinker2code2wav_async_chunk",
custom_process_next_stage_input_func=f"{_PROC}.thinker2code2wav_full_payload",
sampling_constraints={
"detokenize": False,
"stop_token_ids": [
AUDEX_SPEECHGEN_END_TOKEN_ID
],
},
),
StagePipelineConfig(
stage_id=1,
model_stage="audex_code2wav",
execution_type=StageExecutionType.LLM_GENERATION,
input_sources=(0,),
final_output=True,
final_output_type="audio",
engine_output_type="audio",
model_arch="AudexCode2Wav",
model_subdir="audex_causal_speech_decoder",
tokenizer_subdir="checkpoint_folder_audiogen",
sync_process_input_func=f"{_PROC}.thinker2code2wav_token_only",
sampling_constraints={"detokenize": True},
requires_full_payload_input=True,
),
),
)