Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,10 @@ def deepseek_v3_pretrain_256gpu_gb200_bf16_config() -> ConfigContainer:
"NCCL_GRAPH_REGISTER": 0,
"PYTORCH_CUDA_ALLOC_CONF": "expandable_segments:True",
"TORCH_NCCL_AVOID_RECORD_STREAMS": 1,
# PyTorch 2.14 prints a C++ warning on every deprecated CUDAGraph.register_generator_state() call.
# TE layer-graph capture with PP>1 makes thousands per rank, and the stderr flood can stall ranks
# past the NCCL timeout, so keep only C++ errors.
"TORCH_CPP_LOG_LEVEL": "ERROR",
# NCCL user-buffer and launch settings.
"NCCL_NVLS_ENABLE": 0,
# HybridEP topology for the target system.
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -73,6 +73,10 @@ def deepseek_v3_pretrain_256gpu_gb300_bf16_config() -> ConfigContainer:
"NCCL_GRAPH_REGISTER": 0,
"PYTORCH_CUDA_ALLOC_CONF": "expandable_segments:True",
"TORCH_NCCL_AVOID_RECORD_STREAMS": 1,
# PyTorch 2.14 prints a C++ warning on every deprecated CUDAGraph.register_generator_state() call.
# TE layer-graph capture with PP>1 makes thousands per rank, and the stderr flood can stall ranks
# past the NCCL timeout, so keep only C++ errors.
"TORCH_CPP_LOG_LEVEL": "ERROR",
# NCCL user-buffer and launch settings.
"NCCL_NVLS_ENABLE": 0,
# NCCL EP dispatcher mode and one GPU per rank.
Expand Down
4 changes: 4 additions & 0 deletions src/megatron/bridge/perf_recipes/qwen/gb200/qwen3_moe.py
Original file line number Diff line number Diff line change
Expand Up @@ -379,6 +379,10 @@ def qwen3_235b_a22b_pretrain_256gpu_gb200_bf16_config() -> ConfigContainer:
"NCCL_GRAPH_REGISTER": 0,
"PYTORCH_CUDA_ALLOC_CONF": "expandable_segments:True",
"TORCH_NCCL_AVOID_RECORD_STREAMS": 1,
# PyTorch 2.14 prints a C++ warning on every deprecated CUDAGraph.register_generator_state() call.
# TE layer-graph capture with PP>1 makes thousands per rank, and the stderr flood can stall ranks
# past the NCCL timeout, so keep only C++ errors.
"TORCH_CPP_LOG_LEVEL": "ERROR",
# NCCL user-buffer and launch settings.
"NCCL_NVLS_ENABLE": 0,
# HybridEP topology for the target system.
Expand Down
4 changes: 4 additions & 0 deletions src/megatron/bridge/perf_recipes/qwen/gb300/qwen3_moe.py
Original file line number Diff line number Diff line change
Expand Up @@ -418,6 +418,10 @@ def qwen3_235b_a22b_pretrain_256gpu_gb300_bf16_config() -> ConfigContainer:
"NCCL_GRAPH_REGISTER": 0,
"PYTORCH_CUDA_ALLOC_CONF": "expandable_segments:True",
"TORCH_NCCL_AVOID_RECORD_STREAMS": 1,
# PyTorch 2.14 prints a C++ warning on every deprecated CUDAGraph.register_generator_state() call.
# TE layer-graph capture with PP>1 makes thousands per rank, and the stderr flood can stall ranks
# past the NCCL timeout, so keep only C++ errors.
"TORCH_CPP_LOG_LEVEL": "ERROR",
# NCCL user-buffer and launch settings.
"NCCL_NVLS_ENABLE": 0,
# NCCL EP dispatcher mode and one GPU per rank.
Expand Down
16 changes: 16 additions & 0 deletions tests/unit_tests/recipes/test_perf_recipe_environment.py
Original file line number Diff line number Diff line change
Expand Up @@ -274,3 +274,19 @@ def test_representative_recipe_specific_environment_is_visible(relative_path, fu
environment = _explicit_environment(_RECIPE_ROOT / relative_path, function_name)

assert environment.items() >= expected.items()


@pytest.mark.parametrize(
("relative_path", "function_name"),
[
("deepseek/gb200/deepseek_v3.py", "deepseek_v3_pretrain_256gpu_gb200_bf16_config"),
("deepseek/gb300/deepseek_v3.py", "deepseek_v3_pretrain_256gpu_gb300_bf16_config"),
("qwen/gb200/qwen3_moe.py", "qwen3_235b_a22b_pretrain_256gpu_gb200_bf16_config"),
("qwen/gb300/qwen3_moe.py", "qwen3_235b_a22b_pretrain_256gpu_gb300_bf16_config"),
],
)
def test_te_layer_graph_pipeline_recipes_keep_only_cpp_errors(relative_path, function_name):
"""TE layer-graph capture with PP>1 floods stderr with PyTorch 2.14 C++ deprecation warnings."""
environment = _explicit_environment(_RECIPE_ROOT / relative_path, function_name)

assert environment["TORCH_CPP_LOG_LEVEL"] == "ERROR"
Loading