From 5ebc06f8d2df3b461befaace4b194f8c95aa1ad7 Mon Sep 17 00:00:00 2001 From: Xinyu Jiang Date: Tue, 29 Sep 2026 20:28:58 -0400 Subject: [PATCH] [AMD] ci: expand ROCm nightly coverage and refresh runtime estimates (#2541) Co-authored-by: Zhiyao Jiang --- tests/e2e/ckpt/test_qwen3_0.6B_ckpt.py | 2 +- .../fsdp/test_qwen3_4B_fsdp_hybrid_shard_r2s4.py | 3 ++- ...rain__dp2_cp2_pp2__fake_rollout__moe_5layer.py | 3 ++- ...__dp2_cp2_tp2_ep2__fake_rollout__moe_5layer.py | 3 ++- tests/e2e/long/test_qwen2.5_0.5B_gsm8k_async.py | 2 +- tests/e2e/lora/test_lora_qwen2.5_0.5B.py | 2 +- tests/e2e/lora/test_slot_checkpoint_resume.py | 3 ++- .../model_scripts/test_gpt_oss_20b_moe_lora_ci.py | 3 ++- tests/e2e/megatron/test_mimo_7B_mtp_only_grad.py | 2 +- .../megatron/test_qwen3_30B_A3B/test_baseline.py | 2 +- .../test_qwen3_30B_A3B/test_dp_attention.py | 3 ++- .../test_qwen3_30B_A3B/test_r3_baseline.py | 2 +- .../test_qwen3_4B_offload_disaggregated.py | 3 ++- .../e2e/megatron/test_qwen3_5_35b_a3b_lora_ci.py | 15 ++++++++++----- .../precision/test_hf_attention_cp_relayout.py | 2 +- .../e2e/precision/test_qwen3_5_cp_correctness.py | 2 +- .../test_qwen38small.py | 3 ++- tests/e2e/sglang_config/test_sglang_config.py | 2 +- .../test_multi_policy_solver_verifier_gsm8k.py | 3 ++- .../short/test_qwen2.5_0.5B_gsm8k_async_short.py | 2 +- tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short.py | 2 +- .../test_qwen2.5_0.5B_gsm8k_short_legacy_api.py | 3 ++- ....5B_gsm8k_short_ray_rpc_comm_mooncake_store.py | 3 ++- ...2.5_0.5B_gsm8k_short_ray_rpc_comm_ray_store.py | 3 ++- .../short/test_qwen3_0.6B_sft_snapshot_eval.py | 3 ++- tests/e2e/short/test_run_megatron.py | 2 +- tests/fast-gpu/test_fsdp_hybrid_shard.py | 3 ++- tests/fast-gpu/test_nvme_optimizer_main_init.py | 3 ++- tests/fast-gpu/test_qwen3_8_next_long_offsets.py | 3 ++- tests/fast-gpu/test_qwen3_8_next_ops.py | 3 ++- .../test_true_on_policy_logprob_chunking.py | 3 ++- 31 files changed, 58 insertions(+), 35 deletions(-) diff --git a/tests/e2e/ckpt/test_qwen3_0.6B_ckpt.py b/tests/e2e/ckpt/test_qwen3_0.6B_ckpt.py index 1bec921352..3b8898e009 100644 --- a/tests/e2e/ckpt/test_qwen3_0.6B_ckpt.py +++ b/tests/e2e/ckpt/test_qwen3_0.6B_ckpt.py @@ -5,7 +5,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from miles.utils.external_utils import command_utils register_cuda_ci(est_time=900, suite="stage-c-2-gpu-h200", labels=["ckpt"], hardware=["hopper", "blackwell"]) -register_rocm_ci(est_time=1200, suite="nightly-stage-c-2-gpu-mi350", labels=["ckpt"]) +register_rocm_ci(est_time=1500, suite="nightly-stage-c-2-gpu-mi350", labels=["ckpt"]) ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1"))) diff --git a/tests/e2e/fsdp/test_qwen3_4B_fsdp_hybrid_shard_r2s4.py b/tests/e2e/fsdp/test_qwen3_4B_fsdp_hybrid_shard_r2s4.py index 3b921a3703..504a88eeba 100644 --- a/tests/e2e/fsdp/test_qwen3_4B_fsdp_hybrid_shard_r2s4.py +++ b/tests/e2e/fsdp/test_qwen3_4B_fsdp_hybrid_shard_r2s4.py @@ -1,6 +1,6 @@ import os -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from miles.utils.external_utils import command_utils @@ -10,6 +10,7 @@ register_cuda_ci( labels=["fsdp"], hardware=["hopper", "blackwell"], ) +register_rocm_ci(est_time=400, suite="nightly-stage-c-8-gpu-mi350", labels=["fsdp"]) NUM_GPUS = 8 DP_REPLICATE_SIZE = 2 diff --git a/tests/e2e/ft/test_trainer_no_failure__kill_train__dp2_cp2_pp2__fake_rollout__moe_5layer.py b/tests/e2e/ft/test_trainer_no_failure__kill_train__dp2_cp2_pp2__fake_rollout__moe_5layer.py index 96c6228811..c82f99fc69 100644 --- a/tests/e2e/ft/test_trainer_no_failure__kill_train__dp2_cp2_pp2__fake_rollout__moe_5layer.py +++ b/tests/e2e/ft/test_trainer_no_failure__kill_train__dp2_cp2_pp2__fake_rollout__moe_5layer.py @@ -3,7 +3,7 @@ # (the CUDA CI runner's execution model). Scenario logic lives in # tests/e2e/ft/conftest_ft/scenario_trainer_no_failure.py. -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from tests.e2e.ft.conftest_ft.scenario_trainer_no_failure import run_ci register_cuda_ci( @@ -12,6 +12,7 @@ register_cuda_ci( labels=["ft-short"], hardware=["hopper", "blackwell"], ) +register_rocm_ci(est_time=900, suite="nightly-stage-c-8-gpu-mi350", labels=["ft-short"]) _MODE: str = "kill_train__dp2_cp2_pp2__fake_rollout__moe_5layer" diff --git a/tests/e2e/ft/test_trainer_no_failure__kill_train__dp2_cp2_tp2_ep2__fake_rollout__moe_5layer.py b/tests/e2e/ft/test_trainer_no_failure__kill_train__dp2_cp2_tp2_ep2__fake_rollout__moe_5layer.py index be729fe5f1..ee5f27d6ba 100644 --- a/tests/e2e/ft/test_trainer_no_failure__kill_train__dp2_cp2_tp2_ep2__fake_rollout__moe_5layer.py +++ b/tests/e2e/ft/test_trainer_no_failure__kill_train__dp2_cp2_tp2_ep2__fake_rollout__moe_5layer.py @@ -3,7 +3,7 @@ # (the CUDA CI runner's execution model). Scenario logic lives in # tests/e2e/ft/conftest_ft/scenario_trainer_no_failure.py. -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from tests.e2e.ft.conftest_ft.scenario_trainer_no_failure import run_ci register_cuda_ci( @@ -12,6 +12,7 @@ register_cuda_ci( labels=["ft-short"], hardware=["hopper", "blackwell"], ) +register_rocm_ci(est_time=1000, suite="nightly-stage-c-8-gpu-mi350", labels=["ft-short"]) _MODE: str = "kill_train__dp2_cp2_tp2_ep2__fake_rollout__moe_5layer" diff --git a/tests/e2e/long/test_qwen2.5_0.5B_gsm8k_async.py b/tests/e2e/long/test_qwen2.5_0.5B_gsm8k_async.py index a6e3a4e020..79ac726d4f 100644 --- a/tests/e2e/long/test_qwen2.5_0.5B_gsm8k_async.py +++ b/tests/e2e/long/test_qwen2.5_0.5B_gsm8k_async.py @@ -5,7 +5,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from miles.utils.external_utils import command_utils register_cuda_ci(est_time=5400, suite="stage-c-2-gpu-h200", labels=["long"], hardware=["hopper", "blackwell"]) -register_rocm_ci(est_time=5000, suite="nightly-stage-c-2-gpu-mi350", labels=["long"]) +register_rocm_ci(est_time=5400, suite="nightly-stage-c-2-gpu-mi350", labels=["long"]) MODEL_NAME = "Qwen2.5-0.5B-Instruct" MODEL_TYPE = "qwen2.5-0.5B" diff --git a/tests/e2e/lora/test_lora_qwen2.5_0.5B.py b/tests/e2e/lora/test_lora_qwen2.5_0.5B.py index b65cc00b1a..7e9aa56a5c 100644 --- a/tests/e2e/lora/test_lora_qwen2.5_0.5B.py +++ b/tests/e2e/lora/test_lora_qwen2.5_0.5B.py @@ -18,7 +18,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from miles.utils.external_utils import command_utils register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["lora"], hardware=["hopper", "blackwell"]) -register_rocm_ci(est_time=300, suite="nightly-stage-c-4-gpu-mi350", labels=["lora"]) +register_rocm_ci(est_time=500, suite="nightly-stage-c-4-gpu-mi350", labels=["lora"]) ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1"))) diff --git a/tests/e2e/lora/test_slot_checkpoint_resume.py b/tests/e2e/lora/test_slot_checkpoint_resume.py index f5015197e0..dc5bd528db 100644 --- a/tests/e2e/lora/test_slot_checkpoint_resume.py +++ b/tests/e2e/lora/test_slot_checkpoint_resume.py @@ -19,7 +19,7 @@ from megatron.core.tensor_parallel.layers import ColumnParallelLinear from megatron.core.tensor_parallel.random import model_parallel_cuda_manual_seed from megatron.core.transformer.module import MegatronModule from megatron.core.transformer.transformer_config import TransformerConfig -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from torch.utils._pytree import tree_flatten, tree_map from miles.backends.megatron_utils.lora.checkpoint import load_slot, save_slot @@ -27,6 +27,7 @@ from miles.backends.megatron_utils.lora.optimizer import SlotOptimizer, adapter_ from miles.utils.distributed_utils import init_gloo_group register_cuda_ci(est_time=120, suite="stage-b-2-gpu-h200", labels=["lora"], hardware=["hopper"]) +register_rocm_ci(est_time=60, suite="nightly-stage-c-2-gpu-mi350", labels=["lora"]) ADAM = dict(learning_rate=3e-4, beta1=0.9, beta2=0.95, eps=1e-8, weight_decay=0.01, grad_clip_norm=1.0) diff --git a/tests/e2e/megatron/model_scripts/test_gpt_oss_20b_moe_lora_ci.py b/tests/e2e/megatron/model_scripts/test_gpt_oss_20b_moe_lora_ci.py index 019ecd0f8e..0830e0337f 100644 --- a/tests/e2e/megatron/model_scripts/test_gpt_oss_20b_moe_lora_ci.py +++ b/tests/e2e/megatron/model_scripts/test_gpt_oss_20b_moe_lora_ci.py @@ -1,6 +1,6 @@ import os -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from miles.utils.external_utils import command_utils @@ -16,6 +16,7 @@ register_cuda_ci( labels=["megatron", "model-scripts", "lora"], hardware=["hopper", "blackwell"], ) +register_rocm_ci(est_time=1600, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "model-scripts", "lora"]) MODEL_NAME = "gpt-oss-20b-bf16" MODEL_TYPE = "gpt-oss-20b" diff --git a/tests/e2e/megatron/test_mimo_7B_mtp_only_grad.py b/tests/e2e/megatron/test_mimo_7B_mtp_only_grad.py index 9e5b087845..fd611d5c1e 100644 --- a/tests/e2e/megatron/test_mimo_7B_mtp_only_grad.py +++ b/tests/e2e/megatron/test_mimo_7B_mtp_only_grad.py @@ -16,7 +16,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from miles.utils.external_utils import command_utils register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["megatron"], hardware=["hopper", "blackwell"]) -register_rocm_ci(est_time=420, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron"]) +register_rocm_ci(est_time=600, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron"]) MODEL_NAME = "MiMo-7B-RL" MODEL_TYPE = "mimo-7B-rl" diff --git a/tests/e2e/megatron/test_qwen3_30B_A3B/test_baseline.py b/tests/e2e/megatron/test_qwen3_30B_A3B/test_baseline.py index 56e4d691a7..839bd21705 100644 --- a/tests/e2e/megatron/test_qwen3_30B_A3B/test_baseline.py +++ b/tests/e2e/megatron/test_qwen3_30B_A3B/test_baseline.py @@ -18,7 +18,7 @@ register_ci_gate(metric_key="train/train_rollout_kl") register_ci_gate(metric_key="rollout/raw_reward") register_rocm_ci( - est_time=900, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "weight-update", "short", "mooncake"] + est_time=2000, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "weight-update", "short", "mooncake"] ) CASE = CaseConfig( diff --git a/tests/e2e/megatron/test_qwen3_30B_A3B/test_dp_attention.py b/tests/e2e/megatron/test_qwen3_30B_A3B/test_dp_attention.py index 95a89443ab..e5e6ba31de 100644 --- a/tests/e2e/megatron/test_qwen3_30B_A3B/test_dp_attention.py +++ b/tests/e2e/megatron/test_qwen3_30B_A3B/test_dp_attention.py @@ -7,13 +7,14 @@ training partition and per-rollout loss weighting are unchanged.""" import os -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from tests.ci.metric_history import register_ci_gate from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, prepare register_cuda_ci( est_time=1100, suite="stage-c-4-gpu-h200", labels=["megatron", "short"], hardware=["hopper", "blackwell"] ) +register_rocm_ci(est_time=800, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "short"]) register_ci_gate(metric_key="train/grad_norm") register_ci_gate(metric_key="train/ppo_kl") diff --git a/tests/e2e/megatron/test_qwen3_30B_A3B/test_r3_baseline.py b/tests/e2e/megatron/test_qwen3_30B_A3B/test_r3_baseline.py index 3e74fb047e..54eff2d112 100644 --- a/tests/e2e/megatron/test_qwen3_30B_A3B/test_r3_baseline.py +++ b/tests/e2e/megatron/test_qwen3_30B_A3B/test_r3_baseline.py @@ -7,7 +7,7 @@ from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, p register_cuda_ci( est_time=1200, suite="stage-c-4-gpu-h200", labels=["megatron", "replay"], hardware=["hopper", "blackwell"] ) -register_rocm_ci(est_time=1500, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "replay"]) +register_rocm_ci(est_time=1000, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "replay"]) register_ci_gate(metric_key="train/grad_norm") register_ci_gate(metric_key="train/ppo_kl") diff --git a/tests/e2e/megatron/test_qwen3_4B_offload_disaggregated.py b/tests/e2e/megatron/test_qwen3_4B_offload_disaggregated.py index 0eedebefdd..3bf838c990 100644 --- a/tests/e2e/megatron/test_qwen3_4B_offload_disaggregated.py +++ b/tests/e2e/megatron/test_qwen3_4B_offload_disaggregated.py @@ -7,7 +7,7 @@ weights must come from the host backup rather than the unmapped param buffers. --check-weight-update-equal makes the engines verify what they received. """ -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from tests.ci.metric_history import register_ci_gate from miles.utils.external_utils import command_utils @@ -23,6 +23,7 @@ register_cuda_ci( labels=["megatron", "weight-update"], hardware=["hopper", "blackwell"], ) +register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["megatron", "weight-update"]) register_ci_gate(metric_key="train/grad_norm") register_ci_gate(metric_key="rollout/raw_reward") diff --git a/tests/e2e/megatron/test_qwen3_5_35b_a3b_lora_ci.py b/tests/e2e/megatron/test_qwen3_5_35b_a3b_lora_ci.py index 40e7936dde..7620ec06fc 100644 --- a/tests/e2e/megatron/test_qwen3_5_35b_a3b_lora_ci.py +++ b/tests/e2e/megatron/test_qwen3_5_35b_a3b_lora_ci.py @@ -1,7 +1,7 @@ import os from scripts.run_qwen3_5_35b_a3b_lora import ScriptArgs, _prepare_download, _train -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from miles.utils.external_utils import command_utils @@ -19,6 +19,7 @@ register_cuda_ci( labels=["megatron", "model-scripts", "lora"], hardware=["hopper", "blackwell"], ) +register_rocm_ci(est_time=900, suite="nightly-stage-c-8-gpu-mi350", labels=["megatron", "model-scripts", "lora"]) # (name, experts_shared_outer_loras, virtual_experts_serving) _CONFIGS = [ @@ -28,6 +29,13 @@ _CONFIGS = [ def _args(shared_outer: bool, virtual_experts: bool) -> ScriptArgs: + extra_args = "--ci-test --ci-disable-logprobs-checker " + if not virtual_experts: + extra_args += "--no-sglang-lora-use-virtual-experts " + # ROCm shared-expert fusion requires per-expert LoRA factors. + if os.getenv("MILES_HARDWARE_PLATFORM") == "rocm" and shared_outer: + extra_args += "--sglang-disable-shared-experts-fusion " + return ScriptArgs.from_env( model_name="Qwen3.5-35B-A3B", num_nodes=1, @@ -35,10 +43,7 @@ def _args(shared_outer: bool, virtual_experts: bool) -> ScriptArgs: num_rollout=1, experts_shared_outer_loras=shared_outer, enable_wandb=False, - extra_args=( - "--ci-test --ci-disable-logprobs-checker " - + ("" if virtual_experts else "--no-sglang-lora-use-virtual-experts ") - ), + extra_args=extra_args, ) diff --git a/tests/e2e/precision/test_hf_attention_cp_relayout.py b/tests/e2e/precision/test_hf_attention_cp_relayout.py index a2df37def8..3601f51de0 100644 --- a/tests/e2e/precision/test_hf_attention_cp_relayout.py +++ b/tests/e2e/precision/test_hf_attention_cp_relayout.py @@ -14,7 +14,7 @@ import torch.distributed as dist from tests.ci.ci_register import register_cuda_ci, register_rocm_ci register_cuda_ci(est_time=30, suite="stage-c-4-gpu-h200", labels=["precision"], hardware=["hopper", "blackwell"]) -register_rocm_ci(est_time=60, suite="nightly-stage-c-4-gpu-mi350", labels=["precision"]) +register_rocm_ci(est_time=30, suite="nightly-stage-c-4-gpu-mi350", labels=["precision"]) from miles_plugins.models.cp_utils import packed_shard_to_zigzag, zigzag_to_packed_shard diff --git a/tests/e2e/precision/test_qwen3_5_cp_correctness.py b/tests/e2e/precision/test_qwen3_5_cp_correctness.py index b86c10a43e..e33dc125c3 100644 --- a/tests/e2e/precision/test_qwen3_5_cp_correctness.py +++ b/tests/e2e/precision/test_qwen3_5_cp_correctness.py @@ -17,7 +17,7 @@ import torch.distributed as dist from tests.ci.ci_register import register_cuda_ci, register_rocm_ci register_cuda_ci(est_time=300, suite="stage-c-4-gpu-h200", labels=["precision"], hardware=["hopper", "blackwell"]) -register_rocm_ci(est_time=120, suite="nightly-stage-c-4-gpu-mi350", labels=["precision"]) +register_rocm_ci(est_time=200, suite="nightly-stage-c-4-gpu-mi350", labels=["precision"]) def setup_dist(): diff --git a/tests/e2e/sglang/test_session_server_multi_role/test_qwen38small.py b/tests/e2e/sglang/test_session_server_multi_role/test_qwen38small.py index 569ee95d5f..ab0f9176a7 100644 --- a/tests/e2e/sglang/test_session_server_multi_role/test_qwen38small.py +++ b/tests/e2e/sglang/test_session_server_multi_role/test_qwen38small.py @@ -1,8 +1,9 @@ -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from tests.ci.metric_history import register_ci_gate from tests.e2e.sglang.test_session_server_multi_role._common import ModelConfig, run_both_versions register_cuda_ci(est_time=800, suite="stage-c-2-gpu-h200", labels=["sglang"], hardware=["hopper", "blackwell"]) +register_rocm_ci(est_time=700, suite="nightly-stage-c-2-gpu-mi350", labels=["sglang"]) register_ci_gate(metric_key="rollout/tito_session_mismatch_rate/v1/assistant_text") register_ci_gate(metric_key="rollout/tito_session_mismatch_rate/v2/assistant_text") diff --git a/tests/e2e/sglang_config/test_sglang_config.py b/tests/e2e/sglang_config/test_sglang_config.py index 6f03438401..f193270515 100644 --- a/tests/e2e/sglang_config/test_sglang_config.py +++ b/tests/e2e/sglang_config/test_sglang_config.py @@ -5,7 +5,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from miles.utils.external_utils import command_utils register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"]) -register_rocm_ci(est_time=600, suite="nightly-stage-c-4-gpu-mi350", labels=["short"]) +register_rocm_ci(est_time=400, suite="nightly-stage-c-4-gpu-mi350", labels=["short"]) MODEL_NAME = "Qwen2.5-0.5B-Instruct" MODEL_TYPE = "qwen2.5-0.5B" diff --git a/tests/e2e/short/test_multi_policy_solver_verifier_gsm8k.py b/tests/e2e/short/test_multi_policy_solver_verifier_gsm8k.py index c72514123c..47104d9a0c 100644 --- a/tests/e2e/short/test_multi_policy_solver_verifier_gsm8k.py +++ b/tests/e2e/short/test_multi_policy_solver_verifier_gsm8k.py @@ -2,7 +2,7 @@ import dataclasses import os from examples.multi_policy.run_solver_verifier_gsm8k import ScriptArgs, prepare -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from tests.e2e.conftest_multi_policy import execute from miles.utils.external_utils import command_utils @@ -13,6 +13,7 @@ register_cuda_ci( labels=["short", "multi-policy", "fully-async"], hardware=["hopper", "blackwell"], ) +register_rocm_ci(est_time=1700, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "multi-policy", "fully-async"]) NUM_ROLLOUT = int(os.environ.get("MILES_TEST_NUM_ROLLOUT", "5")) SAVE_INTERVAL = 2 diff --git a/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_async_short.py b/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_async_short.py index 07ad891d3a..2672c9b8da 100644 --- a/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_async_short.py +++ b/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_async_short.py @@ -7,7 +7,7 @@ from miles.utils.external_utils import command_utils register_cuda_ci( est_time=400, suite="stage-c-4-gpu-h200", labels=["short", "mooncake"], hardware=["hopper", "blackwell"] ) -register_rocm_ci(est_time=240, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "mooncake"]) +register_rocm_ci(est_time=300, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "mooncake"]) MODEL_NAME = "Qwen2.5-0.5B-Instruct" MODEL_TYPE = "qwen2.5-0.5B" diff --git a/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short.py b/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short.py index 456925c47c..d83c42c97e 100644 --- a/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short.py +++ b/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short.py @@ -10,7 +10,7 @@ from miles.utils.workers.types import WorkerCommBackend register_cuda_ci( est_time=400, suite="stage-c-2-gpu-h200", labels=["short", "mooncake"], hardware=["hopper", "blackwell"] ) -register_rocm_ci(est_time=360, suite="nightly-stage-c-2-gpu-mi350", labels=["short", "mooncake"]) +register_rocm_ci(est_time=300, suite="nightly-stage-c-2-gpu-mi350", labels=["short", "mooncake"]) MODEL_DIR = get_test_model_dir() DATA_DIR = get_test_data_dir() diff --git a/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short_legacy_api.py b/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short_legacy_api.py index 73f395bdc9..8e2c75430b 100644 --- a/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short_legacy_api.py +++ b/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short_legacy_api.py @@ -1,6 +1,6 @@ import os -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci import miles.utils.external_utils.command_utils.legacy as U @@ -11,6 +11,7 @@ register_cuda_ci( labels=["short", "mooncake"], hardware=["hopper", "blackwell"], ) +register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "mooncake"]) FEW_GPU = U.get_bool_env_var("MILES_TEST_FEW_GPU", "0") diff --git a/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short_ray_rpc_comm_mooncake_store.py b/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short_ray_rpc_comm_mooncake_store.py index 330ed4649f..60006a0d12 100644 --- a/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short_ray_rpc_comm_mooncake_store.py +++ b/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short_ray_rpc_comm_mooncake_store.py @@ -2,7 +2,7 @@ import importlib.util from pathlib import Path from types import ModuleType -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci register_cuda_ci( est_time=450, @@ -10,6 +10,7 @@ register_cuda_ci( labels=["short", "mooncake", "rpc-comm"], hardware=["hopper", "blackwell"], ) +register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "mooncake", "rpc-comm"]) def _load_base_test() -> ModuleType: diff --git a/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short_ray_rpc_comm_ray_store.py b/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short_ray_rpc_comm_ray_store.py index 2898bc84f5..9d10fff951 100644 --- a/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short_ray_rpc_comm_ray_store.py +++ b/tests/e2e/short/test_qwen2.5_0.5B_gsm8k_short_ray_rpc_comm_ray_store.py @@ -2,7 +2,7 @@ import importlib.util from pathlib import Path from types import ModuleType -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci register_cuda_ci( est_time=450, @@ -10,6 +10,7 @@ register_cuda_ci( labels=["short", "rpc-comm"], hardware=["hopper", "blackwell"], ) +register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "rpc-comm"]) def _load_base_test() -> ModuleType: diff --git a/tests/e2e/short/test_qwen3_0.6B_sft_snapshot_eval.py b/tests/e2e/short/test_qwen3_0.6B_sft_snapshot_eval.py index 23b447a643..714acf338f 100644 --- a/tests/e2e/short/test_qwen3_0.6B_sft_snapshot_eval.py +++ b/tests/e2e/short/test_qwen3_0.6B_sft_snapshot_eval.py @@ -12,7 +12,7 @@ tests/e2e/megatron/test_qwen3_4b_fully_async_eval.py. import os import pandas as pd -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from transformers import AutoTokenizer from miles.utils.external_utils import command_utils @@ -23,6 +23,7 @@ register_cuda_ci( labels=["short", "eval", "megatron"], hardware=["hopper", "blackwell"], ) +register_rocm_ci(est_time=300, suite="nightly-stage-c-2-gpu-mi350", labels=["short", "eval", "megatron"]) MODEL_NAME = "Qwen3-0.6B" MODEL_TYPE = "qwen3-0.6B" diff --git a/tests/e2e/short/test_run_megatron.py b/tests/e2e/short/test_run_megatron.py index aae84e5b5e..ebdf5230bd 100644 --- a/tests/e2e/short/test_run_megatron.py +++ b/tests/e2e/short/test_run_megatron.py @@ -40,7 +40,7 @@ NUM_LAYERS: int = 5 _RUN_DIR: Path = Path(tempfile.mkdtemp(prefix="test_run_megatron_")) register_cuda_ci(est_time=200, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"]) -register_rocm_ci(est_time=2000, suite="nightly-stage-c-8-gpu-mi350", labels=["short"]) +register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short"]) @dataclasses.dataclass(frozen=True) diff --git a/tests/fast-gpu/test_fsdp_hybrid_shard.py b/tests/fast-gpu/test_fsdp_hybrid_shard.py index 8601cca601..3d1d3fd626 100644 --- a/tests/fast-gpu/test_fsdp_hybrid_shard.py +++ b/tests/fast-gpu/test_fsdp_hybrid_shard.py @@ -1,8 +1,9 @@ """FSDP2 gradient parity across r1s4, r2s2, and r4s1.""" -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci register_cuda_ci(est_time=90, suite="stage-c-4-gpu-h200", labels=["fsdp"], hardware=["hopper", "blackwell"]) +register_rocm_ci(est_time=90, suite="nightly-stage-c-4-gpu-mi350", labels=["fsdp"]) import os import subprocess diff --git a/tests/fast-gpu/test_nvme_optimizer_main_init.py b/tests/fast-gpu/test_nvme_optimizer_main_init.py index c2d3a30bd5..62018b3038 100644 --- a/tests/fast-gpu/test_nvme_optimizer_main_init.py +++ b/tests/fast-gpu/test_nvme_optimizer_main_init.py @@ -4,7 +4,7 @@ import os from types import SimpleNamespace import torch -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci from miles_plugins.optimizers.nvme_stream import NVMeOptimizerStateStore, _Bucket, _Entry, _resize, _Stager @@ -14,6 +14,7 @@ register_cuda_ci( labels=["miles-plugin"], hardware=["hopper", "blackwell"], ) +register_rocm_ci(est_time=30, suite="nightly-stage-c-2-gpu-mi350", labels=["miles-plugin"]) def test_bucketwise_main_initialization_preserves_bytes_and_releases_cuda_storage(tmp_path): diff --git a/tests/fast-gpu/test_qwen3_8_next_long_offsets.py b/tests/fast-gpu/test_qwen3_8_next_long_offsets.py index 7b8db87838..622b8ad236 100644 --- a/tests/fast-gpu/test_qwen3_8_next_long_offsets.py +++ b/tests/fast-gpu/test_qwen3_8_next_long_offsets.py @@ -4,9 +4,10 @@ Only sampled rows have autograd references, keeping the test smaller than a full model. Inputs vary across rows and channels so a wrong in-bounds address also fails. """ -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci register_cuda_ci(est_time=180, suite="stage-b-2-gpu-h200", labels=["miles-plugin"], hardware=["hopper", "blackwell"]) +register_rocm_ci(est_time=40, suite="nightly-stage-c-2-gpu-mi350", labels=["miles-plugin"]) import math diff --git a/tests/fast-gpu/test_qwen3_8_next_ops.py b/tests/fast-gpu/test_qwen3_8_next_ops.py index 13c7171091..d1dc079369 100644 --- a/tests/fast-gpu/test_qwen3_8_next_ops.py +++ b/tests/fast-gpu/test_qwen3_8_next_ops.py @@ -1,8 +1,9 @@ """Qwen3.8-Flash-Next triton kernels must match their torch references, forward and backward.""" -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci register_cuda_ci(est_time=180, suite="stage-b-2-gpu-h200", labels=["miles-plugin"], hardware=["hopper", "blackwell"]) +register_rocm_ci(est_time=500, suite="nightly-stage-c-2-gpu-mi350", labels=["miles-plugin"]) import math diff --git a/tests/fast-gpu/test_true_on_policy_logprob_chunking.py b/tests/fast-gpu/test_true_on_policy_logprob_chunking.py index 9f494387e2..5a7ecfce68 100644 --- a/tests/fast-gpu/test_true_on_policy_logprob_chunking.py +++ b/tests/fast-gpu/test_true_on_policy_logprob_chunking.py @@ -1,6 +1,7 @@ -from tests.ci.ci_register import register_cuda_ci +from tests.ci.ci_register import register_cuda_ci, register_rocm_ci register_cuda_ci(est_time=180, suite="stage-b-2-gpu-h200", labels=["megatron"], hardware=["hopper"]) +register_rocm_ci(est_time=60, suite="nightly-stage-c-2-gpu-mi350", labels=["megatron"]) import gc import os