[AMD] ci: expand ROCm nightly coverage and refresh runtime estimates (#2541)

Co-authored-by: Zhiyao Jiang <jessicajiang324@gmail.com>
This commit is contained in:
Xinyu Jiang
2026-09-29 17:28:58 -07:00
committed by GitHub
co-authored by Zhiyao Jiang
parent 3439ec7513
commit 5ebc06f8d2
31 changed files with 58 additions and 35 deletions
+1 -1
View File
@@ -5,7 +5,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(est_time=900, suite="stage-c-2-gpu-h200", labels=["ckpt"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=1200, suite="nightly-stage-c-2-gpu-mi350", labels=["ckpt"])
register_rocm_ci(est_time=1500, suite="nightly-stage-c-2-gpu-mi350", labels=["ckpt"])
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
@@ -1,6 +1,6 @@
import os
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
@@ -10,6 +10,7 @@ register_cuda_ci(
labels=["fsdp"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(est_time=400, suite="nightly-stage-c-8-gpu-mi350", labels=["fsdp"])
NUM_GPUS = 8
DP_REPLICATE_SIZE = 2
@@ -3,7 +3,7 @@
# (the CUDA CI runner's execution model). Scenario logic lives in
# tests/e2e/ft/conftest_ft/scenario_trainer_no_failure.py.
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from tests.e2e.ft.conftest_ft.scenario_trainer_no_failure import run_ci
register_cuda_ci(
@@ -12,6 +12,7 @@ register_cuda_ci(
labels=["ft-short"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(est_time=900, suite="nightly-stage-c-8-gpu-mi350", labels=["ft-short"])
_MODE: str = "kill_train__dp2_cp2_pp2__fake_rollout__moe_5layer"
@@ -3,7 +3,7 @@
# (the CUDA CI runner's execution model). Scenario logic lives in
# tests/e2e/ft/conftest_ft/scenario_trainer_no_failure.py.
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from tests.e2e.ft.conftest_ft.scenario_trainer_no_failure import run_ci
register_cuda_ci(
@@ -12,6 +12,7 @@ register_cuda_ci(
labels=["ft-short"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(est_time=1000, suite="nightly-stage-c-8-gpu-mi350", labels=["ft-short"])
_MODE: str = "kill_train__dp2_cp2_tp2_ep2__fake_rollout__moe_5layer"
@@ -5,7 +5,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(est_time=5400, suite="stage-c-2-gpu-h200", labels=["long"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=5000, suite="nightly-stage-c-2-gpu-mi350", labels=["long"])
register_rocm_ci(est_time=5400, suite="nightly-stage-c-2-gpu-mi350", labels=["long"])
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
+1 -1
View File
@@ -18,7 +18,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["lora"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=300, suite="nightly-stage-c-4-gpu-mi350", labels=["lora"])
register_rocm_ci(est_time=500, suite="nightly-stage-c-4-gpu-mi350", labels=["lora"])
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
@@ -19,7 +19,7 @@ from megatron.core.tensor_parallel.layers import ColumnParallelLinear
from megatron.core.tensor_parallel.random import model_parallel_cuda_manual_seed
from megatron.core.transformer.module import MegatronModule
from megatron.core.transformer.transformer_config import TransformerConfig
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from torch.utils._pytree import tree_flatten, tree_map
from miles.backends.megatron_utils.lora.checkpoint import load_slot, save_slot
@@ -27,6 +27,7 @@ from miles.backends.megatron_utils.lora.optimizer import SlotOptimizer, adapter_
from miles.utils.distributed_utils import init_gloo_group
register_cuda_ci(est_time=120, suite="stage-b-2-gpu-h200", labels=["lora"], hardware=["hopper"])
register_rocm_ci(est_time=60, suite="nightly-stage-c-2-gpu-mi350", labels=["lora"])
ADAM = dict(learning_rate=3e-4, beta1=0.9, beta2=0.95, eps=1e-8, weight_decay=0.01, grad_clip_norm=1.0)
@@ -1,6 +1,6 @@
import os
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
@@ -16,6 +16,7 @@ register_cuda_ci(
labels=["megatron", "model-scripts", "lora"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(est_time=1600, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "model-scripts", "lora"])
MODEL_NAME = "gpt-oss-20b-bf16"
MODEL_TYPE = "gpt-oss-20b"
@@ -16,7 +16,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["megatron"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=420, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron"])
register_rocm_ci(est_time=600, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron"])
MODEL_NAME = "MiMo-7B-RL"
MODEL_TYPE = "mimo-7B-rl"
@@ -18,7 +18,7 @@ register_ci_gate(metric_key="train/train_rollout_kl")
register_ci_gate(metric_key="rollout/raw_reward")
register_rocm_ci(
est_time=900, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "weight-update", "short", "mooncake"]
est_time=2000, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "weight-update", "short", "mooncake"]
)
CASE = CaseConfig(
@@ -7,13 +7,14 @@ training partition and per-rollout loss weighting are unchanged."""
import os
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from tests.ci.metric_history import register_ci_gate
from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, prepare
register_cuda_ci(
est_time=1100, suite="stage-c-4-gpu-h200", labels=["megatron", "short"], hardware=["hopper", "blackwell"]
)
register_rocm_ci(est_time=800, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "short"])
register_ci_gate(metric_key="train/grad_norm")
register_ci_gate(metric_key="train/ppo_kl")
@@ -7,7 +7,7 @@ from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, p
register_cuda_ci(
est_time=1200, suite="stage-c-4-gpu-h200", labels=["megatron", "replay"], hardware=["hopper", "blackwell"]
)
register_rocm_ci(est_time=1500, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "replay"])
register_rocm_ci(est_time=1000, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "replay"])
register_ci_gate(metric_key="train/grad_norm")
register_ci_gate(metric_key="train/ppo_kl")
@@ -7,7 +7,7 @@ weights must come from the host backup rather than the unmapped param buffers.
--check-weight-update-equal makes the engines verify what they received.
"""
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from tests.ci.metric_history import register_ci_gate
from miles.utils.external_utils import command_utils
@@ -23,6 +23,7 @@ register_cuda_ci(
labels=["megatron", "weight-update"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["megatron", "weight-update"])
register_ci_gate(metric_key="train/grad_norm")
register_ci_gate(metric_key="rollout/raw_reward")
@@ -1,7 +1,7 @@
import os
from scripts.run_qwen3_5_35b_a3b_lora import ScriptArgs, _prepare_download, _train
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
@@ -19,6 +19,7 @@ register_cuda_ci(
labels=["megatron", "model-scripts", "lora"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(est_time=900, suite="nightly-stage-c-8-gpu-mi350", labels=["megatron", "model-scripts", "lora"])
# (name, experts_shared_outer_loras, virtual_experts_serving)
_CONFIGS = [
@@ -28,6 +29,13 @@ _CONFIGS = [
def _args(shared_outer: bool, virtual_experts: bool) -> ScriptArgs:
extra_args = "--ci-test --ci-disable-logprobs-checker "
if not virtual_experts:
extra_args += "--no-sglang-lora-use-virtual-experts "
# ROCm shared-expert fusion requires per-expert LoRA factors.
if os.getenv("MILES_HARDWARE_PLATFORM") == "rocm" and shared_outer:
extra_args += "--sglang-disable-shared-experts-fusion "
return ScriptArgs.from_env(
model_name="Qwen3.5-35B-A3B",
num_nodes=1,
@@ -35,10 +43,7 @@ def _args(shared_outer: bool, virtual_experts: bool) -> ScriptArgs:
num_rollout=1,
experts_shared_outer_loras=shared_outer,
enable_wandb=False,
extra_args=(
"--ci-test --ci-disable-logprobs-checker "
+ ("" if virtual_experts else "--no-sglang-lora-use-virtual-experts ")
),
extra_args=extra_args,
)
@@ -14,7 +14,7 @@ import torch.distributed as dist
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
register_cuda_ci(est_time=30, suite="stage-c-4-gpu-h200", labels=["precision"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=60, suite="nightly-stage-c-4-gpu-mi350", labels=["precision"])
register_rocm_ci(est_time=30, suite="nightly-stage-c-4-gpu-mi350", labels=["precision"])
from miles_plugins.models.cp_utils import packed_shard_to_zigzag, zigzag_to_packed_shard
@@ -17,7 +17,7 @@ import torch.distributed as dist
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
register_cuda_ci(est_time=300, suite="stage-c-4-gpu-h200", labels=["precision"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=120, suite="nightly-stage-c-4-gpu-mi350", labels=["precision"])
register_rocm_ci(est_time=200, suite="nightly-stage-c-4-gpu-mi350", labels=["precision"])
def setup_dist():
@@ -1,8 +1,9 @@
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from tests.ci.metric_history import register_ci_gate
from tests.e2e.sglang.test_session_server_multi_role._common import ModelConfig, run_both_versions
register_cuda_ci(est_time=800, suite="stage-c-2-gpu-h200", labels=["sglang"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=700, suite="nightly-stage-c-2-gpu-mi350", labels=["sglang"])
register_ci_gate(metric_key="rollout/tito_session_mismatch_rate/v1/assistant_text")
register_ci_gate(metric_key="rollout/tito_session_mismatch_rate/v2/assistant_text")
@@ -5,7 +5,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=600, suite="nightly-stage-c-4-gpu-mi350", labels=["short"])
register_rocm_ci(est_time=400, suite="nightly-stage-c-4-gpu-mi350", labels=["short"])
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
@@ -2,7 +2,7 @@ import dataclasses
import os
from examples.multi_policy.run_solver_verifier_gsm8k import ScriptArgs, prepare
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from tests.e2e.conftest_multi_policy import execute
from miles.utils.external_utils import command_utils
@@ -13,6 +13,7 @@ register_cuda_ci(
labels=["short", "multi-policy", "fully-async"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(est_time=1700, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "multi-policy", "fully-async"])
NUM_ROLLOUT = int(os.environ.get("MILES_TEST_NUM_ROLLOUT", "5"))
SAVE_INTERVAL = 2
@@ -7,7 +7,7 @@ from miles.utils.external_utils import command_utils
register_cuda_ci(
est_time=400, suite="stage-c-4-gpu-h200", labels=["short", "mooncake"], hardware=["hopper", "blackwell"]
)
register_rocm_ci(est_time=240, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "mooncake"])
register_rocm_ci(est_time=300, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "mooncake"])
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
@@ -10,7 +10,7 @@ from miles.utils.workers.types import WorkerCommBackend
register_cuda_ci(
est_time=400, suite="stage-c-2-gpu-h200", labels=["short", "mooncake"], hardware=["hopper", "blackwell"]
)
register_rocm_ci(est_time=360, suite="nightly-stage-c-2-gpu-mi350", labels=["short", "mooncake"])
register_rocm_ci(est_time=300, suite="nightly-stage-c-2-gpu-mi350", labels=["short", "mooncake"])
MODEL_DIR = get_test_model_dir()
DATA_DIR = get_test_data_dir()
@@ -1,6 +1,6 @@
import os
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
import miles.utils.external_utils.command_utils.legacy as U
@@ -11,6 +11,7 @@ register_cuda_ci(
labels=["short", "mooncake"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "mooncake"])
FEW_GPU = U.get_bool_env_var("MILES_TEST_FEW_GPU", "0")
@@ -2,7 +2,7 @@ import importlib.util
from pathlib import Path
from types import ModuleType
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
register_cuda_ci(
est_time=450,
@@ -10,6 +10,7 @@ register_cuda_ci(
labels=["short", "mooncake", "rpc-comm"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "mooncake", "rpc-comm"])
def _load_base_test() -> ModuleType:
@@ -2,7 +2,7 @@ import importlib.util
from pathlib import Path
from types import ModuleType
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
register_cuda_ci(
est_time=450,
@@ -10,6 +10,7 @@ register_cuda_ci(
labels=["short", "rpc-comm"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "rpc-comm"])
def _load_base_test() -> ModuleType:
@@ -12,7 +12,7 @@ tests/e2e/megatron/test_qwen3_4b_fully_async_eval.py.
import os
import pandas as pd
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from transformers import AutoTokenizer
from miles.utils.external_utils import command_utils
@@ -23,6 +23,7 @@ register_cuda_ci(
labels=["short", "eval", "megatron"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(est_time=300, suite="nightly-stage-c-2-gpu-mi350", labels=["short", "eval", "megatron"])
MODEL_NAME = "Qwen3-0.6B"
MODEL_TYPE = "qwen3-0.6B"
+1 -1
View File
@@ -40,7 +40,7 @@ NUM_LAYERS: int = 5
_RUN_DIR: Path = Path(tempfile.mkdtemp(prefix="test_run_megatron_"))
register_cuda_ci(est_time=200, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=2000, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
@dataclasses.dataclass(frozen=True)
+2 -1
View File
@@ -1,8 +1,9 @@
"""FSDP2 gradient parity across r1s4, r2s2, and r4s1."""
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
register_cuda_ci(est_time=90, suite="stage-c-4-gpu-h200", labels=["fsdp"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=90, suite="nightly-stage-c-4-gpu-mi350", labels=["fsdp"])
import os
import subprocess
@@ -4,7 +4,7 @@ import os
from types import SimpleNamespace
import torch
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles_plugins.optimizers.nvme_stream import NVMeOptimizerStateStore, _Bucket, _Entry, _resize, _Stager
@@ -14,6 +14,7 @@ register_cuda_ci(
labels=["miles-plugin"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(est_time=30, suite="nightly-stage-c-2-gpu-mi350", labels=["miles-plugin"])
def test_bucketwise_main_initialization_preserves_bytes_and_releases_cuda_storage(tmp_path):
@@ -4,9 +4,10 @@ Only sampled rows have autograd references, keeping the test smaller than a full
model. Inputs vary across rows and channels so a wrong in-bounds address also fails.
"""
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
register_cuda_ci(est_time=180, suite="stage-b-2-gpu-h200", labels=["miles-plugin"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=40, suite="nightly-stage-c-2-gpu-mi350", labels=["miles-plugin"])
import math
+2 -1
View File
@@ -1,8 +1,9 @@
"""Qwen3.8-Flash-Next triton kernels must match their torch references, forward and backward."""
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
register_cuda_ci(est_time=180, suite="stage-b-2-gpu-h200", labels=["miles-plugin"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=500, suite="nightly-stage-c-2-gpu-mi350", labels=["miles-plugin"])
import math
@@ -1,6 +1,7 @@
from tests.ci.ci_register import register_cuda_ci
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
register_cuda_ci(est_time=180, suite="stage-b-2-gpu-h200", labels=["megatron"], hardware=["hopper"])
register_rocm_ci(est_time=60, suite="nightly-stage-c-2-gpu-mi350", labels=["megatron"])
import gc
import os