test(e2e): move 8x H100 e2e tests to H200 stages and widen parallelism coverage (#3660)

Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Jiajun Li
2026-09-27 20:41:43 -07:00
committed by GitHub
co-authored by Claude Opus 5.5
parent 833d5bf78b
commit f84b1496e1
27 changed files with 149 additions and 134 deletions
+2 -6
View File
@@ -233,6 +233,8 @@ jobs:
${{ github.event_name == 'workflow_dispatch' && '--match-all-labels' || '' }}
secrets: inherit
# No partition matrix: the tests left on 8x H100 fit one runner's run, so one
# job keeps the second H100 host free for other runs.
stage-c-8-gpu-h100:
needs: [resolve-ci-policy, resolve-ci-image, stage-a-cpu]
if: |
@@ -242,11 +244,6 @@ jobs:
!contains(fromJSON(needs.resolve-ci-policy.outputs.skipped_stages || '[]'), 'stage-c-8-gpu-h100') &&
(needs.stage-a-cpu.result == 'success' ||
needs.resolve-ci-policy.outputs.bypass_fastfail == 'true')
strategy:
fail-fast: false
max-parallel: ${{ needs.resolve-ci-policy.outputs.cadence == 'weekly' && 1 || 2 }}
matrix:
partition_id: [0, 1]
uses: ./.github/workflows/_run-ci.yml
with:
runs_on: '["h100", "8gpu"]'
@@ -254,7 +251,6 @@ jobs:
ref: ${{ inputs.ref || '' }}
execute_command: >-
python tests/ci/run_suite.py --hw cuda --suite stage-c-8-gpu-h100
--auto-partition-id ${{ matrix.partition_id }} --auto-partition-size 2
--cadence ${{ needs.resolve-ci-policy.outputs.cadence }}
--labels ${{ needs.resolve-ci-policy.outputs.raw_labels }}
${{ github.event_name == 'workflow_dispatch' && '--match-all-labels' || '' }}
+2 -2
View File
@@ -162,8 +162,8 @@ visible before a rollout host reads them.
The maintained end-to-end coverage is
[`tests/e2e/megatron/test_qwen3_4B_disk_delta.py`](https://github.com/radixark/miles/blob/main/tests/e2e/megatron/test_qwen3_4B_disk_delta.py).
It exercises a Qwen3-4B Megatron trainer and two SGLang rollout engines on a
single 8-GPU node. The same storage contract supports separate hosts, but the
It exercises a Qwen3-4B Megatron trainer (TP2 on two GPUs) and two TP1 SGLang
rollout engines on the other two GPUs of a single 4-GPU node. The same storage contract supports separate hosts, but the
registered test does not reproduce a cross-cluster deployment.
Current `main` rejects disk-delta with `--colocate`, LoRA, or PD
+1 -1
View File
@@ -23,7 +23,7 @@ Stage names follow `stage-<tier>-<gpus>-<hw>` (or `stage-<tier>-<hw>` for CPU, e
| `stage-b-2-gpu-h200` | 2× H200 | `["h200","2gpu"]` | 1 | both resolvers, `stage-a-cpu` |
| `stage-c-2-gpu-h200` | 2× H200 | `["h200","2gpu"]` | 2 | both resolvers, `stage-a-cpu` |
| `stage-c-4-gpu-h200` | 4× H200 | `["h200","4gpu"]` | 3 | both resolvers, `stage-a-cpu` |
| `stage-c-8-gpu-h100` | 8× H100 | `["h100","8gpu"]` | 2 | both resolvers, `stage-a-cpu` |
| `stage-c-8-gpu-h100` | 8× H100 | `["h100","8gpu"]` | 1 | both resolvers, `stage-a-cpu` |
| `stage-c-8-gpu-h200` | 8× H200 | `["h200","8gpu"]` | 2 | both resolvers, `stage-a-cpu` |
| `stage-c-8-gpu-b200` | 8× B200 | `["b200","8gpu"]` | 1 | both resolvers, `stage-a-cpu` |
| `stage-c-4-gpu-mi350` | 4× MI350 | `["self-hosted","amd","mi350","4gpu"]` | 2 | both resolvers |
-1
View File
@@ -462,7 +462,6 @@ class TestWorkflowScopeSeam:
def test_weekly_serializes_each_gpu_matrix(self):
workflow = self._workflow()
normal_parallelism = {
"stage-c-8-gpu-h100": 2,
"stage-c-8-gpu-h200": 2,
"stage-c-4-gpu-h200": 3,
"stage-c-2-gpu-h200": 2,
@@ -4,18 +4,20 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(est_time=1400, suite="stage-c-8-gpu-h100", labels=["ckpt"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=1200, suite="nightly-stage-c-8-gpu-mi350", labels=["ckpt"])
register_cuda_ci(est_time=900, suite="stage-c-2-gpu-h200", labels=["ckpt"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=1200, suite="nightly-stage-c-2-gpu-mi350", labels=["ckpt"])
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
MODEL_NAME = "Qwen3-4B"
MODEL_TYPE = "qwen3-4B"
NUM_GPUS = 8
MODEL_NAME = "Qwen3-0.6B"
MODEL_TYPE = "qwen3-0.6B"
NUM_GPUS = 2
# Container-local: /root/models is a host directory shared by every runner on the host.
SAVE_DIR = f"/root/checkpoints/{MODEL_NAME}_miles"
def _get_latest_checkpointed_iteration() -> int:
latest_path = f"/root/models/{MODEL_NAME}_miles/latest_checkpointed_iteration.txt"
latest_path = f"{SAVE_DIR}/latest_checkpointed_iteration.txt"
with open(latest_path, encoding="utf-8") as f:
latest_text = f.read().strip()
if not latest_text.isdigit():
@@ -27,7 +29,7 @@ def prepare():
U = command_utils.default_config().create_backend()
U.exec_command_cpu("mkdir -p /root/models /root/datasets")
U.exec_command_cpu(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.exec_command_cpu(f"rm -rf /root/models/{MODEL_NAME}_miles")
U.exec_command_cpu(f"rm -rf {SAVE_DIR}")
U.hf_download_dataset("zhuzilin/dapo-math-17k")
U.hf_download_dataset("zhuzilin/aime-2024")
@@ -40,15 +42,15 @@ def execute(mode: str = "", ckpt_step: int | None = None):
U = command_utils.default_config().create_backend()
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}_torch_dist "
if mode == "save":
ckpt_args += f"--save /root/models/{MODEL_NAME}_miles "
ckpt_args += f"--save {SAVE_DIR} "
ckpt_args += "--save-interval 2 "
elif mode == "async_save":
ckpt_args += f"--save /root/models/{MODEL_NAME}_miles "
ckpt_args += f"--save {SAVE_DIR} "
ckpt_args += "--save-interval 2 "
ckpt_args += "--async-save "
ckpt_args += "--use-persistent-ckpt-worker "
elif mode == "load":
ckpt_args += f"--load /root/models/{MODEL_NAME}_miles "
ckpt_args += f"--load {SAVE_DIR} "
ckpt_args += f"--ckpt-step {ckpt_step} "
ckpt_args += "--low-memory-resume "
@@ -71,8 +73,8 @@ def execute(mode: str = "", ckpt_step: int | None = None):
perf_args = (
"--tensor-model-parallel-size 2 "
"--sequence-parallel "
"--pipeline-model-parallel-size 2 "
"--context-parallel-size 2 "
"--pipeline-model-parallel-size 1 "
"--context-parallel-size 1 "
"--recompute-granularity full "
"--recompute-method uniform "
"--recompute-num-layers 1 "
@@ -1,12 +1,12 @@
"""AMD 4-GPU variant of test_r3_mtp.py.
Standalone rather than an IS_HIP branch in the original: the MI300X fleet is
split into two 4-GPU runners, so the 8-GPU CUDA case cannot run there as
written, and keeping the variant separate means neither side's parallelism
constrains the other.
split into two 4-GPU runners, and keeping the variant separate means neither
side's parallelism constrains the other. The CUDA test_r3_mtp now runs the same
4-GPU shape on 4x H200.
Difference from the CUDA case: cp_size 2 -> 1 and ep_size 4 -> 2, halving the
world size from 8 to 4. TP=2 and PP=2 are unchanged, so the MTP layer placement
Difference from the original 8-GPU layout (now test_r3_mtp_deepep's): cp_size
2 -> 1 and ep_size 4 -> 2, halving the world size from 8 to 4. TP=2 and PP=2 are unchanged, so the MTP layer placement
under test is unaffected. Both axes have to shrink because Megatron sizes the
dense and expert grids independently -- world_size % (tp * cp * pp) == 0 and
world_size % (etp * ep * pp) == 0 -- so dropping CP alone leaves the expert grid
@@ -4,7 +4,7 @@ from tests.ci.ci_register import register_cuda_ci
from tests.ci.metric_history import register_ci_gate
from tests.e2e.megatron.test_glm47_flash._common import CaseConfig, execute, prepare
register_cuda_ci(est_time=1100, suite="stage-c-8-gpu-h100", labels=["megatron"], hardware=["hopper", "blackwell"])
register_cuda_ci(est_time=1200, suite="stage-c-4-gpu-h200", labels=["megatron"], hardware=["hopper", "blackwell"])
register_ci_gate(metric_key="train/grad_norm")
register_ci_gate(metric_key="train/ppo_kl")
@@ -14,11 +14,13 @@ register_ci_gate(metric_key="rollout/raw_reward")
CASE = CaseConfig(
use_deepep=False,
num_gpus_per_node=8,
cp_size=2,
# tp2/pp2/cp1/ep2 on 4 GPUs (same shape as test_amd_r3_mtp); the 8-GPU tp2/pp2/cp2/ep4
# shape is test_r3_mtp_deepep's on 8x H200, which is disabled.
num_gpus_per_node=4,
cp_size=1,
pp_size=2,
tp_size=2,
ep_size=4,
ep_size=2,
# GLM-4.7-Flash has 20 attention heads; non-EP SGLang TP must divide it.
rollout_num_gpus_per_engine=4,
)
@@ -6,7 +6,7 @@ from tests.e2e.megatron.test_glm47_flash._common import CaseConfig, execute, pre
# FIXME: sglang deepep code path bug.
register_cuda_ci(
est_time=900,
suite="stage-c-8-gpu-h100",
suite="stage-c-8-gpu-h200",
labels=["megatron"],
hardware=["hopper", "blackwell"],
disabled="Disabled due to sglang deepep code path bug.",
@@ -17,9 +17,11 @@ class CaseConfig:
tp_size: int
ep_size: int
rollout_num_gpus_per_engine: int
etp_size: int = 1
sglang_ep_size: int = None
sglang_dp_size: int = None
sglang_enable_dp_attention: bool = False
sglang_max_running_requests: int = 512
use_deepep: bool = False
use_fp8_rollout: bool = False
use_int4_rollout: bool = False
@@ -135,7 +137,7 @@ def build_train_args(case: CaseConfig, *, wandb_file: str) -> str:
f"--pipeline-model-parallel-size {case.pp_size} "
f"--context-parallel-size {case.cp_size} "
f"--expert-model-parallel-size {case.ep_size} "
"--expert-tensor-parallel-size 1 "
f"--expert-tensor-parallel-size {case.etp_size} "
"--recompute-granularity full "
"--recompute-method uniform "
"--recompute-num-layers 1 "
@@ -177,7 +179,7 @@ def build_train_args(case: CaseConfig, *, wandb_file: str) -> str:
sglang_args = (
f"--rollout-num-gpus-per-engine {case.rollout_num_gpus_per_engine} "
"--sglang-mem-fraction-static 0.7 "
"--sglang-max-running-requests 512 "
f"--sglang-max-running-requests {case.sglang_max_running_requests} "
"--sglang-enable-metrics "
)
@@ -4,7 +4,7 @@ from tests.ci.ci_register import register_cuda_ci
from tests.ci.metric_history import register_ci_gate
from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, prepare
register_cuda_ci(est_time=900, suite="stage-c-8-gpu-h100", labels=["megatron"], hardware=["hopper", "blackwell"])
register_cuda_ci(est_time=1500, suite="stage-c-4-gpu-h200", labels=["megatron"], hardware=["hopper", "blackwell"])
register_ci_gate(metric_key="train/grad_norm")
register_ci_gate(metric_key="train/ppo_kl")
@@ -18,13 +18,19 @@ CASE = CaseConfig(
use_int4_rollout=False,
use_bridge=False,
use_r3=False,
num_gpus_per_node=8,
cp_size=2,
pp_size=2,
# tp2/etp2 x dense dp2/ep2 on 4 GPUs: the Megatron DeepEP group is etp2 x ep2 = 4 ranks.
# SGLang runs DP attention (dp4, attention TP1) with DeepEP EP4 = engine TP; 512 running
# requests / dp4 = 128 tokens per rank, the DeepEP low-latency cap.
num_gpus_per_node=4,
cp_size=1,
pp_size=1,
tp_size=2,
ep_size=4,
rollout_num_gpus_per_engine=8,
sglang_ep_size=8,
ep_size=2,
etp_size=2,
rollout_num_gpus_per_engine=4,
sglang_ep_size=4,
sglang_dp_size=4,
sglang_enable_dp_attention=True,
)
@@ -5,7 +5,7 @@ from tests.ci.metric_history import register_ci_gate
from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, prepare
# Limited by host memory
register_cuda_ci(est_time=800, suite="stage-c-8-gpu-h100", labels=["megatron"], hardware=["hopper", "blackwell"])
register_cuda_ci(est_time=1300, suite="stage-c-4-gpu-h200", labels=["megatron"], hardware=["hopper", "blackwell"])
register_ci_gate(metric_key="train/grad_norm")
register_ci_gate(metric_key="train/ppo_kl")
@@ -19,14 +19,18 @@ CASE = CaseConfig(
use_int4_rollout=False,
use_bridge=True,
use_r3=False,
num_gpus_per_node=8,
# tp2/pp2/ep2 on 4 GPUs keeps TP, PP and EP all > 1 on the bridge path. Two SGLang engines
# at TP2/EP2 DeepEP; 256 running requests keep decode at 128 tokens per rank, the DeepEP
# low-latency cap.
num_gpus_per_node=4,
cp_size=1,
pp_size=2,
tp_size=4,
ep_size=4,
rollout_num_gpus_per_engine=8,
sglang_ep_size=8,
tp_size=2,
ep_size=2,
rollout_num_gpus_per_engine=2,
sglang_ep_size=2,
max_tokens_per_gpu=2048,
sglang_max_running_requests=256,
)
@@ -5,9 +5,9 @@ from tests.ci.metric_history import register_ci_gate
from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, prepare
register_cuda_ci(
est_time=1300, suite="stage-c-8-gpu-h100", labels=["megatron", "weight-update"], hardware=["hopper", "blackwell"]
est_time=1500, suite="stage-c-4-gpu-h200", labels=["megatron", "weight-update"], hardware=["hopper", "blackwell"]
)
register_rocm_ci(est_time=900, suite="nightly-stage-c-8-gpu-mi350", labels=["megatron", "weight-update"])
register_rocm_ci(est_time=900, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "weight-update"])
register_ci_gate(metric_key="train/grad_norm")
register_ci_gate(metric_key="train/ppo_kl")
@@ -21,15 +21,20 @@ CASE = CaseConfig(
use_int4_rollout=False,
use_bridge=False,
use_r3=False,
num_gpus_per_node=6,
cp_size=2,
# 3 train + 1 rollout: PP3 broadcast from three senders with VPP2 (8 layers per virtual chunk).
# VPP rounds each step's micro-batch count up to a multiple of PP; 16k tokens packs two ~8k
# samples per micro-batch so a 32-sample step can reach that multiple.
num_gpus_per_node=3,
cp_size=1,
pp_size=3,
tp_size=1,
ep_size=2,
ep_size=1,
colocate=False,
rollout_num_gpus=2,
rollout_num_gpus_per_engine=2,
rollout_num_gpus=1,
rollout_num_gpus_per_engine=1,
update_weight_transfer_mode="broadcast",
max_tokens_per_gpu=16384,
extra_args="--num-layers-per-virtual-pipeline-stage 8 ",
)
@@ -6,12 +6,12 @@ from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, p
register_cuda_ci(
est_time=1500,
suite="stage-c-8-gpu-h100",
suite="stage-c-4-gpu-h200",
labels=["megatron", "weight-update", "fully-async"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(
est_time=800, suite="nightly-stage-c-8-gpu-mi350", labels=["megatron", "weight-update", "fully-async"]
est_time=800, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "weight-update", "fully-async"]
)
register_ci_gate(metric_key="train/grad_norm")
@@ -30,17 +30,20 @@ CASE = CaseConfig(
use_int4_rollout=False,
use_bridge=False,
use_r3=False,
num_gpus_per_node=6,
cp_size=2,
# 3 train + 1 rollout with an uneven PP3 split (17/17/14): layer offsets 0/17/34 differ
# from the even split on both non-first stages.
num_gpus_per_node=3,
cp_size=1,
pp_size=3,
tp_size=1,
ep_size=2,
ep_size=1,
colocate=False,
rollout_num_gpus=2,
rollout_num_gpus_per_engine=2,
rollout_num_gpus=1,
rollout_num_gpus_per_engine=1,
update_weight_transfer_mode="broadcast",
num_rollout=3,
fully_async=True,
extra_args="--decoder-first-pipeline-num-layers 17 --decoder-last-pipeline-num-layers 14 ",
)
@@ -5,17 +5,17 @@ from miles.utils.external_utils import command_utils
MODEL_NAME = "Qwen3-4B"
MODEL_TYPE = "qwen3-4B"
NUM_GPUS = 8
NUM_GPUS = 4
register_cuda_ci(
est_time=600,
suite="stage-c-8-gpu-h100",
est_time=400,
suite="stage-c-4-gpu-h200",
labels=["megatron", "weight-update"],
hardware=["hopper", "blackwell"],
)
register_rocm_ci(
est_time=500,
suite="nightly-stage-c-8-gpu-mi350",
suite="nightly-stage-c-4-gpu-mi350",
labels=["megatron", "weight-update"],
)
@@ -58,7 +58,7 @@ def execute():
"--tensor-model-parallel-size 2 "
"--sequence-parallel "
"--pipeline-model-parallel-size 1 "
"--context-parallel-size 2 "
"--context-parallel-size 1 "
"--recompute-granularity full "
"--recompute-method uniform "
"--recompute-num-layers 1 "
@@ -85,7 +85,7 @@ def execute():
)
sglang_args = (
"--rollout-num-gpus-per-engine 2 " f"--rollout-num-gpus {NUM_GPUS // 2} " "--sglang-mem-fraction-static 0.8 "
"--rollout-num-gpus-per-engine 1 " f"--rollout-num-gpus {NUM_GPUS // 2} " "--sglang-mem-fraction-static 0.8 "
)
ci_args = "--ci-test "
@@ -9,7 +9,7 @@ miles has no VLM/vision implementation on the training side, so Qwen3.5's `visua
weights are never synced and must be excluded from the weight-equality check; each case
passes `check_weight_update_skip_list=("visual",)`.
Topology follows scripts/run_qwen3_5_35b_a3b_mtp.py (cp2/ep8 on 8 GPUs).
Each case picks its own topology (see its CASE).
Spec (EAGLE) and spec-v2 (mamba scheduler) are on for the whole suite; R3 is per-case.
"""
@@ -24,8 +24,7 @@ MODEL_TYPE = "qwen3.5-35B-A3B"
@dataclass
class CaseConfig:
# Topology / GPU counts — explicit per case (each test file picks a shape that fits
# the Qwen3.5 GatedDeltaNet backward on 8x80GB; see each file's CASE).
# Topology / GPU counts — explicit per case (see each file's CASE).
num_gpus_per_node: int
cp_size: int
pp_size: int
@@ -16,7 +16,7 @@ from tests.ci.metric_history import register_ci_gate
from tests.e2e.megatron.test_qwen3_5_35B_A3B_mtp._common import CaseConfig, execute, prepare
register_cuda_ci(
est_time=1800, suite="stage-c-8-gpu-h100", labels=["megatron", "qwen35"], hardware=["hopper", "blackwell"]
est_time=1900, suite="stage-c-4-gpu-h200", labels=["megatron", "qwen35"], hardware=["hopper", "blackwell"]
)
register_ci_gate(metric_key="train/grad_norm")
@@ -26,18 +26,18 @@ register_ci_gate(metric_key="train/train_rollout_kl")
register_ci_gate(metric_key="rollout/raw_reward")
CASE = CaseConfig(
# tp2/cp2/pp2/ep4: TP=4 hits a Qwen3.5 attention-output-gate sharding bug, so stay at
# TP=2 and use PP=2 to halve the resident layers for OOM headroom (keeps CP=2 coverage).
num_gpus_per_node=8,
# tp2/cp2/ep4: TP=4 hits a Qwen3.5 attention-output-gate sharding bug, so stay at TP=2.
# PP=1 on 4 GPUs: PP=2 stays covered by test_mtp1_spec_v2_r3.
num_gpus_per_node=4,
cp_size=2,
pp_size=2,
pp_size=1,
tp_size=2,
ep_size=4,
# 4096 (mtp1 keeps 8192): CP=2 routes the GatedDeltaNet backward through the heavier fla
# CP kernel, whose Triton autotune OOMs at 8192 even with PP=2; halve the budget for headroom.
max_tokens_per_gpu=4096,
rollout_num_gpus_per_engine=8,
sglang_ep_size=8,
rollout_num_gpus_per_engine=4,
sglang_ep_size=4,
enable_mtp_training=False,
use_r3=False,
check_weight_update_selector="target",
@@ -12,7 +12,7 @@ from tests.ci.metric_history import register_ci_gate
from tests.e2e.megatron.test_qwen3_5_35B_A3B_mtp._common import CaseConfig, execute, prepare
register_cuda_ci(
est_time=1600, suite="stage-c-8-gpu-h100", labels=["megatron", "qwen35"], hardware=["hopper", "blackwell"]
est_time=1500, suite="stage-c-8-gpu-h200", labels=["megatron", "qwen35"], hardware=["hopper", "blackwell"]
)
register_ci_gate(metric_key="train/grad_norm")
@@ -22,9 +22,9 @@ register_ci_gate(metric_key="train/train_rollout_kl")
register_ci_gate(metric_key="rollout/raw_reward")
CASE = CaseConfig(
# tp2/pp2/cp1/ep4: TP=4 hits a Qwen3.5 attention-output-gate sharding bug, so stay at
# TP=2; CP=1 avoids the memory-heavy GatedDeltaNet CP backward kernel and PP=2 halves
# the resident layers, together fitting the MTP-training run on 8x80GB.
# tp2/pp2/cp1/ep4 on 8x H200: TP=4 hits a Qwen3.5 attention-output-gate sharding bug, so
# stay at TP=2; PP=2 with dense DP2 keeps Megatron DeepEP x DP x PP (a 4-rank DeepEP group
# per PP stage), which no 4-GPU case can hold.
num_gpus_per_node=8,
cp_size=1,
pp_size=2,
@@ -14,8 +14,8 @@ from miles.utils.external_utils import command_utils
register_cuda_ci(
est_time=1300,
suite="stage-c-8-gpu-h100",
est_time=1200,
suite="stage-c-8-gpu-h200",
labels=["megatron", "model-scripts", "lora"],
hardware=["hopper", "blackwell"],
)
@@ -4,16 +4,16 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(est_time=400, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=600, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=600, suite="nightly-stage-c-4-gpu-mi350", labels=["short"])
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
NUM_GPUS = 8
NUM_GPUS = 4
# Inline sglang config: same model, 2 engine groups with different sizes.
# Group 1: 4 GPUs, 1 GPU/engine (tp=1) -> 4 engines
# Group 2: 4 GPUs, 1 GPU/engine (tp=1) -> 4 engines
# Group 1: 2 GPUs, 1 GPU/engine (tp=1) -> 2 engines
# Group 2: 2 GPUs, 1 GPU/engine (tp=1) -> 2 engines
# Tests that RolloutServer correctly manages multiple engine groups
# behind a single router, with separate port cursors per group.
SGLANG_CONFIG_YAML = """\
@@ -21,10 +21,10 @@ sglang:
- name: default
server_groups:
- worker_type: regular
num_gpus: 4
num_gpus: 2
num_gpus_per_engine: 1
- worker_type: regular
num_gpus: 4
num_gpus: 2
num_gpus_per_engine: 1
"""
@@ -1,9 +1,9 @@
"""E2E test: mixed offload with updatable + frozen models.
Deploys two models via --sglang-config in colocate mode:
- "actor": update_weights=true, 4 GPUs -> overlaps with megatron, gets offloaded
- "actor": update_weights=true, 2 GPUs -> overlaps with megatron, gets offloaded
and weights updated from training.
- "ref": update_weights=false, 4 GPUs -> overlaps with megatron, gets offloaded
- "ref": update_weights=false, 2 GPUs -> overlaps with megatron, gets offloaded
and weights restored from disk (update_weights_from_disk).
Key coverage:
@@ -19,27 +19,27 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(est_time=400, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=300, suite="nightly-stage-c-4-gpu-mi350", labels=["short"])
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
NUM_GPUS = 8
NUM_GPUS = 4
# Two models on 8 GPUs (colocate): actor gets weight updates, ref is frozen.
# Two models on 4 GPUs (colocate): actor gets weight updates, ref is frozen.
SGLANG_CONFIG_YAML = """\
sglang:
- name: actor
update_weights: true
server_groups:
- worker_type: regular
num_gpus: 4
num_gpus: 2
num_gpus_per_engine: 1
- name: ref
update_weights: false
server_groups:
- worker_type: regular
num_gpus: 4
num_gpus: 2
num_gpus_per_engine: 1
"""
@@ -16,27 +16,27 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(est_time=500, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=600, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=600, suite="nightly-stage-c-4-gpu-mi350", labels=["short"])
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
NUM_GPUS = 8
NUM_GPUS = 4
# Two models on 8 GPUs (colocate): actor gets weight updates, ref is frozen.
# Two models on 4 GPUs (colocate): actor gets weight updates, ref is frozen.
SGLANG_CONFIG_YAML = """\
sglang:
- name: actor
update_weights: true
server_groups:
- worker_type: regular
num_gpus: 4
num_gpus: 2
num_gpus_per_engine: 1
- name: ref
update_weights: false
server_groups:
- worker_type: regular
num_gpus: 4
num_gpus: 2
num_gpus_per_engine: 1
"""
+1 -1
View File
@@ -31,7 +31,7 @@ from tests.e2e.conftest_dumper import (
from miles.utils.external_utils import command_utils
register_cuda_ci(est_time=1100, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
register_cuda_ci(est_time=1200, suite="stage-c-8-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=1800, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
@@ -9,12 +9,12 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(est_time=400, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=400, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
register_cuda_ci(est_time=400, suite="stage-c-2-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=400, suite="nightly-stage-c-2-gpu-mi350", labels=["short"])
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
NUM_GPUS = 8
NUM_GPUS = 2
def prepare():
@@ -1,19 +1,18 @@
import os
from tempfile import TemporaryDirectory
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(
est_time=400, suite="stage-c-8-gpu-h100", labels=["short", "eval", "fully-async"], hardware=["hopper", "blackwell"]
est_time=400, suite="stage-c-4-gpu-h200", labels=["short", "eval", "fully-async"], hardware=["hopper", "blackwell"]
)
register_rocm_ci(est_time=400, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "eval", "fully-async"])
FEW_GPU = command_utils.get_bool_env_var("MILES_TEST_FEW_GPU", "0")
register_rocm_ci(est_time=400, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "eval", "fully-async"])
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
NUM_GPUS = 4 if FEW_GPU else 8
NUM_GPUS = 4
def prepare():
@@ -23,7 +22,7 @@ def prepare():
U.hf_download_dataset("zhuzilin/gsm8k")
def execute():
def execute(eval_hf_dir: str):
U = command_utils.default_config().create_backend()
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}/ "
@@ -56,7 +55,7 @@ def execute():
"--eval-top-k 1 "
"--eval-num-gpus 1 "
"--eval-num-gpus-per-engine 1 "
"--eval-hf-dir /dev/shm/miles_e2e_eval_hf "
f"--eval-hf-dir {eval_hf_dir} "
"--eval-keep-snapshots 2 "
)
@@ -104,8 +103,8 @@ def execute():
"--attention-softmax-in-fp32 "
"--attention-backend flash "
"--actor-num-nodes 1 "
f"--actor-num-gpus-per-node {1 if FEW_GPU else 2} "
f"--rollout-num-gpus {2 if FEW_GPU else 5} "
"--actor-num-gpus-per-node 2 "
"--rollout-num-gpus 1 "
# HF-format --ref-load requires the bridge loader; eval snapshots are
# exported through the bridge path as well (marker-gated).
"--megatron-to-hf-mode bridge "
@@ -136,4 +135,6 @@ if __name__ == "__main__":
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
# CI containers share host IPC, so each run must own its snapshot directory.
with TemporaryDirectory(prefix="miles_e2e_eval_hf_", dir="/dev/shm") as eval_hf_dir:
execute(eval_hf_dir)
@@ -5,15 +5,13 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(
est_time=400, suite="stage-c-8-gpu-h100", labels=["short", "mooncake"], hardware=["hopper", "blackwell"]
est_time=400, suite="stage-c-4-gpu-h200", labels=["short", "mooncake"], hardware=["hopper", "blackwell"]
)
register_rocm_ci(est_time=240, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "mooncake"])
FEW_GPU = command_utils.get_bool_env_var("MILES_TEST_FEW_GPU", "0")
register_rocm_ci(est_time=240, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "mooncake"])
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
NUM_GPUS = 4 if FEW_GPU else 8
NUM_GPUS = 4
def prepare():
@@ -102,8 +100,8 @@ def execute():
"--attention-softmax-in-fp32 "
"--attention-backend flash "
"--actor-num-nodes 1 "
f"--actor-num-gpus-per-node {1 if FEW_GPU else 2} "
f"--rollout-num-gpus {3 if FEW_GPU else 6} "
"--actor-num-gpus-per-node 2 "
"--rollout-num-gpus 2 "
"--megatron-to-hf-mode bridge "
)
@@ -8,17 +8,15 @@ from miles.utils.object_store import ObjectStoreBackend
from miles.utils.workers.types import WorkerCommBackend
register_cuda_ci(
est_time=400, suite="stage-c-8-gpu-h100", labels=["short", "mooncake"], hardware=["hopper", "blackwell"]
est_time=400, suite="stage-c-2-gpu-h200", labels=["short", "mooncake"], hardware=["hopper", "blackwell"]
)
register_rocm_ci(est_time=360, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "mooncake"])
FEW_GPU = command_utils.get_bool_env_var("MILES_TEST_FEW_GPU", "0")
register_rocm_ci(est_time=360, suite="nightly-stage-c-2-gpu-mi350", labels=["short", "mooncake"])
MODEL_DIR = get_test_model_dir()
DATA_DIR = get_test_data_dir()
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
NUM_GPUS = 4 if FEW_GPU else 8
NUM_GPUS = 2
def entrypoint(
@@ -7,15 +7,15 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
from miles.utils.external_utils import command_utils
register_cuda_ci(est_time=400, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
register_cuda_ci(est_time=300, suite="stage-c-4-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
register_rocm_ci(est_time=300, suite="nightly-stage-c-4-gpu-mi350", labels=["short"])
TIGHT_DEVICE_MEMORY = command_utils.get_bool_env_var("MILES_TEST_TIGHT_DEVICE_MEMORY", "1")
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
NUM_GPUS = 8
NUM_TRAIN_GPUS = 4
NUM_GPUS = 4
NUM_TRAIN_GPUS = 2
TEACHER_HOST = "127.0.0.1"
TEACHER_PORT = 13141