mirror of
https://github.com/radixark/miles.git
synced 2026-10-01 23:06:14 +08:00
test(e2e): move 8x H100 e2e tests to H200 stages and widen parallelism coverage (#3660)
Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
co-authored by
Claude Opus 5.5
parent
833d5bf78b
commit
f84b1496e1
@@ -233,6 +233,8 @@ jobs:
|
||||
${{ github.event_name == 'workflow_dispatch' && '--match-all-labels' || '' }}
|
||||
secrets: inherit
|
||||
|
||||
# No partition matrix: the tests left on 8x H100 fit one runner's run, so one
|
||||
# job keeps the second H100 host free for other runs.
|
||||
stage-c-8-gpu-h100:
|
||||
needs: [resolve-ci-policy, resolve-ci-image, stage-a-cpu]
|
||||
if: |
|
||||
@@ -242,11 +244,6 @@ jobs:
|
||||
!contains(fromJSON(needs.resolve-ci-policy.outputs.skipped_stages || '[]'), 'stage-c-8-gpu-h100') &&
|
||||
(needs.stage-a-cpu.result == 'success' ||
|
||||
needs.resolve-ci-policy.outputs.bypass_fastfail == 'true')
|
||||
strategy:
|
||||
fail-fast: false
|
||||
max-parallel: ${{ needs.resolve-ci-policy.outputs.cadence == 'weekly' && 1 || 2 }}
|
||||
matrix:
|
||||
partition_id: [0, 1]
|
||||
uses: ./.github/workflows/_run-ci.yml
|
||||
with:
|
||||
runs_on: '["h100", "8gpu"]'
|
||||
@@ -254,7 +251,6 @@ jobs:
|
||||
ref: ${{ inputs.ref || '' }}
|
||||
execute_command: >-
|
||||
python tests/ci/run_suite.py --hw cuda --suite stage-c-8-gpu-h100
|
||||
--auto-partition-id ${{ matrix.partition_id }} --auto-partition-size 2
|
||||
--cadence ${{ needs.resolve-ci-policy.outputs.cadence }}
|
||||
--labels ${{ needs.resolve-ci-policy.outputs.raw_labels }}
|
||||
${{ github.event_name == 'workflow_dispatch' && '--match-all-labels' || '' }}
|
||||
|
||||
@@ -162,8 +162,8 @@ visible before a rollout host reads them.
|
||||
|
||||
The maintained end-to-end coverage is
|
||||
[`tests/e2e/megatron/test_qwen3_4B_disk_delta.py`](https://github.com/radixark/miles/blob/main/tests/e2e/megatron/test_qwen3_4B_disk_delta.py).
|
||||
It exercises a Qwen3-4B Megatron trainer and two SGLang rollout engines on a
|
||||
single 8-GPU node. The same storage contract supports separate hosts, but the
|
||||
It exercises a Qwen3-4B Megatron trainer (TP2 on two GPUs) and two TP1 SGLang
|
||||
rollout engines on the other two GPUs of a single 4-GPU node. The same storage contract supports separate hosts, but the
|
||||
registered test does not reproduce a cross-cluster deployment.
|
||||
|
||||
Current `main` rejects disk-delta with `--colocate`, LoRA, or PD
|
||||
|
||||
@@ -23,7 +23,7 @@ Stage names follow `stage-<tier>-<gpus>-<hw>` (or `stage-<tier>-<hw>` for CPU, e
|
||||
| `stage-b-2-gpu-h200` | 2× H200 | `["h200","2gpu"]` | 1 | both resolvers, `stage-a-cpu` |
|
||||
| `stage-c-2-gpu-h200` | 2× H200 | `["h200","2gpu"]` | 2 | both resolvers, `stage-a-cpu` |
|
||||
| `stage-c-4-gpu-h200` | 4× H200 | `["h200","4gpu"]` | 3 | both resolvers, `stage-a-cpu` |
|
||||
| `stage-c-8-gpu-h100` | 8× H100 | `["h100","8gpu"]` | 2 | both resolvers, `stage-a-cpu` |
|
||||
| `stage-c-8-gpu-h100` | 8× H100 | `["h100","8gpu"]` | 1 | both resolvers, `stage-a-cpu` |
|
||||
| `stage-c-8-gpu-h200` | 8× H200 | `["h200","8gpu"]` | 2 | both resolvers, `stage-a-cpu` |
|
||||
| `stage-c-8-gpu-b200` | 8× B200 | `["b200","8gpu"]` | 1 | both resolvers, `stage-a-cpu` |
|
||||
| `stage-c-4-gpu-mi350` | 4× MI350 | `["self-hosted","amd","mi350","4gpu"]` | 2 | both resolvers |
|
||||
|
||||
@@ -462,7 +462,6 @@ class TestWorkflowScopeSeam:
|
||||
def test_weekly_serializes_each_gpu_matrix(self):
|
||||
workflow = self._workflow()
|
||||
normal_parallelism = {
|
||||
"stage-c-8-gpu-h100": 2,
|
||||
"stage-c-8-gpu-h200": 2,
|
||||
"stage-c-4-gpu-h200": 3,
|
||||
"stage-c-2-gpu-h200": 2,
|
||||
|
||||
@@ -4,18 +4,20 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(est_time=1400, suite="stage-c-8-gpu-h100", labels=["ckpt"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=1200, suite="nightly-stage-c-8-gpu-mi350", labels=["ckpt"])
|
||||
register_cuda_ci(est_time=900, suite="stage-c-2-gpu-h200", labels=["ckpt"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=1200, suite="nightly-stage-c-2-gpu-mi350", labels=["ckpt"])
|
||||
|
||||
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
|
||||
|
||||
MODEL_NAME = "Qwen3-4B"
|
||||
MODEL_TYPE = "qwen3-4B"
|
||||
NUM_GPUS = 8
|
||||
MODEL_NAME = "Qwen3-0.6B"
|
||||
MODEL_TYPE = "qwen3-0.6B"
|
||||
NUM_GPUS = 2
|
||||
# Container-local: /root/models is a host directory shared by every runner on the host.
|
||||
SAVE_DIR = f"/root/checkpoints/{MODEL_NAME}_miles"
|
||||
|
||||
|
||||
def _get_latest_checkpointed_iteration() -> int:
|
||||
latest_path = f"/root/models/{MODEL_NAME}_miles/latest_checkpointed_iteration.txt"
|
||||
latest_path = f"{SAVE_DIR}/latest_checkpointed_iteration.txt"
|
||||
with open(latest_path, encoding="utf-8") as f:
|
||||
latest_text = f.read().strip()
|
||||
if not latest_text.isdigit():
|
||||
@@ -27,7 +29,7 @@ def prepare():
|
||||
U = command_utils.default_config().create_backend()
|
||||
U.exec_command_cpu("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command_cpu(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.exec_command_cpu(f"rm -rf /root/models/{MODEL_NAME}_miles")
|
||||
U.exec_command_cpu(f"rm -rf {SAVE_DIR}")
|
||||
U.hf_download_dataset("zhuzilin/dapo-math-17k")
|
||||
U.hf_download_dataset("zhuzilin/aime-2024")
|
||||
|
||||
@@ -40,15 +42,15 @@ def execute(mode: str = "", ckpt_step: int | None = None):
|
||||
U = command_utils.default_config().create_backend()
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}_torch_dist "
|
||||
if mode == "save":
|
||||
ckpt_args += f"--save /root/models/{MODEL_NAME}_miles "
|
||||
ckpt_args += f"--save {SAVE_DIR} "
|
||||
ckpt_args += "--save-interval 2 "
|
||||
elif mode == "async_save":
|
||||
ckpt_args += f"--save /root/models/{MODEL_NAME}_miles "
|
||||
ckpt_args += f"--save {SAVE_DIR} "
|
||||
ckpt_args += "--save-interval 2 "
|
||||
ckpt_args += "--async-save "
|
||||
ckpt_args += "--use-persistent-ckpt-worker "
|
||||
elif mode == "load":
|
||||
ckpt_args += f"--load /root/models/{MODEL_NAME}_miles "
|
||||
ckpt_args += f"--load {SAVE_DIR} "
|
||||
ckpt_args += f"--ckpt-step {ckpt_step} "
|
||||
ckpt_args += "--low-memory-resume "
|
||||
|
||||
@@ -71,8 +73,8 @@ def execute(mode: str = "", ckpt_step: int | None = None):
|
||||
perf_args = (
|
||||
"--tensor-model-parallel-size 2 "
|
||||
"--sequence-parallel "
|
||||
"--pipeline-model-parallel-size 2 "
|
||||
"--context-parallel-size 2 "
|
||||
"--pipeline-model-parallel-size 1 "
|
||||
"--context-parallel-size 1 "
|
||||
"--recompute-granularity full "
|
||||
"--recompute-method uniform "
|
||||
"--recompute-num-layers 1 "
|
||||
@@ -1,12 +1,12 @@
|
||||
"""AMD 4-GPU variant of test_r3_mtp.py.
|
||||
|
||||
Standalone rather than an IS_HIP branch in the original: the MI300X fleet is
|
||||
split into two 4-GPU runners, so the 8-GPU CUDA case cannot run there as
|
||||
written, and keeping the variant separate means neither side's parallelism
|
||||
constrains the other.
|
||||
split into two 4-GPU runners, and keeping the variant separate means neither
|
||||
side's parallelism constrains the other. The CUDA test_r3_mtp now runs the same
|
||||
4-GPU shape on 4x H200.
|
||||
|
||||
Difference from the CUDA case: cp_size 2 -> 1 and ep_size 4 -> 2, halving the
|
||||
world size from 8 to 4. TP=2 and PP=2 are unchanged, so the MTP layer placement
|
||||
Difference from the original 8-GPU layout (now test_r3_mtp_deepep's): cp_size
|
||||
2 -> 1 and ep_size 4 -> 2, halving the world size from 8 to 4. TP=2 and PP=2 are unchanged, so the MTP layer placement
|
||||
under test is unaffected. Both axes have to shrink because Megatron sizes the
|
||||
dense and expert grids independently -- world_size % (tp * cp * pp) == 0 and
|
||||
world_size % (etp * ep * pp) == 0 -- so dropping CP alone leaves the expert grid
|
||||
|
||||
@@ -4,7 +4,7 @@ from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.metric_history import register_ci_gate
|
||||
from tests.e2e.megatron.test_glm47_flash._common import CaseConfig, execute, prepare
|
||||
|
||||
register_cuda_ci(est_time=1100, suite="stage-c-8-gpu-h100", labels=["megatron"], hardware=["hopper", "blackwell"])
|
||||
register_cuda_ci(est_time=1200, suite="stage-c-4-gpu-h200", labels=["megatron"], hardware=["hopper", "blackwell"])
|
||||
|
||||
register_ci_gate(metric_key="train/grad_norm")
|
||||
register_ci_gate(metric_key="train/ppo_kl")
|
||||
@@ -14,11 +14,13 @@ register_ci_gate(metric_key="rollout/raw_reward")
|
||||
|
||||
CASE = CaseConfig(
|
||||
use_deepep=False,
|
||||
num_gpus_per_node=8,
|
||||
cp_size=2,
|
||||
# tp2/pp2/cp1/ep2 on 4 GPUs (same shape as test_amd_r3_mtp); the 8-GPU tp2/pp2/cp2/ep4
|
||||
# shape is test_r3_mtp_deepep's on 8x H200, which is disabled.
|
||||
num_gpus_per_node=4,
|
||||
cp_size=1,
|
||||
pp_size=2,
|
||||
tp_size=2,
|
||||
ep_size=4,
|
||||
ep_size=2,
|
||||
# GLM-4.7-Flash has 20 attention heads; non-EP SGLang TP must divide it.
|
||||
rollout_num_gpus_per_engine=4,
|
||||
)
|
||||
|
||||
@@ -6,7 +6,7 @@ from tests.e2e.megatron.test_glm47_flash._common import CaseConfig, execute, pre
|
||||
# FIXME: sglang deepep code path bug.
|
||||
register_cuda_ci(
|
||||
est_time=900,
|
||||
suite="stage-c-8-gpu-h100",
|
||||
suite="stage-c-8-gpu-h200",
|
||||
labels=["megatron"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
disabled="Disabled due to sglang deepep code path bug.",
|
||||
|
||||
@@ -17,9 +17,11 @@ class CaseConfig:
|
||||
tp_size: int
|
||||
ep_size: int
|
||||
rollout_num_gpus_per_engine: int
|
||||
etp_size: int = 1
|
||||
sglang_ep_size: int = None
|
||||
sglang_dp_size: int = None
|
||||
sglang_enable_dp_attention: bool = False
|
||||
sglang_max_running_requests: int = 512
|
||||
use_deepep: bool = False
|
||||
use_fp8_rollout: bool = False
|
||||
use_int4_rollout: bool = False
|
||||
@@ -135,7 +137,7 @@ def build_train_args(case: CaseConfig, *, wandb_file: str) -> str:
|
||||
f"--pipeline-model-parallel-size {case.pp_size} "
|
||||
f"--context-parallel-size {case.cp_size} "
|
||||
f"--expert-model-parallel-size {case.ep_size} "
|
||||
"--expert-tensor-parallel-size 1 "
|
||||
f"--expert-tensor-parallel-size {case.etp_size} "
|
||||
"--recompute-granularity full "
|
||||
"--recompute-method uniform "
|
||||
"--recompute-num-layers 1 "
|
||||
@@ -177,7 +179,7 @@ def build_train_args(case: CaseConfig, *, wandb_file: str) -> str:
|
||||
sglang_args = (
|
||||
f"--rollout-num-gpus-per-engine {case.rollout_num_gpus_per_engine} "
|
||||
"--sglang-mem-fraction-static 0.7 "
|
||||
"--sglang-max-running-requests 512 "
|
||||
f"--sglang-max-running-requests {case.sglang_max_running_requests} "
|
||||
"--sglang-enable-metrics "
|
||||
)
|
||||
|
||||
|
||||
@@ -4,7 +4,7 @@ from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.metric_history import register_ci_gate
|
||||
from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, prepare
|
||||
|
||||
register_cuda_ci(est_time=900, suite="stage-c-8-gpu-h100", labels=["megatron"], hardware=["hopper", "blackwell"])
|
||||
register_cuda_ci(est_time=1500, suite="stage-c-4-gpu-h200", labels=["megatron"], hardware=["hopper", "blackwell"])
|
||||
|
||||
register_ci_gate(metric_key="train/grad_norm")
|
||||
register_ci_gate(metric_key="train/ppo_kl")
|
||||
@@ -18,13 +18,19 @@ CASE = CaseConfig(
|
||||
use_int4_rollout=False,
|
||||
use_bridge=False,
|
||||
use_r3=False,
|
||||
num_gpus_per_node=8,
|
||||
cp_size=2,
|
||||
pp_size=2,
|
||||
# tp2/etp2 x dense dp2/ep2 on 4 GPUs: the Megatron DeepEP group is etp2 x ep2 = 4 ranks.
|
||||
# SGLang runs DP attention (dp4, attention TP1) with DeepEP EP4 = engine TP; 512 running
|
||||
# requests / dp4 = 128 tokens per rank, the DeepEP low-latency cap.
|
||||
num_gpus_per_node=4,
|
||||
cp_size=1,
|
||||
pp_size=1,
|
||||
tp_size=2,
|
||||
ep_size=4,
|
||||
rollout_num_gpus_per_engine=8,
|
||||
sglang_ep_size=8,
|
||||
ep_size=2,
|
||||
etp_size=2,
|
||||
rollout_num_gpus_per_engine=4,
|
||||
sglang_ep_size=4,
|
||||
sglang_dp_size=4,
|
||||
sglang_enable_dp_attention=True,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -5,7 +5,7 @@ from tests.ci.metric_history import register_ci_gate
|
||||
from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, prepare
|
||||
|
||||
# Limited by host memory
|
||||
register_cuda_ci(est_time=800, suite="stage-c-8-gpu-h100", labels=["megatron"], hardware=["hopper", "blackwell"])
|
||||
register_cuda_ci(est_time=1300, suite="stage-c-4-gpu-h200", labels=["megatron"], hardware=["hopper", "blackwell"])
|
||||
|
||||
register_ci_gate(metric_key="train/grad_norm")
|
||||
register_ci_gate(metric_key="train/ppo_kl")
|
||||
@@ -19,14 +19,18 @@ CASE = CaseConfig(
|
||||
use_int4_rollout=False,
|
||||
use_bridge=True,
|
||||
use_r3=False,
|
||||
num_gpus_per_node=8,
|
||||
# tp2/pp2/ep2 on 4 GPUs keeps TP, PP and EP all > 1 on the bridge path. Two SGLang engines
|
||||
# at TP2/EP2 DeepEP; 256 running requests keep decode at 128 tokens per rank, the DeepEP
|
||||
# low-latency cap.
|
||||
num_gpus_per_node=4,
|
||||
cp_size=1,
|
||||
pp_size=2,
|
||||
tp_size=4,
|
||||
ep_size=4,
|
||||
rollout_num_gpus_per_engine=8,
|
||||
sglang_ep_size=8,
|
||||
tp_size=2,
|
||||
ep_size=2,
|
||||
rollout_num_gpus_per_engine=2,
|
||||
sglang_ep_size=2,
|
||||
max_tokens_per_gpu=2048,
|
||||
sglang_max_running_requests=256,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -5,9 +5,9 @@ from tests.ci.metric_history import register_ci_gate
|
||||
from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, prepare
|
||||
|
||||
register_cuda_ci(
|
||||
est_time=1300, suite="stage-c-8-gpu-h100", labels=["megatron", "weight-update"], hardware=["hopper", "blackwell"]
|
||||
est_time=1500, suite="stage-c-4-gpu-h200", labels=["megatron", "weight-update"], hardware=["hopper", "blackwell"]
|
||||
)
|
||||
register_rocm_ci(est_time=900, suite="nightly-stage-c-8-gpu-mi350", labels=["megatron", "weight-update"])
|
||||
register_rocm_ci(est_time=900, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "weight-update"])
|
||||
|
||||
register_ci_gate(metric_key="train/grad_norm")
|
||||
register_ci_gate(metric_key="train/ppo_kl")
|
||||
@@ -21,15 +21,20 @@ CASE = CaseConfig(
|
||||
use_int4_rollout=False,
|
||||
use_bridge=False,
|
||||
use_r3=False,
|
||||
num_gpus_per_node=6,
|
||||
cp_size=2,
|
||||
# 3 train + 1 rollout: PP3 broadcast from three senders with VPP2 (8 layers per virtual chunk).
|
||||
# VPP rounds each step's micro-batch count up to a multiple of PP; 16k tokens packs two ~8k
|
||||
# samples per micro-batch so a 32-sample step can reach that multiple.
|
||||
num_gpus_per_node=3,
|
||||
cp_size=1,
|
||||
pp_size=3,
|
||||
tp_size=1,
|
||||
ep_size=2,
|
||||
ep_size=1,
|
||||
colocate=False,
|
||||
rollout_num_gpus=2,
|
||||
rollout_num_gpus_per_engine=2,
|
||||
rollout_num_gpus=1,
|
||||
rollout_num_gpus_per_engine=1,
|
||||
update_weight_transfer_mode="broadcast",
|
||||
max_tokens_per_gpu=16384,
|
||||
extra_args="--num-layers-per-virtual-pipeline-stage 8 ",
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -6,12 +6,12 @@ from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, p
|
||||
|
||||
register_cuda_ci(
|
||||
est_time=1500,
|
||||
suite="stage-c-8-gpu-h100",
|
||||
suite="stage-c-4-gpu-h200",
|
||||
labels=["megatron", "weight-update", "fully-async"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(
|
||||
est_time=800, suite="nightly-stage-c-8-gpu-mi350", labels=["megatron", "weight-update", "fully-async"]
|
||||
est_time=800, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "weight-update", "fully-async"]
|
||||
)
|
||||
|
||||
register_ci_gate(metric_key="train/grad_norm")
|
||||
@@ -30,17 +30,20 @@ CASE = CaseConfig(
|
||||
use_int4_rollout=False,
|
||||
use_bridge=False,
|
||||
use_r3=False,
|
||||
num_gpus_per_node=6,
|
||||
cp_size=2,
|
||||
# 3 train + 1 rollout with an uneven PP3 split (17/17/14): layer offsets 0/17/34 differ
|
||||
# from the even split on both non-first stages.
|
||||
num_gpus_per_node=3,
|
||||
cp_size=1,
|
||||
pp_size=3,
|
||||
tp_size=1,
|
||||
ep_size=2,
|
||||
ep_size=1,
|
||||
colocate=False,
|
||||
rollout_num_gpus=2,
|
||||
rollout_num_gpus_per_engine=2,
|
||||
rollout_num_gpus=1,
|
||||
rollout_num_gpus_per_engine=1,
|
||||
update_weight_transfer_mode="broadcast",
|
||||
num_rollout=3,
|
||||
fully_async=True,
|
||||
extra_args="--decoder-first-pipeline-num-layers 17 --decoder-last-pipeline-num-layers 14 ",
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -5,17 +5,17 @@ from miles.utils.external_utils import command_utils
|
||||
|
||||
MODEL_NAME = "Qwen3-4B"
|
||||
MODEL_TYPE = "qwen3-4B"
|
||||
NUM_GPUS = 8
|
||||
NUM_GPUS = 4
|
||||
|
||||
register_cuda_ci(
|
||||
est_time=600,
|
||||
suite="stage-c-8-gpu-h100",
|
||||
est_time=400,
|
||||
suite="stage-c-4-gpu-h200",
|
||||
labels=["megatron", "weight-update"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(
|
||||
est_time=500,
|
||||
suite="nightly-stage-c-8-gpu-mi350",
|
||||
suite="nightly-stage-c-4-gpu-mi350",
|
||||
labels=["megatron", "weight-update"],
|
||||
)
|
||||
|
||||
@@ -58,7 +58,7 @@ def execute():
|
||||
"--tensor-model-parallel-size 2 "
|
||||
"--sequence-parallel "
|
||||
"--pipeline-model-parallel-size 1 "
|
||||
"--context-parallel-size 2 "
|
||||
"--context-parallel-size 1 "
|
||||
"--recompute-granularity full "
|
||||
"--recompute-method uniform "
|
||||
"--recompute-num-layers 1 "
|
||||
@@ -85,7 +85,7 @@ def execute():
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 2 " f"--rollout-num-gpus {NUM_GPUS // 2} " "--sglang-mem-fraction-static 0.8 "
|
||||
"--rollout-num-gpus-per-engine 1 " f"--rollout-num-gpus {NUM_GPUS // 2} " "--sglang-mem-fraction-static 0.8 "
|
||||
)
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
@@ -9,7 +9,7 @@ miles has no VLM/vision implementation on the training side, so Qwen3.5's `visua
|
||||
weights are never synced and must be excluded from the weight-equality check; each case
|
||||
passes `check_weight_update_skip_list=("visual",)`.
|
||||
|
||||
Topology follows scripts/run_qwen3_5_35b_a3b_mtp.py (cp2/ep8 on 8 GPUs).
|
||||
Each case picks its own topology (see its CASE).
|
||||
Spec (EAGLE) and spec-v2 (mamba scheduler) are on for the whole suite; R3 is per-case.
|
||||
"""
|
||||
|
||||
@@ -24,8 +24,7 @@ MODEL_TYPE = "qwen3.5-35B-A3B"
|
||||
|
||||
@dataclass
|
||||
class CaseConfig:
|
||||
# Topology / GPU counts — explicit per case (each test file picks a shape that fits
|
||||
# the Qwen3.5 GatedDeltaNet backward on 8x80GB; see each file's CASE).
|
||||
# Topology / GPU counts — explicit per case (see each file's CASE).
|
||||
num_gpus_per_node: int
|
||||
cp_size: int
|
||||
pp_size: int
|
||||
|
||||
@@ -16,7 +16,7 @@ from tests.ci.metric_history import register_ci_gate
|
||||
from tests.e2e.megatron.test_qwen3_5_35B_A3B_mtp._common import CaseConfig, execute, prepare
|
||||
|
||||
register_cuda_ci(
|
||||
est_time=1800, suite="stage-c-8-gpu-h100", labels=["megatron", "qwen35"], hardware=["hopper", "blackwell"]
|
||||
est_time=1900, suite="stage-c-4-gpu-h200", labels=["megatron", "qwen35"], hardware=["hopper", "blackwell"]
|
||||
)
|
||||
|
||||
register_ci_gate(metric_key="train/grad_norm")
|
||||
@@ -26,18 +26,18 @@ register_ci_gate(metric_key="train/train_rollout_kl")
|
||||
register_ci_gate(metric_key="rollout/raw_reward")
|
||||
|
||||
CASE = CaseConfig(
|
||||
# tp2/cp2/pp2/ep4: TP=4 hits a Qwen3.5 attention-output-gate sharding bug, so stay at
|
||||
# TP=2 and use PP=2 to halve the resident layers for OOM headroom (keeps CP=2 coverage).
|
||||
num_gpus_per_node=8,
|
||||
# tp2/cp2/ep4: TP=4 hits a Qwen3.5 attention-output-gate sharding bug, so stay at TP=2.
|
||||
# PP=1 on 4 GPUs: PP=2 stays covered by test_mtp1_spec_v2_r3.
|
||||
num_gpus_per_node=4,
|
||||
cp_size=2,
|
||||
pp_size=2,
|
||||
pp_size=1,
|
||||
tp_size=2,
|
||||
ep_size=4,
|
||||
# 4096 (mtp1 keeps 8192): CP=2 routes the GatedDeltaNet backward through the heavier fla
|
||||
# CP kernel, whose Triton autotune OOMs at 8192 even with PP=2; halve the budget for headroom.
|
||||
max_tokens_per_gpu=4096,
|
||||
rollout_num_gpus_per_engine=8,
|
||||
sglang_ep_size=8,
|
||||
rollout_num_gpus_per_engine=4,
|
||||
sglang_ep_size=4,
|
||||
enable_mtp_training=False,
|
||||
use_r3=False,
|
||||
check_weight_update_selector="target",
|
||||
|
||||
@@ -12,7 +12,7 @@ from tests.ci.metric_history import register_ci_gate
|
||||
from tests.e2e.megatron.test_qwen3_5_35B_A3B_mtp._common import CaseConfig, execute, prepare
|
||||
|
||||
register_cuda_ci(
|
||||
est_time=1600, suite="stage-c-8-gpu-h100", labels=["megatron", "qwen35"], hardware=["hopper", "blackwell"]
|
||||
est_time=1500, suite="stage-c-8-gpu-h200", labels=["megatron", "qwen35"], hardware=["hopper", "blackwell"]
|
||||
)
|
||||
|
||||
register_ci_gate(metric_key="train/grad_norm")
|
||||
@@ -22,9 +22,9 @@ register_ci_gate(metric_key="train/train_rollout_kl")
|
||||
register_ci_gate(metric_key="rollout/raw_reward")
|
||||
|
||||
CASE = CaseConfig(
|
||||
# tp2/pp2/cp1/ep4: TP=4 hits a Qwen3.5 attention-output-gate sharding bug, so stay at
|
||||
# TP=2; CP=1 avoids the memory-heavy GatedDeltaNet CP backward kernel and PP=2 halves
|
||||
# the resident layers, together fitting the MTP-training run on 8x80GB.
|
||||
# tp2/pp2/cp1/ep4 on 8x H200: TP=4 hits a Qwen3.5 attention-output-gate sharding bug, so
|
||||
# stay at TP=2; PP=2 with dense DP2 keeps Megatron DeepEP x DP x PP (a 4-rank DeepEP group
|
||||
# per PP stage), which no 4-GPU case can hold.
|
||||
num_gpus_per_node=8,
|
||||
cp_size=1,
|
||||
pp_size=2,
|
||||
|
||||
@@ -14,8 +14,8 @@ from miles.utils.external_utils import command_utils
|
||||
|
||||
|
||||
register_cuda_ci(
|
||||
est_time=1300,
|
||||
suite="stage-c-8-gpu-h100",
|
||||
est_time=1200,
|
||||
suite="stage-c-8-gpu-h200",
|
||||
labels=["megatron", "model-scripts", "lora"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
|
||||
@@ -4,16 +4,16 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(est_time=400, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=600, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
|
||||
register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=600, suite="nightly-stage-c-4-gpu-mi350", labels=["short"])
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
NUM_GPUS = 8
|
||||
NUM_GPUS = 4
|
||||
|
||||
# Inline sglang config: same model, 2 engine groups with different sizes.
|
||||
# Group 1: 4 GPUs, 1 GPU/engine (tp=1) -> 4 engines
|
||||
# Group 2: 4 GPUs, 1 GPU/engine (tp=1) -> 4 engines
|
||||
# Group 1: 2 GPUs, 1 GPU/engine (tp=1) -> 2 engines
|
||||
# Group 2: 2 GPUs, 1 GPU/engine (tp=1) -> 2 engines
|
||||
# Tests that RolloutServer correctly manages multiple engine groups
|
||||
# behind a single router, with separate port cursors per group.
|
||||
SGLANG_CONFIG_YAML = """\
|
||||
@@ -21,10 +21,10 @@ sglang:
|
||||
- name: default
|
||||
server_groups:
|
||||
- worker_type: regular
|
||||
num_gpus: 4
|
||||
num_gpus: 2
|
||||
num_gpus_per_engine: 1
|
||||
- worker_type: regular
|
||||
num_gpus: 4
|
||||
num_gpus: 2
|
||||
num_gpus_per_engine: 1
|
||||
"""
|
||||
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
"""E2E test: mixed offload with updatable + frozen models.
|
||||
|
||||
Deploys two models via --sglang-config in colocate mode:
|
||||
- "actor": update_weights=true, 4 GPUs -> overlaps with megatron, gets offloaded
|
||||
- "actor": update_weights=true, 2 GPUs -> overlaps with megatron, gets offloaded
|
||||
and weights updated from training.
|
||||
- "ref": update_weights=false, 4 GPUs -> overlaps with megatron, gets offloaded
|
||||
- "ref": update_weights=false, 2 GPUs -> overlaps with megatron, gets offloaded
|
||||
and weights restored from disk (update_weights_from_disk).
|
||||
|
||||
Key coverage:
|
||||
@@ -19,27 +19,27 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(est_time=400, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
|
||||
register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-4-gpu-mi350", labels=["short"])
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
NUM_GPUS = 8
|
||||
NUM_GPUS = 4
|
||||
|
||||
# Two models on 8 GPUs (colocate): actor gets weight updates, ref is frozen.
|
||||
# Two models on 4 GPUs (colocate): actor gets weight updates, ref is frozen.
|
||||
SGLANG_CONFIG_YAML = """\
|
||||
sglang:
|
||||
- name: actor
|
||||
update_weights: true
|
||||
server_groups:
|
||||
- worker_type: regular
|
||||
num_gpus: 4
|
||||
num_gpus: 2
|
||||
num_gpus_per_engine: 1
|
||||
- name: ref
|
||||
update_weights: false
|
||||
server_groups:
|
||||
- worker_type: regular
|
||||
num_gpus: 4
|
||||
num_gpus: 2
|
||||
num_gpus_per_engine: 1
|
||||
"""
|
||||
|
||||
|
||||
@@ -16,27 +16,27 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(est_time=500, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=600, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
|
||||
register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=600, suite="nightly-stage-c-4-gpu-mi350", labels=["short"])
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
NUM_GPUS = 8
|
||||
NUM_GPUS = 4
|
||||
|
||||
# Two models on 8 GPUs (colocate): actor gets weight updates, ref is frozen.
|
||||
# Two models on 4 GPUs (colocate): actor gets weight updates, ref is frozen.
|
||||
SGLANG_CONFIG_YAML = """\
|
||||
sglang:
|
||||
- name: actor
|
||||
update_weights: true
|
||||
server_groups:
|
||||
- worker_type: regular
|
||||
num_gpus: 4
|
||||
num_gpus: 2
|
||||
num_gpus_per_engine: 1
|
||||
- name: ref
|
||||
update_weights: false
|
||||
server_groups:
|
||||
- worker_type: regular
|
||||
num_gpus: 4
|
||||
num_gpus: 2
|
||||
num_gpus_per_engine: 1
|
||||
"""
|
||||
|
||||
|
||||
@@ -31,7 +31,7 @@ from tests.e2e.conftest_dumper import (
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(est_time=1100, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_cuda_ci(est_time=1200, suite="stage-c-8-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=1800, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
|
||||
|
||||
|
||||
|
||||
@@ -9,12 +9,12 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(est_time=400, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=400, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
|
||||
register_cuda_ci(est_time=400, suite="stage-c-2-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=400, suite="nightly-stage-c-2-gpu-mi350", labels=["short"])
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
NUM_GPUS = 8
|
||||
NUM_GPUS = 2
|
||||
|
||||
|
||||
def prepare():
|
||||
|
||||
@@ -1,19 +1,18 @@
|
||||
import os
|
||||
from tempfile import TemporaryDirectory
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(
|
||||
est_time=400, suite="stage-c-8-gpu-h100", labels=["short", "eval", "fully-async"], hardware=["hopper", "blackwell"]
|
||||
est_time=400, suite="stage-c-4-gpu-h200", labels=["short", "eval", "fully-async"], hardware=["hopper", "blackwell"]
|
||||
)
|
||||
register_rocm_ci(est_time=400, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "eval", "fully-async"])
|
||||
|
||||
FEW_GPU = command_utils.get_bool_env_var("MILES_TEST_FEW_GPU", "0")
|
||||
register_rocm_ci(est_time=400, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "eval", "fully-async"])
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
NUM_GPUS = 4 if FEW_GPU else 8
|
||||
NUM_GPUS = 4
|
||||
|
||||
|
||||
def prepare():
|
||||
@@ -23,7 +22,7 @@ def prepare():
|
||||
U.hf_download_dataset("zhuzilin/gsm8k")
|
||||
|
||||
|
||||
def execute():
|
||||
def execute(eval_hf_dir: str):
|
||||
U = command_utils.default_config().create_backend()
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}/ "
|
||||
|
||||
@@ -56,7 +55,7 @@ def execute():
|
||||
"--eval-top-k 1 "
|
||||
"--eval-num-gpus 1 "
|
||||
"--eval-num-gpus-per-engine 1 "
|
||||
"--eval-hf-dir /dev/shm/miles_e2e_eval_hf "
|
||||
f"--eval-hf-dir {eval_hf_dir} "
|
||||
"--eval-keep-snapshots 2 "
|
||||
)
|
||||
|
||||
@@ -104,8 +103,8 @@ def execute():
|
||||
"--attention-softmax-in-fp32 "
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
f"--actor-num-gpus-per-node {1 if FEW_GPU else 2} "
|
||||
f"--rollout-num-gpus {2 if FEW_GPU else 5} "
|
||||
"--actor-num-gpus-per-node 2 "
|
||||
"--rollout-num-gpus 1 "
|
||||
# HF-format --ref-load requires the bridge loader; eval snapshots are
|
||||
# exported through the bridge path as well (marker-gated).
|
||||
"--megatron-to-hf-mode bridge "
|
||||
@@ -136,4 +135,6 @@ if __name__ == "__main__":
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
# CI containers share host IPC, so each run must own its snapshot directory.
|
||||
with TemporaryDirectory(prefix="miles_e2e_eval_hf_", dir="/dev/shm") as eval_hf_dir:
|
||||
execute(eval_hf_dir)
|
||||
|
||||
@@ -5,15 +5,13 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(
|
||||
est_time=400, suite="stage-c-8-gpu-h100", labels=["short", "mooncake"], hardware=["hopper", "blackwell"]
|
||||
est_time=400, suite="stage-c-4-gpu-h200", labels=["short", "mooncake"], hardware=["hopper", "blackwell"]
|
||||
)
|
||||
register_rocm_ci(est_time=240, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "mooncake"])
|
||||
|
||||
FEW_GPU = command_utils.get_bool_env_var("MILES_TEST_FEW_GPU", "0")
|
||||
register_rocm_ci(est_time=240, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "mooncake"])
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
NUM_GPUS = 4 if FEW_GPU else 8
|
||||
NUM_GPUS = 4
|
||||
|
||||
|
||||
def prepare():
|
||||
@@ -102,8 +100,8 @@ def execute():
|
||||
"--attention-softmax-in-fp32 "
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
f"--actor-num-gpus-per-node {1 if FEW_GPU else 2} "
|
||||
f"--rollout-num-gpus {3 if FEW_GPU else 6} "
|
||||
"--actor-num-gpus-per-node 2 "
|
||||
"--rollout-num-gpus 2 "
|
||||
"--megatron-to-hf-mode bridge "
|
||||
)
|
||||
|
||||
|
||||
@@ -8,17 +8,15 @@ from miles.utils.object_store import ObjectStoreBackend
|
||||
from miles.utils.workers.types import WorkerCommBackend
|
||||
|
||||
register_cuda_ci(
|
||||
est_time=400, suite="stage-c-8-gpu-h100", labels=["short", "mooncake"], hardware=["hopper", "blackwell"]
|
||||
est_time=400, suite="stage-c-2-gpu-h200", labels=["short", "mooncake"], hardware=["hopper", "blackwell"]
|
||||
)
|
||||
register_rocm_ci(est_time=360, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "mooncake"])
|
||||
|
||||
FEW_GPU = command_utils.get_bool_env_var("MILES_TEST_FEW_GPU", "0")
|
||||
register_rocm_ci(est_time=360, suite="nightly-stage-c-2-gpu-mi350", labels=["short", "mooncake"])
|
||||
|
||||
MODEL_DIR = get_test_model_dir()
|
||||
DATA_DIR = get_test_data_dir()
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
NUM_GPUS = 4 if FEW_GPU else 8
|
||||
NUM_GPUS = 2
|
||||
|
||||
|
||||
def entrypoint(
|
||||
|
||||
@@ -7,15 +7,15 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(est_time=400, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
|
||||
register_cuda_ci(est_time=300, suite="stage-c-4-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-4-gpu-mi350", labels=["short"])
|
||||
|
||||
TIGHT_DEVICE_MEMORY = command_utils.get_bool_env_var("MILES_TEST_TIGHT_DEVICE_MEMORY", "1")
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
NUM_GPUS = 8
|
||||
NUM_TRAIN_GPUS = 4
|
||||
NUM_GPUS = 4
|
||||
NUM_TRAIN_GPUS = 2
|
||||
|
||||
TEACHER_HOST = "127.0.0.1"
|
||||
TEACHER_PORT = 13141
|
||||
|
||||
Reference in New Issue
Block a user