mirror of
https://github.com/radixark/miles.git
synced 2026-10-02 07:14:53 +08:00
[AMD] ci: expand ROCm nightly coverage and refresh runtime estimates (#2541)
Co-authored-by: Zhiyao Jiang <jessicajiang324@gmail.com>
This commit is contained in:
co-authored by
Zhiyao Jiang
parent
3439ec7513
commit
5ebc06f8d2
@@ -5,7 +5,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(est_time=900, suite="stage-c-2-gpu-h200", labels=["ckpt"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=1200, suite="nightly-stage-c-2-gpu-mi350", labels=["ckpt"])
|
||||
register_rocm_ci(est_time=1500, suite="nightly-stage-c-2-gpu-mi350", labels=["ckpt"])
|
||||
|
||||
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import os
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
@@ -10,6 +10,7 @@ register_cuda_ci(
|
||||
labels=["fsdp"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(est_time=400, suite="nightly-stage-c-8-gpu-mi350", labels=["fsdp"])
|
||||
|
||||
NUM_GPUS = 8
|
||||
DP_REPLICATE_SIZE = 2
|
||||
|
||||
+2
-1
@@ -3,7 +3,7 @@
|
||||
# (the CUDA CI runner's execution model). Scenario logic lives in
|
||||
# tests/e2e/ft/conftest_ft/scenario_trainer_no_failure.py.
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from tests.e2e.ft.conftest_ft.scenario_trainer_no_failure import run_ci
|
||||
|
||||
register_cuda_ci(
|
||||
@@ -12,6 +12,7 @@ register_cuda_ci(
|
||||
labels=["ft-short"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(est_time=900, suite="nightly-stage-c-8-gpu-mi350", labels=["ft-short"])
|
||||
|
||||
_MODE: str = "kill_train__dp2_cp2_pp2__fake_rollout__moe_5layer"
|
||||
|
||||
|
||||
+2
-1
@@ -3,7 +3,7 @@
|
||||
# (the CUDA CI runner's execution model). Scenario logic lives in
|
||||
# tests/e2e/ft/conftest_ft/scenario_trainer_no_failure.py.
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from tests.e2e.ft.conftest_ft.scenario_trainer_no_failure import run_ci
|
||||
|
||||
register_cuda_ci(
|
||||
@@ -12,6 +12,7 @@ register_cuda_ci(
|
||||
labels=["ft-short"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(est_time=1000, suite="nightly-stage-c-8-gpu-mi350", labels=["ft-short"])
|
||||
|
||||
_MODE: str = "kill_train__dp2_cp2_tp2_ep2__fake_rollout__moe_5layer"
|
||||
|
||||
|
||||
@@ -5,7 +5,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(est_time=5400, suite="stage-c-2-gpu-h200", labels=["long"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=5000, suite="nightly-stage-c-2-gpu-mi350", labels=["long"])
|
||||
register_rocm_ci(est_time=5400, suite="nightly-stage-c-2-gpu-mi350", labels=["long"])
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
|
||||
@@ -18,7 +18,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["lora"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-4-gpu-mi350", labels=["lora"])
|
||||
register_rocm_ci(est_time=500, suite="nightly-stage-c-4-gpu-mi350", labels=["lora"])
|
||||
|
||||
|
||||
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
|
||||
|
||||
@@ -19,7 +19,7 @@ from megatron.core.tensor_parallel.layers import ColumnParallelLinear
|
||||
from megatron.core.tensor_parallel.random import model_parallel_cuda_manual_seed
|
||||
from megatron.core.transformer.module import MegatronModule
|
||||
from megatron.core.transformer.transformer_config import TransformerConfig
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from torch.utils._pytree import tree_flatten, tree_map
|
||||
|
||||
from miles.backends.megatron_utils.lora.checkpoint import load_slot, save_slot
|
||||
@@ -27,6 +27,7 @@ from miles.backends.megatron_utils.lora.optimizer import SlotOptimizer, adapter_
|
||||
from miles.utils.distributed_utils import init_gloo_group
|
||||
|
||||
register_cuda_ci(est_time=120, suite="stage-b-2-gpu-h200", labels=["lora"], hardware=["hopper"])
|
||||
register_rocm_ci(est_time=60, suite="nightly-stage-c-2-gpu-mi350", labels=["lora"])
|
||||
|
||||
ADAM = dict(learning_rate=3e-4, beta1=0.9, beta2=0.95, eps=1e-8, weight_decay=0.01, grad_clip_norm=1.0)
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import os
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
@@ -16,6 +16,7 @@ register_cuda_ci(
|
||||
labels=["megatron", "model-scripts", "lora"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(est_time=1600, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "model-scripts", "lora"])
|
||||
|
||||
MODEL_NAME = "gpt-oss-20b-bf16"
|
||||
MODEL_TYPE = "gpt-oss-20b"
|
||||
|
||||
@@ -16,7 +16,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["megatron"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=420, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron"])
|
||||
register_rocm_ci(est_time=600, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron"])
|
||||
|
||||
MODEL_NAME = "MiMo-7B-RL"
|
||||
MODEL_TYPE = "mimo-7B-rl"
|
||||
|
||||
@@ -18,7 +18,7 @@ register_ci_gate(metric_key="train/train_rollout_kl")
|
||||
register_ci_gate(metric_key="rollout/raw_reward")
|
||||
|
||||
register_rocm_ci(
|
||||
est_time=900, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "weight-update", "short", "mooncake"]
|
||||
est_time=2000, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "weight-update", "short", "mooncake"]
|
||||
)
|
||||
|
||||
CASE = CaseConfig(
|
||||
|
||||
@@ -7,13 +7,14 @@ training partition and per-rollout loss weighting are unchanged."""
|
||||
|
||||
import os
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from tests.ci.metric_history import register_ci_gate
|
||||
from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, prepare
|
||||
|
||||
register_cuda_ci(
|
||||
est_time=1100, suite="stage-c-4-gpu-h200", labels=["megatron", "short"], hardware=["hopper", "blackwell"]
|
||||
)
|
||||
register_rocm_ci(est_time=800, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "short"])
|
||||
|
||||
register_ci_gate(metric_key="train/grad_norm")
|
||||
register_ci_gate(metric_key="train/ppo_kl")
|
||||
|
||||
@@ -7,7 +7,7 @@ from tests.e2e.megatron.test_qwen3_30B_A3B._common import CaseConfig, execute, p
|
||||
register_cuda_ci(
|
||||
est_time=1200, suite="stage-c-4-gpu-h200", labels=["megatron", "replay"], hardware=["hopper", "blackwell"]
|
||||
)
|
||||
register_rocm_ci(est_time=1500, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "replay"])
|
||||
register_rocm_ci(est_time=1000, suite="nightly-stage-c-4-gpu-mi350", labels=["megatron", "replay"])
|
||||
|
||||
register_ci_gate(metric_key="train/grad_norm")
|
||||
register_ci_gate(metric_key="train/ppo_kl")
|
||||
|
||||
@@ -7,7 +7,7 @@ weights must come from the host backup rather than the unmapped param buffers.
|
||||
--check-weight-update-equal makes the engines verify what they received.
|
||||
"""
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from tests.ci.metric_history import register_ci_gate
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
@@ -23,6 +23,7 @@ register_cuda_ci(
|
||||
labels=["megatron", "weight-update"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["megatron", "weight-update"])
|
||||
register_ci_gate(metric_key="train/grad_norm")
|
||||
register_ci_gate(metric_key="rollout/raw_reward")
|
||||
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
import os
|
||||
|
||||
from scripts.run_qwen3_5_35b_a3b_lora import ScriptArgs, _prepare_download, _train
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
@@ -19,6 +19,7 @@ register_cuda_ci(
|
||||
labels=["megatron", "model-scripts", "lora"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(est_time=900, suite="nightly-stage-c-8-gpu-mi350", labels=["megatron", "model-scripts", "lora"])
|
||||
|
||||
# (name, experts_shared_outer_loras, virtual_experts_serving)
|
||||
_CONFIGS = [
|
||||
@@ -28,6 +29,13 @@ _CONFIGS = [
|
||||
|
||||
|
||||
def _args(shared_outer: bool, virtual_experts: bool) -> ScriptArgs:
|
||||
extra_args = "--ci-test --ci-disable-logprobs-checker "
|
||||
if not virtual_experts:
|
||||
extra_args += "--no-sglang-lora-use-virtual-experts "
|
||||
# ROCm shared-expert fusion requires per-expert LoRA factors.
|
||||
if os.getenv("MILES_HARDWARE_PLATFORM") == "rocm" and shared_outer:
|
||||
extra_args += "--sglang-disable-shared-experts-fusion "
|
||||
|
||||
return ScriptArgs.from_env(
|
||||
model_name="Qwen3.5-35B-A3B",
|
||||
num_nodes=1,
|
||||
@@ -35,10 +43,7 @@ def _args(shared_outer: bool, virtual_experts: bool) -> ScriptArgs:
|
||||
num_rollout=1,
|
||||
experts_shared_outer_loras=shared_outer,
|
||||
enable_wandb=False,
|
||||
extra_args=(
|
||||
"--ci-test --ci-disable-logprobs-checker "
|
||||
+ ("" if virtual_experts else "--no-sglang-lora-use-virtual-experts ")
|
||||
),
|
||||
extra_args=extra_args,
|
||||
)
|
||||
|
||||
|
||||
|
||||
@@ -14,7 +14,7 @@ import torch.distributed as dist
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
register_cuda_ci(est_time=30, suite="stage-c-4-gpu-h200", labels=["precision"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=60, suite="nightly-stage-c-4-gpu-mi350", labels=["precision"])
|
||||
register_rocm_ci(est_time=30, suite="nightly-stage-c-4-gpu-mi350", labels=["precision"])
|
||||
|
||||
from miles_plugins.models.cp_utils import packed_shard_to_zigzag, zigzag_to_packed_shard
|
||||
|
||||
|
||||
@@ -17,7 +17,7 @@ import torch.distributed as dist
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
register_cuda_ci(est_time=300, suite="stage-c-4-gpu-h200", labels=["precision"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=120, suite="nightly-stage-c-4-gpu-mi350", labels=["precision"])
|
||||
register_rocm_ci(est_time=200, suite="nightly-stage-c-4-gpu-mi350", labels=["precision"])
|
||||
|
||||
|
||||
def setup_dist():
|
||||
|
||||
@@ -1,8 +1,9 @@
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from tests.ci.metric_history import register_ci_gate
|
||||
from tests.e2e.sglang.test_session_server_multi_role._common import ModelConfig, run_both_versions
|
||||
|
||||
register_cuda_ci(est_time=800, suite="stage-c-2-gpu-h200", labels=["sglang"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=700, suite="nightly-stage-c-2-gpu-mi350", labels=["sglang"])
|
||||
register_ci_gate(metric_key="rollout/tito_session_mismatch_rate/v1/assistant_text")
|
||||
register_ci_gate(metric_key="rollout/tito_session_mismatch_rate/v2/assistant_text")
|
||||
|
||||
|
||||
@@ -5,7 +5,7 @@ from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from miles.utils.external_utils import command_utils
|
||||
|
||||
register_cuda_ci(est_time=400, suite="stage-c-4-gpu-h200", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=600, suite="nightly-stage-c-4-gpu-mi350", labels=["short"])
|
||||
register_rocm_ci(est_time=400, suite="nightly-stage-c-4-gpu-mi350", labels=["short"])
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
|
||||
@@ -2,7 +2,7 @@ import dataclasses
|
||||
import os
|
||||
|
||||
from examples.multi_policy.run_solver_verifier_gsm8k import ScriptArgs, prepare
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from tests.e2e.conftest_multi_policy import execute
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
@@ -13,6 +13,7 @@ register_cuda_ci(
|
||||
labels=["short", "multi-policy", "fully-async"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(est_time=1700, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "multi-policy", "fully-async"])
|
||||
|
||||
NUM_ROLLOUT = int(os.environ.get("MILES_TEST_NUM_ROLLOUT", "5"))
|
||||
SAVE_INTERVAL = 2
|
||||
|
||||
@@ -7,7 +7,7 @@ from miles.utils.external_utils import command_utils
|
||||
register_cuda_ci(
|
||||
est_time=400, suite="stage-c-4-gpu-h200", labels=["short", "mooncake"], hardware=["hopper", "blackwell"]
|
||||
)
|
||||
register_rocm_ci(est_time=240, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "mooncake"])
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-4-gpu-mi350", labels=["short", "mooncake"])
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
|
||||
@@ -10,7 +10,7 @@ from miles.utils.workers.types import WorkerCommBackend
|
||||
register_cuda_ci(
|
||||
est_time=400, suite="stage-c-2-gpu-h200", labels=["short", "mooncake"], hardware=["hopper", "blackwell"]
|
||||
)
|
||||
register_rocm_ci(est_time=360, suite="nightly-stage-c-2-gpu-mi350", labels=["short", "mooncake"])
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-2-gpu-mi350", labels=["short", "mooncake"])
|
||||
|
||||
MODEL_DIR = get_test_model_dir()
|
||||
DATA_DIR = get_test_data_dir()
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import os
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
import miles.utils.external_utils.command_utils.legacy as U
|
||||
|
||||
@@ -11,6 +11,7 @@ register_cuda_ci(
|
||||
labels=["short", "mooncake"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "mooncake"])
|
||||
|
||||
FEW_GPU = U.get_bool_env_var("MILES_TEST_FEW_GPU", "0")
|
||||
|
||||
|
||||
@@ -2,7 +2,7 @@ import importlib.util
|
||||
from pathlib import Path
|
||||
from types import ModuleType
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
register_cuda_ci(
|
||||
est_time=450,
|
||||
@@ -10,6 +10,7 @@ register_cuda_ci(
|
||||
labels=["short", "mooncake", "rpc-comm"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "mooncake", "rpc-comm"])
|
||||
|
||||
|
||||
def _load_base_test() -> ModuleType:
|
||||
|
||||
@@ -2,7 +2,7 @@ import importlib.util
|
||||
from pathlib import Path
|
||||
from types import ModuleType
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
register_cuda_ci(
|
||||
est_time=450,
|
||||
@@ -10,6 +10,7 @@ register_cuda_ci(
|
||||
labels=["short", "rpc-comm"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short", "rpc-comm"])
|
||||
|
||||
|
||||
def _load_base_test() -> ModuleType:
|
||||
|
||||
@@ -12,7 +12,7 @@ tests/e2e/megatron/test_qwen3_4b_fully_async_eval.py.
|
||||
import os
|
||||
|
||||
import pandas as pd
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
from transformers import AutoTokenizer
|
||||
|
||||
from miles.utils.external_utils import command_utils
|
||||
@@ -23,6 +23,7 @@ register_cuda_ci(
|
||||
labels=["short", "eval", "megatron"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-2-gpu-mi350", labels=["short", "eval", "megatron"])
|
||||
|
||||
MODEL_NAME = "Qwen3-0.6B"
|
||||
MODEL_TYPE = "qwen3-0.6B"
|
||||
|
||||
@@ -40,7 +40,7 @@ NUM_LAYERS: int = 5
|
||||
_RUN_DIR: Path = Path(tempfile.mkdtemp(prefix="test_run_megatron_"))
|
||||
|
||||
register_cuda_ci(est_time=200, suite="stage-c-8-gpu-h100", labels=["short"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=2000, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
|
||||
register_rocm_ci(est_time=300, suite="nightly-stage-c-8-gpu-mi350", labels=["short"])
|
||||
|
||||
|
||||
@dataclasses.dataclass(frozen=True)
|
||||
|
||||
@@ -1,8 +1,9 @@
|
||||
"""FSDP2 gradient parity across r1s4, r2s2, and r4s1."""
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
register_cuda_ci(est_time=90, suite="stage-c-4-gpu-h200", labels=["fsdp"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=90, suite="nightly-stage-c-4-gpu-mi350", labels=["fsdp"])
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
@@ -4,7 +4,7 @@ import os
|
||||
from types import SimpleNamespace
|
||||
|
||||
import torch
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
from miles_plugins.optimizers.nvme_stream import NVMeOptimizerStateStore, _Bucket, _Entry, _resize, _Stager
|
||||
|
||||
@@ -14,6 +14,7 @@ register_cuda_ci(
|
||||
labels=["miles-plugin"],
|
||||
hardware=["hopper", "blackwell"],
|
||||
)
|
||||
register_rocm_ci(est_time=30, suite="nightly-stage-c-2-gpu-mi350", labels=["miles-plugin"])
|
||||
|
||||
|
||||
def test_bucketwise_main_initialization_preserves_bytes_and_releases_cuda_storage(tmp_path):
|
||||
|
||||
@@ -4,9 +4,10 @@ Only sampled rows have autograd references, keeping the test smaller than a full
|
||||
model. Inputs vary across rows and channels so a wrong in-bounds address also fails.
|
||||
"""
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
register_cuda_ci(est_time=180, suite="stage-b-2-gpu-h200", labels=["miles-plugin"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=40, suite="nightly-stage-c-2-gpu-mi350", labels=["miles-plugin"])
|
||||
|
||||
import math
|
||||
|
||||
|
||||
@@ -1,8 +1,9 @@
|
||||
"""Qwen3.8-Flash-Next triton kernels must match their torch references, forward and backward."""
|
||||
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
register_cuda_ci(est_time=180, suite="stage-b-2-gpu-h200", labels=["miles-plugin"], hardware=["hopper", "blackwell"])
|
||||
register_rocm_ci(est_time=500, suite="nightly-stage-c-2-gpu-mi350", labels=["miles-plugin"])
|
||||
|
||||
import math
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
from tests.ci.ci_register import register_cuda_ci
|
||||
from tests.ci.ci_register import register_cuda_ci, register_rocm_ci
|
||||
|
||||
register_cuda_ci(est_time=180, suite="stage-b-2-gpu-h200", labels=["megatron"], hardware=["hopper"])
|
||||
register_rocm_ci(est_time=60, suite="nightly-stage-c-2-gpu-mi350", labels=["megatron"])
|
||||
|
||||
import gc
|
||||
import os
|
||||
|
||||
Reference in New Issue
Block a user