[CI] Re-organize and enable necessary end2end CI cases (#499)

Co-authored-by: Yusheng Su <radixark@ac-h200-user-3.tail134ba0.ts.net>
This commit is contained in:
Ethan (Yusheng) Su
2026-01-27 14:12:08 -08:00
committed by GitHub
co-authored by Yusheng Su
parent a8c8687ca7
commit 81ea4a82e4
38 changed files with 2400 additions and 50 deletions
+112 -9
View File
@@ -33,7 +33,7 @@ jobs:
options: >
--gpus all
--ipc=host
--shm-size=16g
--shm-size=32g
--ulimit memlock=-1
--ulimit stack=67108864
--memory=0
@@ -41,6 +41,9 @@ jobs:
-v /mnt/nvme0n1/miles_ci:/data/miles_ci
-v /mnt/nvme0n1/miles_ci/models:/root/models
-v /mnt/nvme0n1/miles_ci/datasets:/root/datasets
--privileged
--ulimit nofile=65535:65535
-v /tmp:/tmp
strategy:
fail-fast: false
matrix:
@@ -52,11 +55,26 @@ jobs:
GITHUB_COMMIT_NAME: ${{ github.sha }}_${{ github.event.pull_request.number || 'non-pr' }}
WANDB_API_KEY: ${{ secrets.WANDB_API_KEY }}
MILES_TEST_ENABLE_INFINITE_RUN: ${{ (github.event_name == 'workflow_dispatch' && github.event.inputs.infinite_run) || 'false' }}
MILES_TEST_USE_DEEPEP: ${{ matrix.info.use_deepep || '0' }}
MILES_TEST_USE_FP8_ROLLOUT: ${{ matrix.info.use_fp8_rollout || '0' }}
MILES_TEST_ENABLE_EVAL: ${{ matrix.info.enable_eval || '1' }}
steps:
- name: Checkout repository
uses: actions/checkout@v4
- name: Cleanup Ray processes
shell: bash
run: |
pkill -9 -f 'ray::' 2>/dev/null || true
pkill -9 -f raylet 2>/dev/null || true
pkill -9 -f gcs_server 2>/dev/null || true
pkill -9 -f 'ray-dashboard' 2>/dev/null || true
pkill -9 sglang 2>/dev/null || true
ray stop --force 2>/dev/null || true
rm -rf /tmp/ray/* 2>/dev/null || true
sleep 3
- name: Install
shell: bash
run: cd $GITHUB_WORKSPACE && pip install -e . --no-deps --break-system-packages
@@ -65,6 +83,84 @@ jobs:
shell: bash
run: python tests/ci/gpu_lock_exec.py --count ${{ matrix.info.num_gpus }} -- pytest tests/${{ matrix.info.test_file }}
- name: Post-test cleanup
if: always()
shell: bash
run: |
pkill -9 -f 'ray::' 2>/dev/null || true
pkill -9 -f raylet 2>/dev/null || true
ray stop --force 2>/dev/null || true
rm -rf /tmp/ray/* 2>/dev/null || true
unit-test:
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-unit-test'))
runs-on: self-hosted
container:
image: radixark/miles:latest
options: >
--gpus all
--ipc=host
--shm-size=32g
--ulimit memlock=-1
--ulimit stack=67108864
--memory=0
--memory-swap=0
-v /mnt/nvme0n1/miles_ci:/data/miles_ci
-v /mnt/nvme0n1/miles_ci/models:/root/models
-v /mnt/nvme0n1/miles_ci/datasets:/root/datasets
--privileged
--ulimit nofile=65535:65535
-v /tmp:/tmp
strategy:
fail-fast: false
matrix:
info: [{"num_gpus": 2, "test_file": "e2e/fsdp/test_qwen3_4B_fsdp_true_on_policy.py"}]
defaults:
run:
working-directory: ${{ github.workspace }}
env:
GITHUB_COMMIT_NAME: ${{ github.sha }}_${{ github.event.pull_request.number || 'non-pr' }}
WANDB_API_KEY: ${{ secrets.WANDB_API_KEY }}
MILES_TEST_ENABLE_INFINITE_RUN: ${{ (github.event_name == 'workflow_dispatch' && github.event.inputs.infinite_run) || 'false' }}
MILES_TEST_USE_DEEPEP: ${{ matrix.info.use_deepep || '0' }}
MILES_TEST_USE_FP8_ROLLOUT: ${{ matrix.info.use_fp8_rollout || '0' }}
MILES_TEST_ENABLE_EVAL: ${{ matrix.info.enable_eval || '1' }}
steps:
- name: Checkout repository
uses: actions/checkout@v4
- name: Cleanup Ray processes
shell: bash
run: |
pkill -9 -f 'ray::' 2>/dev/null || true
pkill -9 -f raylet 2>/dev/null || true
pkill -9 -f gcs_server 2>/dev/null || true
pkill -9 -f 'ray-dashboard' 2>/dev/null || true
pkill -9 sglang 2>/dev/null || true
ray stop --force 2>/dev/null || true
rm -rf /tmp/ray/* 2>/dev/null || true
sleep 3
- name: Install
shell: bash
run: cd $GITHUB_WORKSPACE && pip install -e . --no-deps --break-system-packages
- name: Execute
shell: bash
run: python tests/ci/gpu_lock_exec.py --count ${{ matrix.info.num_gpus }} -- python tests/${{ matrix.info.test_file }}
- name: Post-test cleanup
if: always()
shell: bash
run: |
pkill -9 -f 'ray::' 2>/dev/null || true
pkill -9 -f raylet 2>/dev/null || true
ray stop --force 2>/dev/null || true
rm -rf /tmp/ray/* 2>/dev/null || true
e2e-test-short:
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-short'))
runs-on: self-hosted
@@ -87,7 +183,7 @@ jobs:
strategy:
fail-fast: false
matrix:
info: [{"num_gpus": 4, "test_file": "test_qwen2.5_0.5B_gsm8k_async_short.py"}, {"num_gpus": 4, "test_file": "test_qwen2.5_0.5B_gsm8k_short.py"}, {"num_gpus": 2, "test_file": "test_qwen3_0.6B_fsdp_colocated_2xGPU.py"}]
info: [{"num_gpus": 4, "test_file": "e2e/short/test_qwen2.5_0.5B_gsm8k_async_short.py"}, {"num_gpus": 4, "test_file": "e2e/short/test_qwen2.5_0.5B_gsm8k_short.py"}, {"num_gpus": 2, "test_file": "e2e/short/test_qwen3_0.6B_fsdp_colocated_2xGPU.py"}]
defaults:
run:
working-directory: ${{ github.workspace }}
@@ -132,6 +228,7 @@ jobs:
ray stop --force 2>/dev/null || true
rm -rf /tmp/ray/* 2>/dev/null || true
e2e-test-fsdp:
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-fsdp'))
runs-on: self-hosted
@@ -154,7 +251,7 @@ jobs:
strategy:
fail-fast: false
matrix:
info: [{"num_gpus": 2, "test_file": "test_qwen3_4B_fsdp_true_on_policy.py"}, {"num_gpus": 8, "test_file": "test_qwen3_vl_4B_fsdp.py"}, {"num_gpus": 2, "test_file": "test_qwen3_0.6B_fsdp_distributed.py"}, {"num_gpus": 4, "test_file": "test_qwen3_0.6B_megatron_fsdp_align.py"}]
info: [{"num_gpus": 2, "test_file": "e2e/fsdp/test_qwen3_4B_fsdp_true_on_policy.py"}, {"num_gpus": 8, "test_file": "e2e/fsdp/test_qwen3_vl_4B_fsdp.py"}, {"num_gpus": 2, "test_file": "e2e/fsdp/test_qwen3_0.6B_fsdp_distributed.py"}, {"num_gpus": 4, "test_file": "e2e/fsdp/test_qwen3_0.6B_megatron_fsdp_align.py"}]
defaults:
run:
working-directory: ${{ github.workspace }}
@@ -199,6 +296,7 @@ jobs:
ray stop --force 2>/dev/null || true
rm -rf /tmp/ray/* 2>/dev/null || true
e2e-test-megatron:
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-megatron'))
runs-on: self-hosted
@@ -221,7 +319,7 @@ jobs:
strategy:
fail-fast: false
matrix:
info: [{"num_gpus": 8, "test_file": "test_quick_start_glm4_9B.py"}, {"num_gpus": 8, "test_file": "test_qwen3_30B_A3B.py", "use_deepep": "1", "use_fp8_rollout": "1"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "test_qwen3_30B_A3B_r3.py", "use_deepep": "1", "use_fp8_rollout": "1"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "test_qwen3_30B_A3B_r3.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ppo.py"}, {"num_gpus": 8, "test_file": "test_moonlight_16B_A3B.py"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "test_moonlight_16B_A3B_r3.py"}, {"num_gpus": 8, "test_file": "test_mimo_7B_mtp_only_grad.py"}]
info: [{"num_gpus": 8, "test_file": "e2e/megatron/test_quick_start_glm4_9B.py"}, {"num_gpus": 8, "test_file": "e2e/megatron/test_qwen3_30B_A3B.py", "use_deepep": "1", "use_fp8_rollout": "1"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "e2e/megatron/test_qwen3_30B_A3B_r3.py", "use_deepep": "1", "use_fp8_rollout": "1"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "e2e/megatron/test_qwen3_30B_A3B_r3.py"}, {"num_gpus": 8, "test_file": "e2e/megatron/test_qwen3_4B_ppo.py"}, {"num_gpus": 8, "test_file": "e2e/megatron/test_moonlight_16B_A3B.py"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "e2e/megatron/test_moonlight_16B_A3B_r3.py"}, {"num_gpus": 8, "test_file": "e2e/megatron/test_mimo_7B_mtp_only_grad.py"}]
defaults:
run:
working-directory: ${{ github.workspace }}
@@ -266,6 +364,7 @@ jobs:
ray stop --force 2>/dev/null || true
rm -rf /tmp/ray/* 2>/dev/null || true
e2e-test-precision:
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-precision'))
runs-on: self-hosted
@@ -288,7 +387,7 @@ jobs:
strategy:
fail-fast: false
matrix:
info: [{"num_gpus": 8, "test_file": "test_qwen3_0.6B_parallel_check.py"}, {"num_gpus": 4, "test_file": "test_qwen3_0.6B_megatron_fsdp_align.py"}]
info: [{"num_gpus": 8, "test_file": "e2e/precision/test_qwen3_0.6B_parallel_check.py"}, {"num_gpus": 4, "test_file": "e2e/precision/test_qwen3_0.6B_megatron_fsdp_align.py"}]
defaults:
run:
working-directory: ${{ github.workspace }}
@@ -333,6 +432,7 @@ jobs:
ray stop --force 2>/dev/null || true
rm -rf /tmp/ray/* 2>/dev/null || true
e2e-test-ckpt:
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-ckpt'))
runs-on: self-hosted
@@ -355,7 +455,7 @@ jobs:
strategy:
fail-fast: false
matrix:
info: [{"num_gpus": 8, "test_file": "test_qwen3_4B_ckpt.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ckpt.py --async-save"}]
info: [{"num_gpus": 8, "test_file": "e2e/ckpt/test_qwen3_4B_ckpt.py"}, {"num_gpus": 8, "test_file": "e2e/ckpt/test_qwen3_4B_ckpt.py --async-save"}]
defaults:
run:
working-directory: ${{ github.workspace }}
@@ -400,6 +500,7 @@ jobs:
ray stop --force 2>/dev/null || true
rm -rf /tmp/ray/* 2>/dev/null || true
e2e-test-long:
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-long'))
runs-on: self-hosted
@@ -422,7 +523,7 @@ jobs:
strategy:
fail-fast: false
matrix:
info: [{"num_gpus": 2, "test_file": "test_qwen2.5_0.5B_gsm8k.py"}, {"num_gpus": 2, "test_file": "test_qwen2.5_0.5B_gsm8k_async.py"}]
info: [{"num_gpus": 2, "test_file": "e2e/long/test_qwen2.5_0.5B_gsm8k.py"}, {"num_gpus": 2, "test_file": "e2e/long/test_qwen2.5_0.5B_gsm8k_async.py"}]
defaults:
run:
working-directory: ${{ github.workspace }}
@@ -467,11 +568,12 @@ jobs:
ray stop --force 2>/dev/null || true
rm -rf /tmp/ray/* 2>/dev/null || true
e2e-test-image:
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-image'))
runs-on: self-hosted
container:
image: radixark/miles-test:latest
image: radixark/miles:latest
options: >
--gpus all
--ipc=host
@@ -489,7 +591,7 @@ jobs:
strategy:
fail-fast: false
matrix:
info: [{"num_gpus": 4, "test_file": "test_qwen2.5_0.5B_gsm8k_async_short.py"}, {"num_gpus": 4, "test_file": "test_qwen2.5_0.5B_gsm8k_short.py"}, {"num_gpus": 2, "test_file": "test_qwen3_0.6B_fsdp_colocated_2xGPU.py"}, {"num_gpus": 2, "test_file": "test_qwen3_4B_fsdp_true_on_policy.py"}, {"num_gpus": 8, "test_file": "test_qwen3_vl_4B_fsdp.py"}, {"num_gpus": 2, "test_file": "test_qwen3_0.6B_fsdp_distributed.py"}, {"num_gpus": 8, "test_file": "test_quick_start_glm4_9B.py"}, {"num_gpus": 8, "test_file": "test_qwen3_30B_A3B.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ppo.py"}, {"num_gpus": 8, "test_file": "test_moonlight_16B_A3B.py"}, {"num_gpus": 8, "test_file": "test_mimo_7B_mtp_only_grad.py"}, {"num_gpus": 8, "test_file": "test_qwen3_0.6B_parallel_check.py"}, {"num_gpus": 4, "test_file": "test_qwen3_0.6B_megatron_fsdp_align.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ckpt.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ckpt.py --async-save"}, {"num_gpus": 2, "test_file": "test_qwen2.5_0.5B_gsm8k.py"}, {"num_gpus": 2, "test_file": "test_qwen2.5_0.5B_gsm8k_async.py"}]
info: [{"num_gpus": 4, "test_file": "e2e/image/test_qwen2.5_0.5B_gsm8k_async_short.py"}, {"num_gpus": 4, "test_file": "e2e/image/test_qwen2.5_0.5B_gsm8k_short.py"}, {"num_gpus": 2, "test_file": "e2e/image/test_qwen3_0.6B_fsdp_colocated_2xGPU.py"}, {"num_gpus": 2, "test_file": "e2e/image/test_qwen3_4B_fsdp_true_on_policy.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_qwen3_vl_4B_fsdp.py"}, {"num_gpus": 2, "test_file": "e2e/image/test_qwen3_0.6B_fsdp_distributed.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_quick_start_glm4_9B.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_qwen3_30B_A3B.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_qwen3_4B_ppo.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_moonlight_16B_A3B.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_mimo_7B_mtp_only_grad.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_qwen3_0.6B_parallel_check.py"}, {"num_gpus": 4, "test_file": "e2e/image/test_qwen3_0.6B_megatron_fsdp_align.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_qwen3_4B_ckpt.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_qwen3_4B_ckpt.py --async-save"}, {"num_gpus": 2, "test_file": "e2e/image/test_qwen2.5_0.5B_gsm8k.py"}, {"num_gpus": 2, "test_file": "e2e/image/test_qwen2.5_0.5B_gsm8k_async.py"}]
defaults:
run:
working-directory: ${{ github.workspace }}
@@ -533,3 +635,4 @@ jobs:
pkill -9 -f raylet 2>/dev/null || true
ray stop --force 2>/dev/null || true
rm -rf /tmp/ray/* 2>/dev/null || true
+56 -40
View File
@@ -5,78 +5,84 @@
{'test_file': 'fast', 'num_gpus': 0},
],
},
'unit-test': {
'label': 'run-unit-test',
'tests': [
{'test_file': 'e2e/fsdp/test_qwen3_4B_fsdp_true_on_policy.py', 'num_gpus': 2}
],
},
'e2e-test-short': {
'label': 'run-ci-short',
'tests': [
{'test_file': 'test_qwen2.5_0.5B_gsm8k_async_short.py', 'num_gpus': 4},
{'test_file': 'test_qwen2.5_0.5B_gsm8k_short.py', 'num_gpus': 4},
{'test_file': 'test_qwen3_0.6B_fsdp_colocated_2xGPU.py', 'num_gpus': 2},
{'test_file': 'e2e/short/test_qwen2.5_0.5B_gsm8k_async_short.py', 'num_gpus': 4},
{'test_file': 'e2e/short/test_qwen2.5_0.5B_gsm8k_short.py', 'num_gpus': 4},
{'test_file': 'e2e/short/test_qwen3_0.6B_fsdp_colocated_2xGPU.py', 'num_gpus': 2},
],
},
'e2e-test-fsdp': {
'label': 'run-ci-fsdp',
'tests': [
{'test_file': 'test_qwen3_4B_fsdp_true_on_policy.py', 'num_gpus': 2},
{'test_file': 'test_qwen3_vl_4B_fsdp.py', 'num_gpus': 8},
{'test_file': 'test_qwen3_0.6B_fsdp_distributed.py', 'num_gpus': 2},
{'test_file': 'test_qwen3_0.6B_megatron_fsdp_align.py', 'num_gpus': 4},
{'test_file': 'e2e/fsdp/test_qwen3_4B_fsdp_true_on_policy.py', 'num_gpus': 2},
{'test_file': 'e2e/fsdp/test_qwen3_vl_4B_fsdp.py', 'num_gpus': 8},
{'test_file': 'e2e/fsdp/test_qwen3_0.6B_fsdp_distributed.py', 'num_gpus': 2},
{'test_file': 'e2e/fsdp/test_qwen3_0.6B_megatron_fsdp_align.py', 'num_gpus': 4},
],
},
'e2e-test-megatron': {
'label': 'run-ci-megatron',
'tests': [
{'test_file': 'test_quick_start_glm4_9B.py', 'num_gpus': 8},
{'test_file': 'test_qwen3_30B_A3B.py', 'num_gpus': 8, 'use_deepep': '1', 'use_fp8_rollout': '1'},
{'test_file': 'test_qwen3_30B_A3B_r3.py', 'num_gpus': 8, 'use_deepep': '1', 'use_fp8_rollout': '1', 'enable_eval': '0'},
{'test_file': 'test_qwen3_30B_A3B_r3.py', 'num_gpus': 8, 'enable_eval': '0'},
{'test_file': 'test_qwen3_4B_ppo.py', 'num_gpus': 8},
{'test_file': 'test_moonlight_16B_A3B.py', 'num_gpus': 8},
{'test_file': 'test_moonlight_16B_A3B_r3.py', 'num_gpus': 8, 'enable_eval': '0'},
{'test_file': 'test_mimo_7B_mtp_only_grad.py', 'num_gpus': 8},
{'test_file': 'e2e/megatron/test_quick_start_glm4_9B.py', 'num_gpus': 8},
{'test_file': 'e2e/megatron/test_qwen3_30B_A3B.py', 'num_gpus': 8, 'use_deepep': '1', 'use_fp8_rollout': '1'},
{'test_file': 'e2e/megatron/test_qwen3_30B_A3B_r3.py', 'num_gpus': 8, 'use_deepep': '1', 'use_fp8_rollout': '1', 'enable_eval': '0'},
{'test_file': 'e2e/megatron/test_qwen3_30B_A3B_r3.py', 'num_gpus': 8, 'enable_eval': '0'},
{'test_file': 'e2e/megatron/test_qwen3_4B_ppo.py', 'num_gpus': 8},
{'test_file': 'e2e/megatron/test_moonlight_16B_A3B.py', 'num_gpus': 8},
{'test_file': 'e2e/megatron/test_moonlight_16B_A3B_r3.py', 'num_gpus': 8, 'enable_eval': '0'},
{'test_file': 'e2e/megatron/test_mimo_7B_mtp_only_grad.py', 'num_gpus': 8},
],
},
'e2e-test-precision': {
'label': 'run-ci-precision',
'tests': [
{'test_file': 'test_qwen3_0.6B_parallel_check.py', 'num_gpus': 8},
{'test_file': 'test_qwen3_0.6B_megatron_fsdp_align.py', 'num_gpus': 4},
{'test_file': 'e2e/precision/test_qwen3_0.6B_parallel_check.py', 'num_gpus': 8},
{'test_file': 'e2e/precision/test_qwen3_0.6B_megatron_fsdp_align.py', 'num_gpus': 4},
],
},
'e2e-test-ckpt': {
'label': 'run-ci-ckpt',
'tests': [
{'test_file': 'test_qwen3_4B_ckpt.py', 'num_gpus': 8},
{'test_file': 'test_qwen3_4B_ckpt.py --async-save', 'num_gpus': 8},
{'test_file': 'e2e/ckpt/test_qwen3_4B_ckpt.py', 'num_gpus': 8},
{'test_file': 'e2e/ckpt/test_qwen3_4B_ckpt.py --async-save', 'num_gpus': 8},
],
},
'e2e-test-long': {
'label': 'run-ci-long',
'tests': [
{'test_file': 'test_qwen2.5_0.5B_gsm8k.py', 'num_gpus': 2},
{'test_file': 'test_qwen2.5_0.5B_gsm8k_async.py', 'num_gpus': 2},
{'test_file': 'e2e/long/test_qwen2.5_0.5B_gsm8k.py', 'num_gpus': 2},
{'test_file': 'e2e/long/test_qwen2.5_0.5B_gsm8k_async.py', 'num_gpus': 2},
],
},
'e2e-test-image': {
'label': 'run-ci-image',
'image': 'radixark/miles-test:latest',
'image': 'radixark/miles:latest',
'tests': [
{'test_file': 'test_qwen2.5_0.5B_gsm8k_async_short.py', 'num_gpus': 4},
{'test_file': 'test_qwen2.5_0.5B_gsm8k_short.py', 'num_gpus': 4},
{'test_file': 'test_qwen3_0.6B_fsdp_colocated_2xGPU.py', 'num_gpus': 2},
{'test_file': 'test_qwen3_4B_fsdp_true_on_policy.py', 'num_gpus': 2},
{'test_file': 'test_qwen3_vl_4B_fsdp.py', 'num_gpus': 8},
{'test_file': 'test_qwen3_0.6B_fsdp_distributed.py', 'num_gpus': 2},
{'test_file': 'test_quick_start_glm4_9B.py', 'num_gpus': 8},
{'test_file': 'test_qwen3_30B_A3B.py', 'num_gpus': 8},
{'test_file': 'test_qwen3_4B_ppo.py', 'num_gpus': 8},
{'test_file': 'test_moonlight_16B_A3B.py', 'num_gpus': 8},
{'test_file': 'test_mimo_7B_mtp_only_grad.py', 'num_gpus': 8},
{'test_file': 'test_qwen3_0.6B_parallel_check.py', 'num_gpus': 8},
{'test_file': 'test_qwen3_0.6B_megatron_fsdp_align.py', 'num_gpus': 4},
{'test_file': 'test_qwen3_4B_ckpt.py', 'num_gpus': 8},
{'test_file': 'test_qwen3_4B_ckpt.py --async-save', 'num_gpus': 8},
{'test_file': 'test_qwen2.5_0.5B_gsm8k.py', 'num_gpus': 2},
{'test_file': 'test_qwen2.5_0.5B_gsm8k_async.py', 'num_gpus': 2},
{'test_file': 'e2e/image/test_qwen2.5_0.5B_gsm8k_async_short.py', 'num_gpus': 4},
{'test_file': 'e2e/image/test_qwen2.5_0.5B_gsm8k_short.py', 'num_gpus': 4},
{'test_file': 'e2e/image/test_qwen3_0.6B_fsdp_colocated_2xGPU.py', 'num_gpus': 2},
{'test_file': 'e2e/image/test_qwen3_4B_fsdp_true_on_policy.py', 'num_gpus': 2},
{'test_file': 'e2e/image/test_qwen3_vl_4B_fsdp.py', 'num_gpus': 8},
{'test_file': 'e2e/image/test_qwen3_0.6B_fsdp_distributed.py', 'num_gpus': 2},
{'test_file': 'e2e/image/test_quick_start_glm4_9B.py', 'num_gpus': 8},
{'test_file': 'e2e/image/test_qwen3_30B_A3B.py', 'num_gpus': 8},
{'test_file': 'e2e/image/test_qwen3_4B_ppo.py', 'num_gpus': 8},
{'test_file': 'e2e/image/test_moonlight_16B_A3B.py', 'num_gpus': 8},
{'test_file': 'e2e/image/test_mimo_7B_mtp_only_grad.py', 'num_gpus': 8},
{'test_file': 'e2e/image/test_qwen3_0.6B_parallel_check.py', 'num_gpus': 8},
{'test_file': 'e2e/image/test_qwen3_0.6B_megatron_fsdp_align.py', 'num_gpus': 4},
{'test_file': 'e2e/image/test_qwen3_4B_ckpt.py', 'num_gpus': 8},
{'test_file': 'e2e/image/test_qwen3_4B_ckpt.py --async-save', 'num_gpus': 8},
{'test_file': 'e2e/image/test_qwen2.5_0.5B_gsm8k.py', 'num_gpus': 2},
{'test_file': 'e2e/image/test_qwen2.5_0.5B_gsm8k_async.py', 'num_gpus': 2},
],
},
} %>
@@ -160,4 +166,14 @@ jobs:
- name: Execute
shell: bash
run: python tests/ci/gpu_lock_exec.py --count ${{ matrix.info.num_gpus }} -- << config.test_executor | default('python') >> tests/${{ matrix.info.test_file }}
<% endfor %>
- name: Post-test cleanup
if: always()
shell: bash
run: |
pkill -9 -f 'ray::' 2>/dev/null || true
pkill -9 -f raylet 2>/dev/null || true
ray stop --force 2>/dev/null || true
rm -rf /tmp/ray/* 2>/dev/null || true
<% endfor %>
+9 -1
View File
@@ -136,7 +136,15 @@ def log_rollout_data(
# NOTE: Here we have to do the clone().detach(), otherwise the tensor will be
# modified in place and will cause problem for the next rollout.
val = torch.cat(val).clone().detach()
if key in ["log_probs", "ref_log_probs", "rollout_log_probs", "returns", "advantages", "values"]:
if key in [
"log_probs",
"ref_log_probs",
"rollout_log_probs",
"returns",
"advantages",
"values",
"entropy",
]:
sum_of_sample_mean = get_sum_of_sample_mean(
total_lengths,
response_lengths,
@@ -0,0 +1,106 @@
import os
import miles.utils.external_utils.command_utils as U
MODEL_NAME = "Qwen3-0.6B"
FEW_GPU = U.get_bool_env_var("MILES_TEST_FEW_GPU", "1")
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"huggingface-cli download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.hf_download_dataset("zhuzilin/gsm8k")
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} "
rollout_args = (
"--prompt-data /root/datasets/gsm8k/train.parquet "
"--input-key messages "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type math "
# NOTE cannot be exactly multiple of eval-interval, since async causes some offsets
f"--num-rollout {3000 if U.get_env_enable_infinite_run() else 65} "
"--rollout-batch-size 32 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 1024 "
"--rollout-temperature 1 "
"--over-sampling-batch-size 64 "
"--dynamic-sampling-filter-path miles.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std "
"--global-batch-size 256 "
)
eval_args = (
"--eval-interval 20 "
"--eval-prompt-data gsm8k /root/datasets/gsm8k/test.parquet "
"--n-samples-per-eval-prompt 1 "
"--eval-max-response-len 1024 "
"--eval-top-k 1 "
)
grpo_args = (
"--advantage-estimator grpo "
# "--use-kl-loss "
"--kl-loss-coef 0.00 "
"--kl-loss-type low_var_kl "
"--kl-coef 0.00 "
"--entropy-coef 0.00 "
"--eps-clip 0.2 "
"--eps-clip-high 0.28 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = "--rollout-num-gpus-per-engine 1 " "--sglang-enable-metrics "
misc_args = (
"--actor-num-nodes 1 "
f"--actor-num-gpus-per-node {1 if FEW_GPU else 2} "
f"--rollout-num-gpus {1 if FEW_GPU else 2} "
"--train-backend fsdp "
)
ci_args = (
"--ci-test "
"--ci-disable-kl-checker "
"--ci-metric-checker-key eval/gsm8k "
"--ci-metric-checker-threshold 0.71 " # loose threshold at 60 step
)
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{grpo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{eval_args} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
U.execute_train(
train_args=train_args,
num_gpus_per_node=2 if FEW_GPU else 4,
megatron_model_type=None,
train_script="train_async.py",
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
if __name__ == "__main__":
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
@@ -0,0 +1,155 @@
import os
import miles.utils.external_utils.command_utils as U
MODEL_NAME = "Qwen3-0.6B"
MODEL_TYPE = "qwen3-0.6B"
NUM_GPUS = 4
CP_SIZE = 1
MEGATRON_TP_SIZE = 1
MEGATRON_PP_SIZE = 1
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.hf_download_dataset("zhuzilin/dapo-math-17k")
U.convert_checkpoint(
model_name=MODEL_NAME,
megatron_model_type=MODEL_TYPE,
num_gpus_per_node=NUM_GPUS,
dir_dst="/root/models",
)
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/"
rollout_args = (
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
"--input-key prompt "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type deepscaler "
"--num-rollout 1 "
"--rollout-batch-size 8 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 8192 "
"--rollout-temperature 1 "
"--global-batch-size 64 "
"--use-dynamic-batch-size "
"--max-tokens-per-gpu 8192 "
)
ppo_args = (
"--advantage-estimator grpo "
"--kl-loss-coef 0.00 "
"--kl-loss-type k1 "
"--kl-coef 0.00 "
"--entropy-coef 0.00 "
"--eps-clip 4e-4 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = (
"--rollout-num-gpus-per-engine 1 " "--sglang-chunked-prefill-size 4096 " "--sglang-mem-fraction-static 0.75 "
)
ci_args = "--ci-test "
misc_args = "--actor-num-nodes 1 " "--colocate " f"--actor-num-gpus-per-node {NUM_GPUS} "
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{ppo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
debug_data_path = "test_rollout_data_megatron_fsdp_align.pt"
grad_norm_path = "grad_norm_fsdp.pt"
fsdp_args = (
"--train-backend fsdp "
"--attn-implementation flash_attention_2 "
"--gradient-checkpointing "
f"--context-parallel-size {CP_SIZE} "
f"--update-weight-buffer-size {512 * 1024 * 1024} "
"""--train-env-vars '{"PYTORCH_CUDA_ALLOC_CONF":"expandable_segments:True"}' """
)
try:
U.execute_train(
train_args=train_args + (f"{fsdp_args}" f"--save-debug-rollout-data {debug_data_path} "),
num_gpus_per_node=NUM_GPUS,
megatron_model_type=None,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
U.execute_train(
train_args=train_args
+ (
f"{fsdp_args}"
f"--load-debug-rollout-data {debug_data_path} "
f"--ci-save-grad-norm {grad_norm_path} "
"--debug-train-only "
),
num_gpus_per_node=NUM_GPUS,
megatron_model_type=None,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
U.execute_train(
train_args=train_args
+ (
f"--ref-load /root/models/{MODEL_NAME}_torch_dist "
f"--tensor-model-parallel-size {MEGATRON_TP_SIZE} "
"--sequence-parallel "
f"--pipeline-model-parallel-size {MEGATRON_PP_SIZE} "
f"--context-parallel-size {CP_SIZE} "
"--expert-model-parallel-size 1 "
"--expert-tensor-parallel-size 1 "
"--recompute-granularity full "
"--recompute-method uniform "
"--recompute-num-layers 1 "
"--train-memory-margin-bytes 3221225472 "
f"--load-debug-rollout-data {debug_data_path} "
f"--ci-load-grad-norm {grad_norm_path} "
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
"--attention-backend flash "
"--debug-train-only "
),
num_gpus_per_node=NUM_GPUS,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
megatron_model_type=MODEL_TYPE,
)
finally:
if os.path.exists(grad_norm_path):
os.remove(grad_norm_path)
if os.path.exists(debug_data_path):
os.remove(debug_data_path)
if __name__ == "__main__":
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
+138
View File
@@ -0,0 +1,138 @@
import os
from argparse import ArgumentParser
import miles.utils.external_utils.command_utils as U
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
TIGHT_HOST_MEMORY = bool(int(os.environ.get("MILES_TEST_TIGHT_HOST_MEMORY", "1")))
MODEL_NAME = "Qwen3-4B"
MODEL_TYPE = "qwen3-4B"
NUM_GPUS = 8
parser = ArgumentParser()
parser.add_argument("--async-save", action="store_true", help="Whether to test async save/load.")
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.exec_command(f"rm -rf /root/models/{MODEL_NAME}_miles")
U.hf_download_dataset("zhuzilin/dapo-math-17k")
U.hf_download_dataset("zhuzilin/aime-2024")
U.convert_checkpoint(
model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS, dir_dst="/root/models"
)
def execute(mode: str = ""):
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}_torch_dist "
if mode == "save":
ckpt_args += f"--save /root/models/{MODEL_NAME}_miles "
ckpt_args += "--save-interval 2 "
elif mode == "async_save":
ckpt_args += f"--save /root/models/{MODEL_NAME}_miles "
ckpt_args += "--save-interval 2 "
ckpt_args += "--async-save "
elif mode == "load":
ckpt_args += f"--load /root/models/{MODEL_NAME}_miles "
ckpt_args += "--ckpt-step 1 "
rollout_args = (
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
"--input-key prompt "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type deepscaler "
"--num-rollout 3 "
"--rollout-batch-size 4 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 1024 "
"--rollout-temperature 0.8 "
"--global-batch-size 32 "
"--balance-data "
)
perf_args = (
"--tensor-model-parallel-size 2 "
"--sequence-parallel "
"--pipeline-model-parallel-size 1 "
"--context-parallel-size 2 "
"--recompute-granularity full "
"--recompute-method uniform "
"--recompute-num-layers 1 "
"--use-dynamic-batch-size "
f"--max-tokens-per-gpu {2048 if TIGHT_HOST_MEMORY else 16384} "
)
ppo_args = (
"--advantage-estimator grpo "
"--kl-loss-coef 0.00 "
"--kl-loss-type k1 "
"--kl-coef 0.00 "
"--entropy-coef 0.00 "
"--eps-clip 0.2 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
"--optimizer-cpu-offload "
"--overlap-cpu-optimizer-d2h-h2d "
"--use-precision-aware-optimizer "
)
sglang_args = "--rollout-num-gpus-per-engine 2 --sglang-mem-fraction-static 0.8 --sglang-cuda-graph-bs 1 2 4 8 16 "
ci_args = "--ci-test "
misc_args = (
# default dropout in megatron is 0.1
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
# should be good for model performance
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
# need to comment this when using model with MLA
"--attention-backend flash "
"--actor-num-nodes 1 "
"--actor-num-gpus-per-node 8 "
"--colocate "
)
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{ppo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{perf_args} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
U.execute_train(
train_args=train_args,
num_gpus_per_node=NUM_GPUS,
megatron_model_type=MODEL_TYPE,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
if __name__ == "__main__":
args = parser.parse_args()
# TODO also use typer
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute("save" if not args.async_save else "async_save")
execute("load")
@@ -0,0 +1,113 @@
import os
import miles.utils.external_utils.command_utils as U
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
NUM_GPUS = 2
MODEL_NAME = "Qwen3-4B"
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.hf_download_dataset("zhuzilin/dapo-math-17k")
U.hf_download_dataset("zhuzilin/aime-2024")
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} "
rollout_args = (
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
"--input-key prompt "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type math "
"--num-rollout 3 "
"--rollout-batch-size 8 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 4096 "
"--rollout-temperature 1 "
"--global-batch-size 32 "
)
eval_args = (
f"{'--eval-interval 20 ' if ENABLE_EVAL else ''}"
"--eval-prompt-data aime /root/datasets/aime-2024/aime-2024.jsonl "
"--n-samples-per-eval-prompt 1 "
"--eval-max-response-len 4096 "
"--eval-top-p 0.7 "
)
fsdp_args = "--train-backend fsdp " "--update-weight-buffer-size 536870912 "
grpo_args = (
"--advantage-estimator grpo "
"--kl-loss-coef 0.00 "
"--kl-loss-type low_var_kl "
"--kl-coef 0.00 "
"--entropy-coef 0.00 "
"--eps-clip 0.2 "
"--eps-clip-high 0.28 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = (
"--rollout-num-gpus-per-engine 1 "
"--sglang-decode-log-interval 1000 "
"--sglang-enable-metrics "
"--sglang-enable-deterministic-inference "
"--sglang-rl-on-policy-target fsdp "
"--sglang-attention-backend fa3 "
"--attn-implementation flash_attention_3 "
"--deterministic-mode "
"--true-on-policy-mode "
)
ci_args = "--ci-test "
misc_args = "--actor-num-nodes 1 " f"--actor-num-gpus-per-node {NUM_GPUS} " "--colocate "
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{grpo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{fsdp_args} "
f"{eval_args} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
extra_env_vars = {
"NCCL_ALGO": "allreduce:tree",
"NVTE_ALLOW_NONDETERMINISTIC_ALGO": "0",
"CUBLAS_WORKSPACE_CONFIG": ":4096:8",
"CUDA_DEVICE_MAX_CONNECTIONS": "1",
"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1",
}
U.execute_train(
train_args=train_args,
num_gpus_per_node=NUM_GPUS,
megatron_model_type=None,
extra_env_vars=extra_env_vars,
)
if __name__ == "__main__":
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
+112
View File
@@ -0,0 +1,112 @@
import os
import miles.utils.external_utils.command_utils as U
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
NUM_GPUS = 8
MODEL_NAME = "Qwen3-VL-4B-Instruct"
DATASET_NAME = "chenhegu/geo3k_imgurl"
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.hf_download_dataset(DATASET_NAME)
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} "
rollout_args = (
"--prompt-data /root/datasets/geo3k_imgurl/train.parquet "
"--input-key problem "
"--label-key answer "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type math "
"--num-rollout 3 "
"--rollout-batch-size 8 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 4096 "
"--rollout-temperature 1 "
"--global-batch-size 32 "
)
# multimodal keys required for vlm datasets
multimodal_args = '--multimodal-keys \'{"image": "images"}\' '
eval_args = (
f"{'--eval-interval 20 ' if ENABLE_EVAL else ''}"
"--eval-prompt-data geo3k /root/datasets/geo3k_imgurl/test.parquet "
"--n-samples-per-eval-prompt 1 "
"--eval-max-response-len 4096 "
)
fsdp_args = "--train-backend fsdp " "--gradient-checkpointing " "--update-weight-buffer-size 536870912 "
grpo_args = (
"--advantage-estimator grpo "
"--kl-loss-coef 0.00 "
"--kl-loss-type low_var_kl "
"--kl-coef 0.00 "
"--entropy-coef 0.00 "
"--eps-clip 0.2 "
"--eps-clip-high 0.28 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = (
"--rollout-num-gpus-per-engine 1 "
"--sglang-mem-fraction-static 0.6 "
"--sglang-decode-log-interval 1000 "
"--sglang-enable-metrics "
"--sglang-attention-backend fa3 "
"--attn-implementation flash_attention_3 "
)
ci_args = "--ci-test "
misc_args = "--actor-num-nodes 1 " f"--actor-num-gpus-per-node {NUM_GPUS} " "--colocate "
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{multimodal_args} "
f"{optimizer_args} "
f"{grpo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{fsdp_args} "
f"{eval_args} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
extra_env_vars = {
"CUDA_DEVICE_MAX_CONNECTIONS": "1",
"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1",
}
U.execute_train(
train_args=train_args,
num_gpus_per_node=NUM_GPUS,
megatron_model_type=None,
extra_env_vars=extra_env_vars,
)
if __name__ == "__main__":
prepare()
os.environ.pop("http_proxy", None)
os.environ.pop("https_proxy", None)
os.environ.pop("HTTP_PROXY", None)
os.environ.pop("HTTPS_PROXY", None)
execute()
+131
View File
@@ -0,0 +1,131 @@
import os
import miles.utils.external_utils.command_utils as U
FEW_GPU = U.get_bool_env_var("MILES_TEST_FEW_GPU", "1")
TIGHT_DEVICE_MEMORY = U.get_bool_env_var("MILES_TEST_TIGHT_DEVICE_MEMORY", "1")
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
NUM_GPUS = 2 if FEW_GPU else 4
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"huggingface-cli download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.hf_download_dataset("zhuzilin/gsm8k")
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}/ "
rollout_args = (
"--prompt-data /root/datasets/gsm8k/train.parquet "
"--input-key messages "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type math "
f"--num-rollout {3000 if U.get_env_enable_infinite_run() else 250} "
"--rollout-batch-size 32 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 1024 "
"--rollout-temperature 1 "
"--over-sampling-batch-size 64 "
"--dynamic-sampling-filter-path miles.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std "
"--global-batch-size 256 "
)
eval_args = (
"--eval-interval 20 "
"--eval-prompt-data gsm8k /root/datasets/gsm8k/test.parquet "
"--n-samples-per-eval-prompt 1 "
"--eval-max-response-len 1024 "
"--eval-top-k 1 "
)
perf_args = (
"--tensor-model-parallel-size 1 "
"--sequence-parallel "
"--pipeline-model-parallel-size 1 "
"--context-parallel-size 1 "
"--expert-model-parallel-size 1 "
"--expert-tensor-parallel-size 1 "
# "--micro-batch-size 1 "
"--use-dynamic-batch-size "
"--max-tokens-per-gpu 9216 "
)
grpo_args = (
"--advantage-estimator grpo "
"--use-kl-loss "
"--kl-loss-coef 0.00 "
"--kl-loss-type low_var_kl "
"--entropy-coef 0.00 "
"--eps-clip 0.2 "
"--eps-clip-high 0.28 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = (
"--rollout-num-gpus-per-engine 1 "
f"--sglang-mem-fraction-static {0.6 if TIGHT_DEVICE_MEMORY else 0.7} "
"--sglang-enable-metrics "
)
ci_args = (
"--ci-test "
"--ci-disable-kl-checker "
"--ci-metric-checker-key eval/gsm8k "
"--ci-metric-checker-threshold 0.55 " # loose threshold at 250 step
)
misc_args = (
# default dropout in megatron is 0.1
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
# should be good for model performance
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
# need to comment this when using model with MLA
"--attention-backend flash "
"--actor-num-nodes 1 "
f"--actor-num-gpus-per-node {2 if FEW_GPU else 4} "
"--colocate "
"--megatron-to-hf-mode bridge "
)
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{grpo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{perf_args} "
f"{eval_args} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
U.execute_train(
train_args=train_args,
num_gpus_per_node=NUM_GPUS,
megatron_model_type=MODEL_TYPE,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
if __name__ == "__main__":
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
@@ -0,0 +1,131 @@
import os
import miles.utils.external_utils.command_utils as U
FEW_GPU = U.get_bool_env_var("MILES_TEST_FEW_GPU", "1")
TIGHT_DEVICE_MEMORY = U.get_bool_env_var("MILES_TEST_TIGHT_DEVICE_MEMORY", "1")
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
NUM_GPUS = 2 if FEW_GPU else 4
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"huggingface-cli download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.hf_download_dataset("zhuzilin/gsm8k")
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}/ "
rollout_args = (
"--prompt-data /root/datasets/gsm8k/train.parquet "
"--input-key messages "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type math "
f"--num-rollout {3000 if U.get_env_enable_infinite_run() else 250} "
"--rollout-batch-size 32 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 1024 "
"--rollout-temperature 1 "
"--over-sampling-batch-size 64 "
"--dynamic-sampling-filter-path miles.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std "
"--global-batch-size 256 "
)
eval_args = (
"--eval-interval 20 "
"--eval-prompt-data gsm8k /root/datasets/gsm8k/test.parquet "
"--n-samples-per-eval-prompt 1 "
"--eval-max-response-len 1024 "
"--eval-top-k 1 "
)
perf_args = (
"--tensor-model-parallel-size 1 "
"--sequence-parallel "
"--pipeline-model-parallel-size 1 "
"--context-parallel-size 1 "
"--expert-model-parallel-size 1 "
"--expert-tensor-parallel-size 1 "
# "--micro-batch-size 1 "
"--use-dynamic-batch-size "
"--max-tokens-per-gpu 9216 "
)
grpo_args = (
"--advantage-estimator grpo "
"--use-kl-loss "
"--kl-loss-coef 0.00 "
"--kl-loss-type low_var_kl "
"--entropy-coef 0.00 "
"--eps-clip 0.2 "
"--eps-clip-high 0.28 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = (
"--rollout-num-gpus-per-engine 1 "
f"--sglang-mem-fraction-static {0.6 if TIGHT_DEVICE_MEMORY else 0.7} "
"--sglang-enable-metrics "
)
ci_args = (
"--ci-test "
"--ci-disable-kl-checker "
"--ci-metric-checker-key eval/gsm8k "
"--ci-metric-checker-threshold 0.55 " # loose threshold at 250 step
)
misc_args = (
# default dropout in megatron is 0.1
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
# should be good for model performance
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
# need to comment this when using model with MLA
"--attention-backend flash "
"--actor-num-nodes 1 "
f"--actor-num-gpus-per-node {1 if FEW_GPU else 2} "
f"--rollout-num-gpus {1 if FEW_GPU else 2} "
"--megatron-to-hf-mode bridge "
)
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{grpo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{perf_args} "
f"{eval_args} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
U.execute_train(
train_args=train_args,
num_gpus_per_node=NUM_GPUS,
megatron_model_type=MODEL_TYPE,
train_script="train_async.py",
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
if __name__ == "__main__":
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
@@ -0,0 +1,147 @@
"""End-to-end test for MTP-only gradient verification.
This test verifies that when MTP training is enabled and all outputs are truncated
(due to very short max response length), only MTP parameters receive non-zero
gradients while all other model parameters have zero gradients.
This validates that the MTP loss computation correctly isolates gradient flow
to only the MTP layers when the main model loss is zero (due to truncation).
"""
import os
import miles.utils.external_utils.command_utils as U
MODEL_NAME = "MiMo-7B-RL"
MODEL_TYPE = "mimo-7B-rl"
NUM_GPUS = 8
def prepare():
"""Download model and convert checkpoint with MTP layers."""
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"hf download XiaomiMiMo/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.hf_download_dataset("zhuzilin/dapo-math-17k")
# Convert checkpoint with MTP layers enabled
U.convert_checkpoint(
model_name=MODEL_NAME,
megatron_model_type=MODEL_TYPE,
num_gpus_per_node=NUM_GPUS,
extra_args=" --mtp-num-layers 1",
dir_dst="/root/models",
)
def execute():
"""Run training with MTP enabled and very short output length to cause truncation."""
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}_torch_dist "
# Use very short rollout-max-response-len to ensure all outputs are truncated
# This should result in zero loss for the main model, leaving only MTP loss
rollout_args = (
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
"--input-key prompt "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type deepscaler "
"--num-rollout 1 "
"--rollout-batch-size 4 "
"--n-samples-per-prompt 2 "
# Very short max response length to cause all outputs to be truncated
"--rollout-max-response-len 128 "
"--rollout-temperature 0.8 "
"--global-batch-size 8 "
)
perf_args = (
"--tensor-model-parallel-size 2 "
"--sequence-parallel "
"--pipeline-model-parallel-size 1 "
"--context-parallel-size 1 "
"--recompute-granularity full "
"--recompute-method uniform "
"--recompute-num-layers 1 "
"--use-dynamic-batch-size "
"--max-tokens-per-gpu 4096 "
)
grpo_args = (
"--advantage-estimator grpo "
"--kl-loss-coef 0.00 "
"--kl-loss-type low_var_kl "
"--kl-coef 0.00 "
"--entropy-coef 0.00 "
"--eps-clip 0.2 "
"--eps-clip-high 0.28 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = (
"--rollout-num-gpus-per-engine 2 "
"--rollout-num-gpus 8 "
"--sglang-mem-fraction-static 0.8 "
"--sglang-enable-metrics "
"--sglang-speculative-algorithm EAGLE "
"--sglang-speculative-num-steps 2 "
"--sglang-speculative-eagle-topk 1 "
"--sglang-speculative-num-draft-tokens 3 "
)
# Enable MTP training with loss scaling
mtp_args = "--mtp-num-layers 1 " "--enable-mtp-training " "--mtp-loss-scaling-factor 0.2 "
ci_args = (
"--ci-test "
"--ci-disable-kl-checker "
# MTP grad check is automatically triggered when ci_test and enable_mtp_training are both set
)
misc_args = (
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
"--attention-backend flash "
"--actor-num-nodes 1 "
"--actor-num-gpus-per-node 8 "
"--colocate "
)
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{grpo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{perf_args} "
f"{sglang_args} "
f"{mtp_args} "
f"{ci_args} "
f"{misc_args} "
)
U.execute_train(
train_args=train_args,
num_gpus_per_node=NUM_GPUS,
megatron_model_type=MODEL_TYPE,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
if __name__ == "__main__":
prepare()
# Remove proxy settings that might interfere with local operations
for key in ["http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"]:
os.environ.pop(key, None)
execute()
@@ -0,0 +1,124 @@
import os
import miles.utils.external_utils.command_utils as U
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
TIGHT_HOST_MEMORY = bool(int(os.environ.get("MILES_TEST_TIGHT_HOST_MEMORY", "1")))
MODEL_NAME = "Moonlight-16B-A3B-Instruct"
MODEL_TYPE = "moonlight"
NUM_GPUS = 8
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(
"hf download moonshotai/Moonlight-16B-A3B-Instruct --local-dir /root/models/Moonlight-16B-A3B-Instruct"
)
U.hf_download_dataset("zhuzilin/dapo-math-17k")
U.hf_download_dataset("zhuzilin/aime-2024")
U.convert_checkpoint(model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS)
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} " f"--ref-load /root/{MODEL_NAME}_torch_dist "
rollout_args = (
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
"--input-key prompt "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type math "
"--num-rollout 3 "
"--rollout-batch-size 8 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 4096 "
"--rollout-temperature 1 "
"--global-batch-size 32 "
)
eval_args = (
f"{'--eval-interval 20 ' if ENABLE_EVAL else ''}"
"--eval-prompt-data aime /root/datasets/aime-2024/aime-2024.jsonl "
"--n-samples-per-eval-prompt 1 "
"--eval-max-response-len 4096 "
"--eval-top-k 1 "
)
perf_args = (
"--tensor-model-parallel-size 2 "
"--sequence-parallel "
"--pipeline-model-parallel-size 1 "
"--context-parallel-size 2 "
"--expert-model-parallel-size 8 "
"--expert-tensor-parallel-size 1 "
"--recompute-granularity full "
"--recompute-method uniform "
"--recompute-num-layers 1 "
"--use-dynamic-batch-size "
f"--max-tokens-per-gpu {2048 if TIGHT_HOST_MEMORY else 2048} "
)
grpo_args = (
"--advantage-estimator gspo "
f"{'' if TIGHT_HOST_MEMORY else '--use-kl-loss '}"
"--kl-loss-coef 0.00 "
"--kl-loss-type low_var_kl "
"--kl-coef 0.00 "
"--entropy-coef 0.00 "
"--eps-clip 4e-4 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = (
"--rollout-num-gpus-per-engine 2 " "--sglang-mem-fraction-static 0.8 " "--sglang-max-running-requests 512 "
)
ci_args = "--ci-test "
misc_args = (
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
"--attention-backend flash "
"--actor-num-nodes 1 "
"--actor-num-gpus-per-node 8 "
"--colocate "
)
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{grpo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{perf_args} "
f"{eval_args} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
U.execute_train(
train_args=train_args,
num_gpus_per_node=NUM_GPUS,
megatron_model_type=MODEL_TYPE,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
if __name__ == "__main__":
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
@@ -0,0 +1,127 @@
import os
import miles.utils.external_utils.command_utils as U
ENABLE_EVAL = U.get_bool_env_var("MILES_TEST_ENABLE_EVAL", "1")
TIGHT_DEVICE_MEMORY = U.get_bool_env_var("MILES_TEST_TIGHT_DEVICE_MEMORY", "1")
MODEL_NAME = "GLM-Z1-9B-0414"
MODEL_TYPE = "glm4-9B"
NUM_GPUS = 8
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command("hf download zai-org/GLM-Z1-9B-0414 --local-dir /root/models/GLM-Z1-9B-0414")
U.hf_download_dataset("zhuzilin/dapo-math-17k")
U.hf_download_dataset("zhuzilin/aime-2024")
U.convert_checkpoint(model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS)
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/{MODEL_NAME}_torch_dist "
rollout_args = (
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
"--input-key prompt "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type deepscaler "
"--num-rollout 3 "
"--rollout-batch-size 8 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 8192 "
"--rollout-temperature 1 "
"--global-batch-size 32 "
"--balance-data "
)
eval_args = (
f"{'--eval-interval 20 ' if ENABLE_EVAL else ''}"
"--eval-prompt-data aime24 /root/datasets/aime-2024/aime-2024.jsonl "
"--n-samples-per-eval-prompt 1 "
"--eval-max-response-len 16384 "
"--eval-top-k 1 "
)
perf_args = (
"--tensor-model-parallel-size 2 "
"--sequence-parallel "
"--pipeline-model-parallel-size 1 "
"--context-parallel-size 2 "
"--expert-model-parallel-size 1 "
"--expert-tensor-parallel-size 1 "
"--recompute-granularity full "
"--recompute-method uniform "
"--recompute-num-layers 1 "
"--use-dynamic-batch-size "
f"--max-tokens-per-gpu {2048 if TIGHT_DEVICE_MEMORY else 4608} "
)
grpo_args = (
"--advantage-estimator grpo "
"--use-kl-loss "
"--kl-loss-coef 0.00 "
"--kl-loss-type low_var_kl "
"--entropy-coef 0.00 "
"--eps-clip 0.2 "
"--eps-clip-high 0.28 "
"--use-tis "
"--calculate-per-token-loss "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = "--rollout-num-gpus-per-engine 2 " "--use-miles-router "
ci_args = "--ci-test "
misc_args = (
# default dropout in megatron is 0.1
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
# should be good for model performance
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
# need to comment this when using model with MLA
"--attention-backend flash "
"--actor-num-nodes 1 "
"--actor-num-gpus-per-node 4 "
"--rollout-num-gpus 4 "
)
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{grpo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{perf_args} "
f"{eval_args} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
U.execute_train(
train_args=train_args,
num_gpus_per_node=NUM_GPUS,
megatron_model_type=MODEL_TYPE,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
if __name__ == "__main__":
# TODO also use typer
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
+151
View File
@@ -0,0 +1,151 @@
import os
import miles.utils.external_utils.command_utils as U
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
TIGHT_HOST_MEMORY = bool(int(os.environ.get("MILES_TEST_TIGHT_HOST_MEMORY", "1")))
USE_DEEPEP = bool(int(os.environ.get("MILES_TEST_USE_DEEPEP", "1")))
USE_FP8_ROLLOUT = bool(int(os.environ.get("MILES_TEST_USE_FP8_ROLLOUT", "1")))
MODEL_NAME = "Qwen3-30B-A3B"
MODEL_TYPE = "qwen3-30B-A3B"
NUM_GPUS = 8
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command("hf download Qwen/Qwen3-30B-A3B --local-dir /root/models/Qwen3-30B-A3B")
U.exec_command("hf download Qwen/Qwen3-30B-A3B-FP8 --local-dir /root/models/Qwen3-30B-A3B-FP8")
U.hf_download_dataset("zhuzilin/dapo-math-17k")
U.hf_download_dataset("zhuzilin/aime-2024")
U.convert_checkpoint(model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS)
def execute():
if USE_FP8_ROLLOUT:
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}-FP8 " f"--ref-load /root/{MODEL_NAME}_torch_dist "
else:
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} " f"--ref-load /root/{MODEL_NAME}_torch_dist "
rollout_args = (
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
"--input-key prompt "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type deepscaler "
"--num-rollout 3 "
"--rollout-batch-size 8 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 8192 "
"--rollout-temperature 1 "
"--global-batch-size 32 "
"--balance-data "
)
eval_args = (
f"{'--eval-interval 20 ' if ENABLE_EVAL else ''}"
"--eval-prompt-data aime24 /root/datasets/aime-2024/aime-2024.jsonl "
"--n-samples-per-eval-prompt 1 "
"--eval-max-response-len 16384 "
"--eval-top-k 1 "
)
perf_args = (
"--tensor-model-parallel-size 4 "
"--sequence-parallel "
"--pipeline-model-parallel-size 1 "
"--context-parallel-size 2 "
"--expert-model-parallel-size 8 "
"--expert-tensor-parallel-size 1 "
"--recompute-granularity full "
"--recompute-method uniform "
"--recompute-num-layers 1 "
"--use-dynamic-batch-size "
f"--max-tokens-per-gpu {2048 if TIGHT_HOST_MEMORY else 16384} "
)
grpo_args = (
"--advantage-estimator gspo "
f"{'' if TIGHT_HOST_MEMORY else '--use-kl-loss '}"
"--kl-loss-coef 0.00 "
"--kl-loss-type low_var_kl "
"--kl-coef 0.00 "
"--entropy-coef 0.00 "
"--eps-clip 4e-4 "
"--use-tis "
"--use-routing-replay "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
"--optimizer-cpu-offload "
"--overlap-cpu-optimizer-d2h-h2d "
"--use-precision-aware-optimizer "
)
sglang_args = (
"--rollout-num-gpus-per-engine 8 "
"--sglang-mem-fraction-static 0.8 "
"--sglang-max-running-requests 512 "
"--sglang-enable-metrics "
)
if USE_DEEPEP:
sglang_args += "--sglang-moe-a2a-backend deepep --sglang-deepep-mode auto "
ci_args = "--ci-test "
misc_args = (
# default dropout in megatron is 0.1
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
# should be good for model performance
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
# need to comment this when using model with MLA
"--attention-backend flash "
"--actor-num-nodes 1 "
"--actor-num-gpus-per-node 8 "
"--colocate "
)
if USE_DEEPEP:
misc_args += "--moe-token-dispatcher-type flex --moe-enable-deepep "
else:
misc_args += "--moe-token-dispatcher-type alltoall "
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{grpo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{perf_args} "
f"{eval_args} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
U.execute_train(
train_args=train_args,
num_gpus_per_node=NUM_GPUS,
megatron_model_type=MODEL_TYPE,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
if __name__ == "__main__":
# TODO also use typer
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
+134
View File
@@ -0,0 +1,134 @@
import os
import miles.utils.external_utils.command_utils as U
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
TIGHT_HOST_MEMORY = bool(int(os.environ.get("MILES_TEST_TIGHT_HOST_MEMORY", "1")))
MODEL_NAME = "Qwen3-4B"
MODEL_TYPE = "qwen3-4B"
NUM_GPUS = 8
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command("hf download Qwen/Qwen3-4B --local-dir /root/models/Qwen3-4B")
U.hf_download_dataset("zhuzilin/dapo-math-17k")
U.hf_download_dataset("zhuzilin/aime-2024")
U.convert_checkpoint(model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS)
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/{MODEL_NAME}_torch_dist "
rollout_args = (
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
"--input-key prompt "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type deepscaler "
"--num-rollout 3 "
"--rollout-batch-size 8 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 8192 "
"--rollout-temperature 0.8 "
"--global-batch-size 32 "
"--balance-data "
)
eval_args = (
f"{'--eval-interval 20 ' if ENABLE_EVAL else ''}"
"--eval-prompt-data aime24 /root/datasets/aime-2024/aime-2024.jsonl "
"--n-samples-per-eval-prompt 1 "
"--eval-max-response-len 16384 "
"--eval-top-k 1 "
)
perf_args = (
"--tensor-model-parallel-size 2 "
"--sequence-parallel "
"--pipeline-model-parallel-size 1 "
"--context-parallel-size 2 "
"--recompute-granularity full "
"--recompute-method uniform "
"--recompute-num-layers 1 "
"--use-dynamic-batch-size "
f"--max-tokens-per-gpu {2048 if TIGHT_HOST_MEMORY else 16384} "
)
ppo_args = (
"--advantage-estimator ppo "
f"{'' if TIGHT_HOST_MEMORY else '--use-kl-loss '}"
"--kl-loss-coef 0.00 "
"--kl-loss-type k1 "
"--kl-coef 0.00 "
"--entropy-coef 0.00 "
"--eps-clip 4e-4 "
"--num-critic-only-steps 1 "
"--normalize-advantages "
"--critic-lr 1e-5 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = (
"--rollout-num-gpus-per-engine 2 "
"--rollout-num-gpus 8 "
"--sglang-mem-fraction-static 0.8 "
"--sglang-max-running-requests 512 "
"--sglang-enable-metrics "
)
ci_args = "--ci-test "
misc_args = (
# default dropout in megatron is 0.1
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
# should be good for model performance
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
# need to comment this when using model with MLA
"--attention-backend flash "
"--actor-num-nodes 1 "
"--actor-num-gpus-per-node 4 "
"--colocate "
)
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{ppo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{perf_args} "
f"{eval_args} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
U.execute_train(
train_args=train_args,
num_gpus_per_node=NUM_GPUS,
megatron_model_type=MODEL_TYPE,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
if __name__ == "__main__":
# TODO also use typer
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
@@ -0,0 +1,155 @@
import os
import miles.utils.external_utils.command_utils as U
MODEL_NAME = "Qwen3-0.6B"
MODEL_TYPE = "qwen3-0.6B"
NUM_GPUS = 4
CP_SIZE = 1
MEGATRON_TP_SIZE = 1
MEGATRON_PP_SIZE = 1
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.hf_download_dataset("zhuzilin/dapo-math-17k")
U.convert_checkpoint(
model_name=MODEL_NAME,
megatron_model_type=MODEL_TYPE,
num_gpus_per_node=NUM_GPUS,
dir_dst="/root/models",
)
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/"
rollout_args = (
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
"--input-key prompt "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type deepscaler "
"--num-rollout 1 "
"--rollout-batch-size 8 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 8192 "
"--rollout-temperature 1 "
"--global-batch-size 64 "
"--use-dynamic-batch-size "
"--max-tokens-per-gpu 8192 "
)
ppo_args = (
"--advantage-estimator grpo "
"--kl-loss-coef 0.00 "
"--kl-loss-type k1 "
"--kl-coef 0.00 "
"--entropy-coef 0.00 "
"--eps-clip 4e-4 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = (
"--rollout-num-gpus-per-engine 1 " "--sglang-chunked-prefill-size 4096 " "--sglang-mem-fraction-static 0.75 "
)
ci_args = "--ci-test "
misc_args = "--actor-num-nodes 1 " "--colocate " f"--actor-num-gpus-per-node {NUM_GPUS} "
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{ppo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
debug_data_path = "test_rollout_data_megatron_fsdp_align.pt"
grad_norm_path = "grad_norm_fsdp.pt"
fsdp_args = (
"--train-backend fsdp "
"--attn-implementation flash_attention_2 "
"--gradient-checkpointing "
f"--context-parallel-size {CP_SIZE} "
f"--update-weight-buffer-size {512 * 1024 * 1024} "
"""--train-env-vars '{"PYTORCH_CUDA_ALLOC_CONF":"expandable_segments:True"}' """
)
try:
U.execute_train(
train_args=train_args + (f"{fsdp_args}" f"--save-debug-rollout-data {debug_data_path} "),
num_gpus_per_node=NUM_GPUS,
megatron_model_type=None,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
U.execute_train(
train_args=train_args
+ (
f"{fsdp_args}"
f"--load-debug-rollout-data {debug_data_path} "
f"--ci-save-grad-norm {grad_norm_path} "
"--debug-train-only "
),
num_gpus_per_node=NUM_GPUS,
megatron_model_type=None,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
U.execute_train(
train_args=train_args
+ (
f"--ref-load /root/models/{MODEL_NAME}_torch_dist "
f"--tensor-model-parallel-size {MEGATRON_TP_SIZE} "
"--sequence-parallel "
f"--pipeline-model-parallel-size {MEGATRON_PP_SIZE} "
f"--context-parallel-size {CP_SIZE} "
"--expert-model-parallel-size 1 "
"--expert-tensor-parallel-size 1 "
"--recompute-granularity full "
"--recompute-method uniform "
"--recompute-num-layers 1 "
"--train-memory-margin-bytes 3221225472 "
f"--load-debug-rollout-data {debug_data_path} "
f"--ci-load-grad-norm {grad_norm_path} "
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
"--attention-backend flash "
"--debug-train-only "
),
num_gpus_per_node=NUM_GPUS,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
megatron_model_type=MODEL_TYPE,
)
finally:
if os.path.exists(grad_norm_path):
os.remove(grad_norm_path)
if os.path.exists(debug_data_path):
os.remove(debug_data_path)
if __name__ == "__main__":
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
@@ -0,0 +1,138 @@
import os
import miles.utils.external_utils.command_utils as U
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
TIGHT_HOST_MEMORY = bool(int(os.environ.get("MILES_TEST_TIGHT_HOST_MEMORY", "1")))
MODEL_NAME = "Qwen3-0.6B"
MODEL_TYPE = "qwen3-0.6B"
NUM_GPUS = 8
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.hf_download_dataset("zhuzilin/dapo-math-17k")
U.convert_checkpoint(
model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS, dir_dst="/root/models"
)
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}_torch_dist "
rollout_args = (
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
"--input-key prompt "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type deepscaler "
"--num-rollout 1 "
"--rollout-batch-size 4 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 8192 "
"--rollout-temperature 0.8 "
"--global-batch-size 32 "
)
ppo_args = (
"--advantage-estimator grpo "
"--kl-loss-coef 0.00 "
"--kl-loss-type k1 "
"--kl-coef 0.00 "
"--entropy-coef 0.00 "
"--eps-clip 4e-4 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = "--rollout-num-gpus-per-engine 2 " "--rollout-num-gpus 8 " "--sglang-mem-fraction-static 0.8 "
ci_args = "--ci-test "
misc_args = (
# default dropout in megatron is 0.1
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
# should be good for model performance
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
# need to comment this when using model with MLA
"--attention-backend flash "
"--actor-num-nodes 1 "
"--colocate "
)
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{ppo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{sglang_args} "
f"{ci_args} "
f"{misc_args} "
)
for i in range(2):
U.execute_train(
train_args=train_args
+ (
f"--save-debug-rollout-data data-{i}.pt "
f"--ci-save-grad-norm grad_norms-{i}.pt "
f"--actor-num-gpus-per-node {NUM_GPUS} "
),
num_gpus_per_node=NUM_GPUS,
megatron_model_type=MODEL_TYPE,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
# 8 GPU CPU 1
for num_gpus in [8, 4, 2]:
remaining_gpus = num_gpus
for tp_size in [1, 2, 4, 8]:
remaining_gpus /= tp_size
for pp_size in [1, 2, 4]:
if remaining_gpus < pp_size:
continue
remaining_gpus /= pp_size
for cp_size in [1, 2, 4, 8]:
if remaining_gpus < cp_size:
continue
args = train_args + (
f"--load-debug-rollout-data data-{i}.pt "
f"--ci-load-grad-norm grad_norms-{i}.pt "
f"--context-parallel-size {cp_size} "
f"--tensor-model-parallel-size {tp_size} "
f"--pipeline-model-parallel-size {pp_size} "
"--sequence-parallel "
f"--actor-num-gpus-per-node {num_gpus} "
"--use-dynamic-batch-size "
"--max-tokens-per-gpu 8192 "
)
U.execute_train(
train_args=args,
num_gpus_per_node=num_gpus,
megatron_model_type=MODEL_TYPE,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
train_args += "--calculate-per-token-loss "
if __name__ == "__main__":
# TODO also use typer
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
@@ -0,0 +1,129 @@
import os
import miles.utils.external_utils.command_utils as U
TIGHT_DEVICE_MEMORY = U.get_bool_env_var("MILES_TEST_TIGHT_DEVICE_MEMORY", "1")
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
NUM_GPUS = 4
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"huggingface-cli download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.hf_download_dataset("zhuzilin/gsm8k")
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}/ "
rollout_args = (
"--prompt-data /root/datasets/gsm8k/train.parquet "
"--input-key messages "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type math "
"--num-rollout 3 "
"--rollout-batch-size 8 "
"--n-samples-per-prompt 4 "
"--rollout-max-response-len 1024 "
"--rollout-temperature 0.8 "
"--over-sampling-batch-size 16 "
"--dynamic-sampling-filter-path miles.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std "
"--global-batch-size 32 "
)
eval_args = (
"--eval-interval 8 "
"--eval-prompt-data gsm8k /root/datasets/gsm8k/test.parquet "
"--n-samples-per-eval-prompt 1 "
"--eval-max-response-len 1024 "
"--eval-top-k 1 "
)
perf_args = (
"--tensor-model-parallel-size 1 "
"--sequence-parallel "
"--pipeline-model-parallel-size 1 "
"--context-parallel-size 1 "
"--expert-model-parallel-size 1 "
"--expert-tensor-parallel-size 1 "
"--use-dynamic-batch-size "
"--max-tokens-per-gpu 9216 "
)
grpo_args = (
"--advantage-estimator grpo "
"--use-kl-loss "
"--kl-loss-coef 0.00 "
"--kl-loss-type low_var_kl "
"--entropy-coef 0.00 "
"--eps-clip 0.2 "
"--eps-clip-high 0.28 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = (
"--rollout-num-gpus-per-engine 1 "
f"--sglang-mem-fraction-static {0.55 if TIGHT_DEVICE_MEMORY else 0.65} "
"--sglang-enable-metrics "
)
ci_args = "--ci-test "
fault_tolerance_args = (
"--use-fault-tolerance "
"--rollout-health-check-interval 5 "
"--rollout-health-check-timeout 10 "
"--rollout-health-check-first-wait 0 "
)
misc_args = (
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
"--attention-backend flash "
"--actor-num-nodes 1 "
"--actor-num-gpus-per-node 1 "
"--rollout-num-gpus 3 "
"--megatron-to-hf-mode bridge "
)
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{grpo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{perf_args} "
f"{eval_args} "
f"{sglang_args} "
f"{ci_args} "
f"{fault_tolerance_args} "
f"{misc_args} "
)
U.execute_train(
train_args=train_args,
num_gpus_per_node=NUM_GPUS,
megatron_model_type=MODEL_TYPE,
train_script="train_async.py",
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
if __name__ == "__main__":
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
@@ -0,0 +1,128 @@
import os
import miles.utils.external_utils.command_utils as U
TIGHT_DEVICE_MEMORY = U.get_bool_env_var("MILES_TEST_TIGHT_DEVICE_MEMORY", "1")
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
MODEL_TYPE = "qwen2.5-0.5B"
NUM_GPUS = 4
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"huggingface-cli download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.hf_download_dataset("zhuzilin/gsm8k")
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}/ "
rollout_args = (
"--prompt-data /root/datasets/gsm8k/train.parquet "
"--input-key messages "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type math "
"--num-rollout 3 "
"--rollout-batch-size 8 "
"--n-samples-per-prompt 4 "
"--rollout-max-response-len 1024 "
"--rollout-temperature 0.8 "
"--over-sampling-batch-size 16 "
"--dynamic-sampling-filter-path miles.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std "
"--global-batch-size 32 "
)
eval_args = (
"--eval-interval 20 "
"--eval-prompt-data gsm8k /root/datasets/gsm8k/test.parquet "
"--n-samples-per-eval-prompt 1 "
"--eval-max-response-len 1024 "
"--eval-top-k 1 "
)
perf_args = (
"--tensor-model-parallel-size 1 "
"--sequence-parallel "
"--pipeline-model-parallel-size 1 "
"--context-parallel-size 1 "
"--expert-model-parallel-size 1 "
"--expert-tensor-parallel-size 1 "
"--use-dynamic-batch-size "
"--max-tokens-per-gpu 9216 "
)
grpo_args = (
"--advantage-estimator grpo "
"--use-kl-loss "
"--kl-loss-coef 0.00 "
"--kl-loss-type low_var_kl "
"--entropy-coef 0.00 "
"--eps-clip 0.2 "
"--eps-clip-high 0.28 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = (
"--rollout-num-gpus-per-engine 1 "
f"--sglang-mem-fraction-static {0.6 if TIGHT_DEVICE_MEMORY else 0.7} "
"--sglang-enable-metrics "
)
ci_args = "--ci-test "
fault_tolerance_args = (
"--use-fault-tolerance "
"--rollout-health-check-interval 5 "
"--rollout-health-check-timeout 10 "
"--rollout-health-check-first-wait 0 "
)
misc_args = (
"--attention-dropout 0.0 "
"--hidden-dropout 0.0 "
"--accumulate-allreduce-grads-in-fp32 "
"--attention-softmax-in-fp32 "
"--attention-backend flash "
"--actor-num-nodes 1 "
"--actor-num-gpus-per-node 4 "
"--colocate "
"--megatron-to-hf-mode bridge "
)
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{grpo_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{perf_args} "
f"{eval_args} "
f"{sglang_args} "
f"{ci_args} "
f"{fault_tolerance_args} "
f"{misc_args} "
)
U.execute_train(
train_args=train_args,
num_gpus_per_node=NUM_GPUS,
megatron_model_type=MODEL_TYPE,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
if __name__ == "__main__":
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()
@@ -0,0 +1,104 @@
import os
import miles.utils.external_utils.command_utils as U
MODEL_NAME = "Qwen3-0.6B"
def prepare():
U.exec_command("mkdir -p /root/models /root/datasets")
U.exec_command(f"huggingface-cli download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
U.hf_download_dataset("zhuzilin/gsm8k")
def execute():
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} "
rollout_args = (
"--prompt-data /root/datasets/gsm8k/train.parquet "
"--input-key messages "
"--label-key label "
"--apply-chat-template "
"--rollout-shuffle "
"--rm-type math "
f"--num-rollout {3000 if U.get_env_enable_infinite_run() else 60} "
"--rollout-batch-size 32 "
"--n-samples-per-prompt 8 "
"--rollout-max-response-len 1024 "
"--rollout-temperature 1 "
"--over-sampling-batch-size 64 "
"--dynamic-sampling-filter-path miles.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std "
"--global-batch-size 256 "
)
eval_args = (
"--eval-interval 20 "
"--eval-prompt-data gsm8k /root/datasets/gsm8k/test.parquet "
"--n-samples-per-eval-prompt 1 "
"--eval-max-response-len 1024 "
"--eval-top-k 1 "
)
grpo_args = (
"--advantage-estimator grpo "
# "--use-kl-loss "
"--kl-loss-coef 0.00 "
"--kl-loss-type low_var_kl "
"--kl-coef 0.00 "
"--entropy-coef 0.00 "
"--eps-clip 0.2 "
"--eps-clip-high 0.28 "
)
optimizer_args = (
"--optimizer adam "
"--lr 1e-6 "
"--lr-decay-style constant "
"--weight-decay 0.1 "
"--adam-beta1 0.9 "
"--adam-beta2 0.98 "
)
sglang_args = "--rollout-num-gpus-per-engine 2 " "--sglang-decode-log-interval 1000 " "--sglang-enable-metrics "
fsdp_args = (
# Set to true for FULL_STATE_DICT mode, false for SHARDED_STATE_DICT mode (default)
# "--fsdp-full-params " # Uncomment this line to enable full params mode
# Set the bucket size for weight update
"--update-weight-buffer-size 536870912 " # 512MB
)
ci_args = (
"--ci-test "
"--ci-disable-kl-checker "
"--ci-metric-checker-key eval/gsm8k "
"--ci-metric-checker-threshold 0.71 " # loose threshold at 60 step
)
misc_args = "--actor-num-nodes 1 " "--actor-num-gpus-per-node 2 " "--colocate " "--train-backend fsdp "
train_args = (
f"{ckpt_args} "
f"{rollout_args} "
f"{optimizer_args} "
f"{grpo_args} "
f"{sglang_args} "
f"{U.get_default_wandb_args(__file__)} "
f"{eval_args} "
f"{fsdp_args} "
f"{ci_args} "
f"{misc_args} "
)
U.execute_train(
train_args=train_args,
num_gpus_per_node=2,
megatron_model_type=None,
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
)
if __name__ == "__main__":
prepare()
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
os.environ.pop(proxy_var, None)
execute()