mirror of
https://github.com/radixark/miles.git
synced 2026-10-02 07:14:53 +08:00
[CI] Re-organize and enable necessary end2end CI cases (#499)
Co-authored-by: Yusheng Su <radixark@ac-h200-user-3.tail134ba0.ts.net>
This commit is contained in:
co-authored by
Yusheng Su
parent
a8c8687ca7
commit
81ea4a82e4
@@ -33,7 +33,7 @@ jobs:
|
||||
options: >
|
||||
--gpus all
|
||||
--ipc=host
|
||||
--shm-size=16g
|
||||
--shm-size=32g
|
||||
--ulimit memlock=-1
|
||||
--ulimit stack=67108864
|
||||
--memory=0
|
||||
@@ -41,6 +41,9 @@ jobs:
|
||||
-v /mnt/nvme0n1/miles_ci:/data/miles_ci
|
||||
-v /mnt/nvme0n1/miles_ci/models:/root/models
|
||||
-v /mnt/nvme0n1/miles_ci/datasets:/root/datasets
|
||||
--privileged
|
||||
--ulimit nofile=65535:65535
|
||||
-v /tmp:/tmp
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
@@ -52,11 +55,26 @@ jobs:
|
||||
GITHUB_COMMIT_NAME: ${{ github.sha }}_${{ github.event.pull_request.number || 'non-pr' }}
|
||||
WANDB_API_KEY: ${{ secrets.WANDB_API_KEY }}
|
||||
MILES_TEST_ENABLE_INFINITE_RUN: ${{ (github.event_name == 'workflow_dispatch' && github.event.inputs.infinite_run) || 'false' }}
|
||||
MILES_TEST_USE_DEEPEP: ${{ matrix.info.use_deepep || '0' }}
|
||||
MILES_TEST_USE_FP8_ROLLOUT: ${{ matrix.info.use_fp8_rollout || '0' }}
|
||||
MILES_TEST_ENABLE_EVAL: ${{ matrix.info.enable_eval || '1' }}
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Cleanup Ray processes
|
||||
shell: bash
|
||||
run: |
|
||||
pkill -9 -f 'ray::' 2>/dev/null || true
|
||||
pkill -9 -f raylet 2>/dev/null || true
|
||||
pkill -9 -f gcs_server 2>/dev/null || true
|
||||
pkill -9 -f 'ray-dashboard' 2>/dev/null || true
|
||||
pkill -9 sglang 2>/dev/null || true
|
||||
ray stop --force 2>/dev/null || true
|
||||
rm -rf /tmp/ray/* 2>/dev/null || true
|
||||
sleep 3
|
||||
|
||||
- name: Install
|
||||
shell: bash
|
||||
run: cd $GITHUB_WORKSPACE && pip install -e . --no-deps --break-system-packages
|
||||
@@ -65,6 +83,84 @@ jobs:
|
||||
shell: bash
|
||||
run: python tests/ci/gpu_lock_exec.py --count ${{ matrix.info.num_gpus }} -- pytest tests/${{ matrix.info.test_file }}
|
||||
|
||||
- name: Post-test cleanup
|
||||
if: always()
|
||||
shell: bash
|
||||
run: |
|
||||
pkill -9 -f 'ray::' 2>/dev/null || true
|
||||
pkill -9 -f raylet 2>/dev/null || true
|
||||
ray stop --force 2>/dev/null || true
|
||||
rm -rf /tmp/ray/* 2>/dev/null || true
|
||||
|
||||
|
||||
unit-test:
|
||||
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-unit-test'))
|
||||
runs-on: self-hosted
|
||||
container:
|
||||
image: radixark/miles:latest
|
||||
options: >
|
||||
--gpus all
|
||||
--ipc=host
|
||||
--shm-size=32g
|
||||
--ulimit memlock=-1
|
||||
--ulimit stack=67108864
|
||||
--memory=0
|
||||
--memory-swap=0
|
||||
-v /mnt/nvme0n1/miles_ci:/data/miles_ci
|
||||
-v /mnt/nvme0n1/miles_ci/models:/root/models
|
||||
-v /mnt/nvme0n1/miles_ci/datasets:/root/datasets
|
||||
--privileged
|
||||
--ulimit nofile=65535:65535
|
||||
-v /tmp:/tmp
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
info: [{"num_gpus": 2, "test_file": "e2e/fsdp/test_qwen3_4B_fsdp_true_on_policy.py"}]
|
||||
defaults:
|
||||
run:
|
||||
working-directory: ${{ github.workspace }}
|
||||
env:
|
||||
GITHUB_COMMIT_NAME: ${{ github.sha }}_${{ github.event.pull_request.number || 'non-pr' }}
|
||||
WANDB_API_KEY: ${{ secrets.WANDB_API_KEY }}
|
||||
MILES_TEST_ENABLE_INFINITE_RUN: ${{ (github.event_name == 'workflow_dispatch' && github.event.inputs.infinite_run) || 'false' }}
|
||||
MILES_TEST_USE_DEEPEP: ${{ matrix.info.use_deepep || '0' }}
|
||||
MILES_TEST_USE_FP8_ROLLOUT: ${{ matrix.info.use_fp8_rollout || '0' }}
|
||||
MILES_TEST_ENABLE_EVAL: ${{ matrix.info.enable_eval || '1' }}
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Cleanup Ray processes
|
||||
shell: bash
|
||||
run: |
|
||||
pkill -9 -f 'ray::' 2>/dev/null || true
|
||||
pkill -9 -f raylet 2>/dev/null || true
|
||||
pkill -9 -f gcs_server 2>/dev/null || true
|
||||
pkill -9 -f 'ray-dashboard' 2>/dev/null || true
|
||||
pkill -9 sglang 2>/dev/null || true
|
||||
ray stop --force 2>/dev/null || true
|
||||
rm -rf /tmp/ray/* 2>/dev/null || true
|
||||
sleep 3
|
||||
|
||||
- name: Install
|
||||
shell: bash
|
||||
run: cd $GITHUB_WORKSPACE && pip install -e . --no-deps --break-system-packages
|
||||
|
||||
- name: Execute
|
||||
shell: bash
|
||||
run: python tests/ci/gpu_lock_exec.py --count ${{ matrix.info.num_gpus }} -- python tests/${{ matrix.info.test_file }}
|
||||
|
||||
- name: Post-test cleanup
|
||||
if: always()
|
||||
shell: bash
|
||||
run: |
|
||||
pkill -9 -f 'ray::' 2>/dev/null || true
|
||||
pkill -9 -f raylet 2>/dev/null || true
|
||||
ray stop --force 2>/dev/null || true
|
||||
rm -rf /tmp/ray/* 2>/dev/null || true
|
||||
|
||||
|
||||
e2e-test-short:
|
||||
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-short'))
|
||||
runs-on: self-hosted
|
||||
@@ -87,7 +183,7 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
info: [{"num_gpus": 4, "test_file": "test_qwen2.5_0.5B_gsm8k_async_short.py"}, {"num_gpus": 4, "test_file": "test_qwen2.5_0.5B_gsm8k_short.py"}, {"num_gpus": 2, "test_file": "test_qwen3_0.6B_fsdp_colocated_2xGPU.py"}]
|
||||
info: [{"num_gpus": 4, "test_file": "e2e/short/test_qwen2.5_0.5B_gsm8k_async_short.py"}, {"num_gpus": 4, "test_file": "e2e/short/test_qwen2.5_0.5B_gsm8k_short.py"}, {"num_gpus": 2, "test_file": "e2e/short/test_qwen3_0.6B_fsdp_colocated_2xGPU.py"}]
|
||||
defaults:
|
||||
run:
|
||||
working-directory: ${{ github.workspace }}
|
||||
@@ -132,6 +228,7 @@ jobs:
|
||||
ray stop --force 2>/dev/null || true
|
||||
rm -rf /tmp/ray/* 2>/dev/null || true
|
||||
|
||||
|
||||
e2e-test-fsdp:
|
||||
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-fsdp'))
|
||||
runs-on: self-hosted
|
||||
@@ -154,7 +251,7 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
info: [{"num_gpus": 2, "test_file": "test_qwen3_4B_fsdp_true_on_policy.py"}, {"num_gpus": 8, "test_file": "test_qwen3_vl_4B_fsdp.py"}, {"num_gpus": 2, "test_file": "test_qwen3_0.6B_fsdp_distributed.py"}, {"num_gpus": 4, "test_file": "test_qwen3_0.6B_megatron_fsdp_align.py"}]
|
||||
info: [{"num_gpus": 2, "test_file": "e2e/fsdp/test_qwen3_4B_fsdp_true_on_policy.py"}, {"num_gpus": 8, "test_file": "e2e/fsdp/test_qwen3_vl_4B_fsdp.py"}, {"num_gpus": 2, "test_file": "e2e/fsdp/test_qwen3_0.6B_fsdp_distributed.py"}, {"num_gpus": 4, "test_file": "e2e/fsdp/test_qwen3_0.6B_megatron_fsdp_align.py"}]
|
||||
defaults:
|
||||
run:
|
||||
working-directory: ${{ github.workspace }}
|
||||
@@ -199,6 +296,7 @@ jobs:
|
||||
ray stop --force 2>/dev/null || true
|
||||
rm -rf /tmp/ray/* 2>/dev/null || true
|
||||
|
||||
|
||||
e2e-test-megatron:
|
||||
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-megatron'))
|
||||
runs-on: self-hosted
|
||||
@@ -221,7 +319,7 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
info: [{"num_gpus": 8, "test_file": "test_quick_start_glm4_9B.py"}, {"num_gpus": 8, "test_file": "test_qwen3_30B_A3B.py", "use_deepep": "1", "use_fp8_rollout": "1"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "test_qwen3_30B_A3B_r3.py", "use_deepep": "1", "use_fp8_rollout": "1"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "test_qwen3_30B_A3B_r3.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ppo.py"}, {"num_gpus": 8, "test_file": "test_moonlight_16B_A3B.py"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "test_moonlight_16B_A3B_r3.py"}, {"num_gpus": 8, "test_file": "test_mimo_7B_mtp_only_grad.py"}]
|
||||
info: [{"num_gpus": 8, "test_file": "e2e/megatron/test_quick_start_glm4_9B.py"}, {"num_gpus": 8, "test_file": "e2e/megatron/test_qwen3_30B_A3B.py", "use_deepep": "1", "use_fp8_rollout": "1"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "e2e/megatron/test_qwen3_30B_A3B_r3.py", "use_deepep": "1", "use_fp8_rollout": "1"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "e2e/megatron/test_qwen3_30B_A3B_r3.py"}, {"num_gpus": 8, "test_file": "e2e/megatron/test_qwen3_4B_ppo.py"}, {"num_gpus": 8, "test_file": "e2e/megatron/test_moonlight_16B_A3B.py"}, {"enable_eval": "0", "num_gpus": 8, "test_file": "e2e/megatron/test_moonlight_16B_A3B_r3.py"}, {"num_gpus": 8, "test_file": "e2e/megatron/test_mimo_7B_mtp_only_grad.py"}]
|
||||
defaults:
|
||||
run:
|
||||
working-directory: ${{ github.workspace }}
|
||||
@@ -266,6 +364,7 @@ jobs:
|
||||
ray stop --force 2>/dev/null || true
|
||||
rm -rf /tmp/ray/* 2>/dev/null || true
|
||||
|
||||
|
||||
e2e-test-precision:
|
||||
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-precision'))
|
||||
runs-on: self-hosted
|
||||
@@ -288,7 +387,7 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
info: [{"num_gpus": 8, "test_file": "test_qwen3_0.6B_parallel_check.py"}, {"num_gpus": 4, "test_file": "test_qwen3_0.6B_megatron_fsdp_align.py"}]
|
||||
info: [{"num_gpus": 8, "test_file": "e2e/precision/test_qwen3_0.6B_parallel_check.py"}, {"num_gpus": 4, "test_file": "e2e/precision/test_qwen3_0.6B_megatron_fsdp_align.py"}]
|
||||
defaults:
|
||||
run:
|
||||
working-directory: ${{ github.workspace }}
|
||||
@@ -333,6 +432,7 @@ jobs:
|
||||
ray stop --force 2>/dev/null || true
|
||||
rm -rf /tmp/ray/* 2>/dev/null || true
|
||||
|
||||
|
||||
e2e-test-ckpt:
|
||||
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-ckpt'))
|
||||
runs-on: self-hosted
|
||||
@@ -355,7 +455,7 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
info: [{"num_gpus": 8, "test_file": "test_qwen3_4B_ckpt.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ckpt.py --async-save"}]
|
||||
info: [{"num_gpus": 8, "test_file": "e2e/ckpt/test_qwen3_4B_ckpt.py"}, {"num_gpus": 8, "test_file": "e2e/ckpt/test_qwen3_4B_ckpt.py --async-save"}]
|
||||
defaults:
|
||||
run:
|
||||
working-directory: ${{ github.workspace }}
|
||||
@@ -400,6 +500,7 @@ jobs:
|
||||
ray stop --force 2>/dev/null || true
|
||||
rm -rf /tmp/ray/* 2>/dev/null || true
|
||||
|
||||
|
||||
e2e-test-long:
|
||||
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-long'))
|
||||
runs-on: self-hosted
|
||||
@@ -422,7 +523,7 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
info: [{"num_gpus": 2, "test_file": "test_qwen2.5_0.5B_gsm8k.py"}, {"num_gpus": 2, "test_file": "test_qwen2.5_0.5B_gsm8k_async.py"}]
|
||||
info: [{"num_gpus": 2, "test_file": "e2e/long/test_qwen2.5_0.5B_gsm8k.py"}, {"num_gpus": 2, "test_file": "e2e/long/test_qwen2.5_0.5B_gsm8k_async.py"}]
|
||||
defaults:
|
||||
run:
|
||||
working-directory: ${{ github.workspace }}
|
||||
@@ -467,11 +568,12 @@ jobs:
|
||||
ray stop --force 2>/dev/null || true
|
||||
rm -rf /tmp/ray/* 2>/dev/null || true
|
||||
|
||||
|
||||
e2e-test-image:
|
||||
if: (github.event_name == 'workflow_dispatch') || (github.event.pull_request && contains(github.event.pull_request.labels.*.name, 'run-ci-image'))
|
||||
runs-on: self-hosted
|
||||
container:
|
||||
image: radixark/miles-test:latest
|
||||
image: radixark/miles:latest
|
||||
options: >
|
||||
--gpus all
|
||||
--ipc=host
|
||||
@@ -489,7 +591,7 @@ jobs:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
info: [{"num_gpus": 4, "test_file": "test_qwen2.5_0.5B_gsm8k_async_short.py"}, {"num_gpus": 4, "test_file": "test_qwen2.5_0.5B_gsm8k_short.py"}, {"num_gpus": 2, "test_file": "test_qwen3_0.6B_fsdp_colocated_2xGPU.py"}, {"num_gpus": 2, "test_file": "test_qwen3_4B_fsdp_true_on_policy.py"}, {"num_gpus": 8, "test_file": "test_qwen3_vl_4B_fsdp.py"}, {"num_gpus": 2, "test_file": "test_qwen3_0.6B_fsdp_distributed.py"}, {"num_gpus": 8, "test_file": "test_quick_start_glm4_9B.py"}, {"num_gpus": 8, "test_file": "test_qwen3_30B_A3B.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ppo.py"}, {"num_gpus": 8, "test_file": "test_moonlight_16B_A3B.py"}, {"num_gpus": 8, "test_file": "test_mimo_7B_mtp_only_grad.py"}, {"num_gpus": 8, "test_file": "test_qwen3_0.6B_parallel_check.py"}, {"num_gpus": 4, "test_file": "test_qwen3_0.6B_megatron_fsdp_align.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ckpt.py"}, {"num_gpus": 8, "test_file": "test_qwen3_4B_ckpt.py --async-save"}, {"num_gpus": 2, "test_file": "test_qwen2.5_0.5B_gsm8k.py"}, {"num_gpus": 2, "test_file": "test_qwen2.5_0.5B_gsm8k_async.py"}]
|
||||
info: [{"num_gpus": 4, "test_file": "e2e/image/test_qwen2.5_0.5B_gsm8k_async_short.py"}, {"num_gpus": 4, "test_file": "e2e/image/test_qwen2.5_0.5B_gsm8k_short.py"}, {"num_gpus": 2, "test_file": "e2e/image/test_qwen3_0.6B_fsdp_colocated_2xGPU.py"}, {"num_gpus": 2, "test_file": "e2e/image/test_qwen3_4B_fsdp_true_on_policy.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_qwen3_vl_4B_fsdp.py"}, {"num_gpus": 2, "test_file": "e2e/image/test_qwen3_0.6B_fsdp_distributed.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_quick_start_glm4_9B.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_qwen3_30B_A3B.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_qwen3_4B_ppo.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_moonlight_16B_A3B.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_mimo_7B_mtp_only_grad.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_qwen3_0.6B_parallel_check.py"}, {"num_gpus": 4, "test_file": "e2e/image/test_qwen3_0.6B_megatron_fsdp_align.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_qwen3_4B_ckpt.py"}, {"num_gpus": 8, "test_file": "e2e/image/test_qwen3_4B_ckpt.py --async-save"}, {"num_gpus": 2, "test_file": "e2e/image/test_qwen2.5_0.5B_gsm8k.py"}, {"num_gpus": 2, "test_file": "e2e/image/test_qwen2.5_0.5B_gsm8k_async.py"}]
|
||||
defaults:
|
||||
run:
|
||||
working-directory: ${{ github.workspace }}
|
||||
@@ -533,3 +635,4 @@ jobs:
|
||||
pkill -9 -f raylet 2>/dev/null || true
|
||||
ray stop --force 2>/dev/null || true
|
||||
rm -rf /tmp/ray/* 2>/dev/null || true
|
||||
|
||||
|
||||
@@ -5,78 +5,84 @@
|
||||
{'test_file': 'fast', 'num_gpus': 0},
|
||||
],
|
||||
},
|
||||
'unit-test': {
|
||||
'label': 'run-unit-test',
|
||||
'tests': [
|
||||
{'test_file': 'e2e/fsdp/test_qwen3_4B_fsdp_true_on_policy.py', 'num_gpus': 2}
|
||||
],
|
||||
},
|
||||
'e2e-test-short': {
|
||||
'label': 'run-ci-short',
|
||||
'tests': [
|
||||
{'test_file': 'test_qwen2.5_0.5B_gsm8k_async_short.py', 'num_gpus': 4},
|
||||
{'test_file': 'test_qwen2.5_0.5B_gsm8k_short.py', 'num_gpus': 4},
|
||||
{'test_file': 'test_qwen3_0.6B_fsdp_colocated_2xGPU.py', 'num_gpus': 2},
|
||||
{'test_file': 'e2e/short/test_qwen2.5_0.5B_gsm8k_async_short.py', 'num_gpus': 4},
|
||||
{'test_file': 'e2e/short/test_qwen2.5_0.5B_gsm8k_short.py', 'num_gpus': 4},
|
||||
{'test_file': 'e2e/short/test_qwen3_0.6B_fsdp_colocated_2xGPU.py', 'num_gpus': 2},
|
||||
],
|
||||
},
|
||||
'e2e-test-fsdp': {
|
||||
'label': 'run-ci-fsdp',
|
||||
'tests': [
|
||||
{'test_file': 'test_qwen3_4B_fsdp_true_on_policy.py', 'num_gpus': 2},
|
||||
{'test_file': 'test_qwen3_vl_4B_fsdp.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_qwen3_0.6B_fsdp_distributed.py', 'num_gpus': 2},
|
||||
{'test_file': 'test_qwen3_0.6B_megatron_fsdp_align.py', 'num_gpus': 4},
|
||||
{'test_file': 'e2e/fsdp/test_qwen3_4B_fsdp_true_on_policy.py', 'num_gpus': 2},
|
||||
{'test_file': 'e2e/fsdp/test_qwen3_vl_4B_fsdp.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/fsdp/test_qwen3_0.6B_fsdp_distributed.py', 'num_gpus': 2},
|
||||
{'test_file': 'e2e/fsdp/test_qwen3_0.6B_megatron_fsdp_align.py', 'num_gpus': 4},
|
||||
],
|
||||
},
|
||||
'e2e-test-megatron': {
|
||||
'label': 'run-ci-megatron',
|
||||
'tests': [
|
||||
{'test_file': 'test_quick_start_glm4_9B.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_qwen3_30B_A3B.py', 'num_gpus': 8, 'use_deepep': '1', 'use_fp8_rollout': '1'},
|
||||
{'test_file': 'test_qwen3_30B_A3B_r3.py', 'num_gpus': 8, 'use_deepep': '1', 'use_fp8_rollout': '1', 'enable_eval': '0'},
|
||||
{'test_file': 'test_qwen3_30B_A3B_r3.py', 'num_gpus': 8, 'enable_eval': '0'},
|
||||
{'test_file': 'test_qwen3_4B_ppo.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_moonlight_16B_A3B.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_moonlight_16B_A3B_r3.py', 'num_gpus': 8, 'enable_eval': '0'},
|
||||
{'test_file': 'test_mimo_7B_mtp_only_grad.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/megatron/test_quick_start_glm4_9B.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/megatron/test_qwen3_30B_A3B.py', 'num_gpus': 8, 'use_deepep': '1', 'use_fp8_rollout': '1'},
|
||||
{'test_file': 'e2e/megatron/test_qwen3_30B_A3B_r3.py', 'num_gpus': 8, 'use_deepep': '1', 'use_fp8_rollout': '1', 'enable_eval': '0'},
|
||||
{'test_file': 'e2e/megatron/test_qwen3_30B_A3B_r3.py', 'num_gpus': 8, 'enable_eval': '0'},
|
||||
{'test_file': 'e2e/megatron/test_qwen3_4B_ppo.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/megatron/test_moonlight_16B_A3B.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/megatron/test_moonlight_16B_A3B_r3.py', 'num_gpus': 8, 'enable_eval': '0'},
|
||||
{'test_file': 'e2e/megatron/test_mimo_7B_mtp_only_grad.py', 'num_gpus': 8},
|
||||
],
|
||||
},
|
||||
'e2e-test-precision': {
|
||||
'label': 'run-ci-precision',
|
||||
'tests': [
|
||||
{'test_file': 'test_qwen3_0.6B_parallel_check.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_qwen3_0.6B_megatron_fsdp_align.py', 'num_gpus': 4},
|
||||
{'test_file': 'e2e/precision/test_qwen3_0.6B_parallel_check.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/precision/test_qwen3_0.6B_megatron_fsdp_align.py', 'num_gpus': 4},
|
||||
],
|
||||
},
|
||||
'e2e-test-ckpt': {
|
||||
'label': 'run-ci-ckpt',
|
||||
'tests': [
|
||||
{'test_file': 'test_qwen3_4B_ckpt.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_qwen3_4B_ckpt.py --async-save', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/ckpt/test_qwen3_4B_ckpt.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/ckpt/test_qwen3_4B_ckpt.py --async-save', 'num_gpus': 8},
|
||||
],
|
||||
},
|
||||
'e2e-test-long': {
|
||||
'label': 'run-ci-long',
|
||||
'tests': [
|
||||
{'test_file': 'test_qwen2.5_0.5B_gsm8k.py', 'num_gpus': 2},
|
||||
{'test_file': 'test_qwen2.5_0.5B_gsm8k_async.py', 'num_gpus': 2},
|
||||
{'test_file': 'e2e/long/test_qwen2.5_0.5B_gsm8k.py', 'num_gpus': 2},
|
||||
{'test_file': 'e2e/long/test_qwen2.5_0.5B_gsm8k_async.py', 'num_gpus': 2},
|
||||
],
|
||||
},
|
||||
'e2e-test-image': {
|
||||
'label': 'run-ci-image',
|
||||
'image': 'radixark/miles-test:latest',
|
||||
'image': 'radixark/miles:latest',
|
||||
'tests': [
|
||||
{'test_file': 'test_qwen2.5_0.5B_gsm8k_async_short.py', 'num_gpus': 4},
|
||||
{'test_file': 'test_qwen2.5_0.5B_gsm8k_short.py', 'num_gpus': 4},
|
||||
{'test_file': 'test_qwen3_0.6B_fsdp_colocated_2xGPU.py', 'num_gpus': 2},
|
||||
{'test_file': 'test_qwen3_4B_fsdp_true_on_policy.py', 'num_gpus': 2},
|
||||
{'test_file': 'test_qwen3_vl_4B_fsdp.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_qwen3_0.6B_fsdp_distributed.py', 'num_gpus': 2},
|
||||
{'test_file': 'test_quick_start_glm4_9B.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_qwen3_30B_A3B.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_qwen3_4B_ppo.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_moonlight_16B_A3B.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_mimo_7B_mtp_only_grad.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_qwen3_0.6B_parallel_check.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_qwen3_0.6B_megatron_fsdp_align.py', 'num_gpus': 4},
|
||||
{'test_file': 'test_qwen3_4B_ckpt.py', 'num_gpus': 8},
|
||||
{'test_file': 'test_qwen3_4B_ckpt.py --async-save', 'num_gpus': 8},
|
||||
{'test_file': 'test_qwen2.5_0.5B_gsm8k.py', 'num_gpus': 2},
|
||||
{'test_file': 'test_qwen2.5_0.5B_gsm8k_async.py', 'num_gpus': 2},
|
||||
{'test_file': 'e2e/image/test_qwen2.5_0.5B_gsm8k_async_short.py', 'num_gpus': 4},
|
||||
{'test_file': 'e2e/image/test_qwen2.5_0.5B_gsm8k_short.py', 'num_gpus': 4},
|
||||
{'test_file': 'e2e/image/test_qwen3_0.6B_fsdp_colocated_2xGPU.py', 'num_gpus': 2},
|
||||
{'test_file': 'e2e/image/test_qwen3_4B_fsdp_true_on_policy.py', 'num_gpus': 2},
|
||||
{'test_file': 'e2e/image/test_qwen3_vl_4B_fsdp.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/image/test_qwen3_0.6B_fsdp_distributed.py', 'num_gpus': 2},
|
||||
{'test_file': 'e2e/image/test_quick_start_glm4_9B.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/image/test_qwen3_30B_A3B.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/image/test_qwen3_4B_ppo.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/image/test_moonlight_16B_A3B.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/image/test_mimo_7B_mtp_only_grad.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/image/test_qwen3_0.6B_parallel_check.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/image/test_qwen3_0.6B_megatron_fsdp_align.py', 'num_gpus': 4},
|
||||
{'test_file': 'e2e/image/test_qwen3_4B_ckpt.py', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/image/test_qwen3_4B_ckpt.py --async-save', 'num_gpus': 8},
|
||||
{'test_file': 'e2e/image/test_qwen2.5_0.5B_gsm8k.py', 'num_gpus': 2},
|
||||
{'test_file': 'e2e/image/test_qwen2.5_0.5B_gsm8k_async.py', 'num_gpus': 2},
|
||||
],
|
||||
},
|
||||
} %>
|
||||
@@ -160,4 +166,14 @@ jobs:
|
||||
- name: Execute
|
||||
shell: bash
|
||||
run: python tests/ci/gpu_lock_exec.py --count ${{ matrix.info.num_gpus }} -- << config.test_executor | default('python') >> tests/${{ matrix.info.test_file }}
|
||||
<% endfor %>
|
||||
|
||||
- name: Post-test cleanup
|
||||
if: always()
|
||||
shell: bash
|
||||
run: |
|
||||
pkill -9 -f 'ray::' 2>/dev/null || true
|
||||
pkill -9 -f raylet 2>/dev/null || true
|
||||
ray stop --force 2>/dev/null || true
|
||||
rm -rf /tmp/ray/* 2>/dev/null || true
|
||||
|
||||
<% endfor %>
|
||||
|
||||
@@ -136,7 +136,15 @@ def log_rollout_data(
|
||||
# NOTE: Here we have to do the clone().detach(), otherwise the tensor will be
|
||||
# modified in place and will cause problem for the next rollout.
|
||||
val = torch.cat(val).clone().detach()
|
||||
if key in ["log_probs", "ref_log_probs", "rollout_log_probs", "returns", "advantages", "values"]:
|
||||
if key in [
|
||||
"log_probs",
|
||||
"ref_log_probs",
|
||||
"rollout_log_probs",
|
||||
"returns",
|
||||
"advantages",
|
||||
"values",
|
||||
"entropy",
|
||||
]:
|
||||
sum_of_sample_mean = get_sum_of_sample_mean(
|
||||
total_lengths,
|
||||
response_lengths,
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
import os
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
MODEL_NAME = "Qwen3-0.6B"
|
||||
|
||||
|
||||
FEW_GPU = U.get_bool_env_var("MILES_TEST_FEW_GPU", "1")
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"huggingface-cli download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.hf_download_dataset("zhuzilin/gsm8k")
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/gsm8k/train.parquet "
|
||||
"--input-key messages "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type math "
|
||||
# NOTE cannot be exactly multiple of eval-interval, since async causes some offsets
|
||||
f"--num-rollout {3000 if U.get_env_enable_infinite_run() else 65} "
|
||||
"--rollout-batch-size 32 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 1024 "
|
||||
"--rollout-temperature 1 "
|
||||
"--over-sampling-batch-size 64 "
|
||||
"--dynamic-sampling-filter-path miles.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std "
|
||||
"--global-batch-size 256 "
|
||||
)
|
||||
|
||||
eval_args = (
|
||||
"--eval-interval 20 "
|
||||
"--eval-prompt-data gsm8k /root/datasets/gsm8k/test.parquet "
|
||||
"--n-samples-per-eval-prompt 1 "
|
||||
"--eval-max-response-len 1024 "
|
||||
"--eval-top-k 1 "
|
||||
)
|
||||
|
||||
grpo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
# "--use-kl-loss "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type low_var_kl "
|
||||
"--kl-coef 0.00 "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 0.2 "
|
||||
"--eps-clip-high 0.28 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = "--rollout-num-gpus-per-engine 1 " "--sglang-enable-metrics "
|
||||
|
||||
misc_args = (
|
||||
"--actor-num-nodes 1 "
|
||||
f"--actor-num-gpus-per-node {1 if FEW_GPU else 2} "
|
||||
f"--rollout-num-gpus {1 if FEW_GPU else 2} "
|
||||
"--train-backend fsdp "
|
||||
)
|
||||
|
||||
ci_args = (
|
||||
"--ci-test "
|
||||
"--ci-disable-kl-checker "
|
||||
"--ci-metric-checker-key eval/gsm8k "
|
||||
"--ci-metric-checker-threshold 0.71 " # loose threshold at 60 step
|
||||
)
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{grpo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{eval_args} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=2 if FEW_GPU else 4,
|
||||
megatron_model_type=None,
|
||||
train_script="train_async.py",
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,155 @@
|
||||
import os
|
||||
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
MODEL_NAME = "Qwen3-0.6B"
|
||||
MODEL_TYPE = "qwen3-0.6B"
|
||||
NUM_GPUS = 4
|
||||
CP_SIZE = 1
|
||||
MEGATRON_TP_SIZE = 1
|
||||
MEGATRON_PP_SIZE = 1
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.hf_download_dataset("zhuzilin/dapo-math-17k")
|
||||
|
||||
U.convert_checkpoint(
|
||||
model_name=MODEL_NAME,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
dir_dst="/root/models",
|
||||
)
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/"
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
|
||||
"--input-key prompt "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type deepscaler "
|
||||
"--num-rollout 1 "
|
||||
"--rollout-batch-size 8 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 8192 "
|
||||
"--rollout-temperature 1 "
|
||||
"--global-batch-size 64 "
|
||||
"--use-dynamic-batch-size "
|
||||
"--max-tokens-per-gpu 8192 "
|
||||
)
|
||||
|
||||
ppo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type k1 "
|
||||
"--kl-coef 0.00 "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 4e-4 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 1 " "--sglang-chunked-prefill-size 4096 " "--sglang-mem-fraction-static 0.75 "
|
||||
)
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
misc_args = "--actor-num-nodes 1 " "--colocate " f"--actor-num-gpus-per-node {NUM_GPUS} "
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{ppo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
debug_data_path = "test_rollout_data_megatron_fsdp_align.pt"
|
||||
grad_norm_path = "grad_norm_fsdp.pt"
|
||||
|
||||
fsdp_args = (
|
||||
"--train-backend fsdp "
|
||||
"--attn-implementation flash_attention_2 "
|
||||
"--gradient-checkpointing "
|
||||
f"--context-parallel-size {CP_SIZE} "
|
||||
f"--update-weight-buffer-size {512 * 1024 * 1024} "
|
||||
"""--train-env-vars '{"PYTORCH_CUDA_ALLOC_CONF":"expandable_segments:True"}' """
|
||||
)
|
||||
|
||||
try:
|
||||
U.execute_train(
|
||||
train_args=train_args + (f"{fsdp_args}" f"--save-debug-rollout-data {debug_data_path} "),
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=None,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args
|
||||
+ (
|
||||
f"{fsdp_args}"
|
||||
f"--load-debug-rollout-data {debug_data_path} "
|
||||
f"--ci-save-grad-norm {grad_norm_path} "
|
||||
"--debug-train-only "
|
||||
),
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=None,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args
|
||||
+ (
|
||||
f"--ref-load /root/models/{MODEL_NAME}_torch_dist "
|
||||
f"--tensor-model-parallel-size {MEGATRON_TP_SIZE} "
|
||||
"--sequence-parallel "
|
||||
f"--pipeline-model-parallel-size {MEGATRON_PP_SIZE} "
|
||||
f"--context-parallel-size {CP_SIZE} "
|
||||
"--expert-model-parallel-size 1 "
|
||||
"--expert-tensor-parallel-size 1 "
|
||||
"--recompute-granularity full "
|
||||
"--recompute-method uniform "
|
||||
"--recompute-num-layers 1 "
|
||||
"--train-memory-margin-bytes 3221225472 "
|
||||
f"--load-debug-rollout-data {debug_data_path} "
|
||||
f"--ci-load-grad-norm {grad_norm_path} "
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
"--attention-backend flash "
|
||||
"--debug-train-only "
|
||||
),
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
)
|
||||
|
||||
finally:
|
||||
if os.path.exists(grad_norm_path):
|
||||
os.remove(grad_norm_path)
|
||||
if os.path.exists(debug_data_path):
|
||||
os.remove(debug_data_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,138 @@
|
||||
import os
|
||||
from argparse import ArgumentParser
|
||||
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
|
||||
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
|
||||
TIGHT_HOST_MEMORY = bool(int(os.environ.get("MILES_TEST_TIGHT_HOST_MEMORY", "1")))
|
||||
|
||||
MODEL_NAME = "Qwen3-4B"
|
||||
MODEL_TYPE = "qwen3-4B"
|
||||
NUM_GPUS = 8
|
||||
|
||||
|
||||
parser = ArgumentParser()
|
||||
parser.add_argument("--async-save", action="store_true", help="Whether to test async save/load.")
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.exec_command(f"rm -rf /root/models/{MODEL_NAME}_miles")
|
||||
U.hf_download_dataset("zhuzilin/dapo-math-17k")
|
||||
U.hf_download_dataset("zhuzilin/aime-2024")
|
||||
|
||||
U.convert_checkpoint(
|
||||
model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS, dir_dst="/root/models"
|
||||
)
|
||||
|
||||
|
||||
def execute(mode: str = ""):
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}_torch_dist "
|
||||
if mode == "save":
|
||||
ckpt_args += f"--save /root/models/{MODEL_NAME}_miles "
|
||||
ckpt_args += "--save-interval 2 "
|
||||
elif mode == "async_save":
|
||||
ckpt_args += f"--save /root/models/{MODEL_NAME}_miles "
|
||||
ckpt_args += "--save-interval 2 "
|
||||
ckpt_args += "--async-save "
|
||||
elif mode == "load":
|
||||
ckpt_args += f"--load /root/models/{MODEL_NAME}_miles "
|
||||
ckpt_args += "--ckpt-step 1 "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
|
||||
"--input-key prompt "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type deepscaler "
|
||||
"--num-rollout 3 "
|
||||
"--rollout-batch-size 4 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 1024 "
|
||||
"--rollout-temperature 0.8 "
|
||||
"--global-batch-size 32 "
|
||||
"--balance-data "
|
||||
)
|
||||
|
||||
perf_args = (
|
||||
"--tensor-model-parallel-size 2 "
|
||||
"--sequence-parallel "
|
||||
"--pipeline-model-parallel-size 1 "
|
||||
"--context-parallel-size 2 "
|
||||
"--recompute-granularity full "
|
||||
"--recompute-method uniform "
|
||||
"--recompute-num-layers 1 "
|
||||
"--use-dynamic-batch-size "
|
||||
f"--max-tokens-per-gpu {2048 if TIGHT_HOST_MEMORY else 16384} "
|
||||
)
|
||||
|
||||
ppo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type k1 "
|
||||
"--kl-coef 0.00 "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 0.2 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
"--optimizer-cpu-offload "
|
||||
"--overlap-cpu-optimizer-d2h-h2d "
|
||||
"--use-precision-aware-optimizer "
|
||||
)
|
||||
|
||||
sglang_args = "--rollout-num-gpus-per-engine 2 --sglang-mem-fraction-static 0.8 --sglang-cuda-graph-bs 1 2 4 8 16 "
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
misc_args = (
|
||||
# default dropout in megatron is 0.1
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
# should be good for model performance
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
# need to comment this when using model with MLA
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
"--actor-num-gpus-per-node 8 "
|
||||
"--colocate "
|
||||
)
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{ppo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{perf_args} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
args = parser.parse_args()
|
||||
# TODO also use typer
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute("save" if not args.async_save else "async_save")
|
||||
execute("load")
|
||||
@@ -0,0 +1,113 @@
|
||||
import os
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
|
||||
NUM_GPUS = 2
|
||||
|
||||
MODEL_NAME = "Qwen3-4B"
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.hf_download_dataset("zhuzilin/dapo-math-17k")
|
||||
U.hf_download_dataset("zhuzilin/aime-2024")
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
|
||||
"--input-key prompt "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type math "
|
||||
"--num-rollout 3 "
|
||||
"--rollout-batch-size 8 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 4096 "
|
||||
"--rollout-temperature 1 "
|
||||
"--global-batch-size 32 "
|
||||
)
|
||||
|
||||
eval_args = (
|
||||
f"{'--eval-interval 20 ' if ENABLE_EVAL else ''}"
|
||||
"--eval-prompt-data aime /root/datasets/aime-2024/aime-2024.jsonl "
|
||||
"--n-samples-per-eval-prompt 1 "
|
||||
"--eval-max-response-len 4096 "
|
||||
"--eval-top-p 0.7 "
|
||||
)
|
||||
|
||||
fsdp_args = "--train-backend fsdp " "--update-weight-buffer-size 536870912 "
|
||||
|
||||
grpo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type low_var_kl "
|
||||
"--kl-coef 0.00 "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 0.2 "
|
||||
"--eps-clip-high 0.28 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 1 "
|
||||
"--sglang-decode-log-interval 1000 "
|
||||
"--sglang-enable-metrics "
|
||||
"--sglang-enable-deterministic-inference "
|
||||
"--sglang-rl-on-policy-target fsdp "
|
||||
"--sglang-attention-backend fa3 "
|
||||
"--attn-implementation flash_attention_3 "
|
||||
"--deterministic-mode "
|
||||
"--true-on-policy-mode "
|
||||
)
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
misc_args = "--actor-num-nodes 1 " f"--actor-num-gpus-per-node {NUM_GPUS} " "--colocate "
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{grpo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{fsdp_args} "
|
||||
f"{eval_args} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
extra_env_vars = {
|
||||
"NCCL_ALGO": "allreduce:tree",
|
||||
"NVTE_ALLOW_NONDETERMINISTIC_ALGO": "0",
|
||||
"CUBLAS_WORKSPACE_CONFIG": ":4096:8",
|
||||
"CUDA_DEVICE_MAX_CONNECTIONS": "1",
|
||||
"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1",
|
||||
}
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=None,
|
||||
extra_env_vars=extra_env_vars,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,112 @@
|
||||
import os
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
|
||||
NUM_GPUS = 8
|
||||
|
||||
MODEL_NAME = "Qwen3-VL-4B-Instruct"
|
||||
DATASET_NAME = "chenhegu/geo3k_imgurl"
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.hf_download_dataset(DATASET_NAME)
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/geo3k_imgurl/train.parquet "
|
||||
"--input-key problem "
|
||||
"--label-key answer "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type math "
|
||||
"--num-rollout 3 "
|
||||
"--rollout-batch-size 8 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 4096 "
|
||||
"--rollout-temperature 1 "
|
||||
"--global-batch-size 32 "
|
||||
)
|
||||
|
||||
# multimodal keys required for vlm datasets
|
||||
multimodal_args = '--multimodal-keys \'{"image": "images"}\' '
|
||||
|
||||
eval_args = (
|
||||
f"{'--eval-interval 20 ' if ENABLE_EVAL else ''}"
|
||||
"--eval-prompt-data geo3k /root/datasets/geo3k_imgurl/test.parquet "
|
||||
"--n-samples-per-eval-prompt 1 "
|
||||
"--eval-max-response-len 4096 "
|
||||
)
|
||||
|
||||
fsdp_args = "--train-backend fsdp " "--gradient-checkpointing " "--update-weight-buffer-size 536870912 "
|
||||
|
||||
grpo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type low_var_kl "
|
||||
"--kl-coef 0.00 "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 0.2 "
|
||||
"--eps-clip-high 0.28 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 1 "
|
||||
"--sglang-mem-fraction-static 0.6 "
|
||||
"--sglang-decode-log-interval 1000 "
|
||||
"--sglang-enable-metrics "
|
||||
"--sglang-attention-backend fa3 "
|
||||
"--attn-implementation flash_attention_3 "
|
||||
)
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
misc_args = "--actor-num-nodes 1 " f"--actor-num-gpus-per-node {NUM_GPUS} " "--colocate "
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{multimodal_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{grpo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{fsdp_args} "
|
||||
f"{eval_args} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
extra_env_vars = {
|
||||
"CUDA_DEVICE_MAX_CONNECTIONS": "1",
|
||||
"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1",
|
||||
}
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=None,
|
||||
extra_env_vars=extra_env_vars,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
prepare()
|
||||
os.environ.pop("http_proxy", None)
|
||||
os.environ.pop("https_proxy", None)
|
||||
os.environ.pop("HTTP_PROXY", None)
|
||||
os.environ.pop("HTTPS_PROXY", None)
|
||||
execute()
|
||||
@@ -0,0 +1,131 @@
|
||||
import os
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
|
||||
FEW_GPU = U.get_bool_env_var("MILES_TEST_FEW_GPU", "1")
|
||||
TIGHT_DEVICE_MEMORY = U.get_bool_env_var("MILES_TEST_TIGHT_DEVICE_MEMORY", "1")
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
NUM_GPUS = 2 if FEW_GPU else 4
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"huggingface-cli download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.hf_download_dataset("zhuzilin/gsm8k")
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}/ "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/gsm8k/train.parquet "
|
||||
"--input-key messages "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type math "
|
||||
f"--num-rollout {3000 if U.get_env_enable_infinite_run() else 250} "
|
||||
"--rollout-batch-size 32 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 1024 "
|
||||
"--rollout-temperature 1 "
|
||||
"--over-sampling-batch-size 64 "
|
||||
"--dynamic-sampling-filter-path miles.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std "
|
||||
"--global-batch-size 256 "
|
||||
)
|
||||
|
||||
eval_args = (
|
||||
"--eval-interval 20 "
|
||||
"--eval-prompt-data gsm8k /root/datasets/gsm8k/test.parquet "
|
||||
"--n-samples-per-eval-prompt 1 "
|
||||
"--eval-max-response-len 1024 "
|
||||
"--eval-top-k 1 "
|
||||
)
|
||||
|
||||
perf_args = (
|
||||
"--tensor-model-parallel-size 1 "
|
||||
"--sequence-parallel "
|
||||
"--pipeline-model-parallel-size 1 "
|
||||
"--context-parallel-size 1 "
|
||||
"--expert-model-parallel-size 1 "
|
||||
"--expert-tensor-parallel-size 1 "
|
||||
# "--micro-batch-size 1 "
|
||||
"--use-dynamic-batch-size "
|
||||
"--max-tokens-per-gpu 9216 "
|
||||
)
|
||||
|
||||
grpo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
"--use-kl-loss "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type low_var_kl "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 0.2 "
|
||||
"--eps-clip-high 0.28 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 1 "
|
||||
f"--sglang-mem-fraction-static {0.6 if TIGHT_DEVICE_MEMORY else 0.7} "
|
||||
"--sglang-enable-metrics "
|
||||
)
|
||||
|
||||
ci_args = (
|
||||
"--ci-test "
|
||||
"--ci-disable-kl-checker "
|
||||
"--ci-metric-checker-key eval/gsm8k "
|
||||
"--ci-metric-checker-threshold 0.55 " # loose threshold at 250 step
|
||||
)
|
||||
|
||||
misc_args = (
|
||||
# default dropout in megatron is 0.1
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
# should be good for model performance
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
# need to comment this when using model with MLA
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
f"--actor-num-gpus-per-node {2 if FEW_GPU else 4} "
|
||||
"--colocate "
|
||||
"--megatron-to-hf-mode bridge "
|
||||
)
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{grpo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{perf_args} "
|
||||
f"{eval_args} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,131 @@
|
||||
import os
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
FEW_GPU = U.get_bool_env_var("MILES_TEST_FEW_GPU", "1")
|
||||
TIGHT_DEVICE_MEMORY = U.get_bool_env_var("MILES_TEST_TIGHT_DEVICE_MEMORY", "1")
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
NUM_GPUS = 2 if FEW_GPU else 4
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"huggingface-cli download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.hf_download_dataset("zhuzilin/gsm8k")
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}/ "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/gsm8k/train.parquet "
|
||||
"--input-key messages "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type math "
|
||||
f"--num-rollout {3000 if U.get_env_enable_infinite_run() else 250} "
|
||||
"--rollout-batch-size 32 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 1024 "
|
||||
"--rollout-temperature 1 "
|
||||
"--over-sampling-batch-size 64 "
|
||||
"--dynamic-sampling-filter-path miles.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std "
|
||||
"--global-batch-size 256 "
|
||||
)
|
||||
|
||||
eval_args = (
|
||||
"--eval-interval 20 "
|
||||
"--eval-prompt-data gsm8k /root/datasets/gsm8k/test.parquet "
|
||||
"--n-samples-per-eval-prompt 1 "
|
||||
"--eval-max-response-len 1024 "
|
||||
"--eval-top-k 1 "
|
||||
)
|
||||
|
||||
perf_args = (
|
||||
"--tensor-model-parallel-size 1 "
|
||||
"--sequence-parallel "
|
||||
"--pipeline-model-parallel-size 1 "
|
||||
"--context-parallel-size 1 "
|
||||
"--expert-model-parallel-size 1 "
|
||||
"--expert-tensor-parallel-size 1 "
|
||||
# "--micro-batch-size 1 "
|
||||
"--use-dynamic-batch-size "
|
||||
"--max-tokens-per-gpu 9216 "
|
||||
)
|
||||
|
||||
grpo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
"--use-kl-loss "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type low_var_kl "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 0.2 "
|
||||
"--eps-clip-high 0.28 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 1 "
|
||||
f"--sglang-mem-fraction-static {0.6 if TIGHT_DEVICE_MEMORY else 0.7} "
|
||||
"--sglang-enable-metrics "
|
||||
)
|
||||
|
||||
ci_args = (
|
||||
"--ci-test "
|
||||
"--ci-disable-kl-checker "
|
||||
"--ci-metric-checker-key eval/gsm8k "
|
||||
"--ci-metric-checker-threshold 0.55 " # loose threshold at 250 step
|
||||
)
|
||||
|
||||
misc_args = (
|
||||
# default dropout in megatron is 0.1
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
# should be good for model performance
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
# need to comment this when using model with MLA
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
f"--actor-num-gpus-per-node {1 if FEW_GPU else 2} "
|
||||
f"--rollout-num-gpus {1 if FEW_GPU else 2} "
|
||||
"--megatron-to-hf-mode bridge "
|
||||
)
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{grpo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{perf_args} "
|
||||
f"{eval_args} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
train_script="train_async.py",
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,147 @@
|
||||
"""End-to-end test for MTP-only gradient verification.
|
||||
|
||||
This test verifies that when MTP training is enabled and all outputs are truncated
|
||||
(due to very short max response length), only MTP parameters receive non-zero
|
||||
gradients while all other model parameters have zero gradients.
|
||||
|
||||
This validates that the MTP loss computation correctly isolates gradient flow
|
||||
to only the MTP layers when the main model loss is zero (due to truncation).
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
|
||||
MODEL_NAME = "MiMo-7B-RL"
|
||||
MODEL_TYPE = "mimo-7B-rl"
|
||||
NUM_GPUS = 8
|
||||
|
||||
|
||||
def prepare():
|
||||
"""Download model and convert checkpoint with MTP layers."""
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"hf download XiaomiMiMo/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.hf_download_dataset("zhuzilin/dapo-math-17k")
|
||||
|
||||
# Convert checkpoint with MTP layers enabled
|
||||
U.convert_checkpoint(
|
||||
model_name=MODEL_NAME,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
extra_args=" --mtp-num-layers 1",
|
||||
dir_dst="/root/models",
|
||||
)
|
||||
|
||||
|
||||
def execute():
|
||||
"""Run training with MTP enabled and very short output length to cause truncation."""
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}_torch_dist "
|
||||
|
||||
# Use very short rollout-max-response-len to ensure all outputs are truncated
|
||||
# This should result in zero loss for the main model, leaving only MTP loss
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
|
||||
"--input-key prompt "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type deepscaler "
|
||||
"--num-rollout 1 "
|
||||
"--rollout-batch-size 4 "
|
||||
"--n-samples-per-prompt 2 "
|
||||
# Very short max response length to cause all outputs to be truncated
|
||||
"--rollout-max-response-len 128 "
|
||||
"--rollout-temperature 0.8 "
|
||||
"--global-batch-size 8 "
|
||||
)
|
||||
|
||||
perf_args = (
|
||||
"--tensor-model-parallel-size 2 "
|
||||
"--sequence-parallel "
|
||||
"--pipeline-model-parallel-size 1 "
|
||||
"--context-parallel-size 1 "
|
||||
"--recompute-granularity full "
|
||||
"--recompute-method uniform "
|
||||
"--recompute-num-layers 1 "
|
||||
"--use-dynamic-batch-size "
|
||||
"--max-tokens-per-gpu 4096 "
|
||||
)
|
||||
|
||||
grpo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type low_var_kl "
|
||||
"--kl-coef 0.00 "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 0.2 "
|
||||
"--eps-clip-high 0.28 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 2 "
|
||||
"--rollout-num-gpus 8 "
|
||||
"--sglang-mem-fraction-static 0.8 "
|
||||
"--sglang-enable-metrics "
|
||||
"--sglang-speculative-algorithm EAGLE "
|
||||
"--sglang-speculative-num-steps 2 "
|
||||
"--sglang-speculative-eagle-topk 1 "
|
||||
"--sglang-speculative-num-draft-tokens 3 "
|
||||
)
|
||||
|
||||
# Enable MTP training with loss scaling
|
||||
mtp_args = "--mtp-num-layers 1 " "--enable-mtp-training " "--mtp-loss-scaling-factor 0.2 "
|
||||
|
||||
ci_args = (
|
||||
"--ci-test "
|
||||
"--ci-disable-kl-checker "
|
||||
# MTP grad check is automatically triggered when ci_test and enable_mtp_training are both set
|
||||
)
|
||||
|
||||
misc_args = (
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
"--actor-num-gpus-per-node 8 "
|
||||
"--colocate "
|
||||
)
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{grpo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{perf_args} "
|
||||
f"{sglang_args} "
|
||||
f"{mtp_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
prepare()
|
||||
# Remove proxy settings that might interfere with local operations
|
||||
for key in ["http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"]:
|
||||
os.environ.pop(key, None)
|
||||
execute()
|
||||
@@ -0,0 +1,124 @@
|
||||
import os
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
|
||||
TIGHT_HOST_MEMORY = bool(int(os.environ.get("MILES_TEST_TIGHT_HOST_MEMORY", "1")))
|
||||
|
||||
MODEL_NAME = "Moonlight-16B-A3B-Instruct"
|
||||
MODEL_TYPE = "moonlight"
|
||||
NUM_GPUS = 8
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(
|
||||
"hf download moonshotai/Moonlight-16B-A3B-Instruct --local-dir /root/models/Moonlight-16B-A3B-Instruct"
|
||||
)
|
||||
U.hf_download_dataset("zhuzilin/dapo-math-17k")
|
||||
U.hf_download_dataset("zhuzilin/aime-2024")
|
||||
|
||||
U.convert_checkpoint(model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS)
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} " f"--ref-load /root/{MODEL_NAME}_torch_dist "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
|
||||
"--input-key prompt "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type math "
|
||||
"--num-rollout 3 "
|
||||
"--rollout-batch-size 8 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 4096 "
|
||||
"--rollout-temperature 1 "
|
||||
"--global-batch-size 32 "
|
||||
)
|
||||
|
||||
eval_args = (
|
||||
f"{'--eval-interval 20 ' if ENABLE_EVAL else ''}"
|
||||
"--eval-prompt-data aime /root/datasets/aime-2024/aime-2024.jsonl "
|
||||
"--n-samples-per-eval-prompt 1 "
|
||||
"--eval-max-response-len 4096 "
|
||||
"--eval-top-k 1 "
|
||||
)
|
||||
|
||||
perf_args = (
|
||||
"--tensor-model-parallel-size 2 "
|
||||
"--sequence-parallel "
|
||||
"--pipeline-model-parallel-size 1 "
|
||||
"--context-parallel-size 2 "
|
||||
"--expert-model-parallel-size 8 "
|
||||
"--expert-tensor-parallel-size 1 "
|
||||
"--recompute-granularity full "
|
||||
"--recompute-method uniform "
|
||||
"--recompute-num-layers 1 "
|
||||
"--use-dynamic-batch-size "
|
||||
f"--max-tokens-per-gpu {2048 if TIGHT_HOST_MEMORY else 2048} "
|
||||
)
|
||||
|
||||
grpo_args = (
|
||||
"--advantage-estimator gspo "
|
||||
f"{'' if TIGHT_HOST_MEMORY else '--use-kl-loss '}"
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type low_var_kl "
|
||||
"--kl-coef 0.00 "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 4e-4 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 2 " "--sglang-mem-fraction-static 0.8 " "--sglang-max-running-requests 512 "
|
||||
)
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
misc_args = (
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
"--actor-num-gpus-per-node 8 "
|
||||
"--colocate "
|
||||
)
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{grpo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{perf_args} "
|
||||
f"{eval_args} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,127 @@
|
||||
import os
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
ENABLE_EVAL = U.get_bool_env_var("MILES_TEST_ENABLE_EVAL", "1")
|
||||
TIGHT_DEVICE_MEMORY = U.get_bool_env_var("MILES_TEST_TIGHT_DEVICE_MEMORY", "1")
|
||||
|
||||
MODEL_NAME = "GLM-Z1-9B-0414"
|
||||
MODEL_TYPE = "glm4-9B"
|
||||
NUM_GPUS = 8
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command("hf download zai-org/GLM-Z1-9B-0414 --local-dir /root/models/GLM-Z1-9B-0414")
|
||||
U.hf_download_dataset("zhuzilin/dapo-math-17k")
|
||||
U.hf_download_dataset("zhuzilin/aime-2024")
|
||||
|
||||
U.convert_checkpoint(model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS)
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/{MODEL_NAME}_torch_dist "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
|
||||
"--input-key prompt "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type deepscaler "
|
||||
"--num-rollout 3 "
|
||||
"--rollout-batch-size 8 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 8192 "
|
||||
"--rollout-temperature 1 "
|
||||
"--global-batch-size 32 "
|
||||
"--balance-data "
|
||||
)
|
||||
|
||||
eval_args = (
|
||||
f"{'--eval-interval 20 ' if ENABLE_EVAL else ''}"
|
||||
"--eval-prompt-data aime24 /root/datasets/aime-2024/aime-2024.jsonl "
|
||||
"--n-samples-per-eval-prompt 1 "
|
||||
"--eval-max-response-len 16384 "
|
||||
"--eval-top-k 1 "
|
||||
)
|
||||
|
||||
perf_args = (
|
||||
"--tensor-model-parallel-size 2 "
|
||||
"--sequence-parallel "
|
||||
"--pipeline-model-parallel-size 1 "
|
||||
"--context-parallel-size 2 "
|
||||
"--expert-model-parallel-size 1 "
|
||||
"--expert-tensor-parallel-size 1 "
|
||||
"--recompute-granularity full "
|
||||
"--recompute-method uniform "
|
||||
"--recompute-num-layers 1 "
|
||||
"--use-dynamic-batch-size "
|
||||
f"--max-tokens-per-gpu {2048 if TIGHT_DEVICE_MEMORY else 4608} "
|
||||
)
|
||||
|
||||
grpo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
"--use-kl-loss "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type low_var_kl "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 0.2 "
|
||||
"--eps-clip-high 0.28 "
|
||||
"--use-tis "
|
||||
"--calculate-per-token-loss "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = "--rollout-num-gpus-per-engine 2 " "--use-miles-router "
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
misc_args = (
|
||||
# default dropout in megatron is 0.1
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
# should be good for model performance
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
# need to comment this when using model with MLA
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
"--actor-num-gpus-per-node 4 "
|
||||
"--rollout-num-gpus 4 "
|
||||
)
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{grpo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{perf_args} "
|
||||
f"{eval_args} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# TODO also use typer
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,151 @@
|
||||
import os
|
||||
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
|
||||
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
|
||||
TIGHT_HOST_MEMORY = bool(int(os.environ.get("MILES_TEST_TIGHT_HOST_MEMORY", "1")))
|
||||
USE_DEEPEP = bool(int(os.environ.get("MILES_TEST_USE_DEEPEP", "1")))
|
||||
USE_FP8_ROLLOUT = bool(int(os.environ.get("MILES_TEST_USE_FP8_ROLLOUT", "1")))
|
||||
|
||||
MODEL_NAME = "Qwen3-30B-A3B"
|
||||
MODEL_TYPE = "qwen3-30B-A3B"
|
||||
NUM_GPUS = 8
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command("hf download Qwen/Qwen3-30B-A3B --local-dir /root/models/Qwen3-30B-A3B")
|
||||
U.exec_command("hf download Qwen/Qwen3-30B-A3B-FP8 --local-dir /root/models/Qwen3-30B-A3B-FP8")
|
||||
U.hf_download_dataset("zhuzilin/dapo-math-17k")
|
||||
U.hf_download_dataset("zhuzilin/aime-2024")
|
||||
|
||||
U.convert_checkpoint(model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS)
|
||||
|
||||
|
||||
def execute():
|
||||
if USE_FP8_ROLLOUT:
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}-FP8 " f"--ref-load /root/{MODEL_NAME}_torch_dist "
|
||||
else:
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} " f"--ref-load /root/{MODEL_NAME}_torch_dist "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
|
||||
"--input-key prompt "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type deepscaler "
|
||||
"--num-rollout 3 "
|
||||
"--rollout-batch-size 8 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 8192 "
|
||||
"--rollout-temperature 1 "
|
||||
"--global-batch-size 32 "
|
||||
"--balance-data "
|
||||
)
|
||||
|
||||
eval_args = (
|
||||
f"{'--eval-interval 20 ' if ENABLE_EVAL else ''}"
|
||||
"--eval-prompt-data aime24 /root/datasets/aime-2024/aime-2024.jsonl "
|
||||
"--n-samples-per-eval-prompt 1 "
|
||||
"--eval-max-response-len 16384 "
|
||||
"--eval-top-k 1 "
|
||||
)
|
||||
|
||||
perf_args = (
|
||||
"--tensor-model-parallel-size 4 "
|
||||
"--sequence-parallel "
|
||||
"--pipeline-model-parallel-size 1 "
|
||||
"--context-parallel-size 2 "
|
||||
"--expert-model-parallel-size 8 "
|
||||
"--expert-tensor-parallel-size 1 "
|
||||
"--recompute-granularity full "
|
||||
"--recompute-method uniform "
|
||||
"--recompute-num-layers 1 "
|
||||
"--use-dynamic-batch-size "
|
||||
f"--max-tokens-per-gpu {2048 if TIGHT_HOST_MEMORY else 16384} "
|
||||
)
|
||||
|
||||
grpo_args = (
|
||||
"--advantage-estimator gspo "
|
||||
f"{'' if TIGHT_HOST_MEMORY else '--use-kl-loss '}"
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type low_var_kl "
|
||||
"--kl-coef 0.00 "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 4e-4 "
|
||||
"--use-tis "
|
||||
"--use-routing-replay "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
"--optimizer-cpu-offload "
|
||||
"--overlap-cpu-optimizer-d2h-h2d "
|
||||
"--use-precision-aware-optimizer "
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 8 "
|
||||
"--sglang-mem-fraction-static 0.8 "
|
||||
"--sglang-max-running-requests 512 "
|
||||
"--sglang-enable-metrics "
|
||||
)
|
||||
|
||||
if USE_DEEPEP:
|
||||
sglang_args += "--sglang-moe-a2a-backend deepep --sglang-deepep-mode auto "
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
misc_args = (
|
||||
# default dropout in megatron is 0.1
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
# should be good for model performance
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
# need to comment this when using model with MLA
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
"--actor-num-gpus-per-node 8 "
|
||||
"--colocate "
|
||||
)
|
||||
|
||||
if USE_DEEPEP:
|
||||
misc_args += "--moe-token-dispatcher-type flex --moe-enable-deepep "
|
||||
else:
|
||||
misc_args += "--moe-token-dispatcher-type alltoall "
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{grpo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{perf_args} "
|
||||
f"{eval_args} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# TODO also use typer
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,134 @@
|
||||
import os
|
||||
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
|
||||
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
|
||||
TIGHT_HOST_MEMORY = bool(int(os.environ.get("MILES_TEST_TIGHT_HOST_MEMORY", "1")))
|
||||
|
||||
MODEL_NAME = "Qwen3-4B"
|
||||
MODEL_TYPE = "qwen3-4B"
|
||||
NUM_GPUS = 8
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command("hf download Qwen/Qwen3-4B --local-dir /root/models/Qwen3-4B")
|
||||
U.hf_download_dataset("zhuzilin/dapo-math-17k")
|
||||
U.hf_download_dataset("zhuzilin/aime-2024")
|
||||
|
||||
U.convert_checkpoint(model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS)
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/{MODEL_NAME}_torch_dist "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
|
||||
"--input-key prompt "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type deepscaler "
|
||||
"--num-rollout 3 "
|
||||
"--rollout-batch-size 8 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 8192 "
|
||||
"--rollout-temperature 0.8 "
|
||||
"--global-batch-size 32 "
|
||||
"--balance-data "
|
||||
)
|
||||
|
||||
eval_args = (
|
||||
f"{'--eval-interval 20 ' if ENABLE_EVAL else ''}"
|
||||
"--eval-prompt-data aime24 /root/datasets/aime-2024/aime-2024.jsonl "
|
||||
"--n-samples-per-eval-prompt 1 "
|
||||
"--eval-max-response-len 16384 "
|
||||
"--eval-top-k 1 "
|
||||
)
|
||||
|
||||
perf_args = (
|
||||
"--tensor-model-parallel-size 2 "
|
||||
"--sequence-parallel "
|
||||
"--pipeline-model-parallel-size 1 "
|
||||
"--context-parallel-size 2 "
|
||||
"--recompute-granularity full "
|
||||
"--recompute-method uniform "
|
||||
"--recompute-num-layers 1 "
|
||||
"--use-dynamic-batch-size "
|
||||
f"--max-tokens-per-gpu {2048 if TIGHT_HOST_MEMORY else 16384} "
|
||||
)
|
||||
|
||||
ppo_args = (
|
||||
"--advantage-estimator ppo "
|
||||
f"{'' if TIGHT_HOST_MEMORY else '--use-kl-loss '}"
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type k1 "
|
||||
"--kl-coef 0.00 "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 4e-4 "
|
||||
"--num-critic-only-steps 1 "
|
||||
"--normalize-advantages "
|
||||
"--critic-lr 1e-5 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 2 "
|
||||
"--rollout-num-gpus 8 "
|
||||
"--sglang-mem-fraction-static 0.8 "
|
||||
"--sglang-max-running-requests 512 "
|
||||
"--sglang-enable-metrics "
|
||||
)
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
misc_args = (
|
||||
# default dropout in megatron is 0.1
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
# should be good for model performance
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
# need to comment this when using model with MLA
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
"--actor-num-gpus-per-node 4 "
|
||||
"--colocate "
|
||||
)
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{ppo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{perf_args} "
|
||||
f"{eval_args} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# TODO also use typer
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,155 @@
|
||||
import os
|
||||
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
MODEL_NAME = "Qwen3-0.6B"
|
||||
MODEL_TYPE = "qwen3-0.6B"
|
||||
NUM_GPUS = 4
|
||||
CP_SIZE = 1
|
||||
MEGATRON_TP_SIZE = 1
|
||||
MEGATRON_PP_SIZE = 1
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.hf_download_dataset("zhuzilin/dapo-math-17k")
|
||||
|
||||
U.convert_checkpoint(
|
||||
model_name=MODEL_NAME,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
dir_dst="/root/models",
|
||||
)
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/"
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
|
||||
"--input-key prompt "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type deepscaler "
|
||||
"--num-rollout 1 "
|
||||
"--rollout-batch-size 8 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 8192 "
|
||||
"--rollout-temperature 1 "
|
||||
"--global-batch-size 64 "
|
||||
"--use-dynamic-batch-size "
|
||||
"--max-tokens-per-gpu 8192 "
|
||||
)
|
||||
|
||||
ppo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type k1 "
|
||||
"--kl-coef 0.00 "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 4e-4 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 1 " "--sglang-chunked-prefill-size 4096 " "--sglang-mem-fraction-static 0.75 "
|
||||
)
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
misc_args = "--actor-num-nodes 1 " "--colocate " f"--actor-num-gpus-per-node {NUM_GPUS} "
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{ppo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
debug_data_path = "test_rollout_data_megatron_fsdp_align.pt"
|
||||
grad_norm_path = "grad_norm_fsdp.pt"
|
||||
|
||||
fsdp_args = (
|
||||
"--train-backend fsdp "
|
||||
"--attn-implementation flash_attention_2 "
|
||||
"--gradient-checkpointing "
|
||||
f"--context-parallel-size {CP_SIZE} "
|
||||
f"--update-weight-buffer-size {512 * 1024 * 1024} "
|
||||
"""--train-env-vars '{"PYTORCH_CUDA_ALLOC_CONF":"expandable_segments:True"}' """
|
||||
)
|
||||
|
||||
try:
|
||||
U.execute_train(
|
||||
train_args=train_args + (f"{fsdp_args}" f"--save-debug-rollout-data {debug_data_path} "),
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=None,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args
|
||||
+ (
|
||||
f"{fsdp_args}"
|
||||
f"--load-debug-rollout-data {debug_data_path} "
|
||||
f"--ci-save-grad-norm {grad_norm_path} "
|
||||
"--debug-train-only "
|
||||
),
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=None,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args
|
||||
+ (
|
||||
f"--ref-load /root/models/{MODEL_NAME}_torch_dist "
|
||||
f"--tensor-model-parallel-size {MEGATRON_TP_SIZE} "
|
||||
"--sequence-parallel "
|
||||
f"--pipeline-model-parallel-size {MEGATRON_PP_SIZE} "
|
||||
f"--context-parallel-size {CP_SIZE} "
|
||||
"--expert-model-parallel-size 1 "
|
||||
"--expert-tensor-parallel-size 1 "
|
||||
"--recompute-granularity full "
|
||||
"--recompute-method uniform "
|
||||
"--recompute-num-layers 1 "
|
||||
"--train-memory-margin-bytes 3221225472 "
|
||||
f"--load-debug-rollout-data {debug_data_path} "
|
||||
f"--ci-load-grad-norm {grad_norm_path} "
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
"--attention-backend flash "
|
||||
"--debug-train-only "
|
||||
),
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
)
|
||||
|
||||
finally:
|
||||
if os.path.exists(grad_norm_path):
|
||||
os.remove(grad_norm_path)
|
||||
if os.path.exists(debug_data_path):
|
||||
os.remove(debug_data_path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,138 @@
|
||||
import os
|
||||
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
|
||||
ENABLE_EVAL = bool(int(os.environ.get("MILES_TEST_ENABLE_EVAL", "1")))
|
||||
TIGHT_HOST_MEMORY = bool(int(os.environ.get("MILES_TEST_TIGHT_HOST_MEMORY", "1")))
|
||||
|
||||
MODEL_NAME = "Qwen3-0.6B"
|
||||
MODEL_TYPE = "qwen3-0.6B"
|
||||
NUM_GPUS = 8
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"hf download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.hf_download_dataset("zhuzilin/dapo-math-17k")
|
||||
|
||||
U.convert_checkpoint(
|
||||
model_name=MODEL_NAME, megatron_model_type=MODEL_TYPE, num_gpus_per_node=NUM_GPUS, dir_dst="/root/models"
|
||||
)
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}_torch_dist "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/dapo-math-17k/dapo-math-17k.jsonl "
|
||||
"--input-key prompt "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type deepscaler "
|
||||
"--num-rollout 1 "
|
||||
"--rollout-batch-size 4 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 8192 "
|
||||
"--rollout-temperature 0.8 "
|
||||
"--global-batch-size 32 "
|
||||
)
|
||||
|
||||
ppo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type k1 "
|
||||
"--kl-coef 0.00 "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 4e-4 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = "--rollout-num-gpus-per-engine 2 " "--rollout-num-gpus 8 " "--sglang-mem-fraction-static 0.8 "
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
misc_args = (
|
||||
# default dropout in megatron is 0.1
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
# should be good for model performance
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
# need to comment this when using model with MLA
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
"--colocate "
|
||||
)
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{ppo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
for i in range(2):
|
||||
U.execute_train(
|
||||
train_args=train_args
|
||||
+ (
|
||||
f"--save-debug-rollout-data data-{i}.pt "
|
||||
f"--ci-save-grad-norm grad_norms-{i}.pt "
|
||||
f"--actor-num-gpus-per-node {NUM_GPUS} "
|
||||
),
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
# 8 GPU CPU 1
|
||||
for num_gpus in [8, 4, 2]:
|
||||
remaining_gpus = num_gpus
|
||||
for tp_size in [1, 2, 4, 8]:
|
||||
remaining_gpus /= tp_size
|
||||
for pp_size in [1, 2, 4]:
|
||||
if remaining_gpus < pp_size:
|
||||
continue
|
||||
remaining_gpus /= pp_size
|
||||
for cp_size in [1, 2, 4, 8]:
|
||||
if remaining_gpus < cp_size:
|
||||
continue
|
||||
args = train_args + (
|
||||
f"--load-debug-rollout-data data-{i}.pt "
|
||||
f"--ci-load-grad-norm grad_norms-{i}.pt "
|
||||
f"--context-parallel-size {cp_size} "
|
||||
f"--tensor-model-parallel-size {tp_size} "
|
||||
f"--pipeline-model-parallel-size {pp_size} "
|
||||
"--sequence-parallel "
|
||||
f"--actor-num-gpus-per-node {num_gpus} "
|
||||
"--use-dynamic-batch-size "
|
||||
"--max-tokens-per-gpu 8192 "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=args,
|
||||
num_gpus_per_node=num_gpus,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
train_args += "--calculate-per-token-loss "
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
# TODO also use typer
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,129 @@
|
||||
import os
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
TIGHT_DEVICE_MEMORY = U.get_bool_env_var("MILES_TEST_TIGHT_DEVICE_MEMORY", "1")
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
NUM_GPUS = 4
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"huggingface-cli download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.hf_download_dataset("zhuzilin/gsm8k")
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}/ "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/gsm8k/train.parquet "
|
||||
"--input-key messages "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type math "
|
||||
"--num-rollout 3 "
|
||||
"--rollout-batch-size 8 "
|
||||
"--n-samples-per-prompt 4 "
|
||||
"--rollout-max-response-len 1024 "
|
||||
"--rollout-temperature 0.8 "
|
||||
"--over-sampling-batch-size 16 "
|
||||
"--dynamic-sampling-filter-path miles.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std "
|
||||
"--global-batch-size 32 "
|
||||
)
|
||||
|
||||
eval_args = (
|
||||
"--eval-interval 8 "
|
||||
"--eval-prompt-data gsm8k /root/datasets/gsm8k/test.parquet "
|
||||
"--n-samples-per-eval-prompt 1 "
|
||||
"--eval-max-response-len 1024 "
|
||||
"--eval-top-k 1 "
|
||||
)
|
||||
|
||||
perf_args = (
|
||||
"--tensor-model-parallel-size 1 "
|
||||
"--sequence-parallel "
|
||||
"--pipeline-model-parallel-size 1 "
|
||||
"--context-parallel-size 1 "
|
||||
"--expert-model-parallel-size 1 "
|
||||
"--expert-tensor-parallel-size 1 "
|
||||
"--use-dynamic-batch-size "
|
||||
"--max-tokens-per-gpu 9216 "
|
||||
)
|
||||
|
||||
grpo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
"--use-kl-loss "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type low_var_kl "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 0.2 "
|
||||
"--eps-clip-high 0.28 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 1 "
|
||||
f"--sglang-mem-fraction-static {0.55 if TIGHT_DEVICE_MEMORY else 0.65} "
|
||||
"--sglang-enable-metrics "
|
||||
)
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
fault_tolerance_args = (
|
||||
"--use-fault-tolerance "
|
||||
"--rollout-health-check-interval 5 "
|
||||
"--rollout-health-check-timeout 10 "
|
||||
"--rollout-health-check-first-wait 0 "
|
||||
)
|
||||
|
||||
misc_args = (
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
"--actor-num-gpus-per-node 1 "
|
||||
"--rollout-num-gpus 3 "
|
||||
"--megatron-to-hf-mode bridge "
|
||||
)
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{grpo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{perf_args} "
|
||||
f"{eval_args} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{fault_tolerance_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
train_script="train_async.py",
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,128 @@
|
||||
import os
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
TIGHT_DEVICE_MEMORY = U.get_bool_env_var("MILES_TEST_TIGHT_DEVICE_MEMORY", "1")
|
||||
|
||||
MODEL_NAME = "Qwen2.5-0.5B-Instruct"
|
||||
MODEL_TYPE = "qwen2.5-0.5B"
|
||||
NUM_GPUS = 4
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"huggingface-cli download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.hf_download_dataset("zhuzilin/gsm8k")
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME}/ " f"--ref-load /root/models/{MODEL_NAME}/ "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/gsm8k/train.parquet "
|
||||
"--input-key messages "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type math "
|
||||
"--num-rollout 3 "
|
||||
"--rollout-batch-size 8 "
|
||||
"--n-samples-per-prompt 4 "
|
||||
"--rollout-max-response-len 1024 "
|
||||
"--rollout-temperature 0.8 "
|
||||
"--over-sampling-batch-size 16 "
|
||||
"--dynamic-sampling-filter-path miles.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std "
|
||||
"--global-batch-size 32 "
|
||||
)
|
||||
|
||||
eval_args = (
|
||||
"--eval-interval 20 "
|
||||
"--eval-prompt-data gsm8k /root/datasets/gsm8k/test.parquet "
|
||||
"--n-samples-per-eval-prompt 1 "
|
||||
"--eval-max-response-len 1024 "
|
||||
"--eval-top-k 1 "
|
||||
)
|
||||
|
||||
perf_args = (
|
||||
"--tensor-model-parallel-size 1 "
|
||||
"--sequence-parallel "
|
||||
"--pipeline-model-parallel-size 1 "
|
||||
"--context-parallel-size 1 "
|
||||
"--expert-model-parallel-size 1 "
|
||||
"--expert-tensor-parallel-size 1 "
|
||||
"--use-dynamic-batch-size "
|
||||
"--max-tokens-per-gpu 9216 "
|
||||
)
|
||||
|
||||
grpo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
"--use-kl-loss "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type low_var_kl "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 0.2 "
|
||||
"--eps-clip-high 0.28 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = (
|
||||
"--rollout-num-gpus-per-engine 1 "
|
||||
f"--sglang-mem-fraction-static {0.6 if TIGHT_DEVICE_MEMORY else 0.7} "
|
||||
"--sglang-enable-metrics "
|
||||
)
|
||||
|
||||
ci_args = "--ci-test "
|
||||
|
||||
fault_tolerance_args = (
|
||||
"--use-fault-tolerance "
|
||||
"--rollout-health-check-interval 5 "
|
||||
"--rollout-health-check-timeout 10 "
|
||||
"--rollout-health-check-first-wait 0 "
|
||||
)
|
||||
|
||||
misc_args = (
|
||||
"--attention-dropout 0.0 "
|
||||
"--hidden-dropout 0.0 "
|
||||
"--accumulate-allreduce-grads-in-fp32 "
|
||||
"--attention-softmax-in-fp32 "
|
||||
"--attention-backend flash "
|
||||
"--actor-num-nodes 1 "
|
||||
"--actor-num-gpus-per-node 4 "
|
||||
"--colocate "
|
||||
"--megatron-to-hf-mode bridge "
|
||||
)
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{grpo_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{perf_args} "
|
||||
f"{eval_args} "
|
||||
f"{sglang_args} "
|
||||
f"{ci_args} "
|
||||
f"{fault_tolerance_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=NUM_GPUS,
|
||||
megatron_model_type=MODEL_TYPE,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
@@ -0,0 +1,104 @@
|
||||
import os
|
||||
import miles.utils.external_utils.command_utils as U
|
||||
|
||||
MODEL_NAME = "Qwen3-0.6B"
|
||||
|
||||
|
||||
def prepare():
|
||||
U.exec_command("mkdir -p /root/models /root/datasets")
|
||||
U.exec_command(f"huggingface-cli download Qwen/{MODEL_NAME} --local-dir /root/models/{MODEL_NAME}")
|
||||
U.hf_download_dataset("zhuzilin/gsm8k")
|
||||
|
||||
|
||||
def execute():
|
||||
ckpt_args = f"--hf-checkpoint /root/models/{MODEL_NAME} "
|
||||
|
||||
rollout_args = (
|
||||
"--prompt-data /root/datasets/gsm8k/train.parquet "
|
||||
"--input-key messages "
|
||||
"--label-key label "
|
||||
"--apply-chat-template "
|
||||
"--rollout-shuffle "
|
||||
"--rm-type math "
|
||||
f"--num-rollout {3000 if U.get_env_enable_infinite_run() else 60} "
|
||||
"--rollout-batch-size 32 "
|
||||
"--n-samples-per-prompt 8 "
|
||||
"--rollout-max-response-len 1024 "
|
||||
"--rollout-temperature 1 "
|
||||
"--over-sampling-batch-size 64 "
|
||||
"--dynamic-sampling-filter-path miles.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std "
|
||||
"--global-batch-size 256 "
|
||||
)
|
||||
|
||||
eval_args = (
|
||||
"--eval-interval 20 "
|
||||
"--eval-prompt-data gsm8k /root/datasets/gsm8k/test.parquet "
|
||||
"--n-samples-per-eval-prompt 1 "
|
||||
"--eval-max-response-len 1024 "
|
||||
"--eval-top-k 1 "
|
||||
)
|
||||
|
||||
grpo_args = (
|
||||
"--advantage-estimator grpo "
|
||||
# "--use-kl-loss "
|
||||
"--kl-loss-coef 0.00 "
|
||||
"--kl-loss-type low_var_kl "
|
||||
"--kl-coef 0.00 "
|
||||
"--entropy-coef 0.00 "
|
||||
"--eps-clip 0.2 "
|
||||
"--eps-clip-high 0.28 "
|
||||
)
|
||||
|
||||
optimizer_args = (
|
||||
"--optimizer adam "
|
||||
"--lr 1e-6 "
|
||||
"--lr-decay-style constant "
|
||||
"--weight-decay 0.1 "
|
||||
"--adam-beta1 0.9 "
|
||||
"--adam-beta2 0.98 "
|
||||
)
|
||||
|
||||
sglang_args = "--rollout-num-gpus-per-engine 2 " "--sglang-decode-log-interval 1000 " "--sglang-enable-metrics "
|
||||
|
||||
fsdp_args = (
|
||||
# Set to true for FULL_STATE_DICT mode, false for SHARDED_STATE_DICT mode (default)
|
||||
# "--fsdp-full-params " # Uncomment this line to enable full params mode
|
||||
# Set the bucket size for weight update
|
||||
"--update-weight-buffer-size 536870912 " # 512MB
|
||||
)
|
||||
|
||||
ci_args = (
|
||||
"--ci-test "
|
||||
"--ci-disable-kl-checker "
|
||||
"--ci-metric-checker-key eval/gsm8k "
|
||||
"--ci-metric-checker-threshold 0.71 " # loose threshold at 60 step
|
||||
)
|
||||
|
||||
misc_args = "--actor-num-nodes 1 " "--actor-num-gpus-per-node 2 " "--colocate " "--train-backend fsdp "
|
||||
|
||||
train_args = (
|
||||
f"{ckpt_args} "
|
||||
f"{rollout_args} "
|
||||
f"{optimizer_args} "
|
||||
f"{grpo_args} "
|
||||
f"{sglang_args} "
|
||||
f"{U.get_default_wandb_args(__file__)} "
|
||||
f"{eval_args} "
|
||||
f"{fsdp_args} "
|
||||
f"{ci_args} "
|
||||
f"{misc_args} "
|
||||
)
|
||||
|
||||
U.execute_train(
|
||||
train_args=train_args,
|
||||
num_gpus_per_node=2,
|
||||
megatron_model_type=None,
|
||||
extra_env_vars={"MILES_EXPERIMENTAL_ROLLOUT_REFACTOR": "1"},
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
prepare()
|
||||
for proxy_var in ("http_proxy", "https_proxy", "HTTP_PROXY", "HTTPS_PROXY"):
|
||||
os.environ.pop(proxy_var, None)
|
||||
execute()
|
||||
Reference in New Issue
Block a user