mirror of
https://github.com/radixark/miles.git
synced 2026-10-02 07:14:53 +08:00
[tests] remove and rename tests
This commit is contained in:
@@ -1,133 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
# for rerun the task
|
||||
pkill -9 sglang
|
||||
sleep 3
|
||||
ray stop --force
|
||||
pkill -9 ray
|
||||
pkill -9 python
|
||||
sleep 3
|
||||
pkill -9 ray
|
||||
pkill -9 python
|
||||
|
||||
set -ex
|
||||
|
||||
# will prevent ray from buffering stdout/stderr
|
||||
export PYTHONBUFFERED=16
|
||||
|
||||
SCRIPT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")" &>/dev/null && pwd)"
|
||||
source "${SCRIPT_DIR}/../scripts/models/qwen3-0.6B.sh"
|
||||
|
||||
CKPT_ARGS=(
|
||||
--hf-checkpoint /root/Qwen3-0.6B
|
||||
--ref-load /root/Qwen3-0.6B_torch_dist
|
||||
)
|
||||
|
||||
ROLLOUT_ARGS=(
|
||||
--prompt-data /root/dapo-math-17k/dapo-math-17k.jsonl
|
||||
--input-key prompt
|
||||
--label-key label
|
||||
--apply-chat-template
|
||||
--rollout-shuffle
|
||||
--rm-type deepscaler
|
||||
--num-rollout 3000
|
||||
--rollout-batch-size 32
|
||||
--n-samples-per-prompt 8
|
||||
--rollout-max-response-len 8192
|
||||
--rollout-temperature 0.8
|
||||
|
||||
--over-sampling-batch-size 64
|
||||
--dynamic-sampling-filter-path slime.rollout.filter_hub.dynamic_sampling_filters.check_reward_nonzero_std
|
||||
#--partial-rollout
|
||||
|
||||
--global-batch-size 256
|
||||
#--balance-data
|
||||
)
|
||||
|
||||
EVAL_ARGS=(
|
||||
--eval-interval 20
|
||||
--eval-prompt-data aime /root/aime-2024/aime-2024.jsonl
|
||||
--n-samples-per-eval-prompt 1
|
||||
--eval-max-response-len 16384
|
||||
--eval-temperature 0
|
||||
)
|
||||
|
||||
PERF_ARGS=(
|
||||
--tensor-model-parallel-size 1
|
||||
--sequence-parallel
|
||||
--pipeline-model-parallel-size 1
|
||||
--context-parallel-size 1
|
||||
--expert-model-parallel-size 1
|
||||
--expert-tensor-parallel-size 1
|
||||
|
||||
# --micro-batch-size 1
|
||||
--use-dynamic-batch-size
|
||||
--max-tokens-per-gpu 9216
|
||||
)
|
||||
|
||||
GRPO_ARGS=(
|
||||
--advantage-estimator grpo
|
||||
--use-kl-loss
|
||||
--kl-loss-coef 0.00
|
||||
--kl-loss-type low_var_kl
|
||||
--entropy-coef 0.00
|
||||
--eps-clip 0.2
|
||||
--eps-clip-high 0.28
|
||||
)
|
||||
|
||||
OPTIMIZER_ARGS=(
|
||||
--optimizer adam
|
||||
--lr 1e-6
|
||||
--lr-decay-style constant
|
||||
--weight-decay 0.1
|
||||
--adam-beta1 0.9
|
||||
--adam-beta2 0.98
|
||||
)
|
||||
|
||||
WANDB_ARGS=(
|
||||
#--use-wandb
|
||||
--wandb-project slime-test
|
||||
--wandb-group test-qwen-3-0.6B
|
||||
)
|
||||
|
||||
SGLANG_ARGS=(
|
||||
--rollout-num-gpus-per-engine 1
|
||||
--sglang-mem-fraction-static 0.7
|
||||
)
|
||||
|
||||
MISC_ARGS=(
|
||||
# default dropout in megatron is 0.1
|
||||
--attention-dropout 0.0
|
||||
--hidden-dropout 0.0
|
||||
# should be good for model performance
|
||||
--accumulate-allreduce-grads-in-fp32
|
||||
--attention-softmax-in-fp32
|
||||
# need to comment this when using model with MLA
|
||||
--attention-backend flash
|
||||
)
|
||||
|
||||
# launch the master node of ray in container
|
||||
ray start --head --node-ip-address ${MASTER_ADDR} --num-gpus 8 --disable-usage-stats
|
||||
|
||||
ray job submit --address="http://127.0.0.1:8265" \
|
||||
--runtime-env-json='{
|
||||
"env_vars": {
|
||||
"PYTHONPATH": "/root/Megatron-LM",
|
||||
"CUDA_DEVICE_MAX_CONNECTIONS": "1"
|
||||
}
|
||||
}' \
|
||||
-- python3 train.py \
|
||||
--actor-num-nodes 1 \
|
||||
--actor-num-gpus-per-node 1 \
|
||||
--colocate \
|
||||
${MODEL_ARGS[@]} \
|
||||
${CKPT_ARGS[@]} \
|
||||
${ROLLOUT_ARGS[@]} \
|
||||
${OPTIMIZER_ARGS[@]} \
|
||||
${GRPO_ARGS[@]} \
|
||||
${DISTRIBUTED_ARGS[@]} \
|
||||
${WANDB_ARGS[@]} \
|
||||
${PERF_ARGS[@]} \
|
||||
${EVAL_ARGS[@]} \
|
||||
${SGLANG_ARGS[@]} \
|
||||
${MISC_ARGS[@]}
|
||||
Reference in New Issue
Block a user