# returncode: 0

### 0
"apt-get"
"install"
"-y"
"iproute2"

### 1
"python3"
"-c"
"import socket;s=socket.socket(socket.AF_INET,socket.SOCK_DGRAM);s.connect(('8.8.8.8',53));print(s.getsockname()[0])"

### 2
"ip"
"-o"
"-4"
"addr"
"show"

### 3
"awk"
"-v"
"ip="
"$4 ~ \"^\"ip\"/\" {print $2; exit}"

### 4
"pkill"
"sglang"

### 5
"ray"
"stop"
"--force"

### 6
"sleep"
"5"

### 7
"pkill"
"-9"
"sglang"

### 8
"pkill"
"-9"
"ray"

### 9
"pkill"
"-9"
"python"

### 10
"python3"
"<REPO_ROOT>/examples/lora/../../miles/utils/external_utils/model_args_utils.py"
"qwen2.5-3B"

### 11
"ray"
"start"
"--head"
"--node-ip-address"
"10.0.0.1"
"--num-gpus"
"1"
"--disable-usage-stats"
"--dashboard-host=0.0.0.0"
"--dashboard-port=8265"

### 12
"python3"
"-c"
"import ray; ray.init(address='auto', ignore_reinit_error=True); print(int(ray.cluster_resources().get('GPU', 0))); ray.shutdown()"

### 13
"ray"
"job"
"submit"
"--address=http://127.0.0.1:8265"
"--runtime-env-json={\n        \"env_vars\": {\n           \"PYTHONPATH\": \"/root/Megatron-LM\",\n           \"CUDA_DEVICE_MAX_CONNECTIONS\": \"1\",\n           \"NCCL_ALGO\": \"Ring\",\n           \"NVTE_ALLOW_NONDETERMINISTIC_ALGO\": \"0\",\n           \"CUBLAS_WORKSPACE_CONFIG\": \":4096:8\",\n           \"NVTE_NORM_FWD_USE_CUDNN\": \"1\",\n           \"NVTE_NORM_BWD_USE_CUDNN\": \"1\"\n        }\n      }"
"--"
"python3"
"train.py"
"--calculate-per-token-loss"
"--swiglu"
"--num-layers"
"36"
"--hidden-size"
"2048"
"--ffn-hidden-size"
"11008"
"--num-attention-heads"
"16"
"--use-rotary-position-embeddings"
"--disable-bias-linear"
"--add-qkv-bias"
"--normalization"
"RMSNorm"
"--norm-epsilon"
"1e-6"
"--rotary-base"
"1000000"
"--group-query-attention"
"--num-query-groups"
"2"
"--vocab-size"
"151936"
"--hf-checkpoint"
"/root/Qwen2.5-3B-Instruct/"
"--megatron-to-hf-mode"
"bridge"
"--lora-rank"
"32"
"--lora-alpha"
"32"
"--lora-dropout"
"0.0"
"--target-modules"
"all-linear"
"--megatron-to-hf-mode"
"bridge"
"--optimizer"
"adam"
"--lr"
"1e-5"
"--lr-decay-style"
"constant"
"--weight-decay"
"0.1"
"--adam-beta1"
"0.9"
"--adam-beta2"
"0.98"
"--advantage-estimator"
"grpo"
"--kl-loss-coef"
"0.00"
"--kl-loss-type"
"low_var_kl"
"--kl-coef"
"0.00"
"--entropy-coef"
"0.00"
"--eps-clip"
"0.2"
"--eps-clip-high"
"0.28"
"--use-wandb"
"--wandb-host"
"https://wandb.ai/"
"--wandb-project"
"miles-lora-megatron"
"--wandb-group"
"qwen2.5-3B-lora-disaggregate-2node-p2p"
"--tensor-model-parallel-size"
"1"
"--sequence-parallel"
"--pipeline-model-parallel-size"
"1"
"--context-parallel-size"
"1"
"--expert-model-parallel-size"
"1"
"--expert-tensor-parallel-size"
"1"
"--use-dynamic-batch-size"
"--max-tokens-per-gpu"
"4096"
"--eval-interval"
"10"
"--eval-prompt-data"
"gsm8k"
"/root/gsm8k/test.parquet"
"--n-samples-per-eval-prompt"
"1"
"--eval-max-response-len"
"1024"
"--eval-top-k"
"1"
"--rollout-num-gpus"
"1"
"--rollout-num-gpus-per-engine"
"1"
"--sglang-mem-fraction-static"
"0.7"
"--sglang-remote-instance-weight-loader-start-seed-via-transfer-engine"
"--attention-dropout"
"0.0"
"--hidden-dropout"
"0.0"
"--accumulate-allreduce-grads-in-fp32"
"--attention-softmax-in-fp32"
"--attention-backend"
"flash"
"--no-gradient-accumulation-fusion"
"--actor-num-nodes"
"1"
"--actor-num-gpus-per-node"
"1"
"--update-weight-transfer-mode"
"p2p"
"--update-weight-buffer-size"
"1073741824"
"--check-weight-update-equal"
"--prompt-data"
"/root/gsm8k/train.parquet"
"--input-key"
"messages"
"--label-key"
"label"
"--apply-chat-template"
"--rollout-shuffle"
"--rm-type"
"math"
"--num-rollout"
"100"
"--rollout-batch-size"
"32"
"--n-samples-per-prompt"
"8"
"--rollout-max-response-len"
"1024"
"--rollout-temperature"
"1"
"--global-batch-size"
"256"
