mirror of
https://github.com/radixark/miles.git
synced 2026-10-02 07:14:53 +08:00
333 lines
9.6 KiB
YAML
333 lines
9.6 KiB
YAML
run:
|
|
colocate:
|
|
inference_pools:
|
|
- layout:
|
|
gpu_offset: 0
|
|
num_gpus_per_inference_pod: 8
|
|
num_gpus_per_node: 8
|
|
num_inference_cells: 1
|
|
num_pods_per_inference_cell: 2
|
|
num_pods_per_trainer_cell: 4
|
|
num_trainer_cells: 1
|
|
pool_id: inference-engine-all-0-0
|
|
- layout:
|
|
gpu_offset: 16
|
|
num_gpus_per_inference_pod: 8
|
|
num_gpus_per_node: 8
|
|
num_inference_cells: 2
|
|
num_pods_per_inference_cell: 1
|
|
num_pods_per_trainer_cell: 4
|
|
num_trainer_cells: 1
|
|
pool_id: inference-engine-all-0-1
|
|
namespace: rl
|
|
release: myrun
|
|
trainer_pool_id: trainer-engine-actor
|
|
env:
|
|
CUDA_DEVICE_MAX_CONNECTIONS: '1'
|
|
PYTHONUNBUFFERED: '1'
|
|
id: 260101-000000-000
|
|
inferenceEngines:
|
|
- command:
|
|
- <PYTHON>
|
|
- -m
|
|
- sglang.launch_server
|
|
- --trust-remote-code
|
|
- --model-path
|
|
- <FIXTURE_DIR>/typical-model
|
|
- --host
|
|
- 0.0.0.0
|
|
- --port
|
|
- '8000'
|
|
- --device
|
|
- cuda
|
|
- --random-seed
|
|
- '<RANDOM_SEED>'
|
|
- --enable-memory-saver
|
|
- --nccl-port
|
|
- '10000'
|
|
- --nnodes
|
|
- '2'
|
|
- --node-rank
|
|
- $(LWS_WORKER_INDEX)
|
|
- --dist-init-addr
|
|
- $(LWS_LEADER_ADDRESS):9000
|
|
- --gated-launch-port
|
|
- '13000'
|
|
- --tp-size
|
|
- '16'
|
|
- --skip-server-warmup
|
|
- --enable-draft-weights-cpu-backup
|
|
- --enable-metrics
|
|
- --disaggregation-mode
|
|
- prefill
|
|
- --load-balance-method
|
|
- round_robin
|
|
- --disaggregation-bootstrap-port
|
|
- '11000'
|
|
- --engine-info-bootstrap-port
|
|
- '12000'
|
|
- --mem-fraction-static
|
|
- '0.7'
|
|
- --chunked-prefill-size
|
|
- '4096'
|
|
- --cuda-graph-backend-prefill
|
|
- disabled
|
|
- --lora-use-virtual-experts
|
|
env:
|
|
NVSHMEM_DISABLE_NCCL: '1'
|
|
RAY_EXPERIMENTAL_NOSET_ASCEND_RT_VISIBLE_DEVICES: '1'
|
|
RAY_EXPERIMENTAL_NOSET_CUDA_VISIBLE_DEVICES: '1'
|
|
RAY_EXPERIMENTAL_NOSET_HABANA_VISIBLE_MODULES: '1'
|
|
RAY_EXPERIMENTAL_NOSET_HIP_VISIBLE_DEVICES: '1'
|
|
RAY_EXPERIMENTAL_NOSET_NEURON_RT_VISIBLE_CORES: '1'
|
|
RAY_EXPERIMENTAL_NOSET_ONEAPI_DEVICE_SELECTOR: '1'
|
|
RAY_EXPERIMENTAL_NOSET_TPU_VISIBLE_CHIPS: '1'
|
|
SGLANG_BATCH_INVARIANT_OPS_ENABLE_MM_FALLBACK_VARIANT: 'true'
|
|
SGLANG_DG_CACHE_DIR_PER_PROCESS: '1'
|
|
SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: 'false'
|
|
SGLANG_ENABLE_STRICT_MEM_CHECK_DURING_IDLE: 'false'
|
|
SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: 'false'
|
|
SGLANG_EXPOSE_OWN_ENV_VARS: '1'
|
|
SGLANG_JIT_DEEPGEMM_PRECOMPILE: 'false'
|
|
SGLANG_MEMORY_SAVER_CUDA_GRAPH: 'true'
|
|
SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1'
|
|
meta:
|
|
gpu_ids: 0,1,2,3,4,5,6,7
|
|
name: inference-engine-all-0-0
|
|
objectName: myrun-miles-run-inference-engine-all-0-0
|
|
poolId: inference-engine-all-0-0
|
|
ports:
|
|
- name: primary
|
|
port: 8000
|
|
- name: dist-init
|
|
port: 9000
|
|
- name: nccl
|
|
port: 10000
|
|
- name: disaggregation
|
|
port: 11000
|
|
- name: engine-info-boo
|
|
port: 12000
|
|
- name: gate
|
|
port: 13000
|
|
replicas: 1
|
|
resources:
|
|
limits:
|
|
nvidia.com/gpu: 8
|
|
size: 2
|
|
- command:
|
|
- <PYTHON>
|
|
- -m
|
|
- sglang.launch_server
|
|
- --trust-remote-code
|
|
- --model-path
|
|
- <FIXTURE_DIR>/typical-model
|
|
- --host
|
|
- 0.0.0.0
|
|
- --port
|
|
- '8000'
|
|
- --device
|
|
- cuda
|
|
- --random-seed
|
|
- '<RANDOM_SEED>'
|
|
- --enable-memory-saver
|
|
- --nccl-port
|
|
- '10000'
|
|
- --node-rank
|
|
- $(LWS_WORKER_INDEX)
|
|
- --dist-init-addr
|
|
- $(LWS_LEADER_ADDRESS):9000
|
|
- --gated-launch-port
|
|
- '13000'
|
|
- --tp-size
|
|
- '8'
|
|
- --skip-server-warmup
|
|
- --enable-draft-weights-cpu-backup
|
|
- --enable-metrics
|
|
- --disaggregation-mode
|
|
- decode
|
|
- --engine-info-bootstrap-port
|
|
- '12000'
|
|
- --mem-fraction-static
|
|
- '0.7'
|
|
- --chunked-prefill-size
|
|
- '4096'
|
|
- --cuda-graph-backend-prefill
|
|
- disabled
|
|
- --lora-use-virtual-experts
|
|
env:
|
|
NVSHMEM_DISABLE_NCCL: '1'
|
|
RAY_EXPERIMENTAL_NOSET_ASCEND_RT_VISIBLE_DEVICES: '1'
|
|
RAY_EXPERIMENTAL_NOSET_CUDA_VISIBLE_DEVICES: '1'
|
|
RAY_EXPERIMENTAL_NOSET_HABANA_VISIBLE_MODULES: '1'
|
|
RAY_EXPERIMENTAL_NOSET_HIP_VISIBLE_DEVICES: '1'
|
|
RAY_EXPERIMENTAL_NOSET_NEURON_RT_VISIBLE_CORES: '1'
|
|
RAY_EXPERIMENTAL_NOSET_ONEAPI_DEVICE_SELECTOR: '1'
|
|
RAY_EXPERIMENTAL_NOSET_TPU_VISIBLE_CHIPS: '1'
|
|
SGLANG_BATCH_INVARIANT_OPS_ENABLE_MM_FALLBACK_VARIANT: 'true'
|
|
SGLANG_DG_CACHE_DIR_PER_PROCESS: '1'
|
|
SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION: 'false'
|
|
SGLANG_ENABLE_STRICT_MEM_CHECK_DURING_IDLE: 'false'
|
|
SGLANG_ENABLE_TP_MEMORY_INBALANCE_CHECK: 'false'
|
|
SGLANG_EXPOSE_OWN_ENV_VARS: '1'
|
|
SGLANG_JIT_DEEPGEMM_PRECOMPILE: 'false'
|
|
SGLANG_MEMORY_SAVER_CUDA_GRAPH: 'true'
|
|
SGLANG_OPT_USE_CUSTOM_ALL_REDUCE_V2: '1'
|
|
meta:
|
|
gpu_ids: 0,1,2,3,4,5,6,7
|
|
name: inference-engine-all-0-1
|
|
objectName: myrun-miles-run-inference-engine-all-0-1
|
|
poolId: inference-engine-all-0-1
|
|
ports:
|
|
- name: primary
|
|
port: 8000
|
|
- name: dist-init
|
|
port: 9000
|
|
- name: nccl
|
|
port: 10000
|
|
- name: engine-info-boo
|
|
port: 12000
|
|
- name: gate
|
|
port: 13000
|
|
replicas: 2
|
|
resources:
|
|
limits:
|
|
nvidia.com/gpu: 8
|
|
objectNames:
|
|
colocatePairing: myrun-miles-run-colocate-pairing
|
|
mooncakeMaster: myrun-miles-run-mooncake-master
|
|
orchestrator: myrun-miles-run-orchestrator
|
|
platformRead: myrun-miles-run-platform-read
|
|
platformReadDelete: myrun-miles-run-platform-read-delete
|
|
uninstall: myrun-miles-run-uninstall
|
|
uninstallManifest: myrun-miles-run-uninstall-manifest
|
|
orchestrator:
|
|
command:
|
|
- python
|
|
- scripts/run_qwen3_4b.py
|
|
- train
|
|
- --cluster-backend
|
|
- kubernetes
|
|
stateFile: /cluster-storage/myteam/miles_data/miles-runs/260101-000000-000/state/orchestrator-260101-000000-000001.state
|
|
staticWorkers:
|
|
- command:
|
|
- <PYTHON>
|
|
- -m
|
|
- miles.utils.workers.serving.serve
|
|
- --specs
|
|
- miles.ray.specs.entrypoint.compute_specs_from_argv
|
|
- --pool-id
|
|
- rollout-executor
|
|
- --
|
|
- --cluster-backend
|
|
- kubernetes
|
|
- --rollout-num-gpus
|
|
- '32'
|
|
name: rollout-executor
|
|
objectName: myrun-miles-run-rollout-executor
|
|
poolId: rollout-executor
|
|
ports:
|
|
- name: rpc
|
|
port: 8000
|
|
replicas: 1
|
|
- command:
|
|
- <PYTHON>
|
|
- -m
|
|
- miles.utils.workers.serving.serve
|
|
- --specs
|
|
- miles.ray.specs.entrypoint.compute_specs_from_argv
|
|
- --pool-id
|
|
- inference-controller
|
|
- --
|
|
- --cluster-backend
|
|
- kubernetes
|
|
- --rollout-num-gpus
|
|
- '32'
|
|
name: inference-controller
|
|
objectName: myrun-miles-run-inference-controller
|
|
poolId: inference-controller
|
|
ports:
|
|
- name: rpc
|
|
port: 8000
|
|
replicas: 1
|
|
serviceAccountName: myrun-miles-run-platform-read
|
|
- command:
|
|
- <PYTHON>
|
|
- -m
|
|
- sglang_router.launch_router
|
|
- --policy
|
|
- manual
|
|
- --assignment-mode
|
|
- min_load
|
|
- --pd-disaggregation
|
|
- --log-level
|
|
- warn
|
|
- --prometheus-port
|
|
- '9000'
|
|
- --request-timeout-secs
|
|
- '14400'
|
|
- --port
|
|
- '8000'
|
|
name: inference-router-0
|
|
objectName: myrun-miles-run-inference-router-0
|
|
poolId: inference-router-0
|
|
ports:
|
|
- name: primary
|
|
port: 8000
|
|
- name: prometheus
|
|
port: 9000
|
|
replicas: 1
|
|
- command:
|
|
- <PYTHON>
|
|
- -m
|
|
- miles.rollout.session.server
|
|
- --config-json
|
|
- '{"host":"0.0.0.0","port":8000,"instance_id":"0123456789abcdef-0","backend_url":"http://myrun-miles-run-inference-router-0-0.myrun-miles-run-inference-router-0:8000","timeout":null,"hf_checkpoint":"<FIXTURE_DIR>/typical-model","chat_template_path":null,"tito_model":"default","apply_chat_template_kwargs":{},"use_rollout_routing_replay":false,"use_rollout_indexer_replay":false,"use_sampling_support_replay":false,"sglang_speculative_algorithm":null,"num_layers":36,"moe_router_topk":2,"save_debug_trajectory_data":null,"lora_rank":0,"lora_adapter_path":null,"lora_train_only":false,"use_session_server":true,"session_message_matcher":"strict","pause_generation_mode":"retract","session_sample_picker_path":"miles.rollout.session.v2.picker_hub.drop_same_prompt_retries","session_sample_postprocessor_path":"miles.rollout.session.v2.postprocessor_hub.default_postprocess"}'
|
|
name: session-server
|
|
objectName: myrun-miles-run-session-server
|
|
poolId: session-server
|
|
ports:
|
|
- name: primary
|
|
port: 8000
|
|
replicas: 32
|
|
- command:
|
|
- <PYTHON>
|
|
- -m
|
|
- miles.utils.workers.serving.serve
|
|
- --specs
|
|
- miles.ray.specs.entrypoint.compute_specs_from_argv
|
|
- --pool-id
|
|
- trainer-controller-actor
|
|
- --
|
|
- --cluster-backend
|
|
- kubernetes
|
|
- --rollout-num-gpus
|
|
- '32'
|
|
name: trainer-controller-actor
|
|
objectName: myrun-miles-run-trainer-controller-actor
|
|
poolId: trainer-controller-actor
|
|
ports:
|
|
- name: rpc
|
|
port: 8000
|
|
replicas: 1
|
|
serviceAccountName: myrun-miles-run-platform-read-delete
|
|
trainerEngines:
|
|
- command:
|
|
- bash
|
|
- -c
|
|
- mkdir -p /scratch/Qwen3-4B && rsync -a --info=progress2 /cluster-storage/models/Qwen3-4B/ /scratch/Qwen3-4B && exec <PYTHON> -m miles.utils.workers.process_supervisor --num-subprocesses 8 -- <PYTHON> -m miles.utils.workers.serving.serve --specs miles.ray.specs.entrypoint.compute_specs_from_argv --pool-id trainer-engine-actor -- --cluster-backend kubernetes --rollout-num-gpus 32
|
|
meta:
|
|
gpu_ids: 0,1,2,3,4,5,6,7
|
|
name: trainer-engine-actor
|
|
objectName: myrun-miles-run-trainer-engine-actor
|
|
poolId: trainer-engine-actor
|
|
ports:
|
|
- name: master
|
|
port: 9000
|
|
- name: rpc
|
|
port: 8000
|
|
replicas: 1
|
|
resources:
|
|
limits:
|
|
nvidia.com/gpu: 8
|
|
size: 4
|