mirror of
https://github.com/volcengine/verl.git
synced 2025-10-20 13:43:50 +08:00
83
examples/grpo_trainer/run_qwen3_vl-235b-megatron.sh
Normal file
83
examples/grpo_trainer/run_qwen3_vl-235b-megatron.sh
Normal file
@ -0,0 +1,83 @@
|
||||
set -x
|
||||
ENGINE=${1:-vllm}
|
||||
export CUDA_DEVICE_MAX_CONNECTIONS=1 # For megatron communication/computation overlapping
|
||||
|
||||
# VLLM version >= 0.11.0 for qwen3-vl support, recommend to use container docker://iseekyan/verl:nemo.gptoss_vllm0.11.0
|
||||
# pip install -U git+https://github.com/ISEEKYAN/mbridge.git # for latest mbridge
|
||||
# pip install -U transformers # for qwen3-vl support
|
||||
# pip install --no-deps --no-cache-dir git+https://github.com/NVIDIA/Megatron-LM.git@core_v0.13.1 # for megatron-lm0.13.1
|
||||
|
||||
|
||||
export VLLM_ALLREDUCE_USE_SYMM_MEM=0 # for vllm0.11.0 with TP
|
||||
|
||||
|
||||
HF_MODEL_PATH=${HF_MODEL_PATH:-"${RAY_DATA_HOME}/models/Qwen3-VL-235B-A22B-Instruct"}
|
||||
|
||||
|
||||
train_path=$HOME/data/geo3k/train.parquet
|
||||
test_path=$HOME/data/geo3k/test.parquet
|
||||
|
||||
python3 -m verl.trainer.main_ppo --config-path=config \
|
||||
--config-name='ppo_megatron_trainer.yaml'\
|
||||
algorithm.adv_estimator=grpo \
|
||||
data.train_files="$train_path" \
|
||||
data.val_files="$test_path" \
|
||||
data.train_batch_size=512 \
|
||||
data.max_prompt_length=1024 \
|
||||
data.max_response_length=2048 \
|
||||
data.filter_overlong_prompts=True \
|
||||
data.truncation='error' \
|
||||
actor_rollout_ref.model.path=$HF_MODEL_PATH \
|
||||
actor_rollout_ref.actor.optim.lr=1e-6 \
|
||||
actor_rollout_ref.actor.ppo_mini_batch_size=128 \
|
||||
actor_rollout_ref.actor.ppo_micro_batch_size_per_gpu=1 \
|
||||
actor_rollout_ref.actor.megatron.tensor_model_parallel_size=4 \
|
||||
actor_rollout_ref.actor.megatron.pipeline_model_parallel_size=8 \
|
||||
actor_rollout_ref.actor.megatron.expert_model_parallel_size=8 \
|
||||
actor_rollout_ref.actor.megatron.expert_tensor_parallel_size=1 \
|
||||
actor_rollout_ref.actor.use_kl_loss=True \
|
||||
actor_rollout_ref.actor.kl_loss_coef=0.01 \
|
||||
actor_rollout_ref.actor.kl_loss_type=low_var_kl \
|
||||
actor_rollout_ref.actor.entropy_coeff=0 \
|
||||
actor_rollout_ref.rollout.log_prob_micro_batch_size_per_gpu=1 \
|
||||
actor_rollout_ref.rollout.tensor_model_parallel_size=16 \
|
||||
actor_rollout_ref.actor.use_dynamic_bsz=True \
|
||||
actor_rollout_ref.actor.ppo_max_token_len_per_gpu=5120 \
|
||||
actor_rollout_ref.ref.log_prob_use_dynamic_bsz=True \
|
||||
actor_rollout_ref.ref.log_prob_max_token_len_per_gpu=5120 \
|
||||
actor_rollout_ref.rollout.log_prob_use_dynamic_bsz=True \
|
||||
actor_rollout_ref.rollout.log_prob_max_token_len_per_gpu=5120 \
|
||||
actor_rollout_ref.rollout.name=$ENGINE \
|
||||
+actor_rollout_ref.rollout.engine_kwargs.vllm.disable_mm_preprocessor_cache=True \
|
||||
actor_rollout_ref.rollout.gpu_memory_utilization=0.7 \
|
||||
actor_rollout_ref.rollout.n=5 \
|
||||
actor_rollout_ref.ref.log_prob_micro_batch_size_per_gpu=1 \
|
||||
actor_rollout_ref.actor.megatron.use_mbridge=True \
|
||||
actor_rollout_ref.actor.megatron.param_offload=True \
|
||||
actor_rollout_ref.actor.megatron.optimizer_offload=True \
|
||||
actor_rollout_ref.actor.megatron.grad_offload=True \
|
||||
actor_rollout_ref.ref.megatron.param_offload=True \
|
||||
+actor_rollout_ref.actor.optim.override_optimizer_config.optimizer_offload_fraction=1 \
|
||||
+actor_rollout_ref.actor.optim.override_optimizer_config.overlap_cpu_optimizer_d2h_h2d=True \
|
||||
+actor_rollout_ref.actor.optim.override_optimizer_config.use_precision_aware_optimizer=True \
|
||||
+actor_rollout_ref.actor.optim.override_optimizer_config.optimizer_cpu_offload=True \
|
||||
+actor_rollout_ref.actor.megatron.override_transformer_config.moe_router_dtype=fp32 \
|
||||
+actor_rollout_ref.actor.megatron.override_transformer_config.moe_enable_deepep=True \
|
||||
+actor_rollout_ref.actor.megatron.override_transformer_config.moe_token_dispatcher_type=flex \
|
||||
+actor_rollout_ref.actor.megatron.override_transformer_config.recompute_method=uniform \
|
||||
+actor_rollout_ref.actor.megatron.override_transformer_config.recompute_granularity=full \
|
||||
+actor_rollout_ref.actor.megatron.override_transformer_config.recompute_num_layers=1 \
|
||||
+actor_rollout_ref.actor.megatron.override_transformer_config.gradient_accumulation_fusion=True \
|
||||
+actor_rollout_ref.actor.megatron.override_transformer_config.moe_permute_fusion=True \
|
||||
+actor_rollout_ref.actor.megatron.override_transformer_config.account_for_loss_in_pipeline_split=True \
|
||||
+actor_rollout_ref.actor.megatron.override_transformer_config.account_for_embedding_in_pipeline_split=True \
|
||||
algorithm.use_kl_in_reward=False \
|
||||
trainer.critic_warmup=0 \
|
||||
trainer.logger='["console","wandb"]' \
|
||||
trainer.project_name='verl_grpo_example_geo3k' \
|
||||
trainer.experiment_name='qwen3_vl_235b_megatron' \
|
||||
trainer.n_gpus_per_node=8 \
|
||||
trainer.nnodes=8 \
|
||||
trainer.save_freq=20 \
|
||||
trainer.test_freq=5 \
|
||||
trainer.total_epochs=15 $@
|
Reference in New Issue
Block a user