[megatron] feat: add script for qwen3next training (#4582)

ISEEKYAN · web-flow · commit b85ea89ab49b · 2025-12-18T16:34:31.000+08:00
### What does this PR do?

1. add dockerfile of experimental images
2. add example script to run qwen3next megatron training
diff --git a/docker/verl0.6.1-experimental/Dockerfile.sglang056exp b/docker/verl0.6.1-experimental/Dockerfile.sglang056exp
@@ -0,0 +1,63 @@
+# Dockerfile for verlai/verl:sgl056.exp
+FROM lmsysorg/sglang:v0.5.6.post1
+
+RUN pip install pybind11
+
+RUN pip install nvidia-mathdx
+
+RUN pip install -v --disable-pip-version-check --no-cache-dir --no-build-isolation --config-settings "--build-option=--cpp_ext" --config-settings "--build-option=--cuda_ext" git+https://github.com/NVIDIA/apex.git
+
+RUN export NVTE_FRAMEWORK=pytorch && MAX_JOBS=128 NVTE_BUILD_THREADS_PER_JOB=4 pip3 install --resume-retries 999 --no-cache-dir --no-build-isolation git+https://github.com/NVIDIA/TransformerEngine.git@release_v2.11
+
+RUN pip install --upgrade --no-cache-dir transformers tokenizers
+
+RUN pip install codetiming tensordict mathruler pylatexenc qwen_vl_utils
+
+RUN pip install --no-cache-dir --no-build-isolation flash_attn==2.8.1
+
+RUN wget https://developer.nvidia.com/downloads/assets/tools/secure/nsight-systems/2025_6/nsight-systems-2025.6.1_2025.6.1.190-1_amd64.deb && \
+    apt-get update && apt-get install -y libxcb-cursor0
+
+RUN apt-get install -y ./nsight-systems-2025.6.1_2025.6.1.190-1_amd64.deb && \
+    rm -rf /usr/local/cuda/bin/nsys && \
+    ln -s /opt/nvidia/nsight-systems/2025.6.1/target-linux-x64/nsys  /usr/local/cuda/bin/nsys && \
+    rm -rf /usr/local/cuda/bin/nsys-ui && \
+    ln -s /opt/nvidia/nsight-systems/2025.6.1/target-linux-x64/nsys-ui /usr/local/cuda/bin/nsys-ui && \
+    rm nsight-systems-2025.6.1_2025.6.1.190-1_amd64.deb
+
+
+# =========================
+# Install HybridEP
+# =========================
+WORKDIR /home/
+RUN git clone --branch hybrid-ep https://github.com/deepseek-ai/DeepEP.git && \
+    cd DeepEP && git checkout 3f601f7ac1c062c46502646ff04c535013bfca00 && \
+    TORCH_CUDA_ARCH_LIST="9.0;10.0" pip install --no-build-isolation .
+
+# =========================
+# Install Qwen3-Next dependencies
+# =========================
+WORKDIR /home/
+# Install causal-conv1d and flash-linear-attention
+RUN cd /tmp && \
+    git clone https://github.com/Dao-AILab/causal-conv1d.git && \
+    cd causal-conv1d && \
+    unset PIP_CONSTRAINT && \
+    CAUSAL_CONV1D_FORCE_BUILD=TRUE pip install --no-build-isolation . && \
+    cd .. && \
+    rm -rf causal-conv1d && \
+    pip install flash-linear-attention
+
+RUN pip install --no-cache-dir torch-memory-saver
+
+RUN pip3 install --no-cache-dir --no-deps trl
+
+RUN pip3 install nvtx matplotlib liger_kernel
+
+RUN pip install -U git+https://github.com/ISEEKYAN/mbridge.git
+
+RUN pip install --no-deps --no-cache-dir git+https://github.com/NVIDIA/Megatron-LM.git@1d462bd37dac21cfa14177405d4921eedb987052 # latest dev branch on 20251209
+
+RUN pip install git+https://github.com/volcengine/verl.git@v0.6.1
+
+RUN pip uninstall -y verl
diff --git a/docker/verl0.6.1-experimental/Dockerfile.vllm012exp b/docker/verl0.6.1-experimental/Dockerfile.vllm012exp
@@ -0,0 +1,69 @@
+# dockerfile for verlai/verl:vll012.exp
+FROM nvcr.io/nvidia/pytorch:25.11-py3
+
+RUN git clone -b v0.12.0 --depth 1 https://github.com/vllm-project/vllm.git /opt/vllm
+
+RUN pip install setuptools_scm
+
+RUN cd /opt/vllm && pip install --no-deps --no-build-isolation --no-cache-dir -e .
+
+RUN pip install -r /opt/vllm/requirements/common.txt
+
+
+RUN pip install pybind11
+
+RUN export NVTE_FRAMEWORK=pytorch && MAX_JOBS=128 NVTE_BUILD_THREADS_PER_JOB=4 pip3 install --resume-retries 999 --no-cache-dir --no-build-isolation git+https://github.com/NVIDIA/TransformerEngine.git@release_v2.11
+
+RUN pip install --upgrade --no-cache-dir transformers tokenizers
+
+RUN pip install codetiming tensordict mathruler pylatexenc qwen_vl_utils
+
+RUN pip install flash_attn
+#==2.8.1
+
+RUN apt update && apt install numactl
+
+RUN wget https://developer.nvidia.com/downloads/assets/tools/secure/nsight-systems/2025_6/nsight-systems-2025.6.1_2025.6.1.190-1_amd64.deb && \
+    apt-get update && apt-get install -y libxcb-cursor0
+
+RUN apt-get install -y ./nsight-systems-2025.6.1_2025.6.1.190-1_amd64.deb && \
+    rm -rf /usr/local/cuda/bin/nsys && \
+    ln -s /opt/nvidia/nsight-systems/2025.6.1/target-linux-x64/nsys  /usr/local/cuda/bin/nsys && \
+    rm -rf /usr/local/cuda/bin/nsys-ui && \
+    ln -s /opt/nvidia/nsight-systems/2025.6.1/target-linux-x64/nsys-ui /usr/local/cuda/bin/nsys-ui && \
+    rm nsight-systems-2025.6.1_2025.6.1.190-1_amd64.deb
+
+
+# =========================
+# Install HybridEP
+# =========================
+WORKDIR /home/
+RUN git clone --branch hybrid-ep https://github.com/deepseek-ai/DeepEP.git && \
+    cd DeepEP && git checkout 3f601f7ac1c062c46502646ff04c535013bfca00 && \
+    TORCH_CUDA_ARCH_LIST="9.0;10.0" pip install --no-build-isolation .
+
+# =========================
+# Install Qwen3-Next dependencies
+# =========================
+WORKDIR /home/
+# Install causal-conv1d and flash-linear-attention
+RUN cd /tmp && \
+    git clone https://github.com/Dao-AILab/causal-conv1d.git && \
+    cd causal-conv1d && \
+    unset PIP_CONSTRAINT && \
+    CAUSAL_CONV1D_FORCE_BUILD=TRUE pip install --no-build-isolation . && \
+    cd .. && \
+    rm -rf causal-conv1d && \
+    pip install flash-linear-attention
+
+RUN pip3 install --no-cache-dir --no-deps trl
+
+RUN pip3 install nvtx matplotlib liger_kernel
+
+RUN pip install -U git+https://github.com/ISEEKYAN/mbridge.git
+
+RUN pip install --no-deps --no-cache-dir git+https://github.com/NVIDIA/Megatron-LM.git@1d462bd37dac21cfa14177405d4921eedb987052 # latest dev branch on 20251209
+
+RUN pip install git+https://github.com/volcengine/verl.git@v0.6.1
+
+RUN pip uninstall -y verl
diff --git a/recipe/dapo/test_dapo_gptoss_20b_megatron.sh b/recipe/dapo/test_dapo_gptoss_20b_megatron.sh
@@ -4,12 +4,12 @@ set -xeuo pipefail
 ################################################### document for gptoss ###################################################
 
 ####################### running environment: #######################
-# option 1: use a pre-built docker image dedicated for gptoss: `docker://iseekyan/verl:nemo.gptoss_vllm0.11.0`, which is 
-#           built upon nemo's dedicated image, see Dockerfile at https://github.com/volcengine/verl/blob/main/docker/verl0.6-cu128-torch2.8.0-fa2.7.4/Dockerfile.vllm011.mcore_gpt-oss
+# option 1: use pre-built images verlai/verl:vll012.exp or verlai/verl:sgl056.exp
 #
 # option 2: self build TE>=2.8 with CUDNN>=9.13.1, megatron with branch `core_dev_r0.15.0`, latest vllm or sglang
 #           you can modify the dockerfile to build the image, see Dockerfile at https://github.com/volcengine/verl/blob/main/docker/Dockerfile.stable.vllm or https://github.com/volcengine/verl/blob/main/docker/Dockerfile.stable.sglang
 
+
 ####################### before training: #######################
 # # install matched mbridge version
 # pip uninstall -y mbridge && pip install git+https://github.com/ISEEKYAN/mbridge@gpt-oss
diff --git a/recipe/dapo/test_dapo_qwen3next_80b_megatron.sh b/recipe/dapo/test_dapo_qwen3next_80b_megatron.sh
@@ -0,0 +1,232 @@
+#!/usr/bin/env bash
+set -xeuo pipefail
+
+
+################################################### document for qwen3next ###################################################
+
+####################### running environment: #######################
+
+# option 1: use pre-built docker images verlai/verl:vll012.exp or verlai/verl:sgl056.exp 
+
+# option 2: self build TE>=2.8, megatron with dev branch and megatron-bridge with main branch
+
+####################### how we support qwen3next? #######################
+# we support qwen3next with megatron-bridge, which is enabled by set `vanilla_mbridge=False`
+
+####################### limitations: #######################
+# 1. context parallel(CP) is not supported until this PR is merged: https://github.com/NVIDIA/Megatron-LM/pull/2614
+# 2. sequence packing(aka thd) is not supported, we must set `actor_rollout_ref.actor.megatron.use_remove_padding=False`, until this PR is merged: https://github.com/NVIDIA/Megatron-LM/pull/2644
+
+## if sequence packing is disabled, we recommend to set `use_dynamic_bsz=False` and set micro batchsize to 1, 
+## otherwise the data will be padded to the max length of the batch, which is not efficient. But it's not mandatory
+
+
+
+
+################################################### quick config ###################################################
+
+# pip install --no-deps --no-cache-dir git+https://github.com/NVIDIA/Megatron-LM.git@dev # install megatron from dev branch
+# pip install --no-deps git+https://github.com/NVIDIA-Nemo/Megatron-Bridge.git # install megatron-bridge from main branch
+
+
+rollout_mode="async"
+return_raw_chat="True"
+export VLLM_USE_V1=1
+rollout_name="vllm" # sglang or vllm
+dtype="bfloat16"
+
+
+project_name='DAPO-test'
+exp_name='qwen3next'
+
+adv_estimator=grpo
+
+use_kl_in_reward=False
+kl_coef=0.0
+use_kl_loss=False
+kl_loss_coef=0.0
+
+clip_ratio_low=0.2
+clip_ratio_high=0.28
+
+max_prompt_length=$((1024 * 2))
+max_response_length=$((1024 * 8))
+enable_overlong_buffer=True
+overlong_buffer_len=$((1024 * 4))
+overlong_penalty_factor=1.0
+
+loss_agg_mode="token-mean"
+
+train_prompt_bsz=32
+n_resp_per_prompt=16
+train_prompt_mini_bsz=32
+
+# Ray
+RAY_ADDRESS=${RAY_ADDRESS:-"http://localhost:8265"}
+WORKING_DIR=${WORKING_DIR:-"${PWD}"}
+RUNTIME_ENV=${RUNTIME_ENV:-"${WORKING_DIR}/verl/verl/trainer/runtime_env.yaml"}
+NNODES=${NNODES:-4}
+# Paths
+MODEL_PATH=${MODEL_PATH:-"${RAY_DATA_HOME}/models/Qwen3-Next-80B-A3B-Instruct"}
+CKPTS_DIR=${CKPTS_DIR:-"${RAY_DATA_HOME}/ckpts/${project_name}/${exp_name}"}
+TRAIN_FILE=${TRAIN_FILE:-"${RAY_DATA_HOME}/data/dapo-math-17k.parquet"}
+TEST_FILE=${TEST_FILE:-"${RAY_DATA_HOME}/data/aime-2024.parquet"}
+
+# Algorithm
+temperature=1.0
+top_p=1.0
+top_k=-1 # 0 for HF rollout, -1 for vLLM rollout
+val_top_p=0.7
+
+# Performance Related Parameter
+use_dynamic_bsz=False
+actor_ppo_max_token_len=$(((max_prompt_length + max_response_length) * 10 / 10))
+infer_ppo_max_token_len=$(((max_prompt_length + max_response_length) * 1))
+offload=True
+gen_tp=16
+train_tp=2
+EP=32
+ETP=1
+train_pp=1
+
+################################################### start of config ###################################################
+
+FP8=(
+    # # train
+    # +actor_rollout_ref.actor.megatron.override_transformer_config.fp8="e4m3" # e4m3 or hybrid
+    # +actor_rollout_ref.actor.megatron.override_transformer_config.fp8_recipe="blockwise"
+    # +actor_rollout_ref.actor.optim.override_optimizer_config.fp8_recipe="blockwise"
+    # # rollout
+    # +actor_rollout_ref.rollout.quantization="fp8"
+)
+
+DATA=(
+    #  dddd
+    data.train_files="${TRAIN_FILE}"
+    data.val_files="${TEST_FILE}"
+    data.prompt_key=prompt
+    data.return_raw_chat=$return_raw_chat
+    data.truncation='left'
+    data.max_prompt_length=${max_prompt_length}
+    data.max_response_length=${max_response_length}
+    data.train_batch_size=${train_prompt_bsz}
+)
+
+REWARD_MODEL=(
+    +reward_model.reward_kwargs.overlong_buffer_cfg.enable=${enable_overlong_buffer}
+    +reward_model.reward_kwargs.overlong_buffer_cfg.len=${overlong_buffer_len}
+    +reward_model.reward_kwargs.overlong_buffer_cfg.penalty_factor=${overlong_penalty_factor}
+    +reward_model.reward_kwargs.overlong_buffer_cfg.log=False
+    +reward_model.reward_kwargs.max_resp_len=${max_response_length}
+    reward_model.reward_manager=dapo
+)
+
+PERF_OPT=(
+    +actor_rollout_ref.actor.megatron.override_transformer_config.apply_rope_fusion=True
+    actor_rollout_ref.actor.megatron.use_remove_padding=False
+    +actor_rollout_ref.actor.megatron.override_transformer_config.recompute_method=uniform
+    +actor_rollout_ref.actor.megatron.override_transformer_config.recompute_granularity=full
+    +actor_rollout_ref.actor.megatron.override_transformer_config.recompute_num_layers=1
+    actor_rollout_ref.actor.megatron.override_transformer_config.attention_backend=auto
+    +actor_rollout_ref.actor.optim.override_optimizer_config.optimizer_offload_fraction=1
+    +actor_rollout_ref.actor.optim.override_optimizer_config.overlap_cpu_optimizer_d2h_h2d=True
+    +actor_rollout_ref.actor.optim.override_optimizer_config.use_precision_aware_optimizer=True
+    +actor_rollout_ref.actor.optim.override_optimizer_config.optimizer_cpu_offload=True
+)
+
+ACTOR=(
+    actor_rollout_ref.actor.use_kl_loss=${use_kl_loss}
+    actor_rollout_ref.actor.kl_loss_coef=${kl_loss_coef}
+    actor_rollout_ref.actor.clip_ratio_low=${clip_ratio_low}
+    actor_rollout_ref.actor.clip_ratio_high=${clip_ratio_high}
+    actor_rollout_ref.actor.clip_ratio_c=10.0
+    actor_rollout_ref.actor.ppo_micro_batch_size_per_gpu=2
+    actor_rollout_ref.actor.use_dynamic_bsz=${use_dynamic_bsz}
+    actor_rollout_ref.actor.ppo_max_token_len_per_gpu=${actor_ppo_max_token_len}
+    actor_rollout_ref.actor.optim.lr=1e-6
+    actor_rollout_ref.actor.optim.lr_warmup_steps=10
+    actor_rollout_ref.actor.optim.weight_decay=0.1
+    actor_rollout_ref.actor.optim.clip_grad=1.0
+    actor_rollout_ref.actor.ppo_mini_batch_size=${train_prompt_mini_bsz}
+    actor_rollout_ref.actor.megatron.param_offload=${offload}
+    actor_rollout_ref.actor.megatron.optimizer_offload=${offload}
+    actor_rollout_ref.actor.megatron.grad_offload=${offload}
+    actor_rollout_ref.actor.megatron.pipeline_model_parallel_size=${train_pp}
+    actor_rollout_ref.actor.megatron.tensor_model_parallel_size=${train_tp}
+    actor_rollout_ref.actor.megatron.expert_model_parallel_size=${EP}
+    actor_rollout_ref.actor.megatron.expert_tensor_parallel_size=${ETP}
+    actor_rollout_ref.actor.entropy_coeff=0
+    actor_rollout_ref.actor.loss_agg_mode=${loss_agg_mode}
+    actor_rollout_ref.actor.megatron.use_mbridge=True
+    actor_rollout_ref.actor.megatron.vanilla_mbridge=False
+    actor_rollout_ref.model.use_remove_padding=False
+)
+
+ROLLOUT=(
+    actor_rollout_ref.rollout.name=${rollout_name}
+    actor_rollout_ref.rollout.mode=${rollout_mode}
+    actor_rollout_ref.rollout.dtype=${dtype}
+    actor_rollout_ref.rollout.gpu_memory_utilization=0.7
+    actor_rollout_ref.rollout.tensor_model_parallel_size=${gen_tp}
+    actor_rollout_ref.rollout.enable_chunked_prefill=True
+    actor_rollout_ref.rollout.max_num_batched_tokens=$((max_prompt_length + max_response_length))
+    actor_rollout_ref.rollout.temperature=${temperature}
+    actor_rollout_ref.rollout.top_p=${top_p}
+    actor_rollout_ref.rollout.top_k=${top_k}
+    actor_rollout_ref.rollout.val_kwargs.temperature=${temperature}
+    actor_rollout_ref.rollout.val_kwargs.top_p=${val_top_p}
+    actor_rollout_ref.rollout.val_kwargs.top_k=${top_k}
+    actor_rollout_ref.rollout.val_kwargs.do_sample=True
+    actor_rollout_ref.rollout.val_kwargs.n=1
+    actor_rollout_ref.rollout.calculate_log_probs=True
+    actor_rollout_ref.rollout.n=${n_resp_per_prompt}
+)
+
+TRAINER=(
+    trainer.logger=['console','wandb']
+    trainer.project_name="${project_name}"
+    trainer.experiment_name="${exp_name}"
+    trainer.n_gpus_per_node=8
+    trainer.nnodes="${NNODES}"
+    trainer.val_before_train=False
+    trainer.test_freq=5
+    trainer.save_freq=-1
+    trainer.total_epochs=10
+    trainer.default_local_dir="${CKPTS_DIR}"
+    trainer.resume_mode=auto
+    trainer.log_val_generations=10
+)
+
+FORWARD_ONLY_SETS=(
+    actor_rollout_ref.ref.log_prob_micro_batch_size_per_gpu=4
+    actor_rollout_ref.rollout.log_prob_micro_batch_size_per_gpu=4
+    actor_rollout_ref.ref.log_prob_use_dynamic_bsz=${use_dynamic_bsz}
+    actor_rollout_ref.rollout.log_prob_use_dynamic_bsz=${use_dynamic_bsz}
+    actor_rollout_ref.ref.log_prob_max_token_len_per_gpu=${infer_ppo_max_token_len}
+    actor_rollout_ref.rollout.log_prob_max_token_len_per_gpu=${infer_ppo_max_token_len}
+)
+
+MODEL=(
+    actor_rollout_ref.model.path="${MODEL_PATH}"
+)
+
+ALGORITHM=(
+    algorithm.adv_estimator=${adv_estimator}
+    algorithm.use_kl_in_reward=${use_kl_in_reward}
+    algorithm.kl_ctrl.kl_coef=${kl_coef}
+)
+################################################### start script ###################################################
+
+python3 -m verl.trainer.main_ppo \
+    --config-path=config \
+    --config-name='ppo_megatron_trainer.yaml' \
+    "${DATA[@]}" \
+    "${ALGORITHM[@]}" \
+    "${MODEL[@]}" \
+    "${ROLLOUT[@]}" \
+    "${ACTOR[@]}" \
+    "${REWARD_MODEL[@]}" \
+    "${FP8[@]}" \
+    "${PERF_OPT[@]}" \
+    "${TRAINER[@]}" \
+    "${FORWARD_ONLY_SETS[@]}" \