#!/bin/bash
pkill -9 sglang
sleep 3
ray stop --force
pkill -9 ray
pkill -9 python
sleep 3
pkill -9 ray
pkill -9 python
set -ex
export PYTHONUNBUFFERED=1
CKPT_ARGS=(
--hf-checkpoint /root/Qwen3-0.6B
)
ROLLOUT_ARGS=(
--prompt-data /root/dapo-math-17k/dapo-math-17k.jsonl
--input-key prompt
--label-key label
--apply-chat-template
--rollout-shuffle
--rm-type deepscaler
--num-rollout 2
--rollout-batch-size 4
--n-samples-per-prompt 4
--rollout-max-response-len 8192
--rollout-temperature 0.8
--global-batch-size 16
)
GSPO_ARGS=(
--advantage-estimator gspo
--kl-loss-coef 0.00
--kl-loss-type low_var_kl
--kl-coef 0.00
--entropy-coef 0.00
--eps-clip 3.5e-4
)
OPTIMIZER_ARGS=(
--optimizer adam
--lr 1e-6
--lr-decay-style constant
--weight-decay 0.1
--adam-beta1 0.9
--adam-beta2 0.98
)
SGLANG_ARGS=(
--rollout-num-gpus-per-engine 1
)
ray start --head --node-ip-address 127.0.0.1 --num-gpus 4 --disable-usage-stats
ray job submit --address="http://127.0.0.1:8265" \
--runtime-env-json='{
"env_vars": {
"no_proxy": "localhost,127.0.0.1,0.0.0.0,${MASTER_ADDR}"
}
}' \
-- python3 train.py \
--actor-num-nodes 1 \
--actor-num-gpus-per-node 4 \
--colocate \
--train-backend megatron \
${CKPT_ARGS[@]} \
${ROLLOUT_ARGS[@]} \
${OPTIMIZER_ARGS[@]} \
${GSPO_ARGS[@]} \
${SGLANG_ARGS[@]}