slurm_train.sh 574 B

123456789101112131415161718192021222324
  1. #!/usr/bin/env bash
  2. set -x
  3. PARTITION=$1
  4. JOB_NAME=$2
  5. CONFIG=$3
  6. WORK_DIR=$4
  7. GPUS=${GPUS:-8}
  8. GPUS_PER_NODE=${GPUS_PER_NODE:-8}
  9. CPUS_PER_TASK=${CPUS_PER_TASK:-5}
  10. SRUN_ARGS=${SRUN_ARGS:-""}
  11. PY_ARGS=${@:5}
  12. PYTHONPATH="$(dirname $0)/..":$PYTHONPATH \
  13. srun -p ${PARTITION} \
  14. --job-name=${JOB_NAME} \
  15. --gres=gpu:${GPUS_PER_NODE} \
  16. --ntasks=${GPUS} \
  17. --ntasks-per-node=${GPUS_PER_NODE} \
  18. --cpus-per-task=${CPUS_PER_TASK} \
  19. --kill-on-bad-exit=1 \
  20. ${SRUN_ARGS} \
  21. python -u tools/train.py ${CONFIG} --work-dir=${WORK_DIR} --launcher="slurm" ${PY_ARGS}