slurm_train.sh 622 B

12345678910111213141516171819202122232425
  1. #!/usr/bin/env bash
  2. # Copyright (c) OpenMMLab. All rights reserved.
  3. set -x
  4. PARTITION=$1
  5. JOB_NAME=$2
  6. CONFIG=$3
  7. WORK_DIR=$4
  8. GPUS=${GPUS:-8}
  9. GPUS_PER_NODE=${GPUS_PER_NODE:-8}
  10. CPUS_PER_TASK=${CPUS_PER_TASK:-5}
  11. SRUN_ARGS=${SRUN_ARGS:-""}
  12. PY_ARGS=${@:5}
  13. PYTHONPATH="$(dirname $0)/..":$PYTHONPATH \
  14. srun -p ${PARTITION} \
  15. --job-name=${JOB_NAME} \
  16. --gres=gpu:${GPUS_PER_NODE} \
  17. --ntasks=${GPUS} \
  18. --ntasks-per-node=${GPUS_PER_NODE} \
  19. --cpus-per-task=${CPUS_PER_TASK} \
  20. --kill-on-bad-exit=1 \
  21. ${SRUN_ARGS} \
  22. python -u tools/train.py ${CONFIG} --work-dir=${WORK_DIR} --launcher="slurm" ${PY_ARGS}