run-13b.sh 468 Bytes
Newer Older
zhaoying1's avatar
UPDATE  
zhaoying1 committed
1
2
3
4
5
6
7
8
9
10
ulimit -u 200000
export HSA_FORCE_FINE_GRAIN_PCIE=1
export MIOPEN_FIND_MODE=3
export MIOPEN_COMPILE_PARALLEL_LEVEL=1
export NCCL_DEBUG=INFO
export NCCL_SOCKET_IFNAME=ib0
export NCCL_P2P_LEVEL=5


echo "START TIME: $(date)"
zhaoying1's avatar
zhaoying1 committed
11
hostfile=./hostfile
zhaoying1's avatar
UPDATE  
zhaoying1 committed
12
13

np=$(cat $hostfile|sort|uniq |wc -l)
zhaoying1's avatar
zhaoying1 committed
14
np=$(($np*8))
zhaoying1's avatar
UPDATE  
zhaoying1 committed
15
16
17


which mpirun
zhaoying1's avatar
zhaoying1 committed
18
mpirun -np $np --allow-run-as-root --hostfile hostfile --bind-to none --mca btl_tcp_if_include enp97s0f1 `pwd`/run-13b-single.sh $np
zhaoying1's avatar
zhaoying1 committed
19
echo "END TIME: $(date)"