mirror of
https://github.com/paboyle/Grid.git
synced 2026-08-28 21:39:35 +01:00
150 lines
5.2 KiB
Bash
150 lines
5.2 KiB
Bash
#!/bin/bash -l
|
|
#SBATCH --job-name=schur2d-ladder
|
|
#SBATCH --nodes=36
|
|
#SBATCH --ntasks-per-node=8
|
|
#SBATCH --cpus-per-task=7
|
|
#SBATCH --gpus-per-node=8
|
|
#SBATCH --time=2:00:00
|
|
#SBATCH --account=phy157_dwf
|
|
#SBATCH --gpu-bind=none
|
|
#SBATCH --exclusive
|
|
#SBATCH --mem=0
|
|
#SBATCH -S 0
|
|
##SBATCH -q debug
|
|
|
|
##############################################################################
|
|
# Staged shakeout of the 2D block-cyclic dense inverse (DENSE_SCHUR2D).
|
|
#
|
|
# F1 : unit tests, 1 node / 8 ranks (seconds)
|
|
# F2 : scale rehearsal, 1 node, N=13824 (seconds)
|
|
# F3 : scale rehearsal, 36 nodes, N=138240 (the timing number)
|
|
# F4 : production example, 1D baseline then 2D, same nodes
|
|
#
|
|
# F4's 2D leg is gated on F3 passing: if the synthetic production-shape
|
|
# inverse fails or is slow, do not spend the 400 s multigrid setup on it.
|
|
##############################################################################
|
|
|
|
cat << EOF > select_gpu
|
|
#!/bin/bash
|
|
export GPU_MAP=(0 1 2 3 7 6 5 4)
|
|
export NUMA_MAP=(3 3 1 1 2 2 0 0)
|
|
export GPU=\${GPU_MAP[\$SLURM_LOCALID]}
|
|
export NUMA=\${NUMA_MAP[\$SLURM_LOCALID]}
|
|
export HIP_VISIBLE_DEVICES=\$GPU
|
|
unset ROCR_VISIBLE_DEVICES
|
|
|
|
if [ \$SLURM_PROCID = "0" ]
|
|
then
|
|
echo \$*
|
|
fi
|
|
|
|
exec numactl -m \$NUMA -N \$NUMA \$*
|
|
EOF
|
|
|
|
chmod +x ./select_gpu
|
|
|
|
root=$HOME/ParallelIO/systems/Frontier
|
|
source $root/sourceme-rocm7.2.sh
|
|
|
|
export OMP_NUM_THREADS=7
|
|
export MPICH_GPU_SUPPORT_ENABLED=1
|
|
export MPICH_SMP_SINGLE_COPY_MODE=CMA
|
|
export MPICH_OFI_NIC_POLICY=GPU
|
|
|
|
module load libfabric
|
|
|
|
# Stale-environment protection: none of the gather knobs apply to the 2D
|
|
# path, and DENSE_GATHER=1 without FORCE would (correctly) abort at start.
|
|
unset DENSE_GATHER DENSE_GATHER_FORCE DENSE_GATHER_DEBUG DENSE_GATHER_MIN_BYTES
|
|
|
|
OPTS1="--accelerator-threads 8 --shm 4096 --shm-mpi 1 --device-mem 32000"
|
|
|
|
##############################################################################
|
|
echo "========================================================="
|
|
echo "F1: unit tests, 1 node, 8 ranks"
|
|
echo "========================================================="
|
|
##############################################################################
|
|
for t in Test_blockcyclic Test_summa Test_schur2d Test_schur2d_redist
|
|
do
|
|
echo "----- F1 $t -----"
|
|
srun -N1 -n8 ./select_gpu $root/tests/debug/$t --mpi 1.1.2.4 --grid 16.16.16.16 $OPTS1
|
|
done
|
|
|
|
##############################################################################
|
|
echo "========================================================="
|
|
echo "F2: scale rehearsal, 1 node, N=13824 (nb=1728, grid 2x4)"
|
|
echo "========================================================="
|
|
##############################################################################
|
|
S2D_N=13824 srun -N1 -n8 ./select_gpu $root/tests/debug/Test_schur2d_scale \
|
|
--mpi 1.1.2.4 --grid 16.16.16.16 $OPTS1
|
|
|
|
##############################################################################
|
|
echo "========================================================="
|
|
echo "F3: scale rehearsal, 36 nodes, N=138240 (nb=480, grid 18x16)"
|
|
echo " THE number: invert phase vs the 1D baseline 328 s"
|
|
echo "========================================================="
|
|
##############################################################################
|
|
S2D_N=138240 srun -N36 -n288 ./select_gpu $root/tests/debug/Test_schur2d_scale \
|
|
--mpi 3.6.4.4 --grid 48.48.48.96 $OPTS1
|
|
F3RC=$?
|
|
echo "F3 exit code $F3RC"
|
|
|
|
##############################################################################
|
|
echo "========================================================="
|
|
echo "F4: production example (1D baseline first, then 2D if F3 passed)"
|
|
echo "========================================================="
|
|
##############################################################################
|
|
|
|
vol=48.48.48.96
|
|
MPI_GEOM=3.6.4.4
|
|
|
|
export CoarseSmootherNstep=2
|
|
export FineSmootherOrder=6
|
|
|
|
export SUBSPACE_FILE=/lustre/orion/phy157/proj-shared/phy157_dwf/paboyle/subspace_nb64.scidac
|
|
unset SLAB_FILE # force a fresh import in both legs
|
|
|
|
export DENSE_SCHUR=1
|
|
export MASS=0.00078
|
|
|
|
export BLOCK=2.2.3.3 # L1->L2 blocking
|
|
export BLOCK2=4.4.2.4 # L2->L3 blocking
|
|
|
|
export FineSmootherShift=0.1
|
|
export CoarseSmootherShift=0.1
|
|
export CoarseSolverTol=0.05
|
|
export CoarseSolverOrder=200
|
|
export L3_TOL=3.0e-1
|
|
export L3_MAXIT=2
|
|
export L3_NSTEP=50
|
|
export OuterTol=1e-8
|
|
export OuterMmax=4
|
|
export OuterNstep=8
|
|
export DENSE_CC=1
|
|
export DENSE_APPLY_PROFILE=1
|
|
unset DENSE_CC_CHECK
|
|
export DENSE_SPLITK=128
|
|
export DENSE_DEVICE_SUM=1
|
|
export GRID_ALLOC_NCACHE_LARGE=64
|
|
export NRHS=4
|
|
|
|
echo "----- F4a: 1D baseline (DENSE_SCHUR2D unset) -----"
|
|
unset DENSE_SCHUR2D
|
|
srun -N36 -n288 ./select_gpu $root/examples/Example_pvdagm_v2_3level_DenseCoarseMatrix \
|
|
--mpi ${MPI_GEOM} --grid $vol $OPTS1 --comms-overlap
|
|
|
|
if [ $F3RC -eq 0 ]
|
|
then
|
|
echo "----- F4b: 2D block-cyclic (DENSE_SCHUR2D=1) -----"
|
|
export DENSE_SCHUR2D=1
|
|
srun -N36 -n288 ./select_gpu $root/examples/Example_pvdagm_v2_3level_DenseCoarseMatrix \
|
|
--mpi ${MPI_GEOM} --grid $vol $OPTS1 --comms-overlap
|
|
else
|
|
echo "----- F4b SKIPPED: F3 failed (rc=$F3RC), not spending the setup on it -----"
|
|
fi
|
|
|
|
echo "========================================================="
|
|
echo "ladder complete; binaries were:"
|
|
grep -h "git commit hash" slurm-$SLURM_JOB_ID.out 2>/dev/null | sort -u
|
|
echo "========================================================="
|