Skip to content

Commit 3e22530

Browse files
[code:MHDTurbulence] Add RIKYU support
- Add RIKYU rows, build, and run support (nvhpc-hpcx-cuda13/26.5) - Add patches fixing gfortran random_seed PUT size and CUDA-context-before-MPI - Add the per-rank CUDA_VISIBLE_DEVICES wrapper needed for multi-GPU launch Signed-off-by: William Dawson <william.dawson@riken.jp>
1 parent aaadb53 commit 3e22530

6 files changed

Lines changed: 125 additions & 0 deletions

File tree

‎programs/MHDTurbulence/build.sh‎

Lines changed: 17 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -36,6 +36,23 @@ case "$system" in
3636
echo "Executable is "${BIN}" and copied to "${artdir}
3737
cp ../exe/$BIN ../../${artdir}
3838
;;
39+
RIKYU)
40+
module load nvhpc-hpcx-cuda13/26.5
41+
# Portability fixes; skipped when already upstream
42+
grep -q "integer,dimension(8) :: seed" src_f90_omp_host/main.f90 || \
43+
patch -p1 < ../programs/${code}/patches/random_seed_put_size.patch
44+
grep -q "acc_init(acc_device_nvidia)" src_f90_acc_device/main.f90 || \
45+
patch -p1 < ../programs/${code}/patches/acc_init_before_mpi.patch
46+
# ntiles is compile-time: one binary per list.csv rank count
47+
cd src_f90_acc_device
48+
for tiles in 1 4 8; do
49+
make clean >/dev/null 2>&1 || true
50+
sed -i "s/ntiles(3) = \[.*,.*,.*\]/ntiles(3) = [ ${tiles},1,1 ]/" config.f90
51+
make > /dev/null
52+
echo "Executable is "${BIN}".t"${tiles}" and copied to "${artdir}
53+
cp ../exe/$BIN ../../${artdir}/${BIN}.t${tiles}
54+
done
55+
;;
3956
# in the future, we may add this
4057
# MiyabiG/OpenMP)
4158
# cd src_f90_omp_device

‎programs/MHDTurbulence/list.csv‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -2,3 +2,6 @@ system,enable,nodes,numproc_node,nthreads,elapse
22
Fugaku,no,1,4,12,0:10:00
33
MiyabiC,yes,1,96,1,0:10:00
44
MiyabiG,yes,1,1,1,0:10:00
5+
RIKYU,yes,1,1,1,0:10:00
6+
RIKYU,yes,1,4,1,0:10:00
7+
RIKYU,yes,2,4,1,0:10:00
Lines changed: 22 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,22 @@
1+
# Patches
2+
3+
Applied by `build.sh` only when the fix is not already present in the
4+
fetched source (grep-guarded), so they become no-ops once upstream
5+
merges them. Both originate from `william-dawson/MHDTurbulence`
6+
branch `gfortran-port` (candidate upstream PRs):
7+
8+
- `random_seed_put_size.patch` — commit `d9a19d9`: `random_seed(PUT=)`
9+
requires the array to match the compiler's seed size (8 for gfortran);
10+
the shipped size-2 array breaks strict compilers. Physically inert
11+
(the seeded RNG feeds a disabled perturbation, `rrv=0`).
12+
- `acc_init_before_mpi.patch` — commit `4db3adb`: initialize the CUDA
13+
context before `MPI_Init`. Without it, multi-node runs hang silently
14+
in the first `BoundaryCondition` (UCX "cannot find remote protocol"
15+
on the device-buffer `MPI_ISEND`, then `MPI_WAITALL` on a dead
16+
request; the app's `MPI_ERRORS_RETURN` handler means the failed ISEND
17+
returns an error code that is never checked).
18+
19+
Launch side (in `run.sh`, RIKYU case): each rank pins one GPU via
20+
`CUDA_VISIBLE_DEVICES=$OMPI_COMM_WORLD_LOCAL_RANK`; combined with the
21+
patch this is required for multi-node GPU runs. Validated 2026-09-03 on
22+
Rikyu (8 GPUs / 2 nodes, results bit-consistent with 1-GPU runs).
Lines changed: 19 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,19 @@
1+
diff --git a/src_f90_acc_device/main.f90 b/src_f90_acc_device/main.f90
2+
index c808716..3356a5e 100644
3+
--- a/src_f90_acc_device/main.f90
4+
+++ b/src_f90_acc_device/main.f90
5+
@@ -1,5 +1,6 @@
6+
program main
7+
use omp_lib
8+
+ use openacc
9+
use basicmod
10+
use mpimod
11+
use boundarymod
12+
@@ -9,6 +10,7 @@ program main
13+
logical::is_final
14+
logical,parameter:: forceoutput=.true., usualoutput=.false.
15+
data is_final /.false./
16+
+ call acc_init(acc_device_nvidia)
17+
call InitializeMPI
18+
if(myid_w == 0) print *, "setup grids and fields"
19+
if(myid_w == 0) print *, "grid size for x y z",ngrid1*ntiles(1),ngrid2*ntiles(2),ngrid3*ntiles(3)
Lines changed: 50 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,50 @@
1+
diff --git a/src_f90_acc_device/main.f90 b/src_f90_acc_device/main.f90
2+
index 669188c..c808716 100644
3+
--- a/src_f90_acc_device/main.f90
4+
+++ b/src_f90_acc_device/main.f90
5+
@@ -123,7 +123,7 @@ subroutine GenerateProblem
6+
real(8):: pi
7+
real(8):: den, B0, rho1, rho2, dv, wid, sig
8+
9+
- integer,dimension(2) :: seed
10+
+ integer,dimension(8) :: seed
11+
real(8),dimension(1) :: rnum
12+
real(8),parameter :: rrv =0.0d-2
13+
real(8):: dv_harm
14+
@@ -169,9 +169,10 @@ subroutine GenerateProblem
15+
16+
if(myid_w == 0) write(6,*) rrv*100.0d0 &
17+
& , "% of Randam Perturbation imposed on velocity"
18+
+ seed = 0
19+
seed(1) = 1
20+
seed(2) = 1 + myid_w*in*jn*kn
21+
- call random_seed(PUT=seed(1:2))
22+
+ call random_seed(PUT=seed)
23+
24+
! pert
25+
do k=ks,ke
26+
diff --git a/src_f90_omp_host/main.f90 b/src_f90_omp_host/main.f90
27+
index 62cc30b..d27d395 100644
28+
--- a/src_f90_omp_host/main.f90
29+
+++ b/src_f90_omp_host/main.f90
30+
@@ -120,7 +120,7 @@ subroutine GenerateProblem
31+
real(8):: pi
32+
real(8):: den, B0, rho1, rho2, dv, wid, sig
33+
34+
- integer,dimension(2) :: seed
35+
+ integer,dimension(8) :: seed
36+
real(8),dimension(1) :: rnum
37+
real(8),parameter :: rrv =0.0d-2
38+
39+
@@ -163,9 +163,10 @@ subroutine GenerateProblem
40+
41+
if(myid_w == 0) write(6,*) rrv*100.0d0 &
42+
& , "% of Randam Perturbation imposed on velocity"
43+
+ seed = 0
44+
seed(1) = 1
45+
seed(2) = 1 + myid_w*in*jn*kn
46+
- call random_seed(PUT=seed(1:2))
47+
+ call random_seed(PUT=seed)
48+
49+
! pert
50+
do k=ks,ke

‎programs/MHDTurbulence/run.sh‎

Lines changed: 14 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -40,6 +40,20 @@ case "$system" in
4040
elapsed=$(grep "sim time \[s\]:" ${LOG} | awk '{print $4}' | tail -n 1)
4141
tcc=$(grep "time/count/cell" ${LOG} | awk '{print $2}' | tail -n 1)
4242
;;
43+
RIKYU)
44+
module load nvhpc-hpcx-cuda13/26.5
45+
exedir=exe
46+
mkdir -p $code/$exedir/
47+
NP=$((nodes * numproc_node))
48+
cp $artdir/${BIN}.t${NP} $code/$exedir/${BIN}
49+
cd $code/$exedir/
50+
echo "code is executed in "$code/$exedir/
51+
# CUDA context must exist before MPI_Init and each rank must pin one
52+
# GPU, or the inter-node device exchange hangs (UCX protocol failure)
53+
mpirun -np "$NP" bash -c 'export CUDA_VISIBLE_DEVICES=$((OMPI_COMM_WORLD_LOCAL_RANK % 4)); exec ./Simulation.x' > ${LOG} 2>&1
54+
elapsed=$(grep "sim time \[s\]:" ${LOG} | awk '{print $4}' | tail -n 1)
55+
tcc=$(grep "time/count/cell" ${LOG} | awk '{print $2}' | tail -n 1)
56+
;;
4357
*)
4458
echo "Unknown system: $system"
4559
exit 1

0 commit comments

Comments
 (0)