@@ -8,23 +8,8 @@ REPO_DIR="SALMON2"
88VERSION_TAG=" v.2.2.2"
99BUILD_DIR=" build-benchkit"
1010
11- # RIKYU pin: v.2.2.2 is over a year stale (2025-06-06) and measured
12- # GPU-decomposition numbers on this repo were never actually run against it --
13- # they came from a much newer, hand-patched checkout (~100 commits ahead,
14- # including PR #1276's OpenACC tuning). Rather than let RIKYU silently build
15- # something nobody has benchmarked, pin it to FugakuNEXT-v4 on
16- # william-dawson/SALMON2 (also open as SALMON-TDDFT/SALMON2#1276) --
17- # develop-2.0.0@9b93a8c4 (2026-08-16) plus four commits, one per concern:
18- # 1. PR #1276's stencil/current/pseudo-pt OpenACC tuning (batched cuBLAS
19- # GEMM for pseudo-pt, not the old USE_CUDA hand-written kernels --
20- # see below)
21- # 2. the nvhpc-openacc-gemm.cmake platform file that wires it up
22- # 3. a fix for the Ewald loop tripcounts: nvfortran >= 26.5 evaluates
23- # an OpenACC tripcount on the device, where the derived-type member
24- # the bound came from is not resident, so the loops ran zero times
25- # 4. a fix for a 2+ node hang on nvhpc/26.5: the CUDA context is created
26- # before MPI_Init_thread, so the MPI layer's CUDA-awareness probe does
27- # not cache a transfer protocol that deadlocks on device buffers
11+ # v.2.2.2 is over a year stale and predates the OpenACC tuning, Ewald
12+ # tripcount fix and multi-node CUDA-context fix that RIKYU needs.
2813if [[ " ${system} " == " RIKYU" ]]; then
2914 REPO_URL=" https://github.com/william-dawson/SALMON2"
3015 VERSION_TAG=" FugakuNEXT-v4"
@@ -42,18 +27,14 @@ mkdir -p "${BUILD_LOG_DIR}"
4227bk_fetch_source " ${REPO_URL} " " ${REPO_DIR} " " ${VERSION_TAG} "
4328
4429cd " ${REPO_DIR} "
45- # RIKYU builds a pinned branch that already carries every fix it needs
46- # (see VERSION_TAG above), so it is used as-is -- no patching. The patches
47- # below exist for the systems that build stock v.2.2.2.
48- if [[ " ${system} " == " RIKYU" ]]; then
49- echo " RIKYU: building ${VERSION_TAG} unpatched (all fixes are in the branch)"
50- elif git apply --check " ${FJMPI_PATCH} " ; then
51- git apply " ${FJMPI_PATCH} "
52- elif git apply --reverse --check " ${FJMPI_PATCH} " > /dev/null 2>&1 ; then
53- echo " SALMON Fujitsu MPI topology patch is already applied"
54- else
55- echo " SALMON Fujitsu MPI topology patch does not apply to ${VERSION_TAG} " >&2
56- exit 1
30+ # Only Fugaku compiles the Fujitsu MPI path.
31+ if [[ " ${system} " == " Fugaku" || " ${system} " == " FugakuCN" ]]; then
32+ if git apply --check " ${FJMPI_PATCH} " ; then
33+ git apply " ${FJMPI_PATCH} "
34+ elif ! git apply --reverse --check " ${FJMPI_PATCH} " > /dev/null 2>&1 ; then
35+ echo " SALMON Fujitsu MPI topology patch does not apply to ${VERSION_TAG} " >&2
36+ exit 1
37+ fi
5738fi
5839
5940rm -rf " ${BUILD_DIR} "
@@ -208,48 +189,6 @@ case "${system}" in
208189 -DCMAKE_SYSTEM_PROCESSOR=openacc
209190 -DCMAKE_Fortran_FLAGS=" -O3 -Wall -fstrict-aliasing -acc=strict -gpu=cc100,managed,ptxinfo -cudalib=cublas,cusolver,nccl -cuda -Minfo=accel -DUSE_OPENACC -DUSE_GEMM -DUSE_NCCL"
210191 -DCMAKE_C_FLAGS=" -O3 -Wall -alias=ansi -acc=strict -gpu=cc100,managed,ptxinfo -cudalib=cublas,cusolver,nccl -cuda -Minfo=accel -DUSE_OPENACC -DUSE_GEMM -DUSE_NCCL"
211- # Module is nvhpc-hpcx-cuda13/26.5 (HPC-X 2.50 / OpenMPI 5), not
212- # plain nvhpc/26.5: that is the stack the domain-decomposition and
213- # NCCL numbers were measured on. MPI3 is left at its auto-detected
214- # ON -- no FORTRAN_COMPILER_HAS_MPI_VERSION3 override -- because the
215- # GEMM path skips calc_uVpsi_rdivided entirely, so the HPC-X libnbc
216- # bug that motivated the old 26.3 + MPI3=OFF workaround is
217- # unreachable. Do not reintroduce that workaround.
218- #
219- # FugakuNEXT-v3 also computes the nonlocal current density with a
220- # batched GEMM (it was 82% of calculating-curr): current density
221- # 29.5 -> 3.6 s and rt iterations 44.6 -> 18.7 s at 16 GPUs orbital.
222- #
223- # -DUSE_NCCL routes the pseudo-pt domain-decomposition reduce through
224- # ncclAllReduce instead of MPI_Allreduce. MPI_Allreduce performs the
225- # reduction arithmetic on the host even when handed a device pointer,
226- # so it leaves NVLink: measured 5.6 GB/s vs NCCL's 196-521 GB/s on the
227- # same 222 MB device buffer. In SALMON that is pseudo-pt comm 18.2s ->
228- # 1.7s at 4 GPUs (rt iterations 99.1 -> 80.5), energies bit-exact.
229- # It needs nccl in -cudalib. If the MPI stack ever grows a working
230- # GPU-side allreduce (UCC ships a TL_NCCL that currently never
231- # registers), drop -DUSE_NCCL and this becomes plain MPI again.
232- #
233- # NOT -DUSE_CUDA -- that flag controls a completely different, OLDER
234- # optimization path (src/common/{zpseudo,stencil_current}.cu, hand-
235- # written CUDA kernels) that this pinned source (see VERSION_TAG
236- # above) replaces with PR #1276's tuned pure-OpenACC kernels instead:
237- # a batched cuBLAS GEMM rewrite of pseudo-pt (-DUSE_GEMM, needs
238- # cusolver linked in) and an inlined OpenACC current-density kernel.
239- # That's where the real speedup comes from, not USE_CUDA -- measured
240- # data only ever showed the OLD CUDA kernels net *losing* time
241- # (pseudo-pt 3.1-3.5x slower under USE_CUDA than plain OpenACC; the
242- # 1.4-3x win on current-density wasn't enough to make up for it).
243- # USE_CUDA also has a real, deterministic bug independent of any of
244- # this: stencil_current.cu's host wrapper sizes its device idx/idy/idz
245- # buffers by each rank's LOCAL grid extent but indexes them with the
246- # RAW/global grid coordinate, which only happens to fit when a rank's
247- # is()=1 on that axis (single GPU, or pure orbital decomposition,
248- # where every rank owns the full box). Any real-space (nproc_rgrid>1)
249- # decomposition puts a non-first rank at is()>1 on the split axis and
250- # overruns the buffer -- reproduced as a deterministic Accelerator
251- # Fatal Error / CUDA_ERROR_ILLEGAL_ADDRESS in calc_current
252- # (density_matrix.f90) on every axis and Po x Pg combination tried.
253192 )
254193 ;;
255194 * )
0 commit comments