Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
41 commits
Select commit Hold shift + click to select a range
c127dd3
feat(driver): add distributed driver benchmark tests and options
msimberg Aug 25, 2026
4374272
feat(nox): add benchmark_driver_mpi and bencher upload sessions
msimberg Aug 25, 2026
c53954b
fix(nox): exclude benchmark_only tests from default test_model_mpi se…
msimberg Aug 25, 2026
013811b
fix: rebuild driver and state in timeloop benchmark setup [R1]
msimberg Aug 25, 2026
3721dbe
fix: function-scope grid fixture for clean limited-area xfail [R7]
msimberg Aug 25, 2026
f58b0b5
fix: assert single-step for grid overrides [R4]
msimberg Aug 25, 2026
5d8e9c2
fix: use is_upload_rank directly and drop redundant -k [R5][R6]
msimberg Aug 25, 2026
486dd5b
fix: call resolve_rank once in driver bencher upload sessions [R8]
msimberg Aug 26, 2026
49ec68f
fix: clear rank env in is_upload_rank explicit-none test [R10]
msimberg Aug 26, 2026
a4fb1c8
refactor(ci): factor bencher driver job anchors and drop intra-node v…
msimberg Aug 31, 2026
adb0466
refactor(nox): share bencher upload helpers, fix driver JSON path, dr…
msimberg Aug 31, 2026
2aee9e0
feat(benchmark): append GHEX transport to driver bencher testbed
msimberg Aug 31, 2026
958dc27
feat(ci): use R02B06 global grid and single step for distributed driv…
msimberg Aug 31, 2026
7d3224b
refactor(driver): extract initialize_driver_states helper
msimberg Sep 2, 2026
eaceffa
refactor(benchmark): hardcode JW, 100 steps and 50 s dtime for MPI dr…
msimberg Sep 2, 2026
44ff10b
feat(benchmark): add single-rank driver benchmark
msimberg Sep 2, 2026
9f9fe32
refactor(benchmark): remove testing.benchmark module and driver-bench…
msimberg Sep 2, 2026
2e7a0c7
feat(ci): add single-rank driver bencher jobs and hardcode benchmark …
msimberg Sep 2, 2026
a45264c
feat(benchmark): parametrize driver benchmarks by experiment with ids…
msimberg Sep 2, 2026
b4c26b7
ci(benchmark): temporarily disable serial bencher jobs for driver-ben…
msimberg Sep 2, 2026
1880f2e
Merge remote-tracking branch 'origin/main' into bencher-distributed-d…
msimberg Sep 2, 2026
1b4b726
ci(benchmark): run driver benchmark with 4 ranks / 1 node (mpitask4 d…
msimberg Sep 2, 2026
259418a
ci(benchmark): pin OMP threads and use full-node GPU layout for drive…
msimberg Sep 3, 2026
6921efd
refactor(ci): follow naming conventions and disable MPS for driver be…
msimberg Sep 3, 2026
f54f263
ci(benchmark): pin OMP_NUM_THREADS to 64 for driver bencher
msimberg Sep 3, 2026
1d5da84
ci(benchmark): pin OMP_NUM_THREADS to 32 for driver bencher
msimberg Sep 3, 2026
4c32a6e
ci(benchmark): pin OMP_NUM_THREADS to 8 for driver bencher
msimberg Sep 3, 2026
2d7d16f
ci(benchmark): settle OMP_NUM_THREADS at 72 for driver bencher
msimberg Sep 3, 2026
4665673
refactor: address review round on driver bencher benchmarks
msimberg Sep 4, 2026
1d034e7
ci(benchmark): drop .retry_on_transient_failure from driver bencher jobs
msimberg Sep 4, 2026
417e924
Consolidate driver benchmark fixtures and factor grid resolution
msimberg Sep 4, 2026
f7ffd98
fix: label MPI benchmark process_props param id as distributed
msimberg Sep 16, 2026
b2223b1
fix: reuse _serial_testbed in _driver_mpi_bencher_testbed
msimberg Sep 16, 2026
c803d2d
fix: clarify MPS not enabled comments in benchmark CI
msimberg Sep 16, 2026
bbca49f
fix: address driver-benchmark review round; restore fixtures.py re-ex…
msimberg Sep 17, 2026
a381e86
Apply suggestion from @msimberg
msimberg Sep 17, 2026
f746d52
Apply suggestion from @msimberg
msimberg Sep 17, 2026
81ccc5a
Apply suggestion from @msimberg
msimberg Sep 17, 2026
c9641df
Apply suggestion from @msimberg
msimberg Sep 17, 2026
ff7553f
refactor: parameterize driver_benchmark_experiment fixture directly
msimberg Sep 17, 2026
6c8eeb3
Remove rules section from benchmark_bencher_baseline.yml
msimberg Sep 17, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
38 changes: 29 additions & 9 deletions .cscs-ci/benchmark_bencher.yml
Original file line number Diff line number Diff line change
Expand Up @@ -5,19 +5,29 @@ include:
.bencher_feature_tests:
extends: [.benchmark_nox_job]
script:
- export PR_ID=$(echo "${CI_COMMIT_BRANCH}" | grep -o 'pr[0-9]*' | grep -o '[0-9]*')
- export FEATURE_BRANCH=$(curl -s https://api.github.com/repos/C2SM/icon4py/pulls/$PR_ID | jq -r '.head.ref')
- export GITHUB_ACTIONS=true
- export GITHUB_EVENT_NAME=pull_request
- export GITHUB_STEP_SUMMARY=$CI_PROJECT_DIR/step_summary.log
- export GITHUB_SHA=$CI_COMMIT_SHA
- export GITHUB_EVENT_PATH=$CI_PROJECT_DIR/event.json
- |
echo "{\"pull_request\": {\"head\": {\"repo\": {\"full_name\": \"C2SM/icon4py\"}}}, \"repository\": {\"full_name\": \"C2SM/icon4py\"}, \"number\": $PR_ID}" > $CI_PROJECT_DIR/event.json
- !reference [.bencher_github_event_setup, script]
- !reference [.benchmark_nox_job, script]
variables:
NOX_SESSION: "__bencher_feature_branch_CI-${PYVERSION_SHORT}"

.bencher_driver_mpi_feature_tests:
extends: [.test_runner_mpi, .test_template_aarch64, .benchmark_driver_mpi_nox_job]
stage: benchmark
script:
- !reference [.bencher_github_event_setup, script]
- !reference [.benchmark_driver_mpi_nox_job, script]
variables:
NOX_SESSION: "__bencher_driver_mpi_feature_branch_CI-${PYVERSION_SHORT}"

.bencher_driver_feature_tests:
extends: [.test_runner_serial, .test_template_aarch64, .benchmark_driver_nox_job]
stage: benchmark
script:
- !reference [.bencher_github_event_setup, script]
- !reference [.benchmark_driver_nox_job, script]
variables:
NOX_SESSION: "__bencher_driver_feature_branch_CI-${PYVERSION_SHORT}"

benchmark_bencher_stencils_feature_aarch64:
extends: [.test_runner_serial, .test_template_aarch64, .bencher_feature_tests]
variables:
Expand All @@ -28,3 +38,13 @@ benchmark_bencher_granules_feature_aarch64:
extends: [.test_runner_serial, .test_template_aarch64, .bencher_feature_tests]
variables:
TEST_SELECTION: "${GRANULE_TEST_SELECTION}"

benchmark_bencher_driver_mpi_feature_aarch64:
extends: [.bencher_driver_mpi_feature_tests]
variables:
SLURM_NTASKS: 4
SLURM_JOB_NUM_NODES: 1
ICON4PY_TEST_MPI_SUBCOMM_SIZE: $SLURM_NTASKS

benchmark_bencher_driver_feature_aarch64:
extends: [.bencher_driver_feature_tests]
22 changes: 22 additions & 0 deletions .cscs-ci/benchmark_bencher_baseline.yml
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,18 @@ include:
- local: '.cscs-ci/base.yml'
- local: '.cscs-ci/benchmark_bencher_common.yml'

.bencher_driver_mpi_baseline_tests:
extends: [.benchmark_driver_mpi_nox_job]
stage: benchmark
variables:
NOX_SESSION: "__bencher_driver_mpi_baseline_CI-${PYVERSION_SHORT}"

.bencher_driver_baseline_tests:
extends: [.benchmark_driver_nox_job]
stage: benchmark
variables:
NOX_SESSION: "__bencher_driver_baseline_CI-${PYVERSION_SHORT}"

benchmark_bencher_stencils_baseline_aarch64:
extends: [.test_runner_serial, .test_template_aarch64, .benchmark_nox_job]
variables:
Expand All @@ -14,3 +26,13 @@ benchmark_bencher_granules_baseline_aarch64:
variables:
NOX_SESSION: "__bencher_baseline_CI-${PYVERSION_SHORT}"
TEST_SELECTION: "${GRANULE_TEST_SELECTION}"

benchmark_bencher_driver_mpi_baseline_aarch64:
extends: [.test_runner_mpi, .test_template_aarch64, .bencher_driver_mpi_baseline_tests]
variables:
SLURM_NTASKS: 4
SLURM_JOB_NUM_NODES: 1
ICON4PY_TEST_MPI_SUBCOMM_SIZE: $SLURM_NTASKS

benchmark_bencher_driver_baseline_aarch64:
extends: [.test_runner_serial, .test_template_aarch64, .bencher_driver_baseline_tests]
53 changes: 53 additions & 0 deletions .cscs-ci/benchmark_bencher_common.yml
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,13 @@
OMP_PLACES: cores
OMP_NUM_THREADS: 72

.benchmark_driver_base_variables:
variables:
SLURM_TIMELIMIT: '02:00:00'
OMP_PROC_BIND: close
OMP_PLACES: cores
OMP_NUM_THREADS: 72

.benchmark_nox_job:
stage: benchmark
script:
Expand All @@ -27,6 +34,52 @@
SLURM_PARTITION: normal
SLURM_TIMELIMIT: '01:30:00'

.benchmark_driver_mpi_nox_job:
stage: benchmark
script:
- .cscs-ci/scripts/ci-mpi-wrapper.sh nox -s "${NOX_SESSION}" -- --backend=$BACKEND --grid=$GRID
# MPS is explicitly not enabled here. The job uses one GPU per task.
before_script:
- cd /icon4py
- source .cscs-ci/scripts/gt4py-cache.sh
extends: [.benchmark_driver_base_variables]
variables:
# allow-task-sharing lets NCCL open peer GPUs for GPU-GPU communication.
Comment thread
msimberg marked this conversation as resolved.
SLURM_GPUS_PER_NODE: 4
SLURM_CPUS_PER_TASK: 72
SLURM_GRES_FLAGS: allow-task-sharing
parallel:
matrix:
- BACKEND: [dace_cpu, dace_gpu, gtfn_cpu, gtfn_gpu]
GRID: [R02B06_GLOBAL]
.benchmark_driver_nox_job:

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Only a nit comment. Most other configurations are separate by a blank line.

Suggested change
.benchmark_driver_nox_job:
.benchmark_driver_nox_job:

stage: benchmark
script:
- nox -s "${NOX_SESSION}" -- --backend=$BACKEND --grid=$GRID
# MPS is explicitly not enabled here. The job uses one GPU per task.
before_script:
- cd /icon4py
- source .cscs-ci/scripts/gt4py-cache.sh
extends: [.benchmark_driver_base_variables]
variables:
SLURM_GPUS_PER_TASK: 1
SLURM_CPUS_PER_TASK: 72
parallel:
matrix:
- BACKEND: [dace_cpu, dace_gpu, gtfn_cpu, gtfn_gpu]
GRID: [R02B06_GLOBAL]
.bencher_github_event_setup:

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Suggested change
.bencher_github_event_setup:
.bencher_github_event_setup:

script:
- export PR_ID=$(echo "${CI_COMMIT_BRANCH}" | grep -o 'pr[0-9]*' | grep -o '[0-9]*')
- export FEATURE_BRANCH=$(curl -s https://api.github.com/repos/C2SM/icon4py/pulls/$PR_ID | jq -r '.head.ref')
- export GITHUB_ACTIONS=true
- export GITHUB_EVENT_NAME=pull_request
- export GITHUB_STEP_SUMMARY=$CI_PROJECT_DIR/step_summary.log
- export GITHUB_SHA=$CI_COMMIT_SHA
- export GITHUB_EVENT_PATH=$CI_PROJECT_DIR/event.json
- |
echo "{\"pull_request\": {\"head\": {\"repo\": {\"full_name\": \"C2SM/icon4py\"}}}, \"repository\": {\"full_name\": \"C2SM/icon4py\"}, \"number\": $PR_ID}" > $CI_PROJECT_DIR/event.json


build_baseimage_aarch64:
extends: [.build_baseimage_aarch64]
Expand Down
42 changes: 28 additions & 14 deletions model/driver/src/icon4py/model/driver/driver.py
Original file line number Diff line number Diff line change
Expand Up @@ -797,20 +797,16 @@ def initialize_driver(
return icon4py_driver


def run_driver(
*,
config: driver_config.ExperimentConfig,
grid_manager: gm.GridManager,
process_props: decomposition_defs.ProcessProperties,
backend: gtx.typing.Backend | None,
) -> tuple[driver_states.DriverStates, Icon4pyDriver]:
icon4py_driver = initialize_driver(
config=config,
grid_manager=grid_manager,
process_props=process_props,
backend=backend,
)
allocator = model_backends.get_allocator(backend)
def initialize_driver_states(
icon4py_driver: Icon4pyDriver,
allocator: gtx.typing.Allocator,
) -> driver_states.DriverStates:
Comment thread
msimberg marked this conversation as resolved.
"""Build and validate the initial driver states from a freshly initialized driver.

This wraps the creation of the prognostic/tracer/nonhydro/diagnostic states, the
application of the initial condition, the assembly of ``DriverStates``, and the
consistency check that precedes the time loop.
"""
prognostic_state_now = prognostics.initialize_prognostic_state(
grid=icon4py_driver.grid,
allocator=allocator,
Expand Down Expand Up @@ -859,5 +855,23 @@ def run_driver(
granules=icon4py_driver.granules,
states=ds,
)
return ds


def run_driver(
*,
config: driver_config.ExperimentConfig,
grid_manager: gm.GridManager,
process_props: decomposition_defs.ProcessProperties,
backend: gtx.typing.Backend | None,
) -> tuple[driver_states.DriverStates, Icon4pyDriver]:
icon4py_driver = initialize_driver(
config=config,
grid_manager=grid_manager,
process_props=process_props,
backend=backend,
)
allocator = model_backends.get_allocator(backend)
ds = initialize_driver_states(icon4py_driver=icon4py_driver, allocator=allocator)
icon4py_driver.time_integration(ds)
return ds, icon4py_driver
93 changes: 93 additions & 0 deletions model/driver/tests/driver/fixtures.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,8 +5,16 @@
#
# Please, refer to the LICENSE file in the root directory.
# SPDX-License-Identifier: BSD-3-Clause
from __future__ import annotations

import gt4py.next.typing as gtx_typing
import pytest

from icon4py.model.common import model_backends, time
from icon4py.model.common.decomposition import definitions as decomp_defs
from icon4py.model.common.grid import grid_manager as gm
from icon4py.model.driver import config as driver_config, driver_utils
from icon4py.model.testing import datatest_utils as dt_utils, definitions as test_defs, grid_utils
from icon4py.model.testing.fixtures import (
backend,
backend_like,
Expand Down Expand Up @@ -35,3 +43,88 @@
@pytest.fixture
def linit(timeloop_diffusion_linit_exit: bool) -> bool:
return timeloop_diffusion_linit_exit


BENCHMARK_EXPERIMENTS: list[test_defs.ExperimentDescription] = [test_defs.Experiments.JW]
BENCHMARK_STEPS: int = 100
BENCHMARK_ROUNDS: int = 5
BENCHMARK_WARMUP_ROUNDS: int = 2
Comment thread
msimberg marked this conversation as resolved.


def _resolve_grid(grid_option: str) -> test_defs.GridDescription:
name = grid_option.split(":", maxsplit=1)[0].strip()
try:
return getattr(test_defs.Grids, name)
except AttributeError:
raise pytest.UsageError(f"Unknown grid '{name}' in '--grid' option") from None


def _make_config(
experiment: test_defs.ExperimentDescription,
grid: test_defs.GridDescription,
process_props: decomp_defs.ProcessProperties,
) -> driver_config.ExperimentConfig:
dt_utils.download_experiment(experiment, process_props)
experiment_path = dt_utils.get_path_for_experiment(experiment, process_props)
config = driver_config.read_experiment_config_from_fortran(experiment_path)
return config.with_overrides(
driver={
"dtime": time.RelativeTime(seconds=50),
"enable_output": False,
"end_of_simulation": time.NumTimeSteps(BENCHMARK_STEPS),
}
)


def _make_grid_manager(
config: driver_config.ExperimentConfig,
grid: test_defs.GridDescription,
process_props: decomp_defs.ProcessProperties,
backend: gtx_typing.Backend | None,
) -> gm.GridManager:
allocator = model_backends.get_allocator(backend)
grid_file_path = grid_utils._download_grid_file(grid)
return driver_utils.create_grid_manager(
grid_file_path=grid_file_path,
vertical_grid_config=config.vertical_grid,
allocator=allocator,
process_props=process_props,
)


@pytest.fixture(params=BENCHMARK_EXPERIMENTS, ids=lambda experiment: experiment.name)
def driver_benchmark_experiment(request: pytest.FixtureRequest) -> test_defs.ExperimentDescription:
return request.param


@pytest.fixture
def driver_benchmark_grid(
request: pytest.FixtureRequest,
driver_benchmark_experiment: test_defs.ExperimentDescription,
) -> test_defs.GridDescription:
grid_option = request.config.getoption("--grid")
return driver_benchmark_experiment.grid if grid_option is None else _resolve_grid(grid_option)


@pytest.fixture
def driver_benchmark_config(
driver_benchmark_experiment: test_defs.ExperimentDescription,
driver_benchmark_grid: test_defs.GridDescription,
process_props: decomp_defs.ProcessProperties,
) -> driver_config.ExperimentConfig:
return _make_config(driver_benchmark_experiment, driver_benchmark_grid, process_props)


@pytest.fixture
def driver_benchmark_grid_manager(
driver_benchmark_config: driver_config.ExperimentConfig,
driver_benchmark_grid: test_defs.GridDescription,
process_props: decomp_defs.ProcessProperties,
backend: gtx_typing.Backend | None,
) -> gm.GridManager:
return _make_grid_manager(
config=driver_benchmark_config,
grid=driver_benchmark_grid,
process_props=process_props,
backend=backend,
)
Loading
Loading