Repository navigation
Enable DaCe external workspace for AMD platform #1427
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
Changes from all commits
c1832e4
9f10099
2fe5ad3
3b8a1e3
79d56ed
66c5afd
dfa6755
a21a4f8
6cf1ea0
17541a0
8647b4b
54567df
617bdb4
7a03f68
56c6b04
f90d413
4c3b055
188e33d
37c1917
edd36fd
05e51aa
61680cf
e4e400c
14971c4
a99aa10
1e4d8ae
ff63592
20d1e37
521038f
81c1bf4
bfc4931
4d76e98
5bf29f7
670e45f
0ee19cc
fa2a068
c0ad736
d03d451
ddb44aa
0d5fa81
225e5d1
aafd9f5
25057d7
631b442
c41df38
d86639b
1c14235
88fc81a
5946289
7ca09b4
b213755
a636338
f99e84b
919dd0c
d26f0dd
fe819c0
b063f59
b2fcf95
a4cf8a1
c53f19c
66f190d
71686ee
fe68f2a
d374eec
ce3dd6c
21a063e
f66b3b0
d8c6257
8bdcabf
832723a
b73462f
c2ce6cb
002f84c
abcaffd
147c84e
916d993
9b84632
f026586
6c004f1
486cd2f
fcbce05
937c22d
5908b46
c07d6c3
765e3fa
4ff7844
5a27e39
d4e12ef
046a311
52e295a
e45e2de
a72bde0
7d7fc4a
365953b
73dacfb
733e606
c6a99a2
8e866e4
fe5334e
3af2e9d
83c598e
4f99e6a
34fa3b8
37c98d9
6adc24a
207d024
0de6b32
54e832e
b75972f
eda411f
35ae427
67adeb1
3fa6ab4
dcec1e7
ca9d260
7991377
564c10d
60e78f9
b3f7d51
530f311
041e58b
d50afa0
68e6d05
9929f11
404016a
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -0,0 +1,122 @@ | ||
| # ICON4Py - ICON inspired code in Python and GT4Py | ||
| # | ||
| # Copyright (c) 2022-2024, ETH Zurich and MeteoSwiss | ||
| # All rights reserved. | ||
| # | ||
| # Please, refer to the LICENSE file in the root directory. | ||
| # SPDX-License-Identifier: BSD-3-Clause | ||
| """External workspace allocation for the DaCe backend. | ||
|
|
||
| The DaCe backend in GT4Py can be configured with an | ||
| :class:`~gt4py.next.program_processors.runners.dace.workflow.common.ExternalWorkspace` | ||
| to provide workspace memory for transient SDFG arrays | ||
| (``transient_memory_mode = EXTERNAL``). When a :class:`BackendConfig` is | ||
| provided, :func:`get_dace_options <icon4py.model.common.model_options.get_dace_options>` | ||
| calls the :class:`IconWorkspaceAllocator` defined here — a process-wide | ||
| singleton that caches a single workspace slab per device and reuses it across | ||
| every compiled program. | ||
|
|
||
| The size of the workspace is configurable per experiment via :class:`BackendConfig` | ||
| (see :func:`backend_config_from_env` for an environment-variable based default). | ||
| """ | ||
|
|
||
| from __future__ import annotations | ||
|
|
||
| import dataclasses | ||
| import os | ||
| import typing | ||
| from collections.abc import Iterable | ||
| from typing import ClassVar, Final | ||
|
|
||
| import gt4py.next as gtx | ||
| from gt4py.next.program_processors.runners.dace.workflow import common as gtx_wfdcommon | ||
|
|
||
| from icon4py.model.common.config import options as common_conf_opt | ||
| from icon4py.model.common.utils import data_allocation | ||
|
|
||
|
|
||
| @dataclasses.dataclass(frozen=True, kw_only=True) | ||
| class BackendConfig: | ||
| """External DaCe workspace sizing, configurable per experiment.""" | ||
|
|
||
| workspace_size: typing.Annotated[ | ||
| int, | ||
| common_conf_opt.ConfigOption( | ||
| description=( | ||
| "Size of the workspace memory (in Bytes) for externally allocated " | ||
| "temporary fields. This is a performance feature of the DaCe backend " | ||
| "to avoid runtime allocation of temporary fields, for each program " | ||
| "call. Note that the memory buffer is allocated once and shared " | ||
| "across all compiled programs." | ||
| ), | ||
| icon_equivalent=None, | ||
| ), | ||
| ] = 256 * 1024 * 1024 # 256 MiB | ||
|
|
||
| def __post_init__(self) -> None: | ||
| if self.workspace_size <= 0: | ||
| raise ValueError(f"'workspace_size' must be positive, got {self.workspace_size}.") | ||
|
|
||
|
|
||
| def backend_config_from_env() -> BackendConfig | None: | ||
| """Build a :class:`BackendConfig` from environment variables. | ||
|
|
||
| Reads ``ICON4PY_BACKEND_WORKSPACE_SIZE`` and returns ``None`` when it is | ||
| not set. | ||
| """ | ||
| size = os.environ.get("ICON4PY_BACKEND_WORKSPACE_SIZE") | ||
| if size is None: | ||
| return None | ||
| return BackendConfig(workspace_size=int(size)) | ||
|
|
||
|
|
||
| def _get_slab(nbytes: int, device: gtx.DeviceType) -> data_allocation.NDArray: | ||
| """Allocate a `nbytes`-byte buffer allocated on ``device``.""" | ||
| xp = data_allocation.array_ns(use_cupy=(device != gtx.DeviceType.CPU)) | ||
| return xp.empty(nbytes, dtype=xp.uint8) | ||
|
|
||
|
|
||
| class IconWorkspaceAllocator: | ||
| """Singleton workspace allocator for the DaCe backend. | ||
|
|
||
| Exactly one instance exists per process (enforced by `__new__`); all DaCe | ||
| backends share it via the module-level `ICON_WORKSPACE_ALLOCATOR`. It keeps | ||
| a single private workspace slab per device in `_workspace_slabs`, reused | ||
| across every compiled program. On a cache hit the slab's size is validated | ||
| against the value passed to `allocate`. | ||
| """ | ||
|
|
||
| _instance: ClassVar[IconWorkspaceAllocator | None] = None | ||
| _workspace_slabs: ClassVar[dict[gtx.DeviceType, data_allocation.NDArray]] = {} | ||
|
|
||
| def __new__(cls) -> IconWorkspaceAllocator: | ||
| if cls._instance is None: | ||
| cls._instance = super().__new__(cls) | ||
| return cls._instance | ||
|
|
||
| def allocate( | ||
| self, | ||
| devices: gtx.DeviceType | Iterable[gtx.DeviceType], | ||
| *, | ||
| size: int, | ||
| ) -> gtx_wfdcommon.ExternalWorkspace: | ||
| if isinstance(devices, gtx.DeviceType): | ||
| devices = [devices] | ||
| wsp: gtx_wfdcommon.ExternalWorkspace = {} | ||
| for dev in devices: | ||
| if (cached := self._workspace_slabs.get(dev)) is not None: | ||
| if cached.nbytes != size: | ||
| raise ValueError( | ||
| f"Workspace size mismatch for {dev!s}: cached slab has " | ||
| f"{cached.nbytes} bytes but 'allocate' was called with " | ||
| f"size={size}." | ||
| ) | ||
| wsp[dev] = cached | ||
| else: | ||
| slab = _get_slab(size, dev) | ||
| self._workspace_slabs[dev] = slab | ||
| wsp[dev] = slab | ||
| return wsp | ||
|
|
||
|
|
||
| ICON_WORKSPACE_ALLOCATOR: Final[IconWorkspaceAllocator] = IconWorkspaceAllocator() | ||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -16,7 +16,7 @@ | |
| from gt4py.next import backend as gtx_backend | ||
| from gt4py.next.program_processors.runners.dace import transformations as gtx_transformations | ||
|
|
||
| from icon4py.model.common import model_backends | ||
| from icon4py.model.common import backend_configuration as backend_cfg, model_backends | ||
|
|
||
|
|
||
| log = logging.getLogger(__name__) | ||
|
|
@@ -35,11 +35,25 @@ def _dace_remove_access_node_copies(sdfg: dace.SDFG) -> None: | |
|
|
||
|
|
||
| def get_dace_options( | ||
| program_name: str, **backend_descriptor: Any | ||
| program_name: str, | ||
| backend_config: backend_cfg.BackendConfig | None, | ||
| **backend_descriptor: Any, | ||
| ) -> model_backends.BackendDescriptor: | ||
| is_rocm_device = backend_descriptor.get("device") == model_backends.DeviceType.ROCM | ||
| device = backend_descriptor.get("device") or model_backends.CPU | ||
| optimization_args = backend_descriptor.get("optimization_args", {}) | ||
| optimization_hooks = optimization_args.get("optimization_hooks", {}) | ||
|
|
||
| if backend_config is not None: | ||
| # The workspace memory allows to avoid the overhead of runtime allocations, | ||
| # which are expensive in the AMD runtime. | ||
| backend_descriptor["external_workspace"] = backend_cfg.ICON_WORKSPACE_ALLOCATOR.allocate( | ||
| device, | ||
|
edopao marked this conversation as resolved.
|
||
| size=backend_config.workspace_size, | ||
| ) | ||
| optimization_args["transient_memory_mode"] = ( | ||
| gtx_transformations.TransientMemoryMode.EXTERNAL | ||
| ) | ||
|
|
||
| if program_name in [ | ||
| "vertically_implicit_solver_at_corrector_step", | ||
| "vertically_implicit_solver_at_predictor_step", | ||
|
|
@@ -60,7 +74,7 @@ def get_dace_options( | |
| backend_descriptor["use_zero_origin"] = True | ||
| if program_name == "graupel_run": | ||
| optimization_args["fuse_tasklets"] = True | ||
| if not is_rocm_device: | ||
| if device != model_backends.DeviceType.ROCM: | ||
| optimization_args["gpu_maxnreg"] = 80 | ||
| optimization_args["gpu_block_size_2d"] = (64, 6) | ||
| optimization_args["gpu_memory_pool"] = False | ||
|
|
@@ -78,12 +92,17 @@ def get_gtfn_options( | |
| return backend_descriptor | ||
|
|
||
|
|
||
| def get_options(program_name: str, **backend_descriptor: Any) -> model_backends.BackendDescriptor: | ||
| def get_options( | ||
| program_name: str, | ||
| *, | ||
| backend_config: backend_cfg.BackendConfig | None, | ||
| **backend_descriptor: Any, | ||
| ) -> model_backends.BackendDescriptor: | ||
| if "backend_factory" not in backend_descriptor: | ||
| # here we could set a backend_factory per program | ||
| backend_descriptor["backend_factory"] = model_backends.make_custom_dace_backend | ||
| if backend_descriptor["backend_factory"] == model_backends.make_custom_dace_backend: | ||
| backend_descriptor = get_dace_options(program_name, **backend_descriptor) | ||
| backend_descriptor = get_dace_options(program_name, backend_config, **backend_descriptor) | ||
| if backend_descriptor["backend_factory"] == model_backends.make_custom_gtfn_backend: | ||
| backend_descriptor = get_gtfn_options(program_name, **backend_descriptor) | ||
|
|
||
|
|
@@ -96,7 +115,9 @@ def customize_backend( | |
| | model_backends.DeviceType | ||
| | model_backends.BackendDescriptor | ||
| | None, | ||
| backend_config: backend_cfg.BackendConfig | None = None, | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. Depending on how important it is to remember to set the BackendConfig, would it make sense to make this non-optional? Or alternatively make the default a BackendConfig() so you don't have to deal with the None case later?
Contributor
Author
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. For now, |
||
| ) -> gtx_typing.Backend | None: | ||
| backend_config = backend_config or backend_cfg.backend_config_from_env() | ||
| program_name = program.__name__ if program is not None else "" | ||
| if backend is None or isinstance(backend, gtx_backend.Backend): | ||
| backend_name = backend.name if backend is not None else "embedded" | ||
|
|
@@ -106,7 +127,9 @@ def customize_backend( | |
| backend_descriptor = ( | ||
| {"device": backend} if isinstance(backend, model_backends.DeviceType) else backend | ||
| ) | ||
| backend_descriptor = get_options(program_name, **backend_descriptor) | ||
| backend_descriptor = get_options( | ||
| program_name, backend_config=backend_config, **backend_descriptor | ||
| ) | ||
| backend_descriptor["device"] = backend_descriptor.get( | ||
| "device", model_backends.CPU | ||
| ) # set default device | ||
|
|
@@ -132,10 +155,11 @@ def setup_program( | |
| horizontal_sizes: dict[str, gtx.int32] | None = None, | ||
| vertical_sizes: dict[str, gtx.int32] | None = None, | ||
| offset_provider: gtx_typing.OffsetProvider | None = None, | ||
| backend_config: backend_cfg.BackendConfig | None = None, | ||
| ) -> Callable[..., None]: | ||
| """ | ||
| This function processes arguments to the GT4Py program. It | ||
| - binds arguments that don't change during model run ('constant_args', 'horizontal_sizes', "vertical_sizes'); | ||
| - binds arguments that don't change during model run ('constant_args', 'horizontal_sizes', 'vertical_sizes'); | ||
| - inlines scalar arguments into the GT4Py program at compile-time (via GT4Py's 'compile'). | ||
| Args: | ||
| - backend: GT4Py backend, | ||
|
|
@@ -145,14 +169,16 @@ def setup_program( | |
| - horizontal_sizes: horizontal domain bounds, | ||
| - vertical_sizes: vertical domain bounds, | ||
| - offset_provider: GT4Py offset_provider, | ||
| - backend_config: external DaCe workspace sizing, or `None` to fall back | ||
| to the 'ICON4PY_BACKEND_WORKSPACE_SIZE' environment variable. | ||
| """ | ||
| constant_args = {} if constant_args is None else constant_args | ||
| variants = {} if variants is None else variants | ||
| horizontal_sizes = {} if horizontal_sizes is None else horizontal_sizes | ||
| vertical_sizes = {} if vertical_sizes is None else vertical_sizes | ||
| offset_provider = {} if offset_provider is None else offset_provider | ||
|
|
||
| backend = customize_backend(program, backend) | ||
| backend = customize_backend(program, backend, backend_config=backend_config) | ||
|
|
||
| bound_static_args = {k: v for k, v in constant_args.items() if gtx.is_scalar_type(v)} | ||
| static_args_program = program.with_backend(backend) | ||
|
|
||
Uh oh!
There was an error while loading. Please reload this page.