Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions ci/all.yml
Original file line number Diff line number Diff line change
Expand Up @@ -29,3 +29,7 @@ generate_test_child_pipeline:

trigger_test_child_pipeline:
extends: [.trigger_test_child_pipeline]

update_tolerances_aarch64:
extends: [.update_tolerances_aarch64]
needs: [build_image_aarch64]
24 changes: 24 additions & 0 deletions ci/base.yml
Original file line number Diff line number Diff line change
Expand Up @@ -203,6 +203,30 @@ variables:
paths:
- pytest-log-rank-*.txt

# Measures test tolerances on the deterministic CPU backends and reports which tolerances could be
# tightened, as a pipeline artifact ('tolerance-updates.patch'). This job never commits or pushes:
# a human reviews the patch and applies it. It is instantiated only in the weekly 'all' pipeline
# (see ci/all.yml). See scripts/python/update_tolerances.py.
.update_tolerances_aarch64:
extends: [.test_runner_serial, .test_template_aarch64]
variables:
SLURM_TIMELIMIT: '00:40:00'
script:
- nox -s "update_tolerances-${PYVERSION_SHORT}"
- git diff -- '*tolerances.json' > tolerance-updates.patch
- |
if [ -s tolerance-updates.patch ]; then
echo "Proposed tolerance tightenings (see the tolerance-updates.patch artifact):"
cat tolerance-updates.patch
else
echo "No tolerances to tighten."
fi
artifacts:
when: always
paths:
- tolerance-updates.patch
- tolerance-measurements-*.jsonl

.generate_test_child_pipeline:
stage: generate_test
extends: .container-runner-lightweight-gh200
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,17 @@
{
"Cell": {
"exclaim_ape_R02B04": 3.1e-09,
"exclaim_ch_r04b09_dsl": 0.04,
"exclaim_gauss3d": 1e-14
},
"Edge": {
"exclaim_ape_R02B04": 8e-14,
"exclaim_ch_r04b09_dsl": 2e-09,
"exclaim_gauss3d": 0
},
"Vertex": {
"exclaim_ape_R02B04": 3e-10,
"exclaim_ch_r04b09_dsl": 0.003,
"exclaim_gauss3d": 1e-15
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
from __future__ import annotations

import math
import pathlib
from typing import TYPE_CHECKING

import numpy as np
Expand All @@ -22,6 +23,7 @@
definitions,
grid_utils as gridtest_utils,
test_utils as test_helpers,
tolerances,
)
from icon4py.model.testing.fixtures.datatest import (
backend,
Expand All @@ -41,23 +43,12 @@

from icon4py.model.testing import serialbox

RBF_TOLERANCES = {
dims.CellDim: {
definitions.Experiments.EXCLAIM_APE.name: 3.1e-9,
definitions.Experiments.MCH_CH_R04B09.name: 4e-2,
definitions.Experiments.GAUSS3D.name: 1e-14,
},
dims.EdgeDim: {
definitions.Experiments.EXCLAIM_APE.name: 8e-14,
definitions.Experiments.MCH_CH_R04B09.name: 2e-9,
definitions.Experiments.GAUSS3D.name: 0,
},
dims.VertexDim: {
definitions.Experiments.EXCLAIM_APE.name: 3e-10,
definitions.Experiments.MCH_CH_R04B09.name: 3e-3,
definitions.Experiments.GAUSS3D.name: 1e-15,
},
}
# Absolute tolerances for the RBF interpolation coefficients, keyed by dimension and experiment
# name. Maintained via the tolerance-tightening tooling (see 'scripts/python/update_tolerances.py');
# edit 'rbf_tolerances.json' rather than hardcoding values here.
Comment on lines +46 to +48

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Suggested change
# Absolute tolerances for the RBF interpolation coefficients, keyed by dimension and experiment
# name. Maintained via the tolerance-tightening tooling (see 'scripts/python/update_tolerances.py');
# edit 'rbf_tolerances.json' rather than hardcoding values here.
# Absolute tolerances for the RBF interpolation coefficients, maintained via the tolerance-tightening tooling.

RBF_TOLERANCES = tolerances.load_dimension_keyed_tolerances(
pathlib.Path(__file__).parent / "rbf_tolerances.json"
)


@pytest.mark.level("unit")
Expand Down Expand Up @@ -223,11 +214,13 @@ def test_rbf_interpolation_coeffs_cell(
rbf_vec_coeff_c1[horizontal_start:],
rbf_vec_coeff_c1_ref[horizontal_start:],
atol=RBF_TOLERANCES[dims.CellDim][experiment.name],
key=dims.CellDim.value,
)
assert test_helpers.dallclose(
rbf_vec_coeff_c2[horizontal_start:],
rbf_vec_coeff_c2_ref[horizontal_start:],
atol=RBF_TOLERANCES[dims.CellDim][experiment.name],
key=dims.CellDim.value,
)


Expand Down Expand Up @@ -298,11 +291,13 @@ def test_rbf_interpolation_coeffs_vertex(
rbf_vec_coeff_v1[horizontal_start:],
rbf_vec_coeff_v1_ref.asnumpy()[horizontal_start:],
atol=RBF_TOLERANCES[dims.VertexDim][experiment.name],
key=dims.VertexDim.value,
)
assert test_helpers.dallclose(
rbf_vec_coeff_v2[horizontal_start:],
rbf_vec_coeff_v2_ref.asnumpy()[horizontal_start:],
atol=RBF_TOLERANCES[dims.VertexDim][experiment.name],
key=dims.VertexDim.value,
)


Expand Down Expand Up @@ -369,4 +364,5 @@ def test_rbf_interpolation_coeffs_edge(
rbf_vec_coeff_e[horizontal_start:],
rbf_vec_coeff_e_ref.asnumpy()[horizontal_start:],
atol=RBF_TOLERANCES[dims.EdgeDim][experiment.name],
key=dims.EdgeDim.value,
)
10 changes: 10 additions & 0 deletions model/testing/src/icon4py/model/testing/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,5 +33,15 @@ def _default_download_cache() -> pathlib.Path:
DALLCLOSE_PRINT_INSTEAD_OF_FAIL: bool = env.flag_to_bool(
"ICON4PY_DALLCLOSE_PRINT_INSTEAD_OF_FAIL", False
)
# When set to a file path, 'assert_dallclose' records the measured max absolute/relative
# differences (instead of asserting) so that tolerances can be measured and tightened automatically.
Comment on lines +36 to +37

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Suggested change
# When set to a file path, 'assert_dallclose' records the measured max absolute/relative
# differences (instead of asserting) so that tolerances can be measured and tightened automatically.

You don't need this

RECORD_TOLERANCES_PATH: pathlib.Path | None = (
env.path("ICON4PY_RECORD_TOLERANCES", pathlib.Path())
if "ICON4PY_RECORD_TOLERANCES" in os.environ
else None
)
# When enabled, 'assert_dallclose' emits a non-failing warning whenever a passed tolerance is much
# larger than the measured difference, flagging tolerances that have become too loose.
Comment on lines +43 to +44

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Suggested change
# When enabled, 'assert_dallclose' emits a non-failing warning whenever a passed tolerance is much
# larger than the measured difference, flagging tolerances that have become too loose.

Neither this

TOLERANCE_DRIFT_WARN: bool = env.flag_to_bool("ICON4PY_TOLERANCE_DRIFT_WARN", False)
DOWNLOAD_CACHE_PATH: pathlib.Path = env.path("ICON4PY_DOWNLOAD_CACHE", _default_download_cache())
DRIVER_LOGGING_LEVEL: str = os.environ.get("ICON4PY_DRIVER_LOGGING_LEVEL", "debug")
43 changes: 42 additions & 1 deletion model/testing/src/icon4py/model/testing/pytest_hooks.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,7 @@
import pytest

from icon4py.model.common import model_backends
from icon4py.model.testing import filters
from icon4py.model.testing import config as testing_config, filters, tolerances


__all__ = [
Expand Down Expand Up @@ -51,6 +51,10 @@ def pytest_configure(config):
if config.getoption("--datatest-skip"):
config.option.markexpr = " and ".join(["not datatest", *m_option])

# Activate tolerance recording/drift detection when the corresponding options are enabled.
if testing_config.RECORD_TOLERANCES_PATH is not None or testing_config.TOLERANCE_DRIFT_WARN:

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Suggested change
if testing_config.RECORD_TOLERANCES_PATH is not None or testing_config.TOLERANCE_DRIFT_WARN:
if testing_config.RECORD_TOLERANCES_PATH or testing_config.TOLERANCE_DRIFT_WARN:

I think you can also just say this

tolerances.activate_recorder()

handle_mpi_options(config)


Expand Down Expand Up @@ -159,10 +163,27 @@ def pytest_collection_modifyitems(config, items):
)


def _record_test_context(item: pytest.Item) -> None:
"""Provide the current test id, backend and experiment to the active tolerance recorder."""
recorder = tolerances.get_active_recorder()
if recorder is None:
return
params = getattr(item, "callspec", None)
params = params.params if params is not None else {}
experiment = params.get("experiment_description", params.get("experiment", ""))
recorder.set_context(
nodeid=item.nodeid,
backend=item.config.getoption("--backend"),
experiment=getattr(experiment, "name", str(experiment)),
)


@pytest.hookimpl(trylast=True)
def pytest_runtest_setup(item: pytest.Item) -> None:
"""Apply test item filters as the final test setup step."""

_record_test_context(item)

item_marker_filters = filters.item_marker_filters
for marker_name in set(m.name for m in item.iter_markers()) & item_marker_filters.keys():
item_filter = item_marker_filters[marker_name]
Expand Down Expand Up @@ -236,10 +257,26 @@ def pytest_runtest_makereport(item, call):
report.sections.append(("benchmark-extra", tuple([filtered_benchmark_name, info])))


def _report_tolerance_drift(terminalreporter) -> None:
"""Print a non-failing summary of tolerances that are much looser than the measured difference."""
recorder = tolerances.get_active_recorder()
if recorder is None or not recorder.drift_warnings:
return
terminalreporter.ensure_newline()
terminalreporter.section("Tolerance drift (tolerances too loose)", sep="-", yellow=True)
for warning in recorder.drift_warnings:
terminalreporter.line(
f"{warning.nodeid} [{warning.field}]: atol={warning.atol:g} "
f"but measured max diff {warning.max_abs:g}"
)


def pytest_terminal_summary(terminalreporter, exitstatus, config):
"""
Add a custom section to the terminal summary with GT4Py timer metrics from benchmarks.
"""
_report_tolerance_drift(terminalreporter)

# Gather gtx_metrics
benchmark_gtx_metrics = []
for outcome in ("passed", "failed", "skipped"):
Expand Down Expand Up @@ -398,3 +435,7 @@ def pytest_sessionfinish(session: pytest.Session, exitstatus: int) -> None:
scheduler = getattr(session.config, "_mpi_scheduler", None)
if scheduler is not None:
scheduler.finalize()

recorder = tolerances.get_active_recorder()
if recorder is not None and testing_config.RECORD_TOLERANCES_PATH is not None:
recorder.dump(testing_config.RECORD_TOLERANCES_PATH)
66 changes: 55 additions & 11 deletions model/testing/src/icon4py/model/testing/test_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,35 @@
from typing_extensions import Buffer

from icon4py.model.common import model_backends, model_options
from icon4py.model.testing import config
from icon4py.model.testing import config, tolerances


def _record_measurement_if_active(
actual: npt.ArrayLike, desired: npt.ArrayLike, *, atol: float, rtol: float, label: str
) -> bool:
"""
In recording mode, store the measured differences and return True so the caller skips comparing.

Returns False (and does nothing) when recording is not active.
"""
recorder = tolerances.get_active_recorder()
if config.RECORD_TOLERANCES_PATH is None or recorder is None:
return False
max_abs, max_rel = tolerances.max_differences(actual, desired)
recorder.record_measurement(field=label, atol=atol, rtol=rtol, max_abs=max_abs, max_rel=max_rel)
return True


def _warn_if_tolerance_drifted(
actual: npt.ArrayLike, desired: npt.ArrayLike, *, atol: float, label: str
) -> None:
"""Flag a tolerance that is much larger than the measured difference (a non-failing warning)."""
recorder = tolerances.get_active_recorder()
if not config.TOLERANCE_DRIFT_WARN or recorder is None or atol <= 0.0:
return
max_abs, _ = tolerances.max_differences(actual, desired)
if 0.0 < max_abs * tolerances.DRIFT_FACTOR < atol:
recorder.record_drift(field=label, atol=atol, max_abs=max_abs)


def get_mpi_comparison_tolerance(
Expand Down Expand Up @@ -48,10 +76,16 @@ def dallclose(
rtol: float = 1.0e-12,
atol: float = 0.0,
equal_nan: bool = False,
key: str = "",
) -> bool:
"""
'numpy.allclose', but with double precision default tolerances.

'key' is a stable label for the compared field, used when tolerance recording is enabled.
"""
if _record_measurement_if_active(a, b, atol=atol, rtol=rtol, label=key):
return True
_warn_if_tolerance_drifted(a, b, atol=atol, label=key)
return np.allclose(a, b, rtol=rtol, atol=atol, equal_nan=equal_nan)


Expand All @@ -64,26 +98,36 @@ def assert_dallclose(
equal_nan: bool = False,
err_msg: str = "",
verbose: bool = True,
key: str = "",
) -> None:
"""
'numpy.testing.assert_allclose', but with double precision default tolerances.

'key' is a stable label for the compared field (e.g. 'vn', 'Cell'). It is used to identify the
measurement when tolerance recording is enabled; it falls back to 'err_msg' when not given.
"""
label = key or err_msg
if _record_measurement_if_active(actual, desired, atol=atol, rtol=rtol, label=label):
return

if config.DALLCLOSE_PRINT_INSTEAD_OF_FAIL:
# Non-blocking version: prints max diff instead of raising errors.
# Prints red if delta > 0, green otherwise.
max_diff = np.max(np.abs(np.asarray(actual) - np.asarray(desired)))
color = "\033[1;31m" if max_diff > 0 else "\033[32m"
print(f"{color}{err_msg} max diff {max_diff}\033[0m")
else:
np_testing.assert_allclose(
actual, # type: ignore[arg-type]
desired, # type: ignore[arg-type]
rtol=rtol,
atol=atol,
equal_nan=equal_nan,
err_msg=err_msg,
verbose=verbose,
)
return

_warn_if_tolerance_drifted(actual, desired, atol=atol, label=label)
np_testing.assert_allclose(
actual, # type: ignore[arg-type]
desired, # type: ignore[arg-type]
rtol=rtol,
atol=atol,
equal_nan=equal_nan,
err_msg=err_msg,
verbose=verbose,
)


def is_sorted(array: np.ndarray) -> bool:
Expand Down
Loading
Loading