From 5880c601ca6c937dce7c94e550deb936e177328f Mon Sep 17 00:00:00 2001 From: Jacopo Date: Wed, 1 Jul 2026 19:12:15 +0200 Subject: [PATCH 1/3] Add tolerance recording and drift detection to the testing plugin Introduce a mechanism to measure numeric test tolerances instead of asserting them, so they can be tightened automatically. - config: ICON4PY_RECORD_TOLERANCES (record measured differences to a file) and ICON4PY_TOLERANCE_DRIFT_WARN (non-failing warning when a tolerance is much looser than the measured difference). - tolerances: ToleranceRecorder plus helpers to aggregate measurements (max over deterministic CPU backends) and propose tightened tolerances. - test_utils: dallclose/assert_dallclose record or drift-warn via shared helpers and take a 'key' label identifying the compared field. - pytest_hooks: activate the recorder, attach per-test context (test id, backend, experiment), dump measurements as JSON lines, and print a drift summary. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../src/icon4py/model/testing/config.py | 10 + .../src/icon4py/model/testing/pytest_hooks.py | 43 +++- .../src/icon4py/model/testing/test_utils.py | 66 +++++- .../src/icon4py/model/testing/tolerances.py | 211 ++++++++++++++++++ 4 files changed, 318 insertions(+), 12 deletions(-) create mode 100644 model/testing/src/icon4py/model/testing/tolerances.py diff --git a/model/testing/src/icon4py/model/testing/config.py b/model/testing/src/icon4py/model/testing/config.py index 481453d5e4..36a3c3f264 100644 --- a/model/testing/src/icon4py/model/testing/config.py +++ b/model/testing/src/icon4py/model/testing/config.py @@ -33,5 +33,15 @@ def _default_download_cache() -> pathlib.Path: DALLCLOSE_PRINT_INSTEAD_OF_FAIL: bool = env.flag_to_bool( "ICON4PY_DALLCLOSE_PRINT_INSTEAD_OF_FAIL", False ) +# When set to a file path, 'assert_dallclose' records the measured max absolute/relative +# differences (instead of asserting) so that tolerances can be measured and tightened automatically. +RECORD_TOLERANCES_PATH: pathlib.Path | None = ( + env.path("ICON4PY_RECORD_TOLERANCES", pathlib.Path()) + if "ICON4PY_RECORD_TOLERANCES" in os.environ + else None +) +# When enabled, 'assert_dallclose' emits a non-failing warning whenever a passed tolerance is much +# larger than the measured difference, flagging tolerances that have become too loose. +TOLERANCE_DRIFT_WARN: bool = env.flag_to_bool("ICON4PY_TOLERANCE_DRIFT_WARN", False) DOWNLOAD_CACHE_PATH: pathlib.Path = env.path("ICON4PY_DOWNLOAD_CACHE", _default_download_cache()) DRIVER_LOGGING_LEVEL: str = os.environ.get("ICON4PY_DRIVER_LOGGING_LEVEL", "debug") diff --git a/model/testing/src/icon4py/model/testing/pytest_hooks.py b/model/testing/src/icon4py/model/testing/pytest_hooks.py index 8b1244c406..61460d7e0e 100644 --- a/model/testing/src/icon4py/model/testing/pytest_hooks.py +++ b/model/testing/src/icon4py/model/testing/pytest_hooks.py @@ -13,7 +13,7 @@ import pytest from icon4py.model.common import model_backends -from icon4py.model.testing import filters +from icon4py.model.testing import config as testing_config, filters, tolerances __all__ = [ @@ -51,6 +51,10 @@ def pytest_configure(config): if config.getoption("--datatest-skip"): config.option.markexpr = " and ".join(["not datatest", *m_option]) + # Activate tolerance recording/drift detection when the corresponding options are enabled. + if testing_config.RECORD_TOLERANCES_PATH is not None or testing_config.TOLERANCE_DRIFT_WARN: + tolerances.activate_recorder() + handle_mpi_options(config) @@ -153,10 +157,27 @@ def pytest_collection_modifyitems(config, items): ) +def _record_test_context(item: pytest.Item) -> None: + """Provide the current test id, backend and experiment to the active tolerance recorder.""" + recorder = tolerances.get_active_recorder() + if recorder is None: + return + params = getattr(item, "callspec", None) + params = params.params if params is not None else {} + experiment = params.get("experiment_description", params.get("experiment", "")) + recorder.set_context( + nodeid=item.nodeid, + backend=item.config.getoption("--backend"), + experiment=getattr(experiment, "name", str(experiment)), + ) + + @pytest.hookimpl(trylast=True) def pytest_runtest_setup(item: pytest.Item) -> None: """Apply test item filters as the final test setup step.""" + _record_test_context(item) + item_marker_filters = filters.item_marker_filters for marker_name in set(m.name for m in item.iter_markers()) & item_marker_filters.keys(): item_filter = item_marker_filters[marker_name] @@ -230,10 +251,26 @@ def pytest_runtest_makereport(item, call): report.sections.append(("benchmark-extra", tuple([filtered_benchmark_name, info]))) +def _report_tolerance_drift(terminalreporter) -> None: + """Print a non-failing summary of tolerances that are much looser than the measured difference.""" + recorder = tolerances.get_active_recorder() + if recorder is None or not recorder.drift_warnings: + return + terminalreporter.ensure_newline() + terminalreporter.section("Tolerance drift (tolerances too loose)", sep="-", yellow=True) + for warning in recorder.drift_warnings: + terminalreporter.line( + f"{warning.nodeid} [{warning.field}]: atol={warning.atol:g} " + f"but measured max diff {warning.max_abs:g}" + ) + + def pytest_terminal_summary(terminalreporter, exitstatus, config): """ Add a custom section to the terminal summary with GT4Py timer metrics from benchmarks. """ + _report_tolerance_drift(terminalreporter) + # Gather gtx_metrics benchmark_gtx_metrics = [] for outcome in ("passed", "failed", "skipped"): @@ -392,3 +429,7 @@ def pytest_sessionfinish(session: pytest.Session, exitstatus: int) -> None: scheduler = getattr(session.config, "_mpi_scheduler", None) if scheduler is not None: scheduler.finalize() + + recorder = tolerances.get_active_recorder() + if recorder is not None and testing_config.RECORD_TOLERANCES_PATH is not None: + recorder.dump(testing_config.RECORD_TOLERANCES_PATH) diff --git a/model/testing/src/icon4py/model/testing/test_utils.py b/model/testing/src/icon4py/model/testing/test_utils.py index dd8e17fac0..5e680c1845 100644 --- a/model/testing/src/icon4py/model/testing/test_utils.py +++ b/model/testing/src/icon4py/model/testing/test_utils.py @@ -17,7 +17,35 @@ from typing_extensions import Buffer from icon4py.model.common import model_options -from icon4py.model.testing import config +from icon4py.model.testing import config, tolerances + + +def _record_measurement_if_active( + actual: npt.ArrayLike, desired: npt.ArrayLike, *, atol: float, rtol: float, label: str +) -> bool: + """ + In recording mode, store the measured differences and return True so the caller skips comparing. + + Returns False (and does nothing) when recording is not active. + """ + recorder = tolerances.get_active_recorder() + if config.RECORD_TOLERANCES_PATH is None or recorder is None: + return False + max_abs, max_rel = tolerances.max_differences(actual, desired) + recorder.record_measurement(field=label, atol=atol, rtol=rtol, max_abs=max_abs, max_rel=max_rel) + return True + + +def _warn_if_tolerance_drifted( + actual: npt.ArrayLike, desired: npt.ArrayLike, *, atol: float, label: str +) -> None: + """Flag a tolerance that is much larger than the measured difference (a non-failing warning).""" + recorder = tolerances.get_active_recorder() + if not config.TOLERANCE_DRIFT_WARN or recorder is None or atol <= 0.0: + return + max_abs, _ = tolerances.max_differences(actual, desired) + if 0.0 < max_abs * tolerances.DRIFT_FACTOR < atol: + recorder.record_drift(field=label, atol=atol, max_abs=max_abs) def dallclose( @@ -27,10 +55,16 @@ def dallclose( rtol: float = 1.0e-12, atol: float = 0.0, equal_nan: bool = False, + key: str = "", ) -> bool: """ 'numpy.allclose', but with double precision default tolerances. + + 'key' is a stable label for the compared field, used when tolerance recording is enabled. """ + if _record_measurement_if_active(a, b, atol=atol, rtol=rtol, label=key): + return True + _warn_if_tolerance_drifted(a, b, atol=atol, label=key) return np.allclose(a, b, rtol=rtol, atol=atol, equal_nan=equal_nan) @@ -43,26 +77,36 @@ def assert_dallclose( equal_nan: bool = False, err_msg: str = "", verbose: bool = True, + key: str = "", ) -> None: """ 'numpy.testing.assert_allclose', but with double precision default tolerances. + + 'key' is a stable label for the compared field (e.g. 'vn', 'Cell'). It is used to identify the + measurement when tolerance recording is enabled; it falls back to 'err_msg' when not given. """ + label = key or err_msg + if _record_measurement_if_active(actual, desired, atol=atol, rtol=rtol, label=label): + return + if config.DALLCLOSE_PRINT_INSTEAD_OF_FAIL: # Non-blocking version: prints max diff instead of raising errors. # Prints red if delta > 0, green otherwise. max_diff = np.max(np.abs(np.asarray(actual) - np.asarray(desired))) color = "\033[1;31m" if max_diff > 0 else "\033[32m" print(f"{color}{err_msg} max diff {max_diff}\033[0m") - else: - np_testing.assert_allclose( - actual, # type: ignore[arg-type] - desired, # type: ignore[arg-type] - rtol=rtol, - atol=atol, - equal_nan=equal_nan, - err_msg=err_msg, - verbose=verbose, - ) + return + + _warn_if_tolerance_drifted(actual, desired, atol=atol, label=label) + np_testing.assert_allclose( + actual, # type: ignore[arg-type] + desired, # type: ignore[arg-type] + rtol=rtol, + atol=atol, + equal_nan=equal_nan, + err_msg=err_msg, + verbose=verbose, + ) def is_sorted(array: np.ndarray) -> bool: diff --git a/model/testing/src/icon4py/model/testing/tolerances.py b/model/testing/src/icon4py/model/testing/tolerances.py new file mode 100644 index 0000000000..85196a9022 --- /dev/null +++ b/model/testing/src/icon4py/model/testing/tolerances.py @@ -0,0 +1,211 @@ +# ICON4Py - ICON inspired code in Python and GT4Py +# +# Copyright (c) 2022-2024, ETH Zurich and MeteoSwiss +# All rights reserved. +# +# Please, refer to the LICENSE file in the root directory. +# SPDX-License-Identifier: BSD-3-Clause + +""" +Machinery for measuring and tightening numeric test tolerances. + +'assert_dallclose' (in 'test_utils') can, instead of asserting, record the measured maximum +absolute and relative differences between actual and reference fields. The 'ToleranceRecorder' +collects these measurements together with the current test context (test id, backend, experiment) +so that a downstream script can propose tighter tolerances. It can also collect non-failing +warnings about tolerances that have become much larger than the measured difference. +""" + +from __future__ import annotations + +import collections.abc +import dataclasses +import json +import math +import pathlib + +import gt4py.next as gtx +import numpy as np +import numpy.typing as npt + +from icon4py.model.common import dimension as dims + + +__all__ = [ + "DETERMINISTIC_CPU_BACKENDS", + "DRIFT_FACTOR", + "SAFETY_FACTOR", + "DriftWarning", + "Measurement", + "ToleranceRecorder", + "activate_recorder", + "aggregate_max_abs", + "deactivate_recorder", + "get_active_recorder", + "load_dimension_keyed_tolerances", + "load_measurements", + "max_differences", + "propose_tolerance", +] + +# A stored tolerance is flagged as drifted (too loose) when it exceeds the measured difference by +# more than this factor. +DRIFT_FACTOR = 100.0 +# Proposed tolerances are the measured difference scaled by this factor to leave headroom. +SAFETY_FACTOR = 4.0 +# Backends whose results are deterministic enough that measured differences may be used to tighten +# tolerances automatically. GPU and dace backends are excluded (documented as non-deterministic). +DETERMINISTIC_CPU_BACKENDS = frozenset({"gtfn_cpu", "embedded"}) + + +def max_differences(actual: npt.ArrayLike, desired: npt.ArrayLike) -> tuple[float, float]: + """Return the maximum absolute and relative difference between 'actual' and 'desired'.""" + actual_array = np.asarray(actual, dtype=float) + desired_array = np.asarray(desired, dtype=float) + absolute = np.abs(actual_array - desired_array) + max_absolute = float(np.max(absolute)) if absolute.size else 0.0 + with np.errstate(divide="ignore", invalid="ignore"): + relative = absolute / np.abs(desired_array) + finite = relative[np.isfinite(relative)] + max_relative = float(np.max(finite)) if finite.size else 0.0 + return max_absolute, max_relative + + +@dataclasses.dataclass(frozen=True) +class Measurement: + nodeid: str + backend: str + experiment: str + field: str + max_abs: float + max_rel: float + atol: float + rtol: float + + +@dataclasses.dataclass(frozen=True) +class DriftWarning: + nodeid: str + field: str + atol: float + max_abs: float + + +class ToleranceRecorder: + """Collects tolerance measurements and drift warnings for the current test session.""" + + def __init__(self) -> None: + self.measurements: list[Measurement] = [] + self.drift_warnings: list[DriftWarning] = [] + self._nodeid = "" + self._backend = "" + self._experiment = "" + + def set_context(self, *, nodeid: str, backend: str, experiment: str) -> None: + self._nodeid = nodeid + self._backend = backend + self._experiment = experiment + + def record_measurement( + self, *, field: str, atol: float, rtol: float, max_abs: float, max_rel: float + ) -> None: + self.measurements.append( + Measurement( + nodeid=self._nodeid, + backend=self._backend, + experiment=self._experiment, + field=field, + max_abs=max_abs, + max_rel=max_rel, + atol=atol, + rtol=rtol, + ) + ) + + def record_drift(self, *, field: str, atol: float, max_abs: float) -> None: + self.drift_warnings.append( + DriftWarning(nodeid=self._nodeid, field=field, atol=atol, max_abs=max_abs) + ) + + def dump(self, path: pathlib.Path) -> None: + """Write the collected measurements to 'path' as JSON lines.""" + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8") as stream: + for measurement in self.measurements: + stream.write(json.dumps(dataclasses.asdict(measurement)) + "\n") + + +class _RecorderState: + """Holds the process-wide active recorder (avoids a module-level 'global').""" + + active: ToleranceRecorder | None = None + + +def get_active_recorder() -> ToleranceRecorder | None: + return _RecorderState.active + + +def activate_recorder() -> ToleranceRecorder: + _RecorderState.active = ToleranceRecorder() + return _RecorderState.active + + +def deactivate_recorder() -> None: + _RecorderState.active = None + + +def load_dimension_keyed_tolerances(path: pathlib.Path) -> dict[gtx.Dimension, dict[str, float]]: + """ + Load a tolerance table keyed by horizontal dimension and experiment name. + + The JSON file maps a dimension name ('Cell', 'Edge', 'Vertex') to a mapping of experiment name + to absolute tolerance. The returned dictionary is keyed by the corresponding 'Dimension' object + so it can be indexed as 'table[dims.CellDim][experiment_name]'. + """ + data = json.loads(path.read_text(encoding="utf-8")) + return {getattr(dims, f"{name}Dim"): tolerances for name, tolerances in data.items()} + + +def load_measurements(paths: collections.abc.Iterable[pathlib.Path]) -> list[Measurement]: + """Load recorded measurements from one or more JSON-lines files.""" + return [ + Measurement(**json.loads(line)) + for path in paths + for line in path.read_text(encoding="utf-8").splitlines() + if line.strip() + ] + + +def aggregate_max_abs( + measurements: collections.abc.Iterable[Measurement], + *, + backends: collections.abc.Container[str] = DETERMINISTIC_CPU_BACKENDS, +) -> dict[tuple[str, str], float]: + """ + Reduce measurements to the maximum absolute difference per '(field, experiment)'. + + Only measurements from 'backends' are considered, so tolerances are derived exclusively from + deterministic backends by default. + """ + aggregated: dict[tuple[str, str], float] = {} + for measurement in measurements: + if measurement.backend not in backends: + continue + key = (measurement.field, measurement.experiment) + aggregated[key] = max(aggregated.get(key, 0.0), measurement.max_abs) + return aggregated + + +def propose_tolerance(max_abs: float) -> float: + """ + Propose a tolerance from a measured maximum difference. + + The measured difference is scaled by 'SAFETY_FACTOR' and rounded up to one significant figure. + A measured difference of zero yields an exact (zero) tolerance. + """ + scaled = max_abs * SAFETY_FACTOR + if scaled <= 0.0: + return 0.0 + exponent = math.floor(math.log10(scaled)) + fraction = scaled / 10.0**exponent + return math.ceil(fraction) * 10.0**exponent From dbe4fe9e9c1a939c39ed2dd19098a01c4d46b6a2 Mon Sep 17 00:00:00 2001 From: Jacopo Date: Wed, 1 Jul 2026 19:12:26 +0200 Subject: [PATCH 2/3] Move RBF interpolation tolerances into a JSON store Replace the inline RBF_TOLERANCES dict with a JSON file loaded at import time, and label the comparisons with the dimension name so they can be recorded. This is the pilot for the automatic tolerance-tightening tooling; the loaded table keeps the same shape so consumers are unchanged. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../unit_tests/rbf_tolerances.json | 17 +++++++++++ .../unit_tests/test_rbf_interpolation.py | 30 ++++++++----------- 2 files changed, 30 insertions(+), 17 deletions(-) create mode 100644 model/common/tests/common/interpolation/unit_tests/rbf_tolerances.json diff --git a/model/common/tests/common/interpolation/unit_tests/rbf_tolerances.json b/model/common/tests/common/interpolation/unit_tests/rbf_tolerances.json new file mode 100644 index 0000000000..e24ded747f --- /dev/null +++ b/model/common/tests/common/interpolation/unit_tests/rbf_tolerances.json @@ -0,0 +1,17 @@ +{ + "Cell": { + "exclaim_ape_R02B04": 3.1e-09, + "exclaim_ch_r04b09_dsl": 0.04, + "exclaim_gauss3d": 1e-14 + }, + "Edge": { + "exclaim_ape_R02B04": 8e-14, + "exclaim_ch_r04b09_dsl": 2e-09, + "exclaim_gauss3d": 0 + }, + "Vertex": { + "exclaim_ape_R02B04": 3e-10, + "exclaim_ch_r04b09_dsl": 0.003, + "exclaim_gauss3d": 1e-15 + } +} diff --git a/model/common/tests/common/interpolation/unit_tests/test_rbf_interpolation.py b/model/common/tests/common/interpolation/unit_tests/test_rbf_interpolation.py index 237e434812..96d1915527 100644 --- a/model/common/tests/common/interpolation/unit_tests/test_rbf_interpolation.py +++ b/model/common/tests/common/interpolation/unit_tests/test_rbf_interpolation.py @@ -8,6 +8,7 @@ from __future__ import annotations import math +import pathlib from typing import TYPE_CHECKING import numpy as np @@ -22,6 +23,7 @@ definitions, grid_utils as gridtest_utils, test_utils as test_helpers, + tolerances, ) from icon4py.model.testing.fixtures.datatest import ( backend, @@ -41,23 +43,12 @@ from icon4py.model.testing import serialbox -RBF_TOLERANCES = { - dims.CellDim: { - definitions.Experiments.EXCLAIM_APE.name: 3.1e-9, - definitions.Experiments.MCH_CH_R04B09.name: 4e-2, - definitions.Experiments.GAUSS3D.name: 1e-14, - }, - dims.EdgeDim: { - definitions.Experiments.EXCLAIM_APE.name: 8e-14, - definitions.Experiments.MCH_CH_R04B09.name: 2e-9, - definitions.Experiments.GAUSS3D.name: 0, - }, - dims.VertexDim: { - definitions.Experiments.EXCLAIM_APE.name: 3e-10, - definitions.Experiments.MCH_CH_R04B09.name: 3e-3, - definitions.Experiments.GAUSS3D.name: 1e-15, - }, -} +# Absolute tolerances for the RBF interpolation coefficients, keyed by dimension and experiment +# name. Maintained via the tolerance-tightening tooling (see 'scripts/python/update_tolerances.py'); +# edit 'rbf_tolerances.json' rather than hardcoding values here. +RBF_TOLERANCES = tolerances.load_dimension_keyed_tolerances( + pathlib.Path(__file__).parent / "rbf_tolerances.json" +) @pytest.mark.level("unit") @@ -223,11 +214,13 @@ def test_rbf_interpolation_coeffs_cell( rbf_vec_coeff_c1[horizontal_start:], rbf_vec_coeff_c1_ref[horizontal_start:], atol=RBF_TOLERANCES[dims.CellDim][experiment.name], + key=dims.CellDim.value, ) assert test_helpers.dallclose( rbf_vec_coeff_c2[horizontal_start:], rbf_vec_coeff_c2_ref[horizontal_start:], atol=RBF_TOLERANCES[dims.CellDim][experiment.name], + key=dims.CellDim.value, ) @@ -298,11 +291,13 @@ def test_rbf_interpolation_coeffs_vertex( rbf_vec_coeff_v1[horizontal_start:], rbf_vec_coeff_v1_ref.asnumpy()[horizontal_start:], atol=RBF_TOLERANCES[dims.VertexDim][experiment.name], + key=dims.VertexDim.value, ) assert test_helpers.dallclose( rbf_vec_coeff_v2[horizontal_start:], rbf_vec_coeff_v2_ref.asnumpy()[horizontal_start:], atol=RBF_TOLERANCES[dims.VertexDim][experiment.name], + key=dims.VertexDim.value, ) @@ -369,4 +364,5 @@ def test_rbf_interpolation_coeffs_edge( rbf_vec_coeff_e[horizontal_start:], rbf_vec_coeff_e_ref.asnumpy()[horizontal_start:], atol=RBF_TOLERANCES[dims.EdgeDim][experiment.name], + key=dims.EdgeDim.value, ) From 2f32a3b2bebb0f55d6207eb52e85759ab2e12fbc Mon Sep 17 00:00:00 2001 From: Jacopo Date: Wed, 1 Jul 2026 19:12:26 +0200 Subject: [PATCH 3/3] Add tolerance-tightening script, nox session and CSCS pipeline job - update_tolerances.py: tighten a JSON tolerance store from recorded measurements (tighten-only, deterministic CPU backends, max over backends scaled by a safety factor). - noxfile: 'update_tolerances' session that records tolerances on the deterministic CPU backends and runs the updater. - ci: a job in the weekly 'all' pipeline that runs the session and publishes the proposed tolerance tightenings as an artifact ('tolerance-updates.patch'); it never commits, so a human reviews and applies the change. Co-Authored-By: Claude Opus 4.8 (1M context) --- ci/all.yml | 4 ++ ci/base.yml | 24 +++++++++ noxfile.py | 44 +++++++++++++++++ scripts/python/update_tolerances.py | 77 +++++++++++++++++++++++++++++ 4 files changed, 149 insertions(+) create mode 100755 scripts/python/update_tolerances.py diff --git a/ci/all.yml b/ci/all.yml index f0a3784a9e..2f4f1533e9 100644 --- a/ci/all.yml +++ b/ci/all.yml @@ -29,3 +29,7 @@ generate_test_child_pipeline: trigger_test_child_pipeline: extends: [.trigger_test_child_pipeline] + +update_tolerances_aarch64: + extends: [.update_tolerances_aarch64] + needs: [build_image_aarch64] diff --git a/ci/base.yml b/ci/base.yml index 4643f49f56..e71f21f830 100644 --- a/ci/base.yml +++ b/ci/base.yml @@ -190,6 +190,30 @@ variables: paths: - pytest-log-rank-*.txt +# Measures test tolerances on the deterministic CPU backends and reports which tolerances could be +# tightened, as a pipeline artifact ('tolerance-updates.patch'). This job never commits or pushes: +# a human reviews the patch and applies it. It is instantiated only in the weekly 'all' pipeline +# (see ci/all.yml). See scripts/python/update_tolerances.py. +.update_tolerances_aarch64: + extends: [.test_runner_serial, .test_template_aarch64] + variables: + SLURM_TIMELIMIT: '00:40:00' + script: + - nox -s "update_tolerances-${PYVERSION_SHORT}" + - git diff -- '*tolerances.json' > tolerance-updates.patch + - | + if [ -s tolerance-updates.patch ]; then + echo "Proposed tolerance tightenings (see the tolerance-updates.patch artifact):" + cat tolerance-updates.patch + else + echo "No tolerances to tighten." + fi + artifacts: + when: always + paths: + - tolerance-updates.patch + - tolerance-measurements-*.jsonl + .generate_test_child_pipeline: stage: generate_test extends: .container-runner-lightweight-gh200 diff --git a/noxfile.py b/noxfile.py index 91aa361166..39c7a3e377 100644 --- a/noxfile.py +++ b/noxfile.py @@ -9,6 +9,7 @@ from __future__ import annotations import os +import pathlib import re from collections.abc import Sequence from datetime import datetime @@ -234,6 +235,49 @@ def test_tools_and_bindings(session: nox.Session, selection: ToolsBindingsTestsS session.run(*pytest_base.split(), *session.posargs) +# Deterministic CPU backends and pilot tests used to measure and tighten tolerances. GPU/dace +# backends are excluded because their results are not bitwise reproducible. +TOLERANCE_RECORD_BACKENDS: Final[Sequence[str]] = ("gtfn_cpu", "embedded") +TOLERANCE_PILOT_TEST: Final[str] = ( + "model/common/tests/common/interpolation/unit_tests/test_rbf_interpolation.py" +) +TOLERANCE_PILOT_STORE: Final[str] = ( + "model/common/tests/common/interpolation/unit_tests/rbf_tolerances.json" +) + + +@nox.session(python=SUPPORTED_PYTHON_VERSIONS) +def update_tolerances(session: nox.Session) -> None: + """Measure test tolerances on deterministic CPU backends and tighten the tolerance store.""" + _install_session_venv(session, extras=["fortran", "io", "testing"], groups=["test"]) + + measurement_files = [] + for backend in TOLERANCE_RECORD_BACKENDS: + measurement_file = str(pathlib.Path(f"tolerance-measurements-{backend}.jsonl").resolve()) + measurement_files.append(measurement_file) + # '-n0' is required: xdist workers would otherwise clobber the shared measurements file. + session.run( + "pytest", + "-q", + "--benchmark-disable", + "-n0", + "--datatest-only", + f"--backend={backend}", + TOLERANCE_PILOT_TEST, + env={"ICON4PY_RECORD_TOLERANCES": measurement_file}, + success_codes=[0, NO_TESTS_COLLECTED_EXIT_CODE], + ) + + session.run( + "python", + "scripts/python/update_tolerances.py", + "--store", + TOLERANCE_PILOT_STORE, + *measurement_files, + *session.posargs, + ) + + # -- utils -- def _install_session_venv( session: nox.Session, diff --git a/scripts/python/update_tolerances.py b/scripts/python/update_tolerances.py new file mode 100755 index 0000000000..ba8ec43cfe --- /dev/null +++ b/scripts/python/update_tolerances.py @@ -0,0 +1,77 @@ +#!/usr/bin/env -S uv run -q --frozen --group test python3 +# +# ICON4Py - ICON inspired code in Python and GT4Py +# +# Copyright (c) 2022-2024, ETH Zurich and MeteoSwiss +# All rights reserved. +# +# Please, refer to the LICENSE file in the root directory. +# SPDX-License-Identifier: BSD-3-Clause + +"""Tighten a JSON tolerance store from recorded 'assert_dallclose' measurements. + +Run the tests with 'ICON4PY_RECORD_TOLERANCES=' to record the measured +differences, then pass the store and the measurement file(s) to this script. Tolerances are only +ever tightened (made smaller): a value is updated when the proposed tolerance -- derived from the +measured difference on deterministic CPU backends -- is smaller than the current one. Tolerances +that are already tight, or that would need to be loosened, are left untouched. + +The store is a JSON file mapping 'field -> experiment -> absolute tolerance'. +""" + +from __future__ import annotations + +import argparse +import json +import pathlib + +from icon4py.model.testing import tolerances + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument( + "--store", type=pathlib.Path, required=True, help="JSON tolerance store to update." + ) + parser.add_argument( + "measurements", + type=pathlib.Path, + nargs="+", + help="Recorded measurement JSON-lines file(s).", + ) + parser.add_argument( + "--dry-run", action="store_true", help="Report proposed changes without writing the store." + ) + args = parser.parse_args() + + store = json.loads(args.store.read_text(encoding="utf-8")) + aggregated = tolerances.aggregate_max_abs(tolerances.load_measurements(args.measurements)) + + changes = [] + for (field, experiment), max_abs in sorted(aggregated.items()): + # Only manage tolerances that already exist in the store. + if field not in store or experiment not in store[field]: + continue + current = store[field][experiment] + proposed = tolerances.propose_tolerance(max_abs) + if proposed < current: + changes.append((field, experiment, current, proposed, max_abs)) + store[field][experiment] = proposed + + if not changes: + print("No tolerances to tighten.") + return + + for field, experiment, current, proposed, max_abs in changes: + print(f"{field}/{experiment}: {current:g} -> {proposed:g} (measured max diff {max_abs:g})") + + if args.dry_run: + print("Dry run: store not modified.") + return + + args.store.write_text(json.dumps(store, indent=2, sort_keys=True) + "\n", encoding="utf-8") + print(f"Updated {args.store}.") + + +if __name__ == "__main__": + main()