From 57e200b6a56d32cf129f3180ca26a9b5c34d6974 Mon Sep 17 00:00:00 2001 From: lindicaphxag-tech Date: Sun, 4 Oct 2026 00:34:24 +0800 Subject: [PATCH 1/7] FIX preserve per-session CrossSession result provenance --- docs/source/whats_new.rst | 8 +++++ moabb/evaluations/evaluations.py | 36 ++++++++++--------- moabb/tests/test_evaluations.py | 61 ++++++++++++++++++++++++++++++++ 3 files changed, 89 insertions(+), 16 deletions(-) diff --git a/docs/source/whats_new.rst b/docs/source/whats_new.rst index 000414955..d6871d6c7 100644 --- a/docs/source/whats_new.rst +++ b/docs/source/whats_new.rst @@ -35,6 +35,12 @@ Requirements Bugs ~~~~ +- Keep :class:`moabb.evaluations.CrossSessionEvaluation` result provenance + session-specific when a custom cross-validator holds out more than one recording + session in the same fold. The estimator is still fitted once per fold, but each + held-out session is scored and stored separately, matching the evaluation's + session-level result contract in both the flattened and legacy execution paths + (by `lindicaphxag-tech`_). - Use ``gmean`` in TRCA and TRCSP for compatibility with pyRiemann 0.12 and 0.13, and pass the TRCSP mean metric by keyword (by `Bruno Aristimunha`_). - Fix the ``-e``/``--evaluations`` flag of ``python -m moabb.run``, which used ``type=list`` and so split its value into single characters: ``-e WithinSession`` reached :func:`moabb.benchmark` as ``['W', 'i', 't', ...]`` and raised ``KeyError: 'W'``. It now takes one or more evaluation names, space separated (by `Iain`_) - Fix the two install pages asking for optional extras MOABB does not have: the pip install page gave ``pip install moabb[deepleaning,carbonemission,docs]``, which is missing the ``r`` of ``deeplearning``, and pip only warns about an unrecognised extra, so following that page left ``braindecode`` uninstalled. The from-sources page asked for ``external``, removed in 1.2.0 (by `Iain`_). @@ -1063,3 +1069,5 @@ API changes .. _Iain: https://github.com/NotAFlightRisk .. _Anna Sokolova: https://github.com/ZyntZ .. _Arthur031221: https://github.com/Arthur031221 + +.. _lindicaphxag-tech: https://github.com/lindicaphxag-tech diff --git a/moabb/evaluations/evaluations.py b/moabb/evaluations/evaluations.py index e579264ab..272aa5465 100644 --- a/moabb/evaluations/evaluations.py +++ b/moabb/evaluations/evaluations.py @@ -302,6 +302,7 @@ class CrossSessionEvaluation(BaseEvaluation): """ _eval_type = "CrossSession" + _score_per_session = True def _create_splitter(self): """Create the CrossSessionSplitter for parallel evaluation.""" @@ -391,24 +392,27 @@ def evaluate( eval_type="CrossSession", ) - res = self._build_scored_result( - dataset, - subject, - groups[test][0], - name, - len(train), - nchan, - duration, - scorer, - cvclf, - X[test], - y[test], - ) + test_sessions = groups[test] + for session in np.unique(test_sessions): + session_test = test[test_sessions == session] + res = self._build_scored_result( + dataset, + subject, + session, + name, + len(train), + nchan, + duration, + scorer, + cvclf, + X[session_test], + y[session_test], + ) - if _carbonfootprint: - self._attach_emissions(res, emissions, task_name) + if _carbonfootprint: + self._attach_emissions(res, emissions, task_name) - yield res + yield res if _carbonfootprint: tracker.stop() diff --git a/moabb/tests/test_evaluations.py b/moabb/tests/test_evaluations.py index 4c6566c07..3c0fb58da 100644 --- a/moabb/tests/test_evaluations.py +++ b/moabb/tests/test_evaluations.py @@ -1565,6 +1565,67 @@ def test_cross_session_equivalence(self, tmp_path): """CrossSession parallel matches legacy scores.""" self._compare_parallel_vs_legacy(ev.CrossSessionEvaluation, tmp_path) + def test_cross_session_multisession_fold_equivalence(self, tmp_path): + """Custom folds spanning sessions keep per-session result provenance.""" + paradigm = FakeImageryParadigm() + ds = FakeDataset( + ["left_hand", "right_hand"], + n_subjects=2, + n_sessions=4, + seed=12, + ) + kwargs = { + "cv_class": GroupKFold, + "cv_kwargs": {"n_splits": 2}, + "random_state": 42, + "overwrite": True, + } + + eval_parallel = ev.CrossSessionEvaluation( + paradigm=paradigm, + datasets=[ds], + hdf5_path=str(tmp_path / "parallel_multisession"), + **kwargs, + ) + results_parallel = eval_parallel.process(pipelines) + + eval_legacy = ev.CrossSessionEvaluation( + paradigm=paradigm, + datasets=[ds], + hdf5_path=str(tmp_path / "legacy_multisession"), + **kwargs, + ) + results_legacy = eval_legacy._process_legacy( + pipelines, param_grid=None, postprocess_pipeline=None + ) + + keys = ["subject", "session", "pipeline"] + left = ( + results_parallel[keys + ["score", "n_samples_test"]] + .sort_values(keys) + .reset_index(drop=True) + ) + right = ( + results_legacy[keys + ["score", "n_samples_test"]] + .sort_values(keys) + .reset_index(drop=True) + ) + + assert len(left) == len(right) == 8 + assert left[keys].equals(right[keys]) + assert set(left["session"]) == {"0", "1", "2", "3"} + assert (left["n_samples_test"] > 0).all() + np.testing.assert_allclose( + left["score"].to_numpy(), + right["score"].to_numpy(), + rtol=1e-10, + atol=1e-10, + ) + np.testing.assert_array_equal( + left["n_samples_test"].to_numpy(), + right["n_samples_test"].to_numpy(), + ) + def test_cross_subject_equivalence(self, tmp_path): """CrossSubject parallel matches legacy scores.""" self._compare_parallel_vs_legacy(ev.CrossSubjectEvaluation, tmp_path) From 843a4c167337168f4c024bc7488d5ec8c062ac13 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Sat, 3 Oct 2026 16:35:20 +0000 Subject: [PATCH 2/7] [pre-commit.ci] auto fixes from pre-commit.com hooks --- moabb/tests/test_evaluations.py | 15 +++------------ 1 file changed, 3 insertions(+), 12 deletions(-) diff --git a/moabb/tests/test_evaluations.py b/moabb/tests/test_evaluations.py index 3c0fb58da..0ccac6401 100644 --- a/moabb/tests/test_evaluations.py +++ b/moabb/tests/test_evaluations.py @@ -1568,12 +1568,7 @@ def test_cross_session_equivalence(self, tmp_path): def test_cross_session_multisession_fold_equivalence(self, tmp_path): """Custom folds spanning sessions keep per-session result provenance.""" paradigm = FakeImageryParadigm() - ds = FakeDataset( - ["left_hand", "right_hand"], - n_subjects=2, - n_sessions=4, - seed=12, - ) + ds = FakeDataset(["left_hand", "right_hand"], n_subjects=2, n_sessions=4, seed=12) kwargs = { "cv_class": GroupKFold, "cv_kwargs": {"n_splits": 2}, @@ -1616,14 +1611,10 @@ def test_cross_session_multisession_fold_equivalence(self, tmp_path): assert set(left["session"]) == {"0", "1", "2", "3"} assert (left["n_samples_test"] > 0).all() np.testing.assert_allclose( - left["score"].to_numpy(), - right["score"].to_numpy(), - rtol=1e-10, - atol=1e-10, + left["score"].to_numpy(), right["score"].to_numpy(), rtol=1e-10, atol=1e-10 ) np.testing.assert_array_equal( - left["n_samples_test"].to_numpy(), - right["n_samples_test"].to_numpy(), + left["n_samples_test"].to_numpy(), right["n_samples_test"].to_numpy() ) def test_cross_subject_equivalence(self, tmp_path): From dda7a652361b7a94e5a6342b5292d71eb376bf19 Mon Sep 17 00:00:00 2001 From: lindicaphxag-tech Date: Sun, 4 Oct 2026 02:31:15 +0800 Subject: [PATCH 3/7] Fix GroupKFold regression setup --- moabb/tests/test_evaluations.py | 1 - 1 file changed, 1 deletion(-) diff --git a/moabb/tests/test_evaluations.py b/moabb/tests/test_evaluations.py index 0ccac6401..d8fb73cc2 100644 --- a/moabb/tests/test_evaluations.py +++ b/moabb/tests/test_evaluations.py @@ -1572,7 +1572,6 @@ def test_cross_session_multisession_fold_equivalence(self, tmp_path): kwargs = { "cv_class": GroupKFold, "cv_kwargs": {"n_splits": 2}, - "random_state": 42, "overwrite": True, } From 37fe19b28247e4a829af646ec520c7dc5c4bc465 Mon Sep 17 00:00:00 2001 From: "pre-commit-ci[bot]" <66853113+pre-commit-ci[bot]@users.noreply.github.com> Date: Sat, 3 Oct 2026 18:31:25 +0000 Subject: [PATCH 4/7] [pre-commit.ci] auto fixes from pre-commit.com hooks --- moabb/tests/test_evaluations.py | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/moabb/tests/test_evaluations.py b/moabb/tests/test_evaluations.py index d8fb73cc2..92a1da584 100644 --- a/moabb/tests/test_evaluations.py +++ b/moabb/tests/test_evaluations.py @@ -1569,11 +1569,7 @@ def test_cross_session_multisession_fold_equivalence(self, tmp_path): """Custom folds spanning sessions keep per-session result provenance.""" paradigm = FakeImageryParadigm() ds = FakeDataset(["left_hand", "right_hand"], n_subjects=2, n_sessions=4, seed=12) - kwargs = { - "cv_class": GroupKFold, - "cv_kwargs": {"n_splits": 2}, - "overwrite": True, - } + kwargs = {"cv_class": GroupKFold, "cv_kwargs": {"n_splits": 2}, "overwrite": True} eval_parallel = ev.CrossSessionEvaluation( paradigm=paradigm, From 92609a3cf2d42db656e63bac454625dd42d8fc10 Mon Sep 17 00:00:00 2001 From: lindicaphxag-tech Date: Sun, 4 Oct 2026 02:39:52 +0800 Subject: [PATCH 5/7] Test CrossSession public result provenance --- moabb/tests/test_evaluations.py | 16 ++-------------- 1 file changed, 2 insertions(+), 14 deletions(-) diff --git a/moabb/tests/test_evaluations.py b/moabb/tests/test_evaluations.py index 92a1da584..24439aa24 100644 --- a/moabb/tests/test_evaluations.py +++ b/moabb/tests/test_evaluations.py @@ -1590,27 +1590,15 @@ def test_cross_session_multisession_fold_equivalence(self, tmp_path): ) keys = ["subject", "session", "pipeline"] - left = ( - results_parallel[keys + ["score", "n_samples_test"]] - .sort_values(keys) - .reset_index(drop=True) - ) - right = ( - results_legacy[keys + ["score", "n_samples_test"]] - .sort_values(keys) - .reset_index(drop=True) - ) + left = results_parallel[keys + ["score"]].sort_values(keys).reset_index(drop=True) + right = results_legacy[keys + ["score"]].sort_values(keys).reset_index(drop=True) assert len(left) == len(right) == 8 assert left[keys].equals(right[keys]) assert set(left["session"]) == {"0", "1", "2", "3"} - assert (left["n_samples_test"] > 0).all() np.testing.assert_allclose( left["score"].to_numpy(), right["score"].to_numpy(), rtol=1e-10, atol=1e-10 ) - np.testing.assert_array_equal( - left["n_samples_test"].to_numpy(), right["n_samples_test"].to_numpy() - ) def test_cross_subject_equivalence(self, tmp_path): """CrossSubject parallel matches legacy scores.""" From 150ada97754deff06ef246edcb726101ae69e5c0 Mon Sep 17 00:00:00 2001 From: lindicaphxag-tech Date: Sun, 4 Oct 2026 02:45:06 +0800 Subject: [PATCH 6/7] Test persisted CrossSession result schema correctly From 098c684cb66a02bf24eea8c426eab28b07016ea6 Mon Sep 17 00:00:00 2001 From: lindicaphxag-tech Date: Sun, 4 Oct 2026 21:50:03 +0800 Subject: [PATCH 7/7] DOC note CrossSession custom-fold result API change --- docs/source/whats_new.rst | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/docs/source/whats_new.rst b/docs/source/whats_new.rst index d6871d6c7..8f8f04a27 100644 --- a/docs/source/whats_new.rst +++ b/docs/source/whats_new.rst @@ -27,7 +27,10 @@ Enhancements API changes ~~~~~~~~~~~ -- None yet. +- :class:`moabb.evaluations.CrossSessionEvaluation` custom cross-validation + folds that hold out multiple sessions now emit one result row per held-out + session instead of one aggregate row per fold. The default leave-one-session-out + behavior is unchanged (:gh:`1210` by `lindicaphxag-tech`_). Requirements ~~~~~~~~~~~~