diff --git a/docs/source/whats_new.rst b/docs/source/whats_new.rst index 11551c551..f7f391a1b 100644 --- a/docs/source/whats_new.rst +++ b/docs/source/whats_new.rst @@ -28,7 +28,10 @@ Enhancements API changes ~~~~~~~~~~~ -- None yet. +- :class:`moabb.evaluations.CrossSessionEvaluation` custom cross-validation + folds that hold out multiple sessions now emit one result row per held-out + session instead of one aggregate row per fold. The default leave-one-session-out + behavior is unchanged (:gh:`1210` by `lindicaphxag-tech`_). Requirements ~~~~~~~~~~~~ @@ -36,6 +39,12 @@ Requirements Bugs ~~~~ +- Keep :class:`moabb.evaluations.CrossSessionEvaluation` result provenance + session-specific when a custom cross-validator holds out more than one recording + session in the same fold. The estimator is still fitted once per fold, but each + held-out session is scored and stored separately, matching the evaluation's + session-level result contract in both the flattened and legacy execution paths + (by `lindicaphxag-tech`_). - Fix metadata-aware splitters collapsing distinct compound groups when column values contain the ``-`` separator. Multi-column group identities now use a canonical collision-free encoding, preserving the intended cross-validation and leakage boundary (by `lindicaphxag-tech`_). - Make :func:`moabb.analysis.chance_level.chance_by_chance` independent of result-row order when test-fold sizes vary: dataset-level adjusted thresholds now take the strictest exact-binomial cutoff across the fold sizes actually present (the cutoff is discrete and not strictly monotone in fold size), while inconsistent class counts are rejected (by `lindicaphxag-tech`_). - Fix ``SSVEP_TRCA`` trial centering: the inter-trial covariance step subtracted the mean across channels at each sample instead of each channel's mean over time, and did it in place on the filterbank data it received, which also fed ``Q`` and the class templates. Centering is now per channel over time, on a copy (:gh:`1183` by `Arthur031221`_). diff --git a/moabb/evaluations/evaluations.py b/moabb/evaluations/evaluations.py index e579264ab..272aa5465 100644 --- a/moabb/evaluations/evaluations.py +++ b/moabb/evaluations/evaluations.py @@ -302,6 +302,7 @@ class CrossSessionEvaluation(BaseEvaluation): """ _eval_type = "CrossSession" + _score_per_session = True def _create_splitter(self): """Create the CrossSessionSplitter for parallel evaluation.""" @@ -391,24 +392,27 @@ def evaluate( eval_type="CrossSession", ) - res = self._build_scored_result( - dataset, - subject, - groups[test][0], - name, - len(train), - nchan, - duration, - scorer, - cvclf, - X[test], - y[test], - ) + test_sessions = groups[test] + for session in np.unique(test_sessions): + session_test = test[test_sessions == session] + res = self._build_scored_result( + dataset, + subject, + session, + name, + len(train), + nchan, + duration, + scorer, + cvclf, + X[session_test], + y[session_test], + ) - if _carbonfootprint: - self._attach_emissions(res, emissions, task_name) + if _carbonfootprint: + self._attach_emissions(res, emissions, task_name) - yield res + yield res if _carbonfootprint: tracker.stop() diff --git a/moabb/tests/test_evaluations.py b/moabb/tests/test_evaluations.py index 4c6566c07..24439aa24 100644 --- a/moabb/tests/test_evaluations.py +++ b/moabb/tests/test_evaluations.py @@ -1565,6 +1565,41 @@ def test_cross_session_equivalence(self, tmp_path): """CrossSession parallel matches legacy scores.""" self._compare_parallel_vs_legacy(ev.CrossSessionEvaluation, tmp_path) + def test_cross_session_multisession_fold_equivalence(self, tmp_path): + """Custom folds spanning sessions keep per-session result provenance.""" + paradigm = FakeImageryParadigm() + ds = FakeDataset(["left_hand", "right_hand"], n_subjects=2, n_sessions=4, seed=12) + kwargs = {"cv_class": GroupKFold, "cv_kwargs": {"n_splits": 2}, "overwrite": True} + + eval_parallel = ev.CrossSessionEvaluation( + paradigm=paradigm, + datasets=[ds], + hdf5_path=str(tmp_path / "parallel_multisession"), + **kwargs, + ) + results_parallel = eval_parallel.process(pipelines) + + eval_legacy = ev.CrossSessionEvaluation( + paradigm=paradigm, + datasets=[ds], + hdf5_path=str(tmp_path / "legacy_multisession"), + **kwargs, + ) + results_legacy = eval_legacy._process_legacy( + pipelines, param_grid=None, postprocess_pipeline=None + ) + + keys = ["subject", "session", "pipeline"] + left = results_parallel[keys + ["score"]].sort_values(keys).reset_index(drop=True) + right = results_legacy[keys + ["score"]].sort_values(keys).reset_index(drop=True) + + assert len(left) == len(right) == 8 + assert left[keys].equals(right[keys]) + assert set(left["session"]) == {"0", "1", "2", "3"} + np.testing.assert_allclose( + left["score"].to_numpy(), right["score"].to_numpy(), rtol=1e-10, atol=1e-10 + ) + def test_cross_subject_equivalence(self, tmp_path): """CrossSubject parallel matches legacy scores.""" self._compare_parallel_vs_legacy(ev.CrossSubjectEvaluation, tmp_path)