From 018e6beb2b917dc8c764c9d4ad824f360693f81c Mon Sep 17 00:00:00 2001 From: Bru Date: Mon, 5 Oct 2026 21:43:44 +0200 Subject: [PATCH] DOC document metadata-only paradigm compatibility checks --- docs/source/whats_new.rst | 1 + .../plot_metadata_preflight.py | 93 ++++++++++++++++ moabb/tests/test_metadata_preflight.py | 100 ++++++++++++++++++ 3 files changed, 194 insertions(+) create mode 100644 examples/data_management_and_configuration/plot_metadata_preflight.py create mode 100644 moabb/tests/test_metadata_preflight.py diff --git a/docs/source/whats_new.rst b/docs/source/whats_new.rst index 22a320264..108bfe3ce 100644 --- a/docs/source/whats_new.rst +++ b/docs/source/whats_new.rst @@ -23,6 +23,7 @@ Version 1.8 (Source - GitHub) Enhancements ~~~~~~~~~~~~ +- Document a download-free, paradigm-only metadata preflight using the existing ``MotorImagery.is_valid`` API, including explicit ``n_classes`` semantics and unknown evaluation compatibility; add a no-I/O regression test for the recipe (by `Bruno Aristimunha`_). - Ship MOABB's reference benchmark pipeline configs as package data and add :func:`moabb.pipelines.get_benchmark_pipelines`, so published baseline pipelines are available from normal wheel/sdist installs instead of only from a source checkout. Pipeline parsing now also accepts a single ``.py`` config and rejects valid-but-empty config directories instead of silently returning no pipelines (:gh:`1149` by `lindicaphxag-tech`_). - Allow :class:`~moabb.evaluations.CrossSubjectEvaluation` to accept an optional top-level ``splitter`` instance, enabling transfer-learning protocols to reuse MOABB's existing caching, parallel execution, and result handling while preserving the default protocol (:gh:`1088` by `lindicaphxag-tech`_). - Add Leelakittisin2025 sit-stand transition imagery, PerezBlanco2026 wrist motor-execution, and Vagaja2023 VR motor-imagery datasets (:pr:`1199`) (by `Bruno Aristimunha`_). diff --git a/examples/data_management_and_configuration/plot_metadata_preflight.py b/examples/data_management_and_configuration/plot_metadata_preflight.py new file mode 100644 index 000000000..8ad082e8c --- /dev/null +++ b/examples/data_management_and_configuration/plot_metadata_preflight.py @@ -0,0 +1,93 @@ +""" +===================================== +Check paradigm metadata before loading +===================================== + +Use :meth:`moabb.paradigms.MotorImagery.is_valid` on an existing MOABB dataset +instance to check its declared paradigm and events without loading recordings. +This is a **paradigm-only** check, not a guarantee that an evaluation can run. +The constructors used below only initialize metadata; that is not a guarantee +about every dataset constructor, especially custom ones. + +A remote catalogue record is not a :class:`moabb.datasets.base.BaseDataset`. +Resolve its identity to an existing loader and verify its metadata first. Do not +invent a dataset or substitute a guessed session count for missing information. +""" + +# License: BSD (3-clause) + +import json + +import moabb +from moabb.datasets import BNCI2014_001, AlexMI +from moabb.paradigms import MotorImagery + + +############################################################################### +# Require two overlapping classes explicitly +# ----------------------------------------- +# AlexMI declares right-hand, feet and rest events, but not left-hand events. +# BNCI2014_001 declares both requested hand events. Neither constructor below +# downloads data. The check does not call ``get_data``, ``data_path``, +# ``used_events`` or ``paradigm.datasets`` (which enumerates datasets). + +datasets = [AlexMI(), BNCI2014_001()] +paradigm = MotorImagery(events=["left_hand", "right_hand"], n_classes=2) +report = [ + { + "dataset": dataset.code, + "moabb_version": moabb.__version__, + "paradigm_compatible": paradigm.is_valid(dataset), + "declared_sessions": dataset.n_sessions, + "evaluation_compatible": None, + } + for dataset in datasets +] +print(json.dumps(report, indent=2)) + +############################################################################### +# Preserve the native n_classes semantics +# -------------------------------------- +# With ``n_classes=None`` (the default), naming events does NOT require two +# overlapping classes. Thus this predicate accepts AlexMI. It still rejects a +# non-imagery dataset. Use ``n_classes=2`` when two overlapping classes are your +# intent; use :class:`moabb.paradigms.LeftRightImagery` for the fixed hand pair. +# Do not call ``used_events`` as a preflight: it can update ``n_classes``. + +unspecified_classes = MotorImagery(events=["left_hand", "right_hand"]) +print(unspecified_classes.is_valid(datasets[0])) # True +print(paradigm.is_valid(datasets[0])) # False + +############################################################################### +# Evaluation compatibility stays unknown +# -------------------------------------- +# ``evaluation_compatible`` above is JSON null, not false. Evaluation predicates +# are instance methods. Constructing an evaluation is NOT a pure preflight: +# it creates Results storage and can remove incompatible entries from the +# supplied dataset list. Do not construct an uninitialized evaluation or call +# an instance method with a dummy receiver to avoid those effects. +# +# Cross-session evaluation requires multiple sessions; AlexMI declares one, +# while BNCI2014_001 declares two. These are loader declarations, not evidence +# that a particular subject or selected session subset has sufficient data. +# We report the count without reimplementing evaluation predicates. Subject +# counts, folds, actual trials, channels, pipeline compatibility and runtime +# feasibility remain unchecked. A true paradigm result is not a runnable +# benchmark, a verified licence, or approval for data use. +# +# A synthetic discovery record illustrates missingness only. It is deliberately +# NOT passed to ``is_valid`` or converted into a MOABB dataset. Its unknown +# sessions remain null; missing metadata is not an incompatibility verdict. + +remote_record = {"id": "synthetic-unresolved-record", "n_sessions": None} +print(json.dumps(remote_record)) + +############################################################################### +# References and provenance +# ------------------------- +# See the linked API pages above and :class:`moabb.evaluations.CrossSessionEvaluation` +# for evaluation usage. For reproducibility, record ``moabb.__version__`` plus +# the source revision for a development installation. This recipe's metadata +# and predicate semantics were checked against `MOABB source at 3888687e0 +# `_. +# Version labels alone do not identify a particular development checkout. diff --git a/moabb/tests/test_metadata_preflight.py b/moabb/tests/test_metadata_preflight.py new file mode 100644 index 000000000..7a96861ae --- /dev/null +++ b/moabb/tests/test_metadata_preflight.py @@ -0,0 +1,100 @@ +"""Keep the documented paradigm-only metadata recipe free of runtime work.""" + +import builtins +import io +import json +import os +import socket +from copy import deepcopy +from pathlib import Path + +import pytest +from sklearn.pipeline import Pipeline + +import moabb +from moabb.analysis import Results +from moabb.datasets import BNCI2014_001, BNCI2014_009, AlexMI +from moabb.datasets.base import BaseDataset +from moabb.evaluations.base import BaseEvaluation +from moabb.paradigms import MotorImagery +from moabb.paradigms.base import BaseParadigm + + +_EXAMPLE = ( + Path(__file__).resolve().parents[2] + / "examples/data_management_and_configuration/plot_metadata_preflight.py" +) + + +def _forbid(*args, **kwargs): + raise AssertionError("metadata preflight must not perform I/O or runtime work") + + +def test_documented_metadata_preflight(monkeypatch): + # Load the source before forbidding I/O; imports have already been performed. + source = compile(_EXAMPLE.read_text(), str(_EXAMPLE), "exec") + dataset_types = (AlexMI, BNCI2014_001, BNCI2014_009) + datasets = [dataset_type() for dataset_type in dataset_types] + strict = MotorImagery(events=["left_hand", "right_hand"], n_classes=2) + default = MotorImagery(events=["left_hand", "right_hand"]) + datasets_before = deepcopy([vars(dataset) for dataset in datasets]) + paradigms_before = deepcopy([vars(strict), vars(default)]) + environment_before = dict(os.environ) + output = [] + + with monkeypatch.context() as patch: + for module, name in ( + (builtins, "open"), + (io, "open"), + (os, "open"), + (os, "mkdir"), + (socket, "socket"), + (socket, "create_connection"), + (socket, "getaddrinfo"), + (Results, "__init__"), + (BaseEvaluation, "__init__"), + (BaseDataset, "get_data"), + (BaseDataset, "download"), + (BaseDataset, "data_path"), + (BaseParadigm, "get_data"), + (MotorImagery, "used_events"), + (Pipeline, "fit"), + ): + patch.setattr(module, name, _forbid) + for dataset_type in dataset_types: + patch.setattr(dataset_type, "data_path", _forbid) + patch.setattr(dataset_type, "_get_single_subject_data", _forbid) + patch.setattr(MotorImagery, "datasets", property(_forbid)) + patch.setattr(builtins, "print", lambda value: output.append(value)) + + namespace = {} + exec(source, namespace) + # Existing native predicates are the sole rule source. Check their + # default/explicit class behavior and the non-imagery rejection. + assert [strict.is_valid(dataset) for dataset in datasets] == [False, True, False] + assert [default.is_valid(dataset) for dataset in datasets] == [True, True, False] + with pytest.raises(AttributeError): + strict.is_valid({"n_sessions": None}) + assert [vars(dataset) for dataset in datasets] == datasets_before + assert [vars(strict), vars(default)] == paradigms_before + assert dict(os.environ) == environment_before + + report = json.loads(output[0]) + assert report == [ + { + "dataset": dataset.code, + "moabb_version": moabb.__version__, + "paradigm_compatible": compatible, + "declared_sessions": sessions, + "evaluation_compatible": None, + } + for dataset, compatible, sessions in zip(datasets[:2], (False, True), (1, 2)) + ] + assert output[1:3] == [True, False] + assert json.loads(output[3]) == { + "id": "synthetic-unresolved-record", + "n_sessions": None, + } + assert namespace["paradigm"].n_classes == 2 + assert namespace["unspecified_classes"].n_classes is None + assert len(namespace["datasets"]) == 2