From e1953d17fa857d859b5f4699214ea76e5add15e3 Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Wed, 3 Jun 2026 14:16:09 +1000 Subject: [PATCH 01/14] updated deidentify signature --- extras/fileformats/extras/biosig/eeg.py | 9 +++---- fileformats/biosig/base.py | 33 +++++++++++++++++++------ 2 files changed, 30 insertions(+), 12 deletions(-) diff --git a/extras/fileformats/extras/biosig/eeg.py b/extras/fileformats/extras/biosig/eeg.py index eca3888..037eafb 100644 --- a/extras/fileformats/extras/biosig/eeg.py +++ b/extras/fileformats/extras/biosig/eeg.py @@ -54,14 +54,13 @@ def brain_vision_read_metadata(bv: BrainVision) -> dict[str, ty.Any]: @extra_implementation(Biosig.deidentify) def brain_vision_deidentify( brain_vision: BrainVision, - out_dir: ty.Optional[Path] = None, - new_stem: ty.Optional[str] = None, - copy_mode: FileSet.CopyMode = FileSet.CopyMode.copy, + spec: ty.Any = None, + out_dir: os.PathLike[str] | None = None, ) -> BrainVision: if out_dir is None: out_dir = Path(tempfile.mkdtemp()) - out_dir.mkdir(parents=True, exist_ok=True) - deidentified = brain_vision.copy(out_dir, new_stem=new_stem, mode=copy_mode) + Path(out_dir).mkdir(parents=True, exist_ok=True) + deidentified = brain_vision.copy(Path(out_dir)) raise NotImplementedError( "need to implemnent deidentification techniques and a save method. If there is a standard " "form to load the data into (e.g. MNE) it would be best to implement FileSet.load and FileSet.save" diff --git a/fileformats/biosig/base.py b/fileformats/biosig/base.py index 242785f..2250e9b 100644 --- a/fileformats/biosig/base.py +++ b/fileformats/biosig/base.py @@ -1,5 +1,5 @@ import typing as ty -from pathlib import Path +import os from fileformats.core import FileSet, extra @@ -9,10 +9,29 @@ class Biosig(FileSet): @extra def deidentify( self, - out_dir: Path | None = None, - new_stem: str | None = None, - copy_mode: FileSet.CopyMode = FileSet.CopyMode.copy, - ) -> ty.Self: - """Returns a new copy of the image with any subject-identifying information - stripped from the from the image header""" + spec: ty.Any = None, + out_dir: os.PathLike[str] | None = None, + ) -> tuple[ty.Self, dict[str, ty.Any]]: + """ + Deidentifies the dataset by stripping any subject-identifying information from the + image header. The exact implementation of this method will depend on the + specific image format and the type of identifying information that is present. + + Parameters + ---------- + spec: Any, optional + A specification for the deidentification process, which may include details on + which fields to remove or how to handle certain types of data. The exact + structure of this specification will depend on the specific image format and the + requirements of the deidentification process. + + Returns + ------- + Self + A new instance of the image with any subject-identifying information stripped from + the image header. + dict[str, Any] + A JSON-like nested dictionary containing the original values from the header that + were stripped/modified during the deidentification process. + """ raise NotImplementedError From 9777c6fda658137e920f16b5391cd9a2312cb49b Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Wed, 3 Jun 2026 14:29:15 +1000 Subject: [PATCH 02/14] fixed up imports and function signatures for extras implementations --- extras/fileformats/extras/biosig/__init__.py | 4 ++++ extras/fileformats/extras/biosig/eeg.py | 14 ++++++++------ extras/fileformats/extras/biosig/meg.py | 8 ++++---- 3 files changed, 16 insertions(+), 10 deletions(-) diff --git a/extras/fileformats/extras/biosig/__init__.py b/extras/fileformats/extras/biosig/__init__.py index 8dee4bf..be5db19 100644 --- a/extras/fileformats/extras/biosig/__init__.py +++ b/extras/fileformats/extras/biosig/__init__.py @@ -1 +1,5 @@ from ._version import __version__ +from . import eeg +from . import meg + +__all__ = ["__version__", "eeg", "meg"] diff --git a/extras/fileformats/extras/biosig/eeg.py b/extras/fileformats/extras/biosig/eeg.py index 037eafb..d2ad5cb 100644 --- a/extras/fileformats/extras/biosig/eeg.py +++ b/extras/fileformats/extras/biosig/eeg.py @@ -13,19 +13,19 @@ @extra_implementation(FileSet.read_metadata) -def fif_read_metadata(fif: Fif) -> dict[str, ty.Any]: +def fif_read_metadata(fif: Fif, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: raw = mne.io.read_raw_fif(fif.fspath, preload=False, verbose=False) return _info_to_metadata(raw.info) @extra_implementation(FileSet.read_metadata) -def fif_gz_read_metadata(fif: FifGz) -> dict[str, ty.Any]: +def fif_gz_read_metadata(fif: FifGz, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: raw = mne.io.read_raw_fif(fif.fspath, preload=False, verbose=False) return _info_to_metadata(raw.info) @extra_implementation(FileSet.read_metadata) -def edf_read_metadata(edf: Edf) -> dict[str, ty.Any]: +def edf_read_metadata(edf: Edf, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: raw = mne.io.read_raw_edf(edf.fspath, preload=False, verbose=False) return { **_info_to_metadata(raw.info), @@ -34,7 +34,7 @@ def edf_read_metadata(edf: Edf) -> dict[str, ty.Any]: @extra_implementation(FileSet.read_metadata) -def edf_plus_read_metadata(edf: EdfPlus) -> dict[str, ty.Any]: +def edf_plus_read_metadata(edf: EdfPlus, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: raw = mne.io.read_raw_edf(edf.fspath, preload=False, verbose=False) return { **_info_to_metadata(raw.info), @@ -43,7 +43,9 @@ def edf_plus_read_metadata(edf: EdfPlus) -> dict[str, ty.Any]: @extra_implementation(FileSet.read_metadata) -def brain_vision_read_metadata(bv: BrainVision) -> dict[str, ty.Any]: +def brain_vision_read_metadata( + bv: BrainVision, **kwargs: ty.Any +) -> ty.Mapping[str, ty.Any]: raw = mne.io.read_raw_brainvision(bv.header_file, preload=False, verbose=False) return { **_info_to_metadata(raw.info), @@ -56,7 +58,7 @@ def brain_vision_deidentify( brain_vision: BrainVision, spec: ty.Any = None, out_dir: os.PathLike[str] | None = None, -) -> BrainVision: +) -> tuple[BrainVision, dict[str, ty.Any]]: if out_dir is None: out_dir = Path(tempfile.mkdtemp()) Path(out_dir).mkdir(parents=True, exist_ok=True) diff --git a/extras/fileformats/extras/biosig/meg.py b/extras/fileformats/extras/biosig/meg.py index 8d7b64e..a3a75a5 100644 --- a/extras/fileformats/extras/biosig/meg.py +++ b/extras/fileformats/extras/biosig/meg.py @@ -10,7 +10,7 @@ @extra_implementation(FileSet.read_metadata) -def ctf_read_metadata(ctf: Ctf) -> dict[str, ty.Any]: +def ctf_read_metadata(ctf: Ctf, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: raw = mne.io.read_raw_ctf(ctf.fspath, preload=False, verbose=False) return { **_info_to_metadata(raw.info), @@ -23,9 +23,9 @@ def ctf_read_metadata(ctf: Ctf) -> dict[str, ty.Any]: @extra_implementation(FileSet.read_metadata) -def kit_read_metadata(kit: Kit) -> dict[str, ty.Any]: - mrk_path = kit._find_kit_mrk_file() - return mne.io.read_raw_kit(kit, mrk=mrk_path, verbose=False) +def kit_read_metadata(kit: Kit, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: + mrk_path = kit._find_kit_mrk_file() # type: ignore[attr-defined] + return mne.io.read_raw_kit(kit, mrk=mrk_path, verbose=False) # type: ignore[no-any-return] def _parse_infods(ds_path: Path) -> dict[str, ty.Any]: From 778bba937429dc418f430b2bd7a439cc864c0050 Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Wed, 3 Jun 2026 17:42:58 +1000 Subject: [PATCH 03/14] cleaning up extras implementations and fixed unittests --- conftest.py | 16 ++--- extras/fileformats/extras/biosig/eeg.py | 69 ++++++++++--------- extras/fileformats/extras/biosig/meg.py | 69 ++++++++----------- .../extras/biosig/tests/test_eeg_extras.py | 20 ++---- .../extras/biosig/tests/test_meg_extras.py | 40 ++++++++++- extras/fileformats/extras/biosig/utils.py | 63 ++++++++--------- fileformats/biosig/__init__.py | 3 +- fileformats/biosig/eeg.py | 41 ++++------- fileformats/biosig/meg.py | 58 ++++++++++------ fileformats/biosig/tests/test_eeg.py | 12 +--- fileformats/biosig/tests/test_meg.py | 6 +- 11 files changed, 207 insertions(+), 190 deletions(-) diff --git a/conftest.py b/conftest.py index daba69f..e6259ac 100644 --- a/conftest.py +++ b/conftest.py @@ -50,25 +50,25 @@ def testing_data_path() -> Path: @pytest.fixture(scope="session") -def fif_path(sample_data_path) -> Path: +def fif_path(sample_data_path: Path) -> Path: return sample_data_path / "MEG" / "sample" / "sample_audvis_raw.fif" @pytest.fixture(scope="session") -def edf_path() -> Path: - return Path(mne.datasets.eegbci.load_data(subject=1, runs=[1])[0]) +def edf_plus_path() -> Path: + return Path(mne.datasets.eegbci.load_data(subjects=1, runs=[1])[0]) @pytest.fixture(scope="session") -def bv_vhdr_path(testing_data_path) -> Path: - return testing_data_path / "BrainVision" / "test.vhdr" +def bv_vhdr_path(testing_data_path: Path) -> Path: + return testing_data_path / "BrainVision" / "test_NO.vhdr" @pytest.fixture(scope="session") -def ctf_ds_path(testing_data_path) -> Path: +def ctf_ds_path(testing_data_path: Path) -> Path: return testing_data_path / "CTF" / "testdata_ctf.ds" @pytest.fixture(scope="session") -def kit_sqd_path(testing_data_path) -> Path: - return testing_data_path / "KIT" / "test.sqd" +def kit_sqd_path(testing_data_path: Path) -> Path: + return testing_data_path / "KIT" / "MQKIT_125_2sec.con" diff --git a/extras/fileformats/extras/biosig/eeg.py b/extras/fileformats/extras/biosig/eeg.py index d2ad5cb..457a1d6 100644 --- a/extras/fileformats/extras/biosig/eeg.py +++ b/extras/fileformats/extras/biosig/eeg.py @@ -5,40 +5,29 @@ from pathlib import Path import mne.io +import mne.export from fileformats.core import extra_implementation, FileSet -from fileformats.biosig import Biosig, BrainVision, Edf, EdfPlus, Fif, FifGz +from fileformats.biosig import Biosig, BrainVision, Edf, EdfPlus -from .utils import _info_to_metadata - - -@extra_implementation(FileSet.read_metadata) -def fif_read_metadata(fif: Fif, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: - raw = mne.io.read_raw_fif(fif.fspath, preload=False, verbose=False) - return _info_to_metadata(raw.info) - - -@extra_implementation(FileSet.read_metadata) -def fif_gz_read_metadata(fif: FifGz, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: - raw = mne.io.read_raw_fif(fif.fspath, preload=False, verbose=False) - return _info_to_metadata(raw.info) +from .utils import mne_deidentify @extra_implementation(FileSet.read_metadata) def edf_read_metadata(edf: Edf, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: - raw = mne.io.read_raw_edf(edf.fspath, preload=False, verbose=False) + raw = mne.io.read_raw_edf(edf, preload=False, verbose=False) return { - **_info_to_metadata(raw.info), - **_parse_edf_header(edf.fspath), + **raw.info.to_json_dict(), + **_parse_edf_header(edf), } @extra_implementation(FileSet.read_metadata) def edf_plus_read_metadata(edf: EdfPlus, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: - raw = mne.io.read_raw_edf(edf.fspath, preload=False, verbose=False) + raw = mne.io.read_raw_edf(edf, preload=False, verbose=False) return { - **_info_to_metadata(raw.info), - **_parse_edf_header(edf.fspath), + **raw.info.to_json_dict(), + **_parse_edf_header(edf), } @@ -48,27 +37,41 @@ def brain_vision_read_metadata( ) -> ty.Mapping[str, ty.Any]: raw = mne.io.read_raw_brainvision(bv.header_file, preload=False, verbose=False) return { - **_info_to_metadata(raw.info), + **raw.info.to_json_dict(), **_parse_vhdr(bv.header_file), } +@extra_implementation(Biosig.deidentify) +def edf_deidentify( + edf: Edf, + spec: ty.Any = None, + out_dir: os.PathLike[str] | None = None, +) -> tuple[Edf, dict[str, ty.Any]]: + out_dir = Path(tempfile.mkdtemp() if out_dir is None else out_dir) + out_dir.mkdir(parents=True, exist_ok=True) + raw = mne.io.read_raw_edf(edf, preload=True, verbose=False) + deidentified_info, reid = mne_deidentify(raw, spec) + raw.info = deidentified_info + deid_fspath = out_dir / "eeg.edf" + mne.export.export_raw(deid_fspath, raw, fmt="edf", overwrite=True) + return type(edf)(deid_fspath), reid + + @extra_implementation(Biosig.deidentify) def brain_vision_deidentify( - brain_vision: BrainVision, + bv: BrainVision, spec: ty.Any = None, out_dir: os.PathLike[str] | None = None, ) -> tuple[BrainVision, dict[str, ty.Any]]: - if out_dir is None: - out_dir = Path(tempfile.mkdtemp()) - Path(out_dir).mkdir(parents=True, exist_ok=True) - deidentified = brain_vision.copy(Path(out_dir)) - raise NotImplementedError( - "need to implemnent deidentification techniques and a save method. If there is a standard " - "form to load the data into (e.g. MNE) it would be best to implement FileSet.load and FileSet.save" - "methods" - ) - return deidentified + out_dir = Path(tempfile.mkdtemp() if out_dir is None else out_dir) + out_dir.mkdir(parents=True, exist_ok=True) + raw = mne.io.read_raw_brainvision(bv.header_file, preload=True, verbose=False) + deidentified_info, reid = mne_deidentify(raw, spec) + raw.info = deidentified_info + deid_vhdr = out_dir / "eeg.vhdr" + mne.export.export_raw(deid_vhdr, raw, fmt="brainvision", overwrite=True) + return BrainVision(out_dir / "eeg.eeg"), reid def _parse_edf_header(path: os.PathLike[str]) -> dict[str, ty.Any]: @@ -118,7 +121,7 @@ def _parse_vhdr(path: os.PathLike[str]) -> dict[str, ty.Any]: """ parser = configparser.RawConfigParser() # vhdr files start with a magic line before the first INI section — skip it - with open(path, encoding="utf-8", errors="replace") as f: + with open(path, encoding="utf-8-sig", errors="replace") as f: lines = f.readlines() ini_lines = [line for line in lines if not line.startswith("Brain Vision")] parser.read_string("".join(ini_lines)) diff --git a/extras/fileformats/extras/biosig/meg.py b/extras/fileformats/extras/biosig/meg.py index a3a75a5..9f22346 100644 --- a/extras/fileformats/extras/biosig/meg.py +++ b/extras/fileformats/extras/biosig/meg.py @@ -1,56 +1,43 @@ +import os import typing as ty -import xml.etree.ElementTree as ET from pathlib import Path +import tempfile import mne.io from fileformats.core import extra_implementation, FileSet -from fileformats.biosig import Ctf, Kit -from .utils import _info_to_metadata +from fileformats.biosig import Ctf, Kit, Fif +from fileformats.biosig.base import Biosig + +from .utils import mne_deidentify @extra_implementation(FileSet.read_metadata) def ctf_read_metadata(ctf: Ctf, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: - raw = mne.io.read_raw_ctf(ctf.fspath, preload=False, verbose=False) - return { - **_info_to_metadata(raw.info), - **_parse_infods(Path(ctf.fspath)), - } + return mne.io.read_raw_ctf(ctf, preload=False, verbose=False).info.to_json_dict() # type: ignore[no-any-return] -# elif ext in [".sqd", ".con"]: # KIT/RIKEN main data files -# # For KIT format, we need to find the marker file (.mrk) in the same directory +@extra_implementation(FileSet.read_metadata) +def kit_read_metadata(kit: Kit, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: + return mne.io.read_raw_kit(kit, mrk=kit.mark_file, verbose=False).info.to_json_dict() # type: ignore[no-any-return] @extra_implementation(FileSet.read_metadata) -def kit_read_metadata(kit: Kit, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: - mrk_path = kit._find_kit_mrk_file() # type: ignore[attr-defined] - return mne.io.read_raw_kit(kit, mrk=mrk_path, verbose=False) # type: ignore[no-any-return] - - -def _parse_infods(ds_path: Path) -> dict[str, ty.Any]: - """ - Parse the .infods XML sidecar in a CTF .ds directory for metadata that - MNE does not surface via raw.info (subject name, operator, study description). - Returns an empty dict if no .infods file is present. - """ - infods_files = list(ds_path.glob("*.infods")) - if not infods_files: - return {} - tree = ET.parse(infods_files[0]) - root = tree.getroot() - - def find(tag: str) -> str | None: - el = root.find(f".//{tag}") - return el.text.strip() if el is not None and el.text else None - - return { - "subject_id": find("SUBJECTID"), - "subject_name": find("SUBJECTNAME"), - "operator": find("OPERATOR"), - "institution": find("INSTITUTION"), - "study_description": find("STUDYDESCRIPTION"), - "run_description": find("RUNDESCRIPTION"), - "acquisition_datetime": find("ACQUISITIONDATETIME"), - "date": find("DATE"), - } +def fif_read_metadata(fif: Fif, **kwargs: ty.Any) -> ty.Mapping[str, ty.Any]: + return mne.io.read_raw_fif(fif, preload=False, verbose=False).info.to_json_dict() # type: ignore[no-any-return] + + +@extra_implementation(Biosig.deidentify) +def fif_deidentify( + fif: Fif, + spec: ty.Any = None, + out_dir: os.PathLike[str] | None = None, +) -> tuple[Fif, dict[str, ty.Any]]: + out_dir = Path(tempfile.mkdtemp() if out_dir is None else out_dir) + out_dir.mkdir(parents=True, exist_ok=True) + raw = mne.io.read_raw_fif(fif, preload=True, verbose=False) + deidentified_info, reid = mne_deidentify(raw, spec) + raw.info = deidentified_info + deid_fspath = out_dir / "meg-signals.fif" + raw.save(deid_fspath, overwrite=True) + return Fif(deid_fspath), reid diff --git a/extras/fileformats/extras/biosig/tests/test_eeg_extras.py b/extras/fileformats/extras/biosig/tests/test_eeg_extras.py index 4bad929..1d9716d 100644 --- a/extras/fileformats/extras/biosig/tests/test_eeg_extras.py +++ b/extras/fileformats/extras/biosig/tests/test_eeg_extras.py @@ -12,28 +12,16 @@ from fileformats.biosig import ( BrainVision, - Edf, - Fif, + EdfPlus, ) -# ------------------------------ -# EEG: FIF -# ------------------------------ - - -def test_fif_read_metadata(fif_path): - metadata = Fif(fif_path).metadata - assert isinstance(metadata, dict) - assert metadata["sfreq"] is not None - - # ------------------------------ # EEG: EDF # ------------------------------ -def test_edf_read_metadata(edf_path): - metadata = Edf(edf_path).metadata +def test_edf_plus_read_metadata(edf_plus_path): + metadata = EdfPlus(edf_plus_path).metadata assert isinstance(metadata, dict) assert metadata["sfreq"] is not None assert "edf_patient_code" in metadata @@ -45,7 +33,7 @@ def test_edf_read_metadata(edf_path): def test_brainvision_read_metadata(bv_vhdr_path): - metadata = BrainVision(bv_vhdr_path.iterdir()).metadata + metadata = BrainVision(bv_vhdr_path.with_suffix(".eeg")).metadata assert isinstance(metadata, dict) assert metadata["sfreq"] is not None assert "bv_n_channels" in metadata diff --git a/extras/fileformats/extras/biosig/tests/test_meg_extras.py b/extras/fileformats/extras/biosig/tests/test_meg_extras.py index 27b0e43..c3a2f22 100644 --- a/extras/fileformats/extras/biosig/tests/test_meg_extras.py +++ b/extras/fileformats/extras/biosig/tests/test_meg_extras.py @@ -10,8 +10,7 @@ - miaocao@swin.edu.au """ -from fileformats.biosig import Ctf, Kit - +from fileformats.biosig import Ctf, Kit, Fif # ------------------------------ # MEG: CTF @@ -24,6 +23,43 @@ def test_ctf_read_metadata(ctf_ds_path): assert metadata["sfreq"] is not None +# ------------------------------ +# MEG: FIF +# ------------------------------ + + +def test_fif_read_metadata(fif_path): + metadata = Fif(fif_path).metadata + assert isinstance(metadata, dict) + assert metadata["sfreq"] is not None + + +def test_fif_deidentify(fif_path, tmp_path): + fif = Fif(fif_path) + orig_metadata = fif.metadata + + deid_fif, reid = fif.deidentify(out_dir=tmp_path) + + assert isinstance(deid_fif, Fif) + assert isinstance(reid, dict) + assert reid, "expected at least one field to be stripped or changed" + + deid_metadata = deid_fif.metadata + + # Every field recorded in reid should differ between original and deidentified + for key in reid: + assert orig_metadata.get(key) != deid_metadata.get( + key + ), f"reid claims '{key}' changed but original and deidentified values match" + + # Subject identifying fields should be absent or cleared in the deidentified file + deid_subject_info = deid_metadata.get("subject_info") or {} + for pii_field in ("last_name", "first_name", "birthday"): + assert not deid_subject_info.get( + pii_field + ), f"subject_info.{pii_field} was not cleared by deidentification" + + # ------------------------------ # MEG: KIT # ------------------------------ diff --git a/extras/fileformats/extras/biosig/utils.py b/extras/fileformats/extras/biosig/utils.py index de2df68..6a08437 100644 --- a/extras/fileformats/extras/biosig/utils.py +++ b/extras/fileformats/extras/biosig/utils.py @@ -1,36 +1,37 @@ +import json import typing as ty import mne +import mne.io -def _info_to_metadata(info: mne.Info) -> dict[str, ty.Any]: - """Extract study-sorting fields from an MNE Info object.""" - subj = info.get("subject_info") or {} - dev = info.get("device_info") or {} - return { - # Subject - "subject_id": subj.get("id"), - "subject_first_name": subj.get("first_name"), - "subject_last_name": subj.get("last_name"), - "subject_birthday": subj.get("birthday"), - "subject_sex": subj.get("sex"), - "subject_hand": subj.get("hand"), - # Study - "proj_name": info.get("proj_name"), - "proj_id": info.get("proj_id"), - "experimenter": info.get("experimenter"), - "description": info.get("description"), - # Recording - "meas_date": info.get("meas_date"), - "utc_offset": info.get("utc_offset"), - # Acquisition - "sfreq": info.get("sfreq"), - "nchan": info.get("nchan"), - "highpass": info.get("highpass"), - "lowpass": info.get("lowpass"), - # Device - "device_type": dev.get("type"), - "device_model": dev.get("model"), - "device_serial": dev.get("serial"), - "device_site": dev.get("site"), - } +def mne_deidentify( + raw: mne.io.BaseRaw, + spec: ty.Any = None, +) -> tuple[mne.Info, dict[str, ty.Any]]: + """Anonymize an MNE Raw object and return the deidentified Info alongside + a dict of the original values that were stripped or changed.""" + orig_info_dict = raw.info.to_json_dict() + kwargs = json.load(open(spec)) if spec is not None else {} + deidentified_info = mne.io.anonymize_info(raw.info, verbose=None, **kwargs) + reid = dict_diff(orig_info_dict, deidentified_info.to_json_dict()) + return deidentified_info, reid + + +def dict_diff( + orig: ty.Mapping[str, ty.Any], new: ty.Mapping[str, ty.Any] +) -> dict[str, ty.Any]: + """Get a dict of all fields in orig that are not present or differ in new. + For nested dicts, the diff is applied recursively. + """ + result = {} + for k, v in orig.items(): + if k not in new: + result[k] = v + elif isinstance(v, dict) and isinstance(new[k], dict): + nested = dict_diff(v, new[k]) + if nested: + result[k] = nested + elif v != new[k]: + result[k] = v + return result diff --git a/fileformats/biosig/__init__.py b/fileformats/biosig/__init__.py index 98f48aa..cb2baa8 100644 --- a/fileformats/biosig/__init__.py +++ b/fileformats/biosig/__init__.py @@ -13,8 +13,6 @@ from .base import Biosig from .eeg import ( Eeg, - Fif, - FifGz, Edf, EdfPlus, BrainVisionHeader, @@ -27,6 +25,7 @@ CtfRes4, CtfInfo, Ctf, + Fif, KitMark, KitHeadPosition, KitSensorInfo, diff --git a/fileformats/biosig/eeg.py b/fileformats/biosig/eeg.py index 7f3fae3..e9f092c 100644 --- a/fileformats/biosig/eeg.py +++ b/fileformats/biosig/eeg.py @@ -13,7 +13,6 @@ from fileformats.core.exceptions import FormatMismatchError from fileformats.generic import BinaryFile, UnicodeFile from fileformats.core.mixin import WithMagicNumber, WithAdjacentFiles -from fileformats.application import Gzip from .base import Biosig @@ -24,26 +23,6 @@ class Eeg(Biosig): pass -# ------------------------------ -# Implementation of Specific EEG Formats -# ------------------------------ -class Fif(WithMagicNumber, BinaryFile, Biosig): - """ - MNE FIF format (standard format for NeuroMag/MEGIN MEG/EEG devices) - Most commonly used binary format, supports compression (.fif.gz) - """ - - ext = ".fif" - # FIF file magic number (hex identifier, from MNE official documentation) - magic_number: str | bytes = b"\x46\x49\x46\x32" # "FIF2" - - -class FifGz(Gzip[Fif], Fif): # type: ignore[type-arg, misc] - """Gzip-compressed MNE FIF format""" - - ext = ".fif.gz" - - class Edf(WithMagicNumber, BinaryFile, Eeg): """ EDF format EEG (European Data Format) — binary file with fixed-width ASCII @@ -116,7 +95,19 @@ def edf_type(self) -> str: return self._edf_type -class BrainVisionHeader(WithMagicNumber, UnicodeFile, Biosig): +class WithBrainVisionMagic: + """Checks the 'Brain Vision' magic string in text mode, handling optional BOM via utf-8-sig.""" + + @validated_property + def brain_vision_magic(self) -> None: + first_line = self.fspath.read_text(encoding="utf-8-sig").split("\n")[0] # type: ignore[attr-defined] + if not first_line.startswith("Brain Vision"): + raise FormatMismatchError( + f"File does not start with 'Brain Vision' magic string: {self}" + ) + + +class BrainVisionHeader(WithBrainVisionMagic, UnicodeFile, Biosig): """ BrainVision header file (.vhdr) — plain-text INI file describing channel configuration, sampling rate, amplifier settings, and references to the @@ -124,19 +115,15 @@ class BrainVisionHeader(WithMagicNumber, UnicodeFile, Biosig): """ ext = ".vhdr" - # First 12 bytes of "Brain Vision Data Exchange Header File Version 1.0\r\n" - magic_number = b"Brain Vision" -class BrainVisionMarker(WithMagicNumber, UnicodeFile, Biosig): +class BrainVisionMarker(WithBrainVisionMagic, UnicodeFile, Biosig): """ BrainVision marker file (.vmrk) — plain-text INI file containing event markers and annotations time-stamped to samples in the data file. """ ext = ".vmrk" - # First 12 bytes of "Brain Vision Data Exchange Marker File, Version 1.0\r\n" - magic_number = b"Brain Vision" class BrainVision(WithAdjacentFiles, BinaryFile, Biosig): diff --git a/fileformats/biosig/meg.py b/fileformats/biosig/meg.py index c4abda4..0184bb7 100644 --- a/fileformats/biosig/meg.py +++ b/fileformats/biosig/meg.py @@ -9,6 +9,7 @@ - miaocao@swin.edu.au """ +import struct from fileformats.core import validated_property from fileformats.core.mixin import WithAdjacentFiles, WithMagicNumber from fileformats.core.exceptions import FormatMismatchError @@ -28,6 +29,32 @@ class Meg(Biosig): """ +# ------------------------------ +# Implementation of Specific EEG Formats +# ------------------------------ +class Fif(BinaryFile, Meg): + """ + MNE FIF format (standard format for NeuroMag/MEGIN MEG/EEG devices) + Most commonly used binary format, supports compression (.fif.gz) + """ + + ext = ".fif" + + @validated_property + def fiff_header(self) -> None: + # FIFF files begin with a tag stream; the first tag must be FIFF_FILE_ID (kind=100) + # with data type FIFFT_ID_STRUCT (dtype=31), encoded as big-endian uint32 pairs. + data = self.read_contents(8) + if len(data) < 8: + raise FormatMismatchError(f"File too short to be a valid FIFF file: {self}") + kind, dtype = struct.unpack(">II", data) + if kind != 100 or dtype != 31: + raise FormatMismatchError( + f"First FIFF tag has kind={kind}, dtype={dtype}; expected kind=100 " + f"(FIFF_FILE_ID) and dtype=31 (FIFFT_ID_STRUCT) in {self}" + ) + + class CtfMeg4(WithMagicNumber, BinaryFile, Meg): """ CTF MEG4 binary data file (.meg4) — raw sensor data in CTF's proprietary format. @@ -47,7 +74,7 @@ class CtfRes4(WithMagicNumber, BinaryFile, Meg): ext = ".res4" # First 8 bytes: "MEG41RS\0" (CTF resource file version identifier) - magic_number = b"MEG41RS\x00" + magic_number = b"MEG42RS\x00" class CtfInfo(Xml, Meg): @@ -89,8 +116,7 @@ class Kit(WithAdjacentFiles, Meg, BinaryFile): KIT/RIKEN (Ricon) MEG format (directory-based) Required files: - Main data file (.sqd or .con) - - Marker file (.mrk) - Optional files: .elp (head position), .hsj (sensor info) + Optional files: .mrk (marker/coregistration), .elp (head position), .hsj (sensor info) """ ext = ".sqd" @@ -98,26 +124,18 @@ class Kit(WithAdjacentFiles, Meg, BinaryFile): marker_generic_names = ("marker.mrk", "markers.mrk", "kit.mrk") - # meg_chs = mne.pick_types(raw.info, meg=True, eeg=False) - - @validated_property - def mark_file(self) -> KitMark: - """ - Helper method: Find corresponding .mrk marker file for KIT/RIKEN data - Looks for same prefix with .mrk extension within the same directory - """ + @property + def mark_file(self) -> KitMark | None: try: mrk_path = self.select_by_ext(KitMark) + return KitMark(mrk_path) except FormatMismatchError: - for cand in self.marker_generic_names: - mrk_path = self.parent / cand - if mrk_path.exists(): - break - else: - raise FormatMismatchError( - f"No .mrk marker file found for KIT MEG data {self}\n" - ) - return KitMark(mrk_path) + pass + for cand in self.marker_generic_names: + mrk_path = self.parent / cand + if mrk_path.exists(): + return KitMark(mrk_path) + return None @property def head_position_file(self) -> KitHeadPosition | None: diff --git a/fileformats/biosig/tests/test_eeg.py b/fileformats/biosig/tests/test_eeg.py index 80fb8f5..5c98f23 100644 --- a/fileformats/biosig/tests/test_eeg.py +++ b/fileformats/biosig/tests/test_eeg.py @@ -14,26 +14,20 @@ BrainVision, BrainVisionHeader, BrainVisionMarker, - Edf, - Fif, + EdfPlus, ) # ------------------------------ # EEG: FIF # ------------------------------ - -def test_fif_instantiate(fif_path): - Fif(fif_path) - - # ------------------------------ # EEG: EDF # ------------------------------ -def test_edf_instantiate(edf_path): - Edf(edf_path) +def test_edf_plus_instantiate(edf_plus_path): + EdfPlus(edf_plus_path) # ------------------------------ diff --git a/fileformats/biosig/tests/test_meg.py b/fileformats/biosig/tests/test_meg.py index 9a02e01..465ad66 100644 --- a/fileformats/biosig/tests/test_meg.py +++ b/fileformats/biosig/tests/test_meg.py @@ -10,7 +10,7 @@ - miaocao@swin.edu.au """ -from fileformats.biosig import Ctf, CtfInfo, CtfMeg4, CtfRes4, Kit +from fileformats.biosig import Ctf, CtfInfo, CtfMeg4, CtfRes4, Kit, Fif # ------------------------------ # MEG: CTF @@ -21,6 +21,10 @@ def test_ctf_instantiate(ctf_ds_path): Ctf(ctf_ds_path) +def test_fif_instantiate(fif_path): + Fif(fif_path) + + def test_ctf_meg4_instantiate(ctf_ds_path): meg4_files = list(ctf_ds_path.glob("*.meg4")) assert meg4_files, "No .meg4 file found in CTF .ds directory" From 2eac26d22109f86cb463771c44bbe0261de755f6 Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Wed, 3 Jun 2026 21:07:03 +1000 Subject: [PATCH 04/14] fixing up ci-cd --- .github/workflows/ci-cd.yml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/.github/workflows/ci-cd.yml b/.github/workflows/ci-cd.yml index 8ffbf43..704f3bd 100644 --- a/.github/workflows/ci-cd.yml +++ b/.github/workflows/ci-cd.yml @@ -25,6 +25,7 @@ jobs: shell: bash -l {0} env: PIP_BREAK_SYSTEM_PACKAGES: 1 + MNE_DATA: ~/mne_data steps: - name: Checkout uses: actions/checkout@v2 @@ -55,6 +56,8 @@ jobs: key: mne-data-${{ hashFiles('**/requirements*.txt', '**/pyproject.toml') }} restore-keys: | mne-data- + - name: Download MNE datasets + run: python -c "import mne; mne.datasets.testing.data_path(); mne.datasets.sample.data_path(); mne.datasets.eegbci.load_data(subjects=1, runs=[1])" # - name: MyPy # run: mypy --install-types --non-interactive --no-warn-unused-ignores . - name: Pytest From 684e07fcc74364a8a6662c2b76bc731a11daeec6 Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Wed, 3 Jun 2026 21:14:24 +1000 Subject: [PATCH 05/14] fixing ci-cd --- .github/workflows/ci-cd.yml | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/.github/workflows/ci-cd.yml b/.github/workflows/ci-cd.yml index 704f3bd..104f221 100644 --- a/.github/workflows/ci-cd.yml +++ b/.github/workflows/ci-cd.yml @@ -57,7 +57,9 @@ jobs: restore-keys: | mne-data- - name: Download MNE datasets - run: python -c "import mne; mne.datasets.testing.data_path(); mne.datasets.sample.data_path(); mne.datasets.eegbci.load_data(subjects=1, runs=[1])" + run: | + mkdir $HOME/mne_data + python -c "import mne; mne.datasets.testing.data_path(); mne.datasets.sample.data_path(); mne.datasets.eegbci.load_data(subjects=1, runs=[1])" # - name: MyPy # run: mypy --install-types --non-interactive --no-warn-unused-ignores . - name: Pytest From 149c03c7b146b819d83b4513ba9c006bbabc2c91 Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Wed, 3 Jun 2026 21:29:59 +1000 Subject: [PATCH 06/14] fixing ci --- .github/workflows/ci-cd.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/ci-cd.yml b/.github/workflows/ci-cd.yml index 104f221..cf7486e 100644 --- a/.github/workflows/ci-cd.yml +++ b/.github/workflows/ci-cd.yml @@ -59,7 +59,7 @@ jobs: - name: Download MNE datasets run: | mkdir $HOME/mne_data - python -c "import mne; mne.datasets.testing.data_path(); mne.datasets.sample.data_path(); mne.datasets.eegbci.load_data(subjects=1, runs=[1])" + python -c "import mne; mne.datasets.testing.data_path(); mne.datasets.sample.data_path(); mne.datasets.eegbci.load_data(subjects=1, runs=[1], update_path=True)" # - name: MyPy # run: mypy --install-types --non-interactive --no-warn-unused-ignores . - name: Pytest From 9c6e2488681e64af8334cca5a172bc9af86122a3 Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Thu, 4 Jun 2026 11:50:54 +1000 Subject: [PATCH 07/14] fixing up cache --- .github/workflows/ci-cd.yml | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/.github/workflows/ci-cd.yml b/.github/workflows/ci-cd.yml index cf7486e..5d84e83 100644 --- a/.github/workflows/ci-cd.yml +++ b/.github/workflows/ci-cd.yml @@ -50,6 +50,7 @@ jobs: - name: Install Package run: python3 -m pip install -e .[test] -e ./extras[test] - name: Cache MNE datasets + id: cache-mne-data uses: actions/cache@v4 with: path: ~/mne_data @@ -57,8 +58,9 @@ jobs: restore-keys: | mne-data- - name: Download MNE datasets + if: steps.cache-mne-data.outputs.cache-hit != 'true' run: | - mkdir $HOME/mne_data + mkdir -p $HOME/mne_data python -c "import mne; mne.datasets.testing.data_path(); mne.datasets.sample.data_path(); mne.datasets.eegbci.load_data(subjects=1, runs=[1], update_path=True)" # - name: MyPy # run: mypy --install-types --non-interactive --no-warn-unused-ignores . From 915229bf556a5ebb45b7809b1a296b63065ee6d1 Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Thu, 4 Jun 2026 13:52:43 +1000 Subject: [PATCH 08/14] fixing up mne caching --- .github/workflows/ci-cd.yml | 2 -- 1 file changed, 2 deletions(-) diff --git a/.github/workflows/ci-cd.yml b/.github/workflows/ci-cd.yml index 5d84e83..6617c33 100644 --- a/.github/workflows/ci-cd.yml +++ b/.github/workflows/ci-cd.yml @@ -55,8 +55,6 @@ jobs: with: path: ~/mne_data key: mne-data-${{ hashFiles('**/requirements*.txt', '**/pyproject.toml') }} - restore-keys: | - mne-data- - name: Download MNE datasets if: steps.cache-mne-data.outputs.cache-hit != 'true' run: | From 99b6acadb18d5da6b47c92a6c508922844f7924d Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Thu, 4 Jun 2026 14:31:36 +1000 Subject: [PATCH 09/14] skip tests on eebci download failure in ci-cd --- .github/workflows/ci-cd.yml | 6 +++++- conftest.py | 5 ++++- 2 files changed, 9 insertions(+), 2 deletions(-) diff --git a/.github/workflows/ci-cd.yml b/.github/workflows/ci-cd.yml index 6617c33..c65ec9b 100644 --- a/.github/workflows/ci-cd.yml +++ b/.github/workflows/ci-cd.yml @@ -59,7 +59,11 @@ jobs: if: steps.cache-mne-data.outputs.cache-hit != 'true' run: | mkdir -p $HOME/mne_data - python -c "import mne; mne.datasets.testing.data_path(); mne.datasets.sample.data_path(); mne.datasets.eegbci.load_data(subjects=1, runs=[1], update_path=True)" + python -c "import mne; mne.datasets.testing.data_path(); mne.datasets.sample.data_path()" + - name: Download EEGBCI data + if: steps.cache-mne-data.outputs.cache-hit != 'true' + continue-on-error: true + run: python -c "import mne; mne.datasets.eegbci.load_data(subjects=1, runs=[1], update_path=True)" # - name: MyPy # run: mypy --install-types --non-interactive --no-warn-unused-ignores . - name: Pytest diff --git a/conftest.py b/conftest.py index e6259ac..7d59881 100644 --- a/conftest.py +++ b/conftest.py @@ -56,7 +56,10 @@ def fif_path(sample_data_path: Path) -> Path: @pytest.fixture(scope="session") def edf_plus_path() -> Path: - return Path(mne.datasets.eegbci.load_data(subjects=1, runs=[1])[0]) + try: + return Path(mne.datasets.eegbci.load_data(subjects=1, runs=[1])[0]) + except Exception as e: + pytest.skip(f"EEGBCI data unavailable (physionet.org unreachable?): {e}") @pytest.fixture(scope="session") From d18a1e83f0826c46ed594426b409c5a3c06b1daa Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Thu, 4 Jun 2026 18:43:46 +1000 Subject: [PATCH 10/14] debugging data download --- .github/workflows/ci-cd.yml | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/.github/workflows/ci-cd.yml b/.github/workflows/ci-cd.yml index c65ec9b..30a6c80 100644 --- a/.github/workflows/ci-cd.yml +++ b/.github/workflows/ci-cd.yml @@ -40,7 +40,7 @@ jobs: - name: Install system deps run: | sudo apt-get update -y - sudo apt-get install -y liblzma-dev + sudo apt-get install -y liblzma-dev tree - name: Set up Python ${{ matrix.python-version }} on ${{ matrix.os }} uses: actions/setup-python@v2 with: @@ -66,6 +66,8 @@ jobs: run: python -c "import mne; mne.datasets.eegbci.load_data(subjects=1, runs=[1], update_path=True)" # - name: MyPy # run: mypy --install-types --non-interactive --no-warn-unused-ignores . + - name: Contents of test data directory + run: tree ~/mne_data - name: Pytest run: pytest -vvs --cov fileformats --cov-config .coveragerc --cov-report xml . - name: Upload coverage to Codecov From d5963095f5dbe68337db9b597cad592ffcddd767 Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Thu, 4 Jun 2026 19:01:18 +1000 Subject: [PATCH 11/14] fixed up conftest fixture path --- conftest.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/conftest.py b/conftest.py index 7d59881..8d2c349 100644 --- a/conftest.py +++ b/conftest.py @@ -64,7 +64,7 @@ def edf_plus_path() -> Path: @pytest.fixture(scope="session") def bv_vhdr_path(testing_data_path: Path) -> Path: - return testing_data_path / "BrainVision" / "test_NO.vhdr" + return testing_data_path / "Brainvision" / "test_NO.vhdr" @pytest.fixture(scope="session") From 217b89ce1e44fb4addf6df97a9007f101def522c Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Thu, 4 Jun 2026 22:21:02 +1000 Subject: [PATCH 12/14] upped codecov version --- .github/workflows/ci-cd.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/ci-cd.yml b/.github/workflows/ci-cd.yml index 30a6c80..ac4ff34 100644 --- a/.github/workflows/ci-cd.yml +++ b/.github/workflows/ci-cd.yml @@ -71,7 +71,7 @@ jobs: - name: Pytest run: pytest -vvs --cov fileformats --cov-config .coveragerc --cov-report xml . - name: Upload coverage to Codecov - uses: codecov/codecov-action@v2 + uses: codecov/codecov-action@v5 with: fail_ci_if_error: true token: ${{ secrets.CODECOV_TOKEN }} From e2d9686f75bca29df31e2253e23f058441b574c2 Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Thu, 4 Jun 2026 22:27:32 +1000 Subject: [PATCH 13/14] updated artifacts versions --- .github/workflows/ci-cd.yml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/.github/workflows/ci-cd.yml b/.github/workflows/ci-cd.yml index ac4ff34..722e0f6 100644 --- a/.github/workflows/ci-cd.yml +++ b/.github/workflows/ci-cd.yml @@ -103,7 +103,7 @@ jobs: run: python3 -m build ${{ matrix.pkg[1] }} - name: Check distributions run: twine check ${{ matrix.pkg[1] }}/dist/* - - uses: actions/upload-artifact@v3 + - uses: actions/upload-artifact@v7 with: name: built-${{ matrix.pkg[0] }} path: ${{ matrix.pkg[1] }}/dist @@ -113,7 +113,7 @@ jobs: runs-on: ubuntu-latest steps: - name: Download build - uses: actions/download-artifact@v3 + uses: actions/download-artifact@v7 with: name: built-main path: dist @@ -135,7 +135,7 @@ jobs: runs-on: ubuntu-latest steps: - name: Download build - uses: actions/download-artifact@v3 + uses: actions/download-artifact@v7 with: name: built-extras path: dist From a9020066bb9f09fae100bf69625fc7c0bd17ebf6 Mon Sep 17 00:00:00 2001 From: "Thomas G. Close" Date: Thu, 4 Jun 2026 22:29:47 +1000 Subject: [PATCH 14/14] removed tree command in ci-cd --- .github/workflows/ci-cd.yml | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/.github/workflows/ci-cd.yml b/.github/workflows/ci-cd.yml index 722e0f6..198a129 100644 --- a/.github/workflows/ci-cd.yml +++ b/.github/workflows/ci-cd.yml @@ -40,7 +40,7 @@ jobs: - name: Install system deps run: | sudo apt-get update -y - sudo apt-get install -y liblzma-dev tree + sudo apt-get install -y liblzma-dev - name: Set up Python ${{ matrix.python-version }} on ${{ matrix.os }} uses: actions/setup-python@v2 with: @@ -66,8 +66,6 @@ jobs: run: python -c "import mne; mne.datasets.eegbci.load_data(subjects=1, runs=[1], update_path=True)" # - name: MyPy # run: mypy --install-types --non-interactive --no-warn-unused-ignores . - - name: Contents of test data directory - run: tree ~/mne_data - name: Pytest run: pytest -vvs --cov fileformats --cov-config .coveragerc --cov-report xml . - name: Upload coverage to Codecov