diff --git a/docs/changes/newsfragments/271.feature b/docs/changes/newsfragments/271.feature new file mode 100644 index 000000000..48c8aa888 --- /dev/null +++ b/docs/changes/newsfragments/271.feature @@ -0,0 +1 @@ +Introduce :class:`.DMCC13Benchmark` to access `DMCC13benchmark dataset `_ by `Synchon Mandal`_ diff --git a/junifer/datagrabber/__init__.py b/junifer/datagrabber/__init__.py index 61b9caedf..7986c3ea2 100644 --- a/junifer/datagrabber/__init__.py +++ b/junifer/datagrabber/__init__.py @@ -15,3 +15,4 @@ from .pattern_datalad import PatternDataladDataGrabber from .aomic import DataladAOMICID1000, DataladAOMICPIOP1, DataladAOMICPIOP2 from .hcp1200 import HCP1200, DataladHCP1200 from .multiple import MultipleDataGrabber +from .dmcc13_benchmark import DMCC13Benchmark diff --git a/junifer/datagrabber/dmcc13_benchmark.py b/junifer/datagrabber/dmcc13_benchmark.py new file mode 100644 index 000000000..92379aadf --- /dev/null +++ b/junifer/datagrabber/dmcc13_benchmark.py @@ -0,0 +1,329 @@ +"""Provide concrete implementation for DMCC13Benchmark DataGrabber.""" + +# Authors: Synchon Mandal +# License: AGPL + +from itertools import product +from pathlib import Path +from typing import Dict, List, Union + +from ..api.decorators import register_datagrabber +from ..utils import raise_error +from .pattern_datalad import PatternDataladDataGrabber + + +__all__ = ["DMCC13Benchmark"] + + +@register_datagrabber +class DMCC13Benchmark(PatternDataladDataGrabber): + """Concrete implementation for datalad-based data fetching of DMCC13. + + Parameters + ---------- + datadir : str or Path or None, optional + The directory where the datalad dataset will be cloned. If None, + the datalad dataset will be cloned into a temporary directory + (default None). + types: {"BOLD", "BOLD_confounds", "T1w", "probseg_CSF", "probseg_GM", \ + "probseg_WM"} or a list of the options, optional + DMCC data types. If None, all available data types are selected. + (default None). + sessions: {"wave1bas", "wave1pro", "wave1rea"} or list of the options, \ + optional + DMCC sessions. If None, all available sessions are selected + (default None). + tasks: {"Rest", "Axcpt", "Cuedts", "Stern", "Stroop"} or \ + list of the options, optional + DMCC task sessions. If None, all available task sessions are selected + (default None). + phase_encodings : {"AP", "PA"} or list of the options, optional + DMCC phase encoding directions. If None, all available phase encodings + are selected (default None). + runs : {"1", "2"} or list of the options, optional + DMCC runs. If None, all available runs are selected (default None). + native_t1w : bool, optional + Whether to use T1w in native space (default False). + + Raises + ------ + ValueError + If invalid value is passed for: + * ``sessions`` + * ``tasks`` + * ``phase_encodings`` + * ``runs`` + + """ + + def __init__( + self, + datadir: Union[str, Path, None] = None, + types: Union[str, List[str], None] = None, + sessions: Union[str, List[str], None] = None, + tasks: Union[str, List[str], None] = None, + phase_encodings: Union[str, List[str], None] = None, + runs: Union[str, List[str], None] = None, + native_t1w: bool = False, + ) -> None: + # Declare all sessions + all_sessions = [ + "wave1bas", + "wave1pro", + "wave1rea", + ] + # Set default sessions + if sessions is None: + sessions = all_sessions + else: + # Convert single session into list + if isinstance(sessions, str): + sessions = [sessions] + # Verify valid sessions + for s in sessions: + if s not in all_sessions: + raise_error( + f"{s} is not a valid session in the DMCC dataset" + ) + self.sessions = sessions + # Declare all tasks + all_tasks = [ + "Rest", + "Axcpt", + "Cuedts", + "Stern", + "Stroop", + ] + # Set default tasks + if tasks is None: + tasks = all_tasks + else: + # Convert single task into list + if isinstance(tasks, str): + tasks = [tasks] + # Verify valid tasks + for t in tasks: + if t not in all_tasks: + raise_error(f"{t} is not a valid task in the DMCC dataset") + self.tasks = tasks + # Declare all phase encodings + all_phase_encodings = ["AP", "PA"] + # Set default phase encodings + if phase_encodings is None: + phase_encodings = all_phase_encodings + else: + # Convert single phase encoding into list + if isinstance(phase_encodings, str): + phase_encodings = [phase_encodings] + # Verify valid phase encodings + for p in phase_encodings: + if p not in all_phase_encodings: + raise_error( + f"{p} is not a valid phase encoding in the DMCC " + "dataset" + ) + self.phase_encodings = phase_encodings + # Declare all runs + all_runs = ["1", "2"] + # Set default runs + if runs is None: + runs = all_runs + else: + # Convert single run into list + if isinstance(runs, str): + runs = [runs] + # Verify valid runs + for r in runs: + if r not in all_runs: + raise_error(f"{r} is not a valid run in the DMCC dataset") + self.runs = runs + # The patterns + patterns = { + "BOLD": ( + "derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/" + "func/sub-{subject}_ses-{session}_task-{task}_acq-mb4" + "{phase_encoding}_run-{run}_" + "space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz" + ), + "BOLD_confounds": ( + "derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/" + "func/sub-{subject}_ses-{session}_task-{task}_acq-mb4" + "{phase_encoding}_run-{run}_desc-confounds_regressors.tsv" + ), + "BOLD_mask": ( + "derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/" + "/func/sub-{subject}_ses-{session}_task-{task}_acq-mb4" + "{phase_encoding}_run-{run}_" + "space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz" + ), + "T1w": ( + "derivatives/fmriprep-1.3.2/sub-{subject}/anat/" + "sub-{subject}_space-MNI152NLin2009cAsym_desc-preproc_T1w.nii.gz" + ), + "T1w_mask": ( + "derivatives/fmriprep-1.3.2/sub-{subject}/anat/" + "sub-{subject}_space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz" + ), + "probseg_CSF": ( + "derivatives/fmriprep-1.3.2/sub-{subject}/anat/" + "sub-{subject}_space-MNI152NLin2009cAsym_label-CSF_probseg.nii.gz" + ), + "probseg_GM": ( + "derivatives/fmriprep-1.3.2/sub-{subject}/anat/" + "sub-{subject}_space-MNI152NLin2009cAsym_label-GM_probseg.nii.gz" + ), + "probseg_WM": ( + "derivatives/fmriprep-1.3.2/sub-{subject}/anat/" + "sub-{subject}_space-MNI152NLin2009cAsym_label-WM_probseg.nii.gz" + ), + } + # Use native T1w assets + self.native_t1w = False + if native_t1w: + self.native_t1w = True + patterns.update( + { + "T1w": ( + "derivatives/fmriprep-1.3.2/sub-{subject}/anat/" + "sub-{subject}_desc-preproc_T1w.nii.gz" + ), + "T1w_mask": ( + "derivatives/fmriprep-1.3.2/sub-{subject}/anat/" + "sub-{subject}_desc-brain_mask.nii.gz" + ), + "Warp": ( + "derivatives/fmriprep-1.3.2/sub-{subject}/anat/" + "sub-{subject}_from-MNI152NLin2009cAsym_to-T1w_" + "mode-image_xfm.h5" + ), + } + ) + # Set default types + if types is None: + types = list(patterns.keys()) + # Convert single type into list + else: + if not isinstance(types, list): + types = [types] + # The replacements + replacements = ["subject", "session", "task", "phase_encoding", "run"] + uri = "https://github.com/OpenNeuroDatasets/ds003452.git" + super().__init__( + types=types, + datadir=datadir, + uri=uri, + patterns=patterns, + replacements=replacements, + confounds_format="fmriprep", + ) + + def get_item( + self, + subject: str, + session: str, + task: str, + phase_encoding: str, + run: str, + ) -> Dict: + """Index one element in the dataset. + + Parameters + ---------- + subject : str + The subject ID. + session : {"wave1bas", "wave1pro", "wave1rea"} + The session to get. + task : {"Rest", "Axcpt", "Cuedts", "Stern", "Stroop"} + The task to get. + phase_encoding : {"AP", "PA"} + The phase encoding to get. + run : {"1", "2"} + The run to get. + + Returns + ------- + out : dict + Dictionary of paths for each type of data required for the + specified element. + + """ + # Format run + if phase_encoding == "AP": + run = "1" + else: + run = "2" + # Fetch item + out = super().get_item( + subject=subject, + session=session, + task=task, + phase_encoding=phase_encoding, + run=run, + ) + if out.get("BOLD"): + out["BOLD"]["mask_item"] = "BOLD_mask" + # Add space information + out["BOLD"].update({"space": "MNI152NLin2009cAsym"}) + if out.get("T1w"): + out["T1w"]["mask_item"] = "T1w_mask" + # Add space information + if self.native_t1w: + out["T1w"].update({"space": "native"}) + else: + out["T1w"].update({"space": "MNI152NLin2009cAsym"}) + return out + + def get_elements(self) -> List: + """Implement fetching list of subjects in the dataset. + + Returns + ------- + list of str + The list of subjects in the dataset. + + """ + subjects = [ + "f1031ax", + "f1552xo", + "f1659oa", + "f1670rz", + "f1951tt", + "f3300jh", + "f3720ca", + "f5004cr", + "f5407sl", + "f5416zj", + "F8113do", + "f8570ui", + "f9057kp", + ] + elems = [] + # For wave1bas session + for subject, session, task, phase_encoding in product( + subjects, + ["wave1bas"], + self.tasks, + self.phase_encodings, + ): + if phase_encoding == "AP": + run = "1" + else: + run = "2" + # Bypass for f1951tt not having run 2 for Rest + if subject == "f1951tt" and task == "Rest" and run == "2": + continue + elems.append((subject, session, task, phase_encoding, run)) + # For other sessions + for subject, session, task, phase_encoding in product( + subjects, + ["wave1pro", "wave1rea"], + ["Rest"], + self.phase_encodings, + ): + if phase_encoding == "AP": + run = "1" + else: + run = "2" + elems.append((subject, session, task, phase_encoding, run)) + + return elems diff --git a/junifer/datagrabber/tests/test_dmcc13_benchmark.py b/junifer/datagrabber/tests/test_dmcc13_benchmark.py new file mode 100644 index 000000000..9d7436b95 --- /dev/null +++ b/junifer/datagrabber/tests/test_dmcc13_benchmark.py @@ -0,0 +1,257 @@ +"""Provide tests for DMCC13Benchmark DataGrabber.""" + +# Authors: Synchon Mandal +# License: AGPL + +from typing import List, Optional, Union + +import pytest + +from junifer.datagrabber import DMCC13Benchmark + + +URI = "https://gin.g-node.org/synchon/datalad-example-dmcc13-benchmark" + + +@pytest.mark.parametrize( + "sessions, tasks, phase_encodings, runs, native_t1w", + [ + (None, None, None, None, False), + ("wave1bas", "Rest", "AP", "1", False), + ("wave1bas", "Axcpt", "AP", "1", False), + ("wave1bas", "Cuedts", "AP", "1", False), + ("wave1bas", "Stern", "AP", "1", False), + ("wave1bas", "Stroop", "AP", "1", False), + ("wave1bas", "Rest", "PA", "2", False), + ("wave1bas", "Axcpt", "PA", "2", False), + ("wave1bas", "Cuedts", "PA", "2", False), + ("wave1bas", "Stern", "PA", "2", False), + ("wave1bas", "Stroop", "PA", "2", False), + ("wave1bas", "Rest", "AP", "1", True), + ("wave1bas", "Axcpt", "AP", "1", True), + ("wave1bas", "Cuedts", "AP", "1", True), + ("wave1bas", "Stern", "AP", "1", True), + ("wave1bas", "Stroop", "AP", "1", True), + ("wave1bas", "Rest", "PA", "2", True), + ("wave1bas", "Axcpt", "PA", "2", True), + ("wave1bas", "Cuedts", "PA", "2", True), + ("wave1bas", "Stern", "PA", "2", True), + ("wave1bas", "Stroop", "PA", "2", True), + ("wave1pro", "Rest", "AP", "1", False), + ("wave1pro", "Rest", "PA", "2", False), + ("wave1pro", "Rest", "AP", "1", True), + ("wave1pro", "Rest", "PA", "2", True), + ("wave1rea", "Rest", "AP", "1", False), + ("wave1rea", "Rest", "PA", "2", False), + ("wave1rea", "Rest", "AP", "1", True), + ("wave1rea", "Rest", "PA", "2", True), + ], +) +def test_DMCC13Benchmark( + sessions: Optional[str], + tasks: Optional[str], + phase_encodings: Optional[str], + runs: Optional[str], + native_t1w: bool, +) -> None: + """Test DMCC13Benchmark DataGrabber. + + Parameters + ---------- + sessions : str or None + The parametrized session values. + tasks : str or None + The parametrized task values. + phase_encodings : str or None + The parametrized phase encoding values. + runs : str or None + The parametrized run values. + native_t1w : bool + The parametrized values for fetching native T1w. + + """ + dg = DMCC13Benchmark( + sessions=sessions, + tasks=tasks, + phase_encodings=phase_encodings, + runs=runs, + native_t1w=native_t1w, + ) + # Set URI to Gin + dg.uri = URI + + with dg: + # breakpoint() + # Get all elements + all_elements = dg.get_elements() + # Get test element + test_element = all_elements[0] + # Get test element's access values + _, ses, task, phase, run = test_element + # Access data + out = dg[("01", ses, task, phase, run)] + + # Available data types + data_types = [ + "BOLD", + "BOLD_confounds", + "BOLD_mask", + "probseg_CSF", + "probseg_GM", + "probseg_WM", + "T1w", + "T1w_mask", + ] + # Add Warp if native T1w is accessed + if native_t1w: + data_types.append("Warp") + + # Data type file name formats + data_file_names = [ + ( + f"sub-01_ses-{ses}_task-{task}_acq-mb4{phase}_run-{run}_" + "space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz" + ), + ( + f"sub-01_ses-{ses}_task-{task}_acq-mb4{phase}_run-{run}_" + "desc-confounds_regressors.tsv" + ), + ( + f"sub-01_ses-{ses}_task-{task}_acq-mb4{phase}_run-{run}_" + "space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz" + ), + "sub-01_space-MNI152NLin2009cAsym_label-CSF_probseg.nii.gz", + "sub-01_space-MNI152NLin2009cAsym_label-GM_probseg.nii.gz", + "sub-01_space-MNI152NLin2009cAsym_label-WM_probseg.nii.gz", + ] + if native_t1w: + data_file_names.extend( + [ + "sub-01_desc-preproc_T1w.nii.gz", + "sub-01_desc-brain_mask.nii.gz", + "sub-01_from-MNI152NLin2009cAsym_to-T1w_mode-image_xfm.h5", + ] + ) + else: + data_file_names.extend( + [ + "sub-01_space-MNI152NLin2009cAsym_desc-preproc_T1w.nii.gz", + "sub-01_space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz", + ] + ) + + for data_type, data_file_name in zip(data_types, data_file_names): + # Assert data type + assert data_type in out + # Assert data file path exists + assert out[data_type]["path"].exists() + # Assert data file path is a file + assert out[data_type]["path"].is_file() + # Assert data file name + assert out[data_type]["path"].name == data_file_name + # Assert metadata + assert "meta" in out[data_type] + + +@pytest.mark.parametrize( + "types, native_t1w", + [ + ("BOLD", True), + ("BOLD", False), + ("T1w", True), + ("T1w", False), + ("probseg_CSF", True), + ("probseg_CSF", False), + ("probseg_GM", True), + ("probseg_GM", False), + ("probseg_WM", True), + ("probseg_WM", False), + (["BOLD", "BOLD_confounds"], True), + (["BOLD", "BOLD_confounds"], False), + (["T1w", "probseg_CSF"], True), + (["T1w", "probseg_CSF"], False), + (["probseg_GM", "probseg_WM"], True), + (["probseg_GM", "probseg_WM"], False), + ], +) +def test_DMCC13Benchmark_partial_data_access( + types: Union[str, List[str]], + native_t1w: bool, +) -> None: + """Test DMCC13Benchmark DataGrabber partial data access. + + Parameters + ---------- + types : str or list of str + The parametrized types. + native_t1w : bool + The parametrized values for fetching native T1w. + + """ + dg = DMCC13Benchmark(types=types, native_t1w=native_t1w) + # Set URI to Gin + dg.uri = URI + + with dg: + # Get all elements + all_elements = dg.get_elements() + # Get test element + test_element = all_elements[0] + # Get test element's access values + _, ses, task, phase, run = test_element + # Access data + out = dg[("01", ses, task, phase, run)] + # Assert data type + if isinstance(types, list): + for type_ in types: + assert type_ in out + else: + assert types in out + + +def test_DMCC13Benchmark_incorrect_data_type() -> None: + """Test DMCC13Benchmark DataGrabber incorrect data type.""" + with pytest.raises( + ValueError, match="`patterns` must contain all `types`" + ): + _ = DMCC13Benchmark(types="Orcus") + + +def test_DMCC13Benchmark_invalid_sessions(): + """Test DMCC13Benchmark DataGrabber invalid sessions.""" + with pytest.raises( + ValueError, + match=("phonyses is not a valid session in " "the DMCC dataset"), + ): + DMCC13Benchmark(sessions="phonyses") + + +def test_DMCC13Benchmark_invalid_tasks(): + """Test DMCC13Benchmark DataGrabber invalid tasks.""" + with pytest.raises( + ValueError, + match=( + "thisisnotarealtask is not a valid task in " "the DMCC dataset" + ), + ): + DMCC13Benchmark(tasks="thisisnotarealtask") + + +def test_DMCC13Benchmark_phase_encodings(): + """Test DMCC13Benchmark DataGrabber invalid phase encodings.""" + with pytest.raises( + ValueError, + match=( + "moonphase is not a valid phase encoding in " "the DMCC dataset" + ), + ): + DMCC13Benchmark(phase_encodings="moonphase") + + +def test_DMCC13Benchmark_runs(): + """Test DMCC13Benchmark DataGrabber invalid runs.""" + with pytest.raises( + ValueError, + match=("cerebralrun is not a valid run in " "the DMCC dataset"), + ): + DMCC13Benchmark(runs="cerebralrun") diff --git a/tools/create_dmcc13_benchmark_example_dataset.py b/tools/create_dmcc13_benchmark_example_dataset.py new file mode 100644 index 000000000..8be13edb3 --- /dev/null +++ b/tools/create_dmcc13_benchmark_example_dataset.py @@ -0,0 +1,163 @@ +"""Script to generate example dataset for DMCC13Benchmark.""" + +# Authors: Synchon Mandal +# License: AGPL + +from itertools import product +from pathlib import Path +from tempfile import TemporaryDirectory + +from datalad import api as dl + + +# Set destination URL +DST = "git@gin.g-node.org:/synchon/datalad-example-dmcc13-benchmark.git" + + +if __name__ == "__main__": + with TemporaryDirectory() as tmpdir: + # Convert str to Path + tmpdir_path = Path(tmpdir) + # Set dataset directory + dataset_dir = tmpdir_path / "example_dmcc13_benchmark" + # Create new datalad dataset + dataset = dl.create(path=str(dataset_dir.resolve())) + + # Create base directory directory + basedir = dataset_dir / "derivatives" / "fmriprep-1.3.2" + basedir.mkdir(parents=True) + + # Generate subject directories + for sub in range(1, 10): + subdir = basedir / f"sub-{sub:02d}" + # Create subject directory + subdir.mkdir() + + # Sessions for functional data + sessions = [ + "wave1bas", + "wave1pro", + "wave1rea", + ] + + # Create data kind directories + for kind in ("anat", *[f"ses-{ses}" for ses in sessions]): + (subdir / kind).mkdir() + # Make functional data directories + if kind != "anat": + (subdir / kind / "func").mkdir() + + # Files to write, start with anatomical data + files = [ + ( + f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_desc-preproc" + "_T1w.nii.gz" + ), # T1w + ( + f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_desc-" + "brain_mask.nii.gz" + ), # T1w mask + ( + f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_" + "label-CSF_probseg.nii.gz" + ), # probseg CSF + ( + f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_" + "label-GM_probseg.nii.gz" + ), # probseg GM + ( + f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_" + "label-WM_probseg.nii.gz" + ), # probseg WM + ( + f"anat/sub-{sub:02d}_desc-preproc_T1w.nii.gz" + ), # T1w in native + ( + f"anat/sub-{sub:02d}_desc-brain_mask.nii.gz" + ), # T1w mask in native + ( + f"anat/sub-{sub:02d}_from-MNI152NLin2009cAsym_to-T1w_" + "mode-image_xfm.h5" + ), # Warp + ] + + # Tasks for functional data + tasks = [ + "Rest", + "Axcpt", + "Cuedts", + "Stern", + "Stroop", + ] + # Phase encodings for functional data + phase_encodings = ["AP", "PA"] + + # For wave1bas task + for ses, task, phase_encoding in product( + ["wave1bas"], + tasks, + phase_encodings, + ): + if phase_encoding == "AP": + run = "1" + else: + run = "2" + # BOLD + files.append( + f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4" + f"{phase_encoding}_run-{run}_" + "space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz" + ) + # BOLD confounds + files.append( + f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4" + f"{phase_encoding}_run-{run}_" + "desc-confounds_regressors.tsv" + ) + # BOLD mask + files.append( + f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4" + f"{phase_encoding}_run-{run}_" + "space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz" + ) + + # For wave1bas task + for ses, task, phase_encoding in product( + ["wave1pro", "wave1rea"], + ["Rest"], + phase_encodings, + ): + if phase_encoding == "AP": + run = "1" + else: + run = "2" + # BOLD + files.append( + f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4" + f"{phase_encoding}_run-{run}_" + "space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz" + ) + # BOLD confounds + files.append( + f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4" + f"{phase_encoding}_run-{run}_" + "desc-confounds_regressors.tsv" + ) + # BOLD mask + files.append( + f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4" + f"{phase_encoding}_run-{run}_" + "space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz" + ) + + # Create files + for file in files: + with open(subdir / file, "w") as f: + f.write("placeholder") + + # Save datalad dataset + dataset.save(recursive=True) + # Add datalad sibling + dataset.siblings(action="add", name="gin", url=DST) + # Push dataset to sibling + dataset.push(to="gin", force="all")