[ENH]: Add DMCC13Benchmark DataGrabber #271

Merged
synchon merged 8 commits from update/dmcc13benchmark-dg into main 2023-11-01 14:28:24 +00:00
5 changed files with 751 additions and 0 deletions

View file

@ -0,0 +1 @@
Introduce :class:`.DMCC13Benchmark` to access `DMCC13benchmark dataset <https://openneuro.org/datasets/ds003452/versions/1.0.1>`_ by `Synchon Mandal`_

View file

@ -15,3 +15,4 @@ from .pattern_datalad import PatternDataladDataGrabber
from .aomic import DataladAOMICID1000, DataladAOMICPIOP1, DataladAOMICPIOP2 from .aomic import DataladAOMICID1000, DataladAOMICPIOP1, DataladAOMICPIOP2
from .hcp1200 import HCP1200, DataladHCP1200 from .hcp1200 import HCP1200, DataladHCP1200
from .multiple import MultipleDataGrabber from .multiple import MultipleDataGrabber
from .dmcc13_benchmark import DMCC13Benchmark

View file

@ -0,0 +1,329 @@
"""Provide concrete implementation for DMCC13Benchmark DataGrabber."""
# Authors: Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL
from itertools import product
from pathlib import Path
from typing import Dict, List, Union
from ..api.decorators import register_datagrabber
from ..utils import raise_error
from .pattern_datalad import PatternDataladDataGrabber
__all__ = ["DMCC13Benchmark"]
@register_datagrabber
class DMCC13Benchmark(PatternDataladDataGrabber):
"""Concrete implementation for datalad-based data fetching of DMCC13.
Parameters
----------
datadir : str or Path or None, optional
The directory where the datalad dataset will be cloned. If None,
the datalad dataset will be cloned into a temporary directory
(default None).
types: {"BOLD", "BOLD_confounds", "T1w", "probseg_CSF", "probseg_GM", \
"probseg_WM"} or a list of the options, optional
DMCC data types. If None, all available data types are selected.
(default None).
sessions: {"wave1bas", "wave1pro", "wave1rea"} or list of the options, \
optional
DMCC sessions. If None, all available sessions are selected
(default None).
tasks: {"Rest", "Axcpt", "Cuedts", "Stern", "Stroop"} or \
list of the options, optional
DMCC task sessions. If None, all available task sessions are selected
(default None).
phase_encodings : {"AP", "PA"} or list of the options, optional
DMCC phase encoding directions. If None, all available phase encodings
are selected (default None).
runs : {"1", "2"} or list of the options, optional
DMCC runs. If None, all available runs are selected (default None).
native_t1w : bool, optional
Whether to use T1w in native space (default False).
Raises
------
ValueError
If invalid value is passed for:
* ``sessions``
* ``tasks``
* ``phase_encodings``
* ``runs``
"""
def __init__(
self,
datadir: Union[str, Path, None] = None,
types: Union[str, List[str], None] = None,
sessions: Union[str, List[str], None] = None,
tasks: Union[str, List[str], None] = None,
phase_encodings: Union[str, List[str], None] = None,
runs: Union[str, List[str], None] = None,
native_t1w: bool = False,
) -> None:
# Declare all sessions
all_sessions = [
"wave1bas",
"wave1pro",
"wave1rea",
]
# Set default sessions
if sessions is None:
sessions = all_sessions
else:
# Convert single session into list
if isinstance(sessions, str):
sessions = [sessions]
# Verify valid sessions
for s in sessions:
if s not in all_sessions:
raise_error(
f"{s} is not a valid session in the DMCC dataset"
)
self.sessions = sessions
# Declare all tasks
all_tasks = [
"Rest",
"Axcpt",
"Cuedts",
"Stern",
"Stroop",
]
# Set default tasks
if tasks is None:
tasks = all_tasks
else:
# Convert single task into list
if isinstance(tasks, str):
tasks = [tasks]
# Verify valid tasks
for t in tasks:
if t not in all_tasks:
raise_error(f"{t} is not a valid task in the DMCC dataset")
self.tasks = tasks
# Declare all phase encodings
all_phase_encodings = ["AP", "PA"]
# Set default phase encodings
if phase_encodings is None:
phase_encodings = all_phase_encodings
else:
# Convert single phase encoding into list
if isinstance(phase_encodings, str):
phase_encodings = [phase_encodings]
# Verify valid phase encodings
for p in phase_encodings:
if p not in all_phase_encodings:
raise_error(
f"{p} is not a valid phase encoding in the DMCC "
"dataset"
)
self.phase_encodings = phase_encodings
# Declare all runs
all_runs = ["1", "2"]
# Set default runs
if runs is None:
runs = all_runs
else:
# Convert single run into list
if isinstance(runs, str):
runs = [runs]
# Verify valid runs
for r in runs:
if r not in all_runs:
raise_error(f"{r} is not a valid run in the DMCC dataset")
self.runs = runs
# The patterns
patterns = {
"BOLD": (
"derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/"
"func/sub-{subject}_ses-{session}_task-{task}_acq-mb4"
"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"BOLD_confounds": (
"derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/"
"func/sub-{subject}_ses-{session}_task-{task}_acq-mb4"
"{phase_encoding}_run-{run}_desc-confounds_regressors.tsv"
),
"BOLD_mask": (
"derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/"
"/func/sub-{subject}_ses-{session}_task-{task}_acq-mb4"
"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"T1w": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_desc-preproc_T1w.nii.gz"
),
"T1w_mask": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"probseg_CSF": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-CSF_probseg.nii.gz"
),
"probseg_GM": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-GM_probseg.nii.gz"
),
"probseg_WM": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-WM_probseg.nii.gz"
),
}
# Use native T1w assets
self.native_t1w = False
if native_t1w:
self.native_t1w = True
patterns.update(
{
"T1w": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_desc-preproc_T1w.nii.gz"
),
"T1w_mask": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_desc-brain_mask.nii.gz"
),
"Warp": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_from-MNI152NLin2009cAsym_to-T1w_"
"mode-image_xfm.h5"
),
}
)
# Set default types
if types is None:
types = list(patterns.keys())
# Convert single type into list
else:
if not isinstance(types, list):
types = [types]
# The replacements
replacements = ["subject", "session", "task", "phase_encoding", "run"]
uri = "https://github.com/OpenNeuroDatasets/ds003452.git"
super().__init__(
types=types,
datadir=datadir,
uri=uri,
patterns=patterns,
replacements=replacements,
confounds_format="fmriprep",
)
def get_item(
self,
subject: str,
session: str,
task: str,
phase_encoding: str,
run: str,
) -> Dict:
"""Index one element in the dataset.
Parameters
----------
subject : str
The subject ID.
session : {"wave1bas", "wave1pro", "wave1rea"}
The session to get.
task : {"Rest", "Axcpt", "Cuedts", "Stern", "Stroop"}
The task to get.
phase_encoding : {"AP", "PA"}
The phase encoding to get.
run : {"1", "2"}
The run to get.
Returns
-------
out : dict
Dictionary of paths for each type of data required for the
specified element.
"""
# Format run
if phase_encoding == "AP":
run = "1"
else:
run = "2"
# Fetch item
out = super().get_item(
subject=subject,
session=session,
task=task,
phase_encoding=phase_encoding,
run=run,
)
if out.get("BOLD"):
out["BOLD"]["mask_item"] = "BOLD_mask"
# Add space information
out["BOLD"].update({"space": "MNI152NLin2009cAsym"})
if out.get("T1w"):
out["T1w"]["mask_item"] = "T1w_mask"
# Add space information
if self.native_t1w:
out["T1w"].update({"space": "native"})
else:
out["T1w"].update({"space": "MNI152NLin2009cAsym"})
return out
def get_elements(self) -> List:
"""Implement fetching list of subjects in the dataset.
Returns
-------
list of str
The list of subjects in the dataset.
"""
subjects = [
"f1031ax",
"f1552xo",
"f1659oa",
"f1670rz",
"f1951tt",
"f3300jh",
"f3720ca",
"f5004cr",
"f5407sl",
"f5416zj",
"F8113do",
"f8570ui",
"f9057kp",
]
elems = []
# For wave1bas session
for subject, session, task, phase_encoding in product(
subjects,
["wave1bas"],
self.tasks,
self.phase_encodings,
):
if phase_encoding == "AP":
run = "1"
else:
run = "2"
# Bypass for f1951tt not having run 2 for Rest
if subject == "f1951tt" and task == "Rest" and run == "2":
continue
elems.append((subject, session, task, phase_encoding, run))
# For other sessions
for subject, session, task, phase_encoding in product(
subjects,
["wave1pro", "wave1rea"],
["Rest"],
self.phase_encodings,
):
if phase_encoding == "AP":
run = "1"
else:
run = "2"
elems.append((subject, session, task, phase_encoding, run))
return elems

View file

@ -0,0 +1,257 @@
"""Provide tests for DMCC13Benchmark DataGrabber."""
# Authors: Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL
from typing import List, Optional, Union
import pytest
from junifer.datagrabber import DMCC13Benchmark
URI = "https://gin.g-node.org/synchon/datalad-example-dmcc13-benchmark"
@pytest.mark.parametrize(
"sessions, tasks, phase_encodings, runs, native_t1w",
[
(None, None, None, None, False),
("wave1bas", "Rest", "AP", "1", False),
("wave1bas", "Axcpt", "AP", "1", False),
("wave1bas", "Cuedts", "AP", "1", False),
("wave1bas", "Stern", "AP", "1", False),
("wave1bas", "Stroop", "AP", "1", False),
("wave1bas", "Rest", "PA", "2", False),
("wave1bas", "Axcpt", "PA", "2", False),
("wave1bas", "Cuedts", "PA", "2", False),
("wave1bas", "Stern", "PA", "2", False),
("wave1bas", "Stroop", "PA", "2", False),
("wave1bas", "Rest", "AP", "1", True),
("wave1bas", "Axcpt", "AP", "1", True),
("wave1bas", "Cuedts", "AP", "1", True),
("wave1bas", "Stern", "AP", "1", True),
("wave1bas", "Stroop", "AP", "1", True),
("wave1bas", "Rest", "PA", "2", True),
("wave1bas", "Axcpt", "PA", "2", True),
("wave1bas", "Cuedts", "PA", "2", True),
("wave1bas", "Stern", "PA", "2", True),
("wave1bas", "Stroop", "PA", "2", True),
("wave1pro", "Rest", "AP", "1", False),
("wave1pro", "Rest", "PA", "2", False),
("wave1pro", "Rest", "AP", "1", True),
("wave1pro", "Rest", "PA", "2", True),
("wave1rea", "Rest", "AP", "1", False),
("wave1rea", "Rest", "PA", "2", False),
("wave1rea", "Rest", "AP", "1", True),
("wave1rea", "Rest", "PA", "2", True),
],
)
def test_DMCC13Benchmark(
sessions: Optional[str],
tasks: Optional[str],
phase_encodings: Optional[str],
runs: Optional[str],
native_t1w: bool,
) -> None:
"""Test DMCC13Benchmark DataGrabber.
Parameters
----------
sessions : str or None
The parametrized session values.
tasks : str or None
The parametrized task values.
phase_encodings : str or None
The parametrized phase encoding values.
runs : str or None
The parametrized run values.
native_t1w : bool
The parametrized values for fetching native T1w.
"""
dg = DMCC13Benchmark(
sessions=sessions,
tasks=tasks,
phase_encodings=phase_encodings,
runs=runs,
native_t1w=native_t1w,
)
# Set URI to Gin
dg.uri = URI
with dg:
# breakpoint()
# Get all elements
all_elements = dg.get_elements()
# Get test element
test_element = all_elements[0]
# Get test element's access values
_, ses, task, phase, run = test_element
# Access data
out = dg[("01", ses, task, phase, run)]
# Available data types
data_types = [
"BOLD",
"BOLD_confounds",
"BOLD_mask",
"probseg_CSF",
"probseg_GM",
"probseg_WM",
"T1w",
"T1w_mask",
]
# Add Warp if native T1w is accessed
if native_t1w:
data_types.append("Warp")
# Data type file name formats
data_file_names = [
(
f"sub-01_ses-{ses}_task-{task}_acq-mb4{phase}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
(
f"sub-01_ses-{ses}_task-{task}_acq-mb4{phase}_run-{run}_"
"desc-confounds_regressors.tsv"
),
(
f"sub-01_ses-{ses}_task-{task}_acq-mb4{phase}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"sub-01_space-MNI152NLin2009cAsym_label-CSF_probseg.nii.gz",
"sub-01_space-MNI152NLin2009cAsym_label-GM_probseg.nii.gz",
"sub-01_space-MNI152NLin2009cAsym_label-WM_probseg.nii.gz",
]
if native_t1w:
data_file_names.extend(
[
"sub-01_desc-preproc_T1w.nii.gz",
"sub-01_desc-brain_mask.nii.gz",
"sub-01_from-MNI152NLin2009cAsym_to-T1w_mode-image_xfm.h5",
]
)
else:
data_file_names.extend(
[
"sub-01_space-MNI152NLin2009cAsym_desc-preproc_T1w.nii.gz",
"sub-01_space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz",
]
)
for data_type, data_file_name in zip(data_types, data_file_names):
# Assert data type
assert data_type in out
# Assert data file path exists
assert out[data_type]["path"].exists()
# Assert data file path is a file
assert out[data_type]["path"].is_file()
# Assert data file name
assert out[data_type]["path"].name == data_file_name
# Assert metadata
assert "meta" in out[data_type]
@pytest.mark.parametrize(
"types, native_t1w",
[
("BOLD", True),
("BOLD", False),
("T1w", True),
("T1w", False),
("probseg_CSF", True),
("probseg_CSF", False),
("probseg_GM", True),
("probseg_GM", False),
("probseg_WM", True),
("probseg_WM", False),
(["BOLD", "BOLD_confounds"], True),
(["BOLD", "BOLD_confounds"], False),
(["T1w", "probseg_CSF"], True),
(["T1w", "probseg_CSF"], False),
(["probseg_GM", "probseg_WM"], True),
(["probseg_GM", "probseg_WM"], False),
],
)
def test_DMCC13Benchmark_partial_data_access(
types: Union[str, List[str]],
native_t1w: bool,
) -> None:
"""Test DMCC13Benchmark DataGrabber partial data access.
Parameters
----------
types : str or list of str
The parametrized types.
native_t1w : bool
The parametrized values for fetching native T1w.
"""
dg = DMCC13Benchmark(types=types, native_t1w=native_t1w)
# Set URI to Gin
dg.uri = URI
with dg:
# Get all elements
all_elements = dg.get_elements()
# Get test element
test_element = all_elements[0]
# Get test element's access values
_, ses, task, phase, run = test_element
# Access data
out = dg[("01", ses, task, phase, run)]
# Assert data type
if isinstance(types, list):
for type_ in types:
assert type_ in out
else:
assert types in out
def test_DMCC13Benchmark_incorrect_data_type() -> None:
"""Test DMCC13Benchmark DataGrabber incorrect data type."""
with pytest.raises(
ValueError, match="`patterns` must contain all `types`"
):
_ = DMCC13Benchmark(types="Orcus")
def test_DMCC13Benchmark_invalid_sessions():
"""Test DMCC13Benchmark DataGrabber invalid sessions."""
with pytest.raises(
ValueError,
match=("phonyses is not a valid session in " "the DMCC dataset"),
):
DMCC13Benchmark(sessions="phonyses")
def test_DMCC13Benchmark_invalid_tasks():
"""Test DMCC13Benchmark DataGrabber invalid tasks."""
with pytest.raises(
ValueError,
match=(
"thisisnotarealtask is not a valid task in " "the DMCC dataset"
),
):
DMCC13Benchmark(tasks="thisisnotarealtask")
def test_DMCC13Benchmark_phase_encodings():
"""Test DMCC13Benchmark DataGrabber invalid phase encodings."""
with pytest.raises(
ValueError,
match=(
"moonphase is not a valid phase encoding in " "the DMCC dataset"
),
):
DMCC13Benchmark(phase_encodings="moonphase")
def test_DMCC13Benchmark_runs():
"""Test DMCC13Benchmark DataGrabber invalid runs."""
with pytest.raises(
ValueError,
match=("cerebralrun is not a valid run in " "the DMCC dataset"),
):
DMCC13Benchmark(runs="cerebralrun")

View file

@ -0,0 +1,163 @@
"""Script to generate example dataset for DMCC13Benchmark."""
# Authors: Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL
from itertools import product
from pathlib import Path
from tempfile import TemporaryDirectory
from datalad import api as dl
# Set destination URL
DST = "git@gin.g-node.org:/synchon/datalad-example-dmcc13-benchmark.git"
if __name__ == "__main__":
with TemporaryDirectory() as tmpdir:
# Convert str to Path
tmpdir_path = Path(tmpdir)
# Set dataset directory
dataset_dir = tmpdir_path / "example_dmcc13_benchmark"
# Create new datalad dataset
dataset = dl.create(path=str(dataset_dir.resolve()))
# Create base directory directory
basedir = dataset_dir / "derivatives" / "fmriprep-1.3.2"
basedir.mkdir(parents=True)
# Generate subject directories
for sub in range(1, 10):
subdir = basedir / f"sub-{sub:02d}"
# Create subject directory
subdir.mkdir()
# Sessions for functional data
sessions = [
"wave1bas",
"wave1pro",
"wave1rea",
]
# Create data kind directories
for kind in ("anat", *[f"ses-{ses}" for ses in sessions]):
(subdir / kind).mkdir()
# Make functional data directories
if kind != "anat":
(subdir / kind / "func").mkdir()
# Files to write, start with anatomical data
files = [
(
f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_desc-preproc"
"_T1w.nii.gz"
), # T1w
(
f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_desc-"
"brain_mask.nii.gz"
), # T1w mask
(
f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_"
"label-CSF_probseg.nii.gz"
), # probseg CSF
(
f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_"
"label-GM_probseg.nii.gz"
), # probseg GM
(
f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_"
"label-WM_probseg.nii.gz"
), # probseg WM
(
f"anat/sub-{sub:02d}_desc-preproc_T1w.nii.gz"
), # T1w in native
(
f"anat/sub-{sub:02d}_desc-brain_mask.nii.gz"
), # T1w mask in native
(
f"anat/sub-{sub:02d}_from-MNI152NLin2009cAsym_to-T1w_"
"mode-image_xfm.h5"
), # Warp
]
# Tasks for functional data
tasks = [
"Rest",
"Axcpt",
"Cuedts",
"Stern",
"Stroop",
]
# Phase encodings for functional data
phase_encodings = ["AP", "PA"]
# For wave1bas task
for ses, task, phase_encoding in product(
["wave1bas"],
tasks,
phase_encodings,
):
if phase_encoding == "AP":
run = "1"
else:
run = "2"
# BOLD
files.append(
f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4"
f"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
)
# BOLD confounds
files.append(
f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4"
f"{phase_encoding}_run-{run}_"
"desc-confounds_regressors.tsv"
)
# BOLD mask
files.append(
f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4"
f"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
)
# For wave1bas task
for ses, task, phase_encoding in product(
["wave1pro", "wave1rea"],
["Rest"],
phase_encodings,
):
if phase_encoding == "AP":
run = "1"
else:
run = "2"
# BOLD
files.append(
f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4"
f"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
)
# BOLD confounds
files.append(
f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4"
f"{phase_encoding}_run-{run}_"
"desc-confounds_regressors.tsv"
)
# BOLD mask
files.append(
f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4"
f"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
)
# Create files
for file in files:
with open(subdir / file, "w") as f:
f.write("placeholder")
# Save datalad dataset
dataset.save(recursive=True)
# Add datalad sibling
dataset.siblings(action="add", name="gin", url=DST)
# Push dataset to sibling
dataset.push(to="gin", force="all")