[ENH]: Add DMCC13Benchmark DataGrabber #271

Merged
synchon merged 8 commits from update/dmcc13benchmark-dg into main 2023-11-01 14:28:24 +00:00
5 changed files with 751 additions and 0 deletions

View file

@ -0,0 +1 @@
Introduce :class:`.DMCC13Benchmark` to access `DMCC13benchmark dataset <https://openneuro.org/datasets/ds003452/versions/1.0.1>`_ by `Synchon Mandal`_

View file

@ -15,3 +15,4 @@ from .pattern_datalad import PatternDataladDataGrabber
from .aomic import DataladAOMICID1000, DataladAOMICPIOP1, DataladAOMICPIOP2
from .hcp1200 import HCP1200, DataladHCP1200
from .multiple import MultipleDataGrabber
from .dmcc13_benchmark import DMCC13Benchmark

View file

@ -0,0 +1,329 @@
"""Provide concrete implementation for DMCC13Benchmark DataGrabber."""
# Authors: Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL
from itertools import product
from pathlib import Path
from typing import Dict, List, Union
from ..api.decorators import register_datagrabber
from ..utils import raise_error
from .pattern_datalad import PatternDataladDataGrabber
__all__ = ["DMCC13Benchmark"]
@register_datagrabber
class DMCC13Benchmark(PatternDataladDataGrabber):
"""Concrete implementation for datalad-based data fetching of DMCC13.
Parameters
----------
datadir : str or Path or None, optional
The directory where the datalad dataset will be cloned. If None,
the datalad dataset will be cloned into a temporary directory
(default None).
types: {"BOLD", "BOLD_confounds", "T1w", "probseg_CSF", "probseg_GM", \
"probseg_WM"} or a list of the options, optional
DMCC data types. If None, all available data types are selected.
(default None).
sessions: {"wave1bas", "wave1pro", "wave1rea"} or list of the options, \
optional
DMCC sessions. If None, all available sessions are selected
(default None).
tasks: {"Rest", "Axcpt", "Cuedts", "Stern", "Stroop"} or \
list of the options, optional
DMCC task sessions. If None, all available task sessions are selected
(default None).
phase_encodings : {"AP", "PA"} or list of the options, optional
DMCC phase encoding directions. If None, all available phase encodings
are selected (default None).
runs : {"1", "2"} or list of the options, optional
DMCC runs. If None, all available runs are selected (default None).
native_t1w : bool, optional
Whether to use T1w in native space (default False).
Raises
------
ValueError
If invalid value is passed for:
* ``sessions``
* ``tasks``
* ``phase_encodings``
* ``runs``
"""
def __init__(
self,
datadir: Union[str, Path, None] = None,
types: Union[str, List[str], None] = None,
sessions: Union[str, List[str], None] = None,
tasks: Union[str, List[str], None] = None,
phase_encodings: Union[str, List[str], None] = None,
runs: Union[str, List[str], None] = None,
native_t1w: bool = False,
) -> None:
# Declare all sessions
all_sessions = [
"wave1bas",
"wave1pro",
"wave1rea",
]
# Set default sessions
if sessions is None:
sessions = all_sessions
else:
# Convert single session into list
if isinstance(sessions, str):
sessions = [sessions]
# Verify valid sessions
for s in sessions:
if s not in all_sessions:
raise_error(
f"{s} is not a valid session in the DMCC dataset"
)
self.sessions = sessions
# Declare all tasks
all_tasks = [
"Rest",
"Axcpt",
"Cuedts",
"Stern",
"Stroop",
]
# Set default tasks
if tasks is None:
tasks = all_tasks
else:
# Convert single task into list
if isinstance(tasks, str):
tasks = [tasks]
# Verify valid tasks
for t in tasks:
if t not in all_tasks:
raise_error(f"{t} is not a valid task in the DMCC dataset")
self.tasks = tasks
# Declare all phase encodings
all_phase_encodings = ["AP", "PA"]
# Set default phase encodings
if phase_encodings is None:
phase_encodings = all_phase_encodings
else:
# Convert single phase encoding into list
if isinstance(phase_encodings, str):
phase_encodings = [phase_encodings]
# Verify valid phase encodings
for p in phase_encodings:
if p not in all_phase_encodings:
raise_error(
f"{p} is not a valid phase encoding in the DMCC "
"dataset"
)
self.phase_encodings = phase_encodings
# Declare all runs
all_runs = ["1", "2"]
# Set default runs
if runs is None:
runs = all_runs
else:
# Convert single run into list
if isinstance(runs, str):
runs = [runs]
# Verify valid runs
for r in runs:
if r not in all_runs:
raise_error(f"{r} is not a valid run in the DMCC dataset")
self.runs = runs
# The patterns
patterns = {
"BOLD": (
"derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/"
"func/sub-{subject}_ses-{session}_task-{task}_acq-mb4"
"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"BOLD_confounds": (
"derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/"
"func/sub-{subject}_ses-{session}_task-{task}_acq-mb4"
"{phase_encoding}_run-{run}_desc-confounds_regressors.tsv"
),
"BOLD_mask": (
"derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/"
"/func/sub-{subject}_ses-{session}_task-{task}_acq-mb4"
"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"T1w": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_desc-preproc_T1w.nii.gz"
),
"T1w_mask": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"probseg_CSF": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-CSF_probseg.nii.gz"
),
"probseg_GM": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-GM_probseg.nii.gz"
),
"probseg_WM": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-WM_probseg.nii.gz"
),
}
# Use native T1w assets
self.native_t1w = False
if native_t1w:
self.native_t1w = True
patterns.update(
{
"T1w": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_desc-preproc_T1w.nii.gz"
),
"T1w_mask": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_desc-brain_mask.nii.gz"
),
"Warp": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_from-MNI152NLin2009cAsym_to-T1w_"
"mode-image_xfm.h5"
),
}
)
# Set default types
if types is None:
types = list(patterns.keys())
# Convert single type into list
else:
if not isinstance(types, list):
types = [types]
# The replacements
replacements = ["subject", "session", "task", "phase_encoding", "run"]
uri = "https://github.com/OpenNeuroDatasets/ds003452.git"
super().__init__(
types=types,
datadir=datadir,
uri=uri,
patterns=patterns,
replacements=replacements,
confounds_format="fmriprep",
)
def get_item(
self,
subject: str,
session: str,
task: str,
phase_encoding: str,
run: str,
) -> Dict:
"""Index one element in the dataset.
Parameters
----------
subject : str
The subject ID.
session : {"wave1bas", "wave1pro", "wave1rea"}
The session to get.
task : {"Rest", "Axcpt", "Cuedts", "Stern", "Stroop"}
The task to get.
phase_encoding : {"AP", "PA"}
The phase encoding to get.
run : {"1", "2"}
The run to get.
Returns
-------
out : dict
Dictionary of paths for each type of data required for the
specified element.
"""
# Format run
if phase_encoding == "AP":
run = "1"
else:
run = "2"
# Fetch item
out = super().get_item(
subject=subject,
session=session,
task=task,
phase_encoding=phase_encoding,
run=run,
)
if out.get("BOLD"):
out["BOLD"]["mask_item"] = "BOLD_mask"
# Add space information
out["BOLD"].update({"space": "MNI152NLin2009cAsym"})
if out.get("T1w"):
out["T1w"]["mask_item"] = "T1w_mask"
# Add space information
if self.native_t1w:
out["T1w"].update({"space": "native"})
else:
out["T1w"].update({"space": "MNI152NLin2009cAsym"})
return out
def get_elements(self) -> List:
"""Implement fetching list of subjects in the dataset.
Returns
-------
list of str
The list of subjects in the dataset.
"""
subjects = [
"f1031ax",
"f1552xo",
"f1659oa",
"f1670rz",
"f1951tt",
"f3300jh",
"f3720ca",
"f5004cr",
"f5407sl",
"f5416zj",
"F8113do",
"f8570ui",
"f9057kp",
]
elems = []
# For wave1bas session
for subject, session, task, phase_encoding in product(
subjects,
["wave1bas"],
self.tasks,
self.phase_encodings,
):
if phase_encoding == "AP":
run = "1"
else:
run = "2"
# Bypass for f1951tt not having run 2 for Rest
if subject == "f1951tt" and task == "Rest" and run == "2":
continue
elems.append((subject, session, task, phase_encoding, run))
# For other sessions
for subject, session, task, phase_encoding in product(
subjects,
["wave1pro", "wave1rea"],
["Rest"],
self.phase_encodings,
):
if phase_encoding == "AP":
run = "1"
else:
run = "2"
elems.append((subject, session, task, phase_encoding, run))
return elems

View file

@ -0,0 +1,257 @@
"""Provide tests for DMCC13Benchmark DataGrabber."""
# Authors: Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL
from typing import List, Optional, Union
import pytest
from junifer.datagrabber import DMCC13Benchmark
URI = "https://gin.g-node.org/synchon/datalad-example-dmcc13-benchmark"
@pytest.mark.parametrize(
"sessions, tasks, phase_encodings, runs, native_t1w",
[
(None, None, None, None, False),
("wave1bas", "Rest", "AP", "1", False),
("wave1bas", "Axcpt", "AP", "1", False),
("wave1bas", "Cuedts", "AP", "1", False),
("wave1bas", "Stern", "AP", "1", False),
("wave1bas", "Stroop", "AP", "1", False),
("wave1bas", "Rest", "PA", "2", False),
("wave1bas", "Axcpt", "PA", "2", False),
("wave1bas", "Cuedts", "PA", "2", False),
("wave1bas", "Stern", "PA", "2", False),
("wave1bas", "Stroop", "PA", "2", False),
("wave1bas", "Rest", "AP", "1", True),
("wave1bas", "Axcpt", "AP", "1", True),
("wave1bas", "Cuedts", "AP", "1", True),
("wave1bas", "Stern", "AP", "1", True),
("wave1bas", "Stroop", "AP", "1", True),
("wave1bas", "Rest", "PA", "2", True),
("wave1bas", "Axcpt", "PA", "2", True),
("wave1bas", "Cuedts", "PA", "2", True),
("wave1bas", "Stern", "PA", "2", True),
("wave1bas", "Stroop", "PA", "2", True),
("wave1pro", "Rest", "AP", "1", False),
("wave1pro", "Rest", "PA", "2", False),
("wave1pro", "Rest", "AP", "1", True),
("wave1pro", "Rest", "PA", "2", True),
("wave1rea", "Rest", "AP", "1", False),
("wave1rea", "Rest", "PA", "2", False),
("wave1rea", "Rest", "AP", "1", True),
("wave1rea", "Rest", "PA", "2", True),
],
)
def test_DMCC13Benchmark(
sessions: Optional[str],
tasks: Optional[str],
phase_encodings: Optional[str],
runs: Optional[str],
native_t1w: bool,
) -> None:
"""Test DMCC13Benchmark DataGrabber.
Parameters
----------
sessions : str or None
The parametrized session values.
tasks : str or None
The parametrized task values.
phase_encodings : str or None
The parametrized phase encoding values.
runs : str or None
The parametrized run values.
native_t1w : bool
The parametrized values for fetching native T1w.
"""
dg = DMCC13Benchmark(
sessions=sessions,
tasks=tasks,
phase_encodings=phase_encodings,
runs=runs,
native_t1w=native_t1w,
)
# Set URI to Gin
dg.uri = URI
with dg:
# breakpoint()
# Get all elements
all_elements = dg.get_elements()
# Get test element
test_element = all_elements[0]
# Get test element's access values
_, ses, task, phase, run = test_element
# Access data
out = dg[("01", ses, task, phase, run)]
# Available data types
data_types = [
"BOLD",
"BOLD_confounds",
"BOLD_mask",
"probseg_CSF",
"probseg_GM",
"probseg_WM",
"T1w",
"T1w_mask",
]
# Add Warp if native T1w is accessed
if native_t1w:
data_types.append("Warp")
# Data type file name formats
data_file_names = [
(
f"sub-01_ses-{ses}_task-{task}_acq-mb4{phase}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
(
f"sub-01_ses-{ses}_task-{task}_acq-mb4{phase}_run-{run}_"
"desc-confounds_regressors.tsv"
),
(
f"sub-01_ses-{ses}_task-{task}_acq-mb4{phase}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"sub-01_space-MNI152NLin2009cAsym_label-CSF_probseg.nii.gz",
"sub-01_space-MNI152NLin2009cAsym_label-GM_probseg.nii.gz",
"sub-01_space-MNI152NLin2009cAsym_label-WM_probseg.nii.gz",
]
if native_t1w:
data_file_names.extend(
[
"sub-01_desc-preproc_T1w.nii.gz",
"sub-01_desc-brain_mask.nii.gz",
"sub-01_from-MNI152NLin2009cAsym_to-T1w_mode-image_xfm.h5",
]
)
else:
data_file_names.extend(
[
"sub-01_space-MNI152NLin2009cAsym_desc-preproc_T1w.nii.gz",
"sub-01_space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz",
]
)
for data_type, data_file_name in zip(data_types, data_file_names):
# Assert data type
assert data_type in out
# Assert data file path exists
assert out[data_type]["path"].exists()
# Assert data file path is a file
assert out[data_type]["path"].is_file()
# Assert data file name
assert out[data_type]["path"].name == data_file_name
# Assert metadata
assert "meta" in out[data_type]
@pytest.mark.parametrize(
"types, native_t1w",
[
("BOLD", True),
("BOLD", False),
("T1w", True),
("T1w", False),
("probseg_CSF", True),
("probseg_CSF", False),
("probseg_GM", True),
("probseg_GM", False),
("probseg_WM", True),
("probseg_WM", False),
(["BOLD", "BOLD_confounds"], True),
(["BOLD", "BOLD_confounds"], False),
(["T1w", "probseg_CSF"], True),
(["T1w", "probseg_CSF"], False),
(["probseg_GM", "probseg_WM"], True),
(["probseg_GM", "probseg_WM"], False),
],
)
def test_DMCC13Benchmark_partial_data_access(
types: Union[str, List[str]],
native_t1w: bool,
) -> None:
"""Test DMCC13Benchmark DataGrabber partial data access.
Parameters
----------
types : str or list of str
The parametrized types.
native_t1w : bool
The parametrized values for fetching native T1w.
"""
dg = DMCC13Benchmark(types=types, native_t1w=native_t1w)
# Set URI to Gin
dg.uri = URI
with dg:
# Get all elements
all_elements = dg.get_elements()
# Get test element
test_element = all_elements[0]
# Get test element's access values
_, ses, task, phase, run = test_element
# Access data
out = dg[("01", ses, task, phase, run)]
# Assert data type
if isinstance(types, list):
for type_ in types:
assert type_ in out
else:
assert types in out
def test_DMCC13Benchmark_incorrect_data_type() -> None:
"""Test DMCC13Benchmark DataGrabber incorrect data type."""
with pytest.raises(
ValueError, match="`patterns` must contain all `types`"
):
_ = DMCC13Benchmark(types="Orcus")
def test_DMCC13Benchmark_invalid_sessions():
"""Test DMCC13Benchmark DataGrabber invalid sessions."""
with pytest.raises(
ValueError,
match=("phonyses is not a valid session in " "the DMCC dataset"),
):
DMCC13Benchmark(sessions="phonyses")
def test_DMCC13Benchmark_invalid_tasks():
"""Test DMCC13Benchmark DataGrabber invalid tasks."""
with pytest.raises(
ValueError,
match=(
"thisisnotarealtask is not a valid task in " "the DMCC dataset"
),
):
DMCC13Benchmark(tasks="thisisnotarealtask")
def test_DMCC13Benchmark_phase_encodings():
"""Test DMCC13Benchmark DataGrabber invalid phase encodings."""
with pytest.raises(
ValueError,
match=(
"moonphase is not a valid phase encoding in " "the DMCC dataset"
),
):
DMCC13Benchmark(phase_encodings="moonphase")
def test_DMCC13Benchmark_runs():
"""Test DMCC13Benchmark DataGrabber invalid runs."""
with pytest.raises(
ValueError,
match=("cerebralrun is not a valid run in " "the DMCC dataset"),
):
DMCC13Benchmark(runs="cerebralrun")

View file

@ -0,0 +1,163 @@
"""Script to generate example dataset for DMCC13Benchmark."""
# Authors: Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL
from itertools import product
from pathlib import Path
from tempfile import TemporaryDirectory
from datalad import api as dl
# Set destination URL
DST = "git@gin.g-node.org:/synchon/datalad-example-dmcc13-benchmark.git"
if __name__ == "__main__":
with TemporaryDirectory() as tmpdir:
# Convert str to Path
tmpdir_path = Path(tmpdir)
# Set dataset directory
dataset_dir = tmpdir_path / "example_dmcc13_benchmark"
# Create new datalad dataset
dataset = dl.create(path=str(dataset_dir.resolve()))
# Create base directory directory
basedir = dataset_dir / "derivatives" / "fmriprep-1.3.2"
basedir.mkdir(parents=True)
# Generate subject directories
for sub in range(1, 10):
subdir = basedir / f"sub-{sub:02d}"
# Create subject directory
subdir.mkdir()
# Sessions for functional data
sessions = [
"wave1bas",
"wave1pro",
"wave1rea",
]
# Create data kind directories
for kind in ("anat", *[f"ses-{ses}" for ses in sessions]):
(subdir / kind).mkdir()
# Make functional data directories
if kind != "anat":
(subdir / kind / "func").mkdir()
# Files to write, start with anatomical data
files = [
(
f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_desc-preproc"
"_T1w.nii.gz"
), # T1w
(
f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_desc-"
"brain_mask.nii.gz"
), # T1w mask
(
f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_"
"label-CSF_probseg.nii.gz"
), # probseg CSF
(
f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_"
"label-GM_probseg.nii.gz"
), # probseg GM
(
f"anat/sub-{sub:02d}_space-MNI152NLin2009cAsym_"
"label-WM_probseg.nii.gz"
), # probseg WM
(
f"anat/sub-{sub:02d}_desc-preproc_T1w.nii.gz"
), # T1w in native
(
f"anat/sub-{sub:02d}_desc-brain_mask.nii.gz"
), # T1w mask in native
(
f"anat/sub-{sub:02d}_from-MNI152NLin2009cAsym_to-T1w_"
"mode-image_xfm.h5"
), # Warp
]
# Tasks for functional data
tasks = [
"Rest",
"Axcpt",
"Cuedts",
"Stern",
"Stroop",
]
# Phase encodings for functional data
phase_encodings = ["AP", "PA"]
# For wave1bas task
for ses, task, phase_encoding in product(
["wave1bas"],
tasks,
phase_encodings,
):
if phase_encoding == "AP":
run = "1"
else:
run = "2"
# BOLD
files.append(
f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4"
f"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
)
# BOLD confounds
files.append(
f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4"
f"{phase_encoding}_run-{run}_"
"desc-confounds_regressors.tsv"
)
# BOLD mask
files.append(
f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4"
f"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
)
# For wave1bas task
for ses, task, phase_encoding in product(
["wave1pro", "wave1rea"],
["Rest"],
phase_encodings,
):
if phase_encoding == "AP":
run = "1"
else:
run = "2"
# BOLD
files.append(
f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4"
f"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
)
# BOLD confounds
files.append(
f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4"
f"{phase_encoding}_run-{run}_"
"desc-confounds_regressors.tsv"
)
# BOLD mask
files.append(
f"ses-{ses}/func/sub-{sub:02d}_ses-{ses}_task-{task}_acq-mb4"
f"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
)
# Create files
for file in files:
with open(subdir / file, "w") as f:
f.write("placeholder")
# Save datalad dataset
dataset.save(recursive=True)
# Add datalad sibling
dataset.siblings(action="add", name="gin", url=DST)
# Push dataset to sibling
dataset.push(to="gin", force="all")