[ENH]: Improving DataGrabber patterns #308

Merged
synchon merged 27 commits from refactor/dg-patterns into main 2024-04-04 13:38:56 +00:00
26 changed files with 1357 additions and 739 deletions

View file

@ -0,0 +1 @@
Improve :class:`.PatternDataGrabber` and :class:`.PatternDataladDataGrabber`'s ``patterns`` to enable ``space``, ``format``, ``mask_item`` and other metadata description handling via YAML by `Synchon Mandal`_

View file

@ -61,7 +61,7 @@ Now that we have our element defined, we need to think about the structure of
the dataset. Mainly, because the structure of the dataset will determine how
the DataGrabber needs to be implemented.
``junifer`` provides an abstract class to deal with datasets that can be thought
``junifer`` provides a concrete class to deal with datasets that can be thought
in terms of *patterns*. A *pattern* is a string that contains placeholders that
are replaced by the actual values of the element. In our BIDS example, the path
to the T1w image of subject ``sub-01`` and session ``ses-01``, relative to the
@ -98,13 +98,14 @@ Step 3: Create a Data Grabber
Option A: Extending from PatternDataGrabber
^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
The :class:`.PatternDataGrabber` class is an abstract class that has the
The :class:`.PatternDataGrabber` class is a concrete class that has the
functionality of understanding patterns embedded in it.
Before creating the DataGrabber, we need to define 3 variables:
* ``types``: A list with the available :ref:`data_types` in our dataset.
* ``patterns``: A dictionary that specifies the pattern for each data type.
* ``patterns``: A dictionary that specifies the pattern and some additional
information for each data type.
* ``replacements``: A list indicating which of the elements in the patterns
should be replaced by the values of the element.
@ -114,8 +115,14 @@ For example, in our BIDS example, the variables will be:
types = ["T1w", "BOLD"]
patterns = {
"T1w": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"BOLD": "{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"T1w": {
"pattern": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"space": "native",
},
"BOLD": {
"pattern": "{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"space": "MNI152NLin6Asym",
},
}
replacements = ["subject", "session"]
@ -141,8 +148,14 @@ With the variables defined above, we can create our DataGrabber and name it
def __init__(self, datadir: str | Path) -> None:
types = ["T1w", "BOLD"]
patterns = {
"T1w": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"BOLD": "{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"T1w": {
"pattern": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"space": "native",
},
"BOLD": {
"pattern": "{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"space": "MNI152NLin6Asym",
},
}
replacements = ["subject", "session"]
super().__init__(
@ -171,8 +184,14 @@ use the :func:`.register_datagrabber` decorator.
def __init__(self, datadir: str | Path) -> None:
types = ["T1w", "BOLD"]
patterns = {
"T1w": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"BOLD": "{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"T1w": {
"pattern": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"space": "native",
},
"BOLD": {
"pattern": "{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"space": "MNI152NLin6Asym",
},
}
replacements = ["subject", "session"]
super().__init__(
@ -252,8 +271,14 @@ And we can create our DataGrabber:
def __init__(self) -> None:
types = ["T1w", "BOLD"]
patterns = {
"T1w": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"BOLD": "{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"T1w": {
"pattern": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"space": "native",
},
"BOLD": {
"pattern": "{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"space": "MNI152NLin6Asym",
},
}
replacements = ["subject", "session"]
uri = "https://gin.g-node.org/juaml/datalad-example-bids"
@ -277,13 +302,17 @@ This approach can be used directly from the YAML, like so:
- BOLD
- T1w
patterns:
BOLD: "{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz"
T1w: "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz"
BOLD:
pattern: "{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz"
space: MNI152NLin6Asym
T1w:
pattern: "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz"
space: native
replacements:
- subject
- session
uri: "https://gin.g-node.org/juaml/datalad-example-bids"
rootdir: "example_bids_ses"
rootdir: example_bids_ses
.. _extending_datagrabbers_base:
@ -314,10 +343,16 @@ and ``session``, we will use them as parameters of ``get_item``:
.. code-block:: python
def get_item(self, subject: str, session: str) -> dict[str, str]:
def get_item(self, subject: str, session: str) -> dict[str, dict[str, str]]:
out = {
"T1w": f"{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"BOLD": f"{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"T1w": {
"path": f"{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"space": "native",
},
"BOLD": {
"path": f"{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"space": "MNI152NLin6Asym",
},
}
return out
@ -367,12 +402,18 @@ So, to summarise, our DataGrabber will look like this:
@register_datagrabber
class ExampleBIDSDataGrabber(BaseDataGrabber):
def get_item(self, subject: str, session: str) -> dict[str, str]:
def get_item(self, subject: str, session: str) -> dict[str, dict[str, str]]:
out = {
"T1w": f"{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"BOLD": f"{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"T1w": {
"path": f"{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"space": "native",
},
"BOLD": {
"path": f"{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"space": "MNI152NLin6Asym",
},
}
return out
return out
def get_elements(self) -> list[str]:
subjects = ["sub-01", "sub-02", "sub-03"]
@ -438,16 +479,20 @@ this:
self, subject: str, session: str
) -> dict:
out = {
"BOLD": f"{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"BOLD": {
"path": f"{subject}/{session}/func/{subject}_{session}_task-rest_bold.nii.gz",
"space": "MNI152NLin6Asym",
},
"BOLD_confounds": {
"path": f"{subject}/{session}/func/{subject}_{session}_confounds.tsv",
"format": "adhoc",
"mappings": {
"fmriprep": {
"variable1": "rot_x",
"variable2": "rot_z",
"variable3": "rot_y",
}
"path": f"{subject}/{session}/func/{subject}_{session}_confounds.tsv",
"format": "adhoc",
"mappings": {
"fmriprep": {
"variable1": "rot_x",
"variable2": "rot_z",
"variable3": "rot_y",
},
},
},
}

View file

@ -103,6 +103,9 @@ Data Types
* - ``T1w``
- T1w image (3D)
- Preprocessed or Raw T1w image
* - ``T2w``
- T2w image (3D)
- Preprocessed or Raw T2w image
* - ``BOLD``
- BOLD image (4D)
- Preprocessed or Denoised BOLD image (fMRIPrep output)

View file

@ -25,8 +25,14 @@ configure_logging(level="INFO")
# replaced in the patterns.
types = ["T1w", "BOLD"]
patterns = {
"T1w": "{subject}/anat/{subject}_T1w.nii.gz",
"BOLD": "{subject}/func/{subject}_task-rest_bold.nii.gz",
"T1w": {
"pattern": "{subject}/anat/{subject}_T1w.nii.gz",
"space": "native",
},
"BOLD": {
"pattern": "{subject}/func/{subject}_task-rest_bold.nii.gz",
"space": "MNI152NLin6Asym",
},
}
replacements = ["subject"]
###############################################################################

View file

@ -31,7 +31,12 @@ class JuselessDataladAOMICID1000VBM(PatternDataladDataGrabber):
types = ["VBM_GM"]
replacements = ["subject"]
patterns = {
"VBM_GM": "sub-{subject}/mri/mwp1sub-{subject}_run-2_T1w.nii.gz",
"VBM_GM": {
"pattern": (
"sub-{subject}/mri/mwp1sub-{subject}_run-2_T1w.nii.gz"
),
"space": "IXI549Space",
},
}
super().__init__(
types=types,

View file

@ -34,7 +34,12 @@ class JuselessDataladCamCANVBM(PatternDataladDataGrabber):
)
types = ["VBM_GM"]
replacements = ["subject"]
patterns = {"VBM_GM": "sub-{subject}/mri/m0wp1sub-{subject}.nii.gz"}
patterns = {
"VBM_GM": {
"pattern": "sub-{subject}/mri/m0wp1sub-{subject}.nii.gz",
"space": "IXI549Space",
},
}
super().__init__(
types=types,
datadir=datadir,

View file

@ -43,7 +43,12 @@ class JuselessDataladIXIVBM(PatternDataladDataGrabber):
types = ["VBM_GM"]
replacements = ["site", "subject"]
patterns = {
"VBM_GM": "{site}/sub-{subject}/mri/m0wp1sub-{subject}.nii.gz"
"VBM_GM": {
"pattern": (
"{site}/sub-{subject}/mri/m0wp1sub-{subject}.nii.gz"
),
"space": "IXI549Space",
},
}
# validate and/or transform 'site' input

View file

@ -70,30 +70,48 @@ class JuselessUCLA(PatternDataGrabber):
self.tasks = tasks
# The patterns
patterns = {
"BOLD": (
"sub-{subject}/func/sub-{subject}_task-{task}_bold_space-"
"MNI152NLin2009cAsym_preproc.nii.gz"
),
"BOLD_confounds": (
"sub-{subject}/func/sub-{subject}_"
"task-{task}_bold_confounds.tsv"
),
"T1w": (
"sub-{subject}/anat/sub-{subject}_"
"T1w_space-MNI152NLin2009cAsym_preproc.nii.gz"
),
"probseg_CSF": (
"sub-{subject}/anat/sub-{subject}_T1w_space-"
"MNI152NLin2009cAsym_class-CSF_probtissue.nii.gz"
),
"probseg_GM": (
"sub-{subject}/anat/sub-{subject}_T1w_space-"
"MNI152NLin2009cAsym_class-GM_probtissue.nii.gz"
),
"probseg_WM": (
"sub-{subject}/anat/sub-{subject}_T1w_space"
"-MNI152NLin2009cAsym_class-WM_probtissue.nii.gz"
),
"BOLD": {
"pattern": (
"sub-{subject}/func/sub-{subject}_task-{task}_bold_space-"
"MNI152NLin2009cAsym_preproc.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"BOLD_confounds": {
"pattern": (
"sub-{subject}/func/sub-{subject}_"
"task-{task}_bold_confounds.tsv"
),
"space": "fmriprep",
},
"T1w": {
"pattern": (
"sub-{subject}/anat/sub-{subject}_"
"T1w_space-MNI152NLin2009cAsym_preproc.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_CSF": {
"pattern": (
"sub-{subject}/anat/sub-{subject}_T1w_space-"
"MNI152NLin2009cAsym_class-CSF_probtissue.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_GM": {
"pattern": (
"sub-{subject}/anat/sub-{subject}_T1w_space-"
"MNI152NLin2009cAsym_class-GM_probtissue.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_WM": {
"pattern": (
"sub-{subject}/anat/sub-{subject}_T1w_space"
"-MNI152NLin2009cAsym_class-WM_probtissue.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
}
# Set default types
if types is None:

View file

@ -32,7 +32,12 @@ class JuselessDataladUKBVBM(PatternDataladDataGrabber):
rootdir = "m0wp1"
types = ["VBM_GM"]
replacements = ["subject", "session"]
patterns = {"VBM_GM": "m0wp1sub-{subject}_ses-{session}_T1w.nii.gz"}
patterns = {
"VBM_GM": {
"pattern": "m0wp1sub-{subject}_ses-{session}_T1w.nii.gz",
"space": "IXI549Space",
},
}
super().__init__(
types=types,
datadir=datadir,

View file

@ -8,7 +8,7 @@
# License: AGPL
from pathlib import Path
from typing import Dict, List, Union
from typing import List, Union
from ...api.decorators import register_datagrabber
from ..pattern_datalad import PatternDataladDataGrabber
@ -41,51 +41,79 @@ class DataladAOMICID1000(PatternDataladDataGrabber):
) -> None:
# The patterns
patterns = {
"BOLD": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-moviewatching_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"BOLD_confounds": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-moviewatching_"
"desc-confounds_regressors.tsv"
),
"BOLD_mask": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-moviewatching_"
"space-MNI152NLin2009cAsym_"
"desc-brain_mask.nii.gz"
),
"T1w": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_"
"desc-preproc_T1w.nii.gz"
),
"T1w_mask": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_"
"desc-brain_mask.nii.gz"
),
"probseg_CSF": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"CSF_probseg.nii.gz"
),
"probseg_GM": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"GM_probseg.nii.gz"
),
"probseg_WM": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"WM_probseg.nii.gz"
),
"DWI": (
"derivatives/dwipreproc/sub-{subject}/dwi/"
"sub-{subject}_desc-preproc_dwi.nii.gz"
),
"BOLD": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-moviewatching_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"space": "MNI152NLin2009cAsym",
"mask_item": "BOLD_mask",
},
"BOLD_confounds": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-moviewatching_"
"desc-confounds_regressors.tsv"
),
"format": "fmriprep",
},
"BOLD_mask": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-moviewatching_"
"space-MNI152NLin2009cAsym_"
"desc-brain_mask.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"T1w": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_"
"desc-preproc_T1w.nii.gz"
),
"space": "MNI152NLin2009cAsym",
"mask_item": "T1w_mask",
},
"T1w_mask": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_"
"desc-brain_mask.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_CSF": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"CSF_probseg.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_GM": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"GM_probseg.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_WM": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"WM_probseg.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"DWI": {
"pattern": (
"derivatives/dwipreproc/sub-{subject}/dwi/"
"sub-{subject}_desc-preproc_dwi.nii.gz"
),
},
}
# Use native T1w assets
self.native_t1w = False
@ -93,19 +121,30 @@ class DataladAOMICID1000(PatternDataladDataGrabber):
self.native_t1w = True
patterns.update(
{
"T1w": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_desc-preproc_T1w.nii.gz"
),
"T1w_mask": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_desc-brain_mask.nii.gz"
),
"Warp": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_from-MNI152NLin2009cAsym_to-T1w_"
"mode-image_xfm.h5"
),
"T1w": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_desc-preproc_T1w.nii.gz"
),
"space": "native",
"mask_item": "T1w_mask",
},
"T1w_mask": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_desc-brain_mask.nii.gz"
),
"space": "native",
},
"Warp": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_from-MNI152NLin2009cAsym_to-T1w_"
"mode-image_xfm.h5"
),
"src": "MNI152NLin2009cAsym",
"dst": "native",
},
}
)
# Set default types
@ -126,35 +165,3 @@ class DataladAOMICID1000(PatternDataladDataGrabber):
replacements=replacements,
confounds_format="fmriprep",
)
def get_item(self, subject: str) -> Dict:
"""Index one element in the dataset.
Parameters
----------
subject : str
The subject ID.
Returns
-------
out : dict
Dictionary of paths for each type of data required for the
specified element.
"""
out = super().get_item(subject=subject)
if out.get("BOLD"):
out["BOLD"]["mask_item"] = "BOLD_mask"
# Add space information
out["BOLD"].update({"space": "MNI152NLin2009cAsym"})
if out.get("T1w"):
out["T1w"]["mask_item"] = "T1w_mask"
# Add space information
if self.native_t1w:
out["T1w"].update({"space": "native"})
else:
out["T1w"].update({"space": "MNI152NLin2009cAsym"})
if out.get("Warp"):
# Add source space information
out["Warp"].update({"src": "MNI152NLin2009cAsym"})
return out

View file

@ -77,50 +77,78 @@ class DataladAOMICPIOP1(PatternDataladDataGrabber):
self.tasks = tasks
# The patterns
patterns = {
"BOLD": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"BOLD_confounds": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_"
"desc-confounds_regressors.tsv"
),
"BOLD_mask": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"T1w": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_"
"desc-preproc_T1w.nii.gz"
),
"T1w_mask": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_"
"desc-brain_mask.nii.gz"
),
"probseg_CSF": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"CSF_probseg.nii.gz"
),
"probseg_GM": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"GM_probseg.nii.gz"
),
"probseg_WM": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"WM_probseg.nii.gz"
),
"DWI": (
"derivatives/dwipreproc/sub-{subject}/dwi/"
"sub-{subject}_desc-preproc_dwi.nii.gz"
),
"BOLD": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"space": "MNI152NLin2009cAsym",
"mask_item": "BOLD_mask",
},
"BOLD_confounds": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_"
"desc-confounds_regressors.tsv"
),
"format": "fmriprep",
},
"BOLD_mask": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"T1w": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_"
"desc-preproc_T1w.nii.gz"
),
"space": "MNI152NLin2009cAsym",
"mask_item": "T1w_mask",
},
"T1w_mask": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_"
"desc-brain_mask.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_CSF": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"CSF_probseg.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_GM": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"GM_probseg.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_WM": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"WM_probseg.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"DWI": {
"pattern": (
"derivatives/dwipreproc/sub-{subject}/dwi/"
"sub-{subject}_desc-preproc_dwi.nii.gz"
),
},
}
# Use native T1w assets
self.native_t1w = False
@ -128,19 +156,30 @@ class DataladAOMICPIOP1(PatternDataladDataGrabber):
self.native_t1w = True
patterns.update(
{
"T1w": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_desc-preproc_T1w.nii.gz"
),
"T1w_mask": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_desc-brain_mask.nii.gz"
),
"Warp": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_from-MNI152NLin2009cAsym_to-T1w_"
"mode-image_xfm.h5"
),
"T1w": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_desc-preproc_T1w.nii.gz"
),
"space": "native",
"mask_item": "T1w_mask",
},
"T1w_mask": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_desc-brain_mask.nii.gz"
),
"space": "native",
},
"Warp": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_from-MNI152NLin2009cAsym_to-T1w_"
"mode-image_xfm.h5"
),
"src": "MNI152NLin2009cAsym",
"dst": "native",
},
}
)
# Set default types
@ -192,22 +231,7 @@ class DataladAOMICPIOP1(PatternDataladDataGrabber):
acq = task_acqs[task]
new_task = f"{task}_acq-{acq}"
out = super().get_item(subject=subject, task=new_task)
if out.get("BOLD"):
out["BOLD"]["mask_item"] = "BOLD_mask"
# Add space information
out["BOLD"].update({"space": "MNI152NLin2009cAsym"})
if out.get("T1w"):
out["T1w"]["mask_item"] = "T1w_mask"
# Add space information
if self.native_t1w:
out["T1w"].update({"space": "native"})
else:
out["T1w"].update({"space": "MNI152NLin2009cAsym"})
if out.get("Warp"):
# Add source space information
out["Warp"].update({"src": "MNI152NLin2009cAsym"})
return out
return super().get_item(subject=subject, task=new_task)
def get_elements(self) -> List:
"""Implement fetching list of subjects in the dataset.

View file

@ -74,50 +74,78 @@ class DataladAOMICPIOP2(PatternDataladDataGrabber):
self.tasks = tasks
# The patterns
patterns = {
"BOLD": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"BOLD_confounds": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_"
"desc-confounds_regressors.tsv"
),
"BOLD_mask": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"T1w": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_"
"desc-preproc_T1w.nii.gz"
),
"T1w_mask": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_"
"desc-brain_mask.nii.gz"
),
"probseg_CSF": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"CSF_probseg.nii.gz"
),
"probseg_GM": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"GM_probseg.nii.gz"
),
"probseg_WM": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"WM_probseg.nii.gz"
),
"DWI": (
"derivatives/dwipreproc/sub-{subject}/dwi/"
"sub-{subject}_desc-preproc_dwi.nii.gz"
),
"BOLD": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"space": "MNI152NLin2009cAsym",
"mask_item": "BOLD_mask",
},
"BOLD_confounds": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_"
"desc-confounds_regressors.tsv"
),
"format": "fmriprep",
},
"BOLD_mask": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"T1w": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_"
"desc-preproc_T1w.nii.gz"
),
"space": "MNI152NLin2009cAsym",
"mask_item": "T1w_mask",
},
"T1w_mask": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_"
"desc-brain_mask.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_CSF": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"CSF_probseg.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_GM": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"GM_probseg.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_WM": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-"
"WM_probseg.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"DWI": {
"pattern": (
"derivatives/dwipreproc/sub-{subject}/dwi/"
"sub-{subject}_desc-preproc_dwi.nii.gz"
),
},
}
# Use native T1w assets
self.native_t1w = False
@ -125,19 +153,30 @@ class DataladAOMICPIOP2(PatternDataladDataGrabber):
self.native_t1w = True
patterns.update(
{
"T1w": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_desc-preproc_T1w.nii.gz"
),
"T1w_mask": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_desc-brain_mask.nii.gz"
),
"Warp": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_from-MNI152NLin2009cAsym_to-T1w_"
"mode-image_xfm.h5"
),
"T1w": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_desc-preproc_T1w.nii.gz"
),
"space": "native",
"mask_item": "T1w_mask",
},
"T1w_mask": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_desc-brain_mask.nii.gz"
),
"space": "native",
},
"Warp": {
"pattern": (
"derivatives/fmriprep/sub-{subject}/anat/"
"sub-{subject}_from-MNI152NLin2009cAsym_to-T1w_"
"mode-image_xfm.h5"
),
"src": "MNI152NLin2009cAsym",
"dst": "native",
},
}
)
# Set default types
@ -192,19 +231,4 @@ class DataladAOMICPIOP2(PatternDataladDataGrabber):
specified element.
"""
out = super().get_item(subject=subject, task=f"{task}_acq-seq")
if out.get("BOLD"):
out["BOLD"]["mask_item"] = "BOLD_mask"
# Add space information
out["BOLD"].update({"space": "MNI152NLin2009cAsym"})
if out.get("T1w"):
out["T1w"]["mask_item"] = "T1w_mask"
# Add space information
if self.native_t1w:
out["T1w"].update({"space": "native"})
else:
out["T1w"].update({"space": "MNI152NLin2009cAsym"})
if out.get("Warp"):
# Add source space information
out["Warp"].update({"src": "MNI152NLin2009cAsym"})
return out
return super().get_item(subject=subject, task=f"{task}_acq-seq")

View file

@ -59,7 +59,9 @@ class BaseDataGrabber(ABC, UpdateMetaMixin):
"""
yield from self.get_elements()
def __getitem__(self, element: Union[str, Tuple[str]]) -> Dict[str, Dict]:
def __getitem__(
self, element: Union[str, Tuple[str, ...]]
) -> Dict[str, Dict]:
"""Enable indexing support.
Parameters
@ -183,7 +185,7 @@ class BaseDataGrabber(ABC, UpdateMetaMixin):
raise_error(
msg="Concrete classes need to implement get_element_keys().",
klass=NotImplementedError,
)
) # pragma: no cover
@abstractmethod
def get_elements(self) -> List[Union[str, Tuple[str]]]:
@ -200,7 +202,7 @@ class BaseDataGrabber(ABC, UpdateMetaMixin):
raise_error(
msg="Concrete classes need to implement get_elements().",
klass=NotImplementedError,
)
) # pragma: no cover
@abstractmethod
def get_item(self, **element: Dict) -> Dict[str, Dict]:
@ -221,4 +223,4 @@ class BaseDataGrabber(ABC, UpdateMetaMixin):
raise_error(
msg="Concrete classes need to implement get_item().",
klass=NotImplementedError,
)
) # pragma: no cover

View file

@ -17,12 +17,10 @@ import datalad.api as dl
from datalad.support.exceptions import IncompleteResultsError
from datalad.support.gitrepo import GitRepo
from ..api.decorators import register_datagrabber
from ..utils import logger, raise_error, warn_with_log
from .base import BaseDataGrabber
@register_datagrabber
class DataladDataGrabber(BaseDataGrabber):
"""Abstract base class for datalad-based data fetching.

View file

@ -139,43 +139,69 @@ class DMCC13Benchmark(PatternDataladDataGrabber):
self.runs = runs
# The patterns
patterns = {
"BOLD": (
"derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/"
"func/sub-{subject}_ses-{session}_task-{task}_acq-mb4"
"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"BOLD_confounds": (
"derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/"
"func/sub-{subject}_ses-{session}_task-{task}_acq-mb4"
"{phase_encoding}_run-{run}_desc-confounds_regressors.tsv"
),
"BOLD_mask": (
"derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/"
"/func/sub-{subject}_ses-{session}_task-{task}_acq-mb4"
"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"T1w": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_desc-preproc_T1w.nii.gz"
),
"T1w_mask": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"probseg_CSF": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-CSF_probseg.nii.gz"
),
"probseg_GM": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-GM_probseg.nii.gz"
),
"probseg_WM": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-WM_probseg.nii.gz"
),
"BOLD": {
"pattern": (
"derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/"
"func/sub-{subject}_ses-{session}_task-{task}_acq-mb4"
"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"space": "MNI152NLin2009cAsym",
"mask_item": "BOLD_mask",
},
"BOLD_confounds": {
"pattern": (
"derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/"
"func/sub-{subject}_ses-{session}_task-{task}_acq-mb4"
"{phase_encoding}_run-{run}_desc-confounds_regressors.tsv"
),
"format": "fmriprep",
},
"BOLD_mask": {
"pattern": (
"derivatives/fmriprep-1.3.2/sub-{subject}/ses-{session}/"
"/func/sub-{subject}_ses-{session}_task-{task}_acq-mb4"
"{phase_encoding}_run-{run}_"
"space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"T1w": {
"pattern": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_desc-preproc_T1w.nii.gz"
),
"space": "MNI152NLin2009cAsym",
"mask_item": "T1w_mask",
},
"T1w_mask": {
"pattern": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_CSF": {
"pattern": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-CSF_probseg.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_GM": {
"pattern": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-GM_probseg.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"probseg_WM": {
"pattern": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_space-MNI152NLin2009cAsym_label-WM_probseg.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
}
# Use native T1w assets
self.native_t1w = False
@ -183,19 +209,30 @@ class DMCC13Benchmark(PatternDataladDataGrabber):
self.native_t1w = True
patterns.update(
{
"T1w": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_desc-preproc_T1w.nii.gz"
),
"T1w_mask": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_desc-brain_mask.nii.gz"
),
"Warp": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_from-MNI152NLin2009cAsym_to-T1w_"
"mode-image_xfm.h5"
),
"T1w": {
"pattern": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_desc-preproc_T1w.nii.gz"
),
"space": "native",
"mask_item": "T1w_mask",
},
"T1w_mask": {
"pattern": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_desc-brain_mask.nii.gz"
),
"space": "native",
},
"Warp": {
"pattern": (
"derivatives/fmriprep-1.3.2/sub-{subject}/anat/"
"sub-{subject}_from-MNI152NLin2009cAsym_to-T1w_"
"mode-image_xfm.h5"
),
"src": "MNI152NLin2009cAsym",
"dst": "native",
},
}
)
# Set default types

View file

@ -106,14 +106,26 @@ class HCP1200(PatternDataGrabber):
types = ["BOLD", "T1w", "Warp"]
# The patterns
patterns = {
"BOLD": (
"{subject}/MNINonLinear/Results/"
"{task}_{phase_encoding}/"
"{task}_{phase_encoding}"
f"{suffix}.nii.gz"
),
"T1w": "{subject}/T1w/T1w_acpc_dc_restore.nii.gz",
"Warp": "{subject}/MNINonLinear/xfms/standard2acpc_dc.nii.gz",
"BOLD": {
"pattern": (
"{subject}/MNINonLinear/Results/"
"{task}_{phase_encoding}/"
"{task}_{phase_encoding}"
f"{suffix}.nii.gz"
),
"space": "MNI152NLin6Asym",
},
"T1w": {
"pattern": "{subject}/T1w/T1w_acpc_dc_restore.nii.gz",
"space": "native",
},
"Warp": {
"pattern": (
"{subject}/MNINonLinear/xfms/standard2acpc_dc.nii.gz"
),
"src": "MNI152NLin6Asym",
"dst": "native",
},
}
# The replacements
replacements = ["subject", "task", "phase_encoding"]
@ -150,19 +162,9 @@ class HCP1200(PatternDataGrabber):
else:
new_task = f"tfMRI_{task}"
out = super().get_item(
return super().get_item(
subject=subject, task=new_task, phase_encoding=phase_encoding
)
# Add space for BOLD data type
if "BOLD" in out:
out["BOLD"].update({"space": "MNI152NLin6Asym"})
# Add space for T1w data type
if "T1w" in out:
out["T1w"].update({"space": "native"})
# Add source space for Warp data type
if "Warp" in out:
out["Warp"].update({"src": "MNI152NLin6Asym"})
return out
def get_elements(self) -> List:
"""Implement fetching list of elements in the dataset.

View file

@ -32,11 +32,98 @@ class PatternDataGrabber(BaseDataGrabber):
types : list of str
The types of data to be grabbed.
patterns : dict
Patterns for each type of data as a dictionary. The keys are the types
and the values are the patterns. Each occurrence of the string
``{subject}`` in the pattern will be replaced by the indexed element.
Data type patterns as a dictionary. It has the following schema:
* ``"T1w"`` :
.. code-block:: none
{
"mandatory": ["pattern", "space"],
"optional": []
}
* ``"T2w"`` :
.. code-block:: none
{
"mandatory": ["pattern", "space"],
"optional": []
}
* ``"BOLD"`` :
.. code-block:: none
{
"mandatory": ["pattern", "space"],
"optional": ["mask_item"]
}
* ``"Warp"`` :
.. code-block:: none
{
"mandatory": ["pattern", "src", "dst"],
"optional": []
}
* ``"BOLD_confounds"`` :
.. code-block:: none
{
"mandatory": ["pattern", "format"],
"optional": []
}
* ``"VBM_GM"`` :
.. code-block:: none
{
"mandatory": ["pattern", "space"],
"optional": []
}
* ``"VBM_WM"`` :
.. code-block:: none
{
"mandatory": ["pattern", "space"],
"optional": []
}
Basically, for each data type, one needs to provide ``mandatory`` keys
and can choose to also provide ``optional`` keys. The value for each
key is a string. So, one needs to provide necessary data types as a
dictionary, for example:
.. code-block:: none
{
"BOLD": {
"pattern": "...",
"space": "...",
},
"T1w": {
"pattern": "...",
"space": "...",
},
"Warp": {
"pattern": "...",
"src": "...",
"dst": "...",
}
}
taken from :class:`.HCP1200`.
replacements : str or list of str
Replacements in the patterns for each item in the "element" tuple.
Replacements in the ``pattern`` key of each data type. The value needs
to be a list of all possible replacements.
datadir : str or pathlib.Path
The directory where the data is / will be stored.
confounds_format : {"fmriprep", "adhoc"} or None, optional
@ -52,7 +139,7 @@ class PatternDataGrabber(BaseDataGrabber):
def __init__(
self,
types: List[str],
patterns: Dict[str, str],
patterns: Dict[str, Dict[str, str]],
replacements: Union[List[str], str],
datadir: Union[str, Path],
confounds_format: Optional[str] = None,
@ -69,7 +156,10 @@ class PatternDataGrabber(BaseDataGrabber):
self.replacements = replacements
# Validate confounds format
if confounds_format and confounds_format not in _CONFOUNDS_FORMATS:
if (
confounds_format is not None
and confounds_format not in _CONFOUNDS_FORMATS
):
raise_error(
"Invalid value for `confounds_format`, should be one of "
f"{_CONFOUNDS_FORMATS}."
@ -143,6 +233,11 @@ class PatternDataGrabber(BaseDataGrabber):
str
The pattern with the element replaced.
Raises
------
ValueError
If element keys do not match with replacements.
"""
if list(element.keys()) != self.replacements:
raise_error(
@ -167,7 +262,7 @@ class PatternDataGrabber(BaseDataGrabber):
return self.replacements
def get_item(self, **element: str) -> Dict[str, Dict]:
"""Implement single element indexing in the database.
"""Implement single element indexing for the datagrabber.
This method constructs a real path to the requested item's data, by
replacing the ``patterns`` with actual values passed via ``**element``.
@ -184,20 +279,33 @@ class PatternDataGrabber(BaseDataGrabber):
Dictionary of dictionaries for each type of data required for the
specified element.
Raises
------
RuntimeError
If more than one file matches for a data type's pattern or
if no file matches for a data type's pattern or
if file cannot be accessed for an element.
"""
out = {}
for t_type in self.types:
t_pattern = self.patterns[t_type]
t_replace = self._replace_patterns_glob(element, t_pattern)
t_replace = self._replace_patterns_glob(
element, t_pattern["pattern"]
)
if "*" in t_replace:
t_matches = list(self.datadir.absolute().glob(t_replace))
if len(t_matches) > 1:
raise_error(
f"More than one file matches for {element} / {t_type}:"
f" {t_matches}"
f" {t_matches}",
klass=RuntimeError,
)
elif len(t_matches) == 0:
raise_error(f"No file matches for {element} / {t_type}")
raise_error(
f"No file matches for {element} / {t_type}",
klass=RuntimeError,
)
t_out = t_matches[0]
else:
t_out = self.datadir / t_replace
@ -205,22 +313,13 @@ class PatternDataGrabber(BaseDataGrabber):
if not t_out.exists() and not t_out.is_symlink():
raise_error(
f"Cannot access {t_type} for {element}: "
f"File {t_out} does not exist"
f"File {t_out} does not exist",
klass=RuntimeError,
)
# Update path for the element
out[t_type] = {"path": t_out}
# Update confounds format for BOLD_confounds
# (if found in the datagrabber)
if t_type == "BOLD_confounds":
if not self.confounds_format:
raise_error(
"`confounds_format` needs to be one of "
f"{_CONFOUNDS_FORMATS}, None provided. "
"As the DataGrabber used specifies "
"'BOLD_confounds', None is invalid."
)
# Set the format
out[t_type].update({"format": self.confounds_format})
out[t_type] = t_pattern.copy() # copy data type dictionary
out[t_type].pop("pattern") # remove pattern key
out[t_type].update({"path": t_out}) # add path key
return out
@ -259,7 +358,7 @@ class PatternDataGrabber(BaseDataGrabber):
re_pattern,
glob_pattern,
t_replacements,
) = self._replace_patterns_regex(t_pattern)
) = self._replace_patterns_regex(t_pattern["pattern"])
for fname in self.datadir.glob(glob_pattern):
suffix = fname.relative_to(self.datadir).as_posix()
m = re.match(re_pattern, suffix)

View file

@ -5,12 +5,11 @@
# Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL
from typing import Dict, List
from ..api.decorators import register_datagrabber
from ..utils import logger
from .datalad_base import DataladDataGrabber
from .pattern import PatternDataGrabber
from .utils import validate_patterns
@register_datagrabber
@ -25,11 +24,109 @@ class PatternDataladDataGrabber(DataladDataGrabber, PatternDataGrabber):
types : list of str
The types of data to be grabbed.
patterns : dict
Patterns for each type of data as a dictionary. The keys are the types
and the values are the patterns. Each occurrence of the string
``{subject}`` in the pattern will be replaced by the indexed element.
**kwargs
Keyword arguments passed to superclass.
Data type patterns as a dictionary. It has the following schema:
* ``"T1w"`` :
.. code-block:: none
{
"mandatory": ["pattern", "space"],
"optional": []
}
* ``"T2w"`` :
.. code-block:: none
{
"mandatory": ["pattern", "space"],
"optional": []
}
* ``"BOLD"`` :
.. code-block:: none
{
"mandatory": ["pattern", "space"],
"optional": ["mask_item"]
}
* ``"Warp"`` :
.. code-block:: none
{
"mandatory": ["pattern", "src", "dst"],
"optional": []
}
* ``"BOLD_confounds"`` :
.. code-block:: none
{
"mandatory": ["pattern", "format"],
"optional": []
}
* ``"VBM_GM"`` :
.. code-block:: none
{
"mandatory": ["pattern", "space"],
"optional": []
}
* ``"VBM_WM"`` :
.. code-block:: none
{
"mandatory": ["pattern", "space"],
"optional": []
}
Basically, for each data type, one needs to provide ``mandatory`` keys
and can choose to also provide ``optional`` keys. The value for each
key is a string. So, one needs to provide necessary data types as a
dictionary, for example:
.. code-block:: none
{
"BOLD": {
"pattern": "...",
"space": "...",
},
"T1w": {
"pattern": "...",
"space": "...",
},
"Warp": {
"pattern": "...",
"src": "...",
"dst": "...",
}
}
taken from :class:`.HCP1200`.
replacements : str or list of str
Replacements in the ``pattern`` key of each data type. The value needs
to be a list of all possible replacements.
confounds_format : {"fmriprep", "adhoc"} or None, optional
The format of the confounds for the dataset (default None).
datadir : str or pathlib.Path or None, optional
That directory where the datalad dataset will be cloned. If None,
the datalad dataset will be cloned into a temporary directory
(default None).
rootdir : str or pathlib.Path, optional
The path within the datalad dataset to the root directory
(default ".").
uri : str or None, optional
URI of the datalad sibling (default None).
See Also
--------
@ -42,12 +139,13 @@ class PatternDataladDataGrabber(DataladDataGrabber, PatternDataGrabber):
def __init__(
self,
types: List[str],
patterns: Dict[str, str],
**kwargs,
) -> None:
# Validate patterns
validate_patterns(types=types, patterns=patterns)
# TODO(synchon): needs to be reworked, DataladDataGrabber needs to be
# a mixin to avoid multiple inheritance wherever possible.
super().__init__(types=types, patterns=patterns, **kwargs)
self.patterns = patterns
logger.debug("Initializing PatternDataladDataGrabber")
for key, val in kwargs.items():
logger.debug(f"\t{key} = {val}")
super().__init__(**kwargs)

View file

@ -12,12 +12,6 @@ import pytest
from junifer.datagrabber import BaseDataGrabber
def test_BaseDataGrabber_abstractness() -> None:
"""Test BaseDataGrabber is abstract base class."""
with pytest.raises(TypeError, match=r"abstract"):
BaseDataGrabber(datadir="/tmp", types=["func"]) # type: ignore
def test_BaseDataGrabber() -> None:
"""Test BaseDataGrabber."""

View file

@ -3,6 +3,9 @@
# Authors: Federico Raimondo <f.raimondo@fz-juelich.de>
# License: AGPL
from contextlib import nullcontext
from typing import ContextManager, Dict, List, Union
import pytest
from junifer.datagrabber.utils import (
@ -12,79 +15,204 @@ from junifer.datagrabber.utils import (
)
def test_validate_types() -> None:
"""Test validation of types."""
with pytest.raises(TypeError, match="must be a list"):
validate_types("wrong") # type: ignore
with pytest.raises(TypeError, match="must be a list of strings"):
validate_types([1]) # type: ignore
@pytest.mark.parametrize(
"types, expect",
[
("wrong", pytest.raises(TypeError, match="must be a list")),
([1], pytest.raises(TypeError, match="must be a list of strings")),
(["T1w", "BOLD"], nullcontext()),
],
)
def test_validate_types(
types: Union[str, List[str], List[int]],
expect: ContextManager,
) -> None:
"""Test validation of types.
validate_types(["T1w", "BOLD"])
Parameters
----------
types : str, list of int or str
The parametrized data types to validate.
expect : typing.ContextManager
The parametrized ContextManager object.
"""
with expect:
validate_types(types) # type: ignore
def test_validate_replacements() -> None:
"""Test validation of replacements."""
with pytest.raises(TypeError, match="must be a list"):
validate_replacements("wrong", "also wrong") # type: ignore
with pytest.raises(TypeError, match="must be a dict"):
validate_replacements(["correct"], "wrong") # type: ignore
@pytest.mark.parametrize(
"replacements, patterns, expect",
[
(
"wrong",
"also wrong",
pytest.raises(TypeError, match="must be a list"),
),
(
[1],
{
"T1w": {"pattern": "{subject}/anat/{subject}_T1w.nii.gz"},
"BOLD": {
"pattern": "{subject}/func/{subject}_task-rest_bold.nii.gz"
},
},
pytest.raises(TypeError, match="must be a list of strings"),
),
(
["session"],
{
"T1w": {"pattern": "{subject}/anat/{subject}_T1w.nii.gz"},
"BOLD": {
"pattern": "{subject}/func/{subject}_task-rest_bold.nii.gz"
},
},
pytest.raises(ValueError, match="is not part of"),
),
(
["subject", "session"],
{
"T1w": {"pattern": "{subject}/anat/_T1w.nii.gz"},
"BOLD": {"pattern": "{session}/func/_task-rest_bold.nii.gz"},
},
pytest.raises(ValueError, match="At least one pattern"),
),
(
["subject"],
{
"T1w": {"pattern": "{subject}/anat/{subject}_T1w.nii.gz"},
"BOLD": {
"pattern": "{subject}/func/{subject}_task-rest_bold.nii.gz"
},
},
nullcontext(),
),
],
)
def test_validate_replacements(
replacements: Union[str, List[str], List[int]],
patterns: Union[str, Dict[str, Dict[str, str]]],
expect: ContextManager,
) -> None:
"""Test validation of replacements.
patterns = {
"T1w": "{subject}/anat/{subject}_T1w.nii.gz",
"BOLD": "{subject}/func/{subject}_task-rest_bold.nii.gz",
}
Parameters
----------
replacements : str, list of str or int
The parametrized pattern replacements to validate.
patterns : str, dict
The parametrized patterns to validate against.
expect : typing.ContextManager
The parametrized ContextManager object.
with pytest.raises(TypeError, match="must be a list of strings"):
validate_replacements([1], patterns) # type: ignore
with pytest.raises(ValueError, match="is not part of"):
validate_replacements(["session"], patterns)
wrong_patterns = {
"T1w": "{subject}/anat/_T1w.nii.gz",
"BOLD": "{session}/func/_task-rest_bold.nii.gz",
}
with pytest.raises(ValueError, match="At least one pattern"):
validate_replacements(["subject", "session"], wrong_patterns)
validate_replacements(["subject"], patterns)
"""
with expect:
validate_replacements(replacements=replacements, patterns=patterns) # type: ignore
def test_validate_patterns() -> None:
"""Test validation of patterns."""
types = ["T1w", "BOLD"]
with pytest.raises(TypeError, match="must be a dict"):
validate_patterns(types, "wrong") # type: ignore
@pytest.mark.parametrize(
"types, patterns, expect",
[
(
["T1w", "BOLD"],
"wrong",
pytest.raises(TypeError, match="must be a dict"),
),
(
["T1w", "BOLD"],
{
"T1w": {"pattern": "{subject}/anat/{subject}_T1w.nii.gz"},
},
pytest.raises(
ValueError,
match="Length of `types` more than that of `patterns`.",
),
),
(
["T1w", "BOLD"],
{
"T1w": {"pattern": "{subject}/anat/{subject}_T1w.nii.gz"},
"T2w": {"pattern": "{subject}/anat/{subject}_T2w.nii.gz"},
},
pytest.raises(ValueError, match="contain all"),
),
(
["T3w"],
{
"T3w": {"pattern": "{subject}/anat/{subject}_T3w.nii.gz"},
},
pytest.raises(ValueError, match="Unknown data type"),
),
(
["BOLD"],
{
"BOLD": {"patterns": "{subject}/func/{subject}_BOLD.nii.gz"},
},
pytest.raises(KeyError, match="Mandatory key"),
),
(
["BOLD_confounds"],
{
"BOLD_confounds": {
"pattern": "{subject}/func/{subject}_confounds.tsv",
"format": "fmriprep",
"space": "MNINLin6Asym",
},
},
pytest.raises(RuntimeError, match="not accepted"),
),
(
["T1w"],
{
"T1w": {
"pattern": "{subject}/anat/{subject}*.nii",
"space": "native",
},
},
pytest.raises(ValueError, match="following a replacement"),
),
(
["T1w", "T2w", "BOLD", "BOLD_confounds"],
{
"T1w": {
"pattern": "{subject}/anat/{subject}_T1w.nii.gz",
"space": "native",
},
"T2w": {
"pattern": "{subject}/anat/{subject}_T2w.nii.gz",
"space": "native",
},
"BOLD": {
"pattern": (
"{subject}/func/{subject}_task-rest_bold.nii.gz"
),
"space": "MNI152NLin6Asym",
},
"BOLD_confounds": {
"pattern": "{subject}/func/{subject}_confounds.tsv",
"format": "fmriprep",
},
},
nullcontext(),
),
],
)
def test_validate_patterns(
types: List[str],
patterns: Union[str, Dict[str, Dict[str, str]]],
expect: ContextManager,
) -> None:
"""Test validation of patterns.
wrongpatterns = {
"T1w": "{subject}/anat/{subject}_T1w.nii.gz",
}
Parameters
----------
types : list of str
The parametrized data types.
patterns : str, dict
The patterns to validate.
expect : typing.ContextManager
The parametrized ContextManager object.
with pytest.raises(
ValueError, match="Length of `types` more than that of `patterns`."
):
validate_patterns(types, wrongpatterns) # type: ignore
wrongpatterns = {
"T1w": "{subject}/anat/{subject}_T1w.nii.gz",
"T2": "{subject}/anat/{subject}_T2.nii.gz",
}
with pytest.raises(ValueError, match="contain all"):
validate_patterns(types, wrongpatterns) # type: ignore
patterns = {
"T1w": "{subject}/anat/{subject}_T1w.nii.gz",
"BOLD": "{subject}/func/{subject}_task-rest_bold.nii.gz",
}
wrongpatterns = {
"T1w": "{subject}/anat/{subject}*.nii",
"BOLD": "{subject}/func/{subject}_task-rest_bold.nii.gz",
}
with pytest.raises(ValueError, match="following a replacement"):
validate_patterns(types, wrongpatterns)
validate_patterns(types, patterns)
"""
with expect:
validate_patterns(types=types, patterns=patterns) # type: ignore

View file

@ -26,12 +26,6 @@ _testing_dataset = {
}
def test_DataladDataGrabber_abstractness() -> None:
"""Test DataladDataGrabber is abstract base class."""
with pytest.raises(TypeError, match=r"abstract"):
DataladDataGrabber() # type: ignore
@pytest.fixture
def concrete_datagrabber() -> Type[DataladDataGrabber]:
"""Return a concrete datalad-based DataGrabber.

View file

@ -26,11 +26,21 @@ def test_MultipleDataGrabber() -> None:
rootdir = "example_bids_ses"
replacements = ["subject", "session"]
pattern1 = {
"T1w": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"T1w": {
"pattern": (
"{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz"
),
"space": "native",
},
}
pattern2 = {
"BOLD": "{subject}/{session}/func/"
"{subject}_{session}_task-rest_bold.nii.gz",
"BOLD": {
"pattern": (
"{subject}/{session}/func/"
"{subject}_{session}_task-rest_bold.nii.gz"
),
"space": "MNI152NLin6Asym",
},
}
dg1 = PatternDataladDataGrabber(
rootdir=rootdir,
@ -84,11 +94,21 @@ def test_MultipleDataGrabber_no_intersection() -> None:
rootdir = "example_bids_ses"
replacements = ["subject", "session"]
pattern1 = {
"T1w": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"T1w": {
"pattern": (
"{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz"
),
"space": "native",
},
}
pattern2 = {
"BOLD": "{subject}/{session}/func/"
"{subject}_{session}_task-rest_bold.nii.gz",
"BOLD": {
"pattern": (
"{subject}/{session}/func/"
"{subject}_{session}_task-rest_bold.nii.gz"
),
"space": "MNI152NLin6Asym",
},
}
dg1 = PatternDataladDataGrabber(
rootdir=rootdir,
@ -119,7 +139,12 @@ def test_MultipleDataGrabber_get_item() -> None:
rootdir = "example_bids_ses"
replacements = ["subject", "session"]
pattern1 = {
"T1w": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"T1w": {
"pattern": (
"{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz"
),
"space": "native",
},
}
dg1 = PatternDataladDataGrabber(
rootdir=rootdir,
@ -142,10 +167,18 @@ def test_MultipleDataGrabber_validation() -> None:
replacement1 = ["subject", "session"]
replacement2 = ["subject"]
pattern1 = {
"T1w": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"T1w": {
"pattern": (
"{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz"
),
"space": "native",
},
}
pattern2 = {
"bold": "{subject}/func/{subject}_task-rest_bold.nii.gz",
"BOLD": {
"pattern": "{subject}/func/{subject}_task-rest_bold.nii.gz",
"space": "MNI152NLin6Asym",
},
}
dg1 = PatternDataladDataGrabber(
rootdir=rootdir,
@ -158,7 +191,7 @@ def test_MultipleDataGrabber_validation() -> None:
dg2 = PatternDataladDataGrabber(
rootdir=rootdir,
uri=repo_uri2,
types=["bold"],
types=["BOLD"],
patterns=pattern2,
replacements=replacement2,
)

View file

@ -5,6 +5,7 @@
# Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL
from itertools import product
from pathlib import Path
import pytest
@ -21,144 +22,88 @@ def test_PatternDataGrabber_errors(tmp_path: Path) -> None:
The path to the test directory.
"""
with pytest.raises(TypeError, match=r"`types` must be a list"):
PatternDataGrabber(
datadir="/tmp",
types="wrong", # type: ignore
patterns={"wrong": "pattern"},
replacements="subject", # type: ignore
)
with pytest.raises(TypeError, match=r"`types` must be a list of strings"):
PatternDataGrabber(
datadir="/tmp", # type: ignore
types=[1, 2, 3], # type: ignore
patterns={"1": "pattern", "2": "pattern", "3": "pattern"},
replacements="subject", # type: ignore
)
with pytest.raises(
ValueError, match=r"`patterns` must contain all `types`"
):
PatternDataGrabber(
datadir="/tmp",
types=["func", "anat"],
patterns={"1": "pattern", "2": "pattern", "3": "pattern"},
replacements=1, # type: ignore
)
with pytest.raises(TypeError, match=r"`patterns` must be a dict"):
PatternDataGrabber(
datadir="/tmp",
types=["func", "anat"],
patterns="wrong", # type: ignore
replacements="subject", # type: ignore
)
with pytest.raises(
ValueError, match=r"Length of `types` more than that of `patterns`"
):
PatternDataGrabber(
datadir="/tmp",
types=["func", "anat"],
patterns={"wrong": "pattern"},
replacements="subject", # type: ignore
)
with pytest.raises(
ValueError, match=r"`patterns` must contain all `types`"
):
PatternDataGrabber(
datadir="/tmp",
types=["func", "anat"],
patterns={"wrong": "pattern", "func": "pattern"},
replacements="subject", # type: ignore
)
with pytest.raises(TypeError, match=r"must be a list of strings"):
PatternDataGrabber(
datadir="/tmp",
types=["func", "anat"],
patterns={"func": "func/test", "anat": "anat/test"},
replacements=1, # type: ignore
)
with pytest.raises(ValueError, match=r"not part of any pattern"):
PatternDataGrabber(
datadir="/tmp",
types=["func", "anat"],
patterns={
"func": "func/{subject}.nii",
"anat": "anat/{subject}.nii",
},
replacements=["subject", "wrong"],
)
tmpdir = tmp_path / "pattern_dg_test_errors"
datagrabber = PatternDataGrabber(
datagrabber_no_access = PatternDataGrabber(
datadir=tmpdir,
types=["func", "anat"],
types=["BOLD", "T1w"],
patterns={
"func": "func/{subject}_single.nii",
"anat": "anat/{subject}_{session}_ses.nii",
"BOLD": {
"pattern": "func/{subject}_single.nii",
"space": "MNI152NLin6Asym",
},
"T1w": {
"pattern": "anat/{subject}_{session}_ses.nii",
"space": "MNI152NLin6Asym",
},
},
replacements=["subject", "session"],
)
with pytest.raises(ValueError, match="element keys must be"):
datagrabber["sub001"]
datagrabber_no_access[("sub001")]
# This should not work, file does not exists
with pytest.raises(ValueError, match="Cannot access"):
datagrabber["sub001", "ses001"]
with pytest.raises(RuntimeError, match="Cannot access"):
datagrabber_no_access[("sub001", "ses001")]
# Create directories and files
(tmpdir / "func").mkdir(exist_ok=True, parents=True)
(tmpdir / "anat").mkdir(exist_ok=True, parents=True)
for t_subject in range(3):
for t_session in range(2):
subject = f"sub{t_subject:03d}"
session = f"ses{t_session:03d}"
(tmpdir / "func" / f"{subject}_single.nii").touch()
if t_subject == 2:
(tmpdir / "func" / f"{subject}_extra.nii").touch()
(tmpdir / "anat" / f"{subject}_{session}_ses.nii").touch()
for t_subject, t_session in product(range(3), range(2)):
subject = f"sub{t_subject:03d}"
session = f"ses{t_session:03d}"
(tmpdir / "func" / f"{subject}_single.nii").touch()
if t_subject == 2:
(tmpdir / "func" / f"{subject}_extra.nii").touch()
(tmpdir / "anat" / f"{subject}_{session}_ses.nii").touch()
# This should work, file now exists
datagrabber["sub001", "ses001"]
datagrabber_no_access[("sub001", "ses001")]
datagrabber = PatternDataGrabber(
datagrabber_multi_access = PatternDataGrabber(
datadir=tmpdir,
types=["func", "anat"],
types=["BOLD", "T1w"],
patterns={
"func": "func/{subject}_*.nii",
"anat": "anat/{subject}_{session}_*.nii",
"BOLD": {
"pattern": "func/{subject}_*.nii",
"space": "MNI152NLin6Asym",
},
"T1w": {
"pattern": "anat/{subject}_{session}_*.nii",
"space": "MNI152NLin6Asym",
},
},
replacements=["subject", "session"],
)
# access a subject with a missing session
with pytest.raises(ValueError, match="No file matches"):
datagrabber["sub001", "ses004"]
# Access a subject with a missing session
with pytest.raises(RuntimeError, match="No file matches"):
datagrabber_multi_access[("sub001", "ses004")]
# access a subject with two matching files
with pytest.raises(ValueError, match="More than one"):
datagrabber["sub002", "ses001"]
# Access a subject with two matching files
with pytest.raises(RuntimeError, match="More than one"):
datagrabber_multi_access[("sub002", "ses001")]
# access the right one
datagrabber["sub001", "ses001"]
# Access the right one
datagrabber_multi_access[("sub001", "ses001")]
datagrabber = PatternDataGrabber(
datagrabber_fake_access = PatternDataGrabber(
datadir=tmpdir,
types=["func", "anat2"],
types=["BOLD", "T1w"],
patterns={
"func": "func/{subject}_single.nii",
"anat2": "anat2/{subject}_{session}_ses.nii",
"BOLD": {
"pattern": "func/{subject}_single.nii",
"space": "MNI152NLin6Asym",
},
"T1w": {
"pattern": "anat2/{subject}_{session}_ses.nii",
"space": "MNI152NLin6Asym",
},
},
replacements=["subject", "session"],
)
assert len(datagrabber.get_elements()) == 0
assert len(datagrabber_fake_access.get_elements()) == 0
def test_PatternDataGrabber(tmp_path: Path) -> None:
@ -171,43 +116,59 @@ def test_PatternDataGrabber(tmp_path: Path) -> None:
"""
datagrabber = PatternDataGrabber(
datagrabber_first = PatternDataGrabber(
datadir="/tmp/data",
types=["func", "anat"],
patterns={"func": "func/{subject}.nii", "anat": "anat/{subject}.nii"},
types=["BOLD", "T1w"],
patterns={
"BOLD": {
"pattern": "func/{subject}.nii",
"space": "MNI152NLin6Asym",
},
"T1w": {
"pattern": "anat/{subject}.nii",
"space": "native",
},
},
replacements="subject",
)
assert datagrabber.datadir == Path("/tmp/data")
assert datagrabber.types == ["func", "anat"]
assert datagrabber.replacements == ["subject"]
assert datagrabber_first.datadir == Path("/tmp/data")
assert set(datagrabber_first.types) == {"T1w", "BOLD"}
assert datagrabber_first.replacements == ["subject"]
datagrabber = PatternDataGrabber(
datagrabber_second = PatternDataGrabber(
datadir=Path("/tmp/data"),
types=["func", "anat"],
types=["BOLD", "T1w"],
patterns={
"func": "func/{subject}.nii",
"anat": "anat/{subject}_{session}.nii",
"BOLD": {
"pattern": "func/{subject}.nii",
"space": "MNI152NLin6Asym",
},
"T1w": {
"pattern": "anat/{subject}_{session}.nii",
"space": "native",
},
},
replacements=["subject", "session"],
)
assert datagrabber.datadir == Path("/tmp/data")
assert datagrabber.types == ["func", "anat"]
assert datagrabber.replacements == ["subject", "session"]
assert datagrabber_second.datadir == Path("/tmp/data")
assert set(datagrabber_second.types) == {"T1w", "BOLD"}
assert datagrabber_second.replacements == ["subject", "session"]
# Create directories and files
tmpdir = tmp_path / "pattern_dg_test"
(tmpdir / "func").mkdir(exist_ok=True, parents=True)
(tmpdir / "anat").mkdir(exist_ok=True, parents=True)
(tmpdir / "vbm").mkdir(exist_ok=True, parents=True)
for t_subject in range(3):
for t_session in range(2):
for t_task in range(2, 4):
subject = f"sub{t_subject:03d}"
session = f"ses{t_session:03d}"
task = f"task{t_task:03d}"
if t_subject != 2:
(tmpdir / "func" / f"{subject}.nii").touch()
(tmpdir / "anat" / f"{subject}_{session}.nii").touch()
(tmpdir / "vbm" / f"{subject}_{task}_{session}.nii").touch()
for t_subject, t_session, t_task in product(
range(3), range(2), range(2, 4)
):
subject = f"sub{t_subject:03d}"
session = f"ses{t_session:03d}"
task = f"task{t_task:03d}"
if t_subject != 2:
(tmpdir / "func" / f"{subject}.nii").touch()
(tmpdir / "anat" / f"{subject}_{session}.nii").touch()
(tmpdir / "vbm" / f"{subject}_{task}_{session}.nii").touch()
expected_elements = [
("sub000", "ses000"),
@ -218,16 +179,19 @@ def test_PatternDataGrabber(tmp_path: Path) -> None:
("sub002", "ses001"),
]
datagrabber = PatternDataGrabber(
datagrabber_third = PatternDataGrabber(
datadir=tmpdir,
types=["anat"],
types=["T1w"],
patterns={
"anat": "anat/{subject}_{session}.nii",
"T1w": {
"pattern": "anat/{subject}_{session}.nii",
"space": "native",
},
},
replacements=["subject", "session"],
)
elements = datagrabber.get_elements()
elements = datagrabber_third.get_elements()
assert set(elements) == set(expected_elements)
expected_elements = [
@ -241,26 +205,35 @@ def test_PatternDataGrabber(tmp_path: Path) -> None:
("sub001", "ses001", "task003"),
]
datagrabber = PatternDataGrabber(
datagrabber_fourth = PatternDataGrabber(
datadir=tmpdir,
types=["func", "anat", "vbm"],
types=["T1w", "BOLD", "VBM_GM"],
patterns={
"func": "func/{subject}.nii",
"anat": "anat/{subject}_{session}.nii",
"vbm": "vbm/{subject}_{task}_{session}.nii",
"BOLD": {
"pattern": "func/{subject}.nii",
"space": "MNI152NLin6Asym",
},
"T1w": {
"pattern": "anat/{subject}_{session}.nii",
"space": "native",
},
"VBM_GM": {
"pattern": "vbm/{subject}_{task}_{session}.nii",
"space": "MNI152NLin6Asym",
},
},
replacements=["subject", "session", "task"],
)
elements = datagrabber.get_elements()
elements = datagrabber_fourth.get_elements()
assert set(elements) == set(expected_elements)
out1 = datagrabber[("sub000", "ses000", "task002")]
out2 = datagrabber[("sub000", "ses000", "task003")]
out1 = datagrabber_fourth[("sub000", "ses000", "task002")]
out2 = datagrabber_fourth[("sub000", "ses000", "task003")]
assert out1["func"]["path"] == out2["func"]["path"]
assert out1["anat"]["path"] == out2["anat"]["path"]
assert out1["vbm"]["path"] != out2["vbm"]["path"]
assert out1["BOLD"]["path"] == out2["BOLD"]["path"]
assert out1["T1w"]["path"] == out2["T1w"]["path"]
assert out1["VBM_GM"]["path"] != out2["VBM_GM"]["path"]
def test_PatternDataGrabber_confounds_format_error_on_init() -> None:
@ -269,40 +242,14 @@ def test_PatternDataGrabber_confounds_format_error_on_init() -> None:
ValueError, match="Invalid value for `confounds_format`"
):
PatternDataGrabber(
types=["func"],
patterns={"func": "func/{subject}.nii"},
types=["BOLD"],
patterns={
"BOLD": {
"pattern": "func/{subject}.nii",
"space": "MNI152NLin6Asym",
},
},
replacements=["subject"],
datadir="/tmp",
confounds_format="foobar",
)
def test_PatternDataGrabber_confounds_format_error_on_fetch(
tmp_path: Path,
) -> None:
"""Test PatterDataGrabber confounds format error on fetching.
Parameters
----------
tmp_path : pathlib.Path
The path to the test directory.
"""
# Create test directory path
tmpdir = tmp_path / "pattern_dg_test"
# Create final test directory
(tmpdir / "func" / "confounds").mkdir(exist_ok=True, parents=True)
# Create test confound file
(tmpdir / "func" / "confounds" / "sub-001.nii").touch()
# Initialize datagrabber
datagrabber = PatternDataGrabber(
types=["BOLD_confounds"],
patterns={"BOLD_confounds": "func/confounds/{subject}.nii"},
replacements=["subject"],
datadir=tmpdir,
)
# Check error on fetch
with pytest.raises(
ValueError, match="As the DataGrabber used specifies 'BOLD_confounds'"
):
datagrabber.get_item(subject="sub-001")

View file

@ -43,8 +43,14 @@ def test_bids_PatternDataladDataGrabber() -> None:
types = ["T1w", "BOLD"]
# Define patterns
patterns = {
"T1w": "{subject}/anat/{subject}_T1w.nii.gz",
"BOLD": "{subject}/func/{subject}_task-rest_bold.nii.gz",
"T1w": {
"pattern": "{subject}/anat/{subject}_T1w.nii.gz",
"space": "MNI152NLin6Asym",
},
"BOLD": {
"pattern": "{subject}/func/{subject}_task-rest_bold.nii.gz",
"space": "MNI152NLin6Asym",
},
}
# Define replacements
replacements = ["subject"]
@ -92,31 +98,28 @@ def test_bids_PatternDataladDataGrabber() -> None:
def test_bids_PatternDataladDataGrabber_datadir() -> None:
"""Test PatternDataladDataGrabber with a datadir set to a relative path."""
# Define types
types = ["T1w", "BOLD"]
# Define patterns
patterns = {
"T1w": "{subject}/anat/{subject}_T1w.nii.gz",
"BOLD": "{subject}/func/{subject}_task-rest_bold.nii.gz",
"T1w": {
"pattern": "{subject}/anat/{subject}_T*w.nii.gz",
"space": "MNI152NLin6Asym",
},
"BOLD": {
"pattern": "{subject}/func/{subject}_task-rest_*.nii.gz",
"space": "MNI152NLin6Asym",
},
}
# Define replacements
replacements = ["subject"]
repo_uri = _testing_dataset["example_bids"]["uri"]
# Define datadir
datadir = "dataset" # use string and not absolute path
patterns = {
"T1w": "example_bids/{subject}/anat/{subject}_T*w.nii.gz",
"BOLD": "example_bids/{subject}/func/{subject}_task-rest_*.nii.gz",
}
with PatternDataladDataGrabber(
uri=repo_uri,
types=types,
uri=_testing_dataset["example_bids"]["uri"],
types=["T1w", "BOLD"],
patterns=patterns,
datadir=datadir,
replacements=replacements,
rootdir="example_bids",
replacements=["subject"],
) as dg:
assert dg.datadir == Path(datadir)
assert dg.datadir == Path(datadir) / "example_bids"
for elem in dg:
t_sub = dg[elem]
assert "path" in t_sub["T1w"]
@ -133,12 +136,23 @@ def test_bids_PatternDataladDataGrabber_session():
"""Test a subject and session-based BIDS PatternDataladDataGrabber."""
types = ["T1w", "BOLD"]
patterns = {
"T1w": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"BOLD": "{subject}/{session}/func/"
"{subject}_{session}_task-rest_bold.nii.gz",
"T1w": {
"pattern": (
"{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz"
),
"space": "MNI152NLin6Asym",
},
"BOLD": {
"pattern": (
"{subject}/{session}/func/"
"{subject}_{session}_task-rest_bold.nii.gz"
),
"space": "MNI152NLin6Asym",
},
}
replacements = ["subject", "session"]
# Check error
with pytest.raises(ValueError, match=r"`uri` must be provided"):
PatternDataladDataGrabber(
datadir=None,
@ -147,9 +161,9 @@ def test_bids_PatternDataladDataGrabber_session():
replacements=replacements,
)
# Set parameters
repo_uri = _testing_dataset["example_bids_ses"]["uri"]
rootdir = "example_bids_ses"
# repo_commit = _testing_dataset['example_bids_ses']['id']
# With T1W and bold, only 2 sessions are available
with PatternDataladDataGrabber(
@ -159,7 +173,7 @@ def test_bids_PatternDataladDataGrabber_session():
patterns=patterns,
replacements=replacements,
) as dg:
subs = list(dg)
subs = list(dg.get_elements())
expected_subs = [
(f"sub-{i:02d}", f"ses-{j:02d}")
for j in range(1, 3)
@ -170,7 +184,12 @@ def test_bids_PatternDataladDataGrabber_session():
# Test with a different T1w only, it should have 3 sessions
types = ["T1w"]
patterns = {
"T1w": "{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz",
"T1w": {
"pattern": (
"{subject}/{session}/anat/{subject}_{session}_T1w.nii.gz"
),
"space": "MNI152NLin6Asym",
},
}
with PatternDataladDataGrabber(
rootdir=rootdir,

View file

@ -6,7 +6,68 @@
from typing import Dict, List
from ..utils import raise_error
from ..utils import logger, raise_error
# Define schema for pattern-based datagrabber's patterns
PATTERNS_SCHEMA = {
"T1w": {
"mandatory": ["pattern", "space"],
"optional": ["mask_item"],
},
"T1w_mask": {
"mandatory": ["pattern", "space"],
"optional": [],
},
"T2w": {
"mandatory": ["pattern", "space"],
"optional": ["mask_item"],
},
"T2w_mask": {
"mandatory": ["pattern", "space"],
"optional": [],
},
"BOLD": {
"mandatory": ["pattern", "space"],
"optional": ["mask_item"],
},
"BOLD_confounds": {
"mandatory": ["pattern", "format"],
"optional": [],
},
"BOLD_mask": {
"mandatory": ["pattern", "space"],
"optional": [],
},
"Warp": {
"mandatory": ["pattern", "src", "dst"],
"optional": [],
},
"VBM_GM": {
"mandatory": ["pattern", "space"],
"optional": [],
},
"VBM_WM": {
"mandatory": ["pattern", "space"],
"optional": [],
},
"probseg_CSF": {
"mandatory": ["pattern", "space"],
"optional": [],
},
"probseg_GM": {
"mandatory": ["pattern", "space"],
"optional": [],
},
"probseg_WM": {
"mandatory": ["pattern", "space"],
"optional": [],
},
"DWI": {
"mandatory": ["pattern"],
"optional": [],
},
}
def validate_types(types: List[str]) -> None:
@ -30,7 +91,7 @@ def validate_types(types: List[str]) -> None:
def validate_replacements(
replacements: List[str], patterns: Dict[str, str]
replacements: List[str], patterns: Dict[str, Dict[str, str]]
) -> None:
"""Validate the replacements.
@ -44,38 +105,41 @@ def validate_replacements(
Raises
------
TypeError
If ``replacements`` is not a list or if the values are not string or
if ``patterns`` is not a dictionary.
If ``replacements`` is not a list or if the values are not string.
ValueError
If a value in ``replacements`` is not in ``pattern`` or if no value in
``patterns`` contain all values in ``replacements``.
If a value in ``replacements`` is not part of a data type pattern or
if no data type patterns contain all values in ``replacements``.
"""
if not isinstance(replacements, list):
raise_error(msg="`replacements` must be a list.", klass=TypeError)
if not isinstance(patterns, dict):
raise_error(msg="`patterns` must be a dict.", klass=TypeError)
if any(not isinstance(x, str) for x in replacements):
raise_error(
msg="`replacements` must be a list of strings.", klass=TypeError
)
for x in replacements:
if all(x not in y for y in patterns.values()):
raise_error(msg=f"Replacement {x} is not part of any pattern.")
if all(
x not in y
for y in [
data_type_val["pattern"] for data_type_val in patterns.values()
]
):
raise_error(msg=f"Replacement: {x} is not part of any pattern.")
# Check that at least one pattern has all the replacements
at_least_one = False
for _, v in patterns.items():
if all(x in v for x in replacements):
for data_type_val in patterns.values():
if all(x in data_type_val["pattern"] for x in replacements):
at_least_one = True
if at_least_one is False:
raise_error(msg="At least one pattern must contain all replacements.")
def validate_patterns(types: List[str], patterns: Dict[str, str]) -> None:
def validate_patterns(
types: List[str], patterns: Dict[str, Dict[str, str]]
) -> None:
"""Validate the patterns.
Parameters
@ -87,12 +151,17 @@ def validate_patterns(types: List[str], patterns: Dict[str, str]) -> None:
Raises
------
KeyError
If any mandatory key is missing for a data type.
RuntimeError
If an unknown key is found for a data type.
TypeError
If ``patterns`` is not a dictionary.
ValueError
If length of ``types`` and ``patterns`` are different or
if ``patterns`` is missing entries from ``types`` or
if ``patterns`` contain '*' as value.
if unknown data type is found in ``patterns`` or
if data type pattern key contains '*' as value.
"""
# Validate the types
@ -110,9 +179,60 @@ def validate_patterns(types: List[str], patterns: Dict[str, str]) -> None:
raise_error(
msg="`patterns` must contain all `types`", klass=ValueError
)
# Wildcard check in patterns
if any("}*" in pattern for pattern in patterns.values()):
raise_error(
msg="`patterns` must not contain `*` following a replacement",
klass=ValueError,
)
# Check against schema
for data_type_key, data_type_val in patterns.items():
# Check if valid data type is provided
if data_type_key not in PATTERNS_SCHEMA:
raise_error(
f"Unknown data type: {data_type_key}, "
f"should be one of: {list(PATTERNS_SCHEMA.keys())}"
)
# Check mandatory keys for data type
for mandatory_key in PATTERNS_SCHEMA[data_type_key]["mandatory"]:
if mandatory_key not in data_type_val:
raise_error(
msg=(
f"Mandatory key: `{mandatory_key}` missing for "
f"{data_type_key}"
),
klass=KeyError,
)
else:
logger.debug(
f"Mandatory key: `{mandatory_key}` found for "
f"{data_type_key}"
)
# Check optional keys for data type
for optional_key in PATTERNS_SCHEMA[data_type_key]["optional"]:
if optional_key not in data_type_val:
logger.debug(
f"Optional key: `{optional_key}` missing for "
f"{data_type_key}"
)
else:
logger.debug(
f"Optional key: `{optional_key}` found for "
f"{data_type_key}"
)
# Check stray key for data type
for key in data_type_val.keys():
if key not in (
PATTERNS_SCHEMA[data_type_key]["mandatory"]
+ PATTERNS_SCHEMA[data_type_key]["optional"]
):
raise_error(
msg=(
f"Key: {key} not accepted for {data_type_key} "
"pattern, remove it to proceed"
),
klass=RuntimeError,
)
# Wildcard check in patterns
if "}*" in data_type_val["pattern"]:
raise_error(
msg=(
f"`{data_type_key}.pattern` must not contain `*` "
"following a replacement"
),
klass=ValueError,
)

View file

@ -53,23 +53,22 @@ class DefaultDataReader(PipelineStepMixin, UpdateMetaMixin):
# Nothing to validate, any input is fine
return input
def get_output_type(self, input: List[str]) -> List[str]:
def get_output_type(self, input_type: str) -> str:
"""Get output type.
Parameters
----------
input : list of str
The input to the reader. The list must contain the
available Junifer Data dictionary keys.
input_type : str
The data type input to the reader.
Returns
-------
list of str
The updated list of output types, as reading possibilities.
str
The data type output by the reader.
"""
# It will output the same type of data as the input
return input
return input_type
def _fit_transform(
self,