[ENH]: Allow users to select types for datagrabbers in order to avoid downloading unnecesary data. #132

Merged
synchon merged 18 commits from update/types-for-aomic-dg into main 2023-07-21 11:28:24 +00:00
13 changed files with 584 additions and 340 deletions

View file

@ -0,0 +1 @@
Expose ``types`` parameter for :class:`.DataladAOMICID1000`, :class:`.DataladAOMICPIOP1`, :class:`.DataladAOMICPIOP2` and :class:`.JuselessUCLA` by `Synchon Mandal`_

View file

@ -0,0 +1 @@
Change validation of ``types`` against ``patterns`` to allow a subset of ``patterns``'s types to be used for ``DataGrabber`` data fetch by `Synchon Mandal`_

View file

@ -6,20 +6,17 @@
# License: AGPL # License: AGPL
import socket import socket
from typing import Optional from typing import List, Optional, Union
import pytest import pytest
from junifer.configs.juseless.datagrabbers import JuselessUCLA from junifer.configs.juseless.datagrabbers import JuselessUCLA
from junifer.utils.logging import configure_logging
# Check if the test is running on juseless # Check if the test is running on juseless
if socket.gethostname() != "juseless": if socket.gethostname() != "juseless":
pytest.skip("These tests are only for juseless", allow_module_level=True) pytest.skip("These tests are only for juseless", allow_module_level=True)
configure_logging(level="DEBUG")
def test_JuselessUCLA() -> None: def test_JuselessUCLA() -> None:
"""Test JuselessUCLA.""" """Test JuselessUCLA."""
@ -42,6 +39,57 @@ def test_JuselessUCLA() -> None:
assert out[t]["path"].exists() assert out[t]["path"].exists()
@pytest.mark.parametrize(
"types",
[
"BOLD",
"BOLD_confounds",
"T1w",
"probseg_CSF",
"probseg_GM",
"probseg_WM",
["BOLD", "BOLD_confounds"],
["T1w", "probseg_CSF"],
["probseg_GM", "probseg_WM"],
["BOLD", "T1w"],
],
)
def test_JuselessUCLA_partial_data_access(
types: Union[str, List[str]],
) -> None:
"""Test JuselessUCLA DataGrabber partial data access.
Parameters
----------
types : str or list of str
The parametrized types.
"""
dg = JuselessUCLA(types=types)
with dg:
# Get all elements
all_elements = dg.get_elements()
# Get test element
test_element = all_elements[0]
# Get test element data
out = dg[test_element]
# Assert data type
if isinstance(types, list):
for type_ in types:
assert type_ in out
else:
assert types in out
def test_JuselessUCLA_incorrect_data_type() -> None:
"""Test JuselessUCLA DataGrabber incorrect data type."""
with pytest.raises(
ValueError, match="`patterns` must contain all `types`"
):
_ = JuselessUCLA(types="Eunomia")
@pytest.mark.parametrize( @pytest.mark.parametrize(
"tasks", "tasks",
[None, "rest", ["rest", "stopsignal"]], [None, "rest", ["rest", "stopsignal"]],

View file

@ -20,9 +20,13 @@ class JuselessUCLA(PatternDataGrabber):
Parameters Parameters
---------- ----------
datadir : str or pathlib.Path, optional datadir : str or Path, optional
The directory where the dataset is stored The directory where the dataset is stored.
(default "/data/project/psychosis_thalamus/data/fmriprep"). (default "/data/project/psychosis_thalamus/data/fmriprep").
types: {"BOLD", "BOLD_confounds", "T1w", "probseg_CSF", "probseg_GM", \
"probseg_WM"} or a list of the options, optional
UCLA data types. If None, all available data types are selected.
(default None).
tasks : {"rest", "bart", "bht", "pamenc", "pamret", \ tasks : {"rest", "bart", "bht", "pamenc", "pamret", \
"scap", "taskswitch", "stopsignal"} or \ "scap", "taskswitch", "stopsignal"} or \
list of the options or None, optional list of the options or None, optional
@ -36,20 +40,10 @@ class JuselessUCLA(PatternDataGrabber):
datadir: Union[ datadir: Union[
str, Path str, Path
] = "/data/project/psychosis_thalamus/data/fmriprep", ] = "/data/project/psychosis_thalamus/data/fmriprep",
types: Union[str, List[str], None] = None,
tasks: Union[str, List[str], None] = None, tasks: Union[str, List[str], None] = None,
) -> None: ) -> None:
types = [ # Declare all tasks
"BOLD",
"BOLD_confounds",
"T1w",
"probseg_CSF",
"probseg_GM",
"probseg_WM",
]
if isinstance(tasks, str):
tasks = [tasks]
all_tasks = [ all_tasks = [
"rest", "rest",
"bart", "bart",
@ -60,18 +54,21 @@ class JuselessUCLA(PatternDataGrabber):
"taskswitch", "taskswitch",
"stopsignal", "stopsignal",
] ]
# Set default tasks
if tasks is None: if tasks is None:
tasks = all_tasks tasks = all_tasks
else: else:
# Convert single task into list
if isinstance(tasks, str):
tasks = [tasks]
# Verify valid tasks
for t in tasks: for t in tasks:
if t not in all_tasks: if t not in all_tasks:
raise_error( raise_error(
f"{t} is not a valid task in the UCLA dataset!" f"{t} is not a valid task in the UCLA dataset!"
) )
self.tasks = tasks self.tasks = tasks
# The patterns
patterns = { patterns = {
"BOLD": ( "BOLD": (
"sub-{subject}/func/sub-{subject}_task-{task}_bold_space-" "sub-{subject}/func/sub-{subject}_task-{task}_bold_space-"
@ -98,12 +95,18 @@ class JuselessUCLA(PatternDataGrabber):
"-MNI152NLin2009cAsym_class-WM_probtissue.nii.gz" "-MNI152NLin2009cAsym_class-WM_probtissue.nii.gz"
), ),
} }
# Set default types
if types is None:
types = list(patterns.keys())
# Convert single type into list
else:
if not isinstance(types, list):
types = [types]
# The replacements
replacements = ["subject", "task"]
# the commented out uri leads to new open neuro dataset which does # the commented out uri leads to new open neuro dataset which does
# NOT have preprocessed data # NOT have preprocessed data
# uri = "https://github.com/OpenNeuroDatasets/ds000030.git" # uri = "https://github.com/OpenNeuroDatasets/ds000030.git"
replacements = ["subject", "task"]
super().__init__( super().__init__(
types=types, types=types,
datadir=datadir, datadir=datadir,

View file

@ -4,10 +4,11 @@
# Vera Komeyer <v.komeyer@fz-juelich.de> # Vera Komeyer <v.komeyer@fz-juelich.de>
# Xuan Li <xu.li@fz-juelich.de> # Xuan Li <xu.li@fz-juelich.de>
# Leonard Sasse <l.sasse@fz-juelich.de> # Leonard Sasse <l.sasse@fz-juelich.de>
# Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL # License: AGPL
from pathlib import Path from pathlib import Path
from typing import Dict, Union from typing import Dict, List, Union
from ...api.decorators import register_datagrabber from ...api.decorators import register_datagrabber
from ..pattern_datalad import PatternDataladDataGrabber from ..pattern_datalad import PatternDataladDataGrabber
@ -23,25 +24,18 @@ class DataladAOMICID1000(PatternDataladDataGrabber):
The directory where the datalad dataset will be cloned. If None, The directory where the datalad dataset will be cloned. If None,
the datalad dataset will be cloned into a temporary directory the datalad dataset will be cloned into a temporary directory
(default None). (default None).
types: {"BOLD", "BOLD_confounds", "T1w", "probseg_CSF", "probseg_GM", \
"probseg_WM", "DWI"} or a list of the options, optional
AOMIC data types. If None, all available data types are selected.
(default None).
""" """
def __init__( def __init__(
self, self,
datadir: Union[str, Path, None] = None, datadir: Union[str, Path, None] = None,
types: Union[str, List[str], None] = None,
) -> None: ) -> None:
# The types of data
types = [
"BOLD",
"BOLD_confounds",
"BOLD_mask",
"T1w",
"T1w_mask",
"probseg_CSF",
"probseg_GM",
"probseg_WM",
"DWI",
]
# The patterns # The patterns
patterns = { patterns = {
"BOLD": ( "BOLD": (
@ -90,6 +84,13 @@ class DataladAOMICID1000(PatternDataladDataGrabber):
"sub-{subject}_desc-preproc_dwi.nii.gz" "sub-{subject}_desc-preproc_dwi.nii.gz"
), ),
} }
# Set default types
if types is None:
types = list(patterns.keys())
# Convert single type into list
else:
if not isinstance(types, list):
types = [types]
# The replacements # The replacements
replacements = ["subject"] replacements = ["subject"]
uri = "https://github.com/OpenNeuroDatasets/ds003097.git" uri = "https://github.com/OpenNeuroDatasets/ds003097.git"
@ -118,6 +119,8 @@ class DataladAOMICID1000(PatternDataladDataGrabber):
""" """
out = super().get_item(subject=subject) out = super().get_item(subject=subject)
out["BOLD"]["mask_item"] = "BOLD_mask" if out.get("BOLD"):
out["T1w"]["mask_item"] = "T1w_mask" out["BOLD"]["mask_item"] = "BOLD_mask"
if out.get("T1w"):
out["T1w"]["mask_item"] = "T1w_mask"
return out return out

View file

@ -4,6 +4,7 @@
# Vera Komeyer <v.komeyer@fz-juelich.de> # Vera Komeyer <v.komeyer@fz-juelich.de>
# Xuan Li <xu.li@fz-juelich.de> # Xuan Li <xu.li@fz-juelich.de>
# Leonard Sasse <l.sasse@fz-juelich.de> # Leonard Sasse <l.sasse@fz-juelich.de>
# Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL # License: AGPL
from itertools import product from itertools import product
@ -25,6 +26,10 @@ class DataladAOMICPIOP1(PatternDataladDataGrabber):
The directory where the datalad dataset will be cloned. If None, The directory where the datalad dataset will be cloned. If None,
the datalad dataset will be cloned into a temporary directory the datalad dataset will be cloned into a temporary directory
(default None). (default None).
types: {"BOLD", "BOLD_confounds", "T1w", "probseg_CSF", "probseg_GM", \
"probseg_WM", "DWI"} or a list of the options, optional
AOMIC data types. If None, all available data types are selected.
(default None).
tasks : {"restingstate", "anticipation", "emomatching", "faces", \ tasks : {"restingstate", "anticipation", "emomatching", "faces", \
"gstroop", "workingmemory"} or list of the options, optional "gstroop", "workingmemory"} or list of the options, optional
AOMIC PIOP1 task sessions. If None, all available task sessions are AOMIC PIOP1 task sessions. If None, all available task sessions are
@ -35,24 +40,10 @@ class DataladAOMICPIOP1(PatternDataladDataGrabber):
def __init__( def __init__(
self, self,
datadir: Union[str, Path, None] = None, datadir: Union[str, Path, None] = None,
types: Union[str, List[str], None] = None,
tasks: Union[str, List[str], None] = None, tasks: Union[str, List[str], None] = None,
) -> None: ) -> None:
# The types of data # Declare all tasks
types = [
"BOLD",
"BOLD_confounds",
"BOLD_mask",
"T1w",
"T1w_mask",
"probseg_CSF",
"probseg_GM",
"probseg_WM",
"DWI",
]
if isinstance(tasks, str):
tasks = [tasks]
all_tasks = [ all_tasks = [
"restingstate", "restingstate",
"anticipation", "anticipation",
@ -61,19 +52,22 @@ class DataladAOMICPIOP1(PatternDataladDataGrabber):
"gstroop", "gstroop",
"workingmemory", "workingmemory",
] ]
# Set default tasks
if tasks is None: if tasks is None:
tasks = all_tasks tasks = all_tasks
else: else:
# Convert single task into list
if isinstance(tasks, str):
tasks = [tasks]
# Verify valid tasks
for t in tasks: for t in tasks:
if t not in all_tasks: if t not in all_tasks:
raise_error( raise_error(
f"{t} is not a valid task in the AOMIC PIOP1" f"{t} is not a valid task in the AOMIC PIOP1"
" dataset!" " dataset!"
) )
self.tasks = tasks self.tasks = tasks
# The patterns
patterns = { patterns = {
"BOLD": ( "BOLD": (
"derivatives/fmriprep/sub-{subject}/func/" "derivatives/fmriprep/sub-{subject}/func/"
@ -120,8 +114,16 @@ class DataladAOMICPIOP1(PatternDataladDataGrabber):
"sub-{subject}_desc-preproc_dwi.nii.gz" "sub-{subject}_desc-preproc_dwi.nii.gz"
), ),
} }
uri = "https://github.com/OpenNeuroDatasets/ds002785" # Set default types
if types is None:
types = list(patterns.keys())
# Convert single type into list
else:
if not isinstance(types, list):
types = [types]
# The replacements
replacements = ["subject", "task"] replacements = ["subject", "task"]
uri = "https://github.com/OpenNeuroDatasets/ds002785"
super().__init__( super().__init__(
types=types, types=types,
datadir=datadir, datadir=datadir,
@ -162,8 +164,10 @@ class DataladAOMICPIOP1(PatternDataladDataGrabber):
new_task = f"{task}_acq-{acq}" new_task = f"{task}_acq-{acq}"
out = super().get_item(subject=subject, task=new_task) out = super().get_item(subject=subject, task=new_task)
out["BOLD"]["mask_item"] = "BOLD_mask" if out.get("BOLD"):
out["T1w"]["mask_item"] = "T1w_mask" out["BOLD"]["mask_item"] = "BOLD_mask"
if out.get("T1w"):
out["T1w"]["mask_item"] = "T1w_mask"
return out return out
def get_elements(self) -> List: def get_elements(self) -> List:

View file

@ -4,8 +4,10 @@
# Vera Komeyer <v.komeyer@fz-juelich.de> # Vera Komeyer <v.komeyer@fz-juelich.de>
# Xuan Li <xu.li@fz-juelich.de> # Xuan Li <xu.li@fz-juelich.de>
# Leonard Sasse <l.sasse@fz-juelich.de> # Leonard Sasse <l.sasse@fz-juelich.de>
# Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL # License: AGPL
from itertools import product
from pathlib import Path from pathlib import Path
from typing import Dict, List, Union from typing import Dict, List, Union
@ -24,7 +26,11 @@ class DataladAOMICPIOP2(PatternDataladDataGrabber):
The directory where the datalad dataset will be cloned. If None, The directory where the datalad dataset will be cloned. If None,
the datalad dataset will be cloned into a temporary directory the datalad dataset will be cloned into a temporary directory
(default None). (default None).
tasks : {"restingstate", "stopsignal", "emomatching", "workingmemory"} \ types: {"BOLD", "BOLD_confounds", "T1w", "probseg_CSF", "probseg_GM", \
"probseg_WM", "DWI"} or a list of the options, optional
AOMIC data types. If None, all available data types are selected.
(default None).
tasks : {"restingstate", "stopsignal", "workingmemory"} \
or list of the options, optional or list of the options, optional
AOMIC PIOP2 task sessions. If None, all available task sessions are AOMIC PIOP2 task sessions. If None, all available task sessions are
selected (default None). selected (default None).
@ -34,58 +40,46 @@ class DataladAOMICPIOP2(PatternDataladDataGrabber):
def __init__( def __init__(
self, self,
datadir: Union[str, Path, None] = None, datadir: Union[str, Path, None] = None,
types: Union[str, List[str], None] = None,
tasks: Union[str, List[str], None] = None, tasks: Union[str, List[str], None] = None,
) -> None: ) -> None:
# The types of data # Declare all tasks
types = [
"BOLD",
"BOLD_confounds",
"BOLD_mask",
"T1w",
"T1w_mask",
"probseg_CSF",
"probseg_GM",
"probseg_WM",
"DWI",
]
if isinstance(tasks, str):
tasks = [tasks]
all_tasks = [ all_tasks = [
"restingstate", "restingstate",
"emomatching",
"workingmemory",
"stopsignal", "stopsignal",
"workingmemory",
] ]
# Set default tasks
if tasks is None: if tasks is None:
tasks = all_tasks tasks = all_tasks
else: else:
# Convert single task into list
if isinstance(tasks, str):
tasks = [tasks]
# Verify valid tasks
for t in tasks: for t in tasks:
if t not in all_tasks: if t not in all_tasks:
raise_error( raise_error(
f"{t} is not a valid task in the AOMIC PIOP2" f"{t} is not a valid task in the AOMIC PIOP2"
" dataset!" " dataset!"
) )
self.tasks = tasks self.tasks = tasks
# The patterns
patterns = { patterns = {
"BOLD": ( "BOLD": (
"derivatives/fmriprep/sub-{subject}/func/" "derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_acq-seq_" "sub-{subject}_task-{task}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz" "space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
), ),
"BOLD_confounds": ( "BOLD_confounds": (
"derivatives/fmriprep/sub-{subject}/func/" "derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_acq-seq_" "sub-{subject}_task-{task}_"
"desc-confounds_regressors.tsv" "desc-confounds_regressors.tsv"
), ),
"BOLD_mask": ( "BOLD_mask": (
"derivatives/fmriprep/sub-{subject}/func/" "derivatives/fmriprep/sub-{subject}/func/"
"sub-{subject}_task-{task}_acq-seq_space" "sub-{subject}_task-{task}_"
"-MNI152NLin2009cAsym_desc-brain_mask.nii.gz" "space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
), ),
"T1w": ( "T1w": (
"derivatives/fmriprep/sub-{subject}/anat/" "derivatives/fmriprep/sub-{subject}/anat/"
@ -117,8 +111,16 @@ class DataladAOMICPIOP2(PatternDataladDataGrabber):
"sub-{subject}_desc-preproc_dwi.nii.gz" "sub-{subject}_desc-preproc_dwi.nii.gz"
), ),
} }
uri = "https://github.com/OpenNeuroDatasets/ds002790" # Set default types
if types is None:
types = list(patterns.keys())
# Convert single type into list
else:
if not isinstance(types, list):
types = [types]
# The replacements
replacements = ["subject", "task"] replacements = ["subject", "task"]
uri = "https://github.com/OpenNeuroDatasets/ds002790"
super().__init__( super().__init__(
types=types, types=types,
datadir=datadir, datadir=datadir,
@ -138,8 +140,11 @@ class DataladAOMICPIOP2(PatternDataladDataGrabber):
imposing constraints based on specified tasks. imposing constraints based on specified tasks.
""" """
all_elements = super().get_elements() subjects = [f"{x:04d}" for x in range(1, 227)]
return [x for x in all_elements if x[1] in self.tasks] elems = []
for subject, task in product(subjects, self.tasks):
elems.append((subject, task))
return elems
def get_item(self, subject: str, task: str) -> Dict: def get_item(self, subject: str, task: str) -> Dict:
"""Index one element in the dataset. """Index one element in the dataset.
@ -148,9 +153,8 @@ class DataladAOMICPIOP2(PatternDataladDataGrabber):
---------- ----------
subject : str subject : str
The subject ID. The subject ID.
task : str task : {"restingstate", "stopsignal", "workingmemory"}
The task to get. Possible values are: The task to get.
{"restingstate", "stopsignal", "emomatching", "workingmemory"}
Returns Returns
------- -------
@ -159,7 +163,9 @@ class DataladAOMICPIOP2(PatternDataladDataGrabber):
specified element. specified element.
""" """
out = super().get_item(subject=subject, task=task) out = super().get_item(subject=subject, task=f"{task}_acq-seq")
out["BOLD"]["mask_item"] = "BOLD_mask" if out.get("BOLD"):
out["T1w"]["mask_item"] = "T1w_mask" out["BOLD"]["mask_item"] = "BOLD_mask"
if out.get("T1w"):
out["T1w"]["mask_item"] = "T1w_mask"
return out return out

View file

@ -4,22 +4,24 @@
# Vera Komeyer <v.komeyer@fz-juelich.de> # Vera Komeyer <v.komeyer@fz-juelich.de>
# Xuan Li <xu.li@fz-juelich.de> # Xuan Li <xu.li@fz-juelich.de>
# Leonard Sasse <l.sasse@fz-juelich.de> # Leonard Sasse <l.sasse@fz-juelich.de>
# Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL # License: AGPL
from junifer.datagrabber import DataladAOMICID1000 from typing import List, Union
from junifer.utils import configure_logging
import pytest
from junifer.datagrabber.aomic.id1000 import DataladAOMICID1000
URI = "https://gin.g-node.org/juaml/datalad-example-aomic1000"
def test_DataladAOMICID1000() -> None: def test_DataladAOMICID1000() -> None:
"""Test DataladAOMICID1000 DataGrabber.""" """Test DataladAOMICID1000 DataGrabber."""
uri_ID1000 = "https://gin.g-node.org/juaml/datalad-example-aomic1000"
configure_logging(level="DEBUG")
dg = DataladAOMICID1000() dg = DataladAOMICID1000()
# Set URI to Gin
# change uri here to use fake data instead of real dataset dg.uri = URI
dg.uri = uri_ID1000
with dg: with dg:
all_elements = dg.get_elements() all_elements = dg.get_elements()
@ -122,3 +124,57 @@ def test_DataladAOMICID1000() -> None:
assert "element" in meta assert "element" in meta
assert "subject" in meta["element"] assert "subject" in meta["element"]
assert test_element == meta["element"]["subject"] assert test_element == meta["element"]["subject"]
@pytest.mark.parametrize(
"types",
[
"BOLD",
"BOLD_confounds",
"T1w",
"probseg_CSF",
"probseg_GM",
"probseg_WM",
"DWI",
["BOLD", "BOLD_confounds"],
["T1w", "probseg_CSF"],
["probseg_GM", "probseg_WM"],
["DWI", "BOLD"],
],
)
def test_DataladAOMICID1000_partial_data_access(
types: Union[str, List[str]],
) -> None:
"""Test DataladAOMICID1000 DataGrabber partial data access.
Parameters
----------
types : str or list of str
The parametrized types.
"""
dg = DataladAOMICID1000(types=types)
# Set URI to Gin
dg.uri = URI
with dg:
# Get all elements
all_elements = dg.get_elements()
# Get test element
test_element = all_elements[0]
# Get test element data
out = dg[test_element]
# Assert data type
if isinstance(types, list):
for type_ in types:
assert type_ in out
else:
assert types in out
def test_DataladAOMICID1000_incorrect_data_type() -> None:
"""Test DataladAOMICID1000 DataGrabber incorrect data type."""
with pytest.raises(
ValueError, match="`patterns` must contain all `types`"
):
_ = DataladAOMICID1000(types="Scooby-Doo")

View file

@ -4,142 +4,201 @@
# Vera Komeyer <v.komeyer@fz-juelich.de> # Vera Komeyer <v.komeyer@fz-juelich.de>
# Xuan Li <xu.li@fz-juelich.de> # Xuan Li <xu.li@fz-juelich.de>
# Leonard Sasse <l.sasse@fz-juelich.de> # Leonard Sasse <l.sasse@fz-juelich.de>
# Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL # License: AGPL
from typing import List, Optional, Union
import pytest import pytest
from junifer.datagrabber import DataladAOMICPIOP1 from junifer.datagrabber import DataladAOMICPIOP1
from junifer.utils import configure_logging
def test_DataladAOMICPIOP1() -> None: URI = "https://gin.g-node.org/juaml/datalad-example-aomicpiop1"
"""Test DataladAOMICPIOP1 DataGrabber."""
configure_logging(level="DEBUG")
uri_PIOP1 = "https://gin.g-node.org/juaml/datalad-example-aomicpiop1"
task_params = [None, "restingstate"]
for task_param in task_params: @pytest.mark.parametrize(
dg = DataladAOMICPIOP1(tasks=task_param) "tasks",
[None, "restingstate"],
)
def test_DataladAOMICPIOP1(tasks: Optional[str]) -> None:
"""Test DataladAOMICPIOP1 DataGrabber.
# change uri here to use fake data instead of real dataset Parameters
dg.uri = uri_PIOP1 ----------
tasks : str or None
The parametrized task values.
with dg: """
all_elements = dg.get_elements() dg = DataladAOMICPIOP1(tasks=tasks)
test_element = all_elements[0] # Set URI to Gin
sub, task = test_element dg.uri = URI
out = dg[test_element] with dg:
all_elements = dg.get_elements()
test_element = all_elements[0]
sub, task = test_element
# asserts type "BOLD" out = dg[test_element]
assert "BOLD" in out
# depending on task 'acquisition is different' # asserts type "BOLD"
task_acqs = { assert "BOLD" in out
"anticipation": "seq",
"emomatching": "seq",
"faces": "mb3",
"gstroop": "seq",
"restingstate": "mb3",
"workingmemory": "seq",
}
acq = task_acqs[task]
new_task = f"{task}_acq-{acq}"
assert (
out["BOLD"]["path"].name == f"sub-{sub}_task-{new_task}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
)
assert out["BOLD"]["path"].exists() # depending on task 'acquisition is different'
assert out["BOLD"]["path"].is_file() task_acqs = {
"anticipation": "seq",
"emomatching": "seq",
"faces": "mb3",
"gstroop": "seq",
"restingstate": "mb3",
"workingmemory": "seq",
}
acq = task_acqs[task]
new_task = f"{task}_acq-{acq}"
assert (
out["BOLD"]["path"].name == f"sub-{sub}_task-{new_task}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
)
# asserts type "BOLD_confounds" assert out["BOLD"]["path"].exists()
assert "BOLD_confounds" in out assert out["BOLD"]["path"].is_file()
assert ( # asserts type "BOLD_confounds"
out["BOLD_confounds"]["path"].name assert "BOLD_confounds" in out
== f"sub-{sub}_task-{new_task}_"
"desc-confounds_regressors.tsv"
)
assert out["BOLD_confounds"]["path"].exists() assert (
assert out["BOLD_confounds"]["path"].is_file() out["BOLD_confounds"]["path"].name == f"sub-{sub}_task-{new_task}_"
"desc-confounds_regressors.tsv"
)
# assert BOLD_mask assert out["BOLD_confounds"]["path"].exists()
assert out["BOLD_mask"]["path"].exists() assert out["BOLD_confounds"]["path"].is_file()
# asserts type "T1w" # assert BOLD_mask
assert "T1w" in out assert out["BOLD_mask"]["path"].exists()
assert ( # asserts type "T1w"
out["T1w"]["path"].name assert "T1w" in out
== f"sub-{sub}_space-MNI152NLin2009cAsym_"
"desc-preproc_T1w.nii.gz"
)
assert out["T1w"]["path"].exists() assert (
assert out["T1w"]["path"].is_file() out["T1w"]["path"].name == f"sub-{sub}_space-MNI152NLin2009cAsym_"
"desc-preproc_T1w.nii.gz"
)
# asserts T1w_mask assert out["T1w"]["path"].exists()
assert out["T1w_mask"]["path"].exists() assert out["T1w"]["path"].is_file()
# asserts type "probseg_CSF" # asserts T1w_mask
assert "probseg_CSF" in out assert out["T1w_mask"]["path"].exists()
assert ( # asserts type "probseg_CSF"
out["probseg_CSF"]["path"].name assert "probseg_CSF" in out
== f"sub-{sub}_space-MNI152NLin2009cAsym_label-"
"CSF_probseg.nii.gz"
)
assert out["probseg_CSF"]["path"].exists() assert (
assert out["probseg_CSF"]["path"].is_file() out["probseg_CSF"]["path"].name
== f"sub-{sub}_space-MNI152NLin2009cAsym_label-"
"CSF_probseg.nii.gz"
)
# asserts type "probseg_GM" assert out["probseg_CSF"]["path"].exists()
assert "probseg_GM" in out assert out["probseg_CSF"]["path"].is_file()
assert ( # asserts type "probseg_GM"
out["probseg_GM"]["path"].name assert "probseg_GM" in out
== f"sub-{sub}_space-MNI152NLin2009cAsym_label-"
"GM_probseg.nii.gz"
)
assert out["probseg_GM"]["path"].exists() assert (
assert out["probseg_GM"]["path"].is_file() out["probseg_GM"]["path"].name
== f"sub-{sub}_space-MNI152NLin2009cAsym_label-"
"GM_probseg.nii.gz"
)
# asserts type "probseg_WM" assert out["probseg_GM"]["path"].exists()
assert "probseg_WM" in out assert out["probseg_GM"]["path"].is_file()
assert ( # asserts type "probseg_WM"
out["probseg_WM"]["path"].name assert "probseg_WM" in out
== f"sub-{sub}_space-MNI152NLin2009cAsym_label-"
"WM_probseg.nii.gz"
)
assert out["probseg_WM"]["path"].exists() assert (
assert out["probseg_WM"]["path"].is_file() out["probseg_WM"]["path"].name
== f"sub-{sub}_space-MNI152NLin2009cAsym_label-"
"WM_probseg.nii.gz"
)
# asserts type "DWI" assert out["probseg_WM"]["path"].exists()
assert "DWI" in out assert out["probseg_WM"]["path"].is_file()
assert ( # asserts type "DWI"
out["DWI"]["path"].name == f"sub-{sub}_desc-preproc_dwi.nii.gz" assert "DWI" in out
)
assert out["DWI"]["path"].exists() assert out["DWI"]["path"].name == f"sub-{sub}_desc-preproc_dwi.nii.gz"
assert out["DWI"]["path"].is_file()
# asserts meta assert out["DWI"]["path"].exists()
assert "meta" in out["BOLD"] assert out["DWI"]["path"].is_file()
meta = out["BOLD"]["meta"]
assert "element" in meta # asserts meta
assert "subject" in meta["element"] assert "meta" in out["BOLD"]
assert sub == meta["element"]["subject"] meta = out["BOLD"]["meta"]
assert "element" in meta
assert "subject" in meta["element"]
assert sub == meta["element"]["subject"]
@pytest.mark.parametrize(
"types",
[
"BOLD",
"BOLD_confounds",
"T1w",
"probseg_CSF",
"probseg_GM",
"probseg_WM",
"DWI",
["BOLD", "BOLD_confounds"],
["T1w", "probseg_CSF"],
["probseg_GM", "probseg_WM"],
["DWI", "BOLD"],
],
)
def test_DataladAOMICPIOP1_partial_data_access(
types: Union[str, List[str]],
) -> None:
"""Test DataladAOMICPIOP1 DataGrabber partial data access.
Parameters
----------
types : str or list of str
The parametrized types.
"""
dg = DataladAOMICPIOP1(types=types)
# Set URI to Gin
dg.uri = URI
with dg:
# Get all elements
all_elements = dg.get_elements()
# Get test element
test_element = all_elements[0]
# Get test element data
out = dg[test_element]
# Assert data type
if isinstance(types, list):
for type_ in types:
assert type_ in out
else:
assert types in out
def test_DataladAOMICPIOP1_incorrect_data_type() -> None:
"""Test DataladAOMICPIOP1 DataGrabber incorrect data type."""
with pytest.raises(
ValueError, match="`patterns` must contain all `types`"
):
_ = DataladAOMICPIOP1(types="Ceres")
def test_DataladAOMICPIOP1_invalid_tasks(): def test_DataladAOMICPIOP1_invalid_tasks():
"""Test whether invalid task fails.""" """Test DataladAOMICIDPIOP1 DataGrabber invalid tasks."""
with pytest.raises( with pytest.raises(
ValueError, ValueError,
match=( match=(

View file

@ -4,136 +4,195 @@
# Vera Komeyer <v.komeyer@fz-juelich.de> # Vera Komeyer <v.komeyer@fz-juelich.de>
# Xuan Li <xu.li@fz-juelich.de> # Xuan Li <xu.li@fz-juelich.de>
# Leonard Sasse <l.sasse@fz-juelich.de> # Leonard Sasse <l.sasse@fz-juelich.de>
# Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL # License: AGPL
from typing import List, Optional, Union
import pytest import pytest
from junifer.datagrabber import DataladAOMICPIOP2 from junifer.datagrabber import DataladAOMICPIOP2
from junifer.utils import configure_logging
def test_DataladAOMICPIOP2() -> None: URI = "https://gin.g-node.org/juaml/datalad-example-aomicpiop2"
"""Test DataladAOMICPIOP2 DataGrabber."""
configure_logging(level="DEBUG")
uri_PIOP2 = "https://gin.g-node.org/juaml/datalad-example-aomicpiop2"
task_params = [None, "restingstate"]
for task_param in task_params: @pytest.mark.parametrize(
dg = DataladAOMICPIOP2(tasks=task_param) "tasks",
[None, "restingstate"],
)
def test_DataladAOMICPIOP2(tasks: Optional[str]) -> None:
"""Test DataladAOMICPIOP2 DataGrabber.
# change uri here to use fake data instead of real dataset Parameters
dg.uri = uri_PIOP2 ----------
tasks : str or None
The parametrized task values.
with dg: """
all_elements = dg.get_elements() dg = DataladAOMICPIOP2(tasks=tasks)
# Set URI to Gin
dg.uri = URI
if task_param == "restingstate": with dg:
for el in all_elements: all_elements = dg.get_elements()
assert el[1] == "restingstate"
test_element = all_elements[0] if tasks == "restingstate":
sub, task = test_element for el in all_elements:
out = dg[test_element] assert el[1] == "restingstate"
# asserts type "BOLD" test_element = all_elements[0]
assert "BOLD" in out sub, task = test_element
out = dg[test_element]
new_task = f"{task}_acq-seq" # asserts type "BOLD"
assert ( assert "BOLD" in out
out["BOLD"]["path"].name == f"sub-{sub}_task-{new_task}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
)
assert out["BOLD"]["path"].exists() new_task = f"{task}_acq-seq"
assert out["BOLD"]["path"].is_file() assert (
out["BOLD"]["path"].name == f"sub-{sub}_task-{new_task}_"
"space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
)
# asserts type "BOLD_confounds" assert out["BOLD"]["path"].exists()
assert "BOLD_confounds" in out assert out["BOLD"]["path"].is_file()
assert ( # asserts type "BOLD_confounds"
out["BOLD_confounds"]["path"].name assert "BOLD_confounds" in out
== f"sub-{sub}_task-{new_task}_"
"desc-confounds_regressors.tsv"
)
assert out["BOLD_confounds"]["path"].exists() assert (
assert out["BOLD_confounds"]["path"].is_file() out["BOLD_confounds"]["path"].name == f"sub-{sub}_task-{new_task}_"
"desc-confounds_regressors.tsv"
)
# assert BOLD_mask assert out["BOLD_confounds"]["path"].exists()
assert out["BOLD_mask"]["path"].exists() assert out["BOLD_confounds"]["path"].is_file()
# asserts type "T1w" # assert BOLD_mask
assert "T1w" in out assert out["BOLD_mask"]["path"].exists()
assert ( # asserts type "T1w"
out["T1w"]["path"].name assert "T1w" in out
== f"sub-{sub}_space-MNI152NLin2009cAsym_"
"desc-preproc_T1w.nii.gz"
)
assert out["T1w"]["path"].exists() assert (
assert out["T1w"]["path"].is_file() out["T1w"]["path"].name == f"sub-{sub}_space-MNI152NLin2009cAsym_"
"desc-preproc_T1w.nii.gz"
)
# asserts T1w_mask assert out["T1w"]["path"].exists()
assert out["T1w_mask"]["path"].exists() assert out["T1w"]["path"].is_file()
# asserts type "probseg_CSF" # asserts T1w_mask
assert "probseg_CSF" in out assert out["T1w_mask"]["path"].exists()
assert ( # asserts type "probseg_CSF"
out["probseg_CSF"]["path"].name assert "probseg_CSF" in out
== f"sub-{sub}_space-MNI152NLin2009cAsym_label-"
"CSF_probseg.nii.gz"
)
assert out["probseg_CSF"]["path"].exists() assert (
assert out["probseg_CSF"]["path"].is_file() out["probseg_CSF"]["path"].name
== f"sub-{sub}_space-MNI152NLin2009cAsym_label-"
"CSF_probseg.nii.gz"
)
# asserts type "probseg_GM" assert out["probseg_CSF"]["path"].exists()
assert "probseg_GM" in out assert out["probseg_CSF"]["path"].is_file()
assert ( # asserts type "probseg_GM"
out["probseg_GM"]["path"].name assert "probseg_GM" in out
== f"sub-{sub}_space-MNI152NLin2009cAsym_label-"
"GM_probseg.nii.gz"
)
assert out["probseg_GM"]["path"].exists() assert (
assert out["probseg_GM"]["path"].is_file() out["probseg_GM"]["path"].name
== f"sub-{sub}_space-MNI152NLin2009cAsym_label-"
"GM_probseg.nii.gz"
)
# asserts type "probseg_WM" assert out["probseg_GM"]["path"].exists()
assert "probseg_WM" in out assert out["probseg_GM"]["path"].is_file()
assert ( # asserts type "probseg_WM"
out["probseg_WM"]["path"].name assert "probseg_WM" in out
== f"sub-{sub}_space-MNI152NLin2009cAsym_label-"
"WM_probseg.nii.gz"
)
assert out["probseg_WM"]["path"].exists() assert (
assert out["probseg_WM"]["path"].is_file() out["probseg_WM"]["path"].name
== f"sub-{sub}_space-MNI152NLin2009cAsym_label-"
"WM_probseg.nii.gz"
)
# asserts type "DWI" assert out["probseg_WM"]["path"].exists()
assert "DWI" in out assert out["probseg_WM"]["path"].is_file()
assert ( # asserts type "DWI"
out["DWI"]["path"].name == f"sub-{sub}_desc-preproc_dwi.nii.gz" assert "DWI" in out
)
assert out["DWI"]["path"].exists() assert out["DWI"]["path"].name == f"sub-{sub}_desc-preproc_dwi.nii.gz"
assert out["DWI"]["path"].is_file()
# asserts meta assert out["DWI"]["path"].exists()
assert "meta" in out["BOLD"] assert out["DWI"]["path"].is_file()
meta = out["BOLD"]["meta"]
assert "element" in meta # asserts meta
assert "subject" in meta["element"] assert "meta" in out["BOLD"]
assert sub == meta["element"]["subject"] meta = out["BOLD"]["meta"]
assert "element" in meta
assert "subject" in meta["element"]
assert sub == meta["element"]["subject"]
@pytest.mark.parametrize(
"types",
[
"BOLD",
"BOLD_confounds",
"T1w",
"probseg_CSF",
"probseg_GM",
"probseg_WM",
"DWI",
["BOLD", "BOLD_confounds"],
["T1w", "probseg_CSF"],
["probseg_GM", "probseg_WM"],
["DWI", "BOLD"],
],
)
def test_DataladAOMICPIOP2_partial_data_access(
types: Union[str, List[str]],
) -> None:
"""Test DataladAOMICPIOP2 DataGrabber partial data access.
Parameters
----------
types : str or list of str
The parametrized types.
"""
dg = DataladAOMICPIOP2(types=types)
# Set URI to Gin
dg.uri = URI
with dg:
# Get all elements
all_elements = dg.get_elements()
# Get test element
test_element = all_elements[0]
# Get test element data
out = dg[test_element]
# Assert data type
if isinstance(types, list):
for type_ in types:
assert type_ in out
else:
assert types in out
def test_DataladAOMICPIOP2_incorrect_data_type() -> None:
"""Test DataladAOMICPIOP2 DataGrabber incorrect data type."""
with pytest.raises(
ValueError, match="`patterns` must contain all `types`"
):
_ = DataladAOMICPIOP2(types="Vesta")
def test_DataladAOMICPIOP2_invalid_tasks(): def test_DataladAOMICPIOP2_invalid_tasks():
"""Test whether invalid task fails.""" """Test DataladAOMICIDPIOP2 DataGrabber invalid tasks."""
with pytest.raises( with pytest.raises(
ValueError, ValueError,
match=( match=(

View file

@ -61,7 +61,9 @@ def test_validate_patterns() -> None:
"T1w": "{subject}/anat/{subject}_T1w.nii.gz", "T1w": "{subject}/anat/{subject}_T1w.nii.gz",
} }
with pytest.raises(ValueError, match="same length"): with pytest.raises(
ValueError, match="Length of `types` more than that of `patterns`."
):
validate_patterns(types, wrongpatterns) # type: ignore validate_patterns(types, wrongpatterns) # type: ignore
wrongpatterns = { wrongpatterns = {

View file

@ -38,7 +38,9 @@ def test_PatternDataGrabber_errors(tmp_path: Path) -> None:
replacements="subject", # type: ignore replacements="subject", # type: ignore
) )
with pytest.raises(ValueError, match=r"must have the same length"): with pytest.raises(
ValueError, match=r"`patterns` must contain all `types`"
):
PatternDataGrabber( PatternDataGrabber(
datadir="/tmp", datadir="/tmp",
types=["func", "anat"], types=["func", "anat"],
@ -55,7 +57,7 @@ def test_PatternDataGrabber_errors(tmp_path: Path) -> None:
) )
with pytest.raises( with pytest.raises(
ValueError, match=r"`patterns` must have the same length" ValueError, match=r"Length of `types` more than that of `patterns`"
): ):
PatternDataGrabber( PatternDataGrabber(
datadir="/tmp", datadir="/tmp",

View file

@ -77,17 +77,17 @@ def validate_patterns(types: List[str], patterns: Dict[str, str]) -> None:
if not isinstance(patterns, dict): if not isinstance(patterns, dict):
raise_error(msg="`patterns` must be a dict.", klass=TypeError) raise_error(msg="`patterns` must be a dict.", klass=TypeError)
# Unequal length of objects # Unequal length of objects
if len(types) != len(patterns): if len(types) > len(patterns):
raise_error( raise_error(
msg="`types` and `patterns` must have the same length.", msg="Length of `types` more than that of `patterns`.",
klass=ValueError, klass=ValueError,
) )
# Missing type in patterns
if any(x not in patterns for x in types): if any(x not in patterns for x in types):
raise_error( raise_error(
msg="`patterns` must contain all `types`", klass=ValueError msg="`patterns` must contain all `types`", klass=ValueError
) )
# Wildcard check in patterns
if any("}*" in pattern for pattern in patterns.values()): if any("}*" in pattern for pattern in patterns.values()):
raise_error( raise_error(
msg="`patterns` must not contain `*` following a replacement", msg="`patterns` must not contain `*` following a replacement",