[ENH]: Introduce junifer.api.generate_yaml #498

Open
synchon wants to merge 21 commits from feat/generate-yaml-api into main
25 changed files with 675 additions and 2 deletions

View file

@ -0,0 +1 @@
Introduce :func:`.generate_yaml` to generate feature YAML from metadata by `Synchon Mandal`_

View file

@ -13,6 +13,7 @@
.. _`INM-7`: https://www.fz-juelich.de/inm/inm-7/EN/Home/home_node.html .. _`INM-7`: https://www.fz-juelich.de/inm/inm-7/EN/Home/home_node.html
.. _`julearn`: https://juaml.github.io/julearn .. _`julearn`: https://juaml.github.io/julearn
.. _`junifer-data`: https://github.com/juaml/junifer-data-client .. _`junifer-data`: https://github.com/juaml/junifer-data-client
.. _`julio`: https://github.com/juaml/julio
.. _`pandas`: https://pandas.pydata.org .. _`pandas`: https://pandas.pydata.org
.. _`pandas.DataFrame` : https://pandas.pydata.org/pandas-docs/stable/reference/api/pandas.DataFrame.html .. _`pandas.DataFrame` : https://pandas.pydata.org/pandas-docs/stable/reference/api/pandas.DataFrame.html

View file

@ -0,0 +1,14 @@
.. include:: ../links.inc
.. _generate_yaml:
Generating YAML from metadata
=============================
``junifer`` stores the pipeline metadata for a run along with the extracted feature data.
So, the metadata for all the "elements" processed with a pipeline is unique. The metadata
contains all the necessary information to recreate the configuration used for the processing.
If one wants to generate the processing YAML, :func:`.generate_yaml` can be used for that.
The only requirement is providing the metadata which can be extracted by following the initial steps of
:ref:`analysing results <analysing_extracted_features>`.

View file

@ -20,6 +20,7 @@ to interact with HPC and HTC systems.
queueing queueing
configuring configuring
dumping dumping
generate_yaml
.. _using_components: .. _using_components:

View file

@ -6,6 +6,7 @@ __all__ = [
"reset", "reset",
"list_elements", "list_elements",
"parse_yaml", "parse_yaml",
"generate_yaml",
] ]
from . import decorators from . import decorators
@ -13,6 +14,7 @@ from .functions import (
collect, collect,
list_elements, list_elements,
parse_yaml, parse_yaml,
generate_yaml,
reset, reset,
run, run,
queue, queue,

View file

@ -6,14 +6,18 @@
# License: AGPL # License: AGPL
import atexit import atexit
import datetime as dt
import importlib import importlib
import importlib.util import importlib.util
import io
import os import os
import shutil import shutil
import sys import sys
from pathlib import Path from pathlib import Path
from typing import TYPE_CHECKING, Any
import structlog import structlog
from pydantic import ValidationError
from ..api.queue_context import GnuParallelLocalAdapter, HTCondorAdapter from ..api.queue_context import GnuParallelLocalAdapter, HTCondorAdapter
from ..datagrabber import BaseDataGrabber from ..datagrabber import BaseDataGrabber
@ -35,8 +39,13 @@ from ..typing import (
from ..utils import raise_error, warn_with_log, yaml from ..utils import raise_error, warn_with_log, yaml
if TYPE_CHECKING:
from ruamel.yaml.comments import CommentedMap
__all__ = [ __all__ = [
"collect", "collect",
"generate_yaml",
"list_elements", "list_elements",
"parse_yaml", "parse_yaml",
"queue", "queue",
@ -600,3 +609,175 @@ def parse_yaml(filepath: str | Path) -> dict: # noqa: C901
) )
return contents return contents
def generate_yaml(meta: dict) -> "CommentedMap": # noqa: C901
"""Generate the feature YAML from metadata.
Parameters
----------
meta : dict
Feature metadata as dictionary.
Returns
-------
ruamel.yaml.comments.CommentedMap
Feature YAML.
"""
y: dict[str, Any] = {}
y["workdir"] = ""
# Add "with" section if present
if "with" in meta:
y["with"] = meta["with"].copy()
# Init var for post comment and issues
post = "\nIssues:\n"
issue_ext = (
" - `{0}` is not a built-in component and thus could not be properly "
"regenerated. Some of these entries in the YAML section might be "
"redundant and not needed. Please check the "
"documentation/implementation of this specific component and remove "
"the unnecessary entries.\n"
)
issue_inv = (
" - `{0}` failed to initialise and thus could not be properly "
"regenerated. Some of these entries in the YAML section might be "
"redundant and not needed. Please check the "
"documentation/implementation of this specific component and remove "
"the unnecessary entries.\n"
)
var = ""
# Set datagrabber
meta_dg = meta["datagrabber"].copy()
a = meta_dg.pop("class")
if a not in PipelineComponentRegistry()._components["datagrabber"]:
y["datagrabber"] = {"kind": a, **meta_dg}
post += f"- datagrabber:\n{issue_ext.format(a)}"
else:
dg = PipelineComponentRegistry().get_class(step="datagrabber", name=a)
try:
dg_model = dg.model_validate(meta_dg)
except ValidationError:
y["datagrabber"] = {"kind": a, **meta_dg}
post += f"- datagrabber:\n{issue_inv.format(a)}"
else:
y["datagrabber"] = {
"kind": a,
**dg_model.model_dump(
mode="json",
include=set(dg_model.dump_fields()),
exclude_defaults=True,
exclude_none=True,
),
}
if meta_dg.get("datalad_dirty"):
var = (
"- The dataset was 'dirty', there is no guarantee that the "
"results will be reproducible.\n"
)
# Set preprocessor(s)
if "preprocess" in meta:
y["preprocess"] = []
meta_p = meta["preprocess"].copy()
if not isinstance(meta_p, list):
meta_p = [meta_p]
for mp in meta_p:
b = mp.pop("class")
if (
b
not in PipelineComponentRegistry()._components["preprocessing"]
):
y["preprocess"].append({"kind": b, **mp})
if "- preprocess:\n" in post:
post += f"{issue_ext.format(b)}"
else:
post += f"- preprocess:\n{issue_ext.format(b)}"
else:
p = PipelineComponentRegistry().get_class(
step="preprocessing", name=b
)
try:
p_model = p.model_validate(mp)
except ValidationError:
y["preprocess"].append({"kind": b, **mp})
if "- preprocess:\n" in post:
post += f"{issue_inv.format(b)}"
else:
post += f"- preprocess:\n{issue_inv.format(b)}"
else:
y["preprocess"].append(
{
"kind": b,
**p_model.model_dump(
mode="json",
exclude={"required_data_types"},
exclude_defaults=True,
exclude_none=True,
),
}
)
# Set marker
meta_m = meta["marker"].copy()
c = meta_m.pop("class")
y["markers"] = []
if c not in PipelineComponentRegistry()._components["marker"]:
y["markers"].append({"kind": c, **meta_m})
post += f"- markers:\n{issue_ext.format(c)}"
else:
m = PipelineComponentRegistry().get_class(step="marker", name=c)
try:
m_model = m.model_validate(meta_m)
except ValidationError:
y["markers"].append({"kind": c, **meta_m})
post += f"- markers:\n{issue_inv.format(c)}"
else:
y["markers"].append(
{
"kind": c,
**m_model.model_dump(
mode="json",
exclude_defaults=True,
exclude_none=True,
),
}
)
# Set storage
y["storage"] = {
"kind": "HDF5FeatureStorage",
"uri": "",
}
# Set queue
if "queue" in meta:
y["queue"] = meta["queue"].copy()
else:
y["queue"] = {
"jobname": meta["name"],
"kind": "",
}
# Dump and load yaml to format
f = io.StringIO()
yaml.dump(y, stream=f)
f.seek(0)
d = yaml.load(f)
# Write comments
pre = (
"Auto-generated by junifer on "
f"{dt.datetime.now(tz=dt.timezone.utc).strftime('%Y-%m-%d %H:%M:%S')} "
"UTC\n\n"
)
if "dependencies" in meta:
for k, v in meta["dependencies"].items():
pre += f"{k}=={v}\n"
const = (
"\nNotes:\n"
"- Check the components for possible changes in the API.\n"
"- `datadir` is ignored and not reproduced. "
"If `datadir` used was not a temporary directory, you will have to "
"manually edit this YAML.\n"
)
post = post if post != "\nIssues:\n" else ""
d.yaml_set_start_comment(pre + const + var + post)
# Add newline between sections
for s in d.keys():
d.yaml_set_comment_before_after_key(s, before="\n")
return d

View file

@ -5,6 +5,7 @@
# Synchon Mandal <s.mandal@fz-juelich.de> # Synchon Mandal <s.mandal@fz-juelich.de>
# License: AGPL # License: AGPL
import io
import logging import logging
import sys import sys
from contextlib import AbstractContextManager, nullcontext from contextlib import AbstractContextManager, nullcontext
@ -16,7 +17,15 @@ from nibabel.filebasedimages import ImageFileError
from ruamel.yaml import YAML from ruamel.yaml import YAML
import junifer.testing.registry # noqa: F401 import junifer.testing.registry # noqa: F401
from junifer.api import collect, list_elements, parse_yaml, queue, reset, run from junifer.api import (
collect,
generate_yaml,
list_elements,
parse_yaml,
queue,
reset,
run,
)
from junifer.datagrabber.base import BaseDataGrabber from junifer.datagrabber.base import BaseDataGrabber
from junifer.pipeline import PipelineComponentRegistry from junifer.pipeline import PipelineComponentRegistry
from junifer.typing import Elements from junifer.typing import Elements
@ -1025,3 +1034,351 @@ def test_parse_yaml_queue_venv_relative(tmp_path: Path) -> None:
fname = tmp_path / "test_parse_yaml_queue_venv_relative.yaml" fname = tmp_path / "test_parse_yaml_queue_venv_relative.yaml"
fname.write_text("queue:\n env:\n kind: venv\n name: .venv\n") fname.write_text("queue:\n env:\n kind: venv\n name: .venv\n")
_ = parse_yaml(fname) _ = parse_yaml(fname)
@pytest.mark.parametrize(
"m, exp",
[
(
{
"datagrabber": {
"class": "PartlyCloudyTestingDataGrabber",
"types": ["BOLD"],
"datadir": (
"/var/folders/dv/2lbr8f8j0q12zrx3mz3ll5m40000gp/T/tmpjeqj9nou"
),
"reduce_confounds": False,
"age_group": "both",
},
"dependencies": {"scikit-learn": "1.4.2", "nilearn": "0.10.4"},
"datareader": {"class": "DefaultDataReader"},
"type": "BOLD",
"marker": {
"class": "FunctionalConnectivityParcels",
"on": ["BOLD"],
"name": "fc_mean-shen_2015_268_functional_connectivity",
"agg_method": "mean",
"agg_method_params": None,
"conn_method": "correlation",
"conn_method_params": {"empirical": True},
"masks": None,
"parcellation": ["Shen_2015_268"],
},
"_element_keys": ["subject"],
"name": "BOLD_fc_mean-shen_2015_268_functional_connectivity",
},
[
"Auto-generated by junifer on",
"Check the components for possible changes in the API",
],
),
(
{
"datagrabber": {
"class": "PartlyCloudyTestingDataGrabber",
"types": ["BOLD"],
"datadir": (
"/var/folders/dv/2lbr8f8j0q12zrx3mz3ll5m40000gp/T/tmpjeqj9nou"
),
"reduce_confound": True,
"age": "both",
},
"dependencies": {"scikit-learn": "1.4.2", "nilearn": "0.10.4"},
"datareader": {"class": "DefaultDataReader"},
"type": "BOLD",
"marker": {
"class": "FunctionalConnectivityParcels",
"on": ["BOLD"],
"name": "fc_mean-shen_2015_268_functional_connectivity",
"ag_method": "mean",
"ag_method_params": None,
"con_method": "correlation",
"con_method_params": {"empirical": True},
"masks": None,
"parcellation": ["Shen_2015_268"],
},
"_element_keys": ["subject"],
"name": "BOLD_fc_mean-shen_2015_268_functional_connectivity",
},
[
"Auto-generated by junifer on",
"Check the components for possible changes in the API",
"`PartlyCloudyTestingDataGrabber` failed to initialise and "
"thus could not be properly",
],
),
(
{
"datagrabber": {
"class": "DMCC13Benchmark",
"types": ["BOLD"],
"patterns": {
"BOLD": {
"pattern": (
"derivatives/fmriprep-1.3.2/{subject}/{session}/func/{subject}_{session}_task-{task}_acq-mb4{phase_encoding}_run-{run}_space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"space": "MNI152NLin2009cAsym",
"mask": {
"pattern": (
"derivatives/fmriprep-1.3.2/{subject}/{session}/func/{subject}_{session}_task-{task}_acq-mb4{phase_encoding}_run-{run}_space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"confounds": {
"pattern": (
"derivatives/fmriprep-1.3.2/{subject}/{session}/func/{subject}_{session}_task-{task}_acq-mb4{phase_encoding}_run-{run}_desc-confounds_regressors.tsv"
),
"format": "fmriprep",
},
},
"T1w": {
"pattern": (
"derivatives/fmriprep-1.3.2/{subject}/anat/{subject}_space-MNI152NLin2009cAsym_desc-preproc_T1w.nii.gz"
),
"space": "MNI152NLin2009cAsym",
"mask": {
"pattern": (
"derivatives/fmriprep-1.3.2/{subject}/anat/{subject}_space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
},
},
"replacements": [
"subject",
"session",
"task",
"phase_encoding",
"run",
],
"confounds_format": "fmriprep",
"partial_pattern_ok": False,
"uri": "https://github.com/OpenNeuroDatasets/ds003452.git",
"rootdir": ".",
"datalad_dirty": True,
"datalad_commit_id": (
"8484f21551af53fcf2bc53878f18ce93dc2d29da"
),
"datalad_id": "ade00fb6-636f-46fb-b2e6-60958b1b112d",
"sessions": ["ses-wave1bas"],
"tasks": ["Rest"],
"phase_encodings": ["AP"],
"runs": ["1"],
"native_t1w": False,
},
"dependencies": {
"scikit-learn": "1.4.2",
"nilearn": "0.10.4",
"numpy": "1.26.4",
},
"datareader": {"class": "DefaultDataReader"},
"preprocess": {
"class": "fMRIPrepConfoundRemover",
"on": ["BOLD"],
"required_data_types": ["BOLD"],
"strategy": {
"motion": "full",
"wm_csf": "full",
"global_signal": "full",
},
"spike": None,
"scrub": None,
"fd_threshold": None,
"std_dvars_threshold": None,
"detrend": True,
"standardize": True,
"low_pass": 0.08,
"high_pass": 0.01,
"t_r": None,
"masks": ["compute_epi_mask"],
},
"type": "BOLD",
"marker": {
"class": "FunctionalConnectivitySpheres",
"on": ["BOLD"],
"name": "fc_spheres_functional_connectivity",
"agg_method": "mean",
"agg_method_params": None,
"conn_method": "correlation",
"conn_method_params": {"empirical": True},
"masks": None,
"coords": "DMNBuckner",
"radius": 5.0,
"allow_overlap": False,
},
"_element_keys": [
"subject",
"session",
"task",
"phase_encoding",
"run",
],
"name": "BOLD_fc_spheres_functional_connectivity",
},
[
"Auto-generated by junifer on",
"Check the components for possible changes in the API",
"The dataset was 'dirty'",
],
),
(
{
"datagrabber": {
"class": "DMCC13Benchmark",
"types": ["BOLD"],
"patterns": {
"BOLD": {
"pattern": (
"derivatives/fmriprep-1.3.2/{subject}/{session}/func/{subject}_{session}_task-{task}_acq-mb4{phase_encoding}_run-{run}_space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"space": "MNI152NLin2009cAsym",
"mask": {
"pattern": (
"derivatives/fmriprep-1.3.2/{subject}/{session}/func/{subject}_{session}_task-{task}_acq-mb4{phase_encoding}_run-{run}_space-MNI152NLin2009cAsym_desc-brain_mask.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
"confounds": {
"pattern": (
"derivatives/fmriprep-1.3.2/{subject}/{session}/func/{subject}_{session}_task-{task}_acq-mb4{phase_encoding}_run-{run}_desc-confounds_regressors.tsv"
),
"format": "fmriprep",
},
},
},
"replacements": [
"subject",
"session",
"task",
"phase_encoding",
"run",
],
"confounds_format": "fmriprep",
"partial_pattern_ok": False,
"uri": "https://github.com/OpenNeuroDatasets/ds003452.git",
"rootdir": ".",
"datalad_dirty": False,
"datalad_commit_id": (
"8484f21551af53fcf2bc53878f18ce93dc2d29da"
),
"datalad_id": "ade00fb6-636f-46fb-b2e6-60958b1b112d",
"sessions": ["ses-wave1bas"],
"tasks": ["Rest"],
"phase_encodings": ["AP"],
"runs": ["1"],
"native_t1w": False,
},
"dependencies": {
"scikit-learn": "1.4.2",
"nilearn": "0.10.4",
"numpy": "1.26.4",
},
"datareader": {"class": "DefaultDataReader"},
"type": "BOLD",
"marker": {
"class": "FunctionalConnectivitySpheres",
"on": ["BOLD"],
"name": "fc_spheres_functional_connectivity",
"agg_method": "mean",
"agg_method_params": None,
"conn_method": "correlation",
"conn_method_params": {"empirical": True},
"masks": None,
"coords": "DMNBuckner",
"radius": 5.0,
"allow_overlap": False,
},
"_element_keys": [
"subject",
"session",
"task",
"phase_encoding",
"run",
],
"name": "BOLD_fc_spheres_functional_connectivity",
},
[
"Auto-generated by junifer on",
"Check the components for possible changes in the API",
],
),
(
{
"datagrabber": {
"class": "ExternalDataGrabber",
"types": ["BOLD"],
"patterns": {
"BOLD": {
"pattern": (
"derivatives/fmriprep-1.3.2/{subject}/func/{subject}_task-{task}_space-MNI152NLin2009cAsym_desc-preproc_bold.nii.gz"
),
"space": "MNI152NLin2009cAsym",
},
},
"replacements": [
"subject",
"task",
],
"confounds_format": "fmriprep",
"partial_pattern_ok": True,
"uri": "https://github.com/datasets/ds11.git",
"rootdir": ".",
"datalad_dirty": False,
"datalad_commit_id": (
"8484f21551af53fcf2bc53878f18ce93dc2d29da"
),
"datalad_id": "ade00fb6-636f-46fb-b2e6-60958b1b112d",
"sessions": ["ses-wave1bas"],
"tasks": ["Rest"],
},
"dependencies": {"scikit-learn": "1.4.2", "nilearn": "0.10.4"},
"datareader": {"class": "DefaultDataReader"},
"preprocess": [
{
"class": "ExternalPreprocessor1",
"on": ["BOLD"],
"required_data_types": ["BOLD"],
},
{
"class": "ExternalPreprocessor2",
"on": ["BOLD"],
"required_data_types": ["BOLD"],
},
],
"type": "BOLD",
"marker": {
"class": "ExternalMarker",
"on": ["BOLD"],
"name": "external",
},
"_element_keys": ["subject", "task"],
"name": "BOLD_external",
},
[
"Auto-generated by junifer on",
"Check the components for possible changes in the API",
"`ExternalDataGrabber` is not a built-in component",
"`ExternalPreprocessor1` is not a built-in component",
"`ExternalMarker` is not a built-in component",
],
),
],
)
def test_generate_yaml(m: dict, exp: list[str]) -> None:
"""Test YAML generation from feature metadata.
Parameters
----------
m : dict
The parametrized feature metadata.
exp : list
The parametrized expected comments.
"""
c = generate_yaml(m)
buf = io.StringIO()
yaml.dump(c, stream=buf)
buf.seek(0)
y = buf.read()
for e in exp:
assert e in y

View file

@ -40,3 +40,8 @@ class JuselessDataladAOMICID1000VBM(PatternDataladDataGrabber):
}, },
} }
replacements: list[str] = ["subject"] # noqa: RUF012 replacements: list[str] = ["subject"] # noqa: RUF012
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return []

View file

@ -43,3 +43,8 @@ class JuselessDataladCamCANVBM(PatternDataladDataGrabber):
}, },
} }
replacements: list[str] = ["subject"] # noqa: RUF012 replacements: list[str] = ["subject"] # noqa: RUF012
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return []

View file

@ -66,3 +66,8 @@ class JuselessDataladIXIVBM(PatternDataladDataGrabber):
}, },
} }
replacements: list[str] = ["site", "subject"] # noqa: RUF012 replacements: list[str] = ["site", "subject"] # noqa: RUF012
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return []

View file

@ -143,6 +143,11 @@ class JuselessUCLA(PatternDataGrabber):
replacements: list[str] = ["subject", "task"] # noqa: RUF012 replacements: list[str] = ["subject", "task"] # noqa: RUF012
fraimondo commented 2026-07-21 14:06:18 +00:00 (Migrated from github.com)

This should only be types and tasks. The rest is hard-coded in the parameters.

This should only be `types` and `tasks`. The rest is hard-coded in the parameters.
confounds_format: ConfoundsFormat = ConfoundsFormat.FMRIPrep confounds_format: ConfoundsFormat = ConfoundsFormat.FMRIPrep
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return ["types", "tasks"]
def get_elements(self) -> list: def get_elements(self) -> list:
"""Implement fetching list of elements in the dataset. """Implement fetching list of elements in the dataset.

View file

@ -43,3 +43,8 @@ class JuselessDataladUKBVBM(PatternDataladDataGrabber):
}, },
} }
replacements: list[str] = ["subject", "session"] # noqa: RUF012 replacements: list[str] = ["subject", "session"] # noqa: RUF012
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return []

View file

@ -243,3 +243,8 @@ class DataladAOMICID1000(PatternDataladDataGrabber):
else: else:
self.patterns["BOLD"]["prewarp_space"] = "native" self.patterns["BOLD"]["prewarp_space"] = "native"
super().validate_datagrabber_params() super().validate_datagrabber_params()
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return ["types", "space"]

View file

@ -266,6 +266,11 @@ class DataladAOMICPIOP1(PatternDataladDataGrabber):
self.patterns["BOLD"]["prewarp_space"] = "native" self.patterns["BOLD"]["prewarp_space"] = "native"
super().validate_datagrabber_params() super().validate_datagrabber_params()
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return ["types", "tasks", "space"]
def get_item(self, subject: str, task: str) -> dict: def get_item(self, subject: str, task: str) -> dict:
"""Get the specified item from the dataset. """Get the specified item from the dataset.

View file

@ -262,6 +262,11 @@ class DataladAOMICPIOP2(PatternDataladDataGrabber):
self.patterns["BOLD"]["prewarp_space"] = "native" self.patterns["BOLD"]["prewarp_space"] = "native"
super().validate_datagrabber_params() super().validate_datagrabber_params()
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return ["types", "tasks", "space"]
def get_elements(self) -> list: def get_elements(self) -> list:
"""Implement fetching list of elements in the dataset. """Implement fetching list of elements in the dataset.

View file

@ -85,6 +85,11 @@ class BaseDataGrabber(BaseModel, ABC, UpdateMetaMixin):
""" """
pass pass
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return ["types", "datadir"]
def __iter__(self) -> Iterator[Elements]: def __iter__(self) -> Iterator[Elements]:
"""Enable iterable support. """Enable iterable support.

View file

@ -176,6 +176,11 @@ class DataladDataGrabber(BaseDataGrabber):
) and self.datadir.stem.endswith("juniferauto"): ) and self.datadir.stem.endswith("juniferauto"):
_remove_datadir(self.datadir) _remove_datadir(self.datadir)
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return ["types", "uri", "rootdir"]
fraimondo commented 2026-07-21 14:14:32 +00:00 (Migrated from github.com)

this is where it becomes a bit tricky.

If I used a datadir (not temp) and the dataset is dirty, the YAML should account for that? or not?

Maybe we should include some comments in the generated YAMLs indicating stuff like this:

eg. if we have a "dirty" dataset, then add a comment that while the yaml will reproduce the results, the original dataset was "dirty" and so there is no guarantee that the same results will be obtained as there is no strict data provenance.

this is where it becomes a bit tricky. If I used a datadir (not temp) and the dataset is dirty, the YAML should account for that? or not? Maybe we should include some comments in the generated YAMLs indicating stuff like this: eg. if we have a "dirty" dataset, then add a comment that while the yaml will reproduce the results, the original dataset was "dirty" and so there is no guarantee that the same results will be obtained as there is no strict data provenance.
synchon commented 2026-07-21 15:25:15 +00:00 (Migrated from github.com)

We can add a general comment. Making it conditional would be quite tricky.

We can add a general comment. Making it conditional would be quite tricky.
fraimondo commented 2026-07-22 09:05:51 +00:00 (Migrated from github.com)

I would like that the generated YAML is commented.

I would like that the generated YAML is commented.
@property @property
def fulldir(self) -> Path: def fulldir(self) -> Path:
"""Get complete data directory path. """Get complete data directory path.

View file

@ -276,6 +276,18 @@ class DMCC13Benchmark(PatternDataladDataGrabber):
self.types.append(DataType.Warp) self.types.append(DataType.Warp)
super().validate_datagrabber_params() super().validate_datagrabber_params()
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return [
"types",
"sessions",
"tasks",
"phase_encodings",
"runs",
"native_t1w",
]
def get_item( def get_item(
self, self,
subject: str, subject: str,

View file

@ -61,6 +61,11 @@ class DataladHCP1200(DataladDataGrabber, HCP1200):
] ]
rootdir: Path = Path("HCP1200") rootdir: Path = Path("HCP1200")
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return ["types", "tasks", "phase_encodings", "ica_fix"]
fraimondo commented 2026-07-21 14:10:13 +00:00 (Migrated from github.com)

This can also be super(HCP1200) - datadir. Thus any change in the super will also be accounted for here.

This can also be `super(HCP1200) - datadir`. Thus any change in the super will also be accounted for here.
synchon commented 2026-07-21 15:21:03 +00:00 (Migrated from github.com)

In that case, the MRO for this class will get dump_fields from DataladDataGrabber which will be incorrect.

In that case, the MRO for this class will get `dump_fields` from `DataladDataGrabber` which will be incorrect.
# Needed here as HCP1200's subjects are sub-datasets, so will not be # Needed here as HCP1200's subjects are sub-datasets, so will not be
# found when elements are checked. # found when elements are checked.
@property @property

View file

@ -159,6 +159,11 @@ class HCP1200(PatternDataGrabber):
].replace("{suffix}", suffix) ].replace("{suffix}", suffix)
super().validate_datagrabber_params() super().validate_datagrabber_params()
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return ["types", "datadir", "tasks", "phase_encodings", "ica_fix"]
def get_item(self, subject: str, task: str, phase_encoding: str) -> dict: def get_item(self, subject: str, task: str, phase_encoding: str) -> dict:
"""Get the specified item from the dataset. """Get the specified item from the dataset.

View file

@ -94,6 +94,11 @@ class MultipleDataGrabber(BaseDataGrabber):
klass=RuntimeError, klass=RuntimeError,
) )
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return [*super().dump_fields(), "datagrabbers"]
def __getitem__(self, element: Element) -> dict: def __getitem__(self, element: Element) -> dict:
"""Implement indexing. """Implement indexing.

View file

@ -101,6 +101,17 @@ class PatternDataGrabber(BaseDataGrabber, PatternValidationMixin):
logger.debug(f"\treplacements = {self.replacements}") logger.debug(f"\treplacements = {self.replacements}")
logger.debug(f"\tconfounds_format = {self.confounds_format}") logger.debug(f"\tconfounds_format = {self.confounds_format}")
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return [
*super().dump_fields(),
"patterns",
"replacements",
"confounds_format",
"partial_pattern_ok",
]
@property @property
def skip_file_check(self) -> bool: def skip_file_check(self) -> bool:
"""Skip file check existence.""" """Skip file check existence."""

View file

@ -61,3 +61,16 @@ class PatternDataladDataGrabber(DataladDataGrabber, PatternDataGrabber):
logger.debug("Initializing PatternDataladDataGrabber") logger.debug("Initializing PatternDataladDataGrabber")
for key, val in self.__pydantic_extra__.items(): for key, val in self.__pydantic_extra__.items():
logger.debug(f"\t{key} = {val}") logger.debug(f"\t{key} = {val}")
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return [
fraimondo commented 2026-07-21 14:15:21 +00:00 (Migrated from github.com)

I thinks this should be both "super" fields.

I thinks this should be both "super" fields.
synchon commented 2026-07-21 15:22:31 +00:00 (Migrated from github.com)

It is "both" of them.

It is "both" of them.
fraimondo commented 2026-07-22 09:06:40 +00:00 (Migrated from github.com)

I meant instead of manually placing the fields, to all super. But I understand that might be bothersome.

I meant instead of manually placing the fields, to all super. But I understand that might be bothersome.
synchon commented 2026-07-22 09:07:56 +00:00 (Migrated from github.com)

The MRO would stop at the first dump_fields which would give a partial list.

The MRO would stop at the first `dump_fields` which would give a partial list.
"types",
"patterns",
"replacements",
"confounds_format",
"partial_pattern_ok",
"uri",
"rootdir",
]

View file

@ -19,7 +19,7 @@ from ..utils import raise_error
def _ets( def _ets(
bold_ts: np.ndarray, bold_ts: np.ndarray,
roi_names: None | list[str] = None, roi_names: list[str] | None = None,
) -> tuple[np.ndarray, list[str] | None]: ) -> tuple[np.ndarray, list[str] | None]:
"""Compute the edge-wise time series based on BOLD time series. """Compute the edge-wise time series based on BOLD time series.

View file

@ -34,6 +34,11 @@ class OasisVBMTestingDataGrabber(BaseDataGrabber):
datadir: Path = Path(tempfile.mkdtemp()) datadir: Path = Path(tempfile.mkdtemp())
_dataset: Any = None _dataset: Any = None
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return ["types"]
def get_element_keys(self) -> list[str]: def get_element_keys(self) -> list[str]:
"""Get element keys. """Get element keys.
@ -101,6 +106,11 @@ class SPMAuditoryTestingDataGrabber(BaseDataGrabber):
types: list[DataType] = [DataType.BOLD, DataType.T1w] # noqa: RUF012 types: list[DataType] = [DataType.BOLD, DataType.T1w] # noqa: RUF012
datadir: Path = Path(tempfile.mkdtemp()) datadir: Path = Path(tempfile.mkdtemp())
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return ["types"]
def get_element_keys(self) -> list[str]: def get_element_keys(self) -> list[str]:
"""Get element keys. """Get element keys.
@ -189,6 +199,11 @@ class PartlyCloudyTestingDataGrabber(BaseDataGrabber):
reduce_confounds: bool = True reduce_confounds: bool = True
age_group: PartlyCloudyAgeGroup = PartlyCloudyAgeGroup.Both age_group: PartlyCloudyAgeGroup = PartlyCloudyAgeGroup.Both
@classmethod
def dump_fields(cls) -> list[str]:
"""Fields to include when dumping model."""
return ["types", "reduce_confounds", "age_group"]
def __enter__(self) -> "PartlyCloudyTestingDataGrabber": def __enter__(self) -> "PartlyCloudyTestingDataGrabber":
"""Implement context entry. """Implement context entry.