Skip to content

preflight

tit.pre.preflight

Preprocessing output preflight and rerun policy helpers.

PreprocessingOutput dataclass

PreprocessingOutput(subject_id: str, step: str, label: str, path: Path, cleanup_paths: tuple[Path, ...] = ())

Existing output for one preprocessing step.

PreprocessingInputProblem dataclass

PreprocessingInputProblem(subject_id: str, step: str, label: str, message: str, path: Path)

Missing or invalid input for one selected preprocessing step.

missing_inputs_for_step

missing_inputs_for_step(project_dir: str, subject_id: str, step: str) -> list[PreprocessingInputProblem]

Return missing required inputs for one subject and preprocessing step.

Source code in tit/pre/preflight.py
def missing_inputs_for_step(
    project_dir: str, subject_id: str, step: str
) -> list[PreprocessingInputProblem]:
    """Return missing required inputs for one subject and preprocessing step."""
    pm = get_path_manager(project_dir)
    if step in (STEP_CHARM, STEP_FASTSURFER, STEP_FREESURFER) and not _has_bids_t1(
        project_dir, subject_id
    ):
        return [
            PreprocessingInputProblem(
                subject_id=subject_id,
                step=step,
                label=STEP_LABELS[step],
                message=(
                    f"{STEP_LABELS[step]} requires a BIDS T1w image for "
                    f"sub-{subject_id}. Run DICOM conversion first or place "
                    f"sub-{subject_id}_T1w.nii[.gz] in the subject anat folder."
                ),
                path=Path(pm.bids_anat(subject_id)),
            )
        ]
    if step in FREESURFER_SUBREGION_STEPS.values():
        from .freesurfer import validate_reconstruction
        from .utils import PreprocessError

        try:
            validate_reconstruction(subject_id)
        except PreprocessError as exc:
            return [
                PreprocessingInputProblem(
                    subject_id,
                    step,
                    STEP_LABELS[step],
                    str(exc),
                    Path(pm.freesurfer_subject(subject_id)),
                )
            ]
    if step == STEP_QSIPREP:
        problems: list[PreprocessingInputProblem] = []
        # QSIPrep needs an anatomical reference (default --anat-modality T1w);
        # without one it fails deep in the workflow ("No T1w images found").
        if not _has_bids_t1(project_dir, subject_id):
            problems.append(
                PreprocessingInputProblem(
                    subject_id=subject_id,
                    step=step,
                    label=STEP_LABELS[step],
                    message=(
                        f"QSIPrep requires a BIDS T1w image for sub-{subject_id}. "
                        f"Run DICOM conversion first or place "
                        f"sub-{subject_id}_T1w.nii[.gz] in the subject anat folder."
                    ),
                    path=Path(pm.bids_anat(subject_id)),
                )
            )
        dwi_ok, dwi_error = validate_bids_dwi(
            project_dir, subject_id, logging.getLogger(__name__)
        )
        if not dwi_ok:
            problems.append(
                PreprocessingInputProblem(
                    subject_id=subject_id,
                    step=step,
                    label=STEP_LABELS[step],
                    message=(
                        f"QSIPrep requires BIDS DWI data for sub-{subject_id}: "
                        f"{dwi_error} Run DICOM conversion with DWI DICOMs or "
                        f"place sub-{subject_id}_dwi.nii[.gz] with .bval/.bvec "
                        "in the subject dwi folder."
                    ),
                    path=Path(pm.bids_dwi(subject_id)),
                )
            )
        else:
            sdc = plan_distortion_correction(
                project_dir,
                subject_id,
                logger=logging.getLogger(__name__),
                repair=False,
            )
            if sdc.blocking_error:
                problems.append(
                    PreprocessingInputProblem(
                        subject_id=subject_id,
                        step=step,
                        label=STEP_LABELS[step],
                        message=f"sub-{subject_id}: {sdc.blocking_error}",
                        path=Path(pm.bids_datatype(subject_id, "fmap")),
                    )
                )
        return problems
    if step == STEP_DTI:
        problems = []
        qsiprep_ok, qsiprep_error = validate_qsiprep_output(project_dir, subject_id)
        if not qsiprep_ok:
            problems.append(
                PreprocessingInputProblem(
                    subject_id=subject_id,
                    step=step,
                    label=STEP_LABELS[step],
                    message=(
                        f"DTI extraction fits the tensor from QSIPrep output: "
                        f"{qsiprep_error}. Run QSIPrep for sub-{subject_id} first."
                    ),
                    path=Path(pm.qsiprep_subject(subject_id)),
                )
            )
        t1 = Path(pm.m2m(subject_id)) / const.FILE_T1
        if not t1.is_file():
            problems.append(
                PreprocessingInputProblem(
                    subject_id=subject_id,
                    step=step,
                    label=STEP_LABELS[step],
                    message=(
                        f"DTI extraction writes into the charm head model; run "
                        f"SimNIBS charm for sub-{subject_id} first."
                    ),
                    path=t1,
                )
            )
        return problems
    return []

find_missing_preprocessing_inputs

find_missing_preprocessing_inputs(project_dir: str, subject_ids: Iterable[str], *, steps: Sequence[str] | None = None, convert_dicom: bool = False, create_m2m: bool = False, run_fastsurfer: bool = False, run_freesurfer: bool = False, freesurfer_recon_all: bool = True, freesurfer_subregions: Sequence[str] = (), run_qsiprep: bool = False, run_qsirecon: bool = False, extract_dti: bool = False, skip_existing_outputs: bool = False) -> list[PreprocessingInputProblem]

Return missing inputs that can be detected before running subprocesses.

Source code in tit/pre/preflight.py
def find_missing_preprocessing_inputs(
    project_dir: str,
    subject_ids: Iterable[str],
    *,
    steps: Sequence[str] | None = None,
    convert_dicom: bool = False,
    create_m2m: bool = False,
    run_fastsurfer: bool = False,
    run_freesurfer: bool = False,
    freesurfer_recon_all: bool = True,
    freesurfer_subregions: Sequence[str] = (),
    run_qsiprep: bool = False,
    run_qsirecon: bool = False,
    extract_dti: bool = False,
    skip_existing_outputs: bool = False,
) -> list[PreprocessingInputProblem]:
    """Return missing inputs that can be detected before running subprocesses."""
    subject_ids = list(subject_ids)
    selected_steps = (
        list(steps)
        if steps is not None
        else selected_preprocessing_steps(
            convert_dicom=convert_dicom,
            create_m2m=create_m2m,
            run_fastsurfer=run_fastsurfer,
            run_freesurfer=run_freesurfer,
            freesurfer_recon_all=freesurfer_recon_all,
            freesurfer_subregions=freesurfer_subregions,
            run_qsiprep=run_qsiprep,
            run_qsirecon=run_qsirecon,
            extract_dti=extract_dti,
        )
    )

    problems: list[PreprocessingInputProblem] = []
    for subject_id in subject_ids:
        subject_steps = selected_steps
        if _dicom_conversion_will_run(
            project_dir,
            subject_id,
            convert_dicom=convert_dicom,
            skip_existing_outputs=skip_existing_outputs,
        ):
            # Conversion can produce the T1w and DWI inputs these steps need,
            # so don't flag them as missing yet.
            subject_steps = [
                step
                for step in selected_steps
                if step
                not in (STEP_CHARM, STEP_FASTSURFER, STEP_FREESURFER, STEP_QSIPREP)
            ]
        if STEP_FREESURFER in selected_steps and (
            not skip_existing_outputs
            or not existing_outputs_for_step(project_dir, subject_id, STEP_FREESURFER)
        ):
            # This job's recon-all supplies the subregion prerequisites.
            subject_steps = [
                s for s in subject_steps if s not in FREESURFER_SUBREGION_STEPS.values()
            ]
        if skip_existing_outputs and existing_outputs_for_step(
            project_dir, subject_id, STEP_FREESURFER
        ):
            subject_steps = [s for s in subject_steps if s != STEP_FREESURFER]
        for step in subject_steps:
            found = missing_inputs_for_step(project_dir, subject_id, step)
            if step == STEP_DTI:
                # Stages selected in the same run produce these inputs first.
                produced = [
                    (
                        Path(get_path_manager(project_dir).qsiprep_subject(subject_id))
                        if STEP_QSIPREP in selected_steps
                        else None
                    ),
                    (
                        Path(get_path_manager(project_dir).m2m(subject_id))
                        / const.FILE_T1
                        if STEP_CHARM in selected_steps
                        else None
                    ),
                ]
                found = [p for p in found if p.path not in produced]
            problems.extend(found)
    problems.extend(missing_freesurfer_license(selected_steps, subject_ids))
    return problems

missing_freesurfer_license

missing_freesurfer_license(selected_steps: Sequence[str], subject_ids: Sequence[str] = ()) -> list[PreprocessingInputProblem]

One problem naming exactly which selected stages need a license, or none.

Only :data:LICENSED_STEPS are checked, so a FastSurfer segmentation-only run is never held up by a missing license.

Source code in tit/pre/preflight.py
def missing_freesurfer_license(
    selected_steps: Sequence[str], subject_ids: Sequence[str] = ()
) -> list[PreprocessingInputProblem]:
    """One problem naming exactly which selected stages need a license, or none.

    Only :data:`LICENSED_STEPS` are checked, so a FastSurfer segmentation-only
    run is never held up by a missing license.
    """
    licensed = [step for step in selected_steps if step in LICENSED_STEPS]
    if not licensed:
        return []
    from tit.surfer_settings import (
        BUNDLED_FS_LICENSE_PATH,
        freesurfer_license_status,
    )

    if freesurfer_license_status()["configured"]:
        return []
    names = ", ".join(STEP_LABELS[step] for step in licensed)
    return [
        PreprocessingInputProblem(
            next(iter(subject_ids), ""),
            licensed[0],
            "FreeSurfer license",
            (
                f"{names} cannot find the FreeSurfer license supplied by TI-Toolbox. "
                "Repair or update the TI-Toolbox installation; no personal license is required."
            ),
            BUNDLED_FS_LICENSE_PATH,
        )
    ]

existing_outputs_for_step

existing_outputs_for_step(project_dir: str, subject_id: str, step: str) -> list[PreprocessingOutput]

Return existing outputs for one subject and preprocessing step.

Source code in tit/pre/preflight.py
def existing_outputs_for_step(
    project_dir: str, subject_id: str, step: str
) -> list[PreprocessingOutput]:
    """Return existing outputs for one subject and preprocessing step."""
    pm = get_path_manager(project_dir)
    if step == STEP_DICOM:
        return _dicom_outputs(project_dir, subject_id)
    if step == STEP_CHARM:
        return _single_path_output(
            project_dir, subject_id, step, Path(pm.m2m(subject_id))
        )
    if step == STEP_FASTSURFER:
        return _single_path_output(
            project_dir, subject_id, step, Path(pm.fastsurfer_subject(subject_id))
        )
    if step == STEP_FREESURFER:
        return _single_path_output(
            project_dir, subject_id, step, Path(pm.freesurfer_subject(subject_id))
        )
    if step in FREESURFER_SUBREGION_STEPS.values():
        subject = Path(pm.freesurfer_subject(subject_id))
        # FreeSurfer 7.4.1 segment_subregions, default suffix: exact output names
        # preserve legacy runs and every recon-all input during subset reruns.
        names = (
            [
                "ThalamicNuclei.mgz",
                "ThalamicNuclei.FSvoxelSpace.mgz",
                "ThalamicNuclei.volumes.txt",
            ]
            if step == STEP_FREESURFER_THALAMUS
            else [
                f"{hemi}.hippoAmygLabels{suffix}.mgz"
                for hemi in ("lh", "rh")
                for suffix in (
                    "",
                    ".FSvoxelSpace",
                    ".HBT",
                    ".HBT.FSvoxelSpace",
                    ".FS60",
                    ".FS60.FSvoxelSpace",
                    ".CA",
                    ".CA.FSvoxelSpace",
                )
            ]
            + [
                f"{hemi}.{stem}.txt"
                for hemi in ("lh", "rh")
                for stem in ("hippoSfVolumes", "amygNucVolumes")
            ]
        )
        paths = tuple(
            subject / "mri" / name
            for name in names
            if (subject / "mri" / name).is_file()
        )
        return (
            [PreprocessingOutput(subject_id, step, STEP_LABELS[step], paths[0], paths)]
            if paths
            else []
        )
    if step == STEP_QSIPREP:
        return _single_path_output(
            project_dir, subject_id, step, Path(pm.qsiprep_subject(subject_id))
        )
    if step == STEP_QSIRECON:
        return _single_path_output(
            project_dir, subject_id, step, Path(pm.qsirecon_subject(subject_id))
        )
    if step == STEP_DTI:
        m2m = Path(pm.m2m(subject_id))
        tensor = m2m / const.FILE_DTI_TENSOR
        if not tensor.exists():
            return []
        # The QC JSON and the ACPC intermediate older releases wrote go with it.
        extras = tuple(
            p
            for p in (m2m / const.FILE_DTI_QC, m2m / "DTI_ACPC_tensor.nii.gz")
            if p.exists()
        )
        return [
            PreprocessingOutput(
                subject_id, step, STEP_LABELS[step], tensor, (tensor, *extras)
            )
        ]
    raise ValueError(f"Unknown preprocessing step: {step}")

selected_preprocessing_steps

selected_preprocessing_steps(*, convert_dicom: bool = False, create_m2m: bool = False, run_fastsurfer: bool = False, run_freesurfer: bool = False, freesurfer_recon_all: bool = True, freesurfer_subregions: Sequence[str] = (), run_qsiprep: bool = False, run_qsirecon: bool = False, extract_dti: bool = False) -> list[str]

Return output-producing steps enabled by the current run options.

Source code in tit/pre/preflight.py
def selected_preprocessing_steps(
    *,
    convert_dicom: bool = False,
    create_m2m: bool = False,
    run_fastsurfer: bool = False,
    run_freesurfer: bool = False,
    freesurfer_recon_all: bool = True,
    freesurfer_subregions: Sequence[str] = (),
    run_qsiprep: bool = False,
    run_qsirecon: bool = False,
    extract_dti: bool = False,
) -> list[str]:
    """Return output-producing steps enabled by the current run options."""
    steps: list[str] = []
    if convert_dicom:
        steps.append(STEP_DICOM)
    if create_m2m:
        steps.append(STEP_CHARM)
    if run_fastsurfer:
        steps.append(STEP_FASTSURFER)
    if run_freesurfer:
        if freesurfer_recon_all:
            steps.append(STEP_FREESURFER)
        steps.extend(
            FREESURFER_SUBREGION_STEPS[item]
            for item in dict.fromkeys(freesurfer_subregions)
        )
    if run_qsiprep:
        steps.append(STEP_QSIPREP)
    if run_qsirecon:
        steps.append(STEP_QSIRECON)
    if extract_dti:
        steps.append(STEP_DTI)
    return steps

find_existing_preprocessing_outputs

find_existing_preprocessing_outputs(project_dir: str, subject_ids: Iterable[str], *, steps: Sequence[str] | None = None, convert_dicom: bool = False, create_m2m: bool = False, run_fastsurfer: bool = False, run_freesurfer: bool = False, freesurfer_recon_all: bool = True, freesurfer_subregions: Sequence[str] = (), run_qsiprep: bool = False, run_qsirecon: bool = False, extract_dti: bool = False) -> list[PreprocessingOutput]

Return outputs that already exist for selected subjects and steps.

Source code in tit/pre/preflight.py
def find_existing_preprocessing_outputs(
    project_dir: str,
    subject_ids: Iterable[str],
    *,
    steps: Sequence[str] | None = None,
    convert_dicom: bool = False,
    create_m2m: bool = False,
    run_fastsurfer: bool = False,
    run_freesurfer: bool = False,
    freesurfer_recon_all: bool = True,
    freesurfer_subregions: Sequence[str] = (),
    run_qsiprep: bool = False,
    run_qsirecon: bool = False,
    extract_dti: bool = False,
) -> list[PreprocessingOutput]:
    """Return outputs that already exist for selected subjects and steps."""
    selected_steps = (
        list(steps)
        if steps is not None
        else selected_preprocessing_steps(
            convert_dicom=convert_dicom,
            create_m2m=create_m2m,
            run_fastsurfer=run_fastsurfer,
            run_freesurfer=run_freesurfer,
            freesurfer_recon_all=freesurfer_recon_all,
            freesurfer_subregions=freesurfer_subregions,
            run_qsiprep=run_qsiprep,
            run_qsirecon=run_qsirecon,
            extract_dti=extract_dti,
        )
    )
    outputs: list[PreprocessingOutput] = []
    for subject_id in subject_ids:
        for step in selected_steps:
            outputs.extend(existing_outputs_for_step(project_dir, subject_id, step))
    return outputs

remove_preprocessing_output

remove_preprocessing_output(output: PreprocessingOutput) -> None

Remove an existing preprocessing output before an explicit rerun.

Source code in tit/pre/preflight.py
def remove_preprocessing_output(output: PreprocessingOutput) -> None:
    """Remove an existing preprocessing output before an explicit rerun."""
    for path in output.paths_to_remove:
        if not path.exists():
            continue
        if path.is_dir():
            shutil.rmtree(path)
        else:
            path.unlink()