torch_em.data.datasets.medical.mct_ltdiag

The MCT-LTDiag dataset contains annotations for liver tumor segmentation in multi-phase contrast-enhanced CT.

The dataset consists of 517 patients (one tar archive per patient) with four-phase CT acquisitions (non-contrast 'nc', arterial 'art', portal-venous 'pvp' and delayed 'delay', stored as 'NIFTI/.nii.gz'). Manual tumor and whole-liver segmentation masks ('mask_pvp.nii.gz' and 'liver_mask_pvp.nii.gz') are only available for the portal-venous phase, which this loader pairs with 'NIFTI/pvp.nii.gz'. Each patient is additionally labeled with one of 5 tumor types (e.g. 'BCLM' for breast-cancer liver metastasis) in 'meta_info_patient.tab', which is not used by this loader but kept alongside the downloaded data for reference.

The data is located at https://doi.org/10.7910/DVN/S3RW15, released under a CC0-1.0 license, and is downloadable anonymously via the Harvard Dataverse access API (no personal API token required for these unrestricted files, despite the dataset-level metadata suggesting otherwise).

This dataset is from the publication https://doi.org/10.1148/radiol.232214. Please cite it if you use this dataset for your research.

  1"""The MCT-LTDiag dataset contains annotations for liver tumor segmentation in multi-phase
  2contrast-enhanced CT.
  3
  4The dataset consists of 517 patients (one tar archive per patient) with four-phase CT acquisitions
  5(non-contrast 'nc', arterial 'art', portal-venous 'pvp' and delayed 'delay', stored as
  6'NIFTI/<phase>.nii.gz'). Manual tumor and whole-liver segmentation masks ('mask_pvp.nii.gz' and
  7'liver_mask_pvp.nii.gz') are only available for the portal-venous phase, which this loader pairs
  8with 'NIFTI/pvp.nii.gz'. Each patient is additionally labeled with one of 5 tumor types (e.g. 'BCLM'
  9for breast-cancer liver metastasis) in 'meta_info_patient.tab', which is not used by this loader but
 10kept alongside the downloaded data for reference.
 11
 12The data is located at https://doi.org/10.7910/DVN/S3RW15, released under a CC0-1.0 license, and is
 13downloadable anonymously via the Harvard Dataverse access API (no personal API token required for
 14these unrestricted files, despite the dataset-level metadata suggesting otherwise).
 15
 16This dataset is from the publication https://doi.org/10.1148/radiol.232214.
 17Please cite it if you use this dataset for your research.
 18"""
 19
 20import os
 21from glob import glob
 22from natsort import natsorted
 23from typing import Union, Tuple, Literal, List, Optional
 24
 25from torch.utils.data import Dataset, DataLoader
 26
 27import torch_em
 28
 29from .. import util
 30
 31
 32DATAVERSE_API_URL = "https://dataverse.harvard.edu/api/datasets/:persistentId/?persistentId=doi:10.7910/DVN/S3RW15"
 33DATAVERSE_DOWNLOAD_URL = "https://dataverse.harvard.edu/api/access/datafile/{file_id}"
 34
 35TARGETS = {"liver": "liver_mask_pvp.nii.gz", "tumor": "mask_pvp.nii.gz"}
 36
 37
 38def _list_patient_files():
 39    import requests
 40
 41    # The default python-requests user agent is blocked by the Dataverse API, unlike a browser-like one.
 42    response = requests.get(DATAVERSE_API_URL, headers={"User-Agent": "Mozilla/5.0"})
 43    response.raise_for_status()
 44    files = response.json()["data"]["latestVersion"]["files"]
 45    return {
 46        f["label"][:-len(".tar")]: f["dataFile"]["id"] for f in files if f["label"].endswith(".tar")
 47    }
 48
 49
 50def get_mct_ltdiag_data(
 51    path: Union[os.PathLike, str], n_patients: Optional[int] = None, download: bool = False
 52) -> str:
 53    """Download the MCT-LTDiag dataset.
 54
 55    NOTE: The full collection is about 180 GB. Use `n_patients` to only download a subset for a quick start.
 56
 57    Args:
 58        path: Filepath to a folder where the data is downloaded for further processing.
 59        n_patients: The number of patients to download, sorted by patient id. By default all 517 are downloaded.
 60        download: Whether to download the data if it is not present.
 61
 62    Returns:
 63        Filepath where the data is downloaded.
 64    """
 65    import tarfile
 66
 67    patient_dir = os.path.join(path, "patients")
 68    patient_files = _list_patient_files()
 69    patient_ids = sorted(patient_files)
 70    if n_patients is not None:
 71        patient_ids = patient_ids[:n_patients]
 72    missing = [
 73        patient_id for patient_id in patient_ids
 74        if not os.path.exists(os.path.join(patient_dir, patient_id, "NIFTI", "pvp.nii.gz"))
 75    ]
 76    if not missing:
 77        return patient_dir
 78
 79    os.makedirs(patient_dir, exist_ok=True)
 80    for patient_id in missing:
 81        tar_path = os.path.join(path, f"{patient_id}.tar")
 82        url = DATAVERSE_DOWNLOAD_URL.format(file_id=patient_files[patient_id])
 83        util.download_source(path=tar_path, url=url, download=download)
 84
 85        out_dir = os.path.join(patient_dir, patient_id)
 86        os.makedirs(out_dir, exist_ok=True)
 87        with tarfile.open(tar_path, "r") as tar:
 88            tar.extractall(path=out_dir)
 89        os.remove(tar_path)
 90
 91    return patient_dir
 92
 93
 94def get_mct_ltdiag_paths(
 95    path: Union[os.PathLike, str],
 96    target: Literal["liver", "tumor"] = "tumor",
 97    n_patients: Optional[int] = None,
 98    download: bool = False,
 99) -> Tuple[List[str], List[str]]:
100    """Get paths to the MCT-LTDiag data.
101
102    Args:
103        path: Filepath to a folder where the data is downloaded for further processing.
104        target: The choice of segmentation target. Either 'liver' or 'tumor'.
105        n_patients: The number of patients to use, sorted by patient id. By default all 517 are used.
106        download: Whether to download the data if it is not present.
107
108    Returns:
109        List of filepaths for the image data.
110        List of filepaths for the label data.
111    """
112    if target not in TARGETS:
113        raise ValueError(f"'{target}' is not a valid target. Choose one of {list(TARGETS)}.")
114
115    patient_dir = get_mct_ltdiag_data(path, n_patients, download)
116
117    raw_paths = natsorted(glob(os.path.join(patient_dir, "*", "NIFTI", "pvp.nii.gz")))
118    label_paths = [os.path.join(os.path.dirname(os.path.dirname(p)), TARGETS[target]) for p in raw_paths]
119
120    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
121    assert all(os.path.exists(p) for p in label_paths)
122
123    return raw_paths, label_paths
124
125
126def get_mct_ltdiag_dataset(
127    path: Union[os.PathLike, str],
128    patch_shape: Tuple[int, int, int],
129    target: Literal["liver", "tumor"] = "tumor",
130    n_patients: Optional[int] = None,
131    resize_inputs: bool = False,
132    download: bool = False,
133    **kwargs
134) -> Dataset:
135    """Get the MCT-LTDiag dataset for liver and tumor segmentation in portal-venous phase CT.
136
137    Args:
138        path: Filepath to a folder where the data is downloaded for further processing.
139        patch_shape: The patch shape to use for training.
140        target: The choice of segmentation target. Either 'liver' or 'tumor'.
141        n_patients: The number of patients to use, sorted by patient id. By default all 517 are used.
142        resize_inputs: Whether to resize the inputs to the patch shape.
143        download: Whether to download the data if it is not present.
144        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
145
146    Returns:
147        The segmentation dataset.
148    """
149    raw_paths, label_paths = get_mct_ltdiag_paths(path, target, n_patients, download)
150
151    if resize_inputs:
152        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
153        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
154            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
155        )
156
157    return torch_em.default_segmentation_dataset(
158        raw_paths=raw_paths,
159        raw_key="data",
160        label_paths=label_paths,
161        label_key="data",
162        is_seg_dataset=True,
163        patch_shape=patch_shape,
164        ndim=3,
165        **kwargs
166    )
167
168
169def get_mct_ltdiag_loader(
170    path: Union[os.PathLike, str],
171    batch_size: int,
172    patch_shape: Tuple[int, int, int],
173    target: Literal["liver", "tumor"] = "tumor",
174    n_patients: Optional[int] = None,
175    resize_inputs: bool = False,
176    download: bool = False,
177    **kwargs
178) -> DataLoader:
179    """Get the MCT-LTDiag dataloader for liver and tumor segmentation in portal-venous phase CT.
180
181    Args:
182        path: Filepath to a folder where the data is downloaded for further processing.
183        batch_size: The batch size for training.
184        patch_shape: The patch shape to use for training.
185        target: The choice of segmentation target. Either 'liver' or 'tumor'.
186        n_patients: The number of patients to use, sorted by patient id. By default all 517 are used.
187        resize_inputs: Whether to resize the inputs to the patch shape.
188        download: Whether to download the data if it is not present.
189        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
190
191    Returns:
192        The DataLoader.
193    """
194    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
195    dataset = get_mct_ltdiag_dataset(path, patch_shape, target, n_patients, resize_inputs, download, **ds_kwargs)
196    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
DATAVERSE_API_URL = 'https://dataverse.harvard.edu/api/datasets/:persistentId/?persistentId=doi:10.7910/DVN/S3RW15'
DATAVERSE_DOWNLOAD_URL = 'https://dataverse.harvard.edu/api/access/datafile/{file_id}'
TARGETS = {'liver': 'liver_mask_pvp.nii.gz', 'tumor': 'mask_pvp.nii.gz'}
def get_mct_ltdiag_data( path: Union[os.PathLike, str], n_patients: Optional[int] = None, download: bool = False) -> str:
51def get_mct_ltdiag_data(
52    path: Union[os.PathLike, str], n_patients: Optional[int] = None, download: bool = False
53) -> str:
54    """Download the MCT-LTDiag dataset.
55
56    NOTE: The full collection is about 180 GB. Use `n_patients` to only download a subset for a quick start.
57
58    Args:
59        path: Filepath to a folder where the data is downloaded for further processing.
60        n_patients: The number of patients to download, sorted by patient id. By default all 517 are downloaded.
61        download: Whether to download the data if it is not present.
62
63    Returns:
64        Filepath where the data is downloaded.
65    """
66    import tarfile
67
68    patient_dir = os.path.join(path, "patients")
69    patient_files = _list_patient_files()
70    patient_ids = sorted(patient_files)
71    if n_patients is not None:
72        patient_ids = patient_ids[:n_patients]
73    missing = [
74        patient_id for patient_id in patient_ids
75        if not os.path.exists(os.path.join(patient_dir, patient_id, "NIFTI", "pvp.nii.gz"))
76    ]
77    if not missing:
78        return patient_dir
79
80    os.makedirs(patient_dir, exist_ok=True)
81    for patient_id in missing:
82        tar_path = os.path.join(path, f"{patient_id}.tar")
83        url = DATAVERSE_DOWNLOAD_URL.format(file_id=patient_files[patient_id])
84        util.download_source(path=tar_path, url=url, download=download)
85
86        out_dir = os.path.join(patient_dir, patient_id)
87        os.makedirs(out_dir, exist_ok=True)
88        with tarfile.open(tar_path, "r") as tar:
89            tar.extractall(path=out_dir)
90        os.remove(tar_path)
91
92    return patient_dir

Download the MCT-LTDiag dataset.

NOTE: The full collection is about 180 GB. Use n_patients to only download a subset for a quick start.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • n_patients: The number of patients to download, sorted by patient id. By default all 517 are downloaded.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_mct_ltdiag_paths( path: Union[os.PathLike, str], target: Literal['liver', 'tumor'] = 'tumor', n_patients: Optional[int] = None, download: bool = False) -> Tuple[List[str], List[str]]:
 95def get_mct_ltdiag_paths(
 96    path: Union[os.PathLike, str],
 97    target: Literal["liver", "tumor"] = "tumor",
 98    n_patients: Optional[int] = None,
 99    download: bool = False,
100) -> Tuple[List[str], List[str]]:
101    """Get paths to the MCT-LTDiag data.
102
103    Args:
104        path: Filepath to a folder where the data is downloaded for further processing.
105        target: The choice of segmentation target. Either 'liver' or 'tumor'.
106        n_patients: The number of patients to use, sorted by patient id. By default all 517 are used.
107        download: Whether to download the data if it is not present.
108
109    Returns:
110        List of filepaths for the image data.
111        List of filepaths for the label data.
112    """
113    if target not in TARGETS:
114        raise ValueError(f"'{target}' is not a valid target. Choose one of {list(TARGETS)}.")
115
116    patient_dir = get_mct_ltdiag_data(path, n_patients, download)
117
118    raw_paths = natsorted(glob(os.path.join(patient_dir, "*", "NIFTI", "pvp.nii.gz")))
119    label_paths = [os.path.join(os.path.dirname(os.path.dirname(p)), TARGETS[target]) for p in raw_paths]
120
121    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
122    assert all(os.path.exists(p) for p in label_paths)
123
124    return raw_paths, label_paths

Get paths to the MCT-LTDiag data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • target: The choice of segmentation target. Either 'liver' or 'tumor'.
  • n_patients: The number of patients to use, sorted by patient id. By default all 517 are used.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_mct_ltdiag_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int, int], target: Literal['liver', 'tumor'] = 'tumor', n_patients: Optional[int] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
127def get_mct_ltdiag_dataset(
128    path: Union[os.PathLike, str],
129    patch_shape: Tuple[int, int, int],
130    target: Literal["liver", "tumor"] = "tumor",
131    n_patients: Optional[int] = None,
132    resize_inputs: bool = False,
133    download: bool = False,
134    **kwargs
135) -> Dataset:
136    """Get the MCT-LTDiag dataset for liver and tumor segmentation in portal-venous phase CT.
137
138    Args:
139        path: Filepath to a folder where the data is downloaded for further processing.
140        patch_shape: The patch shape to use for training.
141        target: The choice of segmentation target. Either 'liver' or 'tumor'.
142        n_patients: The number of patients to use, sorted by patient id. By default all 517 are used.
143        resize_inputs: Whether to resize the inputs to the patch shape.
144        download: Whether to download the data if it is not present.
145        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
146
147    Returns:
148        The segmentation dataset.
149    """
150    raw_paths, label_paths = get_mct_ltdiag_paths(path, target, n_patients, download)
151
152    if resize_inputs:
153        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
154        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
155            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
156        )
157
158    return torch_em.default_segmentation_dataset(
159        raw_paths=raw_paths,
160        raw_key="data",
161        label_paths=label_paths,
162        label_key="data",
163        is_seg_dataset=True,
164        patch_shape=patch_shape,
165        ndim=3,
166        **kwargs
167    )

Get the MCT-LTDiag dataset for liver and tumor segmentation in portal-venous phase CT.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • target: The choice of segmentation target. Either 'liver' or 'tumor'.
  • n_patients: The number of patients to use, sorted by patient id. By default all 517 are used.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_mct_ltdiag_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int, int], target: Literal['liver', 'tumor'] = 'tumor', n_patients: Optional[int] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
170def get_mct_ltdiag_loader(
171    path: Union[os.PathLike, str],
172    batch_size: int,
173    patch_shape: Tuple[int, int, int],
174    target: Literal["liver", "tumor"] = "tumor",
175    n_patients: Optional[int] = None,
176    resize_inputs: bool = False,
177    download: bool = False,
178    **kwargs
179) -> DataLoader:
180    """Get the MCT-LTDiag dataloader for liver and tumor segmentation in portal-venous phase CT.
181
182    Args:
183        path: Filepath to a folder where the data is downloaded for further processing.
184        batch_size: The batch size for training.
185        patch_shape: The patch shape to use for training.
186        target: The choice of segmentation target. Either 'liver' or 'tumor'.
187        n_patients: The number of patients to use, sorted by patient id. By default all 517 are used.
188        resize_inputs: Whether to resize the inputs to the patch shape.
189        download: Whether to download the data if it is not present.
190        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
191
192    Returns:
193        The DataLoader.
194    """
195    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
196    dataset = get_mct_ltdiag_dataset(path, patch_shape, target, n_patients, resize_inputs, download, **ds_kwargs)
197    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the MCT-LTDiag dataloader for liver and tumor segmentation in portal-venous phase CT.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • target: The choice of segmentation target. Either 'liver' or 'tumor'.
  • n_patients: The number of patients to use, sorted by patient id. By default all 517 are used.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.