torch_em.data.datasets.medical.ddti

The DDTI dataset contains annotations for thyroid nodule segmentation in ultrasound images.

The dataset consists of 637 thyroid ultrasound images with pixel-level nodule masks. The masks are binary: 0 for background and 1 for thyroid nodule.

The original DDTI (Digital Database Thyroid Image) data was released without pixel-level masks, only with per-case XML annotations. We use the Kaggle mirror at https://www.kaggle.com/datasets/srivarshachivukula/ddti-dataset, which additionally provides the pixel-level masks derived from the original annotations, for the download.

This dataset is from the publication https://doi.org/10.1117/12.2073532. Please cite it if you use this dataset for your research.

  1"""The DDTI dataset contains annotations for thyroid nodule segmentation in ultrasound images.
  2
  3The dataset consists of 637 thyroid ultrasound images with pixel-level nodule masks. The masks are
  4binary: 0 for background and 1 for thyroid nodule.
  5
  6The original DDTI (Digital Database Thyroid Image) data was released without pixel-level masks, only
  7with per-case XML annotations. We use the Kaggle mirror at
  8https://www.kaggle.com/datasets/srivarshachivukula/ddti-dataset, which additionally provides the
  9pixel-level masks derived from the original annotations, for the download.
 10
 11This dataset is from the publication https://doi.org/10.1117/12.2073532.
 12Please cite it if you use this dataset for your research.
 13"""
 14
 15import os
 16from glob import glob
 17from natsort import natsorted
 18from typing import Union, Tuple, List
 19
 20from torch.utils.data import Dataset, DataLoader
 21
 22import torch_em
 23
 24from .. import util
 25
 26
 27KAGGLE_DATASET_NAME = "srivarshachivukula/ddti-dataset"
 28
 29
 30def get_ddti_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 31    """Download the DDTI dataset.
 32
 33    Args:
 34        path: Filepath to a folder where the data is downloaded for further processing.
 35        download: Whether to download the data if it is not present.
 36
 37    Returns:
 38        Filepath where the data is downloaded.
 39    """
 40    data_dir = os.path.join(path, "DDTI dataset", "DDTI", "1_or_data")
 41    if os.path.exists(data_dir):
 42        return data_dir
 43
 44    os.makedirs(path, exist_ok=True)
 45
 46    util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download)
 47
 48    zip_path = os.path.join(path, "ddti-dataset.zip")
 49    util.unzip(zip_path=zip_path, dst=path)
 50
 51    return data_dir
 52
 53
 54def get_ddti_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 55    """Get paths to the DDTI data.
 56
 57    Args:
 58        path: Filepath to a folder where the data is downloaded for further processing.
 59        download: Whether to download the data if it is not present.
 60
 61    Returns:
 62        List of filepaths for the image data.
 63        List of filepaths for the label data.
 64    """
 65    data_dir = get_ddti_data(path=path, download=download)
 66
 67    image_paths = natsorted(glob(os.path.join(data_dir, "image", "*.PNG")))
 68    gt_paths = natsorted(glob(os.path.join(data_dir, "mask", "*.PNG")))
 69
 70    if len(image_paths) == 0 or len(image_paths) != len(gt_paths):
 71        raise RuntimeError("Something went wrong with fetching the image and label paths.")
 72
 73    return image_paths, gt_paths
 74
 75
 76def get_ddti_dataset(
 77    path: Union[os.PathLike, str],
 78    patch_shape: Tuple[int, int],
 79    resize_inputs: bool = False,
 80    download: bool = False,
 81    **kwargs
 82) -> Dataset:
 83    """Get the DDTI dataset for thyroid nodule segmentation.
 84
 85    Args:
 86        path: Filepath to a folder where the data is downloaded for further processing.
 87        patch_shape: The patch shape to use for training.
 88        resize_inputs: Whether to resize the inputs.
 89        download: Whether to download the data if it is not present.
 90        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
 91
 92    Returns:
 93        The segmentation dataset.
 94    """
 95    image_paths, gt_paths = get_ddti_paths(path, download)
 96
 97    if resize_inputs:
 98        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
 99        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
100            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
101        )
102
103    return torch_em.default_segmentation_dataset(
104        raw_paths=image_paths,
105        raw_key=None,
106        label_paths=gt_paths,
107        label_key=None,
108        patch_shape=patch_shape,
109        is_seg_dataset=False,
110        **kwargs
111    )
112
113
114def get_ddti_loader(
115    path: Union[os.PathLike, str],
116    batch_size: int,
117    patch_shape: Tuple[int, int],
118    resize_inputs: bool = False,
119    download: bool = False,
120    **kwargs
121) -> DataLoader:
122    """Get the DDTI dataloader for thyroid nodule segmentation.
123
124    Args:
125        path: Filepath to a folder where the data is downloaded for further processing.
126        batch_size: The batch size for training.
127        patch_shape: The patch shape to use for training.
128        resize_inputs: Whether to resize the inputs.
129        download: Whether to download the data if it is not present.
130        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
131
132    Returns:
133        The DataLoader.
134    """
135    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
136    dataset = get_ddti_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
137    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
KAGGLE_DATASET_NAME = 'srivarshachivukula/ddti-dataset'
def get_ddti_data(path: Union[os.PathLike, str], download: bool = False) -> str:
31def get_ddti_data(path: Union[os.PathLike, str], download: bool = False) -> str:
32    """Download the DDTI dataset.
33
34    Args:
35        path: Filepath to a folder where the data is downloaded for further processing.
36        download: Whether to download the data if it is not present.
37
38    Returns:
39        Filepath where the data is downloaded.
40    """
41    data_dir = os.path.join(path, "DDTI dataset", "DDTI", "1_or_data")
42    if os.path.exists(data_dir):
43        return data_dir
44
45    os.makedirs(path, exist_ok=True)
46
47    util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download)
48
49    zip_path = os.path.join(path, "ddti-dataset.zip")
50    util.unzip(zip_path=zip_path, dst=path)
51
52    return data_dir

Download the DDTI dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_ddti_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
55def get_ddti_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
56    """Get paths to the DDTI data.
57
58    Args:
59        path: Filepath to a folder where the data is downloaded for further processing.
60        download: Whether to download the data if it is not present.
61
62    Returns:
63        List of filepaths for the image data.
64        List of filepaths for the label data.
65    """
66    data_dir = get_ddti_data(path=path, download=download)
67
68    image_paths = natsorted(glob(os.path.join(data_dir, "image", "*.PNG")))
69    gt_paths = natsorted(glob(os.path.join(data_dir, "mask", "*.PNG")))
70
71    if len(image_paths) == 0 or len(image_paths) != len(gt_paths):
72        raise RuntimeError("Something went wrong with fetching the image and label paths.")
73
74    return image_paths, gt_paths

Get paths to the DDTI data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_ddti_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 77def get_ddti_dataset(
 78    path: Union[os.PathLike, str],
 79    patch_shape: Tuple[int, int],
 80    resize_inputs: bool = False,
 81    download: bool = False,
 82    **kwargs
 83) -> Dataset:
 84    """Get the DDTI dataset for thyroid nodule segmentation.
 85
 86    Args:
 87        path: Filepath to a folder where the data is downloaded for further processing.
 88        patch_shape: The patch shape to use for training.
 89        resize_inputs: Whether to resize the inputs.
 90        download: Whether to download the data if it is not present.
 91        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
 92
 93    Returns:
 94        The segmentation dataset.
 95    """
 96    image_paths, gt_paths = get_ddti_paths(path, download)
 97
 98    if resize_inputs:
 99        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
100        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
101            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
102        )
103
104    return torch_em.default_segmentation_dataset(
105        raw_paths=image_paths,
106        raw_key=None,
107        label_paths=gt_paths,
108        label_key=None,
109        patch_shape=patch_shape,
110        is_seg_dataset=False,
111        **kwargs
112    )

Get the DDTI dataset for thyroid nodule segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_ddti_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
115def get_ddti_loader(
116    path: Union[os.PathLike, str],
117    batch_size: int,
118    patch_shape: Tuple[int, int],
119    resize_inputs: bool = False,
120    download: bool = False,
121    **kwargs
122) -> DataLoader:
123    """Get the DDTI dataloader for thyroid nodule segmentation.
124
125    Args:
126        path: Filepath to a folder where the data is downloaded for further processing.
127        batch_size: The batch size for training.
128        patch_shape: The patch shape to use for training.
129        resize_inputs: Whether to resize the inputs.
130        download: Whether to download the data if it is not present.
131        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
132
133    Returns:
134        The DataLoader.
135    """
136    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
137    dataset = get_ddti_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
138    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the DDTI dataloader for thyroid nodule segmentation.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.