torch_em.data.datasets.medical.ddti
The DDTI dataset contains annotations for thyroid nodule segmentation in ultrasound images.
The dataset consists of 637 thyroid ultrasound images with pixel-level nodule masks. The masks are binary: 0 for background and 1 for thyroid nodule.
The original DDTI (Digital Database Thyroid Image) data was released without pixel-level masks, only with per-case XML annotations. We use the Kaggle mirror at https://www.kaggle.com/datasets/srivarshachivukula/ddti-dataset, which additionally provides the pixel-level masks derived from the original annotations, for the download.
This dataset is from the publication https://doi.org/10.1117/12.2073532. Please cite it if you use this dataset for your research.
1"""The DDTI dataset contains annotations for thyroid nodule segmentation in ultrasound images. 2 3The dataset consists of 637 thyroid ultrasound images with pixel-level nodule masks. The masks are 4binary: 0 for background and 1 for thyroid nodule. 5 6The original DDTI (Digital Database Thyroid Image) data was released without pixel-level masks, only 7with per-case XML annotations. We use the Kaggle mirror at 8https://www.kaggle.com/datasets/srivarshachivukula/ddti-dataset, which additionally provides the 9pixel-level masks derived from the original annotations, for the download. 10 11This dataset is from the publication https://doi.org/10.1117/12.2073532. 12Please cite it if you use this dataset for your research. 13""" 14 15import os 16from glob import glob 17from natsort import natsorted 18from typing import Union, Tuple, List 19 20from torch.utils.data import Dataset, DataLoader 21 22import torch_em 23 24from .. import util 25 26 27KAGGLE_DATASET_NAME = "srivarshachivukula/ddti-dataset" 28 29 30def get_ddti_data(path: Union[os.PathLike, str], download: bool = False) -> str: 31 """Download the DDTI dataset. 32 33 Args: 34 path: Filepath to a folder where the data is downloaded for further processing. 35 download: Whether to download the data if it is not present. 36 37 Returns: 38 Filepath where the data is downloaded. 39 """ 40 data_dir = os.path.join(path, "DDTI dataset", "DDTI", "1_or_data") 41 if os.path.exists(data_dir): 42 return data_dir 43 44 os.makedirs(path, exist_ok=True) 45 46 util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download) 47 48 zip_path = os.path.join(path, "ddti-dataset.zip") 49 util.unzip(zip_path=zip_path, dst=path) 50 51 return data_dir 52 53 54def get_ddti_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 55 """Get paths to the DDTI data. 56 57 Args: 58 path: Filepath to a folder where the data is downloaded for further processing. 59 download: Whether to download the data if it is not present. 60 61 Returns: 62 List of filepaths for the image data. 63 List of filepaths for the label data. 64 """ 65 data_dir = get_ddti_data(path=path, download=download) 66 67 image_paths = natsorted(glob(os.path.join(data_dir, "image", "*.PNG"))) 68 gt_paths = natsorted(glob(os.path.join(data_dir, "mask", "*.PNG"))) 69 70 if len(image_paths) == 0 or len(image_paths) != len(gt_paths): 71 raise RuntimeError("Something went wrong with fetching the image and label paths.") 72 73 return image_paths, gt_paths 74 75 76def get_ddti_dataset( 77 path: Union[os.PathLike, str], 78 patch_shape: Tuple[int, int], 79 resize_inputs: bool = False, 80 download: bool = False, 81 **kwargs 82) -> Dataset: 83 """Get the DDTI dataset for thyroid nodule segmentation. 84 85 Args: 86 path: Filepath to a folder where the data is downloaded for further processing. 87 patch_shape: The patch shape to use for training. 88 resize_inputs: Whether to resize the inputs. 89 download: Whether to download the data if it is not present. 90 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 91 92 Returns: 93 The segmentation dataset. 94 """ 95 image_paths, gt_paths = get_ddti_paths(path, download) 96 97 if resize_inputs: 98 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 99 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 100 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 101 ) 102 103 return torch_em.default_segmentation_dataset( 104 raw_paths=image_paths, 105 raw_key=None, 106 label_paths=gt_paths, 107 label_key=None, 108 patch_shape=patch_shape, 109 is_seg_dataset=False, 110 **kwargs 111 ) 112 113 114def get_ddti_loader( 115 path: Union[os.PathLike, str], 116 batch_size: int, 117 patch_shape: Tuple[int, int], 118 resize_inputs: bool = False, 119 download: bool = False, 120 **kwargs 121) -> DataLoader: 122 """Get the DDTI dataloader for thyroid nodule segmentation. 123 124 Args: 125 path: Filepath to a folder where the data is downloaded for further processing. 126 batch_size: The batch size for training. 127 patch_shape: The patch shape to use for training. 128 resize_inputs: Whether to resize the inputs. 129 download: Whether to download the data if it is not present. 130 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 131 132 Returns: 133 The DataLoader. 134 """ 135 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 136 dataset = get_ddti_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 137 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
31def get_ddti_data(path: Union[os.PathLike, str], download: bool = False) -> str: 32 """Download the DDTI dataset. 33 34 Args: 35 path: Filepath to a folder where the data is downloaded for further processing. 36 download: Whether to download the data if it is not present. 37 38 Returns: 39 Filepath where the data is downloaded. 40 """ 41 data_dir = os.path.join(path, "DDTI dataset", "DDTI", "1_or_data") 42 if os.path.exists(data_dir): 43 return data_dir 44 45 os.makedirs(path, exist_ok=True) 46 47 util.download_source_kaggle(path=path, dataset_name=KAGGLE_DATASET_NAME, download=download) 48 49 zip_path = os.path.join(path, "ddti-dataset.zip") 50 util.unzip(zip_path=zip_path, dst=path) 51 52 return data_dir
Download the DDTI dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
55def get_ddti_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]: 56 """Get paths to the DDTI data. 57 58 Args: 59 path: Filepath to a folder where the data is downloaded for further processing. 60 download: Whether to download the data if it is not present. 61 62 Returns: 63 List of filepaths for the image data. 64 List of filepaths for the label data. 65 """ 66 data_dir = get_ddti_data(path=path, download=download) 67 68 image_paths = natsorted(glob(os.path.join(data_dir, "image", "*.PNG"))) 69 gt_paths = natsorted(glob(os.path.join(data_dir, "mask", "*.PNG"))) 70 71 if len(image_paths) == 0 or len(image_paths) != len(gt_paths): 72 raise RuntimeError("Something went wrong with fetching the image and label paths.") 73 74 return image_paths, gt_paths
Get paths to the DDTI data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
77def get_ddti_dataset( 78 path: Union[os.PathLike, str], 79 patch_shape: Tuple[int, int], 80 resize_inputs: bool = False, 81 download: bool = False, 82 **kwargs 83) -> Dataset: 84 """Get the DDTI dataset for thyroid nodule segmentation. 85 86 Args: 87 path: Filepath to a folder where the data is downloaded for further processing. 88 patch_shape: The patch shape to use for training. 89 resize_inputs: Whether to resize the inputs. 90 download: Whether to download the data if it is not present. 91 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 92 93 Returns: 94 The segmentation dataset. 95 """ 96 image_paths, gt_paths = get_ddti_paths(path, download) 97 98 if resize_inputs: 99 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 100 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 101 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 102 ) 103 104 return torch_em.default_segmentation_dataset( 105 raw_paths=image_paths, 106 raw_key=None, 107 label_paths=gt_paths, 108 label_key=None, 109 patch_shape=patch_shape, 110 is_seg_dataset=False, 111 **kwargs 112 )
Get the DDTI dataset for thyroid nodule segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
115def get_ddti_loader( 116 path: Union[os.PathLike, str], 117 batch_size: int, 118 patch_shape: Tuple[int, int], 119 resize_inputs: bool = False, 120 download: bool = False, 121 **kwargs 122) -> DataLoader: 123 """Get the DDTI dataloader for thyroid nodule segmentation. 124 125 Args: 126 path: Filepath to a folder where the data is downloaded for further processing. 127 batch_size: The batch size for training. 128 patch_shape: The patch shape to use for training. 129 resize_inputs: Whether to resize the inputs. 130 download: Whether to download the data if it is not present. 131 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 132 133 Returns: 134 The DataLoader. 135 """ 136 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 137 dataset = get_ddti_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs) 138 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the DDTI dataloader for thyroid nodule segmentation.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- resize_inputs: Whether to resize the inputs.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.