torch_em.data.datasets.medical.pandental
The PanDental dataset contains annotations for mandible segmentation in panoramic dental radiographs.
This dataset is located at https://data.mendeley.com/datasets/hxt48yk462/2 This dataset is from the publication https://doi.org/10.1117/1.JMI.2.4.044003. Please cite it if you use this dataset for your research.
1"""The PanDental dataset contains annotations for mandible segmentation in 2panoramic dental radiographs. 3 4This dataset is located at https://data.mendeley.com/datasets/hxt48yk462/2 5This dataset is from the publication https://doi.org/10.1117/1.JMI.2.4.044003. 6Please cite it if you use this dataset for your research. 7""" 8 9import os 10from glob import glob 11from tqdm import tqdm 12from pathlib import Path 13from natsort import natsorted 14from typing import Union, Literal, Tuple, List 15 16import numpy as np 17import imageio.v3 as imageio 18 19from torch.utils.data import Dataset, DataLoader 20 21import torch_em 22 23from .. import util 24 25 26URL = "https://data.mendeley.com/public-files/datasets/hxt48yk462/files/c2df1e1e-9939-4197-9bac-1eb697a64094/file_downloaded" # noqa 27CHECKSUM = "4ab6f670428df8052ae04aef82011050cd5ad4805f4d28d32fea523880752cf3" 28 29 30def get_pandental_data(path: Union[os.PathLike, str], download: bool = False) -> str: 31 """Download the PanDental dataset. 32 33 Args: 34 path: Filepath to a folder where the data is downloaded for further processing. 35 download: Whether to download the data if it is not present. 36 37 Returns: 38 Filepath where the data is downloaded. 39 """ 40 data_dir = os.path.join(path, "Images") 41 if os.path.exists(data_dir): 42 return path 43 44 os.makedirs(path, exist_ok=True) 45 46 zip_path = os.path.join(path, "DentalPanoramicXrays.zip") 47 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 48 util.unzip(zip_path=zip_path, dst=path) 49 50 return path 51 52 53def get_pandental_paths( 54 path: Union[os.PathLike, str], annotator: Literal["1", "2"] = "1", download: bool = False 55) -> Tuple[List[str], List[str]]: 56 """Get paths to the PanDental data. 57 58 Args: 59 path: Filepath to a folder where the data is downloaded for further processing. 60 annotator: The choice of expert annotator. Either '1' or '2'. 61 download: Whether to download the data if it is not present. 62 63 Returns: 64 List of filepaths for the image data. 65 List of filepaths for the label data. 66 """ 67 data_dir = get_pandental_data(path=path, download=download) 68 69 image_paths = natsorted(glob(os.path.join(data_dir, "Images", "*.png"))) 70 raw_gt_paths = natsorted(glob(os.path.join(data_dir, f"Segmentation{annotator}", "*.png"))) 71 72 neu_gt_dir = os.path.join(data_dir, "preprocessed", f"gt{annotator}") 73 os.makedirs(neu_gt_dir, exist_ok=True) 74 75 gt_paths = [] 76 for raw_gt_path in tqdm(raw_gt_paths, desc="Preprocessing labels"): 77 gt_path = os.path.join(neu_gt_dir, f"{Path(raw_gt_path).stem}.tif") 78 gt_paths.append(gt_path) 79 if os.path.exists(gt_path): 80 continue 81 82 # the ground-truth is the original image masked by the binary mandible region, 83 # i.e. non-zero pixels correspond to the mandible. 84 raw_gt = imageio.imread(raw_gt_path) 85 binary_gt = (raw_gt > 0).astype(np.uint8) 86 imageio.imwrite(gt_path, binary_gt) 87 88 return image_paths, gt_paths 89 90 91def get_pandental_dataset( 92 path: Union[os.PathLike, str], 93 patch_shape: Tuple[int, int], 94 annotator: Literal["1", "2"] = "1", 95 resize_inputs: bool = False, 96 download: bool = False, 97 **kwargs 98) -> Dataset: 99 """Get the PanDental dataset for mandible segmentation in panoramic dental radiographs. 100 101 Args: 102 path: Filepath to a folder where the data is downloaded for further processing. 103 patch_shape: The patch shape to use for training. 104 annotator: The choice of expert annotator. Either '1' or '2'. 105 resize_inputs: Whether to resize the inputs to the patch shape. 106 download: Whether to download the data if it is not present. 107 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 108 109 Returns: 110 The segmentation dataset. 111 """ 112 image_paths, gt_paths = get_pandental_paths(path, annotator, download) 113 114 if resize_inputs: 115 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 116 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 117 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 118 ) 119 120 return torch_em.default_segmentation_dataset( 121 raw_paths=image_paths, 122 raw_key=None, 123 label_paths=gt_paths, 124 label_key=None, 125 is_seg_dataset=False, 126 patch_shape=patch_shape, 127 **kwargs 128 ) 129 130 131def get_pandental_loader( 132 path: Union[os.PathLike, str], 133 batch_size: int, 134 patch_shape: Tuple[int, int], 135 annotator: Literal["1", "2"] = "1", 136 resize_inputs: bool = False, 137 download: bool = False, 138 **kwargs 139) -> DataLoader: 140 """Get the PanDental dataloader for mandible segmentation in panoramic dental radiographs. 141 142 Args: 143 path: Filepath to a folder where the data is downloaded for further processing. 144 batch_size: The batch size for training. 145 patch_shape: The patch shape to use for training. 146 annotator: The choice of expert annotator. Either '1' or '2'. 147 resize_inputs: Whether to resize the inputs to the patch shape. 148 download: Whether to download the data if it is not present. 149 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 150 151 Returns: 152 The DataLoader. 153 """ 154 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 155 dataset = get_pandental_dataset(path, patch_shape, annotator, resize_inputs, download, **ds_kwargs) 156 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL =
'https://data.mendeley.com/public-files/datasets/hxt48yk462/files/c2df1e1e-9939-4197-9bac-1eb697a64094/file_downloaded'
CHECKSUM =
'4ab6f670428df8052ae04aef82011050cd5ad4805f4d28d32fea523880752cf3'
def
get_pandental_data(path: Union[os.PathLike, str], download: bool = False) -> str:
31def get_pandental_data(path: Union[os.PathLike, str], download: bool = False) -> str: 32 """Download the PanDental dataset. 33 34 Args: 35 path: Filepath to a folder where the data is downloaded for further processing. 36 download: Whether to download the data if it is not present. 37 38 Returns: 39 Filepath where the data is downloaded. 40 """ 41 data_dir = os.path.join(path, "Images") 42 if os.path.exists(data_dir): 43 return path 44 45 os.makedirs(path, exist_ok=True) 46 47 zip_path = os.path.join(path, "DentalPanoramicXrays.zip") 48 util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM) 49 util.unzip(zip_path=zip_path, dst=path) 50 51 return path
Download the PanDental dataset.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- download: Whether to download the data if it is not present.
Returns:
Filepath where the data is downloaded.
def
get_pandental_paths( path: Union[os.PathLike, str], annotator: Literal['1', '2'] = '1', download: bool = False) -> Tuple[List[str], List[str]]:
54def get_pandental_paths( 55 path: Union[os.PathLike, str], annotator: Literal["1", "2"] = "1", download: bool = False 56) -> Tuple[List[str], List[str]]: 57 """Get paths to the PanDental data. 58 59 Args: 60 path: Filepath to a folder where the data is downloaded for further processing. 61 annotator: The choice of expert annotator. Either '1' or '2'. 62 download: Whether to download the data if it is not present. 63 64 Returns: 65 List of filepaths for the image data. 66 List of filepaths for the label data. 67 """ 68 data_dir = get_pandental_data(path=path, download=download) 69 70 image_paths = natsorted(glob(os.path.join(data_dir, "Images", "*.png"))) 71 raw_gt_paths = natsorted(glob(os.path.join(data_dir, f"Segmentation{annotator}", "*.png"))) 72 73 neu_gt_dir = os.path.join(data_dir, "preprocessed", f"gt{annotator}") 74 os.makedirs(neu_gt_dir, exist_ok=True) 75 76 gt_paths = [] 77 for raw_gt_path in tqdm(raw_gt_paths, desc="Preprocessing labels"): 78 gt_path = os.path.join(neu_gt_dir, f"{Path(raw_gt_path).stem}.tif") 79 gt_paths.append(gt_path) 80 if os.path.exists(gt_path): 81 continue 82 83 # the ground-truth is the original image masked by the binary mandible region, 84 # i.e. non-zero pixels correspond to the mandible. 85 raw_gt = imageio.imread(raw_gt_path) 86 binary_gt = (raw_gt > 0).astype(np.uint8) 87 imageio.imwrite(gt_path, binary_gt) 88 89 return image_paths, gt_paths
Get paths to the PanDental data.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- annotator: The choice of expert annotator. Either '1' or '2'.
- download: Whether to download the data if it is not present.
Returns:
List of filepaths for the image data. List of filepaths for the label data.
def
get_pandental_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], annotator: Literal['1', '2'] = '1', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
92def get_pandental_dataset( 93 path: Union[os.PathLike, str], 94 patch_shape: Tuple[int, int], 95 annotator: Literal["1", "2"] = "1", 96 resize_inputs: bool = False, 97 download: bool = False, 98 **kwargs 99) -> Dataset: 100 """Get the PanDental dataset for mandible segmentation in panoramic dental radiographs. 101 102 Args: 103 path: Filepath to a folder where the data is downloaded for further processing. 104 patch_shape: The patch shape to use for training. 105 annotator: The choice of expert annotator. Either '1' or '2'. 106 resize_inputs: Whether to resize the inputs to the patch shape. 107 download: Whether to download the data if it is not present. 108 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`. 109 110 Returns: 111 The segmentation dataset. 112 """ 113 image_paths, gt_paths = get_pandental_paths(path, annotator, download) 114 115 if resize_inputs: 116 resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False} 117 kwargs, patch_shape = util.update_kwargs_for_resize_trafo( 118 kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs 119 ) 120 121 return torch_em.default_segmentation_dataset( 122 raw_paths=image_paths, 123 raw_key=None, 124 label_paths=gt_paths, 125 label_key=None, 126 is_seg_dataset=False, 127 patch_shape=patch_shape, 128 **kwargs 129 )
Get the PanDental dataset for mandible segmentation in panoramic dental radiographs.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- patch_shape: The patch shape to use for training.
- annotator: The choice of expert annotator. Either '1' or '2'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_dataset.
Returns:
The segmentation dataset.
def
get_pandental_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], annotator: Literal['1', '2'] = '1', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
132def get_pandental_loader( 133 path: Union[os.PathLike, str], 134 batch_size: int, 135 patch_shape: Tuple[int, int], 136 annotator: Literal["1", "2"] = "1", 137 resize_inputs: bool = False, 138 download: bool = False, 139 **kwargs 140) -> DataLoader: 141 """Get the PanDental dataloader for mandible segmentation in panoramic dental radiographs. 142 143 Args: 144 path: Filepath to a folder where the data is downloaded for further processing. 145 batch_size: The batch size for training. 146 patch_shape: The patch shape to use for training. 147 annotator: The choice of expert annotator. Either '1' or '2'. 148 resize_inputs: Whether to resize the inputs to the patch shape. 149 download: Whether to download the data if it is not present. 150 kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader. 151 152 Returns: 153 The DataLoader. 154 """ 155 ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs) 156 dataset = get_pandental_dataset(path, patch_shape, annotator, resize_inputs, download, **ds_kwargs) 157 return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
Get the PanDental dataloader for mandible segmentation in panoramic dental radiographs.
Arguments:
- path: Filepath to a folder where the data is downloaded for further processing.
- batch_size: The batch size for training.
- patch_shape: The patch shape to use for training.
- annotator: The choice of expert annotator. Either '1' or '2'.
- resize_inputs: Whether to resize the inputs to the patch shape.
- download: Whether to download the data if it is not present.
- kwargs: Additional keyword arguments for
torch_em.default_segmentation_datasetor for the PyTorch DataLoader.
Returns:
The DataLoader.