torch_em.data.datasets.medical.pandental

The PanDental dataset contains annotations for mandible segmentation in panoramic dental radiographs.

This dataset is located at https://data.mendeley.com/datasets/hxt48yk462/2 This dataset is from the publication https://doi.org/10.1117/1.JMI.2.4.044003. Please cite it if you use this dataset for your research.

  1"""The PanDental dataset contains annotations for mandible segmentation in
  2panoramic dental radiographs.
  3
  4This dataset is located at https://data.mendeley.com/datasets/hxt48yk462/2
  5This dataset is from the publication https://doi.org/10.1117/1.JMI.2.4.044003.
  6Please cite it if you use this dataset for your research.
  7"""
  8
  9import os
 10from glob import glob
 11from tqdm import tqdm
 12from pathlib import Path
 13from natsort import natsorted
 14from typing import Union, Literal, Tuple, List
 15
 16import numpy as np
 17import imageio.v3 as imageio
 18
 19from torch.utils.data import Dataset, DataLoader
 20
 21import torch_em
 22
 23from .. import util
 24
 25
 26URL = "https://data.mendeley.com/public-files/datasets/hxt48yk462/files/c2df1e1e-9939-4197-9bac-1eb697a64094/file_downloaded"  # noqa
 27CHECKSUM = "4ab6f670428df8052ae04aef82011050cd5ad4805f4d28d32fea523880752cf3"
 28
 29
 30def get_pandental_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 31    """Download the PanDental dataset.
 32
 33    Args:
 34        path: Filepath to a folder where the data is downloaded for further processing.
 35        download: Whether to download the data if it is not present.
 36
 37    Returns:
 38        Filepath where the data is downloaded.
 39    """
 40    data_dir = os.path.join(path, "Images")
 41    if os.path.exists(data_dir):
 42        return path
 43
 44    os.makedirs(path, exist_ok=True)
 45
 46    zip_path = os.path.join(path, "DentalPanoramicXrays.zip")
 47    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 48    util.unzip(zip_path=zip_path, dst=path)
 49
 50    return path
 51
 52
 53def get_pandental_paths(
 54    path: Union[os.PathLike, str], annotator: Literal["1", "2"] = "1", download: bool = False
 55) -> Tuple[List[str], List[str]]:
 56    """Get paths to the PanDental data.
 57
 58    Args:
 59        path: Filepath to a folder where the data is downloaded for further processing.
 60        annotator: The choice of expert annotator. Either '1' or '2'.
 61        download: Whether to download the data if it is not present.
 62
 63    Returns:
 64        List of filepaths for the image data.
 65        List of filepaths for the label data.
 66    """
 67    data_dir = get_pandental_data(path=path, download=download)
 68
 69    image_paths = natsorted(glob(os.path.join(data_dir, "Images", "*.png")))
 70    raw_gt_paths = natsorted(glob(os.path.join(data_dir, f"Segmentation{annotator}", "*.png")))
 71
 72    neu_gt_dir = os.path.join(data_dir, "preprocessed", f"gt{annotator}")
 73    os.makedirs(neu_gt_dir, exist_ok=True)
 74
 75    gt_paths = []
 76    for raw_gt_path in tqdm(raw_gt_paths, desc="Preprocessing labels"):
 77        gt_path = os.path.join(neu_gt_dir, f"{Path(raw_gt_path).stem}.tif")
 78        gt_paths.append(gt_path)
 79        if os.path.exists(gt_path):
 80            continue
 81
 82        # the ground-truth is the original image masked by the binary mandible region,
 83        # i.e. non-zero pixels correspond to the mandible.
 84        raw_gt = imageio.imread(raw_gt_path)
 85        binary_gt = (raw_gt > 0).astype(np.uint8)
 86        imageio.imwrite(gt_path, binary_gt)
 87
 88    return image_paths, gt_paths
 89
 90
 91def get_pandental_dataset(
 92    path: Union[os.PathLike, str],
 93    patch_shape: Tuple[int, int],
 94    annotator: Literal["1", "2"] = "1",
 95    resize_inputs: bool = False,
 96    download: bool = False,
 97    **kwargs
 98) -> Dataset:
 99    """Get the PanDental dataset for mandible segmentation in panoramic dental radiographs.
100
101    Args:
102        path: Filepath to a folder where the data is downloaded for further processing.
103        patch_shape: The patch shape to use for training.
104        annotator: The choice of expert annotator. Either '1' or '2'.
105        resize_inputs: Whether to resize the inputs to the patch shape.
106        download: Whether to download the data if it is not present.
107        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
108
109    Returns:
110        The segmentation dataset.
111    """
112    image_paths, gt_paths = get_pandental_paths(path, annotator, download)
113
114    if resize_inputs:
115        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
116        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
117            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
118        )
119
120    return torch_em.default_segmentation_dataset(
121        raw_paths=image_paths,
122        raw_key=None,
123        label_paths=gt_paths,
124        label_key=None,
125        is_seg_dataset=False,
126        patch_shape=patch_shape,
127        **kwargs
128    )
129
130
131def get_pandental_loader(
132    path: Union[os.PathLike, str],
133    batch_size: int,
134    patch_shape: Tuple[int, int],
135    annotator: Literal["1", "2"] = "1",
136    resize_inputs: bool = False,
137    download: bool = False,
138    **kwargs
139) -> DataLoader:
140    """Get the PanDental dataloader for mandible segmentation in panoramic dental radiographs.
141
142    Args:
143        path: Filepath to a folder where the data is downloaded for further processing.
144        batch_size: The batch size for training.
145        patch_shape: The patch shape to use for training.
146        annotator: The choice of expert annotator. Either '1' or '2'.
147        resize_inputs: Whether to resize the inputs to the patch shape.
148        download: Whether to download the data if it is not present.
149        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
150
151    Returns:
152        The DataLoader.
153    """
154    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
155    dataset = get_pandental_dataset(path, patch_shape, annotator, resize_inputs, download, **ds_kwargs)
156    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://data.mendeley.com/public-files/datasets/hxt48yk462/files/c2df1e1e-9939-4197-9bac-1eb697a64094/file_downloaded'
CHECKSUM = '4ab6f670428df8052ae04aef82011050cd5ad4805f4d28d32fea523880752cf3'
def get_pandental_data(path: Union[os.PathLike, str], download: bool = False) -> str:
31def get_pandental_data(path: Union[os.PathLike, str], download: bool = False) -> str:
32    """Download the PanDental dataset.
33
34    Args:
35        path: Filepath to a folder where the data is downloaded for further processing.
36        download: Whether to download the data if it is not present.
37
38    Returns:
39        Filepath where the data is downloaded.
40    """
41    data_dir = os.path.join(path, "Images")
42    if os.path.exists(data_dir):
43        return path
44
45    os.makedirs(path, exist_ok=True)
46
47    zip_path = os.path.join(path, "DentalPanoramicXrays.zip")
48    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
49    util.unzip(zip_path=zip_path, dst=path)
50
51    return path

Download the PanDental dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_pandental_paths( path: Union[os.PathLike, str], annotator: Literal['1', '2'] = '1', download: bool = False) -> Tuple[List[str], List[str]]:
54def get_pandental_paths(
55    path: Union[os.PathLike, str], annotator: Literal["1", "2"] = "1", download: bool = False
56) -> Tuple[List[str], List[str]]:
57    """Get paths to the PanDental data.
58
59    Args:
60        path: Filepath to a folder where the data is downloaded for further processing.
61        annotator: The choice of expert annotator. Either '1' or '2'.
62        download: Whether to download the data if it is not present.
63
64    Returns:
65        List of filepaths for the image data.
66        List of filepaths for the label data.
67    """
68    data_dir = get_pandental_data(path=path, download=download)
69
70    image_paths = natsorted(glob(os.path.join(data_dir, "Images", "*.png")))
71    raw_gt_paths = natsorted(glob(os.path.join(data_dir, f"Segmentation{annotator}", "*.png")))
72
73    neu_gt_dir = os.path.join(data_dir, "preprocessed", f"gt{annotator}")
74    os.makedirs(neu_gt_dir, exist_ok=True)
75
76    gt_paths = []
77    for raw_gt_path in tqdm(raw_gt_paths, desc="Preprocessing labels"):
78        gt_path = os.path.join(neu_gt_dir, f"{Path(raw_gt_path).stem}.tif")
79        gt_paths.append(gt_path)
80        if os.path.exists(gt_path):
81            continue
82
83        # the ground-truth is the original image masked by the binary mandible region,
84        # i.e. non-zero pixels correspond to the mandible.
85        raw_gt = imageio.imread(raw_gt_path)
86        binary_gt = (raw_gt > 0).astype(np.uint8)
87        imageio.imwrite(gt_path, binary_gt)
88
89    return image_paths, gt_paths

Get paths to the PanDental data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • annotator: The choice of expert annotator. Either '1' or '2'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_pandental_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], annotator: Literal['1', '2'] = '1', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 92def get_pandental_dataset(
 93    path: Union[os.PathLike, str],
 94    patch_shape: Tuple[int, int],
 95    annotator: Literal["1", "2"] = "1",
 96    resize_inputs: bool = False,
 97    download: bool = False,
 98    **kwargs
 99) -> Dataset:
100    """Get the PanDental dataset for mandible segmentation in panoramic dental radiographs.
101
102    Args:
103        path: Filepath to a folder where the data is downloaded for further processing.
104        patch_shape: The patch shape to use for training.
105        annotator: The choice of expert annotator. Either '1' or '2'.
106        resize_inputs: Whether to resize the inputs to the patch shape.
107        download: Whether to download the data if it is not present.
108        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
109
110    Returns:
111        The segmentation dataset.
112    """
113    image_paths, gt_paths = get_pandental_paths(path, annotator, download)
114
115    if resize_inputs:
116        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
117        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
118            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
119        )
120
121    return torch_em.default_segmentation_dataset(
122        raw_paths=image_paths,
123        raw_key=None,
124        label_paths=gt_paths,
125        label_key=None,
126        is_seg_dataset=False,
127        patch_shape=patch_shape,
128        **kwargs
129    )

Get the PanDental dataset for mandible segmentation in panoramic dental radiographs.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • annotator: The choice of expert annotator. Either '1' or '2'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_pandental_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], annotator: Literal['1', '2'] = '1', resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
132def get_pandental_loader(
133    path: Union[os.PathLike, str],
134    batch_size: int,
135    patch_shape: Tuple[int, int],
136    annotator: Literal["1", "2"] = "1",
137    resize_inputs: bool = False,
138    download: bool = False,
139    **kwargs
140) -> DataLoader:
141    """Get the PanDental dataloader for mandible segmentation in panoramic dental radiographs.
142
143    Args:
144        path: Filepath to a folder where the data is downloaded for further processing.
145        batch_size: The batch size for training.
146        patch_shape: The patch shape to use for training.
147        annotator: The choice of expert annotator. Either '1' or '2'.
148        resize_inputs: Whether to resize the inputs to the patch shape.
149        download: Whether to download the data if it is not present.
150        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
151
152    Returns:
153        The DataLoader.
154    """
155    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
156    dataset = get_pandental_dataset(path, patch_shape, annotator, resize_inputs, download, **ds_kwargs)
157    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the PanDental dataloader for mandible segmentation in panoramic dental radiographs.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • annotator: The choice of expert annotator. Either '1' or '2'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.