torch_em.data.datasets.medical.pfus1

PFUS1 is a dataset for segmentation of pelvic floor anatomical structures in transperineal ultrasound images, acquired in the midsagittal plane.

The dataset consists of 110 patients (P000-P109), each with several video frames. Every frame is annotated with polygons for 8 anatomical structures: pubis, urethra, bladder, vagina, uterus, anus, rectum and levator ani muscle.

This dataset is located at https://doi.org/10.5281/zenodo.10800787 (Zenodo, CC BY 4.0). The dataset is from the publication https://doi.org/10.1016/j.dib.2025.112346. Please cite it if you use this dataset for your research.

  1"""PFUS1 is a dataset for segmentation of pelvic floor anatomical structures in transperineal
  2ultrasound images, acquired in the midsagittal plane.
  3
  4The dataset consists of 110 patients (P000-P109), each with several video frames. Every frame is
  5annotated with polygons for 8 anatomical structures: pubis, urethra, bladder, vagina, uterus, anus,
  6rectum and levator ani muscle.
  7
  8This dataset is located at https://doi.org/10.5281/zenodo.10800787 (Zenodo, CC BY 4.0).
  9The dataset is from the publication https://doi.org/10.1016/j.dib.2025.112346.
 10Please cite it if you use this dataset for your research.
 11"""
 12
 13import os
 14import json
 15from glob import glob
 16from tqdm import tqdm
 17from natsort import natsorted
 18from typing import Union, Tuple, List
 19
 20import numpy as np
 21from PIL import Image
 22import imageio.v3 as imageio
 23
 24from torch.utils.data import Dataset, DataLoader
 25
 26import torch_em
 27
 28from .. import util
 29
 30
 31URL = "https://zenodo.org/records/10800787/files/pfus1.zip"
 32CHECKSUM = "e65bd7ca941b895ca90c5891b8866ab78bee4575fc7a7c276b9a655456eea380"
 33
 34LABEL_MAP = {
 35    "Pubis": 1,
 36    "Urethra": 2,
 37    "Bladder": 3,
 38    "Vagina": 4,
 39    "Uterus": 5,
 40    "Anus": 6,
 41    "Rectum": 7,
 42    "Levator ani muscle": 8,
 43}
 44
 45
 46def get_pfus1_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 47    """Download the PFUS1 data.
 48
 49    Args:
 50        path: Filepath to a folder where the data is downloaded for further processing.
 51        download: Whether to download the data if it is not present.
 52
 53    Returns:
 54        Filepath where the data is downloaded.
 55    """
 56    data_dir = os.path.join(path, "data")
 57    if os.path.exists(data_dir):
 58        return data_dir
 59
 60    os.makedirs(path, exist_ok=True)
 61
 62    zip_path = os.path.join(path, "pfus1.zip")
 63    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 64    util.unzip(zip_path=zip_path, dst=path)
 65
 66    return data_dir
 67
 68
 69def _rasterize_labels(annotation_path, image_path, label_path):
 70    if os.path.exists(label_path):
 71        return
 72
 73    with open(annotation_path) as f:
 74        annotation = json.load(f)
 75
 76    from skimage.draw import polygon as draw_polygon
 77
 78    with Image.open(image_path) as im:
 79        shape = (im.height, im.width)
 80
 81    labels = np.zeros(shape, dtype="uint8")
 82    for shape_annotation in annotation:
 83        label_id = LABEL_MAP[shape_annotation["label"]]
 84        points = np.array(shape_annotation["pol"], dtype=float)
 85        rows, columns = draw_polygon(points[:, 1], points[:, 0], shape=shape)
 86        labels[rows, columns] = label_id
 87
 88    os.makedirs(os.path.dirname(label_path), exist_ok=True)
 89    imageio.imwrite(label_path, labels, compression="zlib")
 90
 91
 92def get_pfus1_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 93    """Get paths to the PFUS1 data.
 94
 95    Args:
 96        path: Filepath to a folder where the data is downloaded for further processing.
 97        download: Whether to download the data if it is not present.
 98
 99    Returns:
100        List of filepaths for the image data.
101        List of filepaths for the label data.
102    """
103    data_dir = get_pfus1_data(path, download)
104
105    image_paths = natsorted(glob(os.path.join(data_dir, "P*", "frame_*.png")))
106    assert len(image_paths) > 0
107
108    label_dir = os.path.join(os.path.dirname(data_dir), "labels")
109    label_paths = []
110    for image_path in tqdm(image_paths, desc="Rasterize the PFUS1 annotations"):
111        patient_id = os.path.basename(os.path.dirname(image_path))
112        frame_id = os.path.splitext(os.path.basename(image_path))[0]
113
114        annotation_path = os.path.join(data_dir, patient_id, f"{frame_id}.json")
115        label_path = os.path.join(label_dir, patient_id, f"{frame_id}.tif")
116
117        _rasterize_labels(annotation_path, image_path, label_path)
118        label_paths.append(label_path)
119
120    assert len(image_paths) == len(label_paths)
121
122    return image_paths, label_paths
123
124
125def get_pfus1_dataset(
126    path: Union[os.PathLike, str],
127    patch_shape: Tuple[int, int],
128    resize_inputs: bool = False,
129    download: bool = False,
130    **kwargs
131) -> Dataset:
132    """Get the PFUS1 dataset for segmentation of pelvic floor anatomical structures in ultrasound.
133
134    Args:
135        path: Filepath to a folder where the data is downloaded for further processing.
136        patch_shape: The patch shape to use for training.
137        resize_inputs: Whether to resize the inputs to the patch shape.
138        download: Whether to download the data if it is not present.
139        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
140
141    Returns:
142        The segmentation dataset.
143    """
144    image_paths, label_paths = get_pfus1_paths(path, download)
145
146    if resize_inputs:
147        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
148        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
149            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
150        )
151
152    return torch_em.default_segmentation_dataset(
153        raw_paths=image_paths,
154        raw_key=None,
155        label_paths=label_paths,
156        label_key=None,
157        patch_shape=patch_shape,
158        is_seg_dataset=False,
159        **kwargs
160    )
161
162
163def get_pfus1_loader(
164    path: Union[os.PathLike, str],
165    batch_size: int,
166    patch_shape: Tuple[int, int],
167    resize_inputs: bool = False,
168    download: bool = False,
169    **kwargs
170) -> DataLoader:
171    """Get the PFUS1 dataloader for segmentation of pelvic floor anatomical structures in ultrasound.
172
173    Args:
174        path: Filepath to a folder where the data is downloaded for further processing.
175        batch_size: The batch size for training.
176        patch_shape: The patch shape to use for training.
177        resize_inputs: Whether to resize the inputs to the patch shape.
178        download: Whether to download the data if it is not present.
179        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
180
181    Returns:
182        The DataLoader.
183    """
184    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
185    dataset = get_pfus1_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
186    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/records/10800787/files/pfus1.zip'
CHECKSUM = 'e65bd7ca941b895ca90c5891b8866ab78bee4575fc7a7c276b9a655456eea380'
LABEL_MAP = {'Pubis': 1, 'Urethra': 2, 'Bladder': 3, 'Vagina': 4, 'Uterus': 5, 'Anus': 6, 'Rectum': 7, 'Levator ani muscle': 8}
def get_pfus1_data(path: Union[os.PathLike, str], download: bool = False) -> str:
47def get_pfus1_data(path: Union[os.PathLike, str], download: bool = False) -> str:
48    """Download the PFUS1 data.
49
50    Args:
51        path: Filepath to a folder where the data is downloaded for further processing.
52        download: Whether to download the data if it is not present.
53
54    Returns:
55        Filepath where the data is downloaded.
56    """
57    data_dir = os.path.join(path, "data")
58    if os.path.exists(data_dir):
59        return data_dir
60
61    os.makedirs(path, exist_ok=True)
62
63    zip_path = os.path.join(path, "pfus1.zip")
64    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
65    util.unzip(zip_path=zip_path, dst=path)
66
67    return data_dir

Download the PFUS1 data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_pfus1_paths( path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 93def get_pfus1_paths(path: Union[os.PathLike, str], download: bool = False) -> Tuple[List[str], List[str]]:
 94    """Get paths to the PFUS1 data.
 95
 96    Args:
 97        path: Filepath to a folder where the data is downloaded for further processing.
 98        download: Whether to download the data if it is not present.
 99
100    Returns:
101        List of filepaths for the image data.
102        List of filepaths for the label data.
103    """
104    data_dir = get_pfus1_data(path, download)
105
106    image_paths = natsorted(glob(os.path.join(data_dir, "P*", "frame_*.png")))
107    assert len(image_paths) > 0
108
109    label_dir = os.path.join(os.path.dirname(data_dir), "labels")
110    label_paths = []
111    for image_path in tqdm(image_paths, desc="Rasterize the PFUS1 annotations"):
112        patient_id = os.path.basename(os.path.dirname(image_path))
113        frame_id = os.path.splitext(os.path.basename(image_path))[0]
114
115        annotation_path = os.path.join(data_dir, patient_id, f"{frame_id}.json")
116        label_path = os.path.join(label_dir, patient_id, f"{frame_id}.tif")
117
118        _rasterize_labels(annotation_path, image_path, label_path)
119        label_paths.append(label_path)
120
121    assert len(image_paths) == len(label_paths)
122
123    return image_paths, label_paths

Get paths to the PFUS1 data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_pfus1_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
126def get_pfus1_dataset(
127    path: Union[os.PathLike, str],
128    patch_shape: Tuple[int, int],
129    resize_inputs: bool = False,
130    download: bool = False,
131    **kwargs
132) -> Dataset:
133    """Get the PFUS1 dataset for segmentation of pelvic floor anatomical structures in ultrasound.
134
135    Args:
136        path: Filepath to a folder where the data is downloaded for further processing.
137        patch_shape: The patch shape to use for training.
138        resize_inputs: Whether to resize the inputs to the patch shape.
139        download: Whether to download the data if it is not present.
140        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
141
142    Returns:
143        The segmentation dataset.
144    """
145    image_paths, label_paths = get_pfus1_paths(path, download)
146
147    if resize_inputs:
148        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": False}
149        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
150            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
151        )
152
153    return torch_em.default_segmentation_dataset(
154        raw_paths=image_paths,
155        raw_key=None,
156        label_paths=label_paths,
157        label_key=None,
158        patch_shape=patch_shape,
159        is_seg_dataset=False,
160        **kwargs
161    )

Get the PFUS1 dataset for segmentation of pelvic floor anatomical structures in ultrasound.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_pfus1_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
164def get_pfus1_loader(
165    path: Union[os.PathLike, str],
166    batch_size: int,
167    patch_shape: Tuple[int, int],
168    resize_inputs: bool = False,
169    download: bool = False,
170    **kwargs
171) -> DataLoader:
172    """Get the PFUS1 dataloader for segmentation of pelvic floor anatomical structures in ultrasound.
173
174    Args:
175        path: Filepath to a folder where the data is downloaded for further processing.
176        batch_size: The batch size for training.
177        patch_shape: The patch shape to use for training.
178        resize_inputs: Whether to resize the inputs to the patch shape.
179        download: Whether to download the data if it is not present.
180        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
181
182    Returns:
183        The DataLoader.
184    """
185    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
186    dataset = get_pfus1_dataset(path, patch_shape, resize_inputs, download, **ds_kwargs)
187    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the PFUS1 dataloader for segmentation of pelvic floor anatomical structures in ultrasound.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.