torch_em.data.datasets.medical.fugc

The FUGC dataset contains annotations for cervical lip segmentation in transvaginal ultrasound images.

The dataset consists of 890 transvaginal ultrasound images (336x544 pixels, RGB) from the Fetal Ultrasound Grand Challenge on semi-supervised cervical segmentation. The anterior and posterior cervical lips were segmented automatically and then manually corrected by an experienced radiologist.

Labeled images are available for three splits: 'train' (50), 'val' (90) and 'test' (300). The training split additionally contains 450 unlabeled images, which are not used by this loader. The label ids are 0 (background), 1 (anterior lip) and 2 (posterior lip). The record only names the two lips without listing the ids, so this order is inferred: the structure with id 1 lies above the one with id 2 in all 140 labeled training and validation images.

The data is located at https://doi.org/10.5281/zenodo.16893174, released under a CC-BY-4.0 license.

Please cite the Zenodo record if you use this dataset for your research.

  1"""The FUGC dataset contains annotations for cervical lip segmentation in transvaginal ultrasound images.
  2
  3The dataset consists of 890 transvaginal ultrasound images (336x544 pixels, RGB) from the Fetal Ultrasound
  4Grand Challenge on semi-supervised cervical segmentation. The anterior and posterior cervical lips were
  5segmented automatically and then manually corrected by an experienced radiologist.
  6
  7Labeled images are available for three splits: 'train' (50), 'val' (90) and 'test' (300). The training split
  8additionally contains 450 unlabeled images, which are not used by this loader. The label ids are 0 (background),
  91 (anterior lip) and 2 (posterior lip). The record only names the two lips without listing the ids, so this
 10order is inferred: the structure with id 1 lies above the one with id 2 in all 140 labeled training and
 11validation images.
 12
 13The data is located at https://doi.org/10.5281/zenodo.16893174, released under a CC-BY-4.0 license.
 14
 15Please cite the Zenodo record if you use this dataset for your research.
 16"""
 17
 18import os
 19from glob import glob
 20from natsort import natsorted
 21from typing import Union, Tuple, Literal, List
 22
 23from torch.utils.data import Dataset, DataLoader
 24
 25import torch_em
 26
 27from .. import util
 28
 29
 30URL = "https://zenodo.org/records/16893174/files/FUGC%20(Dataset).zip"
 31CHECKSUM = "e4dea63dd744885835c9ce8035a640a6ae42780ae21180601ddb09590fa73ce4"
 32
 33SPLITS = {"train": os.path.join("train", "labeled_data"), "val": "val", "test": "test"}
 34
 35
 36def get_fugc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 37    """Download the FUGC dataset.
 38
 39    Args:
 40        path: Filepath to a folder where the data is downloaded for further processing.
 41        download: Whether to download the data if it is not present.
 42
 43    Returns:
 44        Filepath where the data is downloaded.
 45    """
 46    data_dir = os.path.join(path, "FUGC (Dataset)", "dataset")
 47    if os.path.exists(data_dir):
 48        return data_dir
 49
 50    os.makedirs(path, exist_ok=True)
 51
 52    zip_path = os.path.join(path, "FUGC.zip")
 53    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
 54    util.unzip(zip_path=zip_path, dst=path)
 55
 56    assert os.path.exists(data_dir), f"The extraction of the FUGC archive did not create '{data_dir}'."
 57
 58    return data_dir
 59
 60
 61def get_fugc_paths(
 62    path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False,
 63) -> Tuple[List[str], List[str]]:
 64    """Get paths to the FUGC data.
 65
 66    Args:
 67        path: Filepath to a folder where the data is downloaded for further processing.
 68        split: The choice of data split. One of 'train', 'val' or 'test'.
 69        download: Whether to download the data if it is not present.
 70
 71    Returns:
 72        List of filepaths for the image data.
 73        List of filepaths for the label data.
 74    """
 75    if split not in SPLITS:
 76        raise ValueError(f"'{split}' is not a valid split. Choose one of {list(SPLITS)}.")
 77
 78    data_dir = get_fugc_data(path, download)
 79
 80    label_paths = natsorted(glob(os.path.join(data_dir, SPLITS[split], "labels", "*.png")))
 81    raw_paths = [os.path.join(data_dir, SPLITS[split], "images", os.path.basename(p)) for p in label_paths]
 82
 83    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
 84    assert all(os.path.exists(p) for p in raw_paths)
 85
 86    return raw_paths, label_paths
 87
 88
 89def get_fugc_dataset(
 90    path: Union[os.PathLike, str],
 91    patch_shape: Tuple[int, int],
 92    split: Literal["train", "val", "test"],
 93    resize_inputs: bool = False,
 94    download: bool = False,
 95    **kwargs
 96) -> Dataset:
 97    """Get the FUGC dataset for cervical lip segmentation in transvaginal ultrasound.
 98
 99    Args:
100        path: Filepath to a folder where the data is downloaded for further processing.
101        patch_shape: The patch shape to use for training.
102        split: The choice of data split. One of 'train', 'val' or 'test'.
103        resize_inputs: Whether to resize the inputs to the patch shape.
104        download: Whether to download the data if it is not present.
105        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
106
107    Returns:
108        The segmentation dataset.
109    """
110    raw_paths, label_paths = get_fugc_paths(path, split, download)
111
112    if resize_inputs:
113        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
114        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
115            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
116        )
117
118    return torch_em.default_segmentation_dataset(
119        raw_paths=raw_paths,
120        raw_key=None,
121        label_paths=label_paths,
122        label_key=None,
123        is_seg_dataset=False,
124        patch_shape=patch_shape,
125        **kwargs
126    )
127
128
129def get_fugc_loader(
130    path: Union[os.PathLike, str],
131    batch_size: int,
132    patch_shape: Tuple[int, int],
133    split: Literal["train", "val", "test"],
134    resize_inputs: bool = False,
135    download: bool = False,
136    **kwargs
137) -> DataLoader:
138    """Get the FUGC dataloader for cervical lip segmentation in transvaginal ultrasound.
139
140    Args:
141        path: Filepath to a folder where the data is downloaded for further processing.
142        batch_size: The batch size for training.
143        patch_shape: The patch shape to use for training.
144        split: The choice of data split. One of 'train', 'val' or 'test'.
145        resize_inputs: Whether to resize the inputs to the patch shape.
146        download: Whether to download the data if it is not present.
147        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
148
149    Returns:
150        The DataLoader.
151    """
152    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
153    dataset = get_fugc_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs)
154    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URL = 'https://zenodo.org/records/16893174/files/FUGC%20(Dataset).zip'
CHECKSUM = 'e4dea63dd744885835c9ce8035a640a6ae42780ae21180601ddb09590fa73ce4'
SPLITS = {'train': 'train/labeled_data', 'val': 'val', 'test': 'test'}
def get_fugc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
37def get_fugc_data(path: Union[os.PathLike, str], download: bool = False) -> str:
38    """Download the FUGC dataset.
39
40    Args:
41        path: Filepath to a folder where the data is downloaded for further processing.
42        download: Whether to download the data if it is not present.
43
44    Returns:
45        Filepath where the data is downloaded.
46    """
47    data_dir = os.path.join(path, "FUGC (Dataset)", "dataset")
48    if os.path.exists(data_dir):
49        return data_dir
50
51    os.makedirs(path, exist_ok=True)
52
53    zip_path = os.path.join(path, "FUGC.zip")
54    util.download_source(path=zip_path, url=URL, download=download, checksum=CHECKSUM)
55    util.unzip(zip_path=zip_path, dst=path)
56
57    assert os.path.exists(data_dir), f"The extraction of the FUGC archive did not create '{data_dir}'."
58
59    return data_dir

Download the FUGC dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_fugc_paths( path: Union[os.PathLike, str], split: Literal['train', 'val', 'test'], download: bool = False) -> Tuple[List[str], List[str]]:
62def get_fugc_paths(
63    path: Union[os.PathLike, str], split: Literal["train", "val", "test"], download: bool = False,
64) -> Tuple[List[str], List[str]]:
65    """Get paths to the FUGC data.
66
67    Args:
68        path: Filepath to a folder where the data is downloaded for further processing.
69        split: The choice of data split. One of 'train', 'val' or 'test'.
70        download: Whether to download the data if it is not present.
71
72    Returns:
73        List of filepaths for the image data.
74        List of filepaths for the label data.
75    """
76    if split not in SPLITS:
77        raise ValueError(f"'{split}' is not a valid split. Choose one of {list(SPLITS)}.")
78
79    data_dir = get_fugc_data(path, download)
80
81    label_paths = natsorted(glob(os.path.join(data_dir, SPLITS[split], "labels", "*.png")))
82    raw_paths = [os.path.join(data_dir, SPLITS[split], "images", os.path.basename(p)) for p in label_paths]
83
84    assert len(raw_paths) == len(label_paths) and len(raw_paths) > 0
85    assert all(os.path.exists(p) for p in raw_paths)
86
87    return raw_paths, label_paths

Get paths to the FUGC data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • split: The choice of data split. One of 'train', 'val' or 'test'.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_fugc_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
 90def get_fugc_dataset(
 91    path: Union[os.PathLike, str],
 92    patch_shape: Tuple[int, int],
 93    split: Literal["train", "val", "test"],
 94    resize_inputs: bool = False,
 95    download: bool = False,
 96    **kwargs
 97) -> Dataset:
 98    """Get the FUGC dataset for cervical lip segmentation in transvaginal ultrasound.
 99
100    Args:
101        path: Filepath to a folder where the data is downloaded for further processing.
102        patch_shape: The patch shape to use for training.
103        split: The choice of data split. One of 'train', 'val' or 'test'.
104        resize_inputs: Whether to resize the inputs to the patch shape.
105        download: Whether to download the data if it is not present.
106        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
107
108    Returns:
109        The segmentation dataset.
110    """
111    raw_paths, label_paths = get_fugc_paths(path, split, download)
112
113    if resize_inputs:
114        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
115        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
116            kwargs=kwargs, patch_shape=patch_shape, resize_inputs=resize_inputs, resize_kwargs=resize_kwargs
117        )
118
119    return torch_em.default_segmentation_dataset(
120        raw_paths=raw_paths,
121        raw_key=None,
122        label_paths=label_paths,
123        label_key=None,
124        is_seg_dataset=False,
125        patch_shape=patch_shape,
126        **kwargs
127    )

Get the FUGC dataset for cervical lip segmentation in transvaginal ultrasound.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split. One of 'train', 'val' or 'test'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_fugc_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], split: Literal['train', 'val', 'test'], resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
130def get_fugc_loader(
131    path: Union[os.PathLike, str],
132    batch_size: int,
133    patch_shape: Tuple[int, int],
134    split: Literal["train", "val", "test"],
135    resize_inputs: bool = False,
136    download: bool = False,
137    **kwargs
138) -> DataLoader:
139    """Get the FUGC dataloader for cervical lip segmentation in transvaginal ultrasound.
140
141    Args:
142        path: Filepath to a folder where the data is downloaded for further processing.
143        batch_size: The batch size for training.
144        patch_shape: The patch shape to use for training.
145        split: The choice of data split. One of 'train', 'val' or 'test'.
146        resize_inputs: Whether to resize the inputs to the patch shape.
147        download: Whether to download the data if it is not present.
148        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
149
150    Returns:
151        The DataLoader.
152    """
153    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
154    dataset = get_fugc_dataset(path, patch_shape, split, resize_inputs, download, **ds_kwargs)
155    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the FUGC dataloader for cervical lip segmentation in transvaginal ultrasound.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The choice of data split. One of 'train', 'val' or 'test'.
  • resize_inputs: Whether to resize the inputs to the patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.