torch_em.data.datasets.medical.ima_plus_plus

The IMA++ (ISIC Archive Multi-Annotator) dataset contains multiple independent human lesion segmentation masks for dermoscopy images from the ISIC Archive, for the task of studying inter-annotator variability in skin lesion segmentation.

This dataset is located at https://doi.org/10.5281/zenodo.14201693, under the CC BY-NC-ND 4.0 license (non-commercial use, no derivatives). The dataset is from the publication https://doi.org/10.48550/arXiv.2512.21472. Please cite it if you use this dataset for your research.

The dataset contains 17,684+ segmentation masks (single-annotator, majority-vote consensus 'MV', and STAPLE consensus 'ST') spanning 14,967 dermoscopic images, of which 2,394 images have 2-5 independent masks annotated by up to 16 distinct annotators. Zenodo only hosts the masks and metadata; the corresponding raw images are distributed separately via the ISIC Archive (as the dedicated "IMA++" collection, id 482) and are downloaded here from the public ISIC S3 bucket, using the image identifiers listed in the metadata.

NOTE: This is a much larger dataset than the ISIC 2018 challenge data already available in torch_em.data.datasets.medical.isic: it ships multiple independent masks per image (instead of a single ground truth) and its images are not bundled in a single archive, so downloading them requires a separate, per-image acquisition step.

  1"""The IMA++ (ISIC Archive Multi-Annotator) dataset contains multiple independent human
  2lesion segmentation masks for dermoscopy images from the ISIC Archive, for the task of
  3studying inter-annotator variability in skin lesion segmentation.
  4
  5This dataset is located at https://doi.org/10.5281/zenodo.14201693, under the
  6CC BY-NC-ND 4.0 license (non-commercial use, no derivatives). The dataset is from the
  7publication https://doi.org/10.48550/arXiv.2512.21472. Please cite it if you use this
  8dataset for your research.
  9
 10The dataset contains 17,684+ segmentation masks (single-annotator, majority-vote consensus
 11'MV', and STAPLE consensus 'ST') spanning 14,967 dermoscopic images, of which 2,394 images
 12have 2-5 independent masks annotated by up to 16 distinct annotators. Zenodo only hosts the
 13masks and metadata; the corresponding raw images are distributed separately via the ISIC
 14Archive (as the dedicated "IMA++" collection, id 482) and are downloaded here from the
 15public ISIC S3 bucket, using the image identifiers listed in the metadata.
 16
 17NOTE: This is a much larger dataset than the ISIC 2018 challenge data already available in
 18`torch_em.data.datasets.medical.isic`: it ships multiple independent masks per image (instead
 19of a single ground truth) and its images are not bundled in a single archive, so downloading
 20them requires a separate, per-image acquisition step.
 21"""
 22
 23import os
 24import csv
 25from concurrent.futures import ThreadPoolExecutor, as_completed
 26from typing import Union, Tuple, Optional, List
 27
 28import requests
 29from tqdm import tqdm
 30
 31from torch.utils.data import Dataset, DataLoader
 32
 33import torch_em
 34
 35from .. import util
 36from ..light_microscopy.neurips_cell_seg import to_rgb
 37
 38
 39URLS = {
 40    "segs": "https://zenodo.org/api/records/14201693/files/segs.zip/content",
 41    "seg_metadata": "https://zenodo.org/api/records/14201693/files/seg_metadata.csv/content",
 42    "img_metadata": "https://zenodo.org/api/records/14201693/files/img_metadata.csv/content",
 43}
 44CHECKSUMS = {
 45    "segs": "4141feef60a168d5c699599e0a2089e6c689661ba6630459a138721cbf74ddc2",
 46    "seg_metadata": None,
 47    "img_metadata": None,
 48}
 49IMAGE_URL = "https://isic-archive.s3.amazonaws.com/images/{}.jpg"
 50
 51CONSENSUS_ANNOTATORS = ["MV", "ST"]
 52
 53
 54def get_ima_plus_plus_data(path: Union[os.PathLike, str], download: bool = False) -> str:
 55    """Download the segmentation masks and metadata for the IMA++ dataset.
 56
 57    Args:
 58        path: Filepath to a folder where the data is downloaded for further processing.
 59        download: Whether to download the data if it is not present.
 60
 61    Returns:
 62        Filepath where the data is downloaded.
 63    """
 64    os.makedirs(path, exist_ok=True)
 65
 66    seg_dir = os.path.join(path, "segs")
 67    if not os.path.exists(seg_dir):
 68        zip_path = os.path.join(path, "segs.zip")
 69        util.download_source(path=zip_path, url=URLS["segs"], download=download, checksum=CHECKSUMS["segs"])
 70        util.unzip(zip_path=zip_path, dst=seg_dir, remove=False)
 71
 72    for name in ["seg_metadata", "img_metadata"]:
 73        csv_path = os.path.join(path, f"{name}.csv")
 74        util.download_source(path=csv_path, url=URLS[name], download=download, checksum=CHECKSUMS[name])
 75
 76    return path
 77
 78
 79def _read_seg_metadata(path):
 80    with open(os.path.join(path, "seg_metadata.csv")) as f:
 81        return list(csv.DictReader(f))
 82
 83
 84def _select_rows(rows, annotator):
 85    if annotator is None:
 86        # Default to a single mask per image: the STAPLE consensus mask for images that have
 87        # one, and the (unique) single-annotator mask otherwise.
 88        by_image = {}
 89        for row in rows:
 90            by_image.setdefault(row["ISIC_id"], []).append(row)
 91
 92        selected = []
 93        for isic_id, image_rows in by_image.items():
 94            consensus_rows = [row for row in image_rows if row["annotator"] == "ST"]
 95            selected.append(consensus_rows[0] if consensus_rows else image_rows[0])
 96
 97        return selected
 98
 99    elif annotator == "all":
100        return rows
101
102    else:
103        selected = [row for row in rows if row["annotator"] == annotator]
104        assert len(selected) > 0, f"'{annotator}' did not match any masks. See 'seg_metadata.csv' for valid values."
105        return selected
106
107
108def _download_images(isic_ids, image_dir, download):
109    os.makedirs(image_dir, exist_ok=True)
110
111    missing = [isic_id for isic_id in isic_ids if not os.path.exists(os.path.join(image_dir, f"{isic_id}.jpg"))]
112    if len(missing) == 0:
113        return
114
115    if not download:
116        raise RuntimeError(f"Cannot find {len(missing)} image(s) at {image_dir}, but download was set to False")
117
118    def _download_one(isic_id):
119        dst = os.path.join(image_dir, f"{isic_id}.jpg")
120        tmp = f"{dst}.incomplete"
121        response = requests.get(IMAGE_URL.format(isic_id), timeout=60)
122        response.raise_for_status()
123        with open(tmp, "wb") as f:
124            f.write(response.content)
125        os.replace(tmp, dst)
126
127    with ThreadPoolExecutor(max_workers=16) as pool:
128        futures = [pool.submit(_download_one, isic_id) for isic_id in missing]
129        for future in tqdm(as_completed(futures), total=len(futures), desc="Downloading IMA++ images"):
130            future.result()
131
132
133def get_ima_plus_plus_paths(
134    path: Union[os.PathLike, str], annotator: Optional[str] = None, download: bool = False
135) -> Tuple[List[str], List[str]]:
136    """Get paths to the IMA++ data.
137
138    Args:
139        path: Filepath to a folder where the data is downloaded for further processing.
140        annotator: The choice of annotator whose masks are used. By default (`None`), a single
141            mask per image is returned: the 'ST' (STAPLE) consensus mask for the 2,394 images
142            that have multiple annotations, and the unique single-annotator mask otherwise. Pass
143            `'all'` to get every mask (multiple entries per multi-annotator image), or a specific
144            annotator id (e.g. `'A04'`, `'MV'`, `'ST'`) to filter to masks from that source only.
145        download: Whether to download the data if it is not present.
146
147    Returns:
148        List of filepaths for the image data.
149        List of filepaths for the label data.
150    """
151    data_dir = get_ima_plus_plus_data(path=path, download=download)
152
153    rows = _select_rows(_read_seg_metadata(data_dir), annotator)
154
155    isic_ids = sorted(set(row["ISIC_id"] for row in rows))
156    image_dir = os.path.join(data_dir, "images")
157    _download_images(isic_ids, image_dir, download)
158
159    image_paths = [os.path.join(image_dir, f"{row['ISIC_id']}.jpg") for row in rows]
160    gt_paths = [os.path.join(data_dir, "segs", row["seg_filename"]) for row in rows]
161
162    assert len(image_paths) == len(gt_paths) and len(image_paths) > 0
163    for gt_path in gt_paths:
164        assert os.path.exists(gt_path), gt_path
165
166    return image_paths, gt_paths
167
168
169def get_ima_plus_plus_dataset(
170    path: Union[os.PathLike, str],
171    patch_shape: Tuple[int, int],
172    annotator: Optional[str] = None,
173    resize_inputs: bool = False,
174    download: bool = False,
175    **kwargs
176) -> Dataset:
177    """Get the IMA++ dataset for multi-annotator skin lesion segmentation in dermoscopy images.
178
179    Args:
180        path: Filepath to a folder where the downloaded data will be saved.
181        patch_shape: The patch shape to use for training.
182        annotator: The choice of annotator whose masks are used. See `get_ima_plus_plus_paths`.
183        resize_inputs: Whether to resize the inputs to the expected patch shape.
184        download: Whether to download the data if it is not present.
185        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
186
187    Returns:
188        The segmentation dataset.
189    """
190    image_paths, gt_paths = get_ima_plus_plus_paths(path=path, annotator=annotator, download=download)
191
192    if resize_inputs:
193        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
194        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
195            kwargs=kwargs,
196            patch_shape=patch_shape,
197            resize_inputs=resize_inputs,
198            resize_kwargs=resize_kwargs,
199            ensure_rgb=to_rgb,
200        )
201
202    return torch_em.default_segmentation_dataset(
203        raw_paths=image_paths,
204        raw_key=None,
205        label_paths=gt_paths,
206        label_key=None,
207        patch_shape=patch_shape,
208        is_seg_dataset=False,
209        **kwargs
210    )
211
212
213def get_ima_plus_plus_loader(
214    path: Union[os.PathLike, str],
215    batch_size: int,
216    patch_shape: Tuple[int, int],
217    annotator: Optional[str] = None,
218    resize_inputs: bool = False,
219    download: bool = False,
220    **kwargs
221) -> DataLoader:
222    """Get the IMA++ dataloader for multi-annotator skin lesion segmentation in dermoscopy images.
223
224    Args:
225        path: Filepath to a folder where the downloaded data will be saved.
226        batch_size: The batch size for training.
227        patch_shape: The patch shape to use for training.
228        annotator: The choice of annotator whose masks are used. See `get_ima_plus_plus_paths`.
229        resize_inputs: Whether to resize the inputs to the expected patch shape.
230        download: Whether to download the data if it is not present.
231        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
232
233    Returns:
234        The DataLoader.
235    """
236    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
237    dataset = get_ima_plus_plus_dataset(path, patch_shape, annotator, resize_inputs, download, **ds_kwargs)
238    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
URLS = {'segs': 'https://zenodo.org/api/records/14201693/files/segs.zip/content', 'seg_metadata': 'https://zenodo.org/api/records/14201693/files/seg_metadata.csv/content', 'img_metadata': 'https://zenodo.org/api/records/14201693/files/img_metadata.csv/content'}
CHECKSUMS = {'segs': '4141feef60a168d5c699599e0a2089e6c689661ba6630459a138721cbf74ddc2', 'seg_metadata': None, 'img_metadata': None}
IMAGE_URL = 'https://isic-archive.s3.amazonaws.com/images/{}.jpg'
CONSENSUS_ANNOTATORS = ['MV', 'ST']
def get_ima_plus_plus_data(path: Union[os.PathLike, str], download: bool = False) -> str:
55def get_ima_plus_plus_data(path: Union[os.PathLike, str], download: bool = False) -> str:
56    """Download the segmentation masks and metadata for the IMA++ dataset.
57
58    Args:
59        path: Filepath to a folder where the data is downloaded for further processing.
60        download: Whether to download the data if it is not present.
61
62    Returns:
63        Filepath where the data is downloaded.
64    """
65    os.makedirs(path, exist_ok=True)
66
67    seg_dir = os.path.join(path, "segs")
68    if not os.path.exists(seg_dir):
69        zip_path = os.path.join(path, "segs.zip")
70        util.download_source(path=zip_path, url=URLS["segs"], download=download, checksum=CHECKSUMS["segs"])
71        util.unzip(zip_path=zip_path, dst=seg_dir, remove=False)
72
73    for name in ["seg_metadata", "img_metadata"]:
74        csv_path = os.path.join(path, f"{name}.csv")
75        util.download_source(path=csv_path, url=URLS[name], download=download, checksum=CHECKSUMS[name])
76
77    return path

Download the segmentation masks and metadata for the IMA++ dataset.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • download: Whether to download the data if it is not present.
Returns:

Filepath where the data is downloaded.

def get_ima_plus_plus_paths( path: Union[os.PathLike, str], annotator: Optional[str] = None, download: bool = False) -> Tuple[List[str], List[str]]:
134def get_ima_plus_plus_paths(
135    path: Union[os.PathLike, str], annotator: Optional[str] = None, download: bool = False
136) -> Tuple[List[str], List[str]]:
137    """Get paths to the IMA++ data.
138
139    Args:
140        path: Filepath to a folder where the data is downloaded for further processing.
141        annotator: The choice of annotator whose masks are used. By default (`None`), a single
142            mask per image is returned: the 'ST' (STAPLE) consensus mask for the 2,394 images
143            that have multiple annotations, and the unique single-annotator mask otherwise. Pass
144            `'all'` to get every mask (multiple entries per multi-annotator image), or a specific
145            annotator id (e.g. `'A04'`, `'MV'`, `'ST'`) to filter to masks from that source only.
146        download: Whether to download the data if it is not present.
147
148    Returns:
149        List of filepaths for the image data.
150        List of filepaths for the label data.
151    """
152    data_dir = get_ima_plus_plus_data(path=path, download=download)
153
154    rows = _select_rows(_read_seg_metadata(data_dir), annotator)
155
156    isic_ids = sorted(set(row["ISIC_id"] for row in rows))
157    image_dir = os.path.join(data_dir, "images")
158    _download_images(isic_ids, image_dir, download)
159
160    image_paths = [os.path.join(image_dir, f"{row['ISIC_id']}.jpg") for row in rows]
161    gt_paths = [os.path.join(data_dir, "segs", row["seg_filename"]) for row in rows]
162
163    assert len(image_paths) == len(gt_paths) and len(image_paths) > 0
164    for gt_path in gt_paths:
165        assert os.path.exists(gt_path), gt_path
166
167    return image_paths, gt_paths

Get paths to the IMA++ data.

Arguments:
  • path: Filepath to a folder where the data is downloaded for further processing.
  • annotator: The choice of annotator whose masks are used. By default (None), a single mask per image is returned: the 'ST' (STAPLE) consensus mask for the 2,394 images that have multiple annotations, and the unique single-annotator mask otherwise. Pass 'all' to get every mask (multiple entries per multi-annotator image), or a specific annotator id (e.g. 'A04', 'MV', 'ST') to filter to masks from that source only.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data. List of filepaths for the label data.

def get_ima_plus_plus_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], annotator: Optional[str] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
170def get_ima_plus_plus_dataset(
171    path: Union[os.PathLike, str],
172    patch_shape: Tuple[int, int],
173    annotator: Optional[str] = None,
174    resize_inputs: bool = False,
175    download: bool = False,
176    **kwargs
177) -> Dataset:
178    """Get the IMA++ dataset for multi-annotator skin lesion segmentation in dermoscopy images.
179
180    Args:
181        path: Filepath to a folder where the downloaded data will be saved.
182        patch_shape: The patch shape to use for training.
183        annotator: The choice of annotator whose masks are used. See `get_ima_plus_plus_paths`.
184        resize_inputs: Whether to resize the inputs to the expected patch shape.
185        download: Whether to download the data if it is not present.
186        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
187
188    Returns:
189        The segmentation dataset.
190    """
191    image_paths, gt_paths = get_ima_plus_plus_paths(path=path, annotator=annotator, download=download)
192
193    if resize_inputs:
194        resize_kwargs = {"patch_shape": patch_shape, "is_rgb": True}
195        kwargs, patch_shape = util.update_kwargs_for_resize_trafo(
196            kwargs=kwargs,
197            patch_shape=patch_shape,
198            resize_inputs=resize_inputs,
199            resize_kwargs=resize_kwargs,
200            ensure_rgb=to_rgb,
201        )
202
203    return torch_em.default_segmentation_dataset(
204        raw_paths=image_paths,
205        raw_key=None,
206        label_paths=gt_paths,
207        label_key=None,
208        patch_shape=patch_shape,
209        is_seg_dataset=False,
210        **kwargs
211    )

Get the IMA++ dataset for multi-annotator skin lesion segmentation in dermoscopy images.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • patch_shape: The patch shape to use for training.
  • annotator: The choice of annotator whose masks are used. See get_ima_plus_plus_paths.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_ima_plus_plus_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], annotator: Optional[str] = None, resize_inputs: bool = False, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
214def get_ima_plus_plus_loader(
215    path: Union[os.PathLike, str],
216    batch_size: int,
217    patch_shape: Tuple[int, int],
218    annotator: Optional[str] = None,
219    resize_inputs: bool = False,
220    download: bool = False,
221    **kwargs
222) -> DataLoader:
223    """Get the IMA++ dataloader for multi-annotator skin lesion segmentation in dermoscopy images.
224
225    Args:
226        path: Filepath to a folder where the downloaded data will be saved.
227        batch_size: The batch size for training.
228        patch_shape: The patch shape to use for training.
229        annotator: The choice of annotator whose masks are used. See `get_ima_plus_plus_paths`.
230        resize_inputs: Whether to resize the inputs to the expected patch shape.
231        download: Whether to download the data if it is not present.
232        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
233
234    Returns:
235        The DataLoader.
236    """
237    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
238    dataset = get_ima_plus_plus_dataset(path, patch_shape, annotator, resize_inputs, download, **ds_kwargs)
239    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the IMA++ dataloader for multi-annotator skin lesion segmentation in dermoscopy images.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • annotator: The choice of annotator whose masks are used. See get_ima_plus_plus_paths.
  • resize_inputs: Whether to resize the inputs to the expected patch shape.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.