torch_em.data.datasets.light_microscopy.bbbc010

The BBBC010 dataset contains brightfield and GFP images of the roundworm C. elegans from a live/dead assay screen for novel anti-infectives. Animals were exposed to the pathogen Enterococcus faecalis and either treated with ampicillin ("live" phenotype) or left untreated ("dead" phenotype).

The dataset contains 100 wells from a 384-well plate, each imaged in two channels (w1: brightfield, w2: GFP), with instance segmentation ground truth for individual worms.

The dataset is located at https://bbbc.broadinstitute.org/BBBC010. This dataset is CC0 (public domain). If you use it, please cite it as: "We used the C. elegans infection live/dead image set version 1 provided by Fred Ausubel and available from the Broad Bioimage Benchmark Collection [Ljosa et al., Nature Methods, 2012]." https://doi.org/10.1038/nmeth.2083

  1"""The BBBC010 dataset contains brightfield and GFP images of the roundworm
  2C. elegans from a live/dead assay screen for novel anti-infectives. Animals were
  3exposed to the pathogen Enterococcus faecalis and either treated with ampicillin
  4("live" phenotype) or left untreated ("dead" phenotype).
  5
  6The dataset contains 100 wells from a 384-well plate, each imaged in two channels
  7(w1: brightfield, w2: GFP), with instance segmentation ground truth for individual
  8worms.
  9
 10The dataset is located at https://bbbc.broadinstitute.org/BBBC010.
 11This dataset is CC0 (public domain). If you use it, please cite it as:
 12"We used the C. elegans infection live/dead image set version 1 provided by Fred
 13Ausubel and available from the Broad Bioimage Benchmark Collection [Ljosa et al.,
 14Nature Methods, 2012]." https://doi.org/10.1038/nmeth.2083
 15"""
 16
 17import os
 18import re
 19from glob import glob
 20from natsort import natsorted
 21from typing import List, Optional, Tuple, Union
 22
 23import numpy as np
 24import imageio.v3 as imageio
 25from tqdm import tqdm
 26from sklearn.model_selection import train_test_split
 27
 28from torch.utils.data import Dataset, DataLoader
 29
 30import torch_em
 31
 32from .. import util
 33
 34
 35IMAGE_URL = "https://data.broadinstitute.org/bbbc/BBBC010/BBBC010_v2_images.zip"
 36IMAGE_CHECKSUM = None
 37
 38GT_URL = "https://data.broadinstitute.org/bbbc/BBBC010/BBBC010_v1_foreground_eachworm.zip"
 39GT_CHECKSUM = None
 40
 41WELL_PATTERN = re.compile(r"_([A-E]\d{2})_w(\d)_")
 42
 43
 44def _get_well_and_channel(fname: str) -> Tuple[Optional[str], Optional[int]]:
 45    """Extract the well id (e.g. 'A01') and channel number (1 or 2) from a raw image filename."""
 46    match = WELL_PATTERN.search(fname)
 47    if match is None:
 48        return None, None
 49    return match.group(1), int(match.group(2))
 50
 51
 52def _merge_worm_masks(mask_paths: List[str]) -> np.ndarray:
 53    """Merge per-worm binary masks into a single instance segmentation label image."""
 54    mask_paths = natsorted(mask_paths)
 55    ref = imageio.imread(mask_paths[0])
 56    instances = np.zeros(ref.shape, dtype=np.int32)
 57    for i, mask_path in enumerate(mask_paths, start=1):
 58        mask = imageio.imread(mask_path) > 0
 59        instances[mask] = i
 60    return instances
 61
 62
 63def _preprocess(data_dir: str, channel: int) -> str:
 64    """Convert raw TIFs and per-worm ground truth PNGs to preprocessed H5 files."""
 65    import h5py
 66
 67    h5_dir = os.path.join(data_dir, f"h5_data_w{channel}")
 68    if os.path.exists(h5_dir):
 69        return h5_dir
 70    os.makedirs(h5_dir, exist_ok=True)
 71
 72    raw_paths = glob(os.path.join(data_dir, "images", "*.tif"))
 73    well_to_raw = {}
 74    for raw_path in raw_paths:
 75        well, this_channel = _get_well_and_channel(os.path.basename(raw_path))
 76        if well is None or this_channel != channel:
 77            continue
 78        well_to_raw[well] = raw_path
 79
 80    gt_dir = os.path.join(data_dir, "BBBC010_v1_foreground_eachworm")
 81    for well, raw_path in tqdm(sorted(well_to_raw.items()), desc="Preprocessing BBBC010"):
 82        mask_paths = glob(os.path.join(gt_dir, f"{well}_*_ground_truth.png"))
 83        if len(mask_paths) == 0:
 84            continue
 85
 86        raw = imageio.imread(raw_path)
 87        instances = _merge_worm_masks(mask_paths)
 88
 89        h5_path = os.path.join(h5_dir, f"{well}.h5")
 90        with h5py.File(h5_path, "w") as f:
 91            f.create_dataset("raw", data=raw, compression="gzip")
 92            f.create_dataset("labels", data=instances, compression="gzip")
 93
 94    return h5_dir
 95
 96
 97def get_bbbc010_data(path: Union[os.PathLike, str], channel: int = 1, download: bool = False) -> str:
 98    """Download and preprocess the BBBC010 dataset.
 99
100    Args:
101        path: Filepath to a folder where the downloaded data will be saved.
102        channel: The imaging channel to use as raw input. Default: 1 (brightfield).
103            Available channels: 1=brightfield, 2=GFP.
104        download: Whether to download the data if it is not present.
105
106    Returns:
107        The filepath to the preprocessed H5 data directory.
108    """
109    data_dir = os.path.join(path, "BBBC010")
110
111    if not os.path.exists(data_dir):
112        os.makedirs(data_dir, exist_ok=True)
113        img_zip = os.path.join(path, "BBBC010_v2_images.zip")
114        gt_zip = os.path.join(path, "BBBC010_v1_foreground_eachworm.zip")
115        util.download_source(img_zip, IMAGE_URL, download, checksum=IMAGE_CHECKSUM)
116        util.download_source(gt_zip, GT_URL, download, checksum=GT_CHECKSUM)
117        util.unzip(img_zip, os.path.join(data_dir, "images"))
118        util.unzip(gt_zip, data_dir)
119
120    return _preprocess(data_dir, channel)
121
122
123def get_bbbc010_paths(
124    path: Union[os.PathLike, str],
125    split: Optional[str] = None,
126    channel: int = 1,
127    download: bool = False,
128) -> Tuple[List[str], List[str]]:
129    """Get paths to the BBBC010 data.
130
131    Args:
132        path: Filepath to a folder where the downloaded data will be saved.
133        split: The data split to use. One of 'train', 'val', 'test', or None (use all).
134        channel: The imaging channel to use as raw input. Default: 1 (brightfield).
135            Available channels: 1=brightfield, 2=GFP.
136        download: Whether to download the data if it is not present.
137
138    Returns:
139        List of filepaths for the image data (H5, key 'raw').
140        List of filepaths for the label data (H5, key 'labels').
141    """
142    h5_dir = get_bbbc010_data(path, channel, download)
143    h5_paths = natsorted(glob(os.path.join(h5_dir, "*.h5")))
144
145    if len(h5_paths) == 0:
146        raise RuntimeError(f"No preprocessed files found in {h5_dir}.")
147
148    if split is None:
149        return h5_paths, h5_paths
150
151    train_paths, test_paths = train_test_split(h5_paths, test_size=0.2, random_state=42)
152    train_paths, val_paths = train_test_split(train_paths, test_size=0.15, random_state=42)
153
154    split_map = {"train": train_paths, "val": val_paths, "test": test_paths}
155    assert split in split_map, f"'{split}' is not a valid split. Choose from {list(split_map)}."
156    selected = split_map[split]
157    return selected, selected
158
159
160def get_bbbc010_dataset(
161    path: Union[os.PathLike, str],
162    patch_shape: Tuple[int, int],
163    split: Optional[str] = None,
164    channel: int = 1,
165    download: bool = False,
166    **kwargs,
167) -> Dataset:
168    """Get the BBBC010 dataset for C. elegans instance segmentation.
169
170    Args:
171        path: Filepath to a folder where the downloaded data will be saved.
172        patch_shape: The patch shape to use for training.
173        split: The data split to use. One of 'train', 'val', 'test', or None (use all).
174        channel: The imaging channel to use as raw input. Default: 1 (brightfield).
175            Available channels: 1=brightfield, 2=GFP.
176        download: Whether to download the data if it is not present.
177        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
178
179    Returns:
180        The segmentation dataset.
181    """
182    raw_paths, label_paths = get_bbbc010_paths(path, split, channel, download)
183
184    return torch_em.default_segmentation_dataset(
185        raw_paths=raw_paths,
186        raw_key="raw",
187        label_paths=label_paths,
188        label_key="labels",
189        patch_shape=patch_shape,
190        **kwargs,
191    )
192
193
194def get_bbbc010_loader(
195    path: Union[os.PathLike, str],
196    batch_size: int,
197    patch_shape: Tuple[int, int],
198    split: Optional[str] = None,
199    channel: int = 1,
200    download: bool = False,
201    **kwargs,
202) -> DataLoader:
203    """Get the BBBC010 dataloader for C. elegans instance segmentation.
204
205    Args:
206        path: Filepath to a folder where the downloaded data will be saved.
207        batch_size: The batch size for training.
208        patch_shape: The patch shape to use for training.
209        split: The data split to use. One of 'train', 'val', 'test', or None (use all).
210        channel: The imaging channel to use as raw input. Default: 1 (brightfield).
211            Available channels: 1=brightfield, 2=GFP.
212        download: Whether to download the data if it is not present.
213        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
214
215    Returns:
216        The DataLoader.
217    """
218    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
219    dataset = get_bbbc010_dataset(path, patch_shape, split, channel, download, **ds_kwargs)
220    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)
IMAGE_URL = 'https://data.broadinstitute.org/bbbc/BBBC010/BBBC010_v2_images.zip'
IMAGE_CHECKSUM = None
GT_URL = 'https://data.broadinstitute.org/bbbc/BBBC010/BBBC010_v1_foreground_eachworm.zip'
GT_CHECKSUM = None
WELL_PATTERN = re.compile('_([A-E]\\d{2})_w(\\d)_')
def get_bbbc010_data( path: Union[os.PathLike, str], channel: int = 1, download: bool = False) -> str:
 98def get_bbbc010_data(path: Union[os.PathLike, str], channel: int = 1, download: bool = False) -> str:
 99    """Download and preprocess the BBBC010 dataset.
100
101    Args:
102        path: Filepath to a folder where the downloaded data will be saved.
103        channel: The imaging channel to use as raw input. Default: 1 (brightfield).
104            Available channels: 1=brightfield, 2=GFP.
105        download: Whether to download the data if it is not present.
106
107    Returns:
108        The filepath to the preprocessed H5 data directory.
109    """
110    data_dir = os.path.join(path, "BBBC010")
111
112    if not os.path.exists(data_dir):
113        os.makedirs(data_dir, exist_ok=True)
114        img_zip = os.path.join(path, "BBBC010_v2_images.zip")
115        gt_zip = os.path.join(path, "BBBC010_v1_foreground_eachworm.zip")
116        util.download_source(img_zip, IMAGE_URL, download, checksum=IMAGE_CHECKSUM)
117        util.download_source(gt_zip, GT_URL, download, checksum=GT_CHECKSUM)
118        util.unzip(img_zip, os.path.join(data_dir, "images"))
119        util.unzip(gt_zip, data_dir)
120
121    return _preprocess(data_dir, channel)

Download and preprocess the BBBC010 dataset.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • channel: The imaging channel to use as raw input. Default: 1 (brightfield). Available channels: 1=brightfield, 2=GFP.
  • download: Whether to download the data if it is not present.
Returns:

The filepath to the preprocessed H5 data directory.

def get_bbbc010_paths( path: Union[os.PathLike, str], split: Optional[str] = None, channel: int = 1, download: bool = False) -> Tuple[List[str], List[str]]:
124def get_bbbc010_paths(
125    path: Union[os.PathLike, str],
126    split: Optional[str] = None,
127    channel: int = 1,
128    download: bool = False,
129) -> Tuple[List[str], List[str]]:
130    """Get paths to the BBBC010 data.
131
132    Args:
133        path: Filepath to a folder where the downloaded data will be saved.
134        split: The data split to use. One of 'train', 'val', 'test', or None (use all).
135        channel: The imaging channel to use as raw input. Default: 1 (brightfield).
136            Available channels: 1=brightfield, 2=GFP.
137        download: Whether to download the data if it is not present.
138
139    Returns:
140        List of filepaths for the image data (H5, key 'raw').
141        List of filepaths for the label data (H5, key 'labels').
142    """
143    h5_dir = get_bbbc010_data(path, channel, download)
144    h5_paths = natsorted(glob(os.path.join(h5_dir, "*.h5")))
145
146    if len(h5_paths) == 0:
147        raise RuntimeError(f"No preprocessed files found in {h5_dir}.")
148
149    if split is None:
150        return h5_paths, h5_paths
151
152    train_paths, test_paths = train_test_split(h5_paths, test_size=0.2, random_state=42)
153    train_paths, val_paths = train_test_split(train_paths, test_size=0.15, random_state=42)
154
155    split_map = {"train": train_paths, "val": val_paths, "test": test_paths}
156    assert split in split_map, f"'{split}' is not a valid split. Choose from {list(split_map)}."
157    selected = split_map[split]
158    return selected, selected

Get paths to the BBBC010 data.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • split: The data split to use. One of 'train', 'val', 'test', or None (use all).
  • channel: The imaging channel to use as raw input. Default: 1 (brightfield). Available channels: 1=brightfield, 2=GFP.
  • download: Whether to download the data if it is not present.
Returns:

List of filepaths for the image data (H5, key 'raw'). List of filepaths for the label data (H5, key 'labels').

def get_bbbc010_dataset( path: Union[os.PathLike, str], patch_shape: Tuple[int, int], split: Optional[str] = None, channel: int = 1, download: bool = False, **kwargs) -> torch.utils.data.dataset.Dataset:
161def get_bbbc010_dataset(
162    path: Union[os.PathLike, str],
163    patch_shape: Tuple[int, int],
164    split: Optional[str] = None,
165    channel: int = 1,
166    download: bool = False,
167    **kwargs,
168) -> Dataset:
169    """Get the BBBC010 dataset for C. elegans instance segmentation.
170
171    Args:
172        path: Filepath to a folder where the downloaded data will be saved.
173        patch_shape: The patch shape to use for training.
174        split: The data split to use. One of 'train', 'val', 'test', or None (use all).
175        channel: The imaging channel to use as raw input. Default: 1 (brightfield).
176            Available channels: 1=brightfield, 2=GFP.
177        download: Whether to download the data if it is not present.
178        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset`.
179
180    Returns:
181        The segmentation dataset.
182    """
183    raw_paths, label_paths = get_bbbc010_paths(path, split, channel, download)
184
185    return torch_em.default_segmentation_dataset(
186        raw_paths=raw_paths,
187        raw_key="raw",
188        label_paths=label_paths,
189        label_key="labels",
190        patch_shape=patch_shape,
191        **kwargs,
192    )

Get the BBBC010 dataset for C. elegans instance segmentation.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • patch_shape: The patch shape to use for training.
  • split: The data split to use. One of 'train', 'val', 'test', or None (use all).
  • channel: The imaging channel to use as raw input. Default: 1 (brightfield). Available channels: 1=brightfield, 2=GFP.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset.
Returns:

The segmentation dataset.

def get_bbbc010_loader( path: Union[os.PathLike, str], batch_size: int, patch_shape: Tuple[int, int], split: Optional[str] = None, channel: int = 1, download: bool = False, **kwargs) -> torch.utils.data.dataloader.DataLoader:
195def get_bbbc010_loader(
196    path: Union[os.PathLike, str],
197    batch_size: int,
198    patch_shape: Tuple[int, int],
199    split: Optional[str] = None,
200    channel: int = 1,
201    download: bool = False,
202    **kwargs,
203) -> DataLoader:
204    """Get the BBBC010 dataloader for C. elegans instance segmentation.
205
206    Args:
207        path: Filepath to a folder where the downloaded data will be saved.
208        batch_size: The batch size for training.
209        patch_shape: The patch shape to use for training.
210        split: The data split to use. One of 'train', 'val', 'test', or None (use all).
211        channel: The imaging channel to use as raw input. Default: 1 (brightfield).
212            Available channels: 1=brightfield, 2=GFP.
213        download: Whether to download the data if it is not present.
214        kwargs: Additional keyword arguments for `torch_em.default_segmentation_dataset` or for the PyTorch DataLoader.
215
216    Returns:
217        The DataLoader.
218    """
219    ds_kwargs, loader_kwargs = util.split_kwargs(torch_em.default_segmentation_dataset, **kwargs)
220    dataset = get_bbbc010_dataset(path, patch_shape, split, channel, download, **ds_kwargs)
221    return torch_em.get_data_loader(dataset, batch_size, **loader_kwargs)

Get the BBBC010 dataloader for C. elegans instance segmentation.

Arguments:
  • path: Filepath to a folder where the downloaded data will be saved.
  • batch_size: The batch size for training.
  • patch_shape: The patch shape to use for training.
  • split: The data split to use. One of 'train', 'val', 'test', or None (use all).
  • channel: The imaging channel to use as raw input. Default: 1 (brightfield). Available channels: 1=brightfield, 2=GFP.
  • download: Whether to download the data if it is not present.
  • kwargs: Additional keyword arguments for torch_em.default_segmentation_dataset or for the PyTorch DataLoader.
Returns:

The DataLoader.