angelolab/ark_example
This dataset contains 11 Field of Views (FOVs), each with 22 channels.
02.8k
1# Copyright 2020 The HuggingFace Datasets Authors and the current dataset script contributor.2#3# Licensed under the Apache License, Version 2.0 (the "License");4# you may not use this file except in compliance with the License.5# You may obtain a copy of the License at6#7# http://www.apache.org/licenses/LICENSE-2.08#9# Unless required by applicable law or agreed to in writing, software10# distributed under the License is distributed on an "AS IS" BASIS,11# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.12# See the License for the specific language governing permissions and13# limitations under the License.14 15"""16This dataset contains example data for running through the multiplexed imaging data pipeline in17Ark Analysis: https://github.com/angelolab/ark-analysis.18 19 20Dataset Fov renaming:21 22TMA2_R8C3 -> fov023TMA6_R4C5 -> fov124TMA7_R5C4 -> fov225TMA10_R7C3 -> fov326TMA11_R9C6 -> fov427TMA13_R8C5 -> fov528TMA17_R9C2 -> fov629TMA18_R9C2 -> fov730TMA21_R2C5 -> fov831TMA21_R12C6 -> fov932TMA24_R9C1 -> fov1033 34"""35 36import datasets37import pathlib38 39# Find for instance the citation on arxiv or on the dataset repo/website40_CITATION = """\41@InProceedings{huggingface:dataset,42title = {Ark Analysis Example Dataset},43author={Angelo Lab},44year={2022}45}46"""47 48# TODO: Add description of the dataset here49# You can copy an official description50_DESCRIPTION = """\51This dataset contains 11 Field of Views (FOVs), each with 22 channels. 52"""53 54_HOMEPAGE = "https://github.com/angelolab/ark-analysis"55 56_LICENSE = "https://github.com/angelolab/ark-analysis/blob/main/LICENSE"57 58# The HuggingFace Datasets library doesn't host the datasets but only points to the original files.59# This can be an arbitrary nested dict/list of URLs (see below in `_split_generators` method)60 61_URL_DATA = {62 "image_data": "data/image_data.zip",63 "cell_table": "data/segmentation/cell_table.zip",64 "deepcell_output": "data/segmentation/deepcell_output.zip",65 "example_pixel_output_dir": "data/pixie/example_pixel_output_dir.zip",66 "example_cell_output_dir": "data/pixie/example_cell_output_dir.zip",67 "spatial_lda": "data/spatial_analysis/spatial_lda.zip",68 "post_clustering": "data/post_clustering.zip",69 "ome_tiff": "data/ome_tiff.zip",70 "ez_seg_data": "data/ez_seg_data.zip"71}72 73_URL_DATASET_CONFIGS = {74 "segment_image_data": {"image_data": _URL_DATA["image_data"]},75 "cluster_pixels": {76 "image_data": _URL_DATA["image_data"],77 "cell_table": _URL_DATA["cell_table"],78 "deepcell_output": _URL_DATA["deepcell_output"],79 },80 "cluster_cells": {81 "image_data": _URL_DATA["image_data"],82 "cell_table": _URL_DATA["cell_table"],83 "deepcell_output": _URL_DATA["deepcell_output"],84 "example_pixel_output_dir": _URL_DATA["example_pixel_output_dir"],85 },86 "post_clustering": {87 "image_data": _URL_DATA["image_data"],88 "cell_table": _URL_DATA["cell_table"],89 "deepcell_output": _URL_DATA["deepcell_output"],90 "example_cell_output_dir": _URL_DATA["example_cell_output_dir"],91 },92 "fiber_segmentation": {93 "image_data": _URL_DATA["image_data"],94 },95 "LDA_preprocessing": {96 "image_data": _URL_DATA["image_data"],97 "cell_table": _URL_DATA["cell_table"],98 },99 "LDA_training_inference": {100 "image_data": _URL_DATA["image_data"],101 "cell_table": _URL_DATA["cell_table"],102 "spatial_lda": _URL_DATA["spatial_lda"],103 },104 "neighborhood_analysis": {105 "image_data": _URL_DATA["image_data"],106 "cell_table": _URL_DATA["cell_table"],107 "deepcell_output": _URL_DATA["deepcell_output"],108 },109 "pairwise_spatial_enrichment": {110 "image_data": _URL_DATA["image_data"],111 "cell_table": _URL_DATA["cell_table"],112 "deepcell_output": _URL_DATA["deepcell_output"],113 "post_clustering": _URL_DATA["post_clustering"],114 },115 "ome_tiff": {116 "ome_tiff": _URL_DATA["ome_tiff"],117 },118 "ez_seg_data": {119 "ez_seg_data": _URL_DATA["ez_seg_data"]120 }121}122 123 124# Note: Name of the dataset usually match the script name with CamelCase instead of snake_case125class ArkExample(datasets.GeneratorBasedBuilder):126 """The Dataset consists of 11 FOVs"""127 128 VERSION = datasets.Version("0.0.5")129 130 # You will be able to load one or the other configurations in the following list with131 BUILDER_CONFIGS = [132 datasets.BuilderConfig(133 name="segment_image_data",134 version=VERSION,135 description="This configuration contains data used by notebook 1 - Segment Image Data.",136 ),137 datasets.BuilderConfig(138 name="cluster_pixels",139 version=VERSION,140 description="This configuration contains data used by notebook 2 - Pixel Clustering (Pixie Pipeline #1).",141 ),142 datasets.BuilderConfig(143 name="cluster_cells",144 version=VERSION,145 description="This configuration contains data used by notebook 3 - Cell Clustering (Pixie Pipeline #2).",146 ),147 datasets.BuilderConfig(148 name="post_clustering",149 version=VERSION,150 description="This configuration contains data used by notebook 4 - Post Clustering.",151 ),152 datasets.BuilderConfig(153 name="fiber_segmentation",154 version=VERSION,155 description="This configuration contains data used by the Fiber Segmentation Notebook.",156 ),157 datasets.BuilderConfig(158 name="LDA_preprocessing",159 version=VERSION,160 description="This configuration contains data used by the Spatial LDA - Preprocessing Notebook."161 ),162 datasets.BuilderConfig(163 name="LDA_training_inference",164 version=VERSION,165 description="This configuration contains data used by the Spatial LDA - Training and Inference Notebook."166 ),167 datasets.BuilderConfig(168 name="neighborhood_analysis",169 version=VERSION,170 description="This configuration contains data used by the Neighborhood Analysis Notebook."171 ),172 datasets.BuilderConfig(173 name="pairwise_spatial_enrichment",174 version=VERSION,175 description="This configuration contains data used by the Pairwise Spatial Enrichment Notebook."176 ),177 datasets.BuilderConfig(178 name="ome_tiff",179 version=VERSION,180 description="This configuration contains an OME-TIFF format of FOV1. Intended to be used with the OME-TIFF Conversion Notebook."181 ),182 datasets.BuilderConfig(183 name="ez_seg_data",184 version=VERSION,185 description="This configuration contains the data used by the ezSegmenter notebook."186 )187 ]188 189 def _info(self):190 # This is the name of the configuration selected in BUILDER_CONFIGS above191 if self.config.name in list(_URL_DATASET_CONFIGS.keys()):192 features = datasets.Features(193 {f: datasets.Value("string") for f in _URL_DATASET_CONFIGS[self.config.name].keys()}194 )195 else:196 ValueError(f"Dataset name is incorrect, options include {list(_URL_DATASET_CONFIGS.keys())}")197 return datasets.DatasetInfo(198 # This is the description that will appear on the datasets page.199 description=_DESCRIPTION,200 # This defines the different columns of the dataset and their types201 features=features, # Here we define them above because they are different between the two configurations202 # If there's a common (input, target) tuple from the features, uncomment supervised_keys line below and203 # specify them. They'll be used if as_supervised=True in builder.as_dataset.204 # supervised_keys=("sentence", "label"),205 # Homepage of the dataset for documentation206 homepage=_HOMEPAGE,207 # License for the dataset if available208 license=_LICENSE,209 # Citation for the dataset210 citation=_CITATION,211 )212 213 def _split_generators(self, dl_manager):214 # This method is tasked with downloading/extracting the data and defining the splits depending on the configuration215 urls = _URL_DATASET_CONFIGS[self.config.name]216 data_dirs = {}217 for data_name, url in urls.items():218 dl_path = pathlib.Path(dl_manager.download_and_extract(url))219 data_dirs[data_name] = dl_path220 221 return [222 datasets.SplitGenerator(223 name=self.config.name,224 # These kwargs will be passed to _generate_examples225 gen_kwargs={"dataset_paths": data_dirs},226 ),227 ]228 229 # method parameters are unpacked from `gen_kwargs` as given in `_split_generators`230 def _generate_examples(self, dataset_paths):231 yield self.config.name, dataset_paths232 