jglaser/binding_affinity
A dataset to fine-tune language models on protein-ligand binding affinity prediction.
211.8k
1# coding=utf-82# Copyright 2020 The HuggingFace Datasets Authors and the current dataset script contributor.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8# http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15"""TODO: A dataset of protein sequences, ligand SMILES and binding affinities."""16 17import huggingface_hub18import os19import pyarrow.parquet as pq20import datasets21 22 23# TODO: Add BibTeX citation24# Find for instance the citation on arxiv or on the dataset repo/website25_CITATION = """\26@InProceedings{huggingface:dataset,27title = {jglaser/binding_affinity},28author={Jens Glaser, ORNL29},30year={2021}31}32"""33 34# TODO: Add description of the dataset here35# You can copy an official description36_DESCRIPTION = """\37A dataset to fine-tune language models on protein-ligand binding affinity prediction.38"""39 40# TODO: Add a link to an official homepage for the dataset here41_HOMEPAGE = ""42 43# TODO: Add the licence for the dataset here if you can find it44_LICENSE = "BSD two-clause"45 46# TODO: Add link to the official dataset URLs here47# The HuggingFace dataset library don't host the datasets but only point to the original files48# This can be an arbitrary nested dict/list of URLs (see below in `_split_generators` method)49_URL = "https://huggingface.co/datasets/jglaser/binding_affinity/resolve/main/"50_data_dir = "data/"51_file_names = {'default': _data_dir+'all.parquet',52 'no_kras': _data_dir+'all_nokras.parquet',53 'cov': _data_dir+'cov.parquet'}54 55_URLs = {name: _URL+_file_names[name] for name in _file_names}56 57 58# TODO: Name of the dataset usually match the script name with CamelCase instead of snake_case59class BindingAffinity(datasets.ArrowBasedBuilder):60 """List of protein sequences, ligand SMILES and binding affinities."""61 62 VERSION = datasets.Version("1.4.1")63 64 def _info(self):65 # TODO: This method specifies the datasets.DatasetInfo object which contains informations and typings for the dataset66 #if self.config.name == "first_domain": # This is the name of the configuration selected in BUILDER_CONFIGS above67 # features = datasets.Features(68 # {69 # "sentence": datasets.Value("string"),70 # "option1": datasets.Value("string"),71 # "answer": datasets.Value("string")72 # # These are the features of your dataset like images, labels ...73 # }74 # )75 #else: # This is an example to show how to have different features for "first_domain" and "second_domain"76 features = datasets.Features(77 {78 "seq": datasets.Value("string"),79 "smiles": datasets.Value("string"),80 "affinity_uM": datasets.Value("float"),81 "neg_log10_affinity_M": datasets.Value("float"),82 "smiles_can": datasets.Value("string"),83 "affinity": datasets.Value("float"),84 # These are the features of your dataset like images, labels ...85 }86 )87 return datasets.DatasetInfo(88 # This is the description that will appear on the datasets page.89 description=_DESCRIPTION,90 # This defines the different columns of the dataset and their types91 features=features, # Here we define them above because they are different between the two configurations92 # If there's a common (input, target) tuple from the features,93 # specify them here. They'll be used if as_supervised=True in94 # builder.as_dataset.95 supervised_keys=None,96 # Homepage of the dataset for documentation97 homepage=_HOMEPAGE,98 # License for the dataset if available99 license=_LICENSE,100 # Citation for the dataset101 citation=_CITATION,102 )103 104 def _split_generators(self, dl_manager):105 """Returns SplitGenerators."""106 # TODO: This method is tasked with downloading/extracting the data and defining the splits depending on the configuration107 # If several configurations are possible (listed in BUILDER_CONFIGS), the configuration selected by the user is in self.config.name108 109 # dl_manager is a datasets.download.DownloadManager that can be used to download and extract URLs110 # It can accept any type or nested list/dict and will give back the same structure with the url replaced with path to local files.111 # By default the archives will be extracted and a path to a cached folder where they are extracted is returned instead of the archive112 files = dl_manager.download_and_extract(_URLs)113 114 return [115 datasets.SplitGenerator(116 # These kwargs will be passed to _generate_examples117 name=datasets.Split.TRAIN,118 gen_kwargs={119 'filepath': files["default"],120 },121 ),122 123 datasets.SplitGenerator(124 name='no_kras',125 # These kwargs will be passed to _generate_examples126 gen_kwargs={127 "filepath": files["no_kras"],128 },129 ),130 131 datasets.SplitGenerator(132 name='covalent',133 # These kwargs will be passed to _generate_examples134 gen_kwargs={135 "filepath": files["cov"],136 },137 ),138 ]139 140 def _generate_tables(141 self, filepath142 ):143 from pyarrow import fs144 local = fs.LocalFileSystem()145 146 for i, f in enumerate([filepath]):147 yield i, pq.read_table(f,filesystem=local)148 