asahi417/multi-domain-document-classification
multi_domain_document_classification Multi-domain document classification datasets. Biomedical: chemprot, rct-sample Computer Science: citation_intent, sciie Customer Review: amcd, yelp_review Social Media: tweet_eval_irony, tweet_eval_hate, tweet_eval_emotion The yelp_review dataset is randomly downsampled to 2000/2000/8000 for test/validation/train. chemprot citation_intent hyperpartisan_news rct_sample sciie amcd yelp_review tweet_eval_irony tweet_eval_hate… See the full description on the dataset page: https://huggingface.co/datasets/asahi417/multi-domain-document-classification.
0230
1"""Multi domain document classification dataset used in [https://arxiv.org/pdf/2004.10964.pdf](https://arxiv.org/pdf/2004.10964.pdf)"""2import json3from itertools import chain4import datasets5 6logger = datasets.logging.get_logger(__name__)7_DESCRIPTION = """Multi domain document classification dataset used in [https://arxiv.org/pdf/2004.10964.pdf](https://arxiv.org/pdf/2004.10964.pdf)"""8_NAME = "multi_domain_document_classification"9_VERSION = "0.2.3"10_CITATION = """11@inproceedings{dontstoppretraining2020,12 author = {Suchin Gururangan and Ana Marasović and Swabha Swayamdipta and Kyle Lo and Iz Beltagy and Doug Downey and Noah A. Smith},13 title = {Don't Stop Pretraining: Adapt Language Models to Domains and Tasks},14 year = {2020},15 booktitle = {Proceedings of ACL},16}17"""18 19_HOME_PAGE = "https://github.com/asahi417/m3"20_URL = f'https://huggingface.co/datasets/asahi417/{_NAME}/raw/main/dataset'21_DATA_TYPE = ["chemprot", "citation_intent", "hyperpartisan_news", "rct_sample", "sciie", "amcd",22 "yelp_review", "tweet_eval_irony", "tweet_eval_hate", "tweet_eval_emotion"]23_URLS = {24 k:25 {26 str(datasets.Split.TEST): [f'{_URL}/{k}/test.jsonl'],27 str(datasets.Split.TRAIN): [f'{_URL}/{k}/train.jsonl'],28 str(datasets.Split.VALIDATION): [f'{_URL}/{k}/dev.jsonl']29 }30 for k in _DATA_TYPE31}32_LABELS = {33 "chemprot": {"ACTIVATOR": 0, "AGONIST": 1, "AGONIST-ACTIVATOR": 2, "AGONIST-INHIBITOR": 3, "ANTAGONIST": 4, "DOWNREGULATOR": 5, "INDIRECT-DOWNREGULATOR": 6, "INDIRECT-UPREGULATOR": 7, "INHIBITOR": 8, "PRODUCT-OF": 9, "SUBSTRATE": 10, "SUBSTRATE_PRODUCT-OF": 11, "UPREGULATOR": 12},34 "citation_intent": {"Background": 0, "CompareOrContrast": 1, "Extends": 2, "Future": 3, "Motivation": 4, "Uses": 5},35 "hyperpartisan_news": {"false": 0, "true": 1},36 "rct_sample": {"BACKGROUND": 0, "CONCLUSIONS": 1, "METHODS": 2, "OBJECTIVE": 3, "RESULTS": 4},37 "sciie": {"COMPARE": 0, "CONJUNCTION": 1, "EVALUATE-FOR": 2, "FEATURE-OF": 3, "HYPONYM-OF": 4, "PART-OF": 5, "USED-FOR": 6},38 "amcd": {"false": 0, "true": 1},39 "yelp_review": {"5 star": 4, "4 star": 3, "3 star": 2, "2 star": 1, "1 star": 0},40 "tweet_eval_irony": {"non_irony":0, "irony": 1},41 "tweet_eval_hate": {"non_hate": 0, "hate": 1},42 "tweet_eval_emotion": {"anger": 0, "joy": 1, "optimism": 2, "sadness": 3}43}44 45 46class MultiDomainDocumentClassificationConfig(datasets.BuilderConfig):47 """BuilderConfig"""48 49 def __init__(self, **kwargs):50 """BuilderConfig.51 52 Args:53 **kwargs: keyword arguments forwarded to super.54 """55 super(MultiDomainDocumentClassificationConfig, self).__init__(**kwargs)56 57 58class MultiDomainDocumentClassification(datasets.GeneratorBasedBuilder):59 """Dataset."""60 61 BUILDER_CONFIGS = [62 MultiDomainDocumentClassificationConfig(63 name=k, version=datasets.Version(_VERSION), description=_DESCRIPTION64 ) for k in _DATA_TYPE65 ]66 67 def _split_generators(self, dl_manager):68 downloaded_file = dl_manager.download_and_extract(_URLS[self.config.name])69 return [70 datasets.SplitGenerator(name=i, gen_kwargs={"filepaths": downloaded_file[str(i)]})71 for i in [datasets.Split.TRAIN, datasets.Split.VALIDATION, datasets.Split.TEST]72 ]73 74 def _generate_examples(self, filepaths):75 _key = 076 for filepath in filepaths:77 logger.info(f"generating examples from = {filepath}")78 with open(filepath, encoding="utf-8") as f:79 _list = [i for i in f.read().split('\n') if len(i) > 0]80 for i in _list:81 data = json.loads(i)82 yield _key, data83 _key += 184 85 def _info(self):86 label2id = sorted(_LABELS[self.config.name].items(), key=lambda x: x[1])87 label = [i[0] for i in label2id]88 return datasets.DatasetInfo(89 description=_DESCRIPTION,90 features=datasets.Features(91 {92 "text": datasets.Value("string"),93 "label": datasets.features.ClassLabel(names=label),94 }95 ),96 supervised_keys=None,97 homepage=_HOME_PAGE,98 citation=_CITATION,99 )100 