arka0821/multi_document_summarization
Multi-Document, a large-scale multi-document summarization dataset created from scientific articles. Multi-Document introduces a challenging multi-document summarization task: writing the related-work section of a paper based on its abstract and the articles it references.
5108
1# coding=utf-82# Copyright 2020 The TensorFlow Datasets Authors and the HuggingFace Datasets Authors.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8# http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15 16# Lint as: python317"""Multi-Document Dataset."""18 19 20import json21 22import datasets23from datasets import set_caching_enabled24 25 26set_caching_enabled(False)27 28_CITATION = """29@article{lu2020multi,30 title={Multi-Document: A Large-scale Dataset for Extreme Multi-document Summarization of Scientific Articles},31 author={Arka Das, India},32 journal={arXiv preprint arXiv:2010.14235},33 year={2022}34}35"""36 37_DESCRIPTION = """38Multi-Document, a large-scale multi-document summarization dataset created from scientific articles. Multi-Document introduces a challenging multi-document summarization task: writing the related-work section of a paper based on its abstract and the articles it references.39"""40 41_URL_TRAIN = "https://github.com/arka0821/multi_document_summarization/raw/master/data/train.json.gz"42_URL_TEST = "https://github.com/arka0821/multi_document_summarization/raw/master/data/test.json.gz"43_URL_VAL = "https://github.com/arka0821/multi_document_summarization/raw/master/data/val.json.gz"44 45 46class MultiDocumentSum(datasets.GeneratorBasedBuilder):47 """ "Multi-Document Dataset."""48 49 VERSION = datasets.Version("1.1.0")50 def _info(selif):51 return datasets.DatasetInfo(52 description=_DESCRIPTION,53 features=datasets.Features(54 {55 "id": datasets.Value("string"),56 "docs": datasets.Sequence(57 {58 "id": datasets.Value("string"),59 "text": datasets.Value("string")60 },61 ),62 "summary": datasets.Value("string"),63 }64 ),65 supervised_keys=None,66 homepage="https://github.com/arka0821/multi_document_summarization",67 citation=_CITATION,68 )69 70 def _split_generators(self, dl_manager):71 """Returns SplitGenerators."""72 train_path = dl_manager.download_and_extract(_URL_TRAIN)73 test_path = dl_manager.download_and_extract(_URL_TEST)74 val_path = dl_manager.download_and_extract(_URL_VAL)75 76 return [77 datasets.SplitGenerator(78 name=datasets.Split.TRAIN,79 gen_kwargs={"path": train_path},80 ),81 datasets.SplitGenerator(82 name=datasets.Split.TEST,83 gen_kwargs={"path": test_path},84 ),85 datasets.SplitGenerator(86 name=datasets.Split.VALIDATION,87 gen_kwargs={"path": val_path},88 ),89 ]90 def _generate_examples(self, path=None):91 """Yields examples."""92 with open(path, encoding="utf-8") as f:93 data = json.load(f)94 f.close()95 for idx, el in enumerate(data): 96 ids = [id["id"] for id in el["docs"]]97 texts = [text["text"] for text in el["docs"]]98 tmp = {"id": ids, "text": texts}99 d = el.copy()100 d["docs"] = tmp101 yield idx, d102 