Team Ai
Datasetpublic

arka0821/multi_document_summarization

Multi-Document, a large-scale multi-document summarization dataset created from scientific articles. Multi-Document introduces a challenging multi-document summarization task: writing the related-work section of a paper based on its abstract and the articles it references.

sourceHugging Faceunknownupdated 4y agoView on Hugging Face
5likes108downloads
multi_document_summarization.py102 linesDownload Raw Back to root
1# coding=utf-82# Copyright 2020 The TensorFlow Datasets Authors and the HuggingFace Datasets Authors.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8#     http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15 16# Lint as: python317"""Multi-Document Dataset."""18 19 20import json21 22import datasets23from datasets import set_caching_enabled24 25 26set_caching_enabled(False)27 28_CITATION = """29@article{lu2020multi,30  title={Multi-Document: A Large-scale Dataset for Extreme Multi-document Summarization of Scientific Articles},31  author={Arka Das, India},32  journal={arXiv preprint arXiv:2010.14235},33  year={2022}34}35"""36 37_DESCRIPTION = """38Multi-Document, a large-scale multi-document summarization dataset created from scientific articles. Multi-Document introduces a challenging multi-document summarization task: writing the related-work section of a paper based on its abstract and the articles it references.39"""40 41_URL_TRAIN = "https://github.com/arka0821/multi_document_summarization/raw/master/data/train.json.gz"42_URL_TEST = "https://github.com/arka0821/multi_document_summarization/raw/master/data/test.json.gz"43_URL_VAL = "https://github.com/arka0821/multi_document_summarization/raw/master/data/val.json.gz"44 45 46class MultiDocumentSum(datasets.GeneratorBasedBuilder):47    """ "Multi-Document Dataset."""48 49    VERSION = datasets.Version("1.1.0")50    def _info(selif):51        return datasets.DatasetInfo(52            description=_DESCRIPTION,53            features=datasets.Features(54                {55                    "id": datasets.Value("string"),56                    "docs": datasets.Sequence(57                        {58                            "id": datasets.Value("string"),59                            "text": datasets.Value("string")60                        },61                    ),62                    "summary": datasets.Value("string"),63                }64            ),65            supervised_keys=None,66            homepage="https://github.com/arka0821/multi_document_summarization",67            citation=_CITATION,68        )69 70    def _split_generators(self, dl_manager):71        """Returns SplitGenerators."""72        train_path = dl_manager.download_and_extract(_URL_TRAIN)73        test_path = dl_manager.download_and_extract(_URL_TEST)74        val_path = dl_manager.download_and_extract(_URL_VAL)75 76        return [77            datasets.SplitGenerator(78                name=datasets.Split.TRAIN,79                gen_kwargs={"path": train_path},80            ),81            datasets.SplitGenerator(82                name=datasets.Split.TEST,83                gen_kwargs={"path": test_path},84            ),85            datasets.SplitGenerator(86                name=datasets.Split.VALIDATION,87                gen_kwargs={"path": val_path},88            ),89        ]90    def _generate_examples(self, path=None):91        """Yields examples."""92        with open(path, encoding="utf-8") as f:93            data = json.load(f)94            f.close()95        for idx, el in enumerate(data):            96            ids = [id["id"] for id in el["docs"]]97            texts = [text["text"] for text in el["docs"]]98            tmp = {"id": ids, "text": texts}99            d = el.copy()100            d["docs"] = tmp101            yield idx, d102