google-research-datasets/taskmaster2
Taskmaster is dataset for goal oriented conversations. The Taskmaster-2 dataset consists of 17,289 dialogs in the seven domains which include restaurants, food ordering, movies, hotels, flights, music and sports. Unlike Taskmaster-1, which includes both written "self-dialogs" and spoken two-person dialogs, Taskmaster-2 consists entirely of spoken two-person dialogs. In addition, while Taskmaster-1 is almost exclusively task-based, Taskmaster-2 contains a good number of search- and recommendation-oriented dialogs. All dialogs in this release were created using a Wizard of Oz (WOz) methodology in which crowdsourced workers played the role of a 'user' and trained call center operators played the role of the 'assistant'. In this way, users were led to believe they were interacting with an automated system that “spoke” using text-to-speech (TTS) even though it was in fact a human behind the scenes. As a result, users could express themselves however they chose in the context of an automated interface.
1# coding=utf-82# Copyright 2020 The HuggingFace Datasets Authors and the current dataset script contributor.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8# http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15"""Taskmaster: A dataset for goal oriented conversations."""16 17 18import json19 20import datasets21 22 23_CITATION = """\24@inproceedings{48484,25title = {Taskmaster-1: Toward a Realistic and Diverse Dialog Dataset},26author = {Bill Byrne and Karthik Krishnamoorthi and Chinnadhurai Sankar and Arvind Neelakantan and Daniel Duckworth and Semih Yavuz and Ben Goodrich and Amit Dubey and Kyu-Young Kim and Andy Cedilnik},27year = {2019}28}29"""30 31_DESCRIPTION = """\32Taskmaster is dataset for goal oriented conversations. The Taskmaster-2 dataset consists of 17,289 dialogs \33in the seven domains which include restaurants, food ordering, movies, hotels, flights, music and sports. \34Unlike Taskmaster-1, which includes both written "self-dialogs" and spoken two-person dialogs, \35Taskmaster-2 consists entirely of spoken two-person dialogs. In addition, while Taskmaster-1 is \36almost exclusively task-based, Taskmaster-2 contains a good number of search- and recommendation-oriented dialogs. \37All dialogs in this release were created using a Wizard of Oz (WOz) methodology in which crowdsourced \38workers played the role of a 'user' and trained call center operators played the role of the 'assistant'. \39In this way, users were led to believe they were interacting with an automated system that “spoke” \40using text-to-speech (TTS) even though it was in fact a human behind the scenes. \41As a result, users could express themselves however they chose in the context of an automated interface.42"""43 44_HOMEPAGE = "https://github.com/google-research-datasets/Taskmaster/tree/master/TM-2-2020"45 46_BASE_URL = "https://raw.githubusercontent.com/google-research-datasets/Taskmaster/master/TM-2-2020/data"47 48 49class Taskmaster2(datasets.GeneratorBasedBuilder):50 """Taskmaster: A dataset for goal oriented conversations."""51 52 VERSION = datasets.Version("1.0.0")53 BUILDER_CONFIGS = [54 datasets.BuilderConfig(55 name="flights", version=datasets.Version("1.0.0"), description="Taskmaster-2 flights domain."56 ),57 datasets.BuilderConfig(58 name="food-ordering", version=datasets.Version("1.0.0"), description="Taskmaster-2 food-ordering domain"59 ),60 datasets.BuilderConfig(61 name="hotels", version=datasets.Version("1.0.0"), description="Taskmaster-2 hotel domain"62 ),63 datasets.BuilderConfig(64 name="movies", version=datasets.Version("1.0.0"), description="Taskmaster-2 movies domain"65 ),66 datasets.BuilderConfig(67 name="music", version=datasets.Version("1.0.0"), description="Taskmaster-2 music domain"68 ),69 datasets.BuilderConfig(70 name="restaurant-search",71 version=datasets.Version("1.0.0"),72 description="Taskmaster-2 restaurant-search domain",73 ),74 datasets.BuilderConfig(75 name="sports", version=datasets.Version("1.0.0"), description="Taskmaster-2 sports domain"76 ),77 ]78 79 def _info(self):80 features = {81 "conversation_id": datasets.Value("string"),82 "instruction_id": datasets.Value("string"),83 "utterances": [84 {85 "index": datasets.Value("int32"),86 "speaker": datasets.Value("string"),87 "text": datasets.Value("string"),88 "segments": [89 {90 "start_index": datasets.Value("int32"),91 "end_index": datasets.Value("int32"),92 "text": datasets.Value("string"),93 "annotations": [{"name": datasets.Value("string")}],94 }95 ],96 }97 ],98 }99 return datasets.DatasetInfo(100 description=_DESCRIPTION,101 features=datasets.Features(features),102 supervised_keys=None,103 homepage=_HOMEPAGE,104 citation=_CITATION,105 )106 107 def _split_generators(self, dl_manager):108 url = f"{_BASE_URL}/{self.config.name}.json"109 dialogs_file = dl_manager.download(url)110 return [111 datasets.SplitGenerator(112 name=datasets.Split.TRAIN,113 gen_kwargs={"filepath": dialogs_file},114 ),115 ]116 117 def _generate_examples(self, filepath):118 key = 0119 with open(filepath, encoding="utf-8") as f:120 dialogs = json.load(f)121 for dialog in dialogs:122 utterances = dialog["utterances"]123 for utterance in utterances:124 if "segments" not in utterance:125 utterance["segments"] = []126 yield key, dialog127 key += 1128 