datasets
Training and evaluation data, with the modality, task and licence stated up front. Listed live from the Hugging Face Hub.
reclorhttps://whyu.me/reclor/
@inproceedings{yu2020reclor,
author = {Yu, Weihao and Jiang, Zihang and Dong, Yanfei and Feng, Jiashi},
title = {ReClor: A Reading Comprehension Dataset Requiring Logical Reasoning},
booktitle = {International Conference on Learning Representations (ICLR)},
month = {April},
year = {2020}
}
strategy-qafoliohttps://github.com/Yale-LILY/FOLIO
@article{han2022folio,
title={FOLIO: Natural Language Reasoning with First-Order Logic},
author = {Han, Simeng and Schoelkopf, Hailey and Zhao, Yilun and Qi, Zhenting and Riddell, Martin and Benson, Luke and Sun, Lucy and Zubova, Ekaterina and Qiao, Yujie and Burtell, Matthew and Peng, David and Fan, Jonathan and Liu, Yixin and Wong, Brian and Sailor, Malcolm and Ni, Ansong and Nan, Linyong and Kasai, Jungo and Yu, Tao and Zhang, Rui and Joty, Shafiq and… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/folio.logiqa-2.0-nlihttps://github.com/csitfun/LogiQA2.0
Temporary citation:
@article{liu2020logiqa,
title={Logiqa: A challenge dataset for machine reading comprehension with logical reasoning},
author={Liu, Jian and Cui, Leyang and Liu, Hanmeng and Huang, Dandan and Wang, Yile and Zhang, Yue},
journal={arXiv preprint arXiv:2007.08124},
year={2020}
}
commonsense_qa_2.0https://github.com/allenai/csqa2
@article{talmor2022commonsenseqa,
title={CommonsenseQA 2.0: Exposing the limits of AI through gamification},
author={Talmor, Alon and Yoran, Ori and Bras, Ronan Le and Bhagavatula, Chandra and Goldberg, Yoav and Choi, Yejin and Berant, Jonathan},
journal={arXiv preprint arXiv:2201.05320},
year={2022}
}
scruplesrace-cRace-C : additional data for race (high school/middle school) but for college level
https://github.com/mrcdata/race-c
@InProceedings{pmlr-v101-liang19a,
title={A New Multi-choice Reading Comprehension Dataset for Curriculum Learning},
author={Liang, Yichan and Li, Jianheng and Yin, Jian},
booktitle={Proceedings of The Eleventh Asian Conference on Machine Learning},
pages={742--757},
year={2019}
}
mega-acceptability-v2ConTRoL-nlihttps://github.com/csitfun/ConTRoL-dataset
@article{Liu_Cui_Liu_Zhang_2021,
title={Natural Language Inference in Context - Investigating Contextual Reasoning over Long Texts},
volume={35},
url={https://ojs.aaai.org/index.php/AAAI/article/view/17580},
DOI={10.1609/aaai.v35i15.17580},
number={15},
journal={Proceedings of the AAAI Conference on Artificial Intelligence},
author={Liu, Hanmeng and Cui, Leyang and Liu, Jian and Zhang, Yue},
year={2021},
month={May},
pages={13388-13396}
}
com2sensehttps://github.com/PlusLabNLP/Com2Sense
@inproceedings{singh-etal-2021-com2sense,
title = "{COM}2{SENSE}: A Commonsense Reasoning Benchmark with Complementary Sentences",
author = "Singh, Shikhar and
Wen, Nuan and
Hou, Yu and
Alipoormolabashi, Pegah and
Wu, Te-lin and
Ma, Xuezhe and
Peng, Nanyun",
booktitle = "Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021",
month = aug,
year = "2021",
address =… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/com2sense.wicehttps://github.com/ryokamoi/wice
@inproceedings{kamoi-etal-2023-wice,
title = "{W}i{CE}: Real-World Entailment for Claims in {W}ikipedia",
author = "Kamoi, Ryo and
Goyal, Tanya and
Rodriguez, Juan and
Durrett, Greg",
editor = "Bouamor, Houda and
Pino, Juan and
Bali, Kalika",
booktitle = "Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing",
month = dec,
year = "2023",
address = "Singapore"… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/wice.spartqa-mchoicehttps://github.com/HLR/SpartQA-baselines
@inproceedings{mirzaee-etal-2021-spartqa,
title = "{SPARTQA}: A Textual Question Answering Benchmark for Spatial Reasoning",
author = "Mirzaee, Roshanak and
Rajaby Faghihi, Hossein and
Ning, Qiang and
Kordjamshidi, Parisa",
booktitle = "Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies",
month = jun… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/spartqa-mchoice.scinli#SciNLI: A Corpus for Natural Language Inference on Scientific Text
https://github.com/msadat3/SciNLI
@inproceedings{sadat-caragea-2022-scinli,
title = "{S}ci{NLI}: A Corpus for Natural Language Inference on Scientific Text",
author = "Sadat, Mobashir and
Caragea, Cornelia",
booktitle = "Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers)",
month = may,
year = "2022",
address = "Dublin, Ireland"… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/scinli.ambient@misc{liu-etal-2023-afraid,
title = "We're Afraid Language Models Aren't Modeling Ambiguity",
author = "Alisa Liu and Zhaofeng Wu and Julian Michael and Alane Suhr and Peter West and Alexander Koller and Swabha Swayamdipta and Noah A. Smith and Yejin Choi",
month = apr,
year = "2023",
url = "https://arxiv.org/abs/2304.14399",
}
puzztehttps://bitbucket.org/RoxanaSz/puzzte/src/master/
@article{szomiu2021puzzle,
title={A Puzzle-Based Dataset for Natural Language Inference},
author={Szomiu, Roxana and Groza, Adrian},
journal={arXiv preprint arXiv:2112.05742},
year={2021}
}
leandojohttps://github.com/lean-dojo/LeanDojo
@article{yang2023leandojo,
title={{LeanDojo}: Theorem Proving with Retrieval-Augmented Language Models},
author={Yang, Kaiyu and Swope, Aidan and Gu, Alex and Chalamala, Rahul and Song, Peiyang and Yu, Shixing and Godil, Saad and Prenger, Ryan and Anandkumar, Anima},
journal={arXiv preprint arXiv:2306.15626},
year={2023}
}
mutual@inproceedings{mutual,
title = "MuTual: A Dataset for Multi-Turn Dialogue Reasoning",
author = "Cui, Leyang and Wu, Yu and Liu, Shujie and Zhang, Yue and Zhou, Ming" ,
booktitle = "Proceedings of the 58th Conference of the Association for Computational Linguistics",
year = "2020",
publisher = "Association for Computational Linguistics",
}
boolq-natural-perturbationsBoolQ questions with semantic alteration and human verifications
@article{khashabi2020naturalperturbations,
title={Natural Perturbation for Robust Question Answering},
author={D. Khashabi and T. Khot and A. Sabhwaral},
journal={arXiv preprint},
year={2020}
}
QA-FeedbackCREPEhttps://github.com/velocityCavalry/CREPE
@inproceedings{fan2019eli5,
title = "{ELI}5: Long Form Question Answering",
author = "Fan, Angela and Jernite, Yacine and Perez, Ethan and Grangier, David and Weston, Jason and Auli, Michael",
booktitle = "Proceedings of the Annual Meeting of the Association for Computational Linguistics",
year = "2019",
}
spartqa-yn @inproceedings{mirzaee-etal-2021-spartqa,
title = "{SPARTQA}: A Textual Question Answering Benchmark for Spatial Reasoning",
author = "Mirzaee, Roshanak and
Rajaby Faghihi, Hossein and
Ning, Qiang and
Kordjamshidi, Parisa",
booktitle = "Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies",
month = jun,
year = "2021",
address = "Online"… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/spartqa-yn.fool-me-twicehttps://github.com/google-research/fool-me-twice
@inproceedings{eisenschlos-etal-2021-fool,
title = "Fool Me Twice: Entailment from {W}ikipedia Gamification",
author = {Eisenschlos, Julian Martin and
Dhingra, Bhuwan and
Bulian, Jannis and
B{\"o}rschinger, Benjamin and
Boyd-Graber, Jordan},
booktitle = "Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies",
month… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/fool-me-twice.monlihttps://github.com/atticusg/MoNLI
@inproceedings{geiger-etal-2020-neural,
address = {Online},
author = {Geiger, Atticus and Richardson, Kyle and Potts, Christopher},
booktitle = {Proceedings of the Third BlackboxNLP Workshop on Analyzing and Interpreting Neural Networks for NLP},
doi = {10.18653/v1/2020.blackboxnlp-1.16},
month = nov,
pages = {163--173},
publisher = {Association for Computational Linguistics},
title = {Neural Natural Language Inference Models… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/monli.equateEQUATE
EQUATE (Evaluating Quantitative Understanding Aptitude in Textual Entailment) is a new framework for evaluating quantitative reasoning ability in textual entailment. EQUATE consists of five NLI test sets featuring quantities. You can download EQUATE here. Three of these tests for quantitative reasoning feature language from real-world sources such as news articles and social media (RTE, NewsNLI Reddit), and two are controlled synthetic tests, evaluating model ability to reason with… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/equate.CLAIR_email_fraudhttps://aclweb.org/aclwiki/CLAIR_collection_of_fraud_email_(Repository)
@misc{radev2008clair,
author = {Dragomir Radev},
title = {CLAIR Collection of Fraud Email},
year = {2008},
note = {ACL Data and Code Repository, ADCR2008T001},
url = {http://aclweb.org/aclwiki}
}
twentyquestionssen-makinghttps://github.com/wangcunxiang/Sen-Making-and-Explanation
@inproceedings{wang-etal-2019-make,
title = "Does it Make Sense? And Why? A Pilot Study for Sense Making and Explanation",
author = "Wang, Cunxiang and
Liang, Shuailong and
Zhang, Yue and
Li, Xiaonan and
Gao, Tian",
booktitle = "Proceedings of the 57th Annual Meeting of the Association for Computational Linguistics",
month = jul,
year = "2019",
address = "Florence, Italy",
publisher =… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/sen-making.natural-language-satisfiability@misc{https://doi.org/10.48550/arxiv.2211.05417,
doi = {10.48550/ARXIV.2211.05417},
url = {https://arxiv.org/abs/2211.05417},
author = {Schlegel, Viktor and Pavlov, Kamen V. and Pratt-Hartmann, Ian},
keywords = {Computation and Language (cs.CL), Artificial Intelligence (cs.AI), FOS: Computer and information sciences, FOS: Computer and information sciences},
title = {Can Transformers Reason in Fragments of Natural Language?},
publisher = {arXiv},
year = {2022},
copyright =… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/natural-language-satisfiability.wiki-hadescnli
Generalization of Counterfactually-Augmented NLI Data
@inproceedings{huang2020cnligeneralization,
title={Counterfactually-Augmented {SNLI} Training Data Does Not Yield Better Generalization Than Unaugmented Data},
author={William Huang and Haokun Liu and Samuel R. Bowman},
booktitle = {Proceedings of the 2020 EMNLP Workshop on Insights from Negative Results in NLP},
year={2020},
publisher = {The Association for Computational Linguistics}
}
