datasets
Training and evaluation data, with the modality, task and licence stated up front. Listed live from the Hugging Face Hub.
arct
The Argument Reasoning Comprehension Task: Identification and Reconstruction of Implicit Warrants
https://github.com/UKPLab/argument-reasoning-comprehension-task
@InProceedings{Habernal.et.al.2018.NAACL.ARCT,
title = {The Argument Reasoning Comprehension Task: Identification
and Reconstruction of Implicit Warrants},
author = {Habernal, Ivan and Wachsmuth, Henning and
Gurevych, Iryna and Stein, Benno},
publisher = {Association for… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/arct.acceptability-prediction@inproceedings{lau-etal-2015-unsupervised,
title = "Unsupervised Prediction of Acceptability Judgements",
author = "Lau, Jey Han and
Clark, Alexander and
Lappin, Shalom",
booktitle = "Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)",
month = jul,
year = "2015",
address = "Beijing, China",
publisher = "Association for… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/acceptability-prediction.jigsaw_toxicitysocial-chemestry-101blog_authorship_corpustraciehttps://github.com/allenai/aristo-leaderboard/tree/master/tracie/data
@inproceedings{ZRNKSR21,
author = {Ben Zhou and Kyle Richardson and Qiang Ning and Tushar Khot and Ashish Sabharwal and Dan Roth},
title = {Temporal Reasoning on Implicit Events from Distant Supervision},
booktitle = {NAACL},
year = {2021},
}
simlexcounterfactually-augmented-snli@article{kaushik2020learning,
title={Learning the Difference that Makes a Difference with Counterfactually Augmented Data},
author={Kaushik, Divyansh and Hovy, Eduard and Lipton, Zachary C},
journal={International Conference on Learning Representations (ICLR)},
year={2020}
}
nli-veridicality-transitivity@inproceedings{yanaka-etal-2021-exploring,
title = "Exploring Transitivity in Neural {NLI} Models through Veridicality",
author = "Yanaka, Hitomi and
Mineshima, Koji and
Inui, Kentaro",
booktitle = "Proceedings of the 16th Conference of the European Chapter of the Association for Computational Linguistics: Main Volume",
year = "2021",
pages = "920--934",
}
implicit-hate-stg1https://github.com/SALT-NLP/implicit-hate
@inproceedings{elsherief-etal-2021-latent,
title = "Latent Hatred: A Benchmark for Understanding Implicit Hate Speech",
author = "ElSherief, Mai and
Ziems, Caleb and
Muchlinski, David and
Anupindi, Vaishnavi and
Seybolt, Jordyn and
De Choudhury, Munmun and
Yang, Diyi",
booktitle = "Proceedings of the 2021 Conference on Empirical Methods in Natural Language Processing",
month = nov,
year =… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/implicit-hate-stg1.winowhyhttps://github.com/HKUST-KnowComp/WinoWhy
@inproceedings{zhang2020WinoWhy,
author = {Hongming Zhang and Xinran Zhao and Yangqiu Song},
title = {WinoWhy: A Deep Diagnosis of Essential Commonsense Knowledge for Answering Winograd Schema Challenge},
booktitle = {Proceedings of Annual Meeting of the Association for Computational Linguistics (ACL) 2020},
year = {2020}
}
counterfactually-augmented-imdb@article{kaushik2020learning,
title={Learning the Difference that Makes a Difference with Counterfactually Augmented Data},
author={Kaushik, Divyansh and Hovy, Eduard and Lipton, Zachary C},
journal={International Conference on Learning Representations (ICLR)},
year={2020}
}
I2D2code:
https://i2d2.allen.ai/
https://arxiv.org/abs/2212.09246
@inproceedings{Bhagavatula2022GenGen,
title={Generating Generics: Knowledge Induction with NeuroLogic and Self-Imitation},
author={Chandra Bhagavatula, Jena D. Hwang, Doug Downey, Ronan Le Bras, Ximing Lu, Lianhui Qin, Keisuke Sakaguchi, Swabha Swayamdipta, Peter West, Yejin Choi},
booktitle={arXiv},
year={2022}
}
help-nlihttps://github.com/verypluming/HELP
@InProceedings{yanaka-EtAl:2019:starsem,
author = {Yanaka, Hitomi and Mineshima, Koji and Bekki, Daisuke and Inui, Kentaro and Sekine, Satoshi and Abzianidze, Lasha and Bos, Johan},
title = {HELP: A Dataset for Identifying Shortcomings of Neural Models in Monotonicity Reasoning},
booktitle = {Proceedings of the Eighth Joint Conference on Lexical and Computational Semantics (*SEM2019)},
year = {2019},
}
AES2-essay-scoringhttps://www.kaggle.com/competitions/learning-agency-lab-automated-essay-scoring-2/data
offensive-humor@article{tang2022naughtyformer,
title={The Naughtyformer: A Transformer Understands Offensive Humor},
author={Tang, Leonard and Cai, Alexander and Li, Steve and Wang, Jason},
journal={arXiv preprint arXiv:2211.14369},
year={2022}
}
tomi-nlitomi dataset (theory of mind question answering) recasted as natural language inference
https://colab.research.google.com/drive/1J_RqDSw9iPxJSBvCJu-VRbjXnrEjKVvr?usp=sharing
@article{sileo2023tasksource,
title={tasksource: Structured Dataset Preprocessing Annotations for Frictionless Extreme Multi-Task Learning and Evaluation},
author={Sileo, Damien},
url= {https://arxiv.org/abs/2301.05948},
journal={arXiv preprint arXiv:2301.05948},
year={2023}
}… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/tomi-nli.implicaturesImplicature corpus
@article{george2020conversational,
title={Conversational implicatures in English dialogue: Annotated dataset},
author={George, Elizabeth Jasmi and Mamidi, Radhika},
journal={Procedia Computer Science},
volume={171},
pages={2316--2323},
year={2020},
publisher={Elsevier}
}
Augmented with generated distractors https://colab.research.google.com/drive/1ix0FgwzPAjQkIQA2E3ctlylvcmya7vGy?usp=sharing, for tasksource
@article{sileo2023tasksource,
title={tasksource:… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/implicatures.paradehttps://github.com/heyunh2015/PARADE_dataset
@inproceedings{he-etal-2020-parade,
title = "{PARADE}: {A} {N}ew {D}ataset for {P}araphrase {I}dentification {R}equiring {C}omputer {S}cience {D}omain {K}nowledge",
author = "He, Yun and
Wang, Zhuoer and
Zhang, Yin and
Huang, Ruihong and
Caverlee, James",
booktitle = "Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)",
month = nov,
year = "2020",
address… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/parade.temporal-nli@inproceedings{thukral-etal-2021-probing,
title = "Probing Language Models for Understanding of Temporal Expressions",
author = "Thukral, Shivin and
Kukreja, Kunal and
Kavouras, Christian",
booktitle = "Proceedings of the Fourth BlackboxNLP Workshop on Analyzing and Interpreting Neural Networks for NLP",
month = nov,
year = "2021",
address = "Punta Cana, Dominican Republic",
publisher = "Association for Computational Linguistics",
url =… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/temporal-nli.clcd-english@article{salvatore2019logical,
title={A logical-based corpus for cross-lingual evaluation},
author={Salvatore, Felipe and Finger, Marcelo and Hirata Jr, Roberto},
journal={arXiv preprint arXiv:1905.05704},
year={2019}
}
english-gradinghttps://www.kaggle.com/competitions/feedback-prize-english-language-learning
monotonicity-entailment@inproceedings{yanaka-etal-2019-neural,
title = "Can Neural Networks Understand Monotonicity Reasoning?",
author = "Yanaka, Hitomi and
Mineshima, Koji and
Bekki, Daisuke and
Inui, Kentaro and
Sekine, Satoshi and
Abzianidze, Lasha and
Bos, Johan",
booktitle = "Proceedings of the 2019 ACL Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP",
year = "2019",
pages = "31--40",
}
subjectivity@misc{antici2023corpus,
title={A Corpus for Sentence-level Subjectivity Detection on English News Articles},
author={Francesco Antici and Andrea Galassi and Federico Ruggeri and Katerina Korre and Arianna Muti and Alessandra Bardi and Alice Fedotova and Alberto Barrón-Cedeño},
year={2023},
eprint={2305.18034},
archivePrefix={arXiv},
primaryClass={cs.CL}
}
datasheet:… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/subjectivity.syntactic-augmentation-nlihttps://github.com/Aatlantise/syntactic-augmentation-nli/tree/master/datasets
@inproceedings{min-etal-2020-syntactic,
title = "Syntactic Data Augmentation Increases Robustness to Inference Heuristics",
author = "Min, Junghyun and
McCoy, R. Thomas and
Das, Dipanjan and
Pitler, Emily and
Linzen, Tal",
booktitle = "Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics",
month = jul,
year = "2020",
address =… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/syntactic-augmentation-nli.sts-companionhttps://ixa2.si.ehu.eus/stswiki/index.php/STSbenchmark
The companion datasets to the STS Benchmark comprise the rest of the English datasets used in the STS tasks organized by us in the context of SemEval between 2012 and 2017.
Authors collated two datasets, one with pairs of sentences related to machine translation evaluation. Another one with the rest of datasets, which can be used for domain adaptation studies.
@inproceedings{cer-etal-2017-semeval,
title = "{S}em{E}val-2017 Task 1:… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/sts-companion.avicenna@article{aghahadi2022avicenna,
title={Avicenna: a challenge dataset for natural language generation toward commonsense syllogistic reasoning},
author={Aghahadi, Zeinab and Talebpour, Alireza},
journal={Journal of Applied Non-Classical Logics},
pages={1--17},
year={2022},
publisher={Taylor \& Francis}
}
wouldyouratherfigurative-nli @inproceedings{chakrabarty-etal-2021-figurative,
title = "Figurative Language in Recognizing Textual Entailment",
author = "Chakrabarty, Tuhin and
Ghosh, Debanjan and
Poliak, Adam and
Muresan, Smaranda",
booktitle = "Findings of the Association for Computational Linguistics: ACL-IJCNLP 2021",
month = aug,
year = "2021",
address = "Online",
publisher =… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/figurative-nli.apthttps://github.com/Advancing-Machine-Human-Reasoning-Lab/apt
@inproceedings{nighojkar-licato-2021-improving,
title = "Improving Paraphrase Detection with the Adversarial Paraphrasing Task",
author = "Nighojkar, Animesh and
Licato, John",
booktitle = "Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)",
month = aug,
year = "2021"… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/apt.
