datasets
Training and evaluation data, with the modality, task and licence stated up front. Listed live from the Hugging Face Hub.
acceptability-prediction@inproceedings{lau-etal-2015-unsupervised,
title = "Unsupervised Prediction of Acceptability Judgements",
author = "Lau, Jey Han and
Clark, Alexander and
Lappin, Shalom",
booktitle = "Proceedings of the 53rd Annual Meeting of the Association for Computational Linguistics and the 7th International Joint Conference on Natural Language Processing (Volume 1: Long Papers)",
month = jul,
year = "2015",
address = "Beijing, China",
publisher = "Association for… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/acceptability-prediction.jigsaw_toxicitysocial-chemestry-101blog_authorship_corpussimlexnli-veridicality-transitivity@inproceedings{yanaka-etal-2021-exploring,
title = "Exploring Transitivity in Neural {NLI} Models through Veridicality",
author = "Yanaka, Hitomi and
Mineshima, Koji and
Inui, Kentaro",
booktitle = "Proceedings of the 16th Conference of the European Chapter of the Association for Computational Linguistics: Main Volume",
year = "2021",
pages = "920--934",
}
winowhyhttps://github.com/HKUST-KnowComp/WinoWhy
@inproceedings{zhang2020WinoWhy,
author = {Hongming Zhang and Xinran Zhao and Yangqiu Song},
title = {WinoWhy: A Deep Diagnosis of Essential Commonsense Knowledge for Answering Winograd Schema Challenge},
booktitle = {Proceedings of Annual Meeting of the Association for Computational Linguistics (ACL) 2020},
year = {2020}
}
offensive-humor@article{tang2022naughtyformer,
title={The Naughtyformer: A Transformer Understands Offensive Humor},
author={Tang, Leonard and Cai, Alexander and Li, Steve and Wang, Jason},
journal={arXiv preprint arXiv:2211.14369},
year={2022}
}
paradehttps://github.com/heyunh2015/PARADE_dataset
@inproceedings{he-etal-2020-parade,
title = "{PARADE}: {A} {N}ew {D}ataset for {P}araphrase {I}dentification {R}equiring {C}omputer {S}cience {D}omain {K}nowledge",
author = "He, Yun and
Wang, Zhuoer and
Zhang, Yin and
Huang, Ruihong and
Caverlee, James",
booktitle = "Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP)",
month = nov,
year = "2020",
address… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/parade.english-gradinghttps://www.kaggle.com/competitions/feedback-prize-english-language-learning
monotonicity-entailment@inproceedings{yanaka-etal-2019-neural,
title = "Can Neural Networks Understand Monotonicity Reasoning?",
author = "Yanaka, Hitomi and
Mineshima, Koji and
Bekki, Daisuke and
Inui, Kentaro and
Sekine, Satoshi and
Abzianidze, Lasha and
Bos, Johan",
booktitle = "Proceedings of the 2019 ACL Workshop BlackboxNLP: Analyzing and Interpreting Neural Networks for NLP",
year = "2019",
pages = "31--40",
}
sts-companionhttps://ixa2.si.ehu.eus/stswiki/index.php/STSbenchmark
The companion datasets to the STS Benchmark comprise the rest of the English datasets used in the STS tasks organized by us in the context of SemEval between 2012 and 2017.
Authors collated two datasets, one with pairs of sentences related to machine translation evaluation. Another one with the rest of datasets, which can be used for domain adaptation studies.
@inproceedings{cer-etal-2017-semeval,
title = "{S}em{E}val-2017 Task 1:… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/sts-companion.wouldyouratherrankme-nlg-acceptability@inproceedings{novikova-etal-2018-rankme,
title = "RankME: Reliable Human Ratings for Natural Language Generation",
author = "Novikova, Jekaterina and
Duvsek, Ondvrej and
Rieser, Verena",
booktitle = "Proceedings of the NAACL2018",
month = jun,
year = "2018",
address = "New Orleans, Louisiana",
publisher = "Association for Computational Linguistics",
url = "https://aclanthology.org/N18-2012",
doi = "10.18653/v1/N18-2012",
pages = "72--78"… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/rankme-nlg-acceptability.context_toxicityhttps://github.com/ipavlopoulos/context_toxicity/
@inproceedings{xenos-etal-2021-context,
title = "Context Sensitivity Estimation in Toxicity Detection",
author = "Xenos, Alexandros and
Pavlopoulos, John and
Androutsopoulos, Ion",
booktitle = "Proceedings of the 5th Workshop on Online Abuse and Harms (WOAH 2021)",
month = aug,
year = "2021",
address = "Online",
publisher = "Association for Computational Linguistics",
url =… See the full description on the dataset page: https://huggingface.co/datasets/tasksource/context_toxicity.semantic-feature-production-normskaggle-claim-typesimpeval
