datasets
Training and evaluation data, with the modality, task and licence stated up front. Listed live from the Hugging Face Hub.
lm-pragmatics
Citation
@misc{hu2023finegrainedcomparisonpragmaticlanguage, title={A fine-grained comparison of pragmatic language understanding in humans and language models}, author={Jennifer Hu and Sammy Floyd and Olessia Jouravlev and Evelina Fedorenko and Edward Gibson}, year={2023}, eprint={2212.06801}, archivePrefix={arXiv}, primaryClass={cs.CL}, url={https://arxiv.org/abs/2212.06801},}
glue_diagnostics
Citation
@inproceedings{wang2019glue, title={{GLUE}: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding}, author={Wang, Alex and Singh, Amanpreet and Michael, Julian and Hill, Felix and Levy, Omer and Bowman, Samuel R.}, note={In the Proceedings of ICLR.}, year={2019}}
