datasets
Training and evaluation data, with the modality, task and licence stated up front. Listed live from the Hugging Face Hub.
lm-pragmatics
Citation
@misc{hu2023finegrainedcomparisonpragmaticlanguage, title={A fine-grained comparison of pragmatic language understanding in humans and language models}, author={Jennifer Hu and Sammy Floyd and Olessia Jouravlev and Evelina Fedorenko and Edward Gibson}, year={2023}, eprint={2212.06801}, archivePrefix={arXiv}, primaryClass={cs.CL}, url={https://arxiv.org/abs/2212.06801},}
glue_diagnostics
Citation
@inproceedings{wang2019glue, title={{GLUE}: A Multi-Task Benchmark and Analysis Platform for Natural Language Understanding}, author={Wang, Alex and Singh, Amanpreet and Michael, Julian and Hill, Felix and Levy, Omer and Bowman, Samuel R.}, note={In the Proceedings of ICLR.}, year={2019}}
warm-up_synthetic-dataSFT-Final-DatasetcladderTaken from https://huggingface.co/datasets/causal-nlp/CLadder (the original was broken as of and not usable within the lm_eval_harness)
Citation
@inproceedings{jin2023cladder, author = {Zhijing Jin and Yuen Chen and Felix Leeb and Luigi Gresele and Ojasv Kamal and Zhiheng Lyu and Kevin Blin and Fernando Gonzalez and Max Kleiman-Weiner and Mrinmaya Sachan and Bernhard Sch{"{o}}lkopf}, title = "{CL}adder: {A}ssessing Causal Reasoning in Language Models", year = "2023"… See the full description on the dataset page: https://huggingface.co/datasets/clembench-playpen/cladder.
