bigscience/xP3mt
xP3 (Crosslingual Public Pool of Prompts) is a collection of prompts & datasets across 46 of languages & 16 NLP tasks. It is used for the training of BLOOMZ and mT0, multilingual language models capable of following human instructions in dozens of languages zero-shot.
2619k
1from functools import partial2import json3import multiprocessing4import os5import random6 7from datasets import load_dataset8# pip install -q iso-6399from iso639 import languages10from promptsource.templates import DatasetTemplates11 12# Set to False to use multilingual prompts e.g. 'id' for xcopa/id instead of 'en'13USE_ENGLISH_PROMPTS = False14 15MAX_EXAMPLES_PER_DATASET_PROMPT = 100_00016 17STORY_CLOZE_DIR = "/gpfswork/rech/six/commun/code/tr13f-6B3-ml-t0/story_cloze_data"18XSTORY_CLOZE_DIR = "/gpfswork/rech/six/commun/code/tr13f-6B3-ml-t0/xstory_cloze_data"19 20# Some datasets have test sets with hidden labels which will still compile but only to noise21# e.g. piqa test labels are all [-1] which still works on list indices resulting in 22# noise samples where the label is always the same 23SKIP_PROMPTS = {24 "common_gen": {"test": ["all"]},25 "piqa": {"test": ["all"]},26 "qasc": {"test": ["all"]},27 "imdb": {"unsupervised": ["all"]},28 "glue/qqp": {"test": ["all"]},29 "qasc": {"test": ["all"]},30 "cosmos_qa": {"test": [31 "description_context_question_answer_text", 32 "description_context_question_text",33 "description_context_question_answer_id",34 "context_answer_to_question",35 "context_description_question_answer_text",36 "context_description_question_answer_id",37 "context_question_description_answer_id",38 "context_description_question_text",39 "context_question_description_answer_text",40 "only_question_answer",41 "no_prompt_id",42 "context_question_description_text",43 "no_prompt_text",44 ]},45 "clue/tnews": {"test": ["all"]},46 "clue/csl": {"test": ["all"]},47 "clue/cmrc2018": {"test": ["generate_question", "in_an_exam", "answer_in_the_passage", "answer_following_question", "xp3longcontinue"]},48 "clue/drcd": {"test": ["generate_question", "in_an_exam", "answer_in_the_passage", "answer_following_question", "xp3longcontinue"]},49 "hellaswag": {"test": ["complete_first_then", "Topic of the context", "Open-ended completion", "Randomized prompts template", "Appropriate continuation - Yes or No", "Predict ending with hint", "Open-ended start", "Reversed appropriate continuation - Yes or No", "how_ends", "if_begins_how_continues"]},50}51 52DS_TO_ENG_PROMPT = {53 "xcopa": "en",54 "Muennighoff/xstory_cloze": "en",55 "Muennighoff/xwinograd": "en",56 'GEM/wiki_lingua': 'en_en', # Contains correct language names57 'xnli': 'en',58 "paws-x": "en",59 "mlqa": "mlqa.en.en",60 "xquad": "xquad.en",61 "khalidalt/tydiqa-primary": "english",62 "khalidalt/tydiqa-goldp": "english",63 "pasinit/xlwic": "en",64 "GEM/xlsum": "english",65 "GEM/BiSECT": "en",66}67 68BIAS_FAIRNESS = [69 ('crows_pairs', None),70 ('jigsaw_toxicity_pred', None),71 ('super_glue','axg'),72 ('wino_bias','type1_anti'),73 ('wino_bias','type2_anti'),74 ('wino_bias','type1_pro'),75 ('wino_bias','type2_pro'),76]77 78EVAL_DATASETS_L1 = [79 ('super_glue','wsc.fixed'),80 ('winogrande','winogrande_xl'),81 ('super_glue','cb'),82 ('super_glue','rte'),83 ('anli',None),84 ('story_cloze', '2016'),85 ('Muennighoff/xstory_cloze', 'ar'),86 ('Muennighoff/xstory_cloze', 'es'),87 ('Muennighoff/xstory_cloze', 'eu'),88 ('Muennighoff/xstory_cloze', 'id'),89 ('Muennighoff/xstory_cloze', 'hi'),90 ('Muennighoff/xstory_cloze', 'te'),91 ('Muennighoff/xstory_cloze', 'sw'),92 ('Muennighoff/xstory_cloze', 'zh'),93 ('hellaswag', None),94 ('super_glue', 'copa'),95 # Multilingual96 ('Muennighoff/xwinograd','en'),97 ('Muennighoff/xwinograd','fr'),98 ('Muennighoff/xwinograd','pt'),99 ('Muennighoff/xwinograd','zh'),100 ('clue', 'cluewsc2020'),101 ('xcopa','id'),102 ('xcopa','ta'),103 ('xcopa','sw'),104 ('xcopa','vi'),105 ('xcopa','zh'),106 ("xnli", "ar"),107 ("xnli", "en"),108 ("xnli", "es"),109 ("xnli", "fr"),110 ("xnli", "hi"),111 ("xnli", "sw"),112 ("xnli", "ur"),113 ("xnli", "vi"),114 ("xnli", "zh"),115 ("openai_humaneval", None),116 ("multi_eurlex", "all_languages")117]118 119ADD_TRAIN_DATASETS_L1_BLOOMZZ = [120 ('super_glue','wsc.fixed'),121 ('winogrande','winogrande_xl'),122 ('story_cloze', '2016'),123 ('Muennighoff/xstory_cloze', 'ar'),124 ('Muennighoff/xstory_cloze', 'es'),125 ('Muennighoff/xstory_cloze', 'eu'),126 ('Muennighoff/xstory_cloze', 'id'),127 ('Muennighoff/xstory_cloze', 'hi'),128 ('Muennighoff/xstory_cloze', 'te'),129 ('Muennighoff/xstory_cloze', 'sw'),130 ('Muennighoff/xstory_cloze', 'zh'),131 ('hellaswag', None),132 ('super_glue', 'copa'),133 # Multilingual134 ('Muennighoff/xwinograd','en'),135 ('Muennighoff/xwinograd','fr'),136 ('Muennighoff/xwinograd','pt'),137 ('Muennighoff/xwinograd','zh'),138 ('clue', 'cluewsc2020'),139 ('xcopa','id'),140 ('xcopa','ta'),141 ('xcopa','sw'),142 ('xcopa','vi'),143 ('xcopa','zh'),144 ("multi_eurlex", "all_languages")145 # ("openai_humaneval", None), # Low quality prompts146]147 148EVAL_DATASETS_L2 = [149 ('Muennighoff/xwinograd','jp'),150 ('Muennighoff/xwinograd','ru'),151 ('xcopa','et'),152 ('xcopa','ht'),153 ('xcopa','it'),154 ('xcopa','qu'),155 ('xcopa','th'),156 ('xcopa','tr'),157 ("xnli", "bg"),158 ("xnli", "de"),159 ("xnli", "el"),160 ("xnli", "ru"),161 ("xnli", "th"),162 ("xnli", "tr"),163]164 165TRAIN_DATASETS = [166 # English-only167 ('glue','mrpc'), 168 ('glue','qqp'),169 ('paws','labeled_final'),170 ('ai2_arc','ARC-Challenge'),171 ('ai2_arc','ARC-Easy'),172 ('kilt_tasks','hotpotqa'),173 ('trivia_qa','unfiltered'),174 ('web_questions',None),175 ('wiki_qa',None),176 ('adversarial_qa','dbidaf'),177 ('adversarial_qa','dbert'),178 ('adversarial_qa','droberta'),179 ('duorc','SelfRC'),180 ('duorc','ParaphraseRC'),181 ('ropes',None),182 ('squad_v2',None),183 ('super_glue','record'),184 ('quoref',None),185 ('cos_e','v1.11'),186 ('cosmos_qa',None),187 ('dream',None),188 ('openbookqa','main'),189 ('qasc',None),190 ('quail',None),191 ('quarel',None),192 ('quartz',None),193 ('race','high'),194 ('race','middle'),195 ('sciq',None),196 ('social_i_qa',None),197 ('super_glue','boolq'),198 ('super_glue','multirc'),199 ('wiki_hop','original'),200 ('wiqa',None),201 ('piqa',None),202 ('amazon_polarity',None),203 ('app_reviews',None),204 ('imdb',None),205 ('rotten_tomatoes',None),206 ('yelp_review_full',None),207 ('common_gen',None),208 ('wiki_bio',None),209 ('cnn_dailymail','3.0.0'),210 ('gigaword',None),211 ('multi_news',None),212 ('samsum',None),213 ('xsum',None),214 ('ag_news',None),215 ('dbpedia_14',None),216 ('trec',None),217 # Multilingual218 ('GEM/wiki_lingua', 'ar'),219 ('GEM/wiki_lingua', 'en'),220 ('GEM/wiki_lingua', 'es'),221 ('GEM/wiki_lingua', 'fr'),222 ('GEM/wiki_lingua', 'hi'),223 ('GEM/wiki_lingua', 'id'),224 ('GEM/wiki_lingua', 'pt'),225 ('GEM/wiki_lingua', 'vi'),226 ('GEM/wiki_lingua', 'zh'),227 ('Helsinki-NLP/tatoeba_mt', 'ara-eng'),228 ('Helsinki-NLP/tatoeba_mt', 'ara-fra'),229 ('Helsinki-NLP/tatoeba_mt', 'ara-spa'),230 ('Helsinki-NLP/tatoeba_mt', 'ben-eng'),231 ('Helsinki-NLP/tatoeba_mt', 'cat-eng'),232 ('Helsinki-NLP/tatoeba_mt', 'cat-fra'),233 ('Helsinki-NLP/tatoeba_mt', 'cat-por'),234 ('Helsinki-NLP/tatoeba_mt', 'cat-spa'),235 ('Helsinki-NLP/tatoeba_mt', 'eng-cmn_Hans'),236 ('Helsinki-NLP/tatoeba_mt', 'eng-cmn_Hant'),237 ('Helsinki-NLP/tatoeba_mt', 'eng-eus'),238 ('Helsinki-NLP/tatoeba_mt', 'eng-fra'),239 ('Helsinki-NLP/tatoeba_mt', 'eng-hin'),240 ('Helsinki-NLP/tatoeba_mt', 'eng-ind'),241 ('Helsinki-NLP/tatoeba_mt', 'eng-mal'),242 ('Helsinki-NLP/tatoeba_mt', 'eng-mar'),243 ('Helsinki-NLP/tatoeba_mt', 'eng-por'),244 ('Helsinki-NLP/tatoeba_mt', 'eng-run'),245 ('Helsinki-NLP/tatoeba_mt', 'eng-spa'),246 ('Helsinki-NLP/tatoeba_mt', 'eng-swa'),247 ('Helsinki-NLP/tatoeba_mt', 'eng-tam'),248 ('Helsinki-NLP/tatoeba_mt', 'eng-tel'),249 ('Helsinki-NLP/tatoeba_mt', 'eng-urd'),250 ('Helsinki-NLP/tatoeba_mt', 'eng-vie'),251 ('Helsinki-NLP/tatoeba_mt', 'eng-zho'),252 ('Helsinki-NLP/tatoeba_mt', 'eus-spa'),253 ('Helsinki-NLP/tatoeba_mt', 'fra-cmn_Hans'),254 ('Helsinki-NLP/tatoeba_mt', 'fra-cmn_Hant'),255 ('Helsinki-NLP/tatoeba_mt', 'fra-ind'),256 ('Helsinki-NLP/tatoeba_mt', 'fra-por'),257 ('Helsinki-NLP/tatoeba_mt', 'fra-run'),258 ('Helsinki-NLP/tatoeba_mt', 'fra-spa'),259 ('Helsinki-NLP/tatoeba_mt', 'fra-vie'),260 ('Helsinki-NLP/tatoeba_mt', 'fra-zho'),261 ('Helsinki-NLP/tatoeba_mt', 'hin-urd'),262 ('Helsinki-NLP/tatoeba_mt', 'hin-zho'),263 ('Helsinki-NLP/tatoeba_mt', 'por-cmn_Hans'),264 ('Helsinki-NLP/tatoeba_mt', 'por-cmn_Hant'),265 ('Helsinki-NLP/tatoeba_mt', 'por-spa'),266 ('Helsinki-NLP/tatoeba_mt', 'por-zho'),267 ('Helsinki-NLP/tatoeba_mt', 'run-spa'),268 ('Helsinki-NLP/tatoeba_mt', 'spa-cmn_Hans'),269 ('Helsinki-NLP/tatoeba_mt', 'spa-cmn_Hant'),270 ('Helsinki-NLP/tatoeba_mt', 'spa-vie'),271 ('Helsinki-NLP/tatoeba_mt', 'spa-zho'),272 ('Helsinki-NLP/tatoeba_mt', 'vie-cmn_Hans'),273 ('Helsinki-NLP/tatoeba_mt', 'vie-zho'),274 ('xquad', 'xquad.ar'),275 ('xquad', 'xquad.zh'),276 ('xquad', 'xquad.vi'),277 ('xquad', 'xquad.en'),278 ('xquad', 'xquad.es'),279 ('xquad', 'xquad.hi'),280 ('mlqa', 'mlqa.ar.ar'),281 ('mlqa', 'mlqa.vi.vi'),282 ('mlqa', 'mlqa.zh.zh'),283 ('mlqa', 'mlqa.es.es'),284 ('mlqa', 'mlqa.en.en'),285 ('mlqa', 'mlqa.hi.hi'),286 287 ('mlqa', 'mlqa.ar.vi'),288 ('mlqa', 'mlqa.ar.zh'),289 ('mlqa', 'mlqa.ar.es'),290 ('mlqa', 'mlqa.ar.en'),291 ('mlqa', 'mlqa.ar.hi'),292 293 ('mlqa', 'mlqa.vi.ar'),294 ('mlqa', 'mlqa.vi.zh'),295 ('mlqa', 'mlqa.vi.es'),296 ('mlqa', 'mlqa.vi.en'),297 ('mlqa', 'mlqa.vi.hi'),298 299 ('mlqa', 'mlqa.zh.ar'),300 ('mlqa', 'mlqa.zh.vi'),301 ('mlqa', 'mlqa.zh.es'),302 ('mlqa', 'mlqa.zh.en'),303 ('mlqa', 'mlqa.zh.hi'),304 305 ('mlqa', 'mlqa.es.ar'),306 ('mlqa', 'mlqa.es.vi'),307 ('mlqa', 'mlqa.es.zh'),308 ('mlqa', 'mlqa.es.en'),309 ('mlqa', 'mlqa.es.hi'),310 311 ('mlqa', 'mlqa.en.ar'),312 ('mlqa', 'mlqa.es.vi'),313 ('mlqa', 'mlqa.es.zh'),314 ('mlqa', 'mlqa.es.es'),315 ('mlqa', 'mlqa.es.hi'),316 317 ('mlqa', 'mlqa.hi.ar'),318 ('mlqa', 'mlqa.hi.vi'),319 ('mlqa', 'mlqa.hi.zh'),320 ('mlqa', 'mlqa.hi.es'),321 ('mlqa', 'mlqa.hi.en'),322 323 ('paws-x', 'en'),324 ('paws-x', 'es'),325 ('paws-x', 'fr'),326 ('paws-x', 'zh'),327 ('khalidalt/tydiqa-primary', 'arabic'),328 ('khalidalt/tydiqa-primary', 'bengali'),329 ('khalidalt/tydiqa-primary', 'english'),330 ('khalidalt/tydiqa-primary', 'indonesian'),331 ('khalidalt/tydiqa-primary', 'swahili'),332 ('khalidalt/tydiqa-primary', 'telugu'),333 ('khalidalt/tydiqa-goldp', 'arabic'),334 ('khalidalt/tydiqa-goldp', 'bengali'),335 ('khalidalt/tydiqa-goldp', 'english'),336 ('khalidalt/tydiqa-goldp', 'indonesian'),337 ('khalidalt/tydiqa-goldp', 'swahili'),338 ('khalidalt/tydiqa-goldp', 'telugu'),339 ('Muennighoff/mbpp', 'sanitized'),340 ("great_code", None),341 ("neural_code_search", "evaluation_dataset"),342 ("codeparrot/codecomplex", "codeparrot--codecomplex"),343 ("codeparrot/github-jupyter-text-code-pairs", None),344 ("codeparrot/apps", "all"),345 ("codeparrot/xlcost-text-to-code", "Python-program-level"),346 ("codeparrot/xlcost-text-to-code", "C-program-level"),347 ("codeparrot/xlcost-text-to-code", "C++-program-level"),348 ("codeparrot/xlcost-text-to-code", "Csharp-program-level"),349 ("codeparrot/xlcost-text-to-code", "Java-program-level"),350 ("codeparrot/xlcost-text-to-code", "Javascript-program-level"),351 ("codeparrot/xlcost-text-to-code", "PHP-program-level"),352 ("teven/code_contests", None),353 ("teven/code_docstring_corpus", "top_level"),354 ("Fraser/python-state-changes", None),355 ('clue', 'c3'),356 ('clue', 'cmrc2018'),357 ('clue', 'csl'),358 ('clue', 'drcd'),359 ('clue', 'tnews'),360 ('super_glue', 'wic'),361 ('pasinit/xlwic', "xlwic_en_zh"),362 ('pasinit/xlwic', "xlwic_fr_fr"),363 ('GEM/BiSECT', "en"),364 ('GEM/BiSECT', "es"),365 ('GEM/BiSECT', "fr"),366 ('GEM/xlsum', "arabic"),367 ('GEM/xlsum', "bengali"),368 ('GEM/xlsum', "chinese_simplified"),369 ('GEM/xlsum', "chinese_traditional"),370 ('GEM/xlsum', "english"),371 ('GEM/xlsum', "french"),372 ('GEM/xlsum', "gujarati"),373 ('GEM/xlsum', "hindi"),374 ('GEM/xlsum', "igbo"),375 ('GEM/xlsum', "indonesian"),376 ('GEM/xlsum', "kirundi"),377 ('GEM/xlsum', "marathi"),378 ('GEM/xlsum', "nepali"),379 ('GEM/xlsum', "portuguese"),380 ('GEM/xlsum', "punjabi"),381 ('GEM/xlsum', "spanish"),382 ('GEM/xlsum', "swahili"),383 ('GEM/xlsum', "tamil"),384 ('GEM/xlsum', "telugu"),385 ('GEM/xlsum', "urdu"),386 ('GEM/xlsum', "vietnamese"),387 ('GEM/xlsum', "yoruba"),388 # flores200, wmt & more wikilingua added below389]390 391FLORES_LANGS = [392 ("Acehnese (Arabic script)", "ace_Arab"),393 ("Acehnese (Latin script)", "ace_Latn"),394 ("Mesopotamian Arabic", "acm_Arab"),395 ("Ta’izzi-Adeni Arabic", "acq_Arab"),396 ("Tunisian Arabic", "aeb_Arab"),397 ("Afrikaans", "afr_Latn"),398 ("South Levantine Arabic", "ajp_Arab"),399 ("Akan", "aka_Latn"),400 ("Amharic", "amh_Ethi"),401 ("North Levantine Arabic", "apc_Arab"),402 ("Modern Standard Arabic", "arb_Arab"),403 ("Modern Standard Arabic (Romanized)", "arb_Latn"),404 ("Najdi Arabic", "ars_Arab"),405 ("Moroccan Arabic", "ary_Arab"),406 ("Egyptian Arabic", "arz_Arab"),407 ("Assamese", "asm_Beng"),408 ("Asturian", "ast_Latn"),409 ("Awadhi", "awa_Deva"),410 ("Central Aymara", "ayr_Latn"),411 ("South Azerbaijani", "azb_Arab"),412 ("North Azerbaijani", "azj_Latn"),413 ("Bashkir", "bak_Cyrl"),414 ("Bambara", "bam_Latn"),415 ("Balinese", "ban_Latn"),416 ("Belarusian", "bel_Cyrl"),417 ("Bemba", "bem_Latn"),418 ("Bengali", "ben_Beng"),419 ("Bhojpuri", "bho_Deva"),420 ("Banjar (Arabic script)", "bjn_Arab"),421 ("Banjar (Latin script)", "bjn_Latn"),422 ("Standard Tibetan", "bod_Tibt"),423 ("Bosnian", "bos_Latn"),424 ("Buginese", "bug_Latn"),425 ("Bulgarian", "bul_Cyrl"),426 ("Catalan", "cat_Latn"),427 ("Cebuano", "ceb_Latn"),428 ("Czech", "ces_Latn"),429 ("Chokwe", "cjk_Latn"),430 ("Central Kurdish", "ckb_Arab"),431 ("Crimean Tatar", "crh_Latn"),432 ("Welsh", "cym_Latn"),433 ("Danish", "dan_Latn"),434 ("German", "deu_Latn"),435 ("Southwestern Dinka", "dik_Latn"),436 ("Dyula", "dyu_Latn"),437 ("Dzongkha", "dzo_Tibt"),438 ("Greek", "ell_Grek"),439 ("English", "eng_Latn"),440 ("Esperanto", "epo_Latn"),441 ("Estonian", "est_Latn"),442 ("Basque", "eus_Latn"),443 ("Ewe", "ewe_Latn"),444 ("Faroese", "fao_Latn"),445 ("Fijian", "fij_Latn"),446 ("Finnish", "fin_Latn"),447 ("Fon", "fon_Latn"),448 ("French", "fra_Latn"),449 ("Friulian", "fur_Latn"),450 ("Nigerian Fulfulde", "fuv_Latn"),451 ("Scottish Gaelic", "gla_Latn"),452 ("Irish", "gle_Latn"),453 ("Galician", "glg_Latn"),454 ("Guarani", "grn_Latn"),455 ("Gujarati", "guj_Gujr"),456 ("Haitian Creole", "hat_Latn"),457 ("Hausa", "hau_Latn"),458 ("Hebrew", "heb_Hebr"),459 ("Hindi", "hin_Deva"),460 ("Chhattisgarhi", "hne_Deva"),461 ("Croatian", "hrv_Latn"),462 ("Hungarian", "hun_Latn"),463 ("Armenian", "hye_Armn"),464 ("Igbo", "ibo_Latn"),465 ("Ilocano", "ilo_Latn"),466 ("Indonesian", "ind_Latn"),467 ("Icelandic", "isl_Latn"),468 ("Italian", "ita_Latn"),469 ("Javanese", "jav_Latn"),470 ("Japanese", "jpn_Jpan"),471 ("Kabyle", "kab_Latn"),472 ("Jingpho", "kac_Latn"),473 ("Kamba", "kam_Latn"),474 ("Kannada", "kan_Knda"),475 ("Kashmiri (Arabic script)", "kas_Arab"),476 ("Kashmiri (Devanagari script)", "kas_Deva"),477 ("Georgian", "kat_Geor"),478 ("Central Kanuri (Arabic script)", "knc_Arab"),479 ("Central Kanuri (Latin script)", "knc_Latn"),480 ("Kazakh", "kaz_Cyrl"),481 ("Kabiyè", "kbp_Latn"),482 ("Kabuverdianu", "kea_Latn"),483 ("Khmer", "khm_Khmr"),484 ("Kikuyu", "kik_Latn"),485 ("Kinyarwanda", "kin_Latn"),486 ("Kyrgyz", "kir_Cyrl"),487 ("Kimbundu", "kmb_Latn"),488 ("Northern Kurdish", "kmr_Latn"),489 ("Kikongo", "kon_Latn"),490 ("Korean", "kor_Hang"),491 ("Lao", "lao_Laoo"),492 ("Ligurian", "lij_Latn"),493 ("Limburgish", "lim_Latn"),494 ("Lingala", "lin_Latn"),495 ("Lithuanian", "lit_Latn"),496 ("Lombard", "lmo_Latn"),497 ("Latgalian", "ltg_Latn"),498 ("Luxembourgish", "ltz_Latn"),499 ("Luba-Kasai", "lua_Latn"),500 ("Ganda", "lug_Latn"),501 ("Luo", "luo_Latn"),502 ("Mizo", "lus_Latn"),503 ("Standard Latvian", "lvs_Latn"),504 ("Magahi", "mag_Deva"),505 ("Maithili", "mai_Deva"),506 ("Malayalam", "mal_Mlym"),507 ("Marathi", "mar_Deva"),508 ("Minangkabau (Arabic script)", "min_Arab"),509 ("Minangkabau (Latin script)", "min_Latn"),510 ("Macedonian", "mkd_Cyrl"),511 ("Plateau Malagasy", "plt_Latn"),512 ("Maltese", "mlt_Latn"),513 ("Meitei (Bengali script)", "mni_Beng"),514 ("Halh Mongolian", "khk_Cyrl"),515 ("Mossi", "mos_Latn"),516 ("Maori", "mri_Latn"),517 ("Burmese", "mya_Mymr"),518 ("Dutch", "nld_Latn"),519 ("Norwegian Nynorsk", "nno_Latn"),520 ("Norwegian Bokmål", "nob_Latn"),521 ("Nepali", "npi_Deva"),522 ("Northern Sotho", "nso_Latn"),523 ("Nuer", "nus_Latn"),524 ("Nyanja", "nya_Latn"),525 ("Occitan", "oci_Latn"),526 ("West Central Oromo", "gaz_Latn"),527 ("Odia", "ory_Orya"),528 ("Pangasinan", "pag_Latn"),529 ("Eastern Panjabi", "pan_Guru"),530 ("Papiamento", "pap_Latn"),531 ("Western Persian", "pes_Arab"),532 ("Polish", "pol_Latn"),533 ("Portuguese", "por_Latn"),534 ("Dari", "prs_Arab"),535 ("Southern Pashto", "pbt_Arab"),536 ("Ayacucho Quechua", "quy_Latn"),537 ("Romanian", "ron_Latn"),538 ("Rundi", "run_Latn"),539 ("Russian", "rus_Cyrl"),540 ("Sango", "sag_Latn"),541 ("Sanskrit", "san_Deva"),542 ("Santali", "sat_Olck"),543 ("Sicilian", "scn_Latn"),544 ("Shan", "shn_Mymr"),545 ("Sinhala", "sin_Sinh"),546 ("Slovak", "slk_Latn"),547 ("Slovenian", "slv_Latn"),548 ("Samoan", "smo_Latn"),549 ("Shona", "sna_Latn"),550 ("Sindhi", "snd_Arab"),551 ("Somali", "som_Latn"),552 ("Southern Sotho", "sot_Latn"),553 ("Spanish", "spa_Latn"),554 ("Tosk Albanian", "als_Latn"),555 ("Sardinian", "srd_Latn"),556 ("Serbian", "srp_Cyrl"),557 ("Swati", "ssw_Latn"),558 ("Sundanese", "sun_Latn"),559 ("Swedish", "swe_Latn"),560 ("Swahili", "swh_Latn"),561 ("Silesian", "szl_Latn"),562 ("Tamil", "tam_Taml"),563 ("Tatar", "tat_Cyrl"),564 ("Telugu", "tel_Telu"),565 ("Tajik", "tgk_Cyrl"),566 ("Tagalog", "tgl_Latn"),567 ("Thai", "tha_Thai"),568 ("Tigrinya", "tir_Ethi"),569 ("Tamasheq (Latin script)", "taq_Latn"),570 ("Tamasheq (Tifinagh script)", "taq_Tfng"),571 ("Tok Pisin", "tpi_Latn"),572 ("Tswana", "tsn_Latn"),573 ("Tsonga", "tso_Latn"),574 ("Turkmen", "tuk_Latn"),575 ("Tumbuka", "tum_Latn"),576 ("Turkish", "tur_Latn"),577 ("Twi", "twi_Latn"),578 ("Central Atlas Tamazight", "tzm_Tfng"),579 ("Uyghur", "uig_Arab"),580 ("Ukrainian", "ukr_Cyrl"),581 ("Umbundu", "umb_Latn"),582 ("Urdu", "urd_Arab"),583 ("Northern Uzbek", "uzn_Latn"),584 ("Venetian", "vec_Latn"),585 ("Vietnamese", "vie_Latn"),586 ("Waray", "war_Latn"),587 ("Wolof", "wol_Latn"),588 ("Xhosa", "xho_Latn"),589 ("Eastern Yiddish", "ydd_Hebr"),590 ("Yoruba", "yor_Latn"),591 ("Yue Chinese", "yue_Hant"),592 ("Chinese (Simplified)", "zho_Hans"),593 ("Chinese (Traditional)", "zho_Hant"),594 ("Standard Malay", "zsm_Latn"),595 ("Zulu", "zul_Latn"),596]597 598WMT22_LANGS = [599 ("afr", "eng"),600 ("afr", "som"),601 ("amh", "eng"),602 ("amh", "fra"),603 ("amh", "nya"),604 ("amh", "orm"),605 ("amh", "sna"),606 ("amh", "som"),607 ("amh", "ssw"),608 ("amh", "swh"),609 ("amh", "tsn"),610 ("amh", "tso"),611 ("amh", "umb"),612 ("amh", "xho"),613 ("amh", "yor"),614 ("amh", "zul"),615 ("eng", "fuv"),616 ("eng", "hau"),617 ("eng", "ibo"),618 ("eng", "kam"),619 ("eng", "kin"),620 ("eng", "lin"),621 ("eng", "lug"),622 ("eng", "luo"),623 ("eng", "nso"),624 ("eng", "nya"),625 ("eng", "orm"),626 ("eng", "sna"),627 ("eng", "som"),628 ("eng", "ssw"),629 ("eng", "swh"),630 ("eng", "tsn"),631 ("eng", "tso"),632 ("eng", "umb"),633 ("eng", "wol"),634 ("eng", "xho"),635 ("eng", "yor"),636 ("eng", "zul"),637 ("fra", "hau"),638 ("fra", "ibo"),639 ("fra", "kam"),640 ("fra", "kin"),641 ("fra", "lin"),642 ("fra", "lug"),643 ("fra", "luo"),644 ("fra", "nso"),645 ("fra", "nya"),646 ("fra", "orm"),647 ("fra", "som"),648 ("fra", "ssw"),649 ("fra", "swh"),650 ("fra", "tsn"),651 ("fra", "tso"),652 ("fra", "umb"),653 ("fra", "wol"),654 ("fra", "xho"),655 ("fra", "zul"),656 ("fuv", "hau"),657 ("fuv", "ibo"),658 ("fuv", "kam"),659 ("fuv", "kin"),660 ("fuv", "lug"),661 ("fuv", "luo"),662 ("fuv", "nso"),663 ("fuv", "nya"),664 ("fuv", "orm"),665 ("fuv", "sna"),666 ("fuv", "som"),667 ("fuv", "ssw"),668 ("fuv", "swh"),669 ("fuv", "tsn"),670 ("fuv", "tso"),671 ("fuv", "umb"),672 ("fuv", "xho"),673 ("fuv", "yor"),674 ("fuv", "zul"),675 ("hau", "ibo"),676 ("hau", "kam"),677 ("hau", "kin"),678 ("hau", "lug"),679 ("hau", "luo"),680 ("hau", "nso"),681 ("hau", "nya"),682 ("hau", "orm"),683 ("hau", "sna"),684 ("hau", "som"),685 ("hau", "ssw"),686 ("hau", "swh"),687 ("hau", "tsn"),688 ("hau", "tso"),689 ("hau", "umb"),690 ("hau", "xho"),691 ("hau", "yor"),692 ("hau", "zul"),693 ("ibo", "kam"),694 ("ibo", "kin"),695 ("ibo", "lug"),696 ("ibo", "luo"),697 ("ibo", "nso"),698 ("ibo", "nya"),699 ("ibo", "orm"),700 ("ibo", "sna"),701 ("ibo", "som"),702 ("ibo", "ssw"),703 ("ibo", "swh"),704 ("ibo", "tsn"),705 ("ibo", "tso"),706 ("ibo", "umb"),707 ("ibo", "xho"),708 ("ibo", "yor"),709 ("ibo", "zul"),710 ("kam", "kin"),711 ("kam", "lug"),712 ("kam", "luo"),713 ("kam", "nso"),714 ("kam", "nya"),715 ("kam", "orm"),716 ("kam", "sna"),717 ("kam", "som"),718 ("kam", "ssw"),719 ("kam", "swh"),720 ("kam", "tsn"),721 ("kam", "tso"),722 ("kam", "umb"),723 ("kam", "xho"),724 ("kam", "yor"),725 ("kam", "zul"),726 ("kin", "lug"),727 ("kin", "luo"),728 ("kin", "nso"),729 ("kin", "nya"),730 ("kin", "orm"),731 ("kin", "sna"),732 ("kin", "som"),733 ("kin", "ssw"),734 ("kin", "swh"),735 ("kin", "tsn"),736 ("kin", "tso"),737 ("kin", "umb"),738 ("kin", "xho"),739 ("kin", "yor"),740 ("kin", "zul"),741 ("lug", "luo"),742 ("lug", "nso"),743 ("lug", "nya"),744 ("lug", "orm"),745 ("lug", "sna"),746 ("lug", "som"),747 ("lug", "ssw"),748 ("lug", "swh"),749 ("lug", "tsn"),750 ("lug", "tso"),751 ("lug", "umb"),752 ("lug", "xho"),753 ("lug", "yor"),754 ("lug", "zul"),755 ("luo", "nso"),756 ("luo", "nya"),757 ("luo", "orm"),758 ("luo", "sna"),759 ("luo", "som"),760 ("luo", "ssw"),761 ("luo", "swh"),762 ("luo", "tsn"),763 ("luo", "tso"),764 ("luo", "umb"),765 ("luo", "xho"),766 ("luo", "yor"),767 ("luo", "zul"),768 ("nso", "nya"),769 ("nso", "orm"),770 ("nso", "sna"),771 ("nso", "som"),772 ("nso", "ssw"),773 ("nso", "swh"),774 ("nso", "tsn"),775 ("nso", "tso"),776 ("nso", "umb"),777 ("nso", "xho"),778 ("nso", "yor"),779 ("nso", "zul"),780 ("nya", "orm"),781 ("nya", "sna"),782 ("nya", "som"),783 ("nya", "ssw"),784 ("nya", "swh"),785 ("nya", "tsn"),786 ("nya", "tso"),787 ("nya", "umb"),788 ("nya", "xho"),789 ("nya", "yor"),790 ("nya", "zul"),791 ("orm", "sna"),792 ("orm", "som"),793 ("orm", "ssw"),794 ("orm", "swh"),795 ("orm", "tsn"),796 ("orm", "tso"),797 ("orm", "umb"),798 ("orm", "xho"),799 ("orm", "yor"),800 ("orm", "zul"),801 ("sna", "som"),802 ("sna", "ssw"),803 ("sna", "swh"),804 ("sna", "tsn"),805 ("sna", "tso"),806 ("sna", "umb"),807 ("sna", "xho"),808 ("sna", "yor"),809 ("sna", "zul"),810 ("som", "ssw"),811 ("som", "swh"),812 ("som", "tsn"),813 ("som", "tso"),814 ("som", "umb"),815 ("som", "wol"),816 ("som", "xho"),817 ("som", "yor"),818 ("som", "zul"),819 ("ssw", "swh"),820 ("ssw", "tsn"),821 ("ssw", "tso"),822 ("ssw", "umb"),823 ("ssw", "xho"),824 ("ssw", "yor"),825 ("ssw", "zul"),826 ("swh", "tsn"),827 ("swh", "tso"),828 ("swh", "umb"),829 ("swh", "xho"),830 ("swh", "yor"),831 ("swh", "zul"),832 ("tsn", "tso"),833 ("tsn", "umb"),834 ("tsn", "xho"),835 ("tsn", "yor"),836 ("tsn", "zul"),837 ("tso", "umb"),838 ("tso", "xho"),839 ("tso", "yor"),840 ("tso", "zul"),841 ("umb", "xho"),842 ("umb", "yor"),843 ("umb", "zul"),844 ("xho", "yor"),845 ("xho", "zul"),846 ("yor", "zul"),847]848 849# Copied from metadata850BLOOM_LANGS = """851- ak852- ar853- as854- bm855- bn856- ca857- code858- en859- es860- eu861- fon862- fr863- gu864- hi865- id866- ig867- ki868- kn869- lg870- ln871- ml872- mr873- ne874- nso875- ny876- or877- pa878- pt879- rn880- rw881- sn882- st883- sw884- ta885- te886- tn887- ts888- tum889- tw890- ur891- vi892- wo893- xh894- yo895- zh896- zu897"""898 899DS_TO_LANG = {900 'Muennighoff/mbpp': 'code',901 'openai_humaneval': 'code',902 "great_code": "code",903 "neural_code_search": "code",904 "codeparrot/codecomplex": "code",905 "codeparrot/github-jupyter-text-code-pairs": "code",906 "codeparrot/apps": "code",907 "Fraser/python-state-changes": "code",908 "codeparrot/xlcost-text-to-code": "code",909 "teven/code_contests": "code",910 "teven/code_docstring_corpus": "code",911 "clue": "zh",912 "cmn": "zh", # == zho913 "npi": "ne", # == npe914 "ory": "or", # == ori915 "swh": "sw", # == swa916 "kirundi": "rn", # == rundi917 "punjabi": "pa", # == panjabi918 "chinese_simplified": "zh",919 "chinese_traditional": "zh",920}921 922 923 924bloom_lang_codes_iso3 = []925bloom_lang_codes_iso2 = []926for lang in BLOOM_LANGS.split("\n")[1:-1]:927 iso2 = lang.replace("- ", "")928 DS_TO_LANG[iso2] = iso2929 try:930 name = languages.get(alpha2=iso2)931 DS_TO_LANG[name.name.lower()] = iso2932 # name is e.g. 'swahili (macrolanguage)' also add swahili933 DS_TO_LANG[name.name.lower().split(" ")[0]] = iso2934 935 iso3 = name.part3936 DS_TO_LANG[iso3] = iso2937 except KeyError:938 print(f"Could not find iso3 code for {lang}.")939 940# Add GEM multilingual941WIKILINGUA_LANGS = ["ar", "en", "es", "fr", "hi", "id", "pt", "vi", "zh"]942for l1_code in WIKILINGUA_LANGS:943 for l2_code in WIKILINGUA_LANGS:944 if l1_code == l2_code:945 continue946 TRAIN_DATASETS.append(("GEM/wiki_lingua", f"{l1_code}_{l2_code}"))947 948# Add flores200949for (l1_name, l1_code) in FLORES_LANGS:950 for (l2_name, l2_code) in FLORES_LANGS:951 if l1_code.split("_")[0] not in DS_TO_LANG or l2_code.split("_")[0] not in DS_TO_LANG:952 print(f"Skipping as {l1_name} or {l2_name} was not pre-trained on.")953 continue954 elif l1_name == l2_name:955 continue956 TRAIN_DATASETS.append(("facebook/flores", f"{l1_code}-{l2_code}"))957 958# Add wmt22959for (l1_code, l2_code) in WMT22_LANGS:960 if l1_code not in DS_TO_LANG or l2_code not in DS_TO_LANG:961 print(f"Skipping as {l1_code} or {l2_code} was not pre-trained on.")962 continue963 elif l1_code == l2_code:964 continue965 TRAIN_DATASETS.append(("allenai/wmt22_african", f"{l1_code}-{l2_code}"))966 967 968### DATASET CREATION ###969 970 971# Copied from promptsource.utils972def removeHyphen(example):973 example_clean = {}974 for key in example.keys():975 if "-" in key:976 new_key = key.replace("-", "_")977 example_clean[new_key] = example[key]978 else:979 example_clean[key] = example[key]980 example = example_clean981 return example982 983def apply_template(dataset, template, strip_connection=True):984 def map_fn(ex):985 ex = removeHyphen(ex)986 try:987 inputs_and_targets = template.apply(988 ex, 989 strip_connection=strip_connection,990 truncate=True,991 )992 # Skip ValueError("Prompt did not produce an input and at least one target.")993 # which happens for some prompts with if else clauses based on inputs producing occasional994 # empty targets995 except ValueError:996 return {"inputs": "", "targets": ""}997 if len(inputs_and_targets) == 2:998 # Note that the signature changed in promptsource 999 # In 0.1.0 template.apply returned two strings; In >0.3.0 it retuns a str & list1000 inputs, targets = inputs_and_targets1001 if len(targets) > 1:1002 # Safer to skip, as could be a bug1003 print(f"Found targets longer than 1. Inputs: {inputs} ; Targets {targets}. Skipping.")1004 return {"inputs": "", "targets": ""}1005 targets = targets[0]1006 return {"inputs": inputs, "targets": targets}1007 # When template results in an empty example, template.apply returns [""]1008 # Also, if the template gets split wrong, len can be > 21009 # We will filter these out later1010 else:1011 # inputs is a str by default & targets a str1012 return {"inputs": "", "targets": ""}1013 1014 def filter_fn(ex):1015 return len(ex["inputs"]) > 0 and len(ex["targets"]) > 01016 1017 original_columns = dataset.column_names1018 dataset = dataset.map(map_fn).filter(filter_fn)1019 # map keeps original columns, remove them1020 return dataset.remove_columns(set(original_columns) - {"inputs", "targets"})1021 1022def add_language_name_wikilingua(example):1023 example["source_language_name"] = languages.get(alpha2=example["source_language"]).name1024 example["target_language_name"] = languages.get(alpha2=example["target_language"]).name1025 return example1026 1027def filter_l1_l2_wikilingua(example, l1, l2):1028 return example["source_language"] == l1 and example["target_language"] == l21029 1030def filter_empty_solution_apps(example):1031 return bool(example["solutions"])1032 1033def add_solution_apps(example):1034 example["solution"] = random.choice(json.loads(example["solutions"]))1035 return example1036 1037def clean_code_xlcost(example):1038 clean_lines = []1039 cur_indent = 01040 for line in example["code"].split("NEW_LINE"):1041 cur_indent += line.count("INDENT")1042 cur_indent -= line.count("DEDENT")1043 line = line.replace("INDENT", "").replace("DEDENT", "")1044 line = line.replace("STRNEWLINE", "\n")1045 line = line.replace("TABSYMBOL", "\t")1046 clean_lines.append("\t" * cur_indent + line.strip())1047 example["code_clean"] = "\n".join(clean_lines)1048 return example1049 1050def write_to_jsonl_hub(ds, split="train"):1051 1052 ### GET DATASET & LANGUAGE ###1053 1054 ds_name, subset_name = ds1055 1056 is_wikilingua_cross_lingual = (ds_name == "GEM/wiki_lingua") and ("_") in subset_name1057 1058 lang_dir = DS_TO_LANG.get(ds_name, None)1059 if lang_dir is None:1060 lang_dir = DS_TO_LANG.get(subset_name, "en")1061 if ds_name == "facebook/flores":1062 lang_dir = DS_TO_LANG.get(subset_name.split("-")[-1].split("_")[0])1063 elif is_wikilingua_cross_lingual or ds_name == "pasinit/xlwic":1064 lang_dir = DS_TO_LANG.get(subset_name.split("_")[-1])1065 elif ds_name == "xquad":1066 lang_dir = DS_TO_LANG.get(subset_name.split(".")[1])1067 elif ds_name == "mlqa":1068 # Classify it by the target language for cross-lingual (i.e. what the loss is computed on)1069 lang_dir = DS_TO_LANG.get(subset_name.split(".")[1])1070 os.makedirs(lang_dir, exist_ok=True)1071 1072 if ds_name == "Helsinki-NLP/tatoeba_mt":1073 ds = load_dataset(ds_name, subset_name, ignore_verifications=True, revision="49aa20ac768eabc5a106a123549ea58053fc9b40")1074 elif ds_name == "story_cloze": 1075 ds = load_dataset(ds_name, subset_name, data_dir=STORY_CLOZE_DIR)1076 elif ds_name == "Muennighoff/xstory_cloze":1077 ds = load_dataset(ds_name, subset_name, data_dir=XSTORY_CLOZE_DIR)1078 else:1079 ds = load_dataset(ds_name, subset_name)1080 1081 if ds_name == "GEM/wiki_lingua":1082 # Add names, e.g. Chinese for zh to use them in the jinja prompts1083 ds = ds.map(add_language_name_wikilingua)1084 if is_wikilingua_cross_lingual:1085 # Keep only L1 -> L2 (L2 -> L1 will be a separate dataset)1086 ds = ds.filter(partial(filter_l1_l2_wikilingua, l1=subset_name.split("_")[0], l2=subset_name.split("_")[1]))1087 elif ds_name == "codeparrot/apps":1088 ds = ds.filter(filter_empty_solution_apps).map(add_solution_apps)1089 elif ds_name == "codeparrot/xlcost-text-to-code":1090 ds = ds.map(clean_code_xlcost)1091 1092 ### SELECT SPLITS ###1093 1094 dataset_splits = list(ds.keys())1095 if subset_name == "xlwic_en_zh":1096 # Train set is en; val & test are zh1097 dataset_splits.remove("train")1098 elif ds_name == "teven/code_docstring_corpus":1099 # Bad quality split1100 dataset_splits.remove("class_level")1101 1102 if split == "validation":1103 if split not in dataset_splits or len(dataset_splits) == 1:1104 print(f"Validation not found for {ds_name}")1105 return1106 dataset_splits = ["validation"]1107 elif split == "train":1108 # Use as much as possible1109 # Would need to remove e.g. test datasets to benchmark same task performance1110 if len(dataset_splits) > 1 and "validation" in dataset_splits:1111 dataset_splits.remove("validation")1112 # WikiLingua1113 if "sampled_validation" in dataset_splits:1114 dataset_splits.remove("sampled_validation")1115 if "sampled_test" in dataset_splits:1116 dataset_splits.remove("sampled_test")1117 1118 ### SELECT PROMPTS ###1119 1120 if subset_name is None:1121 prompt_dataset_name = ds_name1122 else:1123 subset_name_prompt = subset_name1124 if USE_ENGLISH_PROMPTS and ds_name in DS_TO_ENG_PROMPT:1125 subset_name_prompt = DS_TO_ENG_PROMPT[ds_name]1126 prompt_dataset_name = f"{ds_name}/{subset_name_prompt}"1127 1128 prompts = DatasetTemplates(prompt_dataset_name)1129 1130 ### PROCESS ###1131 1132 for split in dataset_splits:1133 for t_name in prompts.all_template_names:1134 #if not "mt" in t_name:1135 # print(f"Skipping {t_name}")1136 # continue1137 print(f"Running {ds_name}/{subset_name}/{split}/{t_name}")1138 if SKIP_PROMPTS.get(prompt_dataset_name, {}).get(split, False):1139 if ("all" in SKIP_PROMPTS[prompt_dataset_name][split]) or (t_name in SKIP_PROMPTS[prompt_dataset_name][split]):1140 print(f"Skipping DS: {prompt_dataset_name} Split {split} Prompt {t_name}")1141 continue1142 1143 if ds_name == "Helsinki-NLP/tatoeba_mt":1144 # E.g. translate-this-ara-eng, where eng is the target1145 lang_dir = DS_TO_LANG.get(t_name.split("-")[-1].split("_")[0], "en")1146 elif ds_name in ("allenai/wmt22_african", "multi_eurlex"):1147 # One prompt in multi_eurlex has -source+target appended to the languages1148 lang_dir = DS_TO_LANG.get(t_name.replace("-source+target", "").split("-")[-1])1149 1150 out_path = os.path.join(1151 lang_dir, 1152 f'xp3_{ds_name}_{subset_name}_{split}_{t_name}.jsonl'.replace("/", "_").replace(" ", "_")1153 )1154 if os.path.exists(out_path):1155 print("Skipping as exists: ", out_path)1156 continue1157 1158 assert len(ds[split]) > 0, f"Got empty: {ds_name}"1159 1160 try:1161 if ds_name == "allenai/wmt22_african":1162 # Sort by laser score, i.e. by increasing confidence & limit samples due to mediocre quality1163 ds[split] = ds[split].sort("laser_score", reverse=True)1164 max_range = min(len(ds[split]), MAX_EXAMPLES_PER_DATASET_PROMPT // 2)1165 else:1166 # Allow 5x buffer for empty examples1167 max_range = min(len(ds[split]), MAX_EXAMPLES_PER_DATASET_PROMPT * 5)1168 # Shuffle to avoid using the same subset1169 # Leave \n in-between input & targets for code1170 out_ds = apply_template(1171 dataset=ds[split].shuffle().select(list(range(max_range))), 1172 template=prompts[t_name],1173 strip_connection=False if lang_dir == "code" else True1174 )1175 # Keep X shortest examples1176 max_range = min(len(out_ds), MAX_EXAMPLES_PER_DATASET_PROMPT)1177 out_ds = out_ds.sort("inputs").select(list(range(max_range)))1178 except Exception as e:1179 print(f"Skipping due to {e}. DS: {ds_name}/{subset_name} Template: {t_name}")1180 continue1181 # Do not force ascii to allow chars like é1182 if len(out_ds) > 0:1183 out_ds.to_json(out_path, orient="records", lines=True, force_ascii=False)1184 1185# Testing:1186TRAIN_DATASETS = [1187 ('xquad', 'xquad.ar'),1188 ('xquad', 'xquad.vi'),1189 ('xquad', 'xquad.en'),1190 ('xquad', 'xquad.es'),1191 ('xquad', 'xquad.hi'),1192 ('mlqa', 'mlqa.ar.ar'),1193 ('mlqa', 'mlqa.vi.vi'),1194 ('mlqa', 'mlqa.zh.zh'),1195 ('mlqa', 'mlqa.es.es'),1196 ('mlqa', 'mlqa.en.en'),1197 ('mlqa', 'mlqa.hi.hi'),1198 1199 ('mlqa', 'mlqa.ar.vi'),1200 ('mlqa', 'mlqa.ar.zh'),