CoolFace
Datasetpublic

bigscience/xP3mt

xP3 (Crosslingual Public Pool of Prompts) is a collection of prompts & datasets across 46 of languages & 16 NLP tasks. It is used for the training of BLOOMZ and mT0, multilingual language models capable of following human instructions in dozens of languages zero-shot.

sourceHugging Faceapache-2.0updated 3y agoView on Hugging Face
26likes19kdownloads
prep.py1284 linesDownload Raw Back to root
1from functools import partial2import json3import multiprocessing4import os5import random6 7from datasets import load_dataset8# pip install -q iso-6399from iso639 import languages10from promptsource.templates import DatasetTemplates11 12# Set to False to use multilingual prompts e.g. 'id' for xcopa/id instead of 'en'13USE_ENGLISH_PROMPTS = False14 15MAX_EXAMPLES_PER_DATASET_PROMPT = 100_00016 17STORY_CLOZE_DIR = "/gpfswork/rech/six/commun/code/tr13f-6B3-ml-t0/story_cloze_data"18XSTORY_CLOZE_DIR = "/gpfswork/rech/six/commun/code/tr13f-6B3-ml-t0/xstory_cloze_data"19 20# Some datasets have test sets with hidden labels which will still compile but only to noise21# e.g. piqa test labels are all [-1] which still works on list indices resulting in 22# noise samples where the label is always the same  23SKIP_PROMPTS = {24    "common_gen": {"test": ["all"]},25    "piqa": {"test": ["all"]},26    "qasc": {"test": ["all"]},27    "imdb": {"unsupervised": ["all"]},28    "glue/qqp": {"test": ["all"]},29    "qasc": {"test": ["all"]},30    "cosmos_qa": {"test": [31        "description_context_question_answer_text", 32        "description_context_question_text",33        "description_context_question_answer_id",34        "context_answer_to_question",35        "context_description_question_answer_text",36        "context_description_question_answer_id",37        "context_question_description_answer_id",38        "context_description_question_text",39        "context_question_description_answer_text",40        "only_question_answer",41        "no_prompt_id",42        "context_question_description_text",43        "no_prompt_text",44        ]},45    "clue/tnews": {"test": ["all"]},46    "clue/csl": {"test": ["all"]},47    "clue/cmrc2018": {"test": ["generate_question", "in_an_exam", "answer_in_the_passage", "answer_following_question", "xp3longcontinue"]},48    "clue/drcd": {"test": ["generate_question", "in_an_exam", "answer_in_the_passage", "answer_following_question", "xp3longcontinue"]},49    "hellaswag": {"test": ["complete_first_then", "Topic of the context", "Open-ended completion", "Randomized prompts template", "Appropriate continuation - Yes or No", "Predict ending with hint", "Open-ended start", "Reversed appropriate continuation - Yes or No", "how_ends", "if_begins_how_continues"]},50}51 52DS_TO_ENG_PROMPT = {53    "xcopa": "en",54    "Muennighoff/xstory_cloze": "en",55    "Muennighoff/xwinograd": "en",56    'GEM/wiki_lingua': 'en_en', # Contains correct language names57    'xnli': 'en',58    "paws-x": "en",59    "mlqa": "mlqa.en.en",60    "xquad": "xquad.en",61    "khalidalt/tydiqa-primary": "english",62    "khalidalt/tydiqa-goldp": "english",63    "pasinit/xlwic": "en",64    "GEM/xlsum": "english",65    "GEM/BiSECT": "en",66}67 68BIAS_FAIRNESS = [69    ('crows_pairs', None),70    ('jigsaw_toxicity_pred', None),71    ('super_glue','axg'),72    ('wino_bias','type1_anti'),73    ('wino_bias','type2_anti'),74    ('wino_bias','type1_pro'),75    ('wino_bias','type2_pro'),76]77 78EVAL_DATASETS_L1 = [79    ('super_glue','wsc.fixed'),80    ('winogrande','winogrande_xl'),81    ('super_glue','cb'),82    ('super_glue','rte'),83    ('anli',None),84    ('story_cloze', '2016'),85    ('Muennighoff/xstory_cloze', 'ar'),86    ('Muennighoff/xstory_cloze', 'es'),87    ('Muennighoff/xstory_cloze', 'eu'),88    ('Muennighoff/xstory_cloze', 'id'),89    ('Muennighoff/xstory_cloze', 'hi'),90    ('Muennighoff/xstory_cloze', 'te'),91    ('Muennighoff/xstory_cloze', 'sw'),92    ('Muennighoff/xstory_cloze', 'zh'),93    ('hellaswag', None),94    ('super_glue', 'copa'),95    # Multilingual96    ('Muennighoff/xwinograd','en'),97    ('Muennighoff/xwinograd','fr'),98    ('Muennighoff/xwinograd','pt'),99    ('Muennighoff/xwinograd','zh'),100    ('clue', 'cluewsc2020'),101    ('xcopa','id'),102    ('xcopa','ta'),103    ('xcopa','sw'),104    ('xcopa','vi'),105    ('xcopa','zh'),106    ("xnli", "ar"),107    ("xnli", "en"),108    ("xnli", "es"),109    ("xnli", "fr"),110    ("xnli", "hi"),111    ("xnli", "sw"),112    ("xnli", "ur"),113    ("xnli", "vi"),114    ("xnli", "zh"),115    ("openai_humaneval", None),116    ("multi_eurlex", "all_languages")117]118 119ADD_TRAIN_DATASETS_L1_BLOOMZZ = [120    ('super_glue','wsc.fixed'),121    ('winogrande','winogrande_xl'),122    ('story_cloze', '2016'),123    ('Muennighoff/xstory_cloze', 'ar'),124    ('Muennighoff/xstory_cloze', 'es'),125    ('Muennighoff/xstory_cloze', 'eu'),126    ('Muennighoff/xstory_cloze', 'id'),127    ('Muennighoff/xstory_cloze', 'hi'),128    ('Muennighoff/xstory_cloze', 'te'),129    ('Muennighoff/xstory_cloze', 'sw'),130    ('Muennighoff/xstory_cloze', 'zh'),131    ('hellaswag', None),132    ('super_glue', 'copa'),133    # Multilingual134    ('Muennighoff/xwinograd','en'),135    ('Muennighoff/xwinograd','fr'),136    ('Muennighoff/xwinograd','pt'),137    ('Muennighoff/xwinograd','zh'),138    ('clue', 'cluewsc2020'),139    ('xcopa','id'),140    ('xcopa','ta'),141    ('xcopa','sw'),142    ('xcopa','vi'),143    ('xcopa','zh'),144    ("multi_eurlex", "all_languages")145    # ("openai_humaneval", None), # Low quality prompts146]147 148EVAL_DATASETS_L2 = [149    ('Muennighoff/xwinograd','jp'),150    ('Muennighoff/xwinograd','ru'),151    ('xcopa','et'),152    ('xcopa','ht'),153    ('xcopa','it'),154    ('xcopa','qu'),155    ('xcopa','th'),156    ('xcopa','tr'),157    ("xnli", "bg"),158    ("xnli", "de"),159    ("xnli", "el"),160    ("xnli", "ru"),161    ("xnli", "th"),162    ("xnli", "tr"),163]164 165TRAIN_DATASETS = [166    # English-only167    ('glue','mrpc'), 168    ('glue','qqp'),169    ('paws','labeled_final'),170    ('ai2_arc','ARC-Challenge'),171    ('ai2_arc','ARC-Easy'),172    ('kilt_tasks','hotpotqa'),173    ('trivia_qa','unfiltered'),174    ('web_questions',None),175    ('wiki_qa',None),176    ('adversarial_qa','dbidaf'),177    ('adversarial_qa','dbert'),178    ('adversarial_qa','droberta'),179    ('duorc','SelfRC'),180    ('duorc','ParaphraseRC'),181    ('ropes',None),182    ('squad_v2',None),183    ('super_glue','record'),184    ('quoref',None),185    ('cos_e','v1.11'),186    ('cosmos_qa',None),187    ('dream',None),188    ('openbookqa','main'),189    ('qasc',None),190    ('quail',None),191    ('quarel',None),192    ('quartz',None),193    ('race','high'),194    ('race','middle'),195    ('sciq',None),196    ('social_i_qa',None),197    ('super_glue','boolq'),198    ('super_glue','multirc'),199    ('wiki_hop','original'),200    ('wiqa',None),201    ('piqa',None),202    ('amazon_polarity',None),203    ('app_reviews',None),204    ('imdb',None),205    ('rotten_tomatoes',None),206    ('yelp_review_full',None),207    ('common_gen',None),208    ('wiki_bio',None),209    ('cnn_dailymail','3.0.0'),210    ('gigaword',None),211    ('multi_news',None),212    ('samsum',None),213    ('xsum',None),214    ('ag_news',None),215    ('dbpedia_14',None),216    ('trec',None),217    # Multilingual218    ('GEM/wiki_lingua', 'ar'),219    ('GEM/wiki_lingua', 'en'),220    ('GEM/wiki_lingua', 'es'),221    ('GEM/wiki_lingua', 'fr'),222    ('GEM/wiki_lingua', 'hi'),223    ('GEM/wiki_lingua', 'id'),224    ('GEM/wiki_lingua', 'pt'),225    ('GEM/wiki_lingua', 'vi'),226    ('GEM/wiki_lingua', 'zh'),227    ('Helsinki-NLP/tatoeba_mt', 'ara-eng'),228    ('Helsinki-NLP/tatoeba_mt', 'ara-fra'),229    ('Helsinki-NLP/tatoeba_mt', 'ara-spa'),230    ('Helsinki-NLP/tatoeba_mt', 'ben-eng'),231    ('Helsinki-NLP/tatoeba_mt', 'cat-eng'),232    ('Helsinki-NLP/tatoeba_mt', 'cat-fra'),233    ('Helsinki-NLP/tatoeba_mt', 'cat-por'),234    ('Helsinki-NLP/tatoeba_mt', 'cat-spa'),235    ('Helsinki-NLP/tatoeba_mt', 'eng-cmn_Hans'),236    ('Helsinki-NLP/tatoeba_mt', 'eng-cmn_Hant'),237    ('Helsinki-NLP/tatoeba_mt', 'eng-eus'),238    ('Helsinki-NLP/tatoeba_mt', 'eng-fra'),239    ('Helsinki-NLP/tatoeba_mt', 'eng-hin'),240    ('Helsinki-NLP/tatoeba_mt', 'eng-ind'),241    ('Helsinki-NLP/tatoeba_mt', 'eng-mal'),242    ('Helsinki-NLP/tatoeba_mt', 'eng-mar'),243    ('Helsinki-NLP/tatoeba_mt', 'eng-por'),244    ('Helsinki-NLP/tatoeba_mt', 'eng-run'),245    ('Helsinki-NLP/tatoeba_mt', 'eng-spa'),246    ('Helsinki-NLP/tatoeba_mt', 'eng-swa'),247    ('Helsinki-NLP/tatoeba_mt', 'eng-tam'),248    ('Helsinki-NLP/tatoeba_mt', 'eng-tel'),249    ('Helsinki-NLP/tatoeba_mt', 'eng-urd'),250    ('Helsinki-NLP/tatoeba_mt', 'eng-vie'),251    ('Helsinki-NLP/tatoeba_mt', 'eng-zho'),252    ('Helsinki-NLP/tatoeba_mt', 'eus-spa'),253    ('Helsinki-NLP/tatoeba_mt', 'fra-cmn_Hans'),254    ('Helsinki-NLP/tatoeba_mt', 'fra-cmn_Hant'),255    ('Helsinki-NLP/tatoeba_mt', 'fra-ind'),256    ('Helsinki-NLP/tatoeba_mt', 'fra-por'),257    ('Helsinki-NLP/tatoeba_mt', 'fra-run'),258    ('Helsinki-NLP/tatoeba_mt', 'fra-spa'),259    ('Helsinki-NLP/tatoeba_mt', 'fra-vie'),260    ('Helsinki-NLP/tatoeba_mt', 'fra-zho'),261    ('Helsinki-NLP/tatoeba_mt', 'hin-urd'),262    ('Helsinki-NLP/tatoeba_mt', 'hin-zho'),263    ('Helsinki-NLP/tatoeba_mt', 'por-cmn_Hans'),264    ('Helsinki-NLP/tatoeba_mt', 'por-cmn_Hant'),265    ('Helsinki-NLP/tatoeba_mt', 'por-spa'),266    ('Helsinki-NLP/tatoeba_mt', 'por-zho'),267    ('Helsinki-NLP/tatoeba_mt', 'run-spa'),268    ('Helsinki-NLP/tatoeba_mt', 'spa-cmn_Hans'),269    ('Helsinki-NLP/tatoeba_mt', 'spa-cmn_Hant'),270    ('Helsinki-NLP/tatoeba_mt', 'spa-vie'),271    ('Helsinki-NLP/tatoeba_mt', 'spa-zho'),272    ('Helsinki-NLP/tatoeba_mt', 'vie-cmn_Hans'),273    ('Helsinki-NLP/tatoeba_mt', 'vie-zho'),274    ('xquad', 'xquad.ar'),275    ('xquad', 'xquad.zh'),276    ('xquad', 'xquad.vi'),277    ('xquad', 'xquad.en'),278    ('xquad', 'xquad.es'),279    ('xquad', 'xquad.hi'),280    ('mlqa', 'mlqa.ar.ar'),281    ('mlqa', 'mlqa.vi.vi'),282    ('mlqa', 'mlqa.zh.zh'),283    ('mlqa', 'mlqa.es.es'),284    ('mlqa', 'mlqa.en.en'),285    ('mlqa', 'mlqa.hi.hi'),286 287    ('mlqa', 'mlqa.ar.vi'),288    ('mlqa', 'mlqa.ar.zh'),289    ('mlqa', 'mlqa.ar.es'),290    ('mlqa', 'mlqa.ar.en'),291    ('mlqa', 'mlqa.ar.hi'),292 293    ('mlqa', 'mlqa.vi.ar'),294    ('mlqa', 'mlqa.vi.zh'),295    ('mlqa', 'mlqa.vi.es'),296    ('mlqa', 'mlqa.vi.en'),297    ('mlqa', 'mlqa.vi.hi'),298 299    ('mlqa', 'mlqa.zh.ar'),300    ('mlqa', 'mlqa.zh.vi'),301    ('mlqa', 'mlqa.zh.es'),302    ('mlqa', 'mlqa.zh.en'),303    ('mlqa', 'mlqa.zh.hi'),304 305    ('mlqa', 'mlqa.es.ar'),306    ('mlqa', 'mlqa.es.vi'),307    ('mlqa', 'mlqa.es.zh'),308    ('mlqa', 'mlqa.es.en'),309    ('mlqa', 'mlqa.es.hi'),310 311    ('mlqa', 'mlqa.en.ar'),312    ('mlqa', 'mlqa.es.vi'),313    ('mlqa', 'mlqa.es.zh'),314    ('mlqa', 'mlqa.es.es'),315    ('mlqa', 'mlqa.es.hi'),316 317    ('mlqa', 'mlqa.hi.ar'),318    ('mlqa', 'mlqa.hi.vi'),319    ('mlqa', 'mlqa.hi.zh'),320    ('mlqa', 'mlqa.hi.es'),321    ('mlqa', 'mlqa.hi.en'),322 323    ('paws-x', 'en'),324    ('paws-x', 'es'),325    ('paws-x', 'fr'),326    ('paws-x', 'zh'),327    ('khalidalt/tydiqa-primary', 'arabic'),328    ('khalidalt/tydiqa-primary', 'bengali'),329    ('khalidalt/tydiqa-primary', 'english'),330    ('khalidalt/tydiqa-primary', 'indonesian'),331    ('khalidalt/tydiqa-primary', 'swahili'),332    ('khalidalt/tydiqa-primary', 'telugu'),333    ('khalidalt/tydiqa-goldp', 'arabic'),334    ('khalidalt/tydiqa-goldp', 'bengali'),335    ('khalidalt/tydiqa-goldp', 'english'),336    ('khalidalt/tydiqa-goldp', 'indonesian'),337    ('khalidalt/tydiqa-goldp', 'swahili'),338    ('khalidalt/tydiqa-goldp', 'telugu'),339    ('Muennighoff/mbpp', 'sanitized'),340    ("great_code", None),341    ("neural_code_search", "evaluation_dataset"),342    ("codeparrot/codecomplex", "codeparrot--codecomplex"),343    ("codeparrot/github-jupyter-text-code-pairs", None),344    ("codeparrot/apps", "all"),345    ("codeparrot/xlcost-text-to-code", "Python-program-level"),346    ("codeparrot/xlcost-text-to-code", "C-program-level"),347    ("codeparrot/xlcost-text-to-code", "C++-program-level"),348    ("codeparrot/xlcost-text-to-code", "Csharp-program-level"),349    ("codeparrot/xlcost-text-to-code", "Java-program-level"),350    ("codeparrot/xlcost-text-to-code", "Javascript-program-level"),351    ("codeparrot/xlcost-text-to-code", "PHP-program-level"),352    ("teven/code_contests", None),353    ("teven/code_docstring_corpus", "top_level"),354    ("Fraser/python-state-changes", None),355    ('clue', 'c3'),356    ('clue', 'cmrc2018'),357    ('clue', 'csl'),358    ('clue', 'drcd'),359    ('clue', 'tnews'),360    ('super_glue', 'wic'),361    ('pasinit/xlwic', "xlwic_en_zh"),362    ('pasinit/xlwic', "xlwic_fr_fr"),363    ('GEM/BiSECT', "en"),364    ('GEM/BiSECT', "es"),365    ('GEM/BiSECT', "fr"),366    ('GEM/xlsum', "arabic"),367    ('GEM/xlsum', "bengali"),368    ('GEM/xlsum', "chinese_simplified"),369    ('GEM/xlsum', "chinese_traditional"),370    ('GEM/xlsum', "english"),371    ('GEM/xlsum', "french"),372    ('GEM/xlsum', "gujarati"),373    ('GEM/xlsum', "hindi"),374    ('GEM/xlsum', "igbo"),375    ('GEM/xlsum', "indonesian"),376    ('GEM/xlsum', "kirundi"),377    ('GEM/xlsum', "marathi"),378    ('GEM/xlsum', "nepali"),379    ('GEM/xlsum', "portuguese"),380    ('GEM/xlsum', "punjabi"),381    ('GEM/xlsum', "spanish"),382    ('GEM/xlsum', "swahili"),383    ('GEM/xlsum', "tamil"),384    ('GEM/xlsum', "telugu"),385    ('GEM/xlsum', "urdu"),386    ('GEM/xlsum', "vietnamese"),387    ('GEM/xlsum', "yoruba"),388    # flores200, wmt & more wikilingua added below389]390 391FLORES_LANGS = [392    ("Acehnese (Arabic script)", "ace_Arab"),393    ("Acehnese (Latin script)", "ace_Latn"),394    ("Mesopotamian Arabic", "acm_Arab"),395    ("Ta’izzi-Adeni Arabic", "acq_Arab"),396    ("Tunisian Arabic", "aeb_Arab"),397    ("Afrikaans", "afr_Latn"),398    ("South Levantine Arabic", "ajp_Arab"),399    ("Akan", "aka_Latn"),400    ("Amharic", "amh_Ethi"),401    ("North Levantine Arabic", "apc_Arab"),402    ("Modern Standard Arabic", "arb_Arab"),403    ("Modern Standard Arabic (Romanized)", "arb_Latn"),404    ("Najdi Arabic", "ars_Arab"),405    ("Moroccan Arabic", "ary_Arab"),406    ("Egyptian Arabic", "arz_Arab"),407    ("Assamese", "asm_Beng"),408    ("Asturian", "ast_Latn"),409    ("Awadhi", "awa_Deva"),410    ("Central Aymara", "ayr_Latn"),411    ("South Azerbaijani", "azb_Arab"),412    ("North Azerbaijani", "azj_Latn"),413    ("Bashkir", "bak_Cyrl"),414    ("Bambara", "bam_Latn"),415    ("Balinese", "ban_Latn"),416    ("Belarusian", "bel_Cyrl"),417    ("Bemba", "bem_Latn"),418    ("Bengali", "ben_Beng"),419    ("Bhojpuri", "bho_Deva"),420    ("Banjar (Arabic script)", "bjn_Arab"),421    ("Banjar (Latin script)", "bjn_Latn"),422    ("Standard Tibetan", "bod_Tibt"),423    ("Bosnian", "bos_Latn"),424    ("Buginese", "bug_Latn"),425    ("Bulgarian", "bul_Cyrl"),426    ("Catalan", "cat_Latn"),427    ("Cebuano", "ceb_Latn"),428    ("Czech", "ces_Latn"),429    ("Chokwe", "cjk_Latn"),430    ("Central Kurdish", "ckb_Arab"),431    ("Crimean Tatar", "crh_Latn"),432    ("Welsh", "cym_Latn"),433    ("Danish", "dan_Latn"),434    ("German", "deu_Latn"),435    ("Southwestern Dinka", "dik_Latn"),436    ("Dyula", "dyu_Latn"),437    ("Dzongkha", "dzo_Tibt"),438    ("Greek", "ell_Grek"),439    ("English", "eng_Latn"),440    ("Esperanto", "epo_Latn"),441    ("Estonian", "est_Latn"),442    ("Basque", "eus_Latn"),443    ("Ewe", "ewe_Latn"),444    ("Faroese", "fao_Latn"),445    ("Fijian", "fij_Latn"),446    ("Finnish", "fin_Latn"),447    ("Fon", "fon_Latn"),448    ("French", "fra_Latn"),449    ("Friulian", "fur_Latn"),450    ("Nigerian Fulfulde", "fuv_Latn"),451    ("Scottish Gaelic", "gla_Latn"),452    ("Irish", "gle_Latn"),453    ("Galician", "glg_Latn"),454    ("Guarani", "grn_Latn"),455    ("Gujarati", "guj_Gujr"),456    ("Haitian Creole", "hat_Latn"),457    ("Hausa", "hau_Latn"),458    ("Hebrew", "heb_Hebr"),459    ("Hindi", "hin_Deva"),460    ("Chhattisgarhi", "hne_Deva"),461    ("Croatian", "hrv_Latn"),462    ("Hungarian", "hun_Latn"),463    ("Armenian", "hye_Armn"),464    ("Igbo", "ibo_Latn"),465    ("Ilocano", "ilo_Latn"),466    ("Indonesian", "ind_Latn"),467    ("Icelandic", "isl_Latn"),468    ("Italian", "ita_Latn"),469    ("Javanese", "jav_Latn"),470    ("Japanese", "jpn_Jpan"),471    ("Kabyle", "kab_Latn"),472    ("Jingpho", "kac_Latn"),473    ("Kamba", "kam_Latn"),474    ("Kannada", "kan_Knda"),475    ("Kashmiri (Arabic script)", "kas_Arab"),476    ("Kashmiri (Devanagari script)", "kas_Deva"),477    ("Georgian", "kat_Geor"),478    ("Central Kanuri (Arabic script)", "knc_Arab"),479    ("Central Kanuri (Latin script)", "knc_Latn"),480    ("Kazakh", "kaz_Cyrl"),481    ("Kabiyè", "kbp_Latn"),482    ("Kabuverdianu", "kea_Latn"),483    ("Khmer", "khm_Khmr"),484    ("Kikuyu", "kik_Latn"),485    ("Kinyarwanda", "kin_Latn"),486    ("Kyrgyz", "kir_Cyrl"),487    ("Kimbundu", "kmb_Latn"),488    ("Northern Kurdish", "kmr_Latn"),489    ("Kikongo", "kon_Latn"),490    ("Korean", "kor_Hang"),491    ("Lao", "lao_Laoo"),492    ("Ligurian", "lij_Latn"),493    ("Limburgish", "lim_Latn"),494    ("Lingala", "lin_Latn"),495    ("Lithuanian", "lit_Latn"),496    ("Lombard", "lmo_Latn"),497    ("Latgalian", "ltg_Latn"),498    ("Luxembourgish", "ltz_Latn"),499    ("Luba-Kasai", "lua_Latn"),500    ("Ganda", "lug_Latn"),501    ("Luo", "luo_Latn"),502    ("Mizo", "lus_Latn"),503    ("Standard Latvian", "lvs_Latn"),504    ("Magahi", "mag_Deva"),505    ("Maithili", "mai_Deva"),506    ("Malayalam", "mal_Mlym"),507    ("Marathi", "mar_Deva"),508    ("Minangkabau (Arabic script)", "min_Arab"),509    ("Minangkabau (Latin script)", "min_Latn"),510    ("Macedonian", "mkd_Cyrl"),511    ("Plateau Malagasy", "plt_Latn"),512    ("Maltese", "mlt_Latn"),513    ("Meitei (Bengali script)", "mni_Beng"),514    ("Halh Mongolian", "khk_Cyrl"),515    ("Mossi", "mos_Latn"),516    ("Maori", "mri_Latn"),517    ("Burmese", "mya_Mymr"),518    ("Dutch", "nld_Latn"),519    ("Norwegian Nynorsk", "nno_Latn"),520    ("Norwegian Bokmål", "nob_Latn"),521    ("Nepali", "npi_Deva"),522    ("Northern Sotho", "nso_Latn"),523    ("Nuer", "nus_Latn"),524    ("Nyanja", "nya_Latn"),525    ("Occitan", "oci_Latn"),526    ("West Central Oromo", "gaz_Latn"),527    ("Odia", "ory_Orya"),528    ("Pangasinan", "pag_Latn"),529    ("Eastern Panjabi", "pan_Guru"),530    ("Papiamento", "pap_Latn"),531    ("Western Persian", "pes_Arab"),532    ("Polish", "pol_Latn"),533    ("Portuguese", "por_Latn"),534    ("Dari", "prs_Arab"),535    ("Southern Pashto", "pbt_Arab"),536    ("Ayacucho Quechua", "quy_Latn"),537    ("Romanian", "ron_Latn"),538    ("Rundi", "run_Latn"),539    ("Russian", "rus_Cyrl"),540    ("Sango", "sag_Latn"),541    ("Sanskrit", "san_Deva"),542    ("Santali", "sat_Olck"),543    ("Sicilian", "scn_Latn"),544    ("Shan", "shn_Mymr"),545    ("Sinhala", "sin_Sinh"),546    ("Slovak", "slk_Latn"),547    ("Slovenian", "slv_Latn"),548    ("Samoan", "smo_Latn"),549    ("Shona", "sna_Latn"),550    ("Sindhi", "snd_Arab"),551    ("Somali", "som_Latn"),552    ("Southern Sotho", "sot_Latn"),553    ("Spanish", "spa_Latn"),554    ("Tosk Albanian", "als_Latn"),555    ("Sardinian", "srd_Latn"),556    ("Serbian", "srp_Cyrl"),557    ("Swati", "ssw_Latn"),558    ("Sundanese", "sun_Latn"),559    ("Swedish", "swe_Latn"),560    ("Swahili", "swh_Latn"),561    ("Silesian", "szl_Latn"),562    ("Tamil", "tam_Taml"),563    ("Tatar", "tat_Cyrl"),564    ("Telugu", "tel_Telu"),565    ("Tajik", "tgk_Cyrl"),566    ("Tagalog", "tgl_Latn"),567    ("Thai", "tha_Thai"),568    ("Tigrinya", "tir_Ethi"),569    ("Tamasheq (Latin script)", "taq_Latn"),570    ("Tamasheq (Tifinagh script)", "taq_Tfng"),571    ("Tok Pisin", "tpi_Latn"),572    ("Tswana", "tsn_Latn"),573    ("Tsonga", "tso_Latn"),574    ("Turkmen", "tuk_Latn"),575    ("Tumbuka", "tum_Latn"),576    ("Turkish", "tur_Latn"),577    ("Twi", "twi_Latn"),578    ("Central Atlas Tamazight", "tzm_Tfng"),579    ("Uyghur", "uig_Arab"),580    ("Ukrainian", "ukr_Cyrl"),581    ("Umbundu", "umb_Latn"),582    ("Urdu", "urd_Arab"),583    ("Northern Uzbek", "uzn_Latn"),584    ("Venetian", "vec_Latn"),585    ("Vietnamese", "vie_Latn"),586    ("Waray", "war_Latn"),587    ("Wolof", "wol_Latn"),588    ("Xhosa", "xho_Latn"),589    ("Eastern Yiddish", "ydd_Hebr"),590    ("Yoruba", "yor_Latn"),591    ("Yue Chinese", "yue_Hant"),592    ("Chinese (Simplified)", "zho_Hans"),593    ("Chinese (Traditional)", "zho_Hant"),594    ("Standard Malay", "zsm_Latn"),595    ("Zulu", "zul_Latn"),596]597 598WMT22_LANGS = [599    ("afr", "eng"),600    ("afr", "som"),601    ("amh", "eng"),602    ("amh", "fra"),603    ("amh", "nya"),604    ("amh", "orm"),605    ("amh", "sna"),606    ("amh", "som"),607    ("amh", "ssw"),608    ("amh", "swh"),609    ("amh", "tsn"),610    ("amh", "tso"),611    ("amh", "umb"),612    ("amh", "xho"),613    ("amh", "yor"),614    ("amh", "zul"),615    ("eng", "fuv"),616    ("eng", "hau"),617    ("eng", "ibo"),618    ("eng", "kam"),619    ("eng", "kin"),620    ("eng", "lin"),621    ("eng", "lug"),622    ("eng", "luo"),623    ("eng", "nso"),624    ("eng", "nya"),625    ("eng", "orm"),626    ("eng", "sna"),627    ("eng", "som"),628    ("eng", "ssw"),629    ("eng", "swh"),630    ("eng", "tsn"),631    ("eng", "tso"),632    ("eng", "umb"),633    ("eng", "wol"),634    ("eng", "xho"),635    ("eng", "yor"),636    ("eng", "zul"),637    ("fra", "hau"),638    ("fra", "ibo"),639    ("fra", "kam"),640    ("fra", "kin"),641    ("fra", "lin"),642    ("fra", "lug"),643    ("fra", "luo"),644    ("fra", "nso"),645    ("fra", "nya"),646    ("fra", "orm"),647    ("fra", "som"),648    ("fra", "ssw"),649    ("fra", "swh"),650    ("fra", "tsn"),651    ("fra", "tso"),652    ("fra", "umb"),653    ("fra", "wol"),654    ("fra", "xho"),655    ("fra", "zul"),656    ("fuv", "hau"),657    ("fuv", "ibo"),658    ("fuv", "kam"),659    ("fuv", "kin"),660    ("fuv", "lug"),661    ("fuv", "luo"),662    ("fuv", "nso"),663    ("fuv", "nya"),664    ("fuv", "orm"),665    ("fuv", "sna"),666    ("fuv", "som"),667    ("fuv", "ssw"),668    ("fuv", "swh"),669    ("fuv", "tsn"),670    ("fuv", "tso"),671    ("fuv", "umb"),672    ("fuv", "xho"),673    ("fuv", "yor"),674    ("fuv", "zul"),675    ("hau", "ibo"),676    ("hau", "kam"),677    ("hau", "kin"),678    ("hau", "lug"),679    ("hau", "luo"),680    ("hau", "nso"),681    ("hau", "nya"),682    ("hau", "orm"),683    ("hau", "sna"),684    ("hau", "som"),685    ("hau", "ssw"),686    ("hau", "swh"),687    ("hau", "tsn"),688    ("hau", "tso"),689    ("hau", "umb"),690    ("hau", "xho"),691    ("hau", "yor"),692    ("hau", "zul"),693    ("ibo", "kam"),694    ("ibo", "kin"),695    ("ibo", "lug"),696    ("ibo", "luo"),697    ("ibo", "nso"),698    ("ibo", "nya"),699    ("ibo", "orm"),700    ("ibo", "sna"),701    ("ibo", "som"),702    ("ibo", "ssw"),703    ("ibo", "swh"),704    ("ibo", "tsn"),705    ("ibo", "tso"),706    ("ibo", "umb"),707    ("ibo", "xho"),708    ("ibo", "yor"),709    ("ibo", "zul"),710    ("kam", "kin"),711    ("kam", "lug"),712    ("kam", "luo"),713    ("kam", "nso"),714    ("kam", "nya"),715    ("kam", "orm"),716    ("kam", "sna"),717    ("kam", "som"),718    ("kam", "ssw"),719    ("kam", "swh"),720    ("kam", "tsn"),721    ("kam", "tso"),722    ("kam", "umb"),723    ("kam", "xho"),724    ("kam", "yor"),725    ("kam", "zul"),726    ("kin", "lug"),727    ("kin", "luo"),728    ("kin", "nso"),729    ("kin", "nya"),730    ("kin", "orm"),731    ("kin", "sna"),732    ("kin", "som"),733    ("kin", "ssw"),734    ("kin", "swh"),735    ("kin", "tsn"),736    ("kin", "tso"),737    ("kin", "umb"),738    ("kin", "xho"),739    ("kin", "yor"),740    ("kin", "zul"),741    ("lug", "luo"),742    ("lug", "nso"),743    ("lug", "nya"),744    ("lug", "orm"),745    ("lug", "sna"),746    ("lug", "som"),747    ("lug", "ssw"),748    ("lug", "swh"),749    ("lug", "tsn"),750    ("lug", "tso"),751    ("lug", "umb"),752    ("lug", "xho"),753    ("lug", "yor"),754    ("lug", "zul"),755    ("luo", "nso"),756    ("luo", "nya"),757    ("luo", "orm"),758    ("luo", "sna"),759    ("luo", "som"),760    ("luo", "ssw"),761    ("luo", "swh"),762    ("luo", "tsn"),763    ("luo", "tso"),764    ("luo", "umb"),765    ("luo", "xho"),766    ("luo", "yor"),767    ("luo", "zul"),768    ("nso", "nya"),769    ("nso", "orm"),770    ("nso", "sna"),771    ("nso", "som"),772    ("nso", "ssw"),773    ("nso", "swh"),774    ("nso", "tsn"),775    ("nso", "tso"),776    ("nso", "umb"),777    ("nso", "xho"),778    ("nso", "yor"),779    ("nso", "zul"),780    ("nya", "orm"),781    ("nya", "sna"),782    ("nya", "som"),783    ("nya", "ssw"),784    ("nya", "swh"),785    ("nya", "tsn"),786    ("nya", "tso"),787    ("nya", "umb"),788    ("nya", "xho"),789    ("nya", "yor"),790    ("nya", "zul"),791    ("orm", "sna"),792    ("orm", "som"),793    ("orm", "ssw"),794    ("orm", "swh"),795    ("orm", "tsn"),796    ("orm", "tso"),797    ("orm", "umb"),798    ("orm", "xho"),799    ("orm", "yor"),800    ("orm", "zul"),801    ("sna", "som"),802    ("sna", "ssw"),803    ("sna", "swh"),804    ("sna", "tsn"),805    ("sna", "tso"),806    ("sna", "umb"),807    ("sna", "xho"),808    ("sna", "yor"),809    ("sna", "zul"),810    ("som", "ssw"),811    ("som", "swh"),812    ("som", "tsn"),813    ("som", "tso"),814    ("som", "umb"),815    ("som", "wol"),816    ("som", "xho"),817    ("som", "yor"),818    ("som", "zul"),819    ("ssw", "swh"),820    ("ssw", "tsn"),821    ("ssw", "tso"),822    ("ssw", "umb"),823    ("ssw", "xho"),824    ("ssw", "yor"),825    ("ssw", "zul"),826    ("swh", "tsn"),827    ("swh", "tso"),828    ("swh", "umb"),829    ("swh", "xho"),830    ("swh", "yor"),831    ("swh", "zul"),832    ("tsn", "tso"),833    ("tsn", "umb"),834    ("tsn", "xho"),835    ("tsn", "yor"),836    ("tsn", "zul"),837    ("tso", "umb"),838    ("tso", "xho"),839    ("tso", "yor"),840    ("tso", "zul"),841    ("umb", "xho"),842    ("umb", "yor"),843    ("umb", "zul"),844    ("xho", "yor"),845    ("xho", "zul"),846    ("yor", "zul"),847]848 849# Copied from metadata850BLOOM_LANGS = """851- ak852- ar853- as854- bm855- bn856- ca857- code858- en859- es860- eu861- fon862- fr863- gu864- hi865- id866- ig867- ki868- kn869- lg870- ln871- ml872- mr873- ne874- nso875- ny876- or877- pa878- pt879- rn880- rw881- sn882- st883- sw884- ta885- te886- tn887- ts888- tum889- tw890- ur891- vi892- wo893- xh894- yo895- zh896- zu897"""898 899DS_TO_LANG = {900    'Muennighoff/mbpp': 'code',901    'openai_humaneval': 'code',902    "great_code": "code",903    "neural_code_search": "code",904    "codeparrot/codecomplex": "code",905    "codeparrot/github-jupyter-text-code-pairs": "code",906    "codeparrot/apps": "code",907    "Fraser/python-state-changes": "code",908    "codeparrot/xlcost-text-to-code": "code",909    "teven/code_contests": "code",910    "teven/code_docstring_corpus": "code",911    "clue": "zh",912    "cmn": "zh", # == zho913    "npi": "ne", # == npe914    "ory": "or", # == ori915    "swh": "sw", # == swa916    "kirundi": "rn", # == rundi917    "punjabi": "pa", # == panjabi918    "chinese_simplified": "zh",919    "chinese_traditional": "zh",920}921 922        923 924bloom_lang_codes_iso3 = []925bloom_lang_codes_iso2 = []926for lang in BLOOM_LANGS.split("\n")[1:-1]:927    iso2 = lang.replace("- ", "")928    DS_TO_LANG[iso2] = iso2929    try:930        name = languages.get(alpha2=iso2)931        DS_TO_LANG[name.name.lower()] = iso2932        # name is e.g. 'swahili (macrolanguage)' also add swahili933        DS_TO_LANG[name.name.lower().split(" ")[0]] = iso2934 935        iso3 = name.part3936        DS_TO_LANG[iso3] = iso2937    except KeyError:938        print(f"Could not find iso3 code for {lang}.")939 940# Add GEM multilingual941WIKILINGUA_LANGS = ["ar", "en", "es", "fr", "hi", "id", "pt", "vi", "zh"]942for l1_code in WIKILINGUA_LANGS:943    for l2_code in WIKILINGUA_LANGS:944        if l1_code == l2_code:945            continue946        TRAIN_DATASETS.append(("GEM/wiki_lingua", f"{l1_code}_{l2_code}"))947 948# Add flores200949for (l1_name, l1_code) in FLORES_LANGS:950    for (l2_name, l2_code) in FLORES_LANGS:951        if l1_code.split("_")[0] not in DS_TO_LANG or l2_code.split("_")[0] not in DS_TO_LANG:952            print(f"Skipping as {l1_name} or {l2_name} was not pre-trained on.")953            continue954        elif l1_name == l2_name:955            continue956        TRAIN_DATASETS.append(("facebook/flores", f"{l1_code}-{l2_code}"))957 958# Add wmt22959for (l1_code, l2_code) in WMT22_LANGS:960    if l1_code not in DS_TO_LANG or l2_code not in DS_TO_LANG:961        print(f"Skipping as {l1_code} or {l2_code} was not pre-trained on.")962        continue963    elif l1_code == l2_code:964        continue965    TRAIN_DATASETS.append(("allenai/wmt22_african", f"{l1_code}-{l2_code}"))966 967 968### DATASET CREATION ###969 970 971# Copied from promptsource.utils972def removeHyphen(example):973    example_clean = {}974    for key in example.keys():975        if "-" in key:976            new_key = key.replace("-", "_")977            example_clean[new_key] = example[key]978        else:979            example_clean[key] = example[key]980    example = example_clean981    return example982 983def apply_template(dataset, template, strip_connection=True):984    def map_fn(ex):985        ex = removeHyphen(ex)986        try:987            inputs_and_targets = template.apply(988                ex, 989                strip_connection=strip_connection,990                truncate=True,991            )992        # Skip ValueError("Prompt did not produce an input and at least one target.")993        # which happens for some prompts with if else clauses based on inputs producing occasional994        # empty targets995        except ValueError:996            return {"inputs": "", "targets": ""}997        if len(inputs_and_targets) == 2:998            # Note that the signature changed in promptsource 999            # In 0.1.0 template.apply returned two strings; In >0.3.0 it retuns a str & list1000            inputs, targets = inputs_and_targets1001            if len(targets) > 1:1002                # Safer to skip, as could be a bug1003                print(f"Found targets longer than 1. Inputs: {inputs} ; Targets {targets}. Skipping.")1004                return {"inputs": "", "targets": ""}1005            targets = targets[0]1006            return {"inputs": inputs, "targets": targets}1007        # When template results in an empty example, template.apply returns [""]1008        # Also, if the template gets split wrong, len can be > 21009        # We will filter these out later1010        else:1011            # inputs is a str by default & targets a str1012            return {"inputs": "", "targets": ""}1013 1014    def filter_fn(ex):1015        return len(ex["inputs"]) > 0 and len(ex["targets"]) > 01016 1017    original_columns = dataset.column_names1018    dataset = dataset.map(map_fn).filter(filter_fn)1019    # map keeps original columns, remove them1020    return dataset.remove_columns(set(original_columns) - {"inputs", "targets"})1021 1022def add_language_name_wikilingua(example):1023    example["source_language_name"] = languages.get(alpha2=example["source_language"]).name1024    example["target_language_name"] = languages.get(alpha2=example["target_language"]).name1025    return example1026 1027def filter_l1_l2_wikilingua(example, l1, l2):1028    return example["source_language"] == l1 and example["target_language"] == l21029 1030def filter_empty_solution_apps(example):1031    return bool(example["solutions"])1032 1033def add_solution_apps(example):1034    example["solution"] = random.choice(json.loads(example["solutions"]))1035    return example1036 1037def clean_code_xlcost(example):1038    clean_lines = []1039    cur_indent = 01040    for line in example["code"].split("NEW_LINE"):1041        cur_indent += line.count("INDENT")1042        cur_indent -= line.count("DEDENT")1043        line = line.replace("INDENT", "").replace("DEDENT", "")1044        line = line.replace("STRNEWLINE", "\n")1045        line = line.replace("TABSYMBOL", "\t")1046        clean_lines.append("\t" * cur_indent + line.strip())1047    example["code_clean"] = "\n".join(clean_lines)1048    return example1049 1050def write_to_jsonl_hub(ds, split="train"):1051 1052    ### GET DATASET & LANGUAGE ###1053 1054    ds_name, subset_name = ds1055 1056    is_wikilingua_cross_lingual = (ds_name == "GEM/wiki_lingua") and ("_") in subset_name1057    1058    lang_dir = DS_TO_LANG.get(ds_name, None)1059    if lang_dir is None:1060        lang_dir = DS_TO_LANG.get(subset_name, "en")1061    if ds_name == "facebook/flores":1062        lang_dir = DS_TO_LANG.get(subset_name.split("-")[-1].split("_")[0])1063    elif is_wikilingua_cross_lingual or ds_name == "pasinit/xlwic":1064        lang_dir = DS_TO_LANG.get(subset_name.split("_")[-1])1065    elif ds_name == "xquad":1066        lang_dir = DS_TO_LANG.get(subset_name.split(".")[1])1067    elif ds_name == "mlqa":1068        # Classify it by the target language for cross-lingual (i.e. what the loss is computed on)1069        lang_dir = DS_TO_LANG.get(subset_name.split(".")[1])1070    os.makedirs(lang_dir, exist_ok=True)1071 1072    if ds_name == "Helsinki-NLP/tatoeba_mt":1073        ds = load_dataset(ds_name, subset_name, ignore_verifications=True, revision="49aa20ac768eabc5a106a123549ea58053fc9b40")1074    elif ds_name == "story_cloze":   1075        ds = load_dataset(ds_name, subset_name, data_dir=STORY_CLOZE_DIR)1076    elif ds_name == "Muennighoff/xstory_cloze":1077        ds = load_dataset(ds_name, subset_name, data_dir=XSTORY_CLOZE_DIR)1078    else:1079        ds = load_dataset(ds_name, subset_name)1080 1081    if ds_name == "GEM/wiki_lingua":1082        # Add names, e.g. Chinese for zh to use them in the jinja prompts1083        ds = ds.map(add_language_name_wikilingua)1084        if is_wikilingua_cross_lingual:1085            # Keep only L1 -> L2 (L2 -> L1 will be a separate dataset)1086            ds = ds.filter(partial(filter_l1_l2_wikilingua, l1=subset_name.split("_")[0], l2=subset_name.split("_")[1]))1087    elif ds_name == "codeparrot/apps":1088        ds = ds.filter(filter_empty_solution_apps).map(add_solution_apps)1089    elif ds_name == "codeparrot/xlcost-text-to-code":1090        ds = ds.map(clean_code_xlcost)1091 1092    ### SELECT SPLITS ###1093 1094    dataset_splits = list(ds.keys())1095    if subset_name == "xlwic_en_zh":1096        # Train set is en; val & test are zh1097        dataset_splits.remove("train")1098    elif ds_name == "teven/code_docstring_corpus":1099        # Bad quality split1100        dataset_splits.remove("class_level")1101 1102    if split == "validation":1103        if split not in dataset_splits or len(dataset_splits) == 1:1104            print(f"Validation not found for {ds_name}")1105            return1106        dataset_splits = ["validation"]1107    elif split == "train":1108        # Use as much as possible1109        # Would need to remove e.g. test datasets to benchmark same task performance1110        if len(dataset_splits) > 1 and "validation" in dataset_splits:1111            dataset_splits.remove("validation")1112        # WikiLingua1113        if "sampled_validation" in dataset_splits:1114            dataset_splits.remove("sampled_validation")1115        if "sampled_test" in dataset_splits:1116            dataset_splits.remove("sampled_test")1117 1118    ### SELECT PROMPTS ###1119 1120    if subset_name is None:1121        prompt_dataset_name = ds_name1122    else:1123        subset_name_prompt = subset_name1124        if USE_ENGLISH_PROMPTS and ds_name in DS_TO_ENG_PROMPT:1125            subset_name_prompt = DS_TO_ENG_PROMPT[ds_name]1126        prompt_dataset_name = f"{ds_name}/{subset_name_prompt}"1127 1128    prompts = DatasetTemplates(prompt_dataset_name)1129 1130    ### PROCESS ###1131 1132    for split in dataset_splits:1133        for t_name in prompts.all_template_names:1134            #if not "mt" in t_name:1135            #    print(f"Skipping {t_name}")1136            #    continue1137            print(f"Running {ds_name}/{subset_name}/{split}/{t_name}")1138            if SKIP_PROMPTS.get(prompt_dataset_name, {}).get(split, False):1139                if ("all" in SKIP_PROMPTS[prompt_dataset_name][split]) or (t_name in SKIP_PROMPTS[prompt_dataset_name][split]):1140                    print(f"Skipping DS: {prompt_dataset_name} Split {split} Prompt {t_name}")1141                    continue1142 1143            if ds_name == "Helsinki-NLP/tatoeba_mt":1144                # E.g. translate-this-ara-eng, where eng is the target1145                lang_dir = DS_TO_LANG.get(t_name.split("-")[-1].split("_")[0], "en")1146            elif ds_name in ("allenai/wmt22_african", "multi_eurlex"):1147                # One prompt in multi_eurlex has -source+target appended to the languages1148                lang_dir = DS_TO_LANG.get(t_name.replace("-source+target", "").split("-")[-1])1149 1150            out_path = os.path.join(1151                lang_dir, 1152                f'xp3_{ds_name}_{subset_name}_{split}_{t_name}.jsonl'.replace("/", "_").replace(" ", "_")1153            )1154            if os.path.exists(out_path):1155                print("Skipping as exists: ", out_path)1156                continue1157            1158            assert len(ds[split]) > 0, f"Got empty: {ds_name}"1159 1160            try:1161                if ds_name == "allenai/wmt22_african":1162                    # Sort by laser score, i.e. by increasing confidence & limit samples due to mediocre quality1163                    ds[split] = ds[split].sort("laser_score", reverse=True)1164                    max_range = min(len(ds[split]), MAX_EXAMPLES_PER_DATASET_PROMPT // 2)1165                else:1166                    # Allow 5x buffer for empty examples1167                    max_range = min(len(ds[split]), MAX_EXAMPLES_PER_DATASET_PROMPT * 5)1168                # Shuffle to avoid using the same subset1169                # Leave \n in-between input & targets for code1170                out_ds = apply_template(1171                    dataset=ds[split].shuffle().select(list(range(max_range))), 1172                    template=prompts[t_name],1173                    strip_connection=False if lang_dir == "code" else True1174                )1175                # Keep X shortest examples1176                max_range = min(len(out_ds), MAX_EXAMPLES_PER_DATASET_PROMPT)1177                out_ds = out_ds.sort("inputs").select(list(range(max_range)))1178            except Exception as e:1179                print(f"Skipping due to {e}. DS: {ds_name}/{subset_name} Template: {t_name}")1180                continue1181            # Do not force ascii to allow chars like é1182            if len(out_ds) > 0:1183                out_ds.to_json(out_path, orient="records", lines=True, force_ascii=False)1184 1185# Testing:1186TRAIN_DATASETS = [1187    ('xquad', 'xquad.ar'),1188    ('xquad', 'xquad.vi'),1189    ('xquad', 'xquad.en'),1190    ('xquad', 'xquad.es'),1191    ('xquad', 'xquad.hi'),1192    ('mlqa', 'mlqa.ar.ar'),1193    ('mlqa', 'mlqa.vi.vi'),1194    ('mlqa', 'mlqa.zh.zh'),1195    ('mlqa', 'mlqa.es.es'),1196    ('mlqa', 'mlqa.en.en'),1197    ('mlqa', 'mlqa.hi.hi'),1198 1199    ('mlqa', 'mlqa.ar.vi'),1200    ('mlqa', 'mlqa.ar.zh'),

Showing the first 1,200 of 1284 lines. Download the file for the rest.