CoolFace
Apppublic

bigscience-data/bigscience-corpus

sourceHugging Faceapache-2.0updated 3y agoView on Hugging Face
6likes
sources_with_info_cards.json14707 linesDownload Raw Back to resources
1[2  [3    "wudaocorpora",4    {5      "languages": [6        {7          "ln_code": "zh",8          "dataset_name": "lm_zh_wudaocorpora",9          "size": 332.029883935,10          "--filters": "",11          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",12          "--maps-and-filters argument": "",13          "--filter-short-documents": "filter_small_docs_bytes_1024"14        }15      ],16      "total": 332.029883935,17      "hf_info": {18        "description": "WuDaoCorpora is a large-scale and high-quality data set constructed by Beijing Academy of Artificial Intelligence\n(BAAI), which is used to support the research of Wudao large pre-training model. Since the first release of\nWuDaoCorpora on 20th March , 2021, it has immediately attracted the attention of the industry. At present, 61 research\nteams from 41 enterprises, colleges and academies have applied for use.\nWuDaoCorpora2.0 contains the world\u2019s largest text data set, the world\u2019s largest multi-modal data set and the world\u2019s\nlargest Chinese dialogue data set. They are aiming at condensing Chinese language environment, establishing connection\nbetween text and image and concluding core rules of dialogue respectively. Our data set is multi-dimensional and\nworld-class, which is beneficial for the development of artificial general intelligence in China.\nWe open 200GB WDC-text data for academic research, while the full amount of WDC-text, WDC-ImageCaption and WDC-Dialogue\ndatasets only for WuDao team.\n",19        "citation": "@article{YUAN202165,\n  title = {WuDaoCorpora: A super large-scale Chinese corpora for pre-training language models},\n  journal = {AI Open},\n  volume = {2},\n  pages = {65-68},\n  year = {2021},\n  issn = {2666-6510},\n  doi = {https://doi.org/10.1016/j.aiopen.2021.06.001},\n  url = {https://www.sciencedirect.com/science/article/pii/S2666651021000152},\n  author = {Sha Yuan and Hanyu Zhao and Zhengxiao Du and Ming Ding and Xiao Liu and Yukuo Cen and Xu Zou and Zhilin Yang and Jie Tang},\n  keywords = {Pre-trained language models, Chinese corpus, Transformer-XL},\n}\n",20        "license": "For academic research",21        "homepage": "https://data.wudaoai.cn/",22        "hf_id": "wu_dao_corpora"23      },24      "catalogue_info": {25        "uid": "wudaocorpora",26        "type": "processed",27        "description": {28          "name": "WuDaoCorpora",29          "description": "WuDaoCorpora is a super large-scale Chinese corpora for pre-training language models.\nThe base version of WuDaoCorpora contains about 200GB training data and 72 billion Chinese characters.",30          "homepage": "https://resource.wudaoai.cn/home",31          "validated": true32        },33        "languages": {34          "language_names": [35            "Chinese"36          ],37          "language_comments": "",38          "language_locations": [39            "Eastern Asia",40            "China"41          ],42          "validated": false43        },44        "custodian": {45          "name": "Beijing Academy of Artificial Intelligence",46          "in_catalogue": "",47          "type": "A university or research institution",48          "location": "China",49          "contact_name": "",50          "contact_email": "press@baai.ac.cn",51          "contact_submitter": false,52          "additional": "https://www.baai.ac.cn/",53          "validated": false54        },55        "availability": {56          "procurement": {57            "for_download": "Yes - after signing a user agreement",58            "download_url": "https://resource.wudaoai.cn/home",59            "download_email": ""60          },61          "licensing": {62            "has_licenses": "Yes",63            "license_text": "https://resource.wudaoai.cn/use-agreement",64            "license_properties": [65              "non-commercial use"66            ],67            "license_list": []68          },69          "pii": {70            "has_pii": "Unclear",71            "generic_pii_likely": "",72            "generic_pii_list": [],73            "numeric_pii_likely": "",74            "numeric_pii_list": [],75            "sensitive_pii_likely": "",76            "sensitive_pii_list": [],77            "no_pii_justification_class": "other",78            "no_pii_justification_text": "In the paper WuDaoCorpora: A Super Large-scale Chinese Corporafor Pre-training Language Models (https://ks3-cn-beijing.ksyun.com/resources/WuDaoCorpora/WuDaoCorpora__A_Super_Large_scale_Chinese_Corporafor_Pre_training_Language_Models.pdf), the author claims that \"To protect everyone\u2019s privacy security to the greatest extent, we use Regular Expression to match private information (i.e., identity number, phone number, qq number, email address, etc.) and remove them from the dataset.\""79          },80          "validated": false81        },82        "processed_from_primary": {83          "from_primary": "Taken from primary source",84          "primary_availability": "No - the dataset curators describe the primary sources but they are fully private",85          "primary_license": "",86          "primary_types": [],87          "validated": false,88          "from_primary_entries": []89        },90        "media": {91          "category": [92            "text"93          ],94          "text_format": [95            ".TXT"96          ],97          "audiovisual_format": [],98          "image_format": [],99          "database_format": [100            ".JSON"101          ],102          "text_is_transcribed": "No",103          "instance_type": "post",104          "instance_count": "1M<n<1B",105          "instance_size": "100<n<10,000",106          "validated": false107        },108        "fname": "wudaocorpora.json"109      },110      "data_card": "# WuDaoCorpora\n\n- Dataset uid: `wudaocorpora`\n\n### Description\n\nWuDaoCorpora is a super large-scale Chinese corpora for pre-training language models.\nThe base version of WuDaoCorpora contains about 200GB training data and 72 billion Chinese characters.\n\n### Homepage\n\nhttps://resource.wudaoai.cn/home\n\n### Licensing\n\n- non-commercial use\n\nhttps://resource.wudaoai.cn/use-agreement\n\n\n### Speaker Locations\n\n- Eastern Asia\n- China\n\n\n### Sizes\n\n- 27.3830 % of total\n- 95.7839 % of zh\n\n### BigScience processing steps\n\n#### Filters applied to: zh\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n- filter_small_docs_bytes_1024\n\n"111    }112  ],113  [114    "github-no-gpl",115    {116      "languages": [117        {118          "ln_code": "code",119          "dataset_name": "lm_code_github-no-gpl",120          "size": 159.294113344,121          "--filters": "",122          "--dedups": "dedup_document filter_remove_empty_docs",123          "--maps-and-filters argument": "",124          "--filter-short-documents": "filter_small_docs_bytes_1024"125        }126      ],127      "total": 159.294113344,128      "data_card": "# github-no-gpl\n\n- Dataset uid: `github-no-gpl`\n\n### Description\n\n\n\n- C++\n- C#\n- Go\n- Java\n- JavaScript\n- Lua\n- PHP\n- Python 2\n- Python 3\n- Ruby\n- Rust\n- Scala\n- TypeScript\n### Homepage\n\n\n\n### Licensing\n\n\n\n### Speaker Locations\n\n\n\n### Sizes\n\n- 13.1372 % of total\n- 85.2591 % of code\n\n### BigScience processing steps\n\n#### Filters applied to: code\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_1024\n --- \n - <1MB (there’s nothing bigger on BigQuery anyway) \n - whitespace agnostic deduplication \n - files with a line >1000 characters \n - exclude GPL \n - filter_token_len_avg_std \n - filter_text_len \n - filter_special_character_ratio \n - filter_longest_line \n - filter_by_all \n"129    }130  ],131  [132    "s2orc_ai2_pdf_parses",133    {134      "languages": [135        {136          "ln_code": "en",137          "dataset_name": "lm_en_s2orc_ai2_pdf_parses",138          "size": 155.9089287497206,139          "--filters": "",140          "--dedups": "dedup_document filter_remove_empty_docs",141          "--maps-and-filters argument": "",142          "--filter-short-documents": "filter_small_docs_bytes_1024"143        }144      ],145      "total": 155.9089287497206,146      "catalogue_info": {147        "uid": "s2orc_the_semantic_scholar_open_research_corpus",148        "type": "primary",149        "description": {150          "name": "S2ORC: The Semantic Scholar Open Research Corpus",151          "description": "Largest collection of machine-readable English-language open-access scientific literature formatted to support NLP research. 136M papers with titles and abstracts, including 12.7M papers with full text. Unifies popular resources like PubMed Central (Biomedicine) and arXiv (Physics, Math, CS) with papers sourced across many different academic disciplines. Maintained by the Semantic Scholar Research team at AI2. https://aclanthology.org/2020.acl-main.447/",152          "homepage": "https://github.com/allenai/s2orc",153          "validated": true154        },155        "languages": {156          "language_names": [157            "English"158          ],159          "language_comments": "",160          "language_locations": [161            "World-Wide"162          ],163          "validated": false164        },165        "custodian": {166          "name": "Semantic Scholar / Allen Institute for AI",167          "in_catalogue": "",168          "type": "A nonprofit/NGO (other)",169          "location": "United States of America",170          "contact_name": "Kyle Lo",171          "contact_email": "kylel@allenai.org",172          "contact_submitter": true,173          "additional": "http://allenai.org/",174          "validated": false175        },176        "availability": {177          "procurement": {178            "for_download": "Yes - after signing a user agreement",179            "download_url": "https://docs.google.com/forms/d/1fUqUw68dDMnzFt58WgMi-FI33MPcVFpflN2G3Yjfn9c/edit",180            "download_email": ""181          },182          "licensing": {183            "has_licenses": "Yes",184            "license_text": "",185            "license_properties": [186              "non-commercial use"187            ],188            "license_list": [189              "cc-by-nc-2.0: Creative Commons Attribution Non Commercial 2.0 Generic"190            ]191          },192          "pii": {193            "has_pii": "Yes",194            "generic_pii_likely": "very likely",195            "generic_pii_list": [196              "names",197              "email addresses",198              "physical addresses",199              "URLs",200              "website account name or handle"201            ],202            "numeric_pii_likely": "somewhat likely",203            "numeric_pii_list": [204              "telephone numbers"205            ],206            "sensitive_pii_likely": "unlikely",207            "sensitive_pii_list": [],208            "no_pii_justification_class": "",209            "no_pii_justification_text": ""210          },211          "validated": false212        },213        "source_category": {214          "category_type": "collection",215          "category_web": "",216          "category_media": "scientific articles/journal",217          "validated": false218        },219        "media": {220          "category": [221            "text"222          ],223          "text_format": [224            ".XHTML",225            ".TXT",226            ".CSV",227            ".TEX",228            "other",229            ".JSON"230          ],231          "audiovisual_format": [],232          "image_format": [233            "other",234            ".PDF"235          ],236          "database_format": [237            ".TAR",238            ".JSON",239            ".GZIP",240            ".TGZ"241          ],242          "text_is_transcribed": "Yes - image",243          "instance_type": "article",244          "instance_count": "1M<n<1B",245          "instance_size": "100<n<10,000",246          "validated": false247        },248        "fname": "s2orc_the_semantic_scholar_open_research_corpus.json"249      },250      "data_card": "# S2ORC: The Semantic Scholar Open Research Corpus\n\n- Dataset uid: `s2orc_ai2_pdf_parses`\n\n### Description\n\nLargest collection of machine-readable English-language open-access scientific literature formatted to support NLP research. 136M papers with titles and abstracts, including 12.7M papers with full text. Unifies popular resources like PubMed Central (Biomedicine) and arXiv (Physics, Math, CS) with papers sourced across many different academic disciplines. Maintained by the Semantic Scholar Research team at AI2. https://aclanthology.org/2020.acl-main.447/\n\n### Homepage\n\nhttps://github.com/allenai/s2orc\n\n### Licensing\n\n- non-commercial use\n- cc-by-nc-2.0: Creative Commons Attribution Non Commercial 2.0 Generic\n\n\n### Speaker Locations\n\n- World-Wide\n\n\n### Sizes\n\n- 12.8580 % of total\n- 69.6676 % of en\n\n### BigScience processing steps\n\n#### Filters applied to: en\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_1024\n\n"251    }252  ],253  [254    "hal_archives_ouvertes",255    {256      "languages": [257        {258          "ln_code": "fr",259          "dataset_name": "lm_fr_hal_archives_ouvertes",260          "size": 69.700827133,261          "--filters": "",262          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",263          "--maps-and-filters argument": "remove_references_fr",264          "--filter-short-documents": "filter_small_docs_bytes_1024"265        }266      ],267      "total": 69.700827133,268      "hf_info": {269        "description": "HAL is an open archive where authors can deposit scholarly documents from all academic fields.",270        "citation": "",271        "license": "Mixed",272        "homepage": "https://hal.archives-ouvertes.fr/",273        "hf_id": "hal"274      },275      "catalogue_info": {276        "uid": "hal_archives_ouvertes",277        "type": "primary",278        "description": {279          "name": "HAL archives ouvertes",280          "description": "HAL is an open archive where authors can deposit scholarly documents from all academic fields.\n\nFor the attention of the authors\n    The deposit of the full text should be made in agreement with the co-authors and in the respect for the policy of the publishers.\n    The deposit is subject of a control, HAL reserves the right to refuse items that do not meet the criteria of the archive.\n    Any deposit is definitive, no withdrawals will be made after the on-line posting of the publication.\n    Text files in pdf format or image files are sent to CINES for long-term archiving.\n\nFor the attention of the readers\n    In a context of electronic distribution, each author keeps their intellectual property rights.",281          "homepage": "https://hal.archives-ouvertes.fr/",282          "validated": true283        },284        "languages": {285          "language_names": [286            "English",287            "French",288            "Portuguese",289            "Spanish"290          ],291          "language_comments": "also has (much fewer) data in other BigScience languages",292          "language_locations": [293            "Europe"294          ],295          "validated": false296        },297        "custodian": {298          "name": "CNRS",299          "in_catalogue": "",300          "type": "A university or research institution",301          "location": "France",302          "contact_name": "HAL support",303          "contact_email": "hal.support@ccsd.cnrs.fr",304          "contact_submitter": false,305          "additional": "https://en.wikipedia.org/wiki/HAL_(open_archive)",306          "validated": false307        },308        "availability": {309          "procurement": {310            "for_download": "No - but the current owners/custodians have contact information for data queries",311            "download_url": "",312            "download_email": "hal.support@ccsd.cnrs.fr"313          },314          "licensing": {315            "has_licenses": "Yes",316            "license_text": "Moissonnage : conditions d\u2019utilisation des donn\u00e9es\n\nLes m\u00e9tadonn\u00e9es de HAL peuvent \u00eatre consult\u00e9es de fa\u00e7on totale ou partielle par moissonnage dans le respect du code de la propri\u00e9t\u00e9 intellectuelle.\nPas d\u2019utilisation commerciale des donn\u00e9es extraites.\nObligation de citer la source (exemple : hal.archives-ouvertes.fr/hal-00000001).",317            "license_properties": [318              "multiple licenses",319              "copyright - all rights reserved",320              "open license",321              "research use"322            ],323            "license_list": []324          },325          "pii": {326            "has_pii": "Yes",327            "generic_pii_likely": "somewhat likely",328            "generic_pii_list": [329              "names"330            ],331            "numeric_pii_likely": "unlikely",332            "numeric_pii_list": [],333            "sensitive_pii_likely": "very likely",334            "sensitive_pii_list": [335              "racial or ethnic origin",336              "political opinions",337              "religious or philosophical beliefs"338            ],339            "no_pii_justification_class": "",340            "no_pii_justification_text": ""341          },342          "validated": false343        },344        "source_category": {345          "category_type": "collection",346          "category_web": "",347          "category_media": "scientific articles/journal",348          "validated": false349        },350        "media": {351          "category": [352            "text"353          ],354          "text_format": [355            ".PDF"356          ],357          "audiovisual_format": [],358          "image_format": [],359          "database_format": [],360          "text_is_transcribed": "No",361          "instance_type": "article",362          "instance_count": "100K<n<1M",363          "instance_size": "100<n<10,000",364          "validated": false365        },366        "fname": "hal_archives_ouvertes.json"367      },368      "data_card": "# HAL archives ouvertes\n\n- Dataset uid: `hal_archives_ouvertes`\n\n### Description\n\nHAL is an open archive where authors can deposit scholarly documents from all academic fields.\n\nFor the attention of the authors\n    The deposit of the full text should be made in agreement with the co-authors and in the respect for the policy of the publishers.\n    The deposit is subject of a control, HAL reserves the right to refuse items that do not meet the criteria of the archive.\n    Any deposit is definitive, no withdrawals will be made after the on-line posting of the publication.\n    Text files in pdf format or image files are sent to CINES for long-term archiving.\n\nFor the attention of the readers\n    In a context of electronic distribution, each author keeps their intellectual property rights.\n\n### Homepage\n\nhttps://hal.archives-ouvertes.fr/\n\n### Licensing\n\n- multiple licenses\n- copyright - all rights reserved\n- open license\n- research use\n\nMoissonnage : conditions d\u2019utilisation des donn\u00e9es\n\nLes m\u00e9tadonn\u00e9es de HAL peuvent \u00eatre consult\u00e9es de fa\u00e7on totale ou partielle par moissonnage dans le respect du code de la propri\u00e9t\u00e9 intellectuelle.\nPas d\u2019utilisation commerciale des donn\u00e9es extraites.\nObligation de citer la source (exemple : hal.archives-ouvertes.fr/hal-00000001).\n\n\n### Speaker Locations\n\n- Europe\n\n\n### Sizes\n\n- 5.7483 % of total\n- 67.7196 % of fr\n\n### BigScience processing steps\n\n#### Filters applied to: fr\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n- remove_references_fr\n- filter_small_docs_bytes_1024\n\n"369    }370  ],371  [372    "arabic_billion_words",373    {374      "languages": [375        {376          "ln_code": "ar",377          "dataset_name": "lm_ar_arabic_billion_words",378          "size": 44.020136597,379          "--filters": "",380          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",381          "--maps-and-filters argument": "",382          "--filter-short-documents": "filter_small_docs_bytes_300"383        }384      ],385      "total": 44.020136597,386      "hf_info": {387        "description": "Abu El-Khair Corpus is an Arabic text corpus, that includes more than five million newspaper articles.\nIt contains over a billion and a half words in total, out of which, there are about three million unique words.\nThe corpus is encoded with two types of encoding, namely: UTF-8, and Windows CP-1256.\nAlso it was marked with two mark-up languages, namely: SGML, and XML.\n",388        "citation": "@article{el20161,\n  title={1.5 billion words arabic corpus},\n  author={El-Khair, Ibrahim Abu},\n  journal={arXiv preprint arXiv:1611.04033},\n  year={2016}\n}\n",389        "license": "",390        "homepage": "http://abuelkhair.net/index.php/en/arabic/abu-el-khair-corpus",391        "hf_id": "arabic_billion_words"392      },393      "data_card": "# arabic_billion_words\n\n- Dataset uid: `arabic_billion_words`\n\n### Description\n\nAbu El-Khair Corpus is an Arabic text corpus, that includes more than five million newspaper articles.\nIt contains over a billion and a half words in total, out of which, there are about three million unique words.\nThe corpus is encoded with two types of encoding, namely: UTF-8, and Windows CP-1256.\nAlso it was marked with two mark-up languages, namely: SGML, and XML.\n\n\n### Homepage\n\n- https://huggingface.co/datasets/arabic_billion_words\n- http://abuelkhair.net/index.php/en/arabic/abu-el-khair-corpus\n\n### Licensing\n\n\n\n### Speaker Locations\n\n\n\n### Sizes\n\n- 3.6304 % of total\n- 33.4538 % of ar\n\n### BigScience processing steps\n\n#### Filters applied to: ar\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n"394    }395  ],396  [397    "indic_nlp_corpus",398    {399      "languages": [400        {401          "ln_code": "indic-hi",402          "dataset_name": "lm_indic-hi_indic_nlp_corpus",403          "size": 12.276355316,404          "--filters": "",405          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",406          "--maps-and-filters argument": "",407          "--filter-short-documents": "filter_small_docs_bytes_300"408        },409        {410          "ln_code": "indic-ta",411          "dataset_name": "lm_indic-ta_indic_nlp_corpus",412          "size": 10.079553996,413          "--filters": "",414          "--dedups": "dedup_document filter_remove_empty_docs",415          "--maps-and-filters argument": "",416          "--filter-short-documents": "filter_small_docs_bytes_300"417        },418        {419          "ln_code": "indic-ml",420          "dataset_name": "lm_indic-ml_indic_nlp_corpus",421          "size": 5.094168328,422          "--filters": "",423          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",424          "--maps-and-filters argument": "",425          "--filter-short-documents": "filter_small_docs_bytes_300"426        },427        {428          "ln_code": "indic-te",429          "dataset_name": "lm_indic-te_indic_nlp_corpus",430          "size": 3.171844345,431          "--filters": "",432          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",433          "--maps-and-filters argument": "",434          "--filter-short-documents": "filter_small_docs_bytes_300"435        },436        {437          "ln_code": "indic-kn",438          "dataset_name": "lm_indic-kn_indic_nlp_corpus",439          "size": 2.298787095,440          "--filters": "",441          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",442          "--maps-and-filters argument": "",443          "--filter-short-documents": "filter_small_docs_bytes_300"444        },445        {446          "ln_code": "indic-mr",447          "dataset_name": "lm_indic-mr_indic_nlp_corpus",448          "size": 2.133954109,449          "--filters": "",450          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",451          "--maps-and-filters argument": "",452          "--filter-short-documents": "filter_small_docs_bytes_300"453        },454        {455          "ln_code": "indic-pa",456          "dataset_name": "lm_indic-pa_indic_nlp_corpus",457          "size": 2.040944194,458          "--filters": "",459          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",460          "--maps-and-filters argument": "",461          "--filter-short-documents": "filter_small_docs_bytes_300"462        },463        {464          "ln_code": "indic-or",465          "dataset_name": "lm_indic-or_indic_nlp_corpus",466          "size": 1.558714564,467          "--filters": "",468          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",469          "--maps-and-filters argument": "",470          "--filter-short-documents": ""471        },472        {473          "ln_code": "indic-gu",474          "dataset_name": "lm_indic-gu_indic_nlp_corpus",475          "size": 1.501293439,476          "--filters": "",477          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",478          "--maps-and-filters argument": "",479          "--filter-short-documents": "filter_small_docs_bytes_300"480        },481        {482          "ln_code": "indic-bn",483          "dataset_name": "lm_indic-bn_indic_nlp_corpus",484          "size": 1.094234656,485          "--filters": "",486          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",487          "--maps-and-filters argument": "",488          "--filter-short-documents": "filter_small_docs_bytes_300"489        }490      ],491      "total": 41.249850042,492      "hf_info": {493        "description": "The IndicNLP corpus is a largescale, general-domain corpus containing 2.7 billion words for 10 Indian languages from two language families. s (IndoAryan branch and Dravidian). Each language has at least 100 million words (except Oriya).\n",494        "citation": "",495        "license": "cc-by-nc-sa-4.0",496        "homepage": "https://github.com/AI4Bharat/indicnlp_corpus#publicly-available-classification-datasets",497        "hf_id": "indic_nlp_corpus"498      },499      "catalogue_info": {500        "uid": "indic_nlp_corpus",501        "type": "processed",502        "description": {503          "name": "Indic NLP Corpus",504          "description": "The IndicNLP corpus is a largescale, general-domain corpus containing 2.7 billion words for 10 Indian languages from two language families. s (IndoAryan branch and Dravidian). Each language has at least 100 million words (except Oriya).",505          "homepage": "https://github.com/AI4Bharat/indicnlp_corpus#publicly-available-classification-datasets",506          "validated": true507        },508        "languages": {509          "language_names": [510            "Indic",511            "Punjabi",512            "Hindi",513            "Bengali",514            "Odia",515            "Gujarati",516            "Marathi",517            "Kannada",518            "Telugu",519            "Malayalam",520            "Tamil"521          ],522          "language_comments": "",523          "language_locations": [524            "Southern Asia",525            "India"526          ],527          "validated": false528        },529        "custodian": {530          "name": "AI4Bharat",531          "in_catalogue": "",532          "type": "AI4Bharat is a voluntary community founded by faculty members at the Computer Science and Engineering department at Indian Institute of Technology, Madras.",533          "location": "India",534          "contact_name": "Indic NLP Corpus",535          "contact_email": "ankunchu@microsoft.com;  gokulnc@ai4bharat.org; gsatishkumaryadav@gmail.com; avikbhattacharyya.2k@gmail.com; divk@cse.iitm.ac.in; miteshk@cse.iitm.ac.in; pratyush@cse.iitm.ac.in",536          "contact_submitter": true,537          "additional": "https://ai4bharat.org/",538          "validated": false539        },540        "availability": {541          "procurement": {542            "for_download": "Yes - it has a direct download link or links",543            "download_url": "https://github.com/AI4Bharat/indicnlp_corpus#publicly-available-classification-datasets",544            "download_email": ""545          },546          "licensing": {547            "has_licenses": "Yes",548            "license_text": "",549            "license_properties": [550              "non-commercial use"551            ],552            "license_list": [553              "cc-by-nc-sa-4.0: Creative Commons Attribution Non Commercial Share Alike 4.0 International"554            ]555          },556          "pii": {557            "has_pii": "Yes",558            "generic_pii_likely": "somewhat likely",559            "generic_pii_list": [560              "names",561              "email addresses"562            ],563            "numeric_pii_likely": "somewhat likely",564            "numeric_pii_list": [565              "telephone numbers",566              "fax numbers"567            ],568            "sensitive_pii_likely": "somewhat likely",569            "sensitive_pii_list": [],570            "no_pii_justification_class": "",571            "no_pii_justification_text": ""572          },573          "validated": false574        },575        "processed_from_primary": {576          "from_primary": "Taken from primary source",577          "primary_availability": "Yes - their documentation/homepage/description is available",578          "primary_license": "Yes - the source material has an open license that allows re-use",579          "primary_types": [580            "web | news or magazine website",581            "web | wiki",582            "news articles",583            "web | content repository, archive, or collection"584          ],585          "validated": false,586          "from_primary_entries": []587        },588        "media": {589          "category": [590            "text"591          ],592          "text_format": [593            ".TXT"594          ],595          "audiovisual_format": [],596          "image_format": [],597          "database_format": [598            ".GZ"599          ],600          "text_is_transcribed": "No",601          "instance_type": "Each line of the dataset consists of a single sentence.",602          "instance_count": "1M<n<1B",603          "instance_size": "10<n<100",604          "validated": false605        },606        "fname": "indic_nlp_corpus.json"607      },608      "data_card": "# Indic NLP Corpus\n\n- Dataset uid: `indic_nlp_corpus`\n\n### Description\n\nThe IndicNLP corpus is a largescale, general-domain corpus containing 2.7 billion words for 10 Indian languages from two language families. s (IndoAryan branch and Dravidian). Each language has at least 100 million words (except Oriya).\n\n### Homepage\n\nhttps://github.com/AI4Bharat/indicnlp_corpus#publicly-available-classification-datasets\n\n### Licensing\n\n- non-commercial use\n- cc-by-nc-sa-4.0: Creative Commons Attribution Non Commercial Share Alike 4.0 International\n\n\n### Speaker Locations\n\n- Southern Asia\n- India\n\n\n### Sizes\n\n- 3.4019 % of total\n- 44.4368 % of indic-hi\n- 64.2943 % of indic-ta\n- 70.5374 % of indic-ml\n- 54.2394 % of indic-te\n- 55.9105 % of indic-kn\n- 61.6111 % of indic-mr\n- 67.2242 % of indic-pa\n- 68.1470 % of indic-or\n- 64.3879 % of indic-gu\n- 4.1495 % of indic-bn\n\n### BigScience processing steps\n\n#### Filters applied to: indic-hi\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-ta\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-ml\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-te\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-kn\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-mr\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-pa\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-or\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n\n#### Filters applied to: indic-gu\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-bn\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n"609    }610  ],611  [612    "wikipedia",613    {614      "languages": [615        {616          "ln_code": "en",617          "dataset_name": "lm_en_wikipedia",618          "size": 9.414965656,619          "--filters": "",620          "--dedups": "dedup_document filter_remove_empty_docs",621          "--maps-and-filters argument": "",622          "--filter-short-documents": "filter_small_docs_bytes_1024"623        },624        {625          "ln_code": "ar",626          "dataset_name": "lm_ar_wikipedia",627          "size": 7.470392905,628          "--filters": "filter_wiki_user_titles",629          "--dedups": "dedup_document filter_remove_empty_docs",630          "--maps-and-filters argument": "",631          "--filter-short-documents": "filter_small_docs_bytes_300"632        },633        {634          "ln_code": "fr",635          "dataset_name": "lm_fr_wikipedia",636          "size": 3.43939272,637          "--filters": "",638          "--dedups": "dedup_document filter_remove_empty_docs",639          "--maps-and-filters argument": "",640          "--filter-short-documents": "filter_small_docs_bytes_1024"641        },642        {643          "ln_code": "es",644          "dataset_name": "lm_es_wikipedia",645          "size": 2.690323165,646          "--filters": "",647          "--dedups": "dedup_document filter_remove_empty_docs",648          "--maps-and-filters argument": "",649          "--filter-short-documents": "filter_small_docs_bytes_1024"650        },651        {652          "ln_code": "ca",653          "dataset_name": "lm_ca_wikipedia",654          "size": 1.795324275,655          "--filters": "filter_wiki_user_titles",656          "--dedups": "dedup_document filter_remove_empty_docs",657          "--maps-and-filters argument": "",658          "--filter-short-documents": "filter_small_docs_bytes_1024"659        },660        {661          "ln_code": "zh",662          "dataset_name": "lm_zh_wikipedia",663          "size": 1.486580302664        },665        {666          "ln_code": "zh",667          "dataset_name": "lm_zh_wikipedia",668          "size": 1.485830189669        },670        {671          "ln_code": "indic-bn",672          "dataset_name": "lm_indic-bn_wikipedia",673          "size": 1.443588455,674          "--filters": "filter_wiki_user_titles",675          "--dedups": "dedup_document filter_remove_empty_docs",676          "--maps-and-filters argument": "",677          "--filter-short-documents": "filter_small_docs_bytes_300"678        },679        {680          "ln_code": "indic-ta",681          "dataset_name": "lm_indic-ta_wikipedia",682          "size": 1.396246399,683          "--filters": "filter_wiki_user_titles",684          "--dedups": "dedup_document filter_remove_empty_docs",685          "--maps-and-filters argument": "",686          "--filter-short-documents": "filter_small_docs_bytes_300"687        },688        {689          "ln_code": "indic-te",690          "dataset_name": "lm_indic-te_wikipedia",691          "size": 1.247426356,692          "--filters": "filter_wiki_user_titles",693          "--dedups": "dedup_document filter_remove_empty_docs",694          "--maps-and-filters argument": "",695          "--filter-short-documents": "filter_small_docs_bytes_300"696        },697        {698          "ln_code": "pt",699          "dataset_name": "lm_pt_wikipedia",700          "size": 1.175185446,701          "--filters": "",702          "--dedups": "dedup_document filter_remove_empty_docs",703          "--maps-and-filters argument": "",704          "--filter-short-documents": "filter_small_docs_bytes_300"705        },706        {707          "ln_code": "indic-hi",708          "dataset_name": "lm_indic-hi_wikipedia",709          "size": 1.118676659,710          "--filters": "filter_wiki_user_titles",711          "--dedups": "dedup_document filter_remove_empty_docs",712          "--maps-and-filters argument": "",713          "--filter-short-documents": "filter_small_docs_bytes_300"714        },715        {716          "ln_code": "indic-ml",717          "dataset_name": "lm_indic-ml_wikipedia",718          "size": 0.817256611,719          "--filters": "filter_wiki_user_titles",720          "--dedups": "dedup_document filter_remove_empty_docs",721          "--maps-and-filters argument": "",722          "--filter-short-documents": "filter_small_docs_bytes_300"723        },724        {725          "ln_code": "indic-ur",726          "dataset_name": "lm_indic-ur_wikipedia",727          "size": 0.772117731,728          "--filters": "filter_wiki_user_titles",729          "--dedups": "dedup_document filter_remove_empty_docs",730          "--maps-and-filters argument": "",731          "--filter-short-documents": "filter_small_docs_bytes_300"732        },733        {734          "ln_code": "vi",735          "dataset_name": "lm_vi_wikipedia",736          "size": 0.745160292,737          "--filters": "",738          "--dedups": "dedup_document filter_remove_empty_docs",739          "--maps-and-filters argument": "",740          "--filter-short-documents": "filter_small_docs_bytes_300"741        },742        {743          "ln_code": "indic-kn",744          "dataset_name": "lm_indic-kn_wikipedia",745          "size": 0.698618931,746          "--filters": "filter_wiki_user_titles",747          "--dedups": "dedup_document filter_remove_empty_docs",748          "--maps-and-filters argument": "",749          "--filter-short-documents": "filter_small_docs_bytes_300"750        },751        {752          "ln_code": "eu",753          "dataset_name": "lm_eu_wikipedia",754          "size": 0.488339118,755          "--filters": "filter_wiki_user_titles",756          "--dedups": "dedup_document filter_remove_empty_docs",757          "--maps-and-filters argument": "",758          "--filter-short-documents": ""759        },760        {761          "ln_code": "indic-mr",762          "dataset_name": "lm_indic-mr_wikipedia",763          "size": 0.402612187,764          "--filters": "filter_wiki_user_titles",765          "--dedups": "dedup_document filter_remove_empty_docs",766          "--maps-and-filters argument": "",767          "--filter-short-documents": "filter_small_docs_bytes_300"768        },769        {770          "ln_code": "id",771          "dataset_name": "lm_id_wikipedia",772          "size": 0.31454149,773          "--filters": "",774          "--dedups": "dedup_document filter_remove_empty_docs",775          "--maps-and-filters argument": "",776          "--filter-short-documents": "filter_small_docs_bytes_300"777        },778        {779          "ln_code": "indic-pa",780          "dataset_name": "lm_indic-pa_wikipedia",781          "size": 0.283836132,782          "--filters": "filter_wiki_user_titles",783          "--dedups": "dedup_document filter_remove_empty_docs",784          "--maps-and-filters argument": "",785          "--filter-short-documents": "filter_small_docs_bytes_300"786        },787        {788          "ln_code": "indic-gu",789          "dataset_name": "lm_indic-gu_wikipedia",790          "size": 0.220963387,791          "--filters": "filter_wiki_user_titles",792          "--dedups": "dedup_document filter_remove_empty_docs",793          "--maps-and-filters argument": "",794          "--filter-short-documents": "filter_small_docs_bytes_300"795        },796        {797          "ln_code": "indic-as",798          "dataset_name": "lm_indic-as_wikipedia",799          "size": 0.134210507,800          "--filters": "filter_wiki_user_titles",801          "--dedups": "dedup_document filter_remove_empty_docs",802          "--maps-and-filters argument": "",803          "--filter-short-documents": ""804        },805        {806          "ln_code": "indic-or",807          "dataset_name": "lm_indic-or_wikipedia",808          "size": 0.121933809,809          "--filters": "filter_wiki_user_titles",810          "--dedups": "dedup_document filter_remove_empty_docs",811          "--maps-and-filters argument": "",812          "--filter-short-documents": ""813        }814      ],815      "total": 39.163522721999996,816      "data_card": "# wikipedia\n\n- Dataset uid: `wikipedia`\n\n### Description\n\n\n\n### Homepage\n\n\n\n### Licensing\n\n\n\n### Speaker Locations\n\n\n\n### Sizes\n\n- 3.2299 % of total\n- 4.2071 % of en\n- 5.6773 % of ar\n- 3.3416 % of fr\n- 5.2815 % of es\n- 12.4852 % of ca\n- 0.4288 % of zh\n- 0.4286 % of zh\n- 5.4743 % of indic-bn\n- 8.9062 % of indic-ta\n- 21.3313 % of indic-te\n- 4.4845 % of pt\n- 4.0493 % of indic-hi\n- 11.3163 % of indic-ml\n- 22.5300 % of indic-ur\n- 4.4902 % of vi\n- 16.9916 % of indic-kn\n- 24.7820 % of eu\n- 11.6241 % of indic-mr\n- 9.8749 % of id\n- 9.3489 % of indic-pa\n- 9.4767 % of indic-gu\n- 24.1132 % of indic-as\n- 5.3309 % of indic-or\n\n### BigScience processing steps\n\n#### Filters applied to: en\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_1024\n\n#### Filters applied to: ar\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: fr\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_1024\n\n#### Filters applied to: es\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_1024\n\n#### Filters applied to: ca\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_1024\n\n#### Filters applied to: zh\n\n\n\n#### Filters applied to: zh\n\n\n\n#### Filters applied to: indic-bn\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-ta\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-te\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: pt\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-hi\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-ml\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-ur\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: vi\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-kn\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: eu\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n\n#### Filters applied to: indic-mr\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: id\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-pa\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-gu\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-as\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n\n#### Filters applied to: indic-or\n\n- filter_wiki_user_titles\n- dedup_document\n- filter_remove_empty_docs\n\n"817    }818  ],819  [820    "openiti_proc",821    {822      "languages": [823        {824          "ln_code": "ar",825          "dataset_name": "lm_ar_openiti_proc",826          "size": 37.297631457,827          "--filters": "",828          "--dedups": "dedup_document dedup_template_soft filter_remove_empty_docs",829          "--maps-and-filters argument": "remove_html_spans",830          "--filter-short-documents": "filter_small_docs_bytes_300"831        }832      ],833      "total": 37.297631457,834      "catalogue_info": {835        "uid": "openiti",836        "type": "primary",837        "description": {838          "name": "OpenITI",839          "description": "A corpus of Arabic texts that collected from Islamic books from different websites. ",840          "homepage": "https://zenodo.org/record/4075046",841          "validated": true842        },843        "languages": {844          "language_names": [845            "Arabic",846            "ar-MSA"847          ],848          "language_comments": "",849          "language_locations": [850            "Southern Asia",851            "Western Europe",852            "Northern America",853            "Pakistan",854            "Austria",855            "Germany",856            "United States of America"857          ],858          "validated": false859        },860        "custodian": {861          "name": "Matthew Thomas Miller (University of Maryland, College Park), Maxim G. Romanov (University of Vienna), Sarah Bowen Savant (Aga Khan University\u2014ISMC, London).",862          "in_catalogue": "",863          "type": "A university or research institution",864          "location": "United States of America",865          "contact_name": "",866          "contact_email": "",867          "contact_submitter": false,868          "additional": "https://zenodo.org/record/4075046#.YaE8X7tRXEy",869          "validated": false870        },871        "availability": {872          "procurement": {873            "for_download": "Yes - it has a direct download link or links",874            "download_url": "https://zenodo.org/record/4075046/files/OpenITI-v2020.2.3.zip?download=1",875            "download_email": ""876          },877          "licensing": {878            "has_licenses": "Yes",879            "license_text": "By exercising the Licensed Rights (defined below), You accept and agree to be bound by the terms and conditions of this Creative Commons Attribution-NonCommercial-ShareAlike 4.0 International Public License (\"Public License\"). To the extent this Public License may be interpreted as a contract, You are granted the Licensed Rights in consideration of Your acceptance of these terms and conditions, and the Licensor grants You such rights in consideration of benefits the Licensor receives from making the Licensed Material available under these terms and conditions.",880            "license_properties": [881              "non-commercial use"882            ],883            "license_list": [884              "cc-by-nc-sa-4.0: Creative Commons Attribution Non Commercial Share Alike 4.0 International"885            ]886          },887          "pii": {888            "has_pii": "Unclear",889            "generic_pii_likely": "",890            "generic_pii_list": [],891            "numeric_pii_likely": "",892            "numeric_pii_list": [],893            "sensitive_pii_likely": "",894            "sensitive_pii_list": [],895            "no_pii_justification_class": "general knowledge not written by or referring to private persons",896            "no_pii_justification_text": ""897          },898          "validated": false899        },900        "source_category": {901          "category_type": "collection",902          "category_web": "",903          "category_media": "books/book publisher",904          "validated": false905        },906        "media": {907          "category": [908            "text"909          ],910          "text_format": [911            ".CSV",912            ".TXT"913          ],914          "audiovisual_format": [],915          "image_format": [],916          "database_format": [917            ".ZIP"918          ],919          "text_is_transcribed": "No",920          "instance_type": "book",921          "instance_count": "",922          "instance_size": "",923          "validated": false924        },925        "fname": "openiti.json"926      },927      "data_card": "# OpenITI\n\n- Dataset uid: `openiti_proc`\n\n### Description\n\nA corpus of Arabic texts that collected from Islamic books from different websites. \n\n### Homepage\n\nhttps://zenodo.org/record/4075046\n\n### Licensing\n\n- non-commercial use\n- cc-by-nc-sa-4.0: Creative Commons Attribution Non Commercial Share Alike 4.0 International\n\nBy exercising the Licensed Rights (defined below), You accept and agree to be bound by the terms and conditions of this Creative Commons Attribution-NonCommercial-ShareAlike 4.0 International Public License (\"Public License\"). To the extent this Public License may be interpreted as a contract, You are granted the Licensed Rights in consideration of Your acceptance of these terms and conditions, and the Licensor grants You such rights in consideration of benefits the Licensor receives from making the Licensed Material available under these terms and conditions.\n\n\n### Speaker Locations\n\n- Southern Asia\n- Western Europe\n- Northern America\n- Pakistan\n- Austria\n- Germany\n- United States of America\n\n\n### Sizes\n\n- 3.0760 % of total\n- 28.3450 % of ar\n\n### BigScience processing steps\n\n#### Filters applied to: ar\n\n- dedup_document\n- dedup_template_soft\n- filter_remove_empty_docs\n- remove_html_spans\n- filter_small_docs_bytes_300\n\n"928    }929  ],930  [931    "open_subtitles",932    {933      "languages": [934        {935          "ln_code": "en",936          "dataset_name": "lm_en_open_subtitles",937          "size": 11.323497076,938          "--filters": "",939          "--dedups": "dedup_document filter_remove_empty_docs",940          "--maps-and-filters argument": "",941          "--filter-short-documents": "filter_small_docs_bytes_1024"942        },943        {944          "ln_code": "ar",945          "dataset_name": "lm_ar_open_subtitles",946          "size": 8.643261281,947          "--filters": "",948          "--dedups": "dedup_document filter_remove_empty_docs",949          "--maps-and-filters argument": "",950          "--filter-short-documents": "filter_small_docs_bytes_300"951        },952        {953          "ln_code": "es",954          "dataset_name": "lm_es_open_subtitles",955          "size": 6.916561317,956          "--filters": "",957          "--dedups": "dedup_document filter_remove_empty_docs",958          "--maps-and-filters argument": "",959          "--filter-short-documents": "filter_small_docs_bytes_1024"960        },961        {962          "ln_code": "pt",963          "dataset_name": "lm_pt_open_subtitles",964          "size": 3.44014906,965          "--filters": "",966          "--dedups": "dedup_document filter_remove_empty_docs",967          "--maps-and-filters argument": "",968          "--filter-short-documents": "filter_small_docs_bytes_300"969        },970        {971          "ln_code": "fr",972          "dataset_name": "lm_fr_open_subtitles",973          "size": 3.421248727,974          "--filters": "",975          "--dedups": "dedup_document filter_remove_empty_docs",976          "--maps-and-filters argument": "",977          "--filter-short-documents": "filter_small_docs_bytes_1024"978        },979        {980          "ln_code": "zh",981          "dataset_name": "lm_zh_open_subtitles",982          "size": 1.587619688,983          "--filters": "",984          "--dedups": "dedup_document filter_remove_empty_docs",985          "--maps-and-filters argument": "",986          "--filter-short-documents": "filter_small_docs_bytes_1024"987        },988        {989          "ln_code": "id",990          "dataset_name": "lm_id_open_subtitles",991          "size": 0.667604473,992          "--filters": "",993          "--dedups": "dedup_document filter_remove_empty_docs",994          "--maps-and-filters argument": "",995          "--filter-short-documents": "filter_small_docs_bytes_300"996        },997        {998          "ln_code": "vi",999          "dataset_name": "lm_vi_open_subtitles",1000          "size": 0.318329524,1001          "--filters": "",1002          "--dedups": "dedup_document filter_remove_empty_docs",1003          "--maps-and-filters argument": "",1004          "--filter-short-documents": "filter_small_docs_bytes_300"1005        },1006        {1007          "ln_code": "indic-ml",1008          "dataset_name": "lm_indic-ml_open_subtitles",1009          "size": 0.084112887,1010          "--filters": "",1011          "--dedups": "dedup_document filter_remove_empty_docs",1012          "--maps-and-filters argument": "",1013          "--filter-short-documents": "filter_small_docs_bytes_300"1014        },1015        {1016          "ln_code": "indic-bn",1017          "dataset_name": "lm_indic-bn_open_subtitles",1018          "size": 0.073684224,1019          "--filters": "",1020          "--dedups": "dedup_document filter_remove_empty_docs",1021          "--maps-and-filters argument": "",1022          "--filter-short-documents": "filter_small_docs_bytes_300"1023        },1024        {1025          "ln_code": "eu",1026          "dataset_name": "lm_eu_open_subtitles",1027          "size": 0.029221679,1028          "--filters": "",1029          "--dedups": "dedup_document filter_remove_empty_docs",1030          "--maps-and-filters argument": "",1031          "--filter-short-documents": ""1032        },1033        {1034          "ln_code": "ca",1035          "dataset_name": "lm_ca_open_subtitles",1036          "size": 0.022188171,1037          "--filters": "",1038          "--dedups": "dedup_document filter_remove_empty_docs",1039          "--maps-and-filters argument": "",1040          "--filter-short-documents": "filter_small_docs_bytes_1024"1041        },1042        {1043          "ln_code": "indic-hi",1044          "dataset_name": "lm_indic-hi_open_subtitles",1045          "size": 0.017488212,1046          "--filters": "",1047          "--dedups": "dedup_document filter_remove_empty_docs",1048          "--maps-and-filters argument": "",1049          "--filter-short-documents": "filter_small_docs_bytes_300"1050        },1051        {1052          "ln_code": "indic-ta",1053          "dataset_name": "lm_indic-ta_open_subtitles",1054          "size": 0.005367905,1055          "--filters": "",1056          "--dedups": "dedup_document filter_remove_empty_docs",1057          "--maps-and-filters argument": "",1058          "--filter-short-documents": "filter_small_docs_bytes_300"1059        },1060        {1061          "ln_code": "indic-ur",1062          "dataset_name": "lm_indic-ur_open_subtitles",1063          "size": 0.004406184,1064          "--filters": "",1065          "--dedups": "dedup_document filter_remove_empty_docs",1066          "--maps-and-filters argument": "",1067          "--filter-short-documents": "filter_small_docs_bytes_300"1068        },1069        {1070          "ln_code": "indic-te",1071          "dataset_name": "lm_indic-te_open_subtitles",1072          "size": 0.003926207,1073          "--filters": "",1074          "--dedups": "dedup_document filter_remove_empty_docs",1075          "--maps-and-filters argument": "",1076          "--filter-short-documents": "filter_small_docs_bytes_300"1077        }1078      ],1079      "total": 36.55866661499999,1080      "catalogue_info": {1081        "uid": "open_subtitles",1082        "type": "primary",1083        "description": {1084          "name": "Open Subtitles",1085          "description": "A community repository for subtitles, with a total of 3.36 million subtitle files covering more than 60 languages",1086          "homepage": "https://www.opensubtitles.com/en/home",1087          "validated": true1088        },1089        "languages": {1090          "language_names": [1091            "Arabic",1092            "Basque",1093            "Catalan",1094            "English",1095            "French",1096            "Indic",1097            "Indonesian",1098            "Portuguese",1099            "Spanish",1100            "Vietnamese",1101            "Chinese",1102            "Hindi",1103            "Malayalam",1104            "Urdu",1105            "Tamil",1106            "Telugu",1107            "Bengali"1108          ],1109          "language_comments": "European and Brazilian Portuguese",1110          "language_locations": [1111            "World-Wide"1112          ],1113          "validated": false1114        },1115        "custodian": {1116          "name": "Opensubtitles.org",1117          "in_catalogue": "",1118          "type": "A commercial entity",1119          "location": "",1120          "contact_name": "",1121          "contact_email": "admin@opensubtitles.org",1122          "contact_submitter": false,1123          "additional": "https://www.opensubtitles.com/en/privacy/",1124          "validated": false1125        },1126        "availability": {1127          "procurement": {1128            "for_download": "Yes - it has a direct download link or links",1129            "download_url": "https://opus.nlpl.eu/OpenSubtitles-v2018.php",1130            "download_email": ""1131          },1132          "licensing": {1133            "has_licenses": "No",1134            "license_text": "If you use the OpenSubtitle corpus: Please, add a link to http://www.opensubtitles.org/ to your website and to your reports and publications produced with the data! I promised this when I got the data from the providers of that website!",1135            "license_properties": [],1136            "license_list": []1137          },1138          "pii": {1139            "has_pii": "Yes",1140            "generic_pii_likely": "very likely",1141            "generic_pii_list": [1142              "website account name or handle",1143              "names"1144            ],1145            "numeric_pii_likely": "unlikely",1146            "numeric_pii_list": [],1147            "sensitive_pii_likely": "unlikely",1148            "sensitive_pii_list": [],1149            "no_pii_justification_class": "",1150            "no_pii_justification_text": ""1151          },1152          "validated": false1153        },1154        "source_category": {1155          "category_type": "website",1156          "category_web": "content repository, archive, or collection",1157          "category_media": "movies and documentaries",1158          "validated": false1159        },1160        "media": {1161          "category": [1162            "text"1163          ],1164          "text_format": [1165            "other",1166            ".TXT",1167            "srt, sub, ssa"1168          ],1169          "audiovisual_format": [],1170          "image_format": [],1171          "database_format": [],1172          "text_is_transcribed": "No",1173          "instance_type": "dialogue",1174          "instance_count": "1M<n<1B",1175          "instance_size": "n>10,000",1176          "validated": false1177        },1178        "fname": "open_subtitles.json"1179      },1180      "data_card": "# Open Subtitles\n\n- Dataset uid: `open_subtitles`\n\n### Description\n\nA community repository for subtitles, with a total of 3.36 million subtitle files covering more than 60 languages\n\n### Homepage\n\nhttps://www.opensubtitles.com/en/home\n\n### Licensing\n\n\n\n### Speaker Locations\n\n- World-Wide\n\n\n### Sizes\n\n- 3.0150 % of total\n- 5.0599 % of en\n- 6.5686 % of ar\n- 13.5783 % of es\n- 13.1277 % of pt\n- 3.3240 % of fr\n- 0.4580 % of zh\n- 20.9593 % of id\n- 1.9182 % of vi\n- 1.1647 % of indic-ml\n- 0.2794 % of indic-bn\n- 1.4829 % of eu\n- 0.1543 % of ca\n- 0.0633 % of indic-hi\n- 0.0342 % of indic-ta\n- 0.1286 % of indic-ur\n- 0.0671 % of indic-te\n\n### BigScience processing steps\n\n#### Filters applied to: en\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_1024\n\n#### Filters applied to: ar\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: es\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_1024\n\n#### Filters applied to: pt\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: fr\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_1024\n\n#### Filters applied to: zh\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_1024\n\n#### Filters applied to: id\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: vi\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-ml\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-bn\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: eu\n\n- dedup_document\n- filter_remove_empty_docs\n\n#### Filters applied to: ca\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_1024\n\n#### Filters applied to: indic-hi\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-ta\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-ur\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n#### Filters applied to: indic-te\n\n- dedup_document\n- filter_remove_empty_docs\n- filter_small_docs_bytes_300\n\n"1181    }1182  ],1183  [1184    "uncorpus",1185    {1186      "languages": [1187        {1188          "ln_code": "ar",1189          "dataset_name": "lm_ar_uncorpus",1190          "size": 14.130924048,1191          "--filters": "",1192          "--dedups": "dedup_document filter_remove_empty_docs",1193          "--maps-and-filters argument": "",1194          "--filter-short-documents": "filter_small_docs_bytes_300"1195        },1196        {1197          "ln_code": "fr",1198          "dataset_name": "lm_fr_uncorpus",1199          "size": 5.966635998,1200          "--filters": "",

Showing the first 1,200 of 14707 lines. Download the file for the rest.