kgnlp/meld-open
MELD Open MELD is a multilingual and multi-domain dataset for Named Entity Recognition (NER) constructed from 60 existing datasets. It includes gold-standard annotations across 60 languages and 14 domains. This dataset is a subset of 43 datasets for which licenses permit the redistribution of data in a new format. See the MELD GitHub repository for more details. Note: This version of MELD Open retains the original labels from its source datasets. For normalized labels, use… See the full description on the dataset page: https://huggingface.co/datasets/kgnlp/meld-open.
0214
1---2language:3- amh4- ara5- asm6- bam7- bbj8- ben9- ceb10- dan11- deu12- eng13- ewe14- fao15- fas16- fil17- fin18- fra19- guj20- hau21- hin22- hrv23- hun24- hye25- ibo26- ind27- ita28- jpn29- kan30- kin31- lat32- lug33- luo34- mal35- mar36- mos37- nld38- nno39- nob40- nya41- ori42- pan43- pcm44- pol45- por46- qaf47- rus48- slk49- sna50- spa51- srp52- swa53- swe54- tam55- tel56- tha57- tsn58- twi59- ukr60- wol61- xho62- yor63- zho64- zul65multilinguality: multilingual66size_categories: 10M<n<100M67task_categories:68- token-classification69task_ids:70- named-entity-recognition71pretty_name: MELD Open72meld_metadata:73 AgCNER:74 name: AgCNER75 subsets: {}76 languages:77 - zho78 main_splits:79 language: zho80 pre_tokenized: true81 tagsets:82 - ner83 labels:84 ner:85 - BIS86 - COM87 - CRO88 - CUL89 - DIS90 - DRUG91 - FER92 - ORG93 - OTH94 - PAOG95 - PART96 - PER97 - PET98 splits:99 train:100 path: train.parquet101 document_count: 47839102 sequence_count: 47839103 creation_metadata:104 license: CC 0105 annotation:106 type: gold-standard107 annotator_count: 3108 features:109 - expert-guided110 agreement:111 value: 0.966112 metric: fleiss-kappa113 primary_domain: agriculture114 other_domains: []115 finegrained_domains: []116 data_sources:117 - source: China National Knowledge Infrastructure118 url: https://www.cnki.com.cn/119 - source: Wangfang Data120 url: https://www.wanfangdata.com.cn/121 - source: China Agricultural Technology Promotion Information Platform122 url: https://njtg.nercita.org.cn/user/index.shtml123 - source: China Pesticide Information Network124 url: http://www.chinapesticide.org.cn/125 - source: Agricultural Diseases, Pests, and Weeds Image and Text Database126 url: http://bcch.ahnw.cn/127 - source: Chinese Crop Germplasm Information System128 url: https://www.cgris.net/disease/default.html129 - source: Baidu Baike130 url: https://baike.baidu.com/131 dataset_lineage: null132 label_set_standard: null133 document_boundaries: none134 sentence_boundaries: full135 validation:136 path: validation.parquet137 document_count: 5963138 sequence_count: 5963139 creation_metadata:140 license: CC 0141 annotation:142 type: gold-standard143 annotator_count: 3144 features:145 - expert-guided146 agreement:147 value: 0.966148 metric: fleiss-kappa149 primary_domain: agriculture150 other_domains: []151 finegrained_domains: []152 data_sources:153 - source: China National Knowledge Infrastructure154 url: https://www.cnki.com.cn/155 - source: Wangfang Data156 url: https://www.wanfangdata.com.cn/157 - source: China Agricultural Technology Promotion Information Platform158 url: https://njtg.nercita.org.cn/user/index.shtml159 - source: China Pesticide Information Network160 url: http://www.chinapesticide.org.cn/161 - source: Agricultural Diseases, Pests, and Weeds Image and Text Database162 url: http://bcch.ahnw.cn/163 - source: Chinese Crop Germplasm Information System164 url: https://www.cgris.net/disease/default.html165 - source: Baidu Baike166 url: https://baike.baidu.com/167 dataset_lineage: null168 label_set_standard: null169 document_boundaries: none170 sentence_boundaries: full171 test:172 path: test.parquet173 document_count: 5974174 sequence_count: 5974175 creation_metadata:176 license: CC 0177 annotation:178 type: gold-standard179 annotator_count: 3180 features:181 - expert-guided182 agreement:183 value: 0.966184 metric: fleiss-kappa185 primary_domain: agriculture186 other_domains: []187 finegrained_domains: []188 data_sources:189 - source: China National Knowledge Infrastructure190 url: https://www.cnki.com.cn/191 - source: Wangfang Data192 url: https://www.wanfangdata.com.cn/193 - source: China Agricultural Technology Promotion Information Platform194 url: https://njtg.nercita.org.cn/user/index.shtml195 - source: China Pesticide Information Network196 url: http://www.chinapesticide.org.cn/197 - source: Agricultural Diseases, Pests, and Weeds Image and Text Database198 url: http://bcch.ahnw.cn/199 - source: Chinese Crop Germplasm Information System200 url: https://www.cgris.net/disease/default.html201 - source: Baidu Baike202 url: https://baike.baidu.com/203 dataset_lineage: null204 label_set_standard: null205 document_boundaries: none206 sentence_boundaries: full207 bio_labels:208 ner:209 - B-BIS210 - B-COM211 - B-CRO212 - B-CUL213 - B-DIS214 - B-DRUG215 - B-FER216 - B-ORG217 - B-OTH218 - B-PAOG219 - B-PART220 - B-PER221 - B-PET222 - I-BIS223 - I-COM224 - I-CRO225 - I-CUL226 - I-DIS227 - I-DRUG228 - I-FER229 - I-ORG230 - I-OTH231 - I-PAOG232 - I-PART233 - I-PER234 - I-PET235 - O236 AgriNER:237 name: AgriNER238 subsets: {}239 languages:240 - eng241 main_splits:242 language: eng243 pre_tokenized: false244 tagsets:245 - ner246 labels:247 ner:248 - Agri_Method249 - Agri_Pollution250 - Agri_Process251 - Agri_Waste252 - Chemical253 - Citation254 - Crop255 - Date_and_Time256 - Disease257 - Duration258 - Event259 - Field_Area260 - Food_Item261 - Fruit262 - Humidity263 - Location264 - ML_Model265 - Money266 - Natural_Disaster267 - Natural_Resource268 - Nutrient269 - Organism270 - Organization271 - Other272 - Other_Quantity273 - Person274 - Policy275 - Quantity276 - Rainfall277 - Season278 - Soil279 - Technology280 - Temp281 - Treatment282 - Vegetable283 - Weather284 splits:285 train:286 path: train.parquet287 document_count: 21288 sequence_count: 471289 creation_metadata:290 license: CC BY-SA 4.0291 annotation:292 type: gold-standard293 annotator_count: 4+294 features: []295 agreement: null296 primary_domain: agriculture297 other_domains: []298 finegrained_domains: []299 data_sources:300 - source: Asian Journal of Agricultural and Food Sciences (AJAFS)301 url: https://www.ajouronline.com/index.php/AJAFS/index302 - source: The Indian Journal of Agricultural Sciences303 url: https://epubs.icar.org.in/index.php/IJAgS304 - IEEE Journals305 - Springer Nature Journals306 dataset_lineage: null307 label_set_standard: null308 document_boundaries: full309 sentence_boundaries: none310 test:311 path: test.parquet312 document_count: 8313 sequence_count: 46314 creation_metadata:315 license: CC BY-SA 4.0316 annotation:317 type: gold-standard318 annotator_count: 4+319 features: []320 agreement: null321 primary_domain: agriculture322 other_domains: []323 finegrained_domains: []324 data_sources:325 - source: Asian Journal of Agricultural and Food Sciences (AJAFS)326 url: https://www.ajouronline.com/index.php/AJAFS/index327 - source: The Indian Journal of Agricultural Sciences328 url: https://epubs.icar.org.in/index.php/IJAgS329 - IEEE Journals330 - Springer Nature Journals331 dataset_lineage: null332 label_set_standard: null333 document_boundaries: full334 sentence_boundaries: none335 bio_labels: null336 AnatEM:337 name: AnatEM338 subsets: {}339 languages:340 - eng341 main_splits:342 language: eng343 pre_tokenized: true344 tagsets:345 - ner346 labels:347 ner:348 - Anatomical_system349 - Cancer350 - Cell351 - Cellular_component352 - Developing_anatomical_structure353 - Immaterial_anatomical_entity354 - Multi-tissue_structure355 - Organ356 - Organism_subdivision357 - Organism_substance358 - Pathological_formation359 - Tissue360 splits:361 train:362 path: train.parquet363 document_count: 606364 sequence_count: 5861365 creation_metadata:366 license: CC BY-SA 3.0367 annotation:368 type: gold-standard369 annotator_count: null370 features: []371 agreement: null372 primary_domain: biomedical373 other_domains: []374 finegrained_domains:375 - anatomical376 data_sources:377 - PubMed378 dataset_lineage:379 - AnEM380 - Multi-Level Event Extraction (MLEE)381 - New data382 label_set_standard: null383 document_boundaries: full384 sentence_boundaries: full385 validation:386 path: validation.parquet387 document_count: 202388 sequence_count: 2118389 creation_metadata:390 license: CC BY-SA 3.0391 annotation:392 type: gold-standard393 annotator_count: null394 features: []395 agreement: null396 primary_domain: biomedical397 other_domains: []398 finegrained_domains:399 - anatomical400 data_sources:401 - PubMed402 dataset_lineage:403 - AnEM404 - Multi-Level Event Extraction (MLEE)405 - New data406 label_set_standard: null407 document_boundaries: full408 sentence_boundaries: full409 test:410 path: test.parquet411 document_count: 404412 sequence_count: 3830413 creation_metadata:414 license: CC BY-SA 3.0415 annotation:416 type: gold-standard417 annotator_count: null418 features: []419 agreement: null420 primary_domain: biomedical421 other_domains: []422 finegrained_domains:423 - anatomical424 data_sources:425 - PubMed426 dataset_lineage:427 - AnEM428 - Multi-Level Event Extraction (MLEE)429 - New data430 label_set_standard: null431 document_boundaries: full432 sentence_boundaries: full433 bio_labels:434 ner:435 - B-Anatomical_system436 - B-Cancer437 - B-Cell438 - B-Cellular_component439 - B-Developing_anatomical_structure440 - B-Immaterial_anatomical_entity441 - B-Multi-tissue_structure442 - B-Organ443 - B-Organism_subdivision444 - B-Organism_substance445 - B-Pathological_formation446 - B-Tissue447 - I-Anatomical_system448 - I-Cancer449 - I-Cell450 - I-Cellular_component451 - I-Developing_anatomical_structure452 - I-Immaterial_anatomical_entity453 - I-Multi-tissue_structure454 - I-Organ455 - I-Organism_subdivision456 - I-Organism_substance457 - I-Pathological_formation458 - I-Tissue459 - O460 BC2GM:461 name: BC2GM462 subsets: {}463 languages:464 - eng465 main_splits:466 language: eng467 pre_tokenized: true468 tagsets:469 - ner470 labels:471 ner:472 - GENE473 splits:474 train:475 path: train.parquet476 document_count: 12574477 sequence_count: 12574478 creation_metadata:479 license: CC BY 4.0480 annotation:481 type: gold-standard482 annotator_count: null483 features:484 - domain-experts485 agreement: null486 primary_domain: biomedical487 other_domains: []488 finegrained_domains:489 - genetic490 data_sources:491 - MEDLINE492 dataset_lineage:493 - GENETAG->BC1GM494 label_set_standard: null495 document_boundaries: none496 sentence_boundaries: full497 validation:498 path: validation.parquet499 document_count: 2519500 sequence_count: 2519501 creation_metadata:502 license: CC BY 4.0503 annotation:504 type: gold-standard505 annotator_count: null506 features:507 - domain-experts508 agreement: null509 primary_domain: biomedical510 other_domains: []511 finegrained_domains:512 - genetic513 data_sources:514 - MEDLINE515 dataset_lineage:516 - GENETAG->BC1GM517 label_set_standard: null518 document_boundaries: none519 sentence_boundaries: full520 test:521 path: test.parquet522 document_count: 5038523 sequence_count: 5038524 creation_metadata:525 license: CC BY 4.0526 annotation:527 type: gold-standard528 annotator_count: null529 features:530 - domain-experts531 agreement: null532 primary_domain: biomedical533 other_domains: []534 finegrained_domains:535 - genetic536 data_sources:537 - MEDLINE538 dataset_lineage:539 - GENETAG->BC1GM540 label_set_standard: null541 document_boundaries: none542 sentence_boundaries: full543 bio_labels:544 ner:545 - B-GENE546 - I-GENE547 - O548 BC5CDR:549 name: BC5CDR550 subsets: {}551 languages:552 - eng553 main_splits:554 language: eng555 pre_tokenized: false556 tagsets:557 - ner558 labels:559 ner:560 - Chemical561 - Disease562 splits:563 train:564 path: train.parquet565 document_count: 500566 sequence_count: 4941567 creation_metadata:568 license: Public Domain569 annotation:570 type: gold-standard571 annotator_count: 4572 features:573 - annotators-each:2574 agreement:575 value:576 Disease: 0.86577 Chemical: 0.9523578 metric: jaccard-score579 primary_domain: biomedical580 other_domains: []581 finegrained_domains:582 - chemical-disease relations583 data_sources:584 - CTD-Pfizer corpus585 - PubMed586 dataset_lineage: []587 label_set_standard: null588 document_boundaries: full589 sentence_boundaries: sections590 validation:591 path: validation.parquet592 document_count: 500593 sequence_count: 5037594 creation_metadata:595 license: Public Domain596 annotation:597 type: gold-standard598 annotator_count: 4599 features:600 - annotators-each:2601 agreement:602 value:603 Disease: 0.8742604 Chemical: 0.9577605 metric: jaccard-score606 primary_domain: biomedical607 other_domains: []608 finegrained_domains:609 - chemical-disease relations610 data_sources:611 - CTD-Pfizer corpus612 - PubMed613 dataset_lineage: []614 label_set_standard: null615 document_boundaries: full616 sentence_boundaries: sections617 test:618 path: test.parquet619 document_count: 500620 sequence_count: 5349621 creation_metadata:622 license: Public Domain623 annotation:624 type: gold-standard625 annotator_count: 4626 features:627 - annotators-each:2628 agreement:629 value:630 Disease: 0.8875631 Chemical: 0.963632 metric: jaccard-score633 primary_domain: biomedical634 other_domains: []635 finegrained_domains:636 - chemical-disease relations637 data_sources:638 - CTD-Pfizer corpus639 - PubMed640 dataset_lineage: []641 label_set_standard: null642 document_boundaries: full643 sentence_boundaries: sections644 bio_labels: null645 BioRED:646 name: BioRED647 subsets: {}648 languages:649 - eng650 main_splits:651 language: eng652 pre_tokenized: false653 tagsets:654 - ner655 labels:656 ner:657 - CellLine658 - ChemicalEntity659 - DiseaseOrPhenotypicFeature660 - GeneOrGeneProduct661 - OrganismTaxon662 - SequenceVariant663 splits:664 train:665 path: train.parquet666 document_count: 400667 sequence_count: 4811668 creation_metadata:669 license: Public Domain670 annotation:671 type: gold-standard672 annotator_count: 3673 features:674 - domain-experts675 agreement:676 value:677 Gene: 0.9735678 Disease: 0.9606679 Chemical: 0.9612680 Variant: 0.9779681 Species: 0.9943682 Cell Line: 0.9968683 metric: null684 primary_domain: biomedical685 other_domains: []686 finegrained_domains: []687 data_sources:688 - PubMed689 dataset_lineage: []690 label_set_standard: null691 document_boundaries: full692 sentence_boundaries: sections693 validation:694 path: validation.parquet695 document_count: 100696 sequence_count: 1252697 creation_metadata:698 license: Public Domain699 annotation:700 type: gold-standard701 annotator_count: 3702 features:703 - domain-experts704 agreement:705 value:706 Gene: 0.9735707 Disease: 0.9606708 Chemical: 0.9612709 Variant: 0.9779710 Species: 0.9943711 Cell Line: 0.9968712 metric: null713 primary_domain: biomedical714 other_domains: []715 finegrained_domains: []716 data_sources:717 - PubMed718 dataset_lineage: []719 label_set_standard: null720 document_boundaries: full721 sentence_boundaries: sections722 test:723 path: test.parquet724 document_count: 100725 sequence_count: 1237726 creation_metadata:727 license: Public Domain728 annotation:729 type: gold-standard730 annotator_count: 3731 features:732 - domain-experts733 agreement:734 value:735 Gene: 0.9735736 Disease: 0.9606737 Chemical: 0.9612738 Variant: 0.9779739 Species: 0.9943740 Cell Line: 0.9968741 metric: null742 primary_domain: biomedical743 other_domains: []744 finegrained_domains: []745 data_sources:746 - PubMed747 dataset_lineage: []748 label_set_standard: null749 document_boundaries: full750 sentence_boundaries: sections751 bio_labels: null752 CANTEMIST:753 name: CANTEMIST754 subsets: {}755 languages:756 - spa757 main_splits:758 language: spa759 pre_tokenized: false760 tagsets:761 - ner762 labels:763 ner:764 - MORFOLOGIA_NEOPLASIA765 splits:766 train:767 path: train.parquet768 document_count: 501769 sequence_count: 21609770 creation_metadata:771 license: CC BY 4.0772 annotation:773 type: gold-standard774 annotator_count: null775 features:776 - domain-experts777 - expert-guided778 agreement: null779 primary_domain: clinical780 other_domains: []781 finegrained_domains:782 - cancer783 - tumor morphology784 data_sources:785 - Spanish language oncological clinical case reports786 dataset_lineage: []787 label_set_standard: null788 document_boundaries: full789 sentence_boundaries: none790 validation:791 path: validation.parquet792 document_count: 500793 sequence_count: 20385794 creation_metadata:795 license: CC BY 4.0796 annotation:797 type: gold-standard798 annotator_count: null799 features:800 - domain-experts801 - expert-guided802 agreement: null803 primary_domain: clinical804 other_domains: []805 finegrained_domains:806 - cancer807 - tumor morphology808 data_sources:809 - Spanish language oncological clinical case reports810 dataset_lineage: []811 label_set_standard: null812 document_boundaries: full813 sentence_boundaries: none814 test:815 path: test.parquet816 document_count: 300817 sequence_count: 12499818 creation_metadata:819 license: CC BY 4.0820 annotation:821 type: gold-standard822 annotator_count: null823 features:824 - domain-experts825 - expert-guided826 agreement: null827 primary_domain: clinical828 other_domains: []829 finegrained_domains:830 - cancer831 - tumor morphology832 data_sources:833 - Spanish language oncological clinical case reports834 dataset_lineage: []835 label_set_standard: null836 document_boundaries: full837 sentence_boundaries: none838 bio_labels: null839 CLEANANERCorp:840 name: CLEANANERCorp841 subsets: {}842 languages:843 - ara844 main_splits:845 language: ara846 pre_tokenized: true847 tagsets:848 - ner849 labels:850 ner:851 - LOC852 - MISC853 - ORG854 - PERS855 splits:856 train:857 path: train.parquet858 document_count: 3974859 sequence_count: 3974860 creation_metadata:861 license: GPL 3.0862 annotation:863 type: gold-standard864 annotator_count: 1865 features: []866 agreement: null867 primary_domain: news868 other_domains: []869 finegrained_domains: []870 data_sources:871 - source: Al Jazeera872 url: http://www.aljazeera.net873 - source: Raya874 url: http://www.raya.com875 - source: Arabic Wikipedia876 url: http://ar.wikipedia.org877 - source: Al Alam878 url: http://www.alalam.ma879 - source: Ahram880 url: http://www.ahram.eg.org881 - source: Al Ittihad882 url: http://www.alittihad.ae883 - source: BBC Arabic884 url: http://www.bbc.co.uk/arabic/885 - source: CNN Arabic886 url: http://arabic.cnn.com887 - source: Addustour888 url: http://www.addustour.com889 - source: Kassioun890 url: http://kassioun.org891 - Other newspapers and magazines892 dataset_lineage:893 - ANERcorp894 label_set_standard: MUC-6895 document_boundaries: none896 sentence_boundaries: full897 test:898 path: test.parquet899 document_count: 925900 sequence_count: 925901 creation_metadata:902 license: GPL 3.0903 annotation:904 type: gold-standard905 annotator_count: 1906 features: []907 agreement: null908 primary_domain: news909 other_domains: []910 finegrained_domains: []911 data_sources:912 - source: Al Jazeera913 url: http://www.aljazeera.net914 - source: Raya915 url: http://www.raya.com916 - source: Arabic Wikipedia917 url: http://ar.wikipedia.org918 - source: Al Alam919 url: http://www.alalam.ma920 - source: Ahram921 url: http://www.ahram.eg.org922 - source: Al Ittihad923 url: http://www.alittihad.ae924 - source: BBC Arabic925 url: http://www.bbc.co.uk/arabic/926 - source: CNN Arabic927 url: http://arabic.cnn.com928 - source: Addustour929 url: http://www.addustour.com930 - source: Kassioun931 url: http://kassioun.org932 - Other newspapers and magazines933 dataset_lineage:934 - ANERcorp935 label_set_standard: MUC-6936 document_boundaries: none937 sentence_boundaries: full938 bio_labels:939 ner:940 - B-LOC941 - B-MISC942 - B-ORG943 - B-PERS944 - I-LOC945 - I-MISC946 - I-ORG947 - I-PERS948 - O949 CrossNER:950 name: CrossNER951 subsets:952 ai:953 language: eng954 pre_tokenized: true955 tagsets:956 - ner957 labels:958 ner:959 - algorithm960 - conference961 - country962 - field963 - location964 - metrics965 - misc966 - organisation967 - person968 - product969 - programlang970 - researcher971 - task972 - university973 splits:974 train:975 path: ai/train.parquet976 document_count: 100977 sequence_count: 100978 creation_metadata:979 license: MIT980 annotation:981 type: gold-standard982 annotator_count: null983 features:984 - annotators-each:3985 agreement: null986 primary_domain: wikipedia987 other_domains: []988 finegrained_domains:989 - AI990 data_sources:991 - Wikipedia992 dataset_lineage: []993 label_set_standard: null994 document_boundaries: none995 sentence_boundaries: full996 validation:997 path: ai/validation.parquet998 document_count: 350999 sequence_count: 3501000 creation_metadata:1001 license: MIT1002 annotation:1003 type: gold-standard1004 annotator_count: null1005 features:1006 - annotators-each:31007 agreement: null1008 primary_domain: wikipedia1009 other_domains: []1010 finegrained_domains:1011 - AI1012 data_sources:1013 - Wikipedia1014 dataset_lineage: []1015 label_set_standard: null1016 document_boundaries: none1017 sentence_boundaries: full1018 test:1019 path: ai/test.parquet1020 document_count: 4311021 sequence_count: 4311022 creation_metadata:1023 license: MIT1024 annotation:1025 type: gold-standard1026 annotator_count: null1027 features:1028 - annotators-each:31029 agreement: null1030 primary_domain: wikipedia1031 other_domains: []1032 finegrained_domains:1033 - AI1034 data_sources:1035 - Wikipedia1036 dataset_lineage: []1037 label_set_standard: null1038 document_boundaries: none1039 sentence_boundaries: full1040 bio_labels:1041 ner:1042 - B-algorithm1043 - B-conference1044 - B-country1045 - B-field1046 - B-location1047 - B-metrics1048 - B-misc1049 - B-organisation1050 - B-person1051 - B-product1052 - B-programlang1053 - B-researcher1054 - B-task1055 - B-university1056 - I-algorithm1057 - I-conference1058 - I-country1059 - I-field1060 - I-location1061 - I-metrics1062 - I-misc1063 - I-organisation1064 - I-person1065 - I-product1066 - I-programlang1067 - I-researcher1068 - I-task1069 - I-university1070 - O1071 literature:1072 language: eng1073 pre_tokenized: true1074 tagsets:1075 - ner1076 labels:1077 ner:1078 - award1079 - book1080 - country1081 - event1082 - literarygenre1083 - location1084 - magazine1085 - misc1086 - organisation1087 - person1088 - poem1089 - writer1090 splits:1091 train:1092 path: literature/train.parquet1093 document_count: 1001094 sequence_count: 1001095 creation_metadata:1096 license: MIT1097 annotation:1098 type: gold-standard1099 annotator_count: null1100 features:1101 - annotators-each:31102 agreement: null1103 primary_domain: wikipedia1104 other_domains: []1105 finegrained_domains:1106 - Literature1107 data_sources:1108 - Wikipedia1109 dataset_lineage: []1110 label_set_standard: null1111 document_boundaries: none1112 sentence_boundaries: full1113 validation:1114 path: literature/validation.parquet1115 document_count: 4001116 sequence_count: 4001117 creation_metadata:1118 license: MIT1119 annotation:1120 type: gold-standard1121 annotator_count: null1122 features:1123 - annotators-each:31124 agreement: null1125 primary_domain: wikipedia1126 other_domains: []1127 finegrained_domains:1128 - Literature1129 data_sources:1130 - Wikipedia1131 dataset_lineage: []1132 label_set_standard: null1133 document_boundaries: none1134 sentence_boundaries: full1135 test:1136 path: literature/test.parquet1137 document_count: 4161138 sequence_count: 4161139 creation_metadata:1140 license: MIT1141 annotation:1142 type: gold-standard1143 annotator_count: null1144 features:1145 - annotators-each:31146 agreement: null1147 primary_domain: wikipedia1148 other_domains: []1149 finegrained_domains:1150 - Literature1151 data_sources:1152 - Wikipedia1153 dataset_lineage: []1154 label_set_standard: null1155 document_boundaries: none1156 sentence_boundaries: full1157 bio_labels:1158 ner:1159 - B-award1160 - B-book1161 - B-country1162 - B-event1163 - B-literarygenre1164 - B-location1165 - B-magazine1166 - B-misc1167 - B-organisation1168 - B-person1169 - B-poem1170 - B-writer1171 - I-award1172 - I-book1173 - I-country1174 - I-event1175 - I-literarygenre1176 - I-location1177 - I-magazine1178 - I-misc1179 - I-organisation1180 - I-person1181 - I-poem1182 - I-writer1183 - O1184 music:1185 language: eng1186 pre_tokenized: true1187 tagsets:1188 - ner1189 labels:1190 ner:1191 - album1192 - award1193 - band1194 - country1195 - event1196 - location1197 - misc1198 - musicalartist1199 - musicalinstrument1200 - musicgenre