jstack32/LatinAccents
Dataset Card for Dataset Name This dataset card aims to be a base template for new datasets. It has been generated using this raw template. Dataset Details Dataset Description Curated by: [More Information Needed] Funded by [optional]: [More Information Needed] Shared by [optional]: [More Information Needed] Language(s) (NLP): [More Information Needed] License: [More Information Needed] Dataset Sources [optional]… See the full description on the dataset page: https://huggingface.co/datasets/jstack32/LatinAccents.
013
1# coding=utf-82# Copyright 2022 The HuggingFace Datasets Authors and the current dataset script contributor.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8# http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15""" Common Voice Dataset"""16 17 18import csv19import os20import json21 22import datasets23from datasets.utils.py_utils import size_str24from tqdm import tqdm25 26from .languages import LANGUAGES27from .release_stats import STATS28 29 30_CITATION = """\31@inproceedings{commonvoice:2020,32 author = {Ardila, R. and Branson, M. and Davis, K. and Henretty, M. and Kohler, M. and Meyer, J. and Morais, R. and Saunders, L. and Tyers, F. M. and Weber, G.},33 title = {Common Voice: A Massively-Multilingual Speech Corpus},34 booktitle = {Proceedings of the 12th Conference on Language Resources and Evaluation (LREC 2020)},35 pages = {4211--4215},36 year = 202037}38"""39 40_HOMEPAGE = "https://commonvoice.mozilla.org/en/datasets"41 42_LICENSE = "https://creativecommons.org/publicdomain/zero/1.0/"43 44# TODO: change "streaming" to "main" after merge!45_BASE_URL = "https://huggingface.co/datasets/mozilla-foundation/common_voice_11_0/resolve/main/"46 47_AUDIO_URL = _BASE_URL + "audio/{lang}/{split}/{lang}_{split}_{shard_idx}.tar"48 49_TRANSCRIPT_URL = _BASE_URL + "transcript/{lang}/{split}.tsv"50 51_N_SHARDS_URL = _BASE_URL + "n_shards.json"52 53 54class CommonVoiceConfig(datasets.BuilderConfig):55 """BuilderConfig for CommonVoice."""56 57 def __init__(self, name, version, **kwargs):58 self.language = kwargs.pop("language", None)59 self.release_date = kwargs.pop("release_date", None)60 self.num_clips = kwargs.pop("num_clips", None)61 self.num_speakers = kwargs.pop("num_speakers", None)62 self.validated_hr = kwargs.pop("validated_hr", None)63 self.total_hr = kwargs.pop("total_hr", None)64 self.size_bytes = kwargs.pop("size_bytes", None)65 self.size_human = size_str(self.size_bytes)66 description = (67 f"Common Voice speech to text dataset in {self.language} released on {self.release_date}. "68 f"The dataset comprises {self.validated_hr} hours of validated transcribed speech data "69 f"out of {self.total_hr} hours in total from {self.num_speakers} speakers. "70 f"The dataset contains {self.num_clips} audio clips and has a size of {self.size_human}."71 )72 super(CommonVoiceConfig, self).__init__(73 name=name,74 version=datasets.Version(version),75 description=description,76 **kwargs,77 )78 79 80class CommonVoice(datasets.GeneratorBasedBuilder):81 DEFAULT_WRITER_BATCH_SIZE = 100082 83 BUILDER_CONFIGS = [84 CommonVoiceConfig(85 name=lang,86 version=STATS["version"],87 language=LANGUAGES[lang],88 release_date=STATS["date"],89 num_clips=lang_stats["clips"],90 num_speakers=lang_stats["users"],91 validated_hr=float(lang_stats["validHrs"]) if lang_stats["validHrs"] else None,92 total_hr=float(lang_stats["totalHrs"]) if lang_stats["totalHrs"] else None,93 size_bytes=int(lang_stats["size"]) if lang_stats["size"] else None,94 )95 for lang, lang_stats in STATS["locales"].items()96 ]97 98 def _info(self):99 total_languages = len(STATS["locales"])100 total_valid_hours = STATS["totalValidHrs"]101 description = (102 "Common Voice is Mozilla's initiative to help teach machines how real people speak. "103 f"The dataset currently consists of {total_valid_hours} validated hours of speech "104 f" in {total_languages} languages, but more voices and languages are always added."105 )106 features = datasets.Features(107 {108 "client_id": datasets.Value("string"),109 "path": datasets.Value("string"),110 "audio": datasets.features.Audio(sampling_rate=48_000),111 "sentence": datasets.Value("string"),112 "up_votes": datasets.Value("int64"),113 "down_votes": datasets.Value("int64"),114 "age": datasets.Value("string"),115 "gender": datasets.Value("string"),116 "accent": datasets.Value("string"),117 "locale": datasets.Value("string"),118 "segment": datasets.Value("string"),119 }120 )121 122 return datasets.DatasetInfo(123 description=description,124 features=features,125 supervised_keys=None,126 homepage=_HOMEPAGE,127 license=_LICENSE,128 citation=_CITATION,129 version=self.config.version,130 )131 132 def _split_generators(self, dl_manager):133 lang = self.config.name134 n_shards_path = dl_manager.download_and_extract(_N_SHARDS_URL)135 with open(n_shards_path, encoding="utf-8") as f:136 n_shards = json.load(f)137 138 audio_urls = {}139 splits = ("train", "dev", "test", "other", "invalidated")140 for split in splits:141 audio_urls[split] = [142 _AUDIO_URL.format(lang=lang, split=split, shard_idx=i) for i in range(n_shards[lang][split])143 ]144 archive_paths = dl_manager.download(audio_urls)145 local_extracted_archive_paths = dl_manager.extract(archive_paths) if not dl_manager.is_streaming else {}146 147 meta_urls = {split: _TRANSCRIPT_URL.format(lang=lang, split=split) for split in splits}148 meta_paths = dl_manager.download_and_extract(meta_urls)149 150 split_generators = []151 split_names = {152 "train": datasets.Split.TRAIN,153 "dev": datasets.Split.VALIDATION,154 "test": datasets.Split.TEST,155 }156 for split in splits:157 split_generators.append(158 datasets.SplitGenerator(159 name=split_names.get(split, split),160 gen_kwargs={161 "local_extracted_archive_paths": local_extracted_archive_paths.get(split),162 "archives": [dl_manager.iter_archive(path) for path in archive_paths.get(split)],163 "meta_path": meta_paths[split],164 },165 ),166 )167 168 return split_generators169 170 def _generate_examples(self, local_extracted_archive_paths, archives, meta_path):171 data_fields = list(self._info().features.keys())172 metadata = {}173 with open(meta_path, encoding="utf-8") as f:174 reader = csv.DictReader(f, delimiter="\t", quoting=csv.QUOTE_NONE)175 for row in tqdm(reader, desc="Reading metadata..."):176 if not row["path"].endswith(".mp3"):177 row["path"] += ".mp3"178 # accent -> accents in CV 8.0179 if "accents" in row:180 row["accent"] = row["accents"]181 del row["accents"]182 # if data is incomplete, fill with empty values183 for field in data_fields:184 if field not in row:185 row[field] = ""186 metadata[row["path"]] = row187 188 for i, audio_archive in enumerate(archives):189 for path, file in audio_archive:190 _, filename = os.path.split(path)191 if filename in metadata:192 result = dict(metadata[filename])193 # set the audio feature and the path to the extracted file194 path = os.path.join(local_extracted_archive_paths[i], path) if local_extracted_archive_paths else path195 result["audio"] = {"path": path, "bytes": file.read()}196 result["path"] = path197 yield path, result