CoolFace
Datasetpublic

jstack32/LatinAccents

Dataset Card for Dataset Name This dataset card aims to be a base template for new datasets. It has been generated using this raw template. Dataset Details Dataset Description Curated by: [More Information Needed] Funded by [optional]: [More Information Needed] Shared by [optional]: [More Information Needed] Language(s) (NLP): [More Information Needed] License: [More Information Needed] Dataset Sources [optional]… See the full description on the dataset page: https://huggingface.co/datasets/jstack32/LatinAccents.

sourceHugging Faceapache-2.0updated 3y agoView on Hugging Face
0likes13downloads
CommonVoice.py197 linesDownload Raw Back to root
1# coding=utf-82# Copyright 2022 The HuggingFace Datasets Authors and the current dataset script contributor.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8#     http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15""" Common Voice Dataset"""16 17 18import csv19import os20import json21 22import datasets23from datasets.utils.py_utils import size_str24from tqdm import tqdm25 26from .languages import LANGUAGES27from .release_stats import STATS28 29 30_CITATION = """\31@inproceedings{commonvoice:2020,32  author = {Ardila, R. and Branson, M. and Davis, K. and Henretty, M. and Kohler, M. and Meyer, J. and Morais, R. and Saunders, L. and Tyers, F. M. and Weber, G.},33  title = {Common Voice: A Massively-Multilingual Speech Corpus},34  booktitle = {Proceedings of the 12th Conference on Language Resources and Evaluation (LREC 2020)},35  pages = {4211--4215},36  year = 202037}38"""39 40_HOMEPAGE = "https://commonvoice.mozilla.org/en/datasets"41 42_LICENSE = "https://creativecommons.org/publicdomain/zero/1.0/"43 44# TODO: change "streaming" to "main" after merge!45_BASE_URL = "https://huggingface.co/datasets/mozilla-foundation/common_voice_11_0/resolve/main/"46 47_AUDIO_URL = _BASE_URL + "audio/{lang}/{split}/{lang}_{split}_{shard_idx}.tar"48 49_TRANSCRIPT_URL = _BASE_URL + "transcript/{lang}/{split}.tsv"50 51_N_SHARDS_URL = _BASE_URL + "n_shards.json"52 53 54class CommonVoiceConfig(datasets.BuilderConfig):55    """BuilderConfig for CommonVoice."""56 57    def __init__(self, name, version, **kwargs):58        self.language = kwargs.pop("language", None)59        self.release_date = kwargs.pop("release_date", None)60        self.num_clips = kwargs.pop("num_clips", None)61        self.num_speakers = kwargs.pop("num_speakers", None)62        self.validated_hr = kwargs.pop("validated_hr", None)63        self.total_hr = kwargs.pop("total_hr", None)64        self.size_bytes = kwargs.pop("size_bytes", None)65        self.size_human = size_str(self.size_bytes)66        description = (67            f"Common Voice speech to text dataset in {self.language} released on {self.release_date}. "68            f"The dataset comprises {self.validated_hr} hours of validated transcribed speech data "69            f"out of {self.total_hr} hours in total from {self.num_speakers} speakers. "70            f"The dataset contains {self.num_clips} audio clips and has a size of {self.size_human}."71        )72        super(CommonVoiceConfig, self).__init__(73            name=name,74            version=datasets.Version(version),75            description=description,76            **kwargs,77        )78 79 80class CommonVoice(datasets.GeneratorBasedBuilder):81    DEFAULT_WRITER_BATCH_SIZE = 100082 83    BUILDER_CONFIGS = [84        CommonVoiceConfig(85            name=lang,86            version=STATS["version"],87            language=LANGUAGES[lang],88            release_date=STATS["date"],89            num_clips=lang_stats["clips"],90            num_speakers=lang_stats["users"],91            validated_hr=float(lang_stats["validHrs"]) if lang_stats["validHrs"] else None,92            total_hr=float(lang_stats["totalHrs"]) if lang_stats["totalHrs"] else None,93            size_bytes=int(lang_stats["size"]) if lang_stats["size"] else None,94        )95        for lang, lang_stats in STATS["locales"].items()96    ]97 98    def _info(self):99        total_languages = len(STATS["locales"])100        total_valid_hours = STATS["totalValidHrs"]101        description = (102            "Common Voice is Mozilla's initiative to help teach machines how real people speak. "103            f"The dataset currently consists of {total_valid_hours} validated hours of speech "104            f" in {total_languages} languages, but more voices and languages are always added."105        )106        features = datasets.Features(107            {108                "client_id": datasets.Value("string"),109                "path": datasets.Value("string"),110                "audio": datasets.features.Audio(sampling_rate=48_000),111                "sentence": datasets.Value("string"),112                "up_votes": datasets.Value("int64"),113                "down_votes": datasets.Value("int64"),114                "age": datasets.Value("string"),115                "gender": datasets.Value("string"),116                "accent": datasets.Value("string"),117                "locale": datasets.Value("string"),118                "segment": datasets.Value("string"),119            }120        )121 122        return datasets.DatasetInfo(123            description=description,124            features=features,125            supervised_keys=None,126            homepage=_HOMEPAGE,127            license=_LICENSE,128            citation=_CITATION,129            version=self.config.version,130        )131 132    def _split_generators(self, dl_manager):133        lang = self.config.name134        n_shards_path = dl_manager.download_and_extract(_N_SHARDS_URL)135        with open(n_shards_path, encoding="utf-8") as f:136            n_shards = json.load(f)137 138        audio_urls = {}139        splits = ("train", "dev", "test", "other", "invalidated")140        for split in splits:141            audio_urls[split] = [142                _AUDIO_URL.format(lang=lang, split=split, shard_idx=i) for i in range(n_shards[lang][split])143            ]144        archive_paths = dl_manager.download(audio_urls)145        local_extracted_archive_paths = dl_manager.extract(archive_paths) if not dl_manager.is_streaming else {}146 147        meta_urls = {split: _TRANSCRIPT_URL.format(lang=lang, split=split) for split in splits}148        meta_paths = dl_manager.download_and_extract(meta_urls)149 150        split_generators = []151        split_names = {152            "train": datasets.Split.TRAIN,153            "dev": datasets.Split.VALIDATION,154            "test": datasets.Split.TEST,155        }156        for split in splits:157            split_generators.append(158                datasets.SplitGenerator(159                    name=split_names.get(split, split),160                    gen_kwargs={161                        "local_extracted_archive_paths": local_extracted_archive_paths.get(split),162                        "archives": [dl_manager.iter_archive(path) for path in archive_paths.get(split)],163                        "meta_path": meta_paths[split],164                    },165                ),166            )167 168        return split_generators169 170    def _generate_examples(self, local_extracted_archive_paths, archives, meta_path):171        data_fields = list(self._info().features.keys())172        metadata = {}173        with open(meta_path, encoding="utf-8") as f:174            reader = csv.DictReader(f, delimiter="\t", quoting=csv.QUOTE_NONE)175            for row in tqdm(reader, desc="Reading metadata..."):176                if not row["path"].endswith(".mp3"):177                    row["path"] += ".mp3"178                # accent -> accents in CV 8.0179                if "accents" in row:180                    row["accent"] = row["accents"]181                    del row["accents"]182                # if data is incomplete, fill with empty values183                for field in data_fields:184                    if field not in row:185                        row[field] = ""186                metadata[row["path"]] = row187 188        for i, audio_archive in enumerate(archives):189            for path, file in audio_archive:190                _, filename = os.path.split(path)191                if filename in metadata:192                    result = dict(metadata[filename])193                    # set the audio feature and the path to the extracted file194                    path = os.path.join(local_extracted_archive_paths[i], path) if local_extracted_archive_paths else path195                    result["audio"] = {"path": path, "bytes": file.read()}196                    result["path"] = path197                    yield path, result