CoolFace
Apppublic

kevinwang676/FreeVC-OpenAI-TTS

sourceHugging Facemitupdated 2y agoView on Hugging Face
0likes
app.py316 linesDownload Raw Back to root
1import os2import torch3import librosa4import gradio as gr5from scipy.io.wavfile import write6from transformers import WavLMModel7 8import utils9from models import SynthesizerTrn10from mel_processing import mel_spectrogram_torch11from speaker_encoder.voice_encoder import SpeakerEncoder12 13'''14def get_wavlm():15    os.system('gdown https://drive.google.com/uc?id=12-cB34qCTvByWT-QtOcZaqwwO21FLSqU')16    shutil.move('WavLM-Large.pt', 'wavlm')17'''18 19device = torch.device("cuda" if torch.cuda.is_available() else "cpu")20 21print("Loading FreeVC...")22hps = utils.get_hparams_from_file("configs/freevc.json")23freevc = SynthesizerTrn(24    hps.data.filter_length // 2 + 1,25    hps.train.segment_size // hps.data.hop_length,26    **hps.model).to(device)27_ = freevc.eval()28_ = utils.load_checkpoint("checkpoints/freevc.pth", freevc, None)29smodel = SpeakerEncoder('speaker_encoder/ckpt/pretrained_bak_5805000.pt')30 31print("Loading FreeVC(24k)...")32hps = utils.get_hparams_from_file("configs/freevc-24.json")33freevc_24 = SynthesizerTrn(34    hps.data.filter_length // 2 + 1,35    hps.train.segment_size // hps.data.hop_length,36    **hps.model).to(device)37_ = freevc_24.eval()38_ = utils.load_checkpoint("checkpoints/freevc-24.pth", freevc_24, None)39 40print("Loading FreeVC-s...")41hps = utils.get_hparams_from_file("configs/freevc-s.json")42freevc_s = SynthesizerTrn(43    hps.data.filter_length // 2 + 1,44    hps.train.segment_size // hps.data.hop_length,45    **hps.model).to(device)46_ = freevc_s.eval()47_ = utils.load_checkpoint("checkpoints/freevc-s.pth", freevc_s, None)48 49print("Loading WavLM for content...")50cmodel = WavLMModel.from_pretrained("microsoft/wavlm-large").to(device)51 52 53from openai import OpenAI54 55import ffmpeg56import urllib.request57urllib.request.urlretrieve("https://download.openxlab.org.cn/models/Kevin676/rvc-models/weight/UVR-HP2.pth", "uvr5/uvr_model/UVR-HP2.pth")58urllib.request.urlretrieve("https://download.openxlab.org.cn/models/Kevin676/rvc-models/weight/UVR-HP5.pth", "uvr5/uvr_model/UVR-HP5.pth")59 60from uvr5.vr import AudioPre61weight_uvr5_root = "uvr5/uvr_model"62uvr5_names = []63for name in os.listdir(weight_uvr5_root):64    if name.endswith(".pth") or "onnx" in name:65        uvr5_names.append(name.replace(".pth", ""))66 67func = AudioPre68 69pre_fun_hp2 = func(70  agg=int(10),71  model_path=os.path.join(weight_uvr5_root, "UVR-HP2.pth"),72  device="cuda",73  is_half=True,74)75pre_fun_hp5 = func(76  agg=int(10),77  model_path=os.path.join(weight_uvr5_root, "UVR-HP5.pth"),78  device="cuda",79  is_half=True,80)81 82 83def convert(api_key, text, tgt, voice, save_path):84    model = "FreeVC (24kHz)"85    with torch.no_grad():86        # tgt87        wav_tgt, _ = librosa.load(tgt, sr=hps.data.sampling_rate)88        wav_tgt, _ = librosa.effects.trim(wav_tgt, top_db=20)89        if model == "FreeVC" or model == "FreeVC (24kHz)":90            g_tgt = smodel.embed_utterance(wav_tgt)91            g_tgt = torch.from_numpy(g_tgt).unsqueeze(0).to(device)92        else:93            wav_tgt = torch.from_numpy(wav_tgt).unsqueeze(0).to(device)94            mel_tgt = mel_spectrogram_torch(95                wav_tgt,96                hps.data.filter_length,97                hps.data.n_mel_channels,98                hps.data.sampling_rate,99                hps.data.hop_length,100                hps.data.win_length,101                hps.data.mel_fmin,102                hps.data.mel_fmax103            )104        # src105        client = OpenAI(api_key=api_key)106 107        response = client.audio.speech.create(108            model="tts-1-hd",109            voice=voice,110            input=text,111        )112 113        response.stream_to_file("output_openai.mp3")114 115        src = "output_openai.mp3"116        wav_src, _ = librosa.load(src, sr=hps.data.sampling_rate)117        wav_src = torch.from_numpy(wav_src).unsqueeze(0).to(device)118        c = cmodel(wav_src).last_hidden_state.transpose(1, 2).to(device)119        # infer120        if model == "FreeVC":121            audio = freevc.infer(c, g=g_tgt)122        elif model == "FreeVC-s":123            audio = freevc_s.infer(c, mel=mel_tgt)124        else:125            audio = freevc_24.infer(c, g=g_tgt)126        audio = audio[0][0].data.cpu().float().numpy()127        if model == "FreeVC" or model == "FreeVC-s":128            write(f"output/{save_path}.wav", hps.data.sampling_rate, audio)129        else:130            write(f"output/{save_path}.wav", 24000, audio)131    return f"output/{save_path}.wav"132 133 134class subtitle:135    def __init__(self,index:int, start_time, end_time, text:str):136        self.index = int(index)137        self.start_time = start_time138        self.end_time = end_time139        self.text = text.strip()140    def normalize(self,ntype:str,fps=30):141         if ntype=="prcsv":142              h,m,s,fs=(self.start_time.replace(';',':')).split(":")#seconds143              self.start_time=int(h)*3600+int(m)*60+int(s)+round(int(fs)/fps,2)144              h,m,s,fs=(self.end_time.replace(';',':')).split(":")145              self.end_time=int(h)*3600+int(m)*60+int(s)+round(int(fs)/fps,2)146         elif ntype=="srt":147             h,m,s=self.start_time.split(":")148             s=s.replace(",",".")149             self.start_time=int(h)*3600+int(m)*60+round(float(s),2)150             h,m,s=self.end_time.split(":")151             s=s.replace(",",".")152             self.end_time=int(h)*3600+int(m)*60+round(float(s),2)153         else:154             raise ValueError155    def add_offset(self,offset=0):156        self.start_time+=offset157        if self.start_time<0:158            self.start_time=0159        self.end_time+=offset160        if self.end_time<0:161            self.end_time=0162    def __str__(self) -> str:163        return f'id:{self.index},start:{self.start_time},end:{self.end_time},text:{self.text}'164 165def read_srt(uploaded_file):166    offset=0167    with open(uploaded_file.name,"r",encoding="utf-8") as f:168        file=f.readlines()169    subtitle_list=[]170    indexlist=[]171    filelength=len(file)172    for i in range(0,filelength):173        if " --> " in file[i]:174            is_st=True175            for char in file[i-1].strip().replace("\ufeff",""):176                if char not in ['0','1','2','3','4','5','6','7','8','9']:177                    is_st=False178                    break179            if is_st:180                indexlist.append(i) #get line id181    listlength=len(indexlist)182    for i in range(0,listlength-1):183        st,et=file[indexlist[i]].split(" --> ")184        id=int(file[indexlist[i]-1].strip().replace("\ufeff",""))185        text=""186        for x in range(indexlist[i]+1,indexlist[i+1]-2):187            text+=file[x]188        st=subtitle(id,st,et,text)189        st.normalize(ntype="srt")190        st.add_offset(offset=offset)191        subtitle_list.append(st)192    st,et=file[indexlist[-1]].split(" --> ")193    id=file[indexlist[-1]-1]194    text=""195    for x in range(indexlist[-1]+1,filelength):196        text+=file[x]197    st=subtitle(id,st,et,text)198    st.normalize(ntype="srt")199    st.add_offset(offset=offset)200    subtitle_list.append(st)201    return subtitle_list202 203from pydub import AudioSegment204 205def trim_audio(intervals, input_file_path, output_file_path):206    # load the audio file207    audio = AudioSegment.from_file(input_file_path)208 209    # iterate over the list of time intervals210    for i, (start_time, end_time) in enumerate(intervals):211        # extract the segment of the audio212        segment = audio[start_time*1000:end_time*1000]213 214        # construct the output file path215        output_file_path_i = f"{output_file_path}_{i}.wav"216 217        # export the segment to a file218        segment.export(output_file_path_i, format='wav')219 220import re221 222def sort_key(file_name):223    """Extract the last number in the file name for sorting."""224    numbers = re.findall(r'\d+', file_name)225    if numbers:226        return int(numbers[-1])227    return -1  # In case there's no number, this ensures it goes to the start.228 229 230def merge_audios(folder_path):231    output_file = "AI配音版.wav"232    # Get all WAV files in the folder233    files = [f for f in os.listdir(folder_path) if f.endswith('.wav')]234    # Sort files based on the last digit in their names235    sorted_files = sorted(files, key=sort_key)236    237    # Initialize an empty audio segment238    merged_audio = AudioSegment.empty()239    240    # Loop through each file, in order, and concatenate them241    for file in sorted_files:242        audio = AudioSegment.from_wav(os.path.join(folder_path, file))243        merged_audio += audio244        print(f"Merged: {file}")245    246    # Export the merged audio to a new file247    merged_audio.export(output_file, format="wav")248    return "AI配音版.wav"249 250import shutil251 252def convert_from_srt(apikey, filename, video_full, voice, split_model, multilingual):253    subtitle_list = read_srt(filename)254    255    if os.path.exists("audio_full.wav"):256        os.remove("audio_full.wav")257 258    ffmpeg.input(video_full).output("audio_full.wav", ac=2, ar=44100).run()259    260    if split_model=="UVR-HP2":261        pre_fun = pre_fun_hp2262    else:263        pre_fun = pre_fun_hp5264 265    filename = "output"266    pre_fun._path_audio_("audio_full.wav", f"./denoised/{split_model}/{filename}/", f"./denoised/{split_model}/{filename}/", "wav")267    if os.path.isdir("output"):268        shutil.rmtree("output")269 270    try:271        if multilingual==False:272            for i in subtitle_list:273                os.makedirs("output", exist_ok=True)274                trim_audio([[i.start_time, i.end_time]], f"./denoised/{split_model}/{filename}/vocal_audio_full.wav_10.wav", f"sliced_audio_{i.index}")275                print(f"正在合成第{i.index}条语音")276                print(f"语音内容:{i.text}")277                convert(apikey, i.text, f"sliced_audio_{i.index}_0.wav", voice, i.text + " " + str(i.index))278        else:279            for i in subtitle_list:280                os.makedirs("output", exist_ok=True)281                trim_audio([[i.start_time, i.end_time]], f"./denoised/{split_model}/{filename}/vocal_audio_full.wav_10.wav", f"sliced_audio_{i.index}")282                print(f"正在合成第{i.index}条语音")283                print(f"语音内容:{i.text.splitlines()[1]}")284                convert(apikey, i.text.splitlines()[1], f"sliced_audio_{i.index}_0.wav", voice, i.text.splitlines()[1] + " " + str(i.index))285    except Exception:286        pass287        288    return merge_audios("output")289 290 291with gr.Blocks() as app:292    gr.Markdown("# <center>🌊💕🎶 OpenAI TTS - SRT文件一键AI配音</center>")293    gr.Markdown("### <center>🌟 只需上传SRT文件和原版配音文件即可,每次一集视频AI自动配音!Developed by Kevin Wang </center>")294    with gr.Row():295        with gr.Column():296            inp0 = gr.Textbox(type='password', label='请输入您的OpenAI API Key')297            inp1 = gr.File(file_count="single", label="请上传一集视频对应的SRT文件")298            inp2 = gr.Video(label="请上传一集包含原声配音的视频", info="需要是.mp4视频文件")299            inp3 = gr.Dropdown(choices=['alloy', 'echo', 'fable', 'onyx', 'nova', 'shimmer'], label='请选择一个说话人提供基础音色', info="试听音色链接:https://platform.openai.com/docs/guides/text-to-speech/voice-options", value='alloy')300            inp4 = gr.Dropdown(label="请选择用于分离伴奏的模型", info="UVR-HP5去除背景音乐效果更好,但会对人声造成一定的损伤", choices=["UVR-HP2", "UVR-HP5"], value="UVR-HP5")301            inp5 = gr.Checkbox(label="SRT文件是否为双语字幕", info="若为双语字幕,请打勾选择(SRT文件中需要先出现中文字幕,后英文字幕;中英字幕各占一行)")302            btn = gr.Button("一键开启AI配音吧💕", variant="primary")303        with gr.Column():304            out1 = gr.Audio(label="为您生成的AI完整配音", type="filepath")305 306        btn.click(convert_from_srt, [inp0, inp1, inp2, inp3, inp4, inp5], [out1])307        308    gr.Markdown("### <center>注意❗:请勿生成会对任何个人或组织造成侵害的内容,请尊重他人的著作权和知识产权。用户对此程序的任何使用行为与程序开发者无关。</center>")309    gr.HTML('''310        <div class="footer">311                    <p>🌊🏞️🎶 - 江水东流急,滔滔无尽声。 明·顾璘312                    </p>313        </div>314    ''')315 316app.launch(share=True, show_error=True)