kevinwang676/FreeVC-OpenAI-TTS
0
1import os2import torch3import librosa4import gradio as gr5from scipy.io.wavfile import write6from transformers import WavLMModel7 8import utils9from models import SynthesizerTrn10from mel_processing import mel_spectrogram_torch11from speaker_encoder.voice_encoder import SpeakerEncoder12 13'''14def get_wavlm():15 os.system('gdown https://drive.google.com/uc?id=12-cB34qCTvByWT-QtOcZaqwwO21FLSqU')16 shutil.move('WavLM-Large.pt', 'wavlm')17'''18 19device = torch.device("cuda" if torch.cuda.is_available() else "cpu")20 21print("Loading FreeVC...")22hps = utils.get_hparams_from_file("configs/freevc.json")23freevc = SynthesizerTrn(24 hps.data.filter_length // 2 + 1,25 hps.train.segment_size // hps.data.hop_length,26 **hps.model).to(device)27_ = freevc.eval()28_ = utils.load_checkpoint("checkpoints/freevc.pth", freevc, None)29smodel = SpeakerEncoder('speaker_encoder/ckpt/pretrained_bak_5805000.pt')30 31print("Loading FreeVC(24k)...")32hps = utils.get_hparams_from_file("configs/freevc-24.json")33freevc_24 = SynthesizerTrn(34 hps.data.filter_length // 2 + 1,35 hps.train.segment_size // hps.data.hop_length,36 **hps.model).to(device)37_ = freevc_24.eval()38_ = utils.load_checkpoint("checkpoints/freevc-24.pth", freevc_24, None)39 40print("Loading FreeVC-s...")41hps = utils.get_hparams_from_file("configs/freevc-s.json")42freevc_s = SynthesizerTrn(43 hps.data.filter_length // 2 + 1,44 hps.train.segment_size // hps.data.hop_length,45 **hps.model).to(device)46_ = freevc_s.eval()47_ = utils.load_checkpoint("checkpoints/freevc-s.pth", freevc_s, None)48 49print("Loading WavLM for content...")50cmodel = WavLMModel.from_pretrained("microsoft/wavlm-large").to(device)51 52 53from openai import OpenAI54 55import ffmpeg56import urllib.request57urllib.request.urlretrieve("https://download.openxlab.org.cn/models/Kevin676/rvc-models/weight/UVR-HP2.pth", "uvr5/uvr_model/UVR-HP2.pth")58urllib.request.urlretrieve("https://download.openxlab.org.cn/models/Kevin676/rvc-models/weight/UVR-HP5.pth", "uvr5/uvr_model/UVR-HP5.pth")59 60from uvr5.vr import AudioPre61weight_uvr5_root = "uvr5/uvr_model"62uvr5_names = []63for name in os.listdir(weight_uvr5_root):64 if name.endswith(".pth") or "onnx" in name:65 uvr5_names.append(name.replace(".pth", ""))66 67func = AudioPre68 69pre_fun_hp2 = func(70 agg=int(10),71 model_path=os.path.join(weight_uvr5_root, "UVR-HP2.pth"),72 device="cuda",73 is_half=True,74)75pre_fun_hp5 = func(76 agg=int(10),77 model_path=os.path.join(weight_uvr5_root, "UVR-HP5.pth"),78 device="cuda",79 is_half=True,80)81 82 83def convert(api_key, text, tgt, voice, save_path):84 model = "FreeVC (24kHz)"85 with torch.no_grad():86 # tgt87 wav_tgt, _ = librosa.load(tgt, sr=hps.data.sampling_rate)88 wav_tgt, _ = librosa.effects.trim(wav_tgt, top_db=20)89 if model == "FreeVC" or model == "FreeVC (24kHz)":90 g_tgt = smodel.embed_utterance(wav_tgt)91 g_tgt = torch.from_numpy(g_tgt).unsqueeze(0).to(device)92 else:93 wav_tgt = torch.from_numpy(wav_tgt).unsqueeze(0).to(device)94 mel_tgt = mel_spectrogram_torch(95 wav_tgt,96 hps.data.filter_length,97 hps.data.n_mel_channels,98 hps.data.sampling_rate,99 hps.data.hop_length,100 hps.data.win_length,101 hps.data.mel_fmin,102 hps.data.mel_fmax103 )104 # src105 client = OpenAI(api_key=api_key)106 107 response = client.audio.speech.create(108 model="tts-1-hd",109 voice=voice,110 input=text,111 )112 113 response.stream_to_file("output_openai.mp3")114 115 src = "output_openai.mp3"116 wav_src, _ = librosa.load(src, sr=hps.data.sampling_rate)117 wav_src = torch.from_numpy(wav_src).unsqueeze(0).to(device)118 c = cmodel(wav_src).last_hidden_state.transpose(1, 2).to(device)119 # infer120 if model == "FreeVC":121 audio = freevc.infer(c, g=g_tgt)122 elif model == "FreeVC-s":123 audio = freevc_s.infer(c, mel=mel_tgt)124 else:125 audio = freevc_24.infer(c, g=g_tgt)126 audio = audio[0][0].data.cpu().float().numpy()127 if model == "FreeVC" or model == "FreeVC-s":128 write(f"output/{save_path}.wav", hps.data.sampling_rate, audio)129 else:130 write(f"output/{save_path}.wav", 24000, audio)131 return f"output/{save_path}.wav"132 133 134class subtitle:135 def __init__(self,index:int, start_time, end_time, text:str):136 self.index = int(index)137 self.start_time = start_time138 self.end_time = end_time139 self.text = text.strip()140 def normalize(self,ntype:str,fps=30):141 if ntype=="prcsv":142 h,m,s,fs=(self.start_time.replace(';',':')).split(":")#seconds143 self.start_time=int(h)*3600+int(m)*60+int(s)+round(int(fs)/fps,2)144 h,m,s,fs=(self.end_time.replace(';',':')).split(":")145 self.end_time=int(h)*3600+int(m)*60+int(s)+round(int(fs)/fps,2)146 elif ntype=="srt":147 h,m,s=self.start_time.split(":")148 s=s.replace(",",".")149 self.start_time=int(h)*3600+int(m)*60+round(float(s),2)150 h,m,s=self.end_time.split(":")151 s=s.replace(",",".")152 self.end_time=int(h)*3600+int(m)*60+round(float(s),2)153 else:154 raise ValueError155 def add_offset(self,offset=0):156 self.start_time+=offset157 if self.start_time<0:158 self.start_time=0159 self.end_time+=offset160 if self.end_time<0:161 self.end_time=0162 def __str__(self) -> str:163 return f'id:{self.index},start:{self.start_time},end:{self.end_time},text:{self.text}'164 165def read_srt(uploaded_file):166 offset=0167 with open(uploaded_file.name,"r",encoding="utf-8") as f:168 file=f.readlines()169 subtitle_list=[]170 indexlist=[]171 filelength=len(file)172 for i in range(0,filelength):173 if " --> " in file[i]:174 is_st=True175 for char in file[i-1].strip().replace("\ufeff",""):176 if char not in ['0','1','2','3','4','5','6','7','8','9']:177 is_st=False178 break179 if is_st:180 indexlist.append(i) #get line id181 listlength=len(indexlist)182 for i in range(0,listlength-1):183 st,et=file[indexlist[i]].split(" --> ")184 id=int(file[indexlist[i]-1].strip().replace("\ufeff",""))185 text=""186 for x in range(indexlist[i]+1,indexlist[i+1]-2):187 text+=file[x]188 st=subtitle(id,st,et,text)189 st.normalize(ntype="srt")190 st.add_offset(offset=offset)191 subtitle_list.append(st)192 st,et=file[indexlist[-1]].split(" --> ")193 id=file[indexlist[-1]-1]194 text=""195 for x in range(indexlist[-1]+1,filelength):196 text+=file[x]197 st=subtitle(id,st,et,text)198 st.normalize(ntype="srt")199 st.add_offset(offset=offset)200 subtitle_list.append(st)201 return subtitle_list202 203from pydub import AudioSegment204 205def trim_audio(intervals, input_file_path, output_file_path):206 # load the audio file207 audio = AudioSegment.from_file(input_file_path)208 209 # iterate over the list of time intervals210 for i, (start_time, end_time) in enumerate(intervals):211 # extract the segment of the audio212 segment = audio[start_time*1000:end_time*1000]213 214 # construct the output file path215 output_file_path_i = f"{output_file_path}_{i}.wav"216 217 # export the segment to a file218 segment.export(output_file_path_i, format='wav')219 220import re221 222def sort_key(file_name):223 """Extract the last number in the file name for sorting."""224 numbers = re.findall(r'\d+', file_name)225 if numbers:226 return int(numbers[-1])227 return -1 # In case there's no number, this ensures it goes to the start.228 229 230def merge_audios(folder_path):231 output_file = "AI配音版.wav"232 # Get all WAV files in the folder233 files = [f for f in os.listdir(folder_path) if f.endswith('.wav')]234 # Sort files based on the last digit in their names235 sorted_files = sorted(files, key=sort_key)236 237 # Initialize an empty audio segment238 merged_audio = AudioSegment.empty()239 240 # Loop through each file, in order, and concatenate them241 for file in sorted_files:242 audio = AudioSegment.from_wav(os.path.join(folder_path, file))243 merged_audio += audio244 print(f"Merged: {file}")245 246 # Export the merged audio to a new file247 merged_audio.export(output_file, format="wav")248 return "AI配音版.wav"249 250import shutil251 252def convert_from_srt(apikey, filename, video_full, voice, split_model, multilingual):253 subtitle_list = read_srt(filename)254 255 if os.path.exists("audio_full.wav"):256 os.remove("audio_full.wav")257 258 ffmpeg.input(video_full).output("audio_full.wav", ac=2, ar=44100).run()259 260 if split_model=="UVR-HP2":261 pre_fun = pre_fun_hp2262 else:263 pre_fun = pre_fun_hp5264 265 filename = "output"266 pre_fun._path_audio_("audio_full.wav", f"./denoised/{split_model}/{filename}/", f"./denoised/{split_model}/{filename}/", "wav")267 if os.path.isdir("output"):268 shutil.rmtree("output")269 270 try:271 if multilingual==False:272 for i in subtitle_list:273 os.makedirs("output", exist_ok=True)274 trim_audio([[i.start_time, i.end_time]], f"./denoised/{split_model}/{filename}/vocal_audio_full.wav_10.wav", f"sliced_audio_{i.index}")275 print(f"正在合成第{i.index}条语音")276 print(f"语音内容:{i.text}")277 convert(apikey, i.text, f"sliced_audio_{i.index}_0.wav", voice, i.text + " " + str(i.index))278 else:279 for i in subtitle_list:280 os.makedirs("output", exist_ok=True)281 trim_audio([[i.start_time, i.end_time]], f"./denoised/{split_model}/{filename}/vocal_audio_full.wav_10.wav", f"sliced_audio_{i.index}")282 print(f"正在合成第{i.index}条语音")283 print(f"语音内容:{i.text.splitlines()[1]}")284 convert(apikey, i.text.splitlines()[1], f"sliced_audio_{i.index}_0.wav", voice, i.text.splitlines()[1] + " " + str(i.index))285 except Exception:286 pass287 288 return merge_audios("output")289 290 291with gr.Blocks() as app:292 gr.Markdown("# <center>🌊💕🎶 OpenAI TTS - SRT文件一键AI配音</center>")293 gr.Markdown("### <center>🌟 只需上传SRT文件和原版配音文件即可,每次一集视频AI自动配音!Developed by Kevin Wang </center>")294 with gr.Row():295 with gr.Column():296 inp0 = gr.Textbox(type='password', label='请输入您的OpenAI API Key')297 inp1 = gr.File(file_count="single", label="请上传一集视频对应的SRT文件")298 inp2 = gr.Video(label="请上传一集包含原声配音的视频", info="需要是.mp4视频文件")299 inp3 = gr.Dropdown(choices=['alloy', 'echo', 'fable', 'onyx', 'nova', 'shimmer'], label='请选择一个说话人提供基础音色', info="试听音色链接:https://platform.openai.com/docs/guides/text-to-speech/voice-options", value='alloy')300 inp4 = gr.Dropdown(label="请选择用于分离伴奏的模型", info="UVR-HP5去除背景音乐效果更好,但会对人声造成一定的损伤", choices=["UVR-HP2", "UVR-HP5"], value="UVR-HP5")301 inp5 = gr.Checkbox(label="SRT文件是否为双语字幕", info="若为双语字幕,请打勾选择(SRT文件中需要先出现中文字幕,后英文字幕;中英字幕各占一行)")302 btn = gr.Button("一键开启AI配音吧💕", variant="primary")303 with gr.Column():304 out1 = gr.Audio(label="为您生成的AI完整配音", type="filepath")305 306 btn.click(convert_from_srt, [inp0, inp1, inp2, inp3, inp4, inp5], [out1])307 308 gr.Markdown("### <center>注意❗:请勿生成会对任何个人或组织造成侵害的内容,请尊重他人的著作权和知识产权。用户对此程序的任何使用行为与程序开发者无关。</center>")309 gr.HTML('''310 <div class="footer">311 <p>🌊🏞️🎶 - 江水东流急,滔滔无尽声。 明·顾璘312 </p>313 </div>314 ''')315 316app.launch(share=True, show_error=True)