tts

Paused

App Files Files Community

zxsipola123456 commited on Jul 16, 2024

Commit

6a0a2ac

verified ·

1 Parent(s): 6c18768

Update app.py

Browse files

Files changed (1) hide show

app.py +33 -11

app.py CHANGED Viewed

@@ -28,17 +28,39 @@ async def text_to_speech_edge(text, language_code):
     return "语音合成完成：{}".format(text), tmp_path
 # 声音更改函数
 def voice_change(audio_in, audio_ref):
     samplerate1, data1 = wavfile.read(audio_in)
     samplerate2, data2 = wavfile.read(audio_ref)
-    write("./audio_in.wav", samplerate1, data1)
-    write("./audio_ref.wav", samplerate2, data2)
-    query_seq = knn_vc.get_features("./audio_in.wav")
-    matching_set = knn_vc.get_matching_set(["./audio_ref.wav"])
     out_wav = knn_vc.match(query_seq, matching_set, topk=4)
-    torchaudio.save('output.wav', out_wav[None], 16000)
-    return 'output.wav'
 # 文字转语音（OpenAI）
 def tts(text, model, voice, api_key):
@@ -66,11 +88,11 @@ def tts(text, model, voice, api_key):
 app = gr.Blocks()
 with app:
-    gr.Markdown("# <center>🌟 - OpenAI TTS + AI变声</center>")
-    gr.Markdown("### <center>🎶 地表最强文本转语音模型 + 3秒实时AI变声，支持中文！Powered by [OpenAI TTS](https://platform.openai.com/docs/guides/text-to-speech) and [KNN-VC](https://github.com/bshall/knn-vc) </center>")
-    with gr.Tab("🤗 OpenAI TTS"):
         with gr.Row(variant='panel'):
-            api_key = gr.Textbox(type='password', label='OpenAI API Key', placeholder='请在此填写您的OpenAI API Key')
             model = gr.Dropdown(choices=['tts-1','tts-1-hd'], label='请选择模型（tts-1推理更快，tts-1-hd音质更好）', value='tts-1')
             voice = gr.Dropdown(choices=['alloy', 'echo', 'fable', 'onyx', 'nova', 'shimmer'], label='请选择一个说话人', value='alloy')
         with gr.Row():
@@ -102,7 +124,7 @@ with app:
         btn_edge.click(lambda text, lang: anyio.run(text_to_speech_edge, text, lang), [input_text, language], [output_text, output_audio])
         btn_vc.click(voice_change, [output_audio, inp_vc], out_vc)
-    gr.Markdown("### <center>注意❗：请不要生成会对个人以及组织造成侵害的内容，此程序仅供科研、学习及个人娱乐使用。Get your OpenAI API Key [here](https://platform.openai.com/api-keys).</center>")
     gr.HTML('''
         <div class="footer">
          <p>Power by sipola </p>

     return "语音合成完成：{}".format(text), tmp_path
 # 声音更改函数
+#def voice_change(audio_in, audio_ref):
+    #samplerate1, data1 = wavfile.read(audio_in)
+    #samplerate2, data2 = wavfile.read(audio_ref)
+    #write("./audio_in.wav", samplerate1, data1)
+    #write("./audio_ref.wav", samplerate2, data2)
+    #query_seq = knn_vc.get_features("./audio_in.wav")
+    #matching_set = knn_vc.get_matching_set(["./audio_ref.wav"])
+    #out_wav = knn_vc.match(query_seq, matching_set, topk=4)
+    #torchaudio.save('output.wav', out_wav[None], 16000)
+    #return 'output.wav'
 def voice_change(audio_in, audio_ref):
     samplerate1, data1 = wavfile.read(audio_in)
     samplerate2, data2 = wavfile.read(audio_ref)
+    with tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as tmp_audio_in, \
+         tempfile.NamedTemporaryFile(delete=False, suffix=".wav") as tmp_audio_ref:
+        audio_in_path = tmp_audio_in.name
+        audio_ref_path = tmp_audio_ref.name
+        write(audio_in_path, samplerate1, data1)
+        write(audio_ref_path, samplerate2, data2)
+    query_seq = knn_vc.get_features(audio_in_path)
+    matching_set = knn_vc.get_matching_set([audio_ref_path])
     out_wav = knn_vc.match(query_seq, matching_set, topk=4)
+    # 确保 out_wav 是二维张量
+    if len(out_wav.shape) == 1:
+        out_wav = out_wav.unsqueeze(0)
+    output_path = 'output.wav'
+    torchaudio.save(output_path, out_wav, 16000)
+    return output_path
 # 文字转语音（OpenAI）
 def tts(text, model, voice, api_key):
 app = gr.Blocks()
 with app:
+    gr.Markdown("# <center>OpenAI TTS + 3秒实时AI变声+需要使用中转key</center>")
+    gr.Markdown("### <center>中转key购买地址https://buy.sipola.cn</center>")
+    with gr.Tab("TTS"):
         with gr.Row(variant='panel'):
+            api_key = gr.Textbox(type='password', label='API Key', placeholder='请在此填写您的API Key')
             model = gr.Dropdown(choices=['tts-1','tts-1-hd'], label='请选择模型（tts-1推理更快，tts-1-hd音质更好）', value='tts-1')
             voice = gr.Dropdown(choices=['alloy', 'echo', 'fable', 'onyx', 'nova', 'shimmer'], label='请选择一个说话人', value='alloy')
         with gr.Row():
         btn_edge.click(lambda text, lang: anyio.run(text_to_speech_edge, text, lang), [input_text, language], [output_text, output_audio])
         btn_vc.click(voice_change, [output_audio, inp_vc], out_vc)
+    gr.Markdown("### <center>注意获取中转API Key [here](https://buy.sipola.cn).</center>")
     gr.HTML('''
         <div class="footer">
          <p>Power by sipola </p>