styletts2-voice-cloning

Runtime error

App Files Files Community

mrfakename commited on Nov 23, 2023

Commit

4b8ade9

•

1 Parent(s): 6c306f4

Revert "Allow acronym expansion"

Browse files

This reverts commit 6621613498a6b6109c28ff85e028b1a8ab6824a5.

Files changed (2) hide show

app.py +13 -31
requirements.txt +1 -3

app.py CHANGED Viewed

@@ -6,20 +6,6 @@ import os
 # from tortoise.utils.text import split_and_recombine_text
 import numpy as np
 import pickle
-import spacy
-from scispacy.abbreviation import AbbreviationDetector
-nlp = spacy.load("en_core_sci_sm")
-# Add the abbreviation pipe to the spacy pipeline.
-nlp.add_pipe("abbreviation_detector")
-def replace_acronyms(text):
-    doc = nlp(text)
-    altered_tok = [tok.text for tok in doc]
-    for abrv in doc._.abbreviations:
-        altered_tok[abrv.start] = str(abrv._.long_form)
-    return(" ".join(altered_tok))
 theme = gr.themes.Base(
     font=[gr.themes.GoogleFont('Libre Franklin'), gr.themes.GoogleFont('Public Sans'), 'system-ui', 'sans-serif'],
 )
@@ -34,14 +20,12 @@ global_phonemizer = phonemizer.backend.EspeakBackend(language='en-us', preserve_
 # else:
 for v in voicelist:
     voices[v] = styletts2importable.compute_style(f'voices/{v}.wav')
-def synthesize(text, voice, multispeakersteps, msexpand):
     if text.strip() == "":
         raise gr.Error("You must enter some text")
     # if len(global_phonemizer.phonemize([text])) > 300:
     if len(text) > 300:
         raise gr.Error("Text must be under 300 characters")
-    if msexpand:
-        text = replace_acronyms(text)
     v = voice.lower()
     # return (24000, styletts2importable.inference(text, voices[v], alpha=0.3, beta=0.7, diffusion_steps=7, embedding_scale=1))
     return (24000, styletts2importable.inference(text, voices[v], alpha=0.3, beta=0.7, diffusion_steps=multispeakersteps, embedding_scale=1))
@@ -69,14 +53,14 @@ def clsynthesize(text, voice, vcsteps):
         raise gr.Error("Text must be under 400 characters")
     # return (24000, styletts2importable.inference(text, styletts2importable.compute_style(voice), alpha=0.3, beta=0.7, diffusion_steps=20, embedding_scale=1))
     return (24000, styletts2importable.inference(text, styletts2importable.compute_style(voice), alpha=0.3, beta=0.7, diffusion_steps=vcsteps, embedding_scale=1))
-def ljsynthesize(text, steps):
     if text.strip() == "":
         raise gr.Error("You must enter some text")
     # if global_phonemizer.phonemize([text]) > 300:
     if len(text) > 400:
         raise gr.Error("Text must be under 400 characters")
     noise = torch.randn(1,1,256).to('cuda' if torch.cuda.is_available() else 'cpu')
-    return (24000, ljspeechimportable.inference(text, noise, diffusion_steps=steps, embedding_scale=1))
 with gr.Blocks() as vctk: # just realized it isn't vctk but libritts but i'm too lazy to change it rn
@@ -85,12 +69,11 @@ with gr.Blocks() as vctk: # just realized it isn't vctk but libritts but i'm too
             inp = gr.Textbox(label="Text", info="What would you like StyleTTS 2 to read? It works better on full sentences.", interactive=True)
             voice = gr.Dropdown(voicelist, label="Voice", info="Select a default voice.", value='m-us-2', interactive=True)
             multispeakersteps = gr.Slider(minimum=5, maximum=15, value=7, step=1, label="Diffusion Steps", info="Higher = better quality, but slower", interactive=True)
-            msexpand = gr.Checkbox(label="Expand acronyms", info="Expand acronyms using SciSpacy algorithm")
             # use_gruut = gr.Checkbox(label="Use alternate phonemizer (Gruut) - Experimental")
         with gr.Column(scale=1):
             btn = gr.Button("Synthesize", variant="primary")
             audio = gr.Audio(interactive=False, label="Synthesized Audio")
-            btn.click(synthesize, inputs=[inp, voice, multispeakersteps, msexpand], outputs=[audio], concurrency_limit=4)
 with gr.Blocks() as clone:
     with gr.Row():
         with gr.Column(scale=1):
@@ -100,16 +83,7 @@ with gr.Blocks() as clone:
         with gr.Column(scale=1):
             clbtn = gr.Button("Synthesize", variant="primary")
             claudio = gr.Audio(interactive=False, label="Synthesized Audio")
-            clbtn.click(clsynthesize, inputs=[clinp, clvoice, vcsteps], outputs=[claudio], concurrency_limit=2)
-with gr.Blocks() as lj:
-    with gr.Row():
-        with gr.Column(scale=1):
-            ljinp = gr.Textbox(label="Text", info="What would you like StyleTTS 2 to read? It works better on full sentences.", interactive=True)
-        with gr.Column(scale=1):
-            ljbtn = gr.Button("Synthesize", variant="primary")
-            ljaudio = gr.Audio(interactive=False, label="Synthesized Audio")
-            ljsteps = gr.Slider(minimum=5, maximum=15, value=7, step=1, label="Diffusion Steps", info="Higher = better quality, but slower", interactive=True)
-            ljbtn.click(ljsynthesize, inputs=[ljinp, ljsteps], outputs=[ljaudio], concurrency_limit=4)
 # with gr.Blocks() as longText:
 #     with gr.Row():
 #         with gr.Column(scale=1):
@@ -121,6 +95,14 @@ with gr.Blocks() as lj:
 #             lngbtn = gr.Button("Synthesize", variant="primary")
 #             lngaudio = gr.Audio(interactive=False, label="Synthesized Audio")
 #             lngbtn.click(longsynthesize, inputs=[lnginp, lngvoice, lngsteps, lngpwd], outputs=[lngaudio], concurrency_limit=4)
 with gr.Blocks(title="StyleTTS 2", css="footer{display:none !important}", theme=theme) as demo:
     gr.Markdown("""# StyleTTS 2

 # from tortoise.utils.text import split_and_recombine_text
 import numpy as np
 import pickle
 theme = gr.themes.Base(
     font=[gr.themes.GoogleFont('Libre Franklin'), gr.themes.GoogleFont('Public Sans'), 'system-ui', 'sans-serif'],
 )
 # else:
 for v in voicelist:
     voices[v] = styletts2importable.compute_style(f'voices/{v}.wav')
+def synthesize(text, voice, multispeakersteps):
     if text.strip() == "":
         raise gr.Error("You must enter some text")
     # if len(global_phonemizer.phonemize([text])) > 300:
     if len(text) > 300:
         raise gr.Error("Text must be under 300 characters")
     v = voice.lower()
     # return (24000, styletts2importable.inference(text, voices[v], alpha=0.3, beta=0.7, diffusion_steps=7, embedding_scale=1))
     return (24000, styletts2importable.inference(text, voices[v], alpha=0.3, beta=0.7, diffusion_steps=multispeakersteps, embedding_scale=1))
         raise gr.Error("Text must be under 400 characters")
     # return (24000, styletts2importable.inference(text, styletts2importable.compute_style(voice), alpha=0.3, beta=0.7, diffusion_steps=20, embedding_scale=1))
     return (24000, styletts2importable.inference(text, styletts2importable.compute_style(voice), alpha=0.3, beta=0.7, diffusion_steps=vcsteps, embedding_scale=1))
+def ljsynthesize(text):
     if text.strip() == "":
         raise gr.Error("You must enter some text")
     # if global_phonemizer.phonemize([text]) > 300:
     if len(text) > 400:
         raise gr.Error("Text must be under 400 characters")
     noise = torch.randn(1,1,256).to('cuda' if torch.cuda.is_available() else 'cpu')
+    return (24000, ljspeechimportable.inference(text, noise, diffusion_steps=7, embedding_scale=1))
 with gr.Blocks() as vctk: # just realized it isn't vctk but libritts but i'm too lazy to change it rn
             inp = gr.Textbox(label="Text", info="What would you like StyleTTS 2 to read? It works better on full sentences.", interactive=True)
             voice = gr.Dropdown(voicelist, label="Voice", info="Select a default voice.", value='m-us-2', interactive=True)
             multispeakersteps = gr.Slider(minimum=5, maximum=15, value=7, step=1, label="Diffusion Steps", info="Higher = better quality, but slower", interactive=True)
             # use_gruut = gr.Checkbox(label="Use alternate phonemizer (Gruut) - Experimental")
         with gr.Column(scale=1):
             btn = gr.Button("Synthesize", variant="primary")
             audio = gr.Audio(interactive=False, label="Synthesized Audio")
+            btn.click(synthesize, inputs=[inp, voice, multispeakersteps], outputs=[audio], concurrency_limit=4)
 with gr.Blocks() as clone:
     with gr.Row():
         with gr.Column(scale=1):
         with gr.Column(scale=1):
             clbtn = gr.Button("Synthesize", variant="primary")
             claudio = gr.Audio(interactive=False, label="Synthesized Audio")
+            clbtn.click(clsynthesize, inputs=[clinp, clvoice, vcsteps], outputs=[claudio], concurrency_limit=4)
 # with gr.Blocks() as longText:
 #     with gr.Row():
 #         with gr.Column(scale=1):
 #             lngbtn = gr.Button("Synthesize", variant="primary")
 #             lngaudio = gr.Audio(interactive=False, label="Synthesized Audio")
 #             lngbtn.click(longsynthesize, inputs=[lnginp, lngvoice, lngsteps, lngpwd], outputs=[lngaudio], concurrency_limit=4)
+with gr.Blocks() as lj:
+    with gr.Row():
+        with gr.Column(scale=1):
+            ljinp = gr.Textbox(label="Text", info="What would you like StyleTTS 2 to read? It works better on full sentences.", interactive=True)
+        with gr.Column(scale=1):
+            ljbtn = gr.Button("Synthesize", variant="primary")
+            ljaudio = gr.Audio(interactive=False, label="Synthesized Audio")
+            ljbtn.click(ljsynthesize, inputs=[ljinp], outputs=[ljaudio], concurrency_limit=4)
 with gr.Blocks(title="StyleTTS 2", css="footer{display:none !important}", theme=theme) as demo:
     gr.Markdown("""# StyleTTS 2

requirements.txt CHANGED Viewed

@@ -20,6 +20,4 @@ phonemizer
 cached-path
 gradio
 gruut
-# tortoise-tts
-spacy
-scispacy

 cached-path
 gradio
 gruut
+# tortoise-tts