speech-to-speech-translation

Runtime error

App Files Files Community

davidggphy commited on Aug 25, 2023

Commit

245bced

1 Parent(s): 3b47781

Adapt to Whisper (es) + Bark (es)

Browse files

Files changed (1) hide show

app.py +37 -21

app.py CHANGED Viewed

@@ -1,42 +1,58 @@
 import gradio as gr
 import numpy as np
 import torch
-from datasets import load_dataset
-from transformers import SpeechT5ForTextToSpeech, SpeechT5HifiGan, SpeechT5Processor, pipeline
 device = "cuda:0" if torch.cuda.is_available() else "cpu"
-# load speech translation checkpoint
-asr_pipe = pipeline("automatic-speech-recognition", model="openai/whisper-base", device=device)
-# load text-to-speech checkpoint and speaker embeddings
-processor = SpeechT5Processor.from_pretrained("microsoft/speecht5_tts")
-model = SpeechT5ForTextToSpeech.from_pretrained("microsoft/speecht5_tts").to(device)
-vocoder = SpeechT5HifiGan.from_pretrained("microsoft/speecht5_hifigan").to(device)
-embeddings_dataset = load_dataset("Matthijs/cmu-arctic-xvectors", split="validation")
-speaker_embeddings = torch.tensor(embeddings_dataset[7306]["xvector"]).unsqueeze(0)
-def translate(audio):
-    outputs = asr_pipe(audio, max_new_tokens=256, generate_kwargs={"task": "translate"})
-    return outputs["text"]
-def synthesise(text):
-    inputs = processor(text=text, return_tensors="pt")
-    speech = model.generate_speech(inputs["input_ids"].to(device), speaker_embeddings.to(device), vocoder=vocoder)
-    return speech.cpu()
-def speech_to_speech_translation(audio):
     translated_text = translate(audio)
-    synthesised_speech = synthesise(translated_text)
-    synthesised_speech = (synthesised_speech.numpy() * 32767).astype(np.int16)
-    return 16000, synthesised_speech
 title = "Cascaded STST"

 import gradio as gr
 import numpy as np
 import torch
+from transformers import BarkModel
+from transformers import AutoProcessor
+from transformers import pipeline
+import librosa
+processor = AutoProcessor.from_pretrained("suno/bark-small")
+model = BarkModel.from_pretrained("suno/bark-small")
 device = "cuda:0" if torch.cuda.is_available() else "cpu"
+model = model.to(device)
+# https://suno-ai.notion.site/8b8e8749ed514b0cbf3f699013548683?v=bc67cff786b04b50b3ceb756fd05f68c
+language_presets = {"es":"v2/es_speaker_",
+                    "en":"v2/en_speaker_"}
+def tts(text, language="es", style:int = 0):
+    voice_preset = language_presets[language] + str(style)
+    # prepare the inputs
+    inputs = processor(text, voice_preset = voice_preset)
+    # generate speech
+    speech_output = model.generate(**inputs.to(device))
+    sampling_rate = model.generation_config.sample_rate
+    return speech_output[0].cpu().numpy(), sampling_rate
+# load speech translation checkpoint
+asr_pipe = pipeline("automatic-speech-recognition", model="openai/whisper-base", device=device)
+def translate(audio, language:str = "es"):
+    outputs = asr_pipe(audio, max_new_tokens=256, generate_kwargs={"task": "transcribe", "language":language})
+    text = outputs["text"]
+    return text
+def synthesise(text, language="es",style=0):
+    speech, sr = tts(text, language=language, style=style)
+    target_sr = 16_000
+    speech = librosa.resample(speech, orig_sr = sr, target_sr = target_sr)
+    return speech, target_sr
+def speech_to_speech_translation(audio, debug = True):
     translated_text = translate(audio)
+    if debug:
+        print(f"{translated_text=}")
+    synthesised_speech, sampling_rate = synthesise(translated_text)
+    # tranform to int for Gradio
+    synthesised_speech = (np.array(synthesised_speech) * 32767).astype(np.int16)
+    return sampling_rate, synthesised_speech
 title = "Cascaded STST"