speech-to-speech-translation-ca

Runtime error

JanLilan commited on Dec 27, 2023

Commit

e5d5a17

1 Parent(s): dbfdf1a

update app.py adding whisper to catala, also the model text-to-speech

Files changed (1) hide show

app.py CHANGED Viewed

@@ -2,7 +2,6 @@ import gradio as gr
 import numpy as np
 import torch
 from datasets import load_dataset
 from transformers import SpeechT5ForTextToSpeech, SpeechT5HifiGan, SpeechT5Processor, pipeline
@@ -14,15 +13,21 @@ asr_pipe = pipeline("automatic-speech-recognition", model="openai/whisper-base",
 # load text-to-speech checkpoint and speaker embeddings
 processor = SpeechT5Processor.from_pretrained("microsoft/speecht5_tts")
-model = SpeechT5ForTextToSpeech.from_pretrained("microsoft/speecht5_tts").to(device)
 vocoder = SpeechT5HifiGan.from_pretrained("microsoft/speecht5_hifigan").to(device)
 embeddings_dataset = load_dataset("Matthijs/cmu-arctic-xvectors", split="validation")
 speaker_embeddings = torch.tensor(embeddings_dataset[7306]["xvector"]).unsqueeze(0)
 def translate(audio):
-    outputs = asr_pipe(audio, max_new_tokens=256, generate_kwargs={"task": "translate"})
     return outputs["text"]

 import numpy as np
 import torch
 from datasets import load_dataset
 from transformers import SpeechT5ForTextToSpeech, SpeechT5HifiGan, SpeechT5Processor, pipeline
 # load text-to-speech checkpoint and speaker embeddings
 processor = SpeechT5Processor.from_pretrained("microsoft/speecht5_tts")
+# model = SpeechT5ForTextToSpeech.from_pretrained("microsoft/speecht5_tts").to(device)
+model = SpeechT5ForTextToSpeech.from_pretrained(
+    "JanLilan/speecht5_finetuned_openslr-slr69-cat"
+).to(device)
 vocoder = SpeechT5HifiGan.from_pretrained("microsoft/speecht5_hifigan").to(device)
+# we will try to translate with this voice embedding... Let's see what happen. else:
+# dataset = load_dataset("projecte-aina/openslr-slr69-ca-trimmed-denoised", split="train")
+# dataset = dataset.cast_column("audio", Audio(sampling_rate=16000))
+# etc.
 embeddings_dataset = load_dataset("Matthijs/cmu-arctic-xvectors", split="validation")
 speaker_embeddings = torch.tensor(embeddings_dataset[7306]["xvector"]).unsqueeze(0)
 def translate(audio):
+    outputs = asr_pipe(audio, max_new_tokens=256, generate_kwargs={"task": "translate", "language": "cat"})
     return outputs["text"]