caspr / app.py
artificialguybr's picture
Update app.py
5bdcc47
Raw History Blame
5.96 kB
import gradio as gr
from transformers import AutoModelForSeq2SeqLM, AutoTokenizer
from subprocess import run
from faster_whisper import WhisperModel
import json
import tempfile
import os
import ffmpeg
from zipfile import ZipFile
import stat
import uuid
import subprocess
import torch
import bitsandbytes
import scipy
from googletrans import Translator
import re
import subprocess
ZipFile("ffmpeg.zip").extractall()
st = os.stat('ffmpeg')
os.chmod('ffmpeg', st.st_mode | stat.S_IEXEC)
with open('google_lang_codes.json', 'r') as f:
google_lang_codes = json.load(f)
translator = Translator()
#tokenizer = AutoTokenizer.from_pretrained("facebook/nllb-200-3.3B")
#model = AutoModelForSeq2SeqLM.from_pretrained("facebook/nllb-200-3.3B")
whisper_model = WhisperModel("large-v2", device="cuda", compute_type="float16")
print("cwd", os.getcwd())
print(os.listdir())
def process_video(Video, target_language):
current_path = os.getcwd()
print("Iniciando process_video")
common_uuid = uuid.uuid4()
print("Checking FFmpeg availability...")
run(["ffmpeg", "-version"])
audio_file = f"{common_uuid}.wav"
run(["ffmpeg", "-i", Video, audio_file])
# Transcription with Whisper.
print("Iniciando transcrição com Whisper")
segments, _ = whisper_model.transcribe(audio_file, beam_size=5)
segments = list(segments)
transcript_file = f"{current_path}/{common_uuid}.srt"
# Create a list to hold the translated lines.
translated_lines = []
with open(transcript_file, "w+", encoding="utf-8") as f:
counter = 1
for segment in segments:
start_hours = int(segment.start // 3600)
start_minutes = int((segment.start % 3600) // 60)
start_seconds = int(segment.start % 60)
start_milliseconds = int((segment.start - int(segment.start)) * 1000)
end_hours = int(segment.end // 3600)
end_minutes = int((segment.end % 3600) // 60)
end_seconds = int(segment.end % 60)
end_milliseconds = int((segment.end - int(segment.end)) * 1000)
formatted_start = f"{start_hours:02d}:{start_minutes:02d}:{start_seconds:02d},{start_milliseconds:03d}"
formatted_end = f"{end_hours:02d}:{end_minutes:02d}:{end_seconds:02d},{end_milliseconds:03d}"
f.write(f"{counter}\n")
f.write(f"{formatted_start} --> {formatted_end}\n")
f.write(f"{segment.text}\n\n")
counter += 1
# Move the file pointer to the beginning of the file.
f.seek(0)
# Translating the SRT from Whisper with NLLB.
target_language_code = google_lang_codes.get(target_language, "en")
paragraph = ""
for line in f:
if line.strip().isnumeric() or "-->" in line:
translated_lines.append(line)
elif line.strip() != "":
translated_text = translator.translate(line.strip(), dest=target_language_code).text
translated_lines.append(translated_text + "\n")
else:
translated_lines.append("\n")
# Move the file pointer to the beginning of the file and truncate it.
f.seek(0)
f.truncate()
# Write the translated lines back into the original file.
f.writelines(translated_lines)
#return None, None
output_video = f"{common_uuid}_output_video.mp4"
# Debugging: Validate FFmpeg command for subtitle embedding
print("Validating FFmpeg command for subtitle embedding...")
print(f"Translated SRT file: {transcript_file}")
with open(transcript_file, 'r', encoding='utf-8') as f:
print(f"First few lines of translated SRT: {f.readlines()[:10]}")
if os.path.exists(transcript_file):
print(f"{transcript_file} exists.")
else:
print(f"{transcript_file} does not exist.")
#transcript_file_abs_path = os.path.abspath(transcript_file)
try:
if target_language_code == 'ja': # 'ja' é o código de idioma para o japonês
result = subprocess.run(["ffmpeg", "-i", Video, "-vf", f"subtitles={transcript_file}:force_style='FontName=Noto Sans CJK JP',charenc=UTF-8", "-scodec", "mov_text", "-metadata:s:s:0", "language=jpn", output_video], capture_output=True, text=True)
else:
result = subprocess.run(["ffmpeg", "-i", Video, "-vf", f"subtitles={transcript_file}:force_style='FontName=Arial Unicode MS'", output_video], capture_output=True, text=True)
if result.returncode == 0:
print("FFmpeg executado com sucesso.")
else:
print(f"FFmpeg falhou com o código de retorno {result.returncode}.")
print("Stdout:", result.stdout)
print("Stderr:", result.stderr)
except Exception as e:
print(f"Ocorreu uma exceção: {e}")
print("process_video concluído com sucesso")
os.unlink(audio_file)
os.unlink(transcript_file)
print(f"Returning output video path: {output_video}")
return output_video
iface = gr.Interface(
fn=process_video,
inputs=[
gr.Video(),
gr.Dropdown(choices=list(google_lang_codes.keys()), label="Target Language for Dubbing", value="English"),
],
outputs=[
gr.Video(),
#gr.FileExplorer()
],
live=False,
title="VIDEO TRANSCRIPTION AND TRANSLATION",
description="""This tool was developed by [@artificialguybr](https://twitter.com/artificialguybr) using entirely open-source tools. Special thanks to Hugging Face for the GPU support.""",
allow_flagging=False
)
with gr.Blocks() as demo:
iface.render()
gr.Markdown("""
**Note:**
- Video limit is 15 minute. It will do the transcription and translate of subtitles.
- The tool uses open-source models for all models. It's a alpha version.
""")
demo.queue(concurrency_count=1, max_size=15)
demo.launch()