LG3370's picture
Update app.py
1f4859d verified
Raw History Blame
2.93 kB
import gradio as gr
import spaces
from faster_whisper import WhisperModel
# कोलाब वाला ही मॉडल आईडी
MODEL_ID = "collabora/faster-whisper-large-v2-hindi"
@spaces.GPU(duration=120)
def transcribe_hindi(audio):
if audio is None:
return "कृपया ऑडियो प्रदान करें।", None
# हगिंग फेस के नए CUDA ड्राइवर के लिए compute_type="float16" अनिवार्य है
model = WhisperModel(MODEL_ID, device="cuda", compute_type="float16")
# व्हिस्पर लार्ज मॉडल के लिए चंकिंग बैकएंड में खुद होती है
segments, info = model.transcribe(audio, language="hi", beam_size=5)
full_text = "".join([segment.text for segment in segments])
# डाउनलोड के लिए .txt फ़ाइल बनाना
file_path = "transcription.txt"
with open(file_path, "w", encoding="utf-8") as f:
f.write(full_text.strip())
return full_text.strip(), file_path
custom_css = """
footer {visibility: hidden}
.gradio-container {background-color: #fcfcfc}
#header {text-align: center; margin-bottom: 20px}
"""
with gr.Blocks(title="IndicWhisper Collabora GPU") as demo:
gr.HTML("<div id='header'><h1>🎙️ Hindi Whisper (Collabora - faster_whisper)</h1></div>")
with gr.Row():
with gr.Column():
audio_input = gr.Audio(
sources=["microphone", "upload"],
type="filepath",
label="ऑडियो रिकॉर्ड करें या अपलोड करें"
)
submit_btn = gr.Button("अनुवाद करें (Transcribe)", variant="primary")
with gr.Column():
output_text = gr.Textbox(
label="Transcription Output",
lines=10,
placeholder="आपका टेक्स्ट यहाँ दिखाई देगा..."
)
download_file = gr.File(
label="टैक्स्ट फ़ाइल डाउनलोड करें",
visible=True
)
gr.Markdown("""
---
**सुझाव:** यहाँ से प्राप्त आउटपुट को कॉपी करें और अपने **Gemini Gem** में पेस्ट करें ताकि **पञ्चमाक्षर नियमों** (ङ्, ञ्, ण्, न्, म्) के अनुसार शुद्धिकरण किया जा सके।
""")
submit_btn.click(
fn=transcribe_hindi,
inputs=audio_input,
outputs=[output_text, download_file]
)
if __name__ == "__main__":
demo.launch(css=custom_css)