import gradio as gr import spaces from faster_whisper import WhisperModel # कोलाब वाला ही मॉडल आईडी MODEL_ID = "collabora/faster-whisper-large-v2-hindi" @spaces.GPU(duration=120) def transcribe_hindi(audio): if audio is None: return "कृपया ऑडियो प्रदान करें।", None # हगिंग फेस के नए CUDA ड्राइवर के लिए compute_type="float16" अनिवार्य है model = WhisperModel(MODEL_ID, device="cuda", compute_type="float16") # व्हिस्पर लार्ज मॉडल के लिए चंकिंग बैकएंड में खुद होती है segments, info = model.transcribe(audio, language="hi", beam_size=5) full_text = "".join([segment.text for segment in segments]) # डाउनलोड के लिए .txt फ़ाइल बनाना file_path = "transcription.txt" with open(file_path, "w", encoding="utf-8") as f: f.write(full_text.strip()) return full_text.strip(), file_path custom_css = """ footer {visibility: hidden} .gradio-container {background-color: #fcfcfc} #header {text-align: center; margin-bottom: 20px} """ with gr.Blocks(title="IndicWhisper Collabora GPU") as demo: gr.HTML("") with gr.Row(): with gr.Column(): audio_input = gr.Audio( sources=["microphone", "upload"], type="filepath", label="ऑडियो रिकॉर्ड करें या अपलोड करें" ) submit_btn = gr.Button("अनुवाद करें (Transcribe)", variant="primary") with gr.Column(): output_text = gr.Textbox( label="Transcription Output", lines=10, placeholder="आपका टेक्स्ट यहाँ दिखाई देगा..." ) download_file = gr.File( label="टैक्स्ट फ़ाइल डाउनलोड करें", visible=True ) gr.Markdown(""" --- **सुझाव:** यहाँ से प्राप्त आउटपुट को कॉपी करें और अपने **Gemini Gem** में पेस्ट करें ताकि **पञ्चमाक्षर नियमों** (ङ्, ञ्, ण्, न्, म्) के अनुसार शुद्धिकरण किया जा सके। """) submit_btn.click( fn=transcribe_hindi, inputs=audio_input, outputs=[output_text, download_file] ) if __name__ == "__main__": demo.launch(css=custom_css)