LG3370 commited on
Commit
b38ca8a
·
verified ·
1 Parent(s): 176e692

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +67 -26
app.py CHANGED
@@ -1,35 +1,76 @@
 
1
  import spaces
2
- from faster_whisper import WhisperModel
 
 
3
 
4
- # 1. Initialize the model on CPU globally.
5
- # ZeroGPU boots your script on a CPU-only node first.
6
 
7
-
8
-
9
- # 2. Wrap the inference logic with the ZeroGPU decorator.
10
- # You can increase the duration (in seconds) if you process long audio files.
11
- @spaces.GPU(duration=90)
12
- def transcribe_audio(audio_path):
13
- # Dynamically change the device to CUDA during the ZeroGPU allocation window
14
- model = WhisperModel(
15
- "collabora/faster-whisper-large-v2-hindi",
16
- device="cuda",
17
- compute_type="float32"
18
  )
19
 
20
- segments, info = model.transcribe(audio_path)
 
 
21
 
22
- # We consume the generator inside the GPU function to prevent losing the GPU context
23
- results = []
24
- for segment in segments:
25
- results.append("[%.2fs -> %.2fs] %s" % (segment.start, segment.end, segment.text))
26
 
27
- # Optional: Clean up and move model back to CPU when done
28
- model.model.to("cpu")
29
- return results
 
 
 
 
30
 
31
- # 3. Trigger your transcription
32
- output_segments = transcribe_audio("Mytest000.m4a")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
33
 
34
- for line in output_segments:
35
- print(line)
 
1
+ import gradio as gr
2
  import spaces
3
+ import torch
4
+ from transformers import pipeline
5
+ import os
6
 
7
+ # आपका पसंदीदा और सटीक हिंदी मॉडल
8
+ MODEL_ID = "collabora/faster-whisper-large-v2-hindi"
9
 
10
+ @spaces.GPU(duration=120)
11
+ def transcribe_hindi(audio):
12
+ if audio is None:
13
+ return "कृपया ऑडियो प्रदान करें।", None
14
+
15
+ # Standard Transformers Pipeline - ZeroGPU के CUDA ड्राइवर के साथ 100% अनुकूल
16
+ pipe = pipeline(
17
+ "automatic-speech-recognition",
18
+ model=MODEL_ID,
19
+ torch_dtype=torch.float16,
20
+ device="cuda"
21
  )
22
 
23
+ # ट्रांसक्रिप्शन जनरेट करें
24
+ result = pipe(audio, generate_kwargs={"language": "hindi"})
25
+ text_output = result["text"].strip()
26
 
27
+ # डाउनलोड के लिए .txt फ़ाइल बनाना
28
+ file_path = "transcription.txt"
29
+ with open(file_path, "w", encoding="utf-8") as f:
30
+ f.write(text_output)
31
 
32
+ return text_output, file_path
33
+
34
+ custom_css = """
35
+ footer {visibility: hidden}
36
+ .gradio-container {background-color: #fcfcfc}
37
+ #header {text-align: center; margin-bottom: 20px}
38
+ """
39
 
40
+ with gr.Blocks(title="IndicWhisper Collabora GPU") as demo:
41
+ gr.HTML("<div id='header'><h1>🎙️ Hindi Whisper (Collabora v2 - ZeroGPU)</h1></div>")
42
+
43
+ with gr.Row():
44
+ with gr.Column():
45
+ audio_input = gr.Audio(
46
+ sources=["microphone", "upload"],
47
+ type="filepath",
48
+ label="ऑडियो रिकॉर्ड करें या अपलोड करें"
49
+ )
50
+ submit_btn = gr.Button("अनुवाद करें (Transcribe)", variant="primary")
51
+
52
+ with gr.Column():
53
+ output_text = gr.Textbox(
54
+ label="Transcription Output",
55
+ lines=10,
56
+ placeholder="आपका टेक्स्ट यहाँ दिखाई देगा..."
57
+ )
58
+ # डाउनलोड लिंक के लिए Gradio का File कंपोनेंट
59
+ download_file = gr.File(
60
+ label="टैक्स्ट फ़ाइल डाउनलोड करें",
61
+ visible=True
62
+ )
63
+
64
+ gr.Markdown("""
65
+ ---
66
+ **सुझाव:** यहाँ से प्राप्त आउटपुट को कॉपी करें या फ़ाइल डाउनलोड करके अपने **Gemini Gem** में पेस्ट करें ताकि **पञ्चमाक्षर नियमों** (ङ्, ञ्, ण्, न्, म्) के अनुसार शुद्धिकरण किया जा सके।
67
+ """)
68
+
69
+ submit_btn.click(
70
+ fn=transcribe_hindi,
71
+ inputs=audio_input,
72
+ outputs=[output_text, download_file]
73
+ )
74
 
75
+ if __name__ == "__main__":
76
+ demo.launch(css=custom_css)