File size: 4,948 Bytes
b38ca8a
12ec844
957942c
54aa443
624006c
957942c
1f4859d
54aa443
5f430e2
54aa443
 
 
 
 
 
5f430e2
54aa443
5f430e2
54aa443
5f430e2
54aa443
 
b38ca8a
 
 
 
5f430e2
54aa443
 
5f430e2
54aa443
 
 
 
 
 
 
 
 
 
 
 
5f430e2
 
 
54aa443
 
5f430e2
 
 
 
cbf57f1
5f430e2
624b74d
5f430e2
54aa443
5f430e2
54aa443
5f430e2
 
 
b38ca8a
 
 
 
 
 
12ec844
b38ca8a
957942c
b38ca8a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
54aa443
b38ca8a
 
 
 
 
 
 
12ec844
b38ca8a
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
import gradio as gr
import spaces
import torch
from transformers import AutoModelForSpeechSeq2Seq, AutoProcessor, pipeline

MODEL_ID = "collabora/whisper-large-v2-hindi"

print("काव्यात्मक सृजन के लिए मॉडल और प्रोसेसर को रैम में लोड किया जा रहा है...")

# १. मॉडल और प्रोसेसर को पहले ही CPU रैम में लोड कर लें (इससे ६.१७ GB डाउनलोड और लोड होने का समय बच जाएगा)
model = AutoModelForSpeechSeq2Seq.from_pretrained(
    MODEL_ID, 
    torch_dtype=torch.float16, 
    low_cpu_mem_usage=True, 
    use_safetensors=True
)
processor = AutoProcessor.from_pretrained(MODEL_ID)

print("मॉडल और प्रोसेसर सफलतापूर्वक रैम में आ चुके हैं!")

# २. अब केवल इस गणना वाले हिस्से को ZeroGPU सौंपेंगे
@spaces.GPU(duration=120)
def transcribe_hindi(audio):
    if audio is None:
        return "कृपया ऑडियो प्रदान करें।", None
    
    try:
        # मॉडल को तुरंत GPU पर भेजें (यह १ सेकंड से कम समय लेगा क्योंकि यह पहले से रैम में है)
        model.to("cuda")
        
        # फ़ंक्शन के अंदर ही पाइपलाइन का निर्माण (ZeroGPU के नियमों के अनुसार अनिवार्य)
        asr_pipe = pipeline(
            "automatic-speech-recognition",
            model=model,
            tokenizer=processor.tokenizer,
            feature_extractor=processor.feature_extractor,
            chunk_length_s=30,
            device="cuda",
            torch_dtype=torch.float16
        )
        
        # ट्रांसक्रिप्शन प्रक्रिया
        result = asr_pipe(audio, batch_size=8, generate_kwargs={"language": "hindi"})
        text_output = result["text"].strip()
        
        # काम पूरा होते ही मॉडल को वापस CPU पर भेजें ताकि ZeroGPU का कंटेनर मुक्त हो सके
        model.to("cpu")
        
        file_path = "transcription.txt"
        with open(file_path, "w", encoding="utf-8") as f:
            f.write(text_output)
            
        return text_output, file_path

    except Exception as e:
        # किसी भी त्रुटि की स्थिति में मॉडल को वापस CPU पर लाना सुरक्षित है
        try:
            model.to("cpu")
        except:
            pass
        return f"प्रक्रिया में कुछ व्यवधान आया: {str(e)}। कृपया पुनः प्रयास करें।", None

custom_css = """
footer {visibility: hidden}
.gradio-container {background-color: #fcfcfc}
#header {text-align: center; margin-bottom: 20px}
"""

with gr.Blocks(title="IndicWhisper Collabora GPU") as demo:
    gr.HTML("<div id='header'><h1>🎙️ Hindi Whisper (Collabora Pipeline - ZeroGPU)</h1></div>")
    
    with gr.Row():
        with gr.Column():
            audio_input = gr.Audio(
                sources=["microphone", "upload"], 
                type="filepath", 
                label="ऑडियो रिकॉर्ड करें या अपलोड करें"
            )
            submit_btn = gr.Button("अनुवाद करें (Transcribe)", variant="primary")
            
        with gr.Column():
            output_text = gr.Textbox(
                label="Transcription Output", 
                lines=10, 
                placeholder="आपका टेक्स्ट यहाँ दिखाई देगा..."
            )
            download_file = gr.File(
                label="टैक्स्ट फ़ाइल डाउनलोड करें",
                visible=True
            )
            
    gr.Markdown("""
    ---
    **सुझाव:** यहाँ से प्राप्त आउटपुट को कॉपी करें या फ़ाइल डाउनलोड करके अपने **Gemini Gem** में पेस्ट करें ताकि **पञ्चमाक्षर नियमों** (ङ्, ञ्, ण्, न्, म्) के अनुसार शुद्धिकरण किया जा सके。
    """)

    submit_btn.click(
        fn=transcribe_hindi, 
        inputs=audio_input, 
        outputs=[output_text, download_file]
    )

if __name__ == "__main__":
    demo.launch(css=custom_css)