fahadqazi commited on
Commit
bfe6ba9
·
verified ·
1 Parent(s): 8ca4f18

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +64 -54
app.py CHANGED
@@ -3,96 +3,106 @@ import gradio as gr
3
  import torch
4
  import os
5
  import importlib.util
6
-
7
  from huggingface_hub import login, hf_hub_download
8
 
9
  RUN_DEMO = True
10
 
11
  if RUN_DEMO:
12
-
13
  from transformers import AutoModelForSpeechSeq2Seq, AutoProcessor
14
 
15
  device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
16
-
17
  print(f"Is CUDA available: {torch.cuda.is_available()}")
18
- # True
19
- #print(f"CUDA device: {torch.cuda.get_device_name(torch.cuda.current_device())}")
20
- # Tesla T4
21
 
 
22
  auth_token = os.environ.get("hf_token")
23
- print(auth_token)
24
  if not auth_token:
25
- raise ValueError("Hugging Face token is missing! Add it as a secret.")
26
-
27
- login(token=auth_token)
28
-
29
- stt_module_path = hf_hub_download(repo_id="fahadqazi/private-code", filename="sindhi_stt_module.py", token=auth_token)
30
-
31
- # Dynamically import the module
32
- spec = importlib.util.spec_from_file_location("sindhi_stt_module", stt_module_path)
33
- stt_module = importlib.util.module_from_spec(spec)
34
- spec.loader.exec_module(stt_module)
35
-
36
- TranscriberClass = stt_module.Transcriber
37
-
38
- Transcriber = TranscriberClass(
39
- auth_token=auth_token
40
- )
 
 
 
 
 
 
 
 
 
 
 
41
 
 
 
 
 
 
 
 
 
 
 
 
42
 
43
  if torch.cuda.is_available():
44
- # Function to transcribe using the selected model
45
  @spaces.GPU(duration=60)
46
  def decorated_transcribe(audio_arrays):
47
- return Transcriber.transcribe(audio_arrays)
48
  else:
49
  def decorated_transcribe(audio_arrays):
50
- return Transcriber.transcribe(audio_arrays)
51
 
52
- # def transcribe_wrapper(uploaded_file, youtube_link, remove_music):
53
  def transcribe_wrapper(uploaded_file, microphone, remove_music):
54
- youtube_link = None
55
- if youtube_link and youtube_link.strip() != "":
56
- try:
57
- clean_url = Transcriber.clean_youtube_url(youtube_link.strip())
58
- audio_path = Transcriber.download_youtube_audio(clean_url)
59
- is_youtube = True
60
- except Exception as e:
61
- return f"(YouTube download failed: {str(e)})"
62
- elif microphone:
63
- audio_path = microphone
64
- is_youtube = False
65
- elif uploaded_file:
66
- audio_path = uploaded_file.name if hasattr(uploaded_file, "name") else uploaded_file
67
- is_youtube = False
68
- else:
69
- return "Please upload a file, record from microphone, or enter a YouTube URL."
70
-
71
- audio_arrays = Transcriber.prepare_inputs(audio_path, is_youtube=is_youtube, use_demucs=remove_music)
72
- print(audio_arrays)
73
- return decorated_transcribe(audio_arrays)
 
74
 
75
  # Gradio Interface
76
  iface = gr.Interface(
77
  fn=transcribe_wrapper,
78
  inputs=[
79
  gr.File(file_types=[".wav", ".mp3", ".ogg", ".flac", ".mp4", ".mov", ".mkv"], label="Upload Audio or Video"),
80
- # gr.Textbox(label="YouTube URL (optional)", placeholder="https://www.youtube.com/watch?v=..."),
81
  gr.Audio(sources="microphone", type="filepath", label="Or Record from Microphone"),
82
  gr.Checkbox(label="Remove background music / noise (slower, more accurate)", value=False)
83
  ],
84
- # examples=[
85
- # [None, "https://www.youtube.com/watch?v=Dg2-4UX9NZU", False],
86
- # [None, "https://www.youtube.com/watch?v=MaE2zH9ZhQk", False]
87
- # ],
88
  outputs="text",
89
  title="Sindhi Speech to Text",
90
- # description="Upload an audio or video file (or paste a YouTube link) up to a few minutes long. Only one input is needed.",
91
  description="Upload an audio or video file up to a few minutes long. Only one input is needed.",
92
  cache_examples=False
93
  )
94
 
95
- iface.launch(ssr=False)
 
96
 
97
  else:
98
 
 
3
  import torch
4
  import os
5
  import importlib.util
 
6
  from huggingface_hub import login, hf_hub_download
7
 
8
  RUN_DEMO = True
9
 
10
  if RUN_DEMO:
 
11
  from transformers import AutoModelForSpeechSeq2Seq, AutoProcessor
12
 
13
  device = torch.device("cuda" if torch.cuda.is_available() else "cpu")
 
14
  print(f"Is CUDA available: {torch.cuda.is_available()}")
 
 
 
15
 
16
+ # Check token early
17
  auth_token = os.environ.get("hf_token")
 
18
  if not auth_token:
19
+ print("ERROR: Hugging Face token is missing!")
20
+ RUN_DEMO = False
21
+ else:
22
+ try:
23
+ login(token=auth_token)
24
+
25
+ # Download module
26
+ stt_module_path = hf_hub_download(
27
+ repo_id="fahadqazi/private-code",
28
+ filename="sindhi_stt_module.py",
29
+ token=auth_token
30
+ )
31
+
32
+ # Import module
33
+ spec = importlib.util.spec_from_file_location("sindhi_stt_module", stt_module_path)
34
+ stt_module = importlib.util.module_from_spec(spec)
35
+ spec.loader.exec_module(stt_module)
36
+ TranscriberClass = stt_module.Transcriber
37
+
38
+ # LAZY LOADING: Don't initialize here, do it in function
39
+ print("Module loaded successfully, initializing on first use...")
40
+
41
+ except Exception as e:
42
+ print(f"ERROR during initialization: {str(e)}")
43
+ import traceback
44
+ traceback.print_exc()
45
+ RUN_DEMO = False
46
 
47
+ if RUN_DEMO:
48
+ # Initialize transcriber lazily
49
+ transcriber_instance = None
50
+
51
+ def get_transcriber():
52
+ global transcriber_instance
53
+ if transcriber_instance is None:
54
+ print("Initializing Transcriber (first use)...")
55
+ transcriber_instance = TranscriberClass(auth_token=auth_token)
56
+ print("Transcriber initialized!")
57
+ return transcriber_instance
58
 
59
  if torch.cuda.is_available():
 
60
  @spaces.GPU(duration=60)
61
  def decorated_transcribe(audio_arrays):
62
+ return get_transcriber().transcribe(audio_arrays)
63
  else:
64
  def decorated_transcribe(audio_arrays):
65
+ return get_transcriber().transcribe(audio_arrays)
66
 
 
67
  def transcribe_wrapper(uploaded_file, microphone, remove_music):
68
+ try:
69
+ transcriber = get_transcriber()
70
+
71
+ # ... rest of your code ...
72
+ youtube_link = None
73
+ if youtube_link and youtube_link.strip() != "":
74
+ # ... YouTube handling ...
75
+ pass
76
+ elif microphone:
77
+ audio_path = microphone
78
+ is_youtube = False
79
+ elif uploaded_file:
80
+ audio_path = uploaded_file.name if hasattr(uploaded_file, "name") else uploaded_file
81
+ is_youtube = False
82
+ else:
83
+ return "Please upload a file, record from microphone, or enter a YouTube URL."
84
+
85
+ audio_arrays = transcriber.prepare_inputs(audio_path, is_youtube=is_youtube, use_demucs=remove_music)
86
+ return decorated_transcribe(audio_arrays)
87
+ except Exception as e:
88
+ return f"Error: {str(e)}"
89
 
90
  # Gradio Interface
91
  iface = gr.Interface(
92
  fn=transcribe_wrapper,
93
  inputs=[
94
  gr.File(file_types=[".wav", ".mp3", ".ogg", ".flac", ".mp4", ".mov", ".mkv"], label="Upload Audio or Video"),
 
95
  gr.Audio(sources="microphone", type="filepath", label="Or Record from Microphone"),
96
  gr.Checkbox(label="Remove background music / noise (slower, more accurate)", value=False)
97
  ],
 
 
 
 
98
  outputs="text",
99
  title="Sindhi Speech to Text",
 
100
  description="Upload an audio or video file up to a few minutes long. Only one input is needed.",
101
  cache_examples=False
102
  )
103
 
104
+ print("Launching Gradio interface...")
105
+ iface.launch()
106
 
107
  else:
108