Komal133 commited on
Commit
8abb11f
Β·
verified Β·
1 Parent(s): 0cbb12f

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +215 -61
app.py CHANGED
@@ -5,97 +5,178 @@ import threading
5
  import time
6
  import librosa
7
  import requests
 
8
  from datetime import datetime
9
  from transformers import pipeline
10
 
11
  # πŸŽ™οΈ Load detection model
12
  try:
13
  print("[INFO] Loading Hugging Face model...")
 
 
 
 
14
  classifier = pipeline(
15
  "audio-classification",
16
  model="padmalcom/wav2vec2-large-nonverbalvocalization-classification"
17
  )
 
18
  except Exception as e:
19
  print(f"[ERROR] Failed to load model: {e}")
20
  classifier = None
21
 
22
  # === Audio Conversion ===
23
  def convert_audio(input_path, output_path="input.wav"):
 
 
 
 
24
  try:
25
  cmd = [
26
  "ffmpeg", "-i", input_path,
27
  "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1",
28
  output_path, "-y"
29
  ]
30
- subprocess.run(cmd, check=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE)
 
31
  print(f"[DEBUG] Audio converted to WAV: {output_path}")
 
 
 
 
32
  return output_path
33
  except subprocess.CalledProcessError as e:
34
- print(f"[ERROR] ffmpeg conversion failed: {e.stderr.decode()}")
35
- raise RuntimeError("Audio conversion failed.")
 
 
 
 
 
 
36
 
37
  # === Scream Detection ===
38
  def detect_scream(audio_path):
 
 
 
 
 
 
 
39
  try:
 
40
  audio, sr = librosa.load(audio_path, sr=16000)
41
  print(f"[DEBUG] Loaded audio: {len(audio)} samples at {sr} Hz")
 
42
  if len(audio) == 0:
43
- return {"label": "none", "score": 0.0}
 
 
 
44
  results = classifier(audio)
45
  print(f"[DEBUG] Model output: {results}")
 
46
  if not results:
47
- return {"label": "none", "score": 0.0}
48
- top = results[0]
49
- return {"label": top["label"].lower(), "score": float(top["score"]) * 100}
 
 
 
 
 
50
  except Exception as e:
51
- print(f"[ERROR] Detection failed: {e}")
52
- return {"label": "error", "score": 0.0}
53
 
54
  # === Send Alert to Salesforce ===
55
  def send_salesforce_alert(audio_meta, detection):
 
 
 
 
56
  SF_URL = os.getenv("SF_ALERT_URL")
57
  SF_TOKEN = os.getenv("SF_API_TOKEN")
 
58
  if not SF_URL or not SF_TOKEN:
59
- raise RuntimeError("Salesforce config missing.")
 
60
 
61
  headers = {
62
  "Authorization": f"Bearer {SF_TOKEN}",
63
  "Content-Type": "application/json"
64
  }
65
  payload = {
66
- "AudioName": audio_meta["filename"],
67
  "DetectedLabel": detection["label"],
68
- "Score": detection["score"],
69
  "AlertLevel": audio_meta["alert_level"],
70
  "Timestamp": audio_meta["timestamp"],
71
  }
72
 
73
  print(f"[DEBUG] Sending payload to Salesforce: {payload}")
74
- resp = requests.post(SF_URL, json=payload, headers=headers, timeout=5)
75
- resp.raise_for_status()
76
- return resp.json()
 
 
 
 
 
 
 
 
 
 
 
77
 
78
  # === Main Gradio Function ===
79
- def process_uploaded(audio_file, start_stop, high_thresh, med_thresh):
80
- if start_stop != "Start":
81
- return "πŸ›‘ System is stopped."
 
 
 
 
 
 
 
 
 
82
 
83
  try:
 
84
  wav_path = convert_audio(audio_file)
85
- except Exception as e:
86
  return f"❌ Audio conversion error: {e}"
 
 
87
 
 
88
  detection = detect_scream(wav_path)
89
  label = detection["label"]
90
  score = detection["score"]
91
 
92
- # Determine risk level
93
- if label and "scream" in label and score >= high_thresh:
94
- level = "High-Risk"
95
- elif label and "scream" in label and score >= med_thresh:
96
- level = "Medium-Risk"
97
- else:
98
- level = "None"
 
 
 
 
 
 
 
 
 
 
 
 
99
 
100
  audio_meta = {
101
  "filename": os.path.basename(audio_file),
@@ -103,64 +184,137 @@ def process_uploaded(audio_file, start_stop, high_thresh, med_thresh):
103
  "alert_level": level
104
  }
105
 
106
- # Send to Salesforce if needed
107
  if level in ("High-Risk", "Medium-Risk"):
108
  try:
109
  sf_resp = send_salesforce_alert(audio_meta, detection)
110
- return f"βœ… Detection: {label} ({score:.1f}%) β€” {level} β€” Alert sent (ID: {sf_resp.get('id', 'N/A')})"
 
 
111
  except Exception as e:
112
- return f"⚠️ Detection: {label} ({score:.1f}%) β€” {level} β€” ERROR: {e}"
 
 
 
 
 
113
 
114
- return f"🟒 Detection: {label} ({score:.1f}%) β€” Alert Level: {level}"
115
 
116
  # === Gradio UI ===
 
117
  iface = gr.Interface(
118
  fn=process_uploaded,
119
  inputs=[
120
- gr.Audio(type="filepath", label="Upload Audio"),
121
- gr.Radio(["Start", "Stop"], label="System State", value="Start"),
122
- gr.Slider(0, 100, value=80, step=1, label="High-Risk Threshold (%)"),
123
- gr.Slider(0, 100, value=50, step=1, label="Medium-Risk Threshold (%)")
 
 
 
124
  ],
125
  outputs="text",
126
- title="πŸ“’ Scream Detection & Salesforce Alerts",
127
  description="""
128
- 🎧 Upload or record audio. System classifies screams and triggers alerts.
129
- ⚠️ Alerts are sent to Salesforce for High/Medium-Risk detections.
 
130
  """,
131
- allow_flagging="never"
132
  )
133
 
134
- # === Optional Real-Time Listener ===
 
 
 
 
135
  def pi_listener(high_thresh=80, med_thresh=50, interval=1.0):
136
- import sounddevice as sd
137
- import numpy as np
 
 
 
 
 
 
 
 
 
 
 
 
 
138
 
139
  def callback(indata, frames, time_info, status):
 
 
 
 
 
140
  wav = indata.squeeze()
141
- detection = classifier(wav.astype(np.float32))
142
- lbl, sc = (detection[0]["label"].lower(), detection[0]["score"] * 100)
143
- level = "None"
144
- if "scream" in lbl and sc >= high_thresh:
145
- level = "High-Risk"
146
- elif "scream" in lbl and sc >= med_thresh:
147
- level = "Medium-Risk"
148
- if level != "None":
149
- timestamp = datetime.utcnow().isoformat() + "Z"
150
- send_salesforce_alert(
151
- {"filename": "live-stream", "timestamp": timestamp, "alert_level": level},
152
- {"label": lbl, "score": sc}
153
- )
154
- print(f"[{timestamp}] {level} scream detected ({sc:.1f}%) – alert sent.")
155
-
156
- with sd.InputStream(channels=1, samplerate=16000, callback=callback):
157
- print("πŸ”Š Real-time detection started...")
158
- while True:
159
- time.sleep(interval)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
160
 
161
  # === App Entry ===
162
  if __name__ == "__main__":
163
- # Optional: enable real-time listener
 
 
 
164
  # pi_thread = threading.Thread(target=pi_listener, daemon=True)
165
  # pi_thread.start()
166
 
 
5
  import time
6
  import librosa
7
  import requests
8
+ import numpy as np # Added for sounddevice callback
9
  from datetime import datetime
10
  from transformers import pipeline
11
 
12
  # πŸŽ™οΈ Load detection model
13
  try:
14
  print("[INFO] Loading Hugging Face model...")
15
+ # The current model includes 'screaming' as a label, but might misclassify
16
+ # high-pitched screams as 'crying' due to its general non-verbal vocalization training.
17
+ # For higher accuracy in distinguishing screams from crying, fine-tuning on a specific
18
+ # dataset or exploring other specialized models would be recommended.
19
  classifier = pipeline(
20
  "audio-classification",
21
  model="padmalcom/wav2vec2-large-nonverbalvocalization-classification"
22
  )
23
+ print(f"[INFO] Model labels: {classifier.model.config.id2label.values()}")
24
  except Exception as e:
25
  print(f"[ERROR] Failed to load model: {e}")
26
  classifier = None
27
 
28
  # === Audio Conversion ===
29
  def convert_audio(input_path, output_path="input.wav"):
30
+ """
31
+ Converts audio files to a standard WAV format (16kHz, mono, 16-bit PCM).
32
+ This ensures compatibility with the Hugging Face model.
33
+ """
34
  try:
35
  cmd = [
36
  "ffmpeg", "-i", input_path,
37
  "-acodec", "pcm_s16le", "-ar", "16000", "-ac", "1",
38
  output_path, "-y"
39
  ]
40
+ # Use subprocess.run with capture_output=True for better error handling
41
+ result = subprocess.run(cmd, check=True, capture_output=True, text=True)
42
  print(f"[DEBUG] Audio converted to WAV: {output_path}")
43
+ if result.stdout:
44
+ print(f"[DEBUG] ffmpeg stdout: {result.stdout.strip()}")
45
+ if result.stderr:
46
+ print(f"[DEBUG] ffmpeg stderr: {result.stderr.strip()}")
47
  return output_path
48
  except subprocess.CalledProcessError as e:
49
+ print(f"[ERROR] ffmpeg conversion failed: {e.stderr.strip()}")
50
+ raise RuntimeError(f"Audio conversion failed: {e.stderr.strip()}")
51
+ except FileNotFoundError:
52
+ print("[ERROR] ffmpeg command not found. Please ensure ffmpeg is installed and in your PATH.")
53
+ raise RuntimeError("ffmpeg not found. Please install it.")
54
+ except Exception as e:
55
+ print(f"[ERROR] Unexpected error during audio conversion: {e}")
56
+ raise RuntimeError(f"Unexpected audio conversion error: {e}")
57
 
58
  # === Scream Detection ===
59
  def detect_scream(audio_path):
60
+ """
61
+ Detects screams in an audio file using the loaded Hugging Face model.
62
+ Returns the top detected label and its confidence score.
63
+ """
64
+ if classifier is None:
65
+ return {"label": "model_not_loaded", "score": 0.0}
66
+
67
  try:
68
+ # Librosa loads audio, automatically resamples if needed
69
  audio, sr = librosa.load(audio_path, sr=16000)
70
  print(f"[DEBUG] Loaded audio: {len(audio)} samples at {sr} Hz")
71
+
72
  if len(audio) == 0:
73
+ print("[WARNING] Empty audio file provided for detection.")
74
+ return {"label": "no_audio_data", "score": 0.0}
75
+
76
+ # The pipeline expects raw audio data (numpy array)
77
  results = classifier(audio)
78
  print(f"[DEBUG] Model output: {results}")
79
+
80
  if not results:
81
+ print("[WARNING] Model returned no detection results.")
82
+ return {"label": "no_detection", "score": 0.0}
83
+
84
+ # Sort results by score in descending order to get the top prediction
85
+ top_prediction = sorted(results, key=lambda x: x['score'], reverse=True)[0]
86
+
87
+ # Ensure label is lowercase for consistent comparison
88
+ return {"label": top_prediction["label"].lower(), "score": float(top_prediction["score"]) * 100}
89
  except Exception as e:
90
+ print(f"[ERROR] Detection failed for {audio_path}: {e}")
91
+ return {"label": "detection_error", "score": 0.0}
92
 
93
  # === Send Alert to Salesforce ===
94
  def send_salesforce_alert(audio_meta, detection):
95
+ """
96
+ Sends an alert payload to a configured Salesforce endpoint.
97
+ Retrieves Salesforce URL and token from environment variables.
98
+ """
99
  SF_URL = os.getenv("SF_ALERT_URL")
100
  SF_TOKEN = os.getenv("SF_API_TOKEN")
101
+
102
  if not SF_URL or not SF_TOKEN:
103
+ print("[ERROR] Salesforce configuration (SF_ALERT_URL or SF_API_TOKEN) missing.")
104
+ raise RuntimeError("Salesforce configuration missing. Cannot send alert.")
105
 
106
  headers = {
107
  "Authorization": f"Bearer {SF_TOKEN}",
108
  "Content-Type": "application/json"
109
  }
110
  payload = {
111
+ "AudioName": audio_meta.get("filename", "unknown_audio"),
112
  "DetectedLabel": detection["label"],
113
+ "Score": round(detection["score"], 2), # Round score for cleaner data
114
  "AlertLevel": audio_meta["alert_level"],
115
  "Timestamp": audio_meta["timestamp"],
116
  }
117
 
118
  print(f"[DEBUG] Sending payload to Salesforce: {payload}")
119
+ try:
120
+ resp = requests.post(SF_URL, json=payload, headers=headers, timeout=10) # Increased timeout
121
+ resp.raise_for_status() # Raises HTTPError for bad responses (4xx or 5xx)
122
+ print(f"[INFO] Salesforce alert sent successfully. Response: {resp.json()}")
123
+ return resp.json()
124
+ except requests.exceptions.Timeout:
125
+ print("[ERROR] Salesforce alert request timed out.")
126
+ raise RuntimeError("Salesforce alert timed out.")
127
+ except requests.exceptions.RequestException as e:
128
+ print(f"[ERROR] Error sending Salesforce alert: {e}")
129
+ # Attempt to print response content if available for more details
130
+ if hasattr(e, 'response') and e.response is not None:
131
+ print(f"[ERROR] Salesforce response content: {e.response.text}")
132
+ raise RuntimeError(f"Failed to send Salesforce alert: {e}")
133
 
134
  # === Main Gradio Function ===
135
+ def process_uploaded(audio_file, system_state, high_thresh, med_thresh):
136
+ """
137
+ Main function for Gradio interface. Processes uploaded audio,
138
+ performs scream detection, and sends alerts to Salesforce based on thresholds.
139
+ """
140
+ if system_state != "Start":
141
+ return "πŸ›‘ System is stopped. Change 'System State' to 'Start' to enable processing."
142
+
143
+ if audio_file is None:
144
+ return "Please upload an audio file or record one."
145
+
146
+ print(f"[INFO] Processing uploaded audio: {audio_file}")
147
 
148
  try:
149
+ # Convert audio to the required WAV format
150
  wav_path = convert_audio(audio_file)
151
+ except RuntimeError as e:
152
  return f"❌ Audio conversion error: {e}"
153
+ except Exception as e:
154
+ return f"❌ An unexpected error occurred during audio conversion: {e}"
155
 
156
+ # Perform scream detection
157
  detection = detect_scream(wav_path)
158
  label = detection["label"]
159
  score = detection["score"]
160
 
161
+ # Determine risk level based on detected label and score
162
+ alert_message = f"🟒 Detection: {label} ({score:.1f}%) β€” "
163
+ level = "None"
164
+
165
+ # Check for 'scream' or related labels. The model might output 'screaming' or similar.
166
+ # It's important to check if 'scream' is *in* the label, as some models might output
167
+ # variations or combine labels.
168
+ if "scream" in label:
169
+ if score >= high_thresh:
170
+ level = "High-Risk"
171
+ elif score >= med_thresh:
172
+ level = "Medium-Risk"
173
+ # Add explicit check for 'crying' if it's a known misclassification target
174
+ elif "crying" in label and score >= med_thresh: # Consider if crying should also trigger an alert
175
+ # This part can be adjusted based on whether 'crying' is also an alertable event.
176
+ # For now, we'll treat it as 'None' unless explicitly defined as a risk.
177
+ level = "None" # Or set to "Low-Risk" if crying is also a concern.
178
+
179
+ alert_message += f"Alert Level: {level}"
180
 
181
  audio_meta = {
182
  "filename": os.path.basename(audio_file),
 
184
  "alert_level": level
185
  }
186
 
187
+ # Send to Salesforce if a risk level is determined
188
  if level in ("High-Risk", "Medium-Risk"):
189
  try:
190
  sf_resp = send_salesforce_alert(audio_meta, detection)
191
+ alert_message = f"βœ… Detection: {label} ({score:.1f}%) β€” {level} β€” Alert sent to Salesforce (ID: {sf_resp.get('id', 'N/A')})"
192
+ except RuntimeError as e:
193
+ alert_message = f"⚠️ Detection: {label} ({score:.1f}%) β€” {level} β€” Salesforce ERROR: {e}"
194
  except Exception as e:
195
+ alert_message = f"⚠️ Detection: {label} ({score:.1f}%) β€” {level} β€” Unexpected Salesforce error: {e}"
196
+
197
+ # Clean up the converted WAV file
198
+ if os.path.exists(wav_path):
199
+ os.remove(wav_path)
200
+ print(f"[DEBUG] Cleaned up {wav_path}")
201
 
202
+ return alert_message
203
 
204
  # === Gradio UI ===
205
+ # Ensure the title and description align with the requirements
206
  iface = gr.Interface(
207
  fn=process_uploaded,
208
  inputs=[
209
+ gr.Audio(type="filepath", label="Upload Audio (or Record)"),
210
+ gr.Radio(["Start", "Stop"], label="System State", value="Start",
211
+ info="Set to 'Start' to enable audio processing and alerts."),
212
+ gr.Slider(0, 100, value=80, step=1, label="High-Risk Threshold (%)",
213
+ info="Confidence score for High-Risk scream detection."),
214
+ gr.Slider(0, 100, value=50, step=1, label="Medium-Risk Threshold (%)",
215
+ info="Confidence score for Medium-Risk scream detection.")
216
  ],
217
  outputs="text",
218
+ title="πŸ“’ Emotion-Triggered Alarm System",
219
  description="""
220
+ 🎧 Upload or record audio for real-time scream detection.
221
+ ⚠️ Alerts are sent to Salesforce for High-Risk (confidence > 80%) and Medium-Risk (confidence 50-80%) detections.
222
+ The system aims to detect panic-indicating screams.
223
  """,
224
+ allow_flagging="never" # As per requirement
225
  )
226
 
227
+ # === Optional Real-Time Listener (for Raspberry Pi or similar) ===
228
+ # This section demonstrates how a real-time listener could be implemented.
229
+ # It requires `sounddevice` and `numpy`.
230
+ # For actual deployment, environment variables for SF_URL and SF_TOKEN must be set.
231
+ # This part is commented out by default as it requires specific hardware/setup.
232
  def pi_listener(high_thresh=80, med_thresh=50, interval=1.0):
233
+ """
234
+ Simulates a real-time audio listener for devices like Raspberry Pi.
235
+ Captures audio chunks, processes them, and sends alerts.
236
+ """
237
+ try:
238
+ import sounddevice as sd
239
+ import numpy as np
240
+ except ImportError:
241
+ print("[ERROR] sounddevice or numpy not found. Real-time listener cannot be started.")
242
+ print("Please install them: pip install sounddevice numpy")
243
+ return
244
+
245
+ if classifier is None:
246
+ print("[ERROR] Model not loaded. Real-time listener cannot operate.")
247
+ return
248
 
249
  def callback(indata, frames, time_info, status):
250
+ """Callback function for sounddevice to process audio chunks."""
251
+ if status:
252
+ print(f"[WARNING] Sounddevice status: {status}")
253
+
254
+ # Ensure indata is a 1D array of float32
255
  wav = indata.squeeze()
256
+ if wav.ndim > 1:
257
+ wav = wav[:, 0] # Take first channel if stereo
258
+ wav = wav.astype(np.float32)
259
+
260
+ if len(wav) == 0:
261
+ return # Skip if no audio data
262
+
263
+ try:
264
+ # Classify the audio chunk
265
+ detection_results = classifier(wav)
266
+ if not detection_results:
267
+ return
268
+
269
+ # Get the top prediction
270
+ top_prediction = sorted(detection_results, key=lambda x: x['score'], reverse=True)[0]
271
+ lbl, sc = (top_prediction["label"].lower(), float(top_prediction["score"]) * 100)
272
+
273
+ level = "None"
274
+ if "scream" in lbl: # Check if 'scream' is in the label
275
+ if sc >= high_thresh:
276
+ level = "High-Risk"
277
+ elif sc >= med_thresh:
278
+ level = "Medium-Risk"
279
+
280
+ if level != "None":
281
+ timestamp = datetime.utcnow().isoformat() + "Z"
282
+ audio_meta = {
283
+ "filename": f"live-stream-{timestamp}",
284
+ "timestamp": timestamp,
285
+ "alert_level": level
286
+ }
287
+ detection_info = {"label": lbl, "score": sc}
288
+
289
+ try:
290
+ send_salesforce_alert(audio_meta, detection_info)
291
+ print(f"[{timestamp}] {level} scream detected ({sc:.1f}%) – alert sent.")
292
+ except RuntimeError as e:
293
+ print(f"[{timestamp}] {level} scream detected ({sc:.1f}%) – Salesforce alert failed: {e}")
294
+ except Exception as e:
295
+ print(f"[{timestamp}] {level} scream detected ({sc:.1f}%) – Unexpected error sending alert: {e}")
296
+
297
+ except Exception as e:
298
+ print(f"[ERROR] Error in real-time detection callback: {e}")
299
+
300
+ # Start audio stream
301
+ try:
302
+ # Adjust blocksize if needed for performance vs. latency
303
+ with sd.InputStream(channels=1, samplerate=16000, callback=callback, blocksize=16000): # 1 second chunks
304
+ print("πŸ”Š Real-time detection started...")
305
+ while True:
306
+ time.sleep(interval) # Keep the main thread alive
307
+ except sd.PortAudioError as e:
308
+ print(f"[ERROR] PortAudio error: {e}. Check your audio device setup.")
309
+ except Exception as e:
310
+ print(f"[ERROR] An unexpected error occurred in the real-time listener: {e}")
311
 
312
  # === App Entry ===
313
  if __name__ == "__main__":
314
+ # Optional: enable real-time listener for Raspberry Pi or similar.
315
+ # Uncomment the lines below to enable it.
316
+ # Remember to install sounddevice and numpy: pip install sounddevice numpy
317
+ # Also, ensure your system has PortAudio installed for sounddevice to work.
318
  # pi_thread = threading.Thread(target=pi_listener, daemon=True)
319
  # pi_thread.start()
320