huang_de_jun
style: update header to midnight blue
7dbe870
Raw History Blame
17.8 kB
"""
Ask The Right Question - AI-powered Q&A Assistant
Continuous listening mode: Start listening, and questions automatically
generate as the talk progresses.
"""
import os
import tempfile
import numpy as np
import gradio as gr
from dotenv import load_dotenv
from src.transcription import transcribe_audio
from src.context import ConversationContext
from src.streaming_agent import generate_questions_sync, format_questions_html, format_activity_log
# Load environment variables
load_dotenv()
# Constants
SAMPLE_RATE = 16000
CHUNK_DURATION = 5 # Process audio every 5 seconds
MIN_WORDS_FOR_QUESTIONS = 50 # Start generating questions after this many words
WORDS_BETWEEN_GENERATIONS = 30 # Generate new questions every N new words
def get_initial_state():
"""Create fresh initial state."""
return {
"context": ConversationContext(),
"questions": [],
"last_word_count": 0,
"audio_buffer": np.array([], dtype=np.float32),
"activity_log": [],
"transcript": "",
"questions_html": format_questions_html([]),
"log_html": format_activity_log([]),
"speaker_name": ""
}
def process_streaming_audio(
audio_chunk,
state: dict
) -> tuple[str, str, str, str, dict]:
"""
Process streaming audio chunk.
Args:
audio_chunk: Tuple of (sample_rate, audio_data) from gr.Audio streaming
state: Session state
Returns:
Tuple of (transcript, agent_log, questions, status, state)
"""
if audio_chunk is None:
return (
state.get("transcript", ""),
state.get("log_html", format_activity_log([])),
state.get("questions_html", format_questions_html([])),
'<div class="status-bar">Waiting for audio...</div>',
state
)
sample_rate, audio_data = audio_chunk
# Initialize state if needed
if "context" not in state:
state = get_initial_state()
ctx = state["context"]
activity_log = state.get("activity_log", [])
# Convert to float32 if needed
if audio_data.dtype != np.float32:
if audio_data.dtype == np.int16:
audio_data = audio_data.astype(np.float32) / 32768.0
elif audio_data.dtype == np.int32:
audio_data = audio_data.astype(np.float32) / 2147483648.0
else:
audio_data = audio_data.astype(np.float32)
# Handle stereo
if len(audio_data.shape) > 1:
audio_data = audio_data.mean(axis=1)
# Accumulate audio
state["audio_buffer"] = np.concatenate([state.get("audio_buffer", np.array([], dtype=np.float32)), audio_data])
# Check if we have enough audio to process (CHUNK_DURATION seconds)
samples_needed = sample_rate * CHUNK_DURATION
if len(state["audio_buffer"]) < samples_needed:
word_count = len(ctx.get_full_transcript().split())
status = f'<div class="status-bar">🎤 Listening... buffering audio ({len(state["audio_buffer"])}/{samples_needed} samples)</div>'
return (
state.get("transcript", ""),
state.get("log_html", format_activity_log(activity_log)),
state.get("questions_html", format_questions_html(state.get("questions", []))),
status,
state
)
# Process the buffered audio
audio_to_process = state["audio_buffer"][:samples_needed]
state["audio_buffer"] = state["audio_buffer"][samples_needed:]
try:
# Save to temp file for Whisper
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f:
import wave
with wave.open(f.name, 'wb') as wav:
wav.setnchannels(1)
wav.setsampwidth(2)
wav.setframerate(sample_rate)
wav.writeframes((audio_to_process * 32767).astype(np.int16).tobytes())
temp_path = f.name
# Transcribe
new_text = transcribe_audio(temp_path)
os.unlink(temp_path)
if new_text.strip():
ctx.add_transcript(new_text)
state["transcript"] = ctx.get_full_transcript()
current_word_count = len(state["transcript"].split())
words_since_last = current_word_count - state.get("last_word_count", 0)
# Add transcription to activity log
activity_log.append({"type": "transcribe", "words": current_word_count})
state["activity_log"] = activity_log
status = f'<div class="status-bar">🎤 Listening... {current_word_count} words captured</div>'
# Check if we should generate questions
should_generate = (
current_word_count >= MIN_WORDS_FOR_QUESTIONS and
words_since_last >= WORDS_BETWEEN_GENERATIONS
)
if should_generate:
state["last_word_count"] = current_word_count
# Generate questions
speaker = state.get("speaker_name", "")
questions, activity_log, questions_html, log_html = generate_questions_sync(
context=ctx,
existing_questions=state.get("questions", []),
activity_log=activity_log,
speaker_name=speaker
)
state["questions"] = questions
state["questions_html"] = questions_html
state["activity_log"] = activity_log
state["log_html"] = log_html
status = f'<div class="status-bar">✨ Generated {len(questions)} questions | {current_word_count} words</div>'
else:
# Update log HTML
state["log_html"] = format_activity_log(activity_log)
# Show waiting status if not enough words yet
if current_word_count < MIN_WORDS_FOR_QUESTIONS:
need_more = MIN_WORDS_FOR_QUESTIONS - current_word_count
status = f'<div class="status-bar">🎤 Listening... {current_word_count} words (need {need_more} more to start generating)</div>'
return (
state["transcript"],
state["log_html"],
state.get("questions_html", format_questions_html([])),
status,
state
)
except Exception as e:
activity_log.append({"type": "error", "msg": str(e)})
state["activity_log"] = activity_log
state["log_html"] = format_activity_log(activity_log)
return (
state.get("transcript", ""),
state["log_html"],
state.get("questions_html", format_questions_html([])),
f'<div class="status-bar" style="border-color:#ef4444;">❌ Error: {str(e)}</div>',
state
)
return (
state.get("transcript", ""),
state.get("log_html", format_activity_log([])),
state.get("questions_html", format_questions_html([])),
'<div class="status-bar">Processing...</div>',
state
)
def set_speaker_name(name: str, state: dict) -> dict:
"""Store speaker name in state."""
if "context" not in state:
state = get_initial_state()
state["speaker_name"] = name
return state
def add_manual_context(text: str, state: dict) -> tuple[str, str, str, str, dict]:
"""Add manually typed context and potentially trigger question generation."""
if not text.strip():
return (
state.get("transcript", ""),
state.get("log_html", format_activity_log([])),
state.get("questions_html", format_questions_html([])),
'<div class="status-bar">No text provided</div>',
state
)
# Initialize state if needed
if "context" not in state:
state = get_initial_state()
ctx = state["context"]
activity_log = state.get("activity_log", [])
ctx.add_transcript(text)
state["transcript"] = ctx.get_full_transcript()
current_word_count = len(state["transcript"].split())
# Add to activity log
activity_log.append({"type": "transcribe", "words": current_word_count})
state["activity_log"] = activity_log
status = f'<div class="status-bar">Added text. Total: {current_word_count} words</div>'
# Generate questions if enough content
if current_word_count >= MIN_WORDS_FOR_QUESTIONS:
speaker = state.get("speaker_name", "")
questions, activity_log, questions_html, log_html = generate_questions_sync(
context=ctx,
existing_questions=state.get("questions", []),
activity_log=activity_log,
speaker_name=speaker
)
state["questions"] = questions
state["questions_html"] = questions_html
state["activity_log"] = activity_log
state["log_html"] = log_html
state["last_word_count"] = current_word_count
status = f'<div class="status-bar">✨ Generated {len(questions)} questions | {current_word_count} words</div>'
else:
state["log_html"] = format_activity_log(activity_log)
need_more = MIN_WORDS_FOR_QUESTIONS - current_word_count
status = f'<div class="status-bar">Added text. Need {need_more} more words to start generating.</div>'
return (
state["transcript"],
state["log_html"],
state.get("questions_html", format_questions_html([])),
status,
state
)
def clear_session(state: dict) -> tuple[str, str, str, str, dict]:
"""Clear all session data."""
new_state = get_initial_state()
return (
"",
format_activity_log([]),
format_questions_html([]),
'<div class="status-bar">Session cleared. Ready to listen.</div>',
new_state
)
# Custom CSS
custom_css = """
.header {
text-align: center;
padding: 15px;
background: linear-gradient(135deg, #2d4a6f 0%, #1e3a5f 100%);
color: white;
border-radius: 10px;
margin-bottom: 15px;
}
.header { position: relative; }
.header h1 { margin: 0; font-size: 24px; color: white !important; }
.header p { margin: 5px 0 0 0; opacity: 0.9; font-size: 14px; color: white !important; }
.header .info-btn {
display: inline-block;
margin-top: 12px;
background: rgba(255,255,255,0.2);
border: 1px solid rgba(255,255,255,0.4);
color: white;
padding: 6px 16px;
border-radius: 20px;
cursor: pointer;
font-size: 13px;
font-weight: 500;
animation: pulse 2s ease-in-out infinite;
box-shadow: 0 0 0 0 rgba(255,255,255,0.4);
transition: background 0.2s;
}
.header .info-btn:hover {
background: rgba(255,255,255,0.3);
animation: none;
}
@keyframes pulse {
0% { box-shadow: 0 0 0 0 rgba(255,255,255,0.5); }
50% { box-shadow: 0 0 0 8px rgba(255,255,255,0); }
100% { box-shadow: 0 0 0 0 rgba(255,255,255,0); }
}
.info-modal {
display: none;
position: fixed;
top: 0; left: 0; right: 0; bottom: 0;
background: rgba(0,0,0,0.5);
z-index: 1000;
justify-content: center;
align-items: center;
}
.info-modal.show { display: flex; }
.info-modal-content {
background: white;
padding: 25px;
border-radius: 12px;
max-width: 600px;
max-height: 80vh;
overflow-y: auto;
position: relative;
margin: 20px;
line-height: 1.6;
}
.info-modal-content h2 { margin-top: 0; color: #667eea; }
.info-modal-content h3 { color: #764ba2; margin-top: 20px; }
.info-modal-content table { width: 100%; border-collapse: collapse; margin: 15px 0; font-size: 14px; }
.info-modal-content th, .info-modal-content td { padding: 8px 10px; text-align: left; border-bottom: 1px solid #eee; }
.info-modal-content th { background: #f8f9fa; font-weight: 600; }
.info-modal .close-btn {
position: absolute;
top: 10px;
right: 15px;
background: none;
border: none;
font-size: 24px;
cursor: pointer;
color: #666;
}
.info-modal .close-btn:hover { color: #333; }
.status-bar {
background: #f0f9ff;
padding: 8px 12px;
border-radius: 6px;
font-size: 13px;
border-left: 3px solid #3b82f6;
}
"""
# Build the Gradio interface
with gr.Blocks(css=custom_css, title="Ask The Right Question") as demo:
# Session state
state = gr.State(get_initial_state())
# Header with info button
gr.HTML("""
<div class="header">
<h1>Ask The Right Question</h1>
<p>Start listening and questions will generate automatically as the talk progresses</p>
<br>
<button class="info-btn" onclick="document.getElementById('infoModal').classList.add('show')">The problem we solve</button>
</div>
<!-- Info Modal -->
<div id="infoModal" class="info-modal" onclick="if(event.target===this) this.classList.remove('show')">
<div class="info-modal-content">
<button class="close-btn" onclick="document.getElementById('infoModal').classList.remove('show')">&times;</button>
<h2>Ask The Right Question</h2>
<p><strong>AI-powered assistant that helps you ask insightful questions during Q&A sessions at conferences, classes, and talks.</strong></p>
<h3>The Problem</h3>
<p>You're at a conference or talk. The speaker opens the floor for questions. Your mind goes blank. You want to ask something meaningful but coming up with good questions on the spot is hard.</p>
<h3>How It Works</h3>
<ol>
<li><strong>Listen</strong> - Record audio or paste text from the talk</li>
<li><strong>Research</strong> - AI researches speaker background and fact-checks claims</li>
<li><strong>Generate</strong> - Get thoughtful questions with explanations of why they're valuable</li>
</ol>
<h3>Question Types</h3>
<table>
<tr><th>Type</th><th>Purpose</th></tr>
<tr><td>CLARIFICATION</td><td>Seeks to understand better</td></tr>
<tr><td>DEPTH</td><td>Explores a topic more deeply</td></tr>
<tr><td>CONNECTION</td><td>Links to other fields or ideas</td></tr>
<tr><td>CHALLENGE</td><td>Probes assumptions respectfully</td></tr>
<tr><td>PRACTICAL</td><td>Asks about real-world application</td></tr>
<tr><td>FORWARD</td><td>Explores future implications</td></tr>
</table>
<p style="margin-top:20px; font-size:13px; color:#666;">Built for HuggingFace MCP Hackathon | Claude + Whisper + Tavily</p>
</div>
</div>
""")
# Status bar
status_display = gr.HTML(
value='<div class="status-bar">Ready to listen. Click the microphone or paste text to begin.</div>'
)
with gr.Row():
# Left column - Input & Transcript
with gr.Column(scale=1):
gr.Markdown("### Input")
with gr.Tab("Live Audio"):
audio_input = gr.Audio(
sources=["microphone"],
type="numpy",
streaming=True,
label="Click to start listening"
)
gr.Markdown("*Audio is processed every 5 seconds*")
with gr.Tab("Paste Text"):
manual_text = gr.Textbox(
label="Paste transcript or notes",
placeholder="Paste content from the talk here...",
lines=4
)
add_text_btn = gr.Button("Add Context", variant="secondary")
speaker_name = gr.Textbox(
label="Speaker Name (optional)",
placeholder="e.g., Dr. Jane Smith - enables background research"
)
gr.Markdown("### Transcript")
transcript_display = gr.Textbox(
label="Accumulated Content",
lines=10,
interactive=False,
placeholder="Content will appear here as you listen..."
)
clear_btn = gr.Button("Clear & Reset", variant="stop")
# Middle column - Agent Activity
with gr.Column(scale=1):
gr.Markdown("### Agent Activity")
agent_log = gr.HTML(
value=format_activity_log([])
)
# Right column - Questions
with gr.Column(scale=1):
gr.Markdown("### Questions (auto-generated)")
questions_output = gr.HTML(
value=format_questions_html([])
)
# Footer
gr.Markdown("""
---
**How it works:**
1. Click the microphone to start continuous listening (or paste text)
2. As content accumulates (50+ words), questions automatically generate
3. New questions appear every ~30 words of new content
4. Add speaker name for personalized background research
**Question types:** CLARIFY | DEPTH | CONNECT | CHALLENGE | PRACTICAL | FORWARD
---
Built for HuggingFace MCP Hackathon | Claude + Whisper + Tavily
""")
# Event handlers
audio_input.stream(
fn=process_streaming_audio,
inputs=[audio_input, state],
outputs=[transcript_display, agent_log, questions_output, status_display, state]
)
speaker_name.change(
fn=set_speaker_name,
inputs=[speaker_name, state],
outputs=[state]
)
add_text_btn.click(
fn=add_manual_context,
inputs=[manual_text, state],
outputs=[transcript_display, agent_log, questions_output, status_display, state]
).then(
fn=lambda: "",
outputs=[manual_text]
)
clear_btn.click(
fn=clear_session,
inputs=[state],
outputs=[transcript_display, agent_log, questions_output, status_display, state]
)
if __name__ == "__main__":
demo.launch()