carlosduplar commited on
Commit
b05542b
·
1 Parent(s): 36303cf

build-small-hackathon: initial Gradio + Modal app scaffold

Browse files
Files changed (12) hide show
  1. .gitignore +4 -0
  2. app.py +192 -0
  3. llm_engine.py +40 -0
  4. modal_app.py +181 -0
  5. modal_requirements.txt +7 -0
  6. parse_feedback.py +32 -0
  7. prompts.py +26 -0
  8. requirements.txt +3 -0
  9. space_README.md +43 -0
  10. stt_engine.py +23 -0
  11. style.css +156 -0
  12. tts_engine.py +44 -0
.gitignore CHANGED
@@ -6,3 +6,7 @@ coverage/
6
  *.log
7
  .env*
8
  !.env.example
 
 
 
 
 
6
  *.log
7
  .env*
8
  !.env.example
9
+ __pycache__/
10
+ *.pyc
11
+ .gradio/
12
+ test-qwen/
app.py ADDED
@@ -0,0 +1,192 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import re
2
+ import gradio as gr
3
+
4
+ from prompts import SYSTEM_PROMPT, PHASE_SWITCH_REMINDER
5
+ from parse_feedback import parse_feedback, render_feedback_table, strip_markdown
6
+ from stt_engine import transcribe
7
+ from llm_engine import chat as llm_chat
8
+ from tts_engine import synthesize
9
+
10
+ TERMINATE_RE = re.compile(r"(fin\s+de\s+(la\s+)?séance|session\s+terminée)", re.IGNORECASE)
11
+
12
+ # 7 outputs: chatbot, audio_output, state, feedback_intro, feedback_table, feedback_panel, status
13
+
14
+ def _idle_feedback():
15
+ return "", [], gr.update(open=False)
16
+
17
+ def _show_feedback(state, clean):
18
+ entries = parse_feedback(clean)
19
+ table = render_feedback_table(entries) if entries else []
20
+ intro = clean
21
+ if "Disse:" in intro:
22
+ intro = intro.split("Disse:")[0].strip()
23
+ return intro, table, gr.update(open=True)
24
+
25
+ def _chat_val(state):
26
+ return state["messages"]
27
+
28
+ def _make_audio(audio_bytes):
29
+ return audio_bytes if audio_bytes else None
30
+
31
+ def process_turn(audio_path, state):
32
+ state = dict(state)
33
+ if not audio_path:
34
+ yield _chat_val(state), None, state, *_idle_feedback(), ""
35
+ return
36
+
37
+ # 1. STT
38
+ yield _chat_val(state), None, state, *_idle_feedback(), "🎙 Transcription…"
39
+ user_text = transcribe(audio_path)
40
+ if not user_text or len(user_text.strip()) < 2:
41
+ yield _chat_val(state), None, state, *_idle_feedback(), "⛔ Parlez plus fort ou plus longtemps."
42
+ return
43
+
44
+ state["messages"].append({"role": "user", "content": user_text.strip()})
45
+
46
+ if TERMINATE_RE.search(user_text):
47
+ yield from _end_session(state)
48
+ return
49
+
50
+ # 2. LLM
51
+ yield _chat_val(state), None, state, *_idle_feedback(), "🧠 Réflexion…"
52
+ response = llm_chat(state["messages"])
53
+ if not response:
54
+ yield _chat_val(state), None, state, *_idle_feedback(), "⛔ Erreur du modèle. Réessayez."
55
+ return
56
+
57
+ clean = strip_markdown(response)
58
+ state["messages"].append({"role": "assistant", "content": clean})
59
+
60
+ # 3. TTS
61
+ yield _chat_val(state), None, state, *_idle_feedback(), "🔊 Synthèse vocale…"
62
+ audio_bytes = synthesize(clean)
63
+
64
+ yield _chat_val(state), _make_audio(audio_bytes), state, *_idle_feedback(), ""
65
+
66
+
67
+ def _end_session(state):
68
+ state["messages"].append({"role": "user", "content": PHASE_SWITCH_REMINDER})
69
+
70
+ yield _chat_val(state), None, state, *_idle_feedback(), "📝 Génération du récapitulatif…"
71
+ response = llm_chat(state["messages"])
72
+ if not response:
73
+ yield _chat_val(state), None, state, *_idle_feedback(), "⛔ Erreur lors de la génération du bilan."
74
+ return
75
+
76
+ clean = strip_markdown(response)
77
+ state["messages"].append({"role": "assistant", "content": clean})
78
+ state["phase"] = 2
79
+
80
+ intro, table, accordion = _show_feedback(state, clean)
81
+ audio_bytes = synthesize(intro)
82
+
83
+ yield _chat_val(state), _make_audio(audio_bytes), state, intro, table, accordion, ""
84
+
85
+
86
+ def end_session_click(state):
87
+ state = dict(state)
88
+ yield from _end_session(state)
89
+
90
+
91
+ def reset_session():
92
+ state = {"messages": [], "phase": 1, "turn_count": 0}
93
+ state["messages"].append({"role": "system", "content": SYSTEM_PROMPT})
94
+ return [], None, state, *_idle_feedback(), ""
95
+
96
+
97
+ # ---- Init state ----
98
+ initial_messages = [{"role": "system", "content": SYSTEM_PROMPT}]
99
+ initial_state = {"messages": list(initial_messages), "phase": 1, "turn_count": 0}
100
+
101
+ # ---- Gradio UI ----
102
+ custom_css = open("style.css", encoding="utf-8").read()
103
+
104
+ with gr.Blocks(
105
+ css=custom_css,
106
+ title="Patient Virtuel · Hygiéniste Pro",
107
+ theme=gr.themes.Soft(primary_hue="orange"),
108
+ ) as demo:
109
+ gr.HTML('<div class="atmosphere"></div>')
110
+
111
+ gr.Markdown(
112
+ '<h1 class="app-title" style="text-align:center; font-weight:400; '
113
+ 'font-family:Cormorant Garamond,serif; color:white; margin-bottom:0; '
114
+ 'font-size:28px; letter-spacing:0.02em;">'
115
+ "Patient Virtuel · Hygiéniste Pro</h1>"
116
+ )
117
+
118
+ state = gr.State(initial_state)
119
+
120
+ with gr.Row():
121
+ with gr.Column(scale=1, min_width=280):
122
+ audio_input = gr.Audio(
123
+ sources=["microphone"],
124
+ type="filepath",
125
+ show_label=False,
126
+ show_download_button=False,
127
+ waveform_options={"waveform_color": "#ff4e00", "show_controls": False},
128
+ )
129
+
130
+ gr.Markdown(
131
+ '<p style="font-size:13px; color:#888; text-align:center; '
132
+ 'margin-top:4px;">Appuyez pour parler, relâchez pour envoyer</p>'
133
+ )
134
+
135
+ with gr.Row():
136
+ btn_end = gr.Button("🟠 Terminer la séance", variant="stop", scale=2)
137
+ btn_clear = gr.Button("🗑 Nouvelle", variant="secondary", scale=1)
138
+
139
+ status = gr.Markdown("", elem_id="status-bar")
140
+
141
+ with gr.Column(scale=2):
142
+ chatbot = gr.Chatbot(
143
+ value=list(initial_messages),
144
+ type="messages",
145
+ label="Conversation",
146
+ height=480,
147
+ avatar_images=(None, "🤖"),
148
+ show_copy_button=False,
149
+ sanitize_html=True,
150
+ render_markdown=False,
151
+ )
152
+
153
+ audio_output = gr.Audio(
154
+ label="Réponse audio",
155
+ autoplay=True,
156
+ show_download_button=False,
157
+ interactive=False,
158
+ waveform_options={"waveform_color": "#ff4e00", "show_controls": False},
159
+ )
160
+
161
+ with gr.Row():
162
+ feedback_panel = gr.Accordion(
163
+ label="📋 Récapitulatif de la séance",
164
+ open=False,
165
+ )
166
+ with feedback_panel:
167
+ feedback_intro = gr.Markdown("")
168
+ feedback_table = gr.Dataframe(
169
+ headers=["Disse", "Correction", "Explication"],
170
+ datatype=["str", "str", "str"],
171
+ wrap=True,
172
+ interactive=False,
173
+ label="Erreurs relevées",
174
+ show_label=False,
175
+ )
176
+
177
+ gr.Markdown(
178
+ '<p style="font-size:11px; color:#555; text-align:center; margin-top:16px;">'
179
+ "License CC-BY-NC 4.0 (Voxtral TTS) — démonstration non-commerciale. "
180
+ "Modèle: Qwen/Qwen3.6-27B, STT: faster-whisper.</p>"
181
+ )
182
+
183
+ # ---- Event wiring ----
184
+ outputs = [chatbot, audio_output, state, feedback_intro, feedback_table, feedback_panel, status]
185
+
186
+ audio_input.change(fn=process_turn, inputs=[audio_input, state], outputs=outputs)
187
+ btn_end.click(fn=end_session_click, inputs=[state], outputs=outputs)
188
+ btn_clear.click(fn=reset_session, inputs=[], outputs=outputs)
189
+
190
+ # ---- Launch ----
191
+ if __name__ == "__main__":
192
+ demo.launch(show_api=False)
llm_engine.py ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import httpx
3
+
4
+ MODAL_ENDPOINT = os.environ.get("MODAL_ENDPOINT_QWEN", "")
5
+ MODAL_AUTH_TOKEN = os.environ.get("MODAL_AUTH_TOKEN", "")
6
+
7
+ TRUNCATION_LIMIT = 20_000 # tokens; oldest non-system turns trimmed when exceeded
8
+
9
+
10
+ def chat(messages: list[dict]) -> str | None:
11
+ if not MODAL_ENDPOINT:
12
+ raise RuntimeError("MODAL_ENDPOINT_QWEN not set")
13
+
14
+ # Trim history if too long: keep system prompt + last N turns
15
+ _trim(messages)
16
+
17
+ resp = httpx.post(
18
+ MODAL_ENDPOINT,
19
+ json={"messages": messages, "token": MODAL_AUTH_TOKEN},
20
+ timeout=600, # long timeout for cold starts
21
+ )
22
+ resp.raise_for_status()
23
+ return resp.json().get("text")
24
+
25
+
26
+ def _trim(messages: list[dict]):
27
+ """Drop oldest non-system turns if total tokens exceeds TRUNCATION_LIMIT."""
28
+ if len(messages) < 4:
29
+ return
30
+ # rough estimate: 1 token ≈ 3.5 chars
31
+ total_chars = sum(len(m.get("content", "")) for m in messages)
32
+ if total_chars < TRUNCATION_LIMIT * 3.5:
33
+ return
34
+ # keep system prompt, drop oldest user/assistant pairs
35
+ system = [m for m in messages if m.get("role") == "system"]
36
+ rest = [m for m in messages if m.get("role") != "system"]
37
+ while rest and total_chars >= TRUNCATION_LIMIT * 3.5:
38
+ dropped = rest.pop(0)
39
+ total_chars -= len(dropped.get("content", ""))
40
+ messages[:] = system + rest
modal_app.py ADDED
@@ -0,0 +1,181 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import tempfile
3
+ import soundfile as sf
4
+ import modal
5
+
6
+ app = modal.App("patient-virtuel")
7
+
8
+ # ---- Shared volumes for model cache ----
9
+ qwen_vol = modal.Volume.from_name("qwen-cache", create_if_missing=True)
10
+ whisper_vol = modal.Volume.from_name("whisper-cache", create_if_missing=True)
11
+ CACHE_DIR = "/root/.cache/huggingface"
12
+
13
+ # ---- Auth helper ----
14
+ EXPECTED_TOKEN = os.environ.get("EXPECTED_TOKEN", "")
15
+
16
+ def _check_token(token: str | None):
17
+ if not token or token != EXPECTED_TOKEN:
18
+ raise modal.exception.APIError("Unauthorized")
19
+
20
+ default_qwen_timeout = 10 * 60 # 10m for cold start + inference
21
+
22
+ # ---- 1. Qwen/Qwen3.6-27B LLM ----
23
+ qwen_image = (
24
+ modal.Image.debian_slim(python_version="3.12")
25
+ .pip_install(
26
+ "vllm>=0.8.5",
27
+ "huggingface-hub>=0.25",
28
+ "torch>=2.5",
29
+ )
30
+ )
31
+
32
+ HF_QWEN = "Qwen/Qwen3.6-27B"
33
+ GA = 15
34
+ LP = 15
35
+
36
+ @app.function(
37
+ image=qwen_image,
38
+ gpu=modal.gpu.A100(count=1, memory=40),
39
+ volumes={CACHE_DIR: qwen_vol},
40
+ secrets=[modal.Secret.from_name("hf-token")],
41
+ timeout=default_qwen_timeout,
42
+ scaledown_window=600, # keep warm 10 min after last call
43
+ concurrency_limit=1,
44
+ container_idle_timeout=600,
45
+ )
46
+ @modal.web_endpoint(method="POST", label="qwen-chat")
47
+ def qwen_chat(messages: list[dict], token: str | None = None):
48
+ _check_token(token)
49
+
50
+ from vllm import LLM, SamplingParams, TokensPrompt
51
+
52
+ llm = LLM(
53
+ model=HF_QWEN,
54
+ dtype="bfloat16",
55
+ max_model_len=32768,
56
+ trust_remote_code=True,
57
+ )
58
+
59
+ sp = SamplingParams(
60
+ temperature=0.7,
61
+ top_p=0.8,
62
+ top_k=20,
63
+ max_tokens=512,
64
+ stop=["</s>", "<|im_end|>"],
65
+ )
66
+
67
+ outputs = llm.chat(
68
+ messages=messages,
69
+ sampling_params=sp,
70
+ use_tqdm=False,
71
+ )
72
+
73
+ text = outputs[0].outputs[0].text.strip()
74
+
75
+ # Strip reasoning tokens if present
76
+ if "<think>" in text:
77
+ text = text.split("</think>")[-1].strip()
78
+
79
+ return {"text": text}
80
+
81
+
82
+ # ---- 2. Whisper STT ----
83
+ whisper_image = (
84
+ modal.Image.debian_slim(python_version="3.12")
85
+ .pip_install("faster-whisper", "numpy", "soundfile")
86
+ )
87
+
88
+ @app.function(
89
+ image=whisper_image,
90
+ gpu=modal.gpu.A10G(count=1),
91
+ volumes={CACHE_DIR: whisper_vol},
92
+ timeout=120,
93
+ scaledown_window=300,
94
+ concurrency_limit=1,
95
+ container_idle_timeout=300,
96
+ )
97
+ @modal.web_endpoint(method="POST", label="whisper-stt")
98
+ def whisper_stt(token: str | None = None, audio_bytes: bytes | None = None):
99
+ _check_token(token)
100
+
101
+ if not audio_bytes:
102
+ return {"error": "No audio bytes provided"}, 400
103
+
104
+ from faster_whisper import WhisperModel
105
+
106
+ model = WhisperModel(
107
+ "large-v3-turbo",
108
+ device="cuda",
109
+ compute_type="float16",
110
+ download_root=CACHE_DIR,
111
+ )
112
+
113
+ with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f:
114
+ f.write(audio_bytes if isinstance(audio_bytes, bytes) else audio_bytes.encode("latin1"))
115
+ path = f.name
116
+
117
+ segments, info = model.transcribe(
118
+ path,
119
+ language="fr",
120
+ beam_size=5,
121
+ vad_filter=True,
122
+ condition_on_previous_text=False,
123
+ )
124
+ os.unlink(path)
125
+
126
+ text = " ".join(seg.text for seg in segments)
127
+ return {"text": text.strip()}
128
+
129
+
130
+ # ---- 3. Mistral API TTS (wrapped for unified call from Space) ----
131
+ @app.function(
132
+ image=modal.Image.debian_slim(python_version="3.12").pip_install("httpx"),
133
+ secrets=[modal.Secret.from_name("mistral-key")],
134
+ timeout=30,
135
+ scaledown_window=60,
136
+ )
137
+ @modal.web_endpoint(method="POST", label="mistral-tts")
138
+ def mistral_tts(text: str, token: str | None = None):
139
+ _check_token(token)
140
+
141
+ import httpx as _httpx
142
+
143
+ api_key = os.environ.get("MISTRAL_API_KEY", "")
144
+ if not api_key:
145
+ return {"error": "MISTRAL_API_KEY not set"}, 500
146
+
147
+ resp = _httpx.post(
148
+ "https://api.mistral.ai/v1/audio/speech",
149
+ headers={"Authorization": f"Bearer {api_key}"},
150
+ json={
151
+ "model": "mistral-tts", # TODO: confirm exact model name on Mistral API
152
+ "input": text,
153
+ "voice": "french_female",
154
+ "response_format": "wav",
155
+ },
156
+ timeout=25,
157
+ )
158
+ resp.raise_for_status()
159
+ return resp.content # raw WAV bytes
160
+
161
+
162
+ # ---- Local dev helper ----
163
+ @app.local_entrypoint()
164
+ def test():
165
+ import httpx
166
+
167
+ print("Testing Qwen endpoint...")
168
+ r = httpx.post(
169
+ "http://localhost:8000/qwen-chat",
170
+ json={"messages": [{"role": "user", "content": "Dis bonjour en français"}], "token": EXPECTED_TOKEN},
171
+ timeout=30,
172
+ )
173
+ print(f"Qwen: {r.json()}")
174
+
175
+ print("Testing Mistral TTS endpoint...")
176
+ r = httpx.post(
177
+ "http://localhost:8000/mistral-tts",
178
+ json={"text": "Bonjour, comment allez-vous?", "token": EXPECTED_TOKEN},
179
+ timeout=30,
180
+ )
181
+ print(f"TTS: {len(r.content)} bytes")
modal_requirements.txt ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ vllm>=0.8.5
2
+ faster-whisper>=1.1
3
+ huggingface-hub>=0.25
4
+ httpx>=0.28
5
+ modal>=0.70
6
+ soundfile>=0.13
7
+ torch>=2.5
parse_feedback.py ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import re
2
+
3
+ PATTERN = re.compile(
4
+ r"Disse:\s*(.+?)\s*Correction:\s*(.+?)\s*Explication:\s*(.+?)(?=\s*Disse:|\s*$)",
5
+ re.DOTALL | re.IGNORECASE,
6
+ )
7
+
8
+ def parse_feedback(text: str) -> list[dict]:
9
+ entries = []
10
+ for m in PATTERN.finditer(text):
11
+ entries.append({
12
+ "disse": m.group(1).strip(),
13
+ "correction": m.group(2).strip(),
14
+ "explication": m.group(3).strip(),
15
+ })
16
+ return entries
17
+
18
+
19
+ def render_feedback_table(entries: list[dict]) -> list[list[str]]:
20
+ return [[e["disse"], e["correction"], e["explication"]] for e in entries]
21
+
22
+
23
+ def strip_markdown(text: str) -> str:
24
+ text = re.sub(r"\*\*(.+?)\*\*", r"\1", text)
25
+ text = re.sub(r"\*(.+?)\*", r"\1", text)
26
+ text = re.sub(r"`(.+?)`", r"\1", text)
27
+ text = re.sub(r"#{1,6}\s*", "", text)
28
+ text = re.sub(r"^[-*]\s", "", text, flags=re.MULTILINE)
29
+ text = re.sub(r"!?\[.*?\]\(.*?\)", "", text)
30
+ text = text.replace("▌", "")
31
+ text = text.replace("🤖", "")
32
+ return text.strip()
prompts.py ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ SYSTEM_PROMPT = """You are a virtual dental patient in a Swiss clinic. The user is your dental hygienist, currently training to improve their professional French.
2
+
3
+ Phase 1: Voice Roleplay
4
+ Act as a patient undergoing a 60-minute hygiene session. The flow is: Anamnesis -> Scaling (Détartrage) -> Polishing (Polissage) -> Fluoridation.
5
+ Let the user lead the conversation. Do not anticipate their lines.
6
+ Keep your answers highly concise (1-2 short sentences maximum).
7
+ Speak entirely in French.
8
+ Vary your language skills: use Swiss-French regionalisms (septante, huitante, lolette, chanterelle). Occasionally drop in a basic Swiss-German greeting or loanword ("Grüessech", "Merci vielmal").
9
+ Incorporate natural small talk at the beginning or during pauses (weather, public transport, holidays).
10
+ Occasionally ask administrative questions about cost or insurance coverage.
11
+ During the scaling phase, occasionally simulate sudden pain or sensitivity ("Aïe !", "C'est sensible ici") to test their clinical response.
12
+ NEVER break character, offer advice, or correct their French during the roleplay.
13
+ FORMATTING: Do not use emojis, asterisks, bolding, bullet points, or any markdown. Use only plain text and standard punctuation to ensure smooth text-to-speech playback.
14
+ Continue the roleplay until the user explicitly says: "Fin de la séance" or "Session terminée".
15
+
16
+ Phase 2: Feedback & Recap (Post-Roleplay)
17
+ Trigger this phase ONLY when the user uses one of the termination phrases above.
18
+ Drop the patient persona and act as an expert French language tutor.
19
+ First, respond with exactly two spoken sentences in French: encourage them, then highlight one main area of improvement regarding medical vocabulary or grammar progression.
20
+ Then IMMEDIATELY provide a structured text response with the detailed written breakdown of their mistakes. Do NOT wait for any further prompt. List the errors clearly in plain text (not a markdown table) using this format for each mistake:
21
+ Disse: [What they said in FR]
22
+ Correction: [Correction in FR]
23
+ Explication: [Explanation in FR]
24
+ """
25
+
26
+ PHASE_SWITCH_REMINDER = """Reminder: you are now an expert French language tutor providing feedback. Do NOT continue the patient roleplay. Respond with exactly two spoken sentences in French (encouragement + one area of improvement), then immediately provide the structured Disse/Correction/Explication breakdown. Use plain text only — no markdown, no emojis."""
requirements.txt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ gradio>=5.0,<6.0
2
+ httpx>=0.28,<1.0
3
+ numpy
space_README.md ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ title: Patient Virtuel · Hygiéniste Pro
3
+ emoji: 🦷
4
+ colorFrom: orange
5
+ colorTo: dark
6
+ sdk: gradio
7
+ sdk_version: 5.0
8
+ app_file: app.py
9
+ pinned: false
10
+ license: apache-2.0
11
+ ---
12
+
13
+ # Patient Virtuel · Hygiéniste Pro
14
+
15
+ A voice-based French practice tool for dental hygienists. Roleplay a 60-minute hygiene session with a Swiss virtual patient, then receive structured grammar and vocabulary feedback.
16
+
17
+ **Backyard AI** — built for a real learner training at a Swiss clinic.
18
+
19
+ ## How it works
20
+
21
+ 1. Press the mic button and speak in French to the patient
22
+ 2. The patient responds naturally, using Swiss-French regionalisms
23
+ 3. When you say "Fin de la séance", the app switches to tutor mode
24
+ 4. Receive a structured recap with corrections and explanations
25
+
26
+ ## Model credits
27
+
28
+ - **LLM**: [Qwen/Qwen3.6-27B](https://huggingface.co/Qwen/Qwen3.6-27B) via Modal (A100 GPU) — Apache 2.0
29
+ - **TTS**: [mistralai/Voxtral-4B-TTS-2603](https://huggingface.co/mistralai/Voxtral-4B-TTS-2603) via Mistral API — CC-BY-NC 4.0 (non-commercial demo)
30
+ - **STT**: [faster-whisper-large-v3-turbo](https://github.com/SYSTRAN/faster-whisper) via Modal (A10G GPU) — MIT
31
+
32
+ ## License disclaimer
33
+
34
+ The TTS model (Voxtral-4B-TTS-2603) is licensed under CC-BY-NC 4.0. This Space is a non-commercial demonstration. All other components are Apache 2.0 or MIT licensed.
35
+
36
+ ## Environment variables
37
+
38
+ | Variable | Purpose |
39
+ |---|---|
40
+ | `MODAL_ENDPOINT_QWEN` | Modal endpoint for Qwen LLM |
41
+ | `MODAL_ENDPOINT_WHISPER` | Modal endpoint for Whisper STT |
42
+ | `MODAL_AUTH_TOKEN` | Shared auth token (matching Modal's EXPECTED_TOKEN) |
43
+ | `MISTRAL_API_KEY` | Mistral API key for Voxtral TTS |
stt_engine.py ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import httpx
3
+
4
+ MODAL_ENDPOINT = os.environ.get("MODAL_ENDPOINT_WHISPER", "")
5
+ MODAL_AUTH_TOKEN = os.environ.get("MODAL_AUTH_TOKEN", "")
6
+
7
+
8
+ def transcribe(audio_path: str) -> str | None:
9
+ if not MODAL_ENDPOINT:
10
+ raise RuntimeError("MODAL_ENDPOINT_WHISPER not set")
11
+
12
+ with open(audio_path, "rb") as f:
13
+ audio_bytes = f.read()
14
+
15
+ resp = httpx.post(
16
+ MODAL_ENDPOINT,
17
+ data={"token": MODAL_AUTH_TOKEN},
18
+ files={"audio_bytes": audio_bytes},
19
+ timeout=30,
20
+ )
21
+ resp.raise_for_status()
22
+ data = resp.json()
23
+ return data.get("text")
style.css ADDED
@@ -0,0 +1,156 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ :root {
2
+ --color-bg: #0a0502;
3
+ --color-accent: #ff4e00;
4
+ --glass-surface: rgba(255, 80, 20, 0.1);
5
+ --glass-border: rgba(255, 200, 150, 0.1);
6
+ --font-serif: "Cormorant Garamond", serif;
7
+ }
8
+
9
+ html, body {
10
+ background-color: var(--color-bg) !important;
11
+ color: #fff !important;
12
+ font-family: "Inter", ui-sans-serif, system-ui, sans-serif;
13
+ overflow: hidden;
14
+ height: 100%;
15
+ }
16
+
17
+ .gradio-container {
18
+ max-width: 100% !important;
19
+ background: transparent !important;
20
+ }
21
+
22
+ /* Atmosphere gradient behind the app */
23
+ .atmosphere {
24
+ position: fixed;
25
+ top: 0; left: 0;
26
+ width: 100%; height: 100%;
27
+ background:
28
+ radial-gradient(circle at 50% 30%, #3a1510 0%, transparent 60%),
29
+ radial-gradient(circle at 10% 80%, var(--color-accent) 0%, transparent 50%);
30
+ filter: blur(80px);
31
+ opacity: 0.6;
32
+ z-index: -1;
33
+ animation: pulse 15s ease-in-out infinite alternate;
34
+ pointer-events: none;
35
+ }
36
+
37
+ @keyframes pulse {
38
+ 0% { opacity: 0.4; transform: scale(1); }
39
+ 100% { opacity: 0.7; transform: scale(1.1); }
40
+ }
41
+
42
+ /* Glass card look */
43
+ .glass-card {
44
+ background: var(--glass-surface) !important;
45
+ backdrop-filter: blur(40px) !important;
46
+ border: 1px solid var(--glass-border) !important;
47
+ box-shadow: 0 20px 50px rgba(0, 0, 0, 0.3) !important;
48
+ border-radius: 16px !important;
49
+ }
50
+
51
+ /* Feedback shell */
52
+ .feedback-shell {
53
+ background: linear-gradient(
54
+ 135deg, rgba(20, 10, 5, 0.95) 0%, rgba(30, 15, 8, 0.9) 100%
55
+ ) !important;
56
+ backdrop-filter: blur(60px) !important;
57
+ border: 1px solid rgba(255, 150, 80, 0.08) !important;
58
+ box-shadow:
59
+ 0 40px 80px rgba(0, 0, 0, 0.5),
60
+ 0 0 0 1px rgba(255, 255, 255, 0.03) inset,
61
+ 0 1px 0 rgba(255, 200, 150, 0.05) inset !important;
62
+ border-radius: 16px !important;
63
+ }
64
+
65
+ /* Recording glow animation */
66
+ @keyframes glow {
67
+ 0% { box-shadow: 0 0 10px rgba(255, 78, 0, 0.3); }
68
+ 100% { box-shadow: 0 0 30px rgba(255, 78, 0, 0.8); }
69
+ }
70
+
71
+ .recording-glow {
72
+ animation: glow 1.5s ease-in-out infinite alternate;
73
+ }
74
+
75
+ /* Brand accent button */
76
+ button.primary, button.variant-primary {
77
+ background: var(--color-accent) !important;
78
+ border-color: var(--color-accent) !important;
79
+ color: #fff !important;
80
+ }
81
+
82
+ /* Stop button */
83
+ button.variant-stop {
84
+ background: #8b0000 !important;
85
+ border-color: #8b0000 !important;
86
+ color: #fff !important;
87
+ }
88
+
89
+ /* Status bar */
90
+ #status-bar {
91
+ color: var(--color-accent);
92
+ font-size: 14px;
93
+ font-weight: 500;
94
+ height: 24px;
95
+ transition: opacity 0.2s;
96
+ }
97
+
98
+ /* Chat bubbles */
99
+ .chat-message {
100
+ border-radius: 12px !important;
101
+ font-size: 15px !important;
102
+ line-height: 1.5 !important;
103
+ }
104
+
105
+ .user-message {
106
+ background: var(--glass-surface) !important;
107
+ border: 1px solid var(--glass-border) !important;
108
+ color: #e0d0c0 !important;
109
+ }
110
+
111
+ .assistant-message {
112
+ background: rgba(255, 255, 255, 0.04) !important;
113
+ border: 1px solid rgba(255, 255, 255, 0.06) !important;
114
+ color: #f0e8dc !important;
115
+ }
116
+
117
+ /* Serif for the title only */
118
+ h1, .app-title {
119
+ font-family: var(--font-serif) !important;
120
+ font-weight: 400 !important;
121
+ letter-spacing: 0.02em !important;
122
+ }
123
+
124
+ /* Dataframe (feedback table) */
125
+ table.dataframe {
126
+ background: transparent !important;
127
+ border: none !important;
128
+ }
129
+
130
+ table.dataframe td, table.dataframe th {
131
+ background: transparent !important;
132
+ border-color: rgba(255, 200, 150, 0.1) !important;
133
+ color: #d0c0b0 !important;
134
+ }
135
+
136
+ /* Scrollbar styling */
137
+ ::-webkit-scrollbar {
138
+ width: 6px;
139
+ }
140
+ ::-webkit-scrollbar-track {
141
+ background: transparent;
142
+ }
143
+ ::-webkit-scrollbar-thumb {
144
+ background: rgba(255, 78, 0, 0.3);
145
+ border-radius: 3px;
146
+ }
147
+ ::-webkit-scrollbar-thumb:hover {
148
+ background: rgba(255, 78, 0, 0.5);
149
+ }
150
+
151
+ /* Accordion */
152
+ .accordion {
153
+ background: transparent !important;
154
+ border: 1px solid var(--glass-border) !important;
155
+ border-radius: 12px !important;
156
+ }
tts_engine.py ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import httpx
3
+
4
+ MODAL_ENDPOINT = os.environ.get("MODAL_ENDPOINT_MISTRAL_TTS", "")
5
+ MISTRAL_API_KEY = os.environ.get("MISTRAL_API_KEY", "")
6
+ MODAL_AUTH_TOKEN = os.environ.get("MODAL_AUTH_TOKEN", "")
7
+
8
+ VOICE = "french_female"
9
+ MODEL = "mistral-tts"
10
+
11
+
12
+ def synthesize(text: str) -> bytes | None:
13
+ """Tier-1: call Mistral API directly from the Space (simplest, free tier)."""
14
+ if not MISTRAL_API_KEY:
15
+ return _fallback_modal(text)
16
+
17
+ resp = httpx.post(
18
+ "https://api.mistral.ai/v1/audio/speech",
19
+ headers={"Authorization": f"Bearer {MISTRAL_API_KEY}"},
20
+ json={
21
+ "model": MODEL,
22
+ "input": text,
23
+ "voice": VOICE,
24
+ "response_format": "wav",
25
+ },
26
+ timeout=25,
27
+ )
28
+ if resp.is_error:
29
+ return _fallback_modal(text)
30
+ return resp.content
31
+
32
+
33
+ def _fallback_modal(text: str) -> bytes | None:
34
+ """Tier-2: fall back to Modal-hosted Mistral TTS."""
35
+ if not MODAL_ENDPOINT:
36
+ raise RuntimeError("No TTS endpoint available")
37
+
38
+ resp = httpx.post(
39
+ MODAL_ENDPOINT,
40
+ json={"text": text, "token": MODAL_AUTH_TOKEN},
41
+ timeout=30,
42
+ )
43
+ resp.raise_for_status()
44
+ return resp.content