carlosduplar commited on
Commit ·
b05542b
1
Parent(s): 36303cf
build-small-hackathon: initial Gradio + Modal app scaffold
Browse files- .gitignore +4 -0
- app.py +192 -0
- llm_engine.py +40 -0
- modal_app.py +181 -0
- modal_requirements.txt +7 -0
- parse_feedback.py +32 -0
- prompts.py +26 -0
- requirements.txt +3 -0
- space_README.md +43 -0
- stt_engine.py +23 -0
- style.css +156 -0
- tts_engine.py +44 -0
.gitignore
CHANGED
|
@@ -6,3 +6,7 @@ coverage/
|
|
| 6 |
*.log
|
| 7 |
.env*
|
| 8 |
!.env.example
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 6 |
*.log
|
| 7 |
.env*
|
| 8 |
!.env.example
|
| 9 |
+
__pycache__/
|
| 10 |
+
*.pyc
|
| 11 |
+
.gradio/
|
| 12 |
+
test-qwen/
|
app.py
ADDED
|
@@ -0,0 +1,192 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import re
|
| 2 |
+
import gradio as gr
|
| 3 |
+
|
| 4 |
+
from prompts import SYSTEM_PROMPT, PHASE_SWITCH_REMINDER
|
| 5 |
+
from parse_feedback import parse_feedback, render_feedback_table, strip_markdown
|
| 6 |
+
from stt_engine import transcribe
|
| 7 |
+
from llm_engine import chat as llm_chat
|
| 8 |
+
from tts_engine import synthesize
|
| 9 |
+
|
| 10 |
+
TERMINATE_RE = re.compile(r"(fin\s+de\s+(la\s+)?séance|session\s+terminée)", re.IGNORECASE)
|
| 11 |
+
|
| 12 |
+
# 7 outputs: chatbot, audio_output, state, feedback_intro, feedback_table, feedback_panel, status
|
| 13 |
+
|
| 14 |
+
def _idle_feedback():
|
| 15 |
+
return "", [], gr.update(open=False)
|
| 16 |
+
|
| 17 |
+
def _show_feedback(state, clean):
|
| 18 |
+
entries = parse_feedback(clean)
|
| 19 |
+
table = render_feedback_table(entries) if entries else []
|
| 20 |
+
intro = clean
|
| 21 |
+
if "Disse:" in intro:
|
| 22 |
+
intro = intro.split("Disse:")[0].strip()
|
| 23 |
+
return intro, table, gr.update(open=True)
|
| 24 |
+
|
| 25 |
+
def _chat_val(state):
|
| 26 |
+
return state["messages"]
|
| 27 |
+
|
| 28 |
+
def _make_audio(audio_bytes):
|
| 29 |
+
return audio_bytes if audio_bytes else None
|
| 30 |
+
|
| 31 |
+
def process_turn(audio_path, state):
|
| 32 |
+
state = dict(state)
|
| 33 |
+
if not audio_path:
|
| 34 |
+
yield _chat_val(state), None, state, *_idle_feedback(), ""
|
| 35 |
+
return
|
| 36 |
+
|
| 37 |
+
# 1. STT
|
| 38 |
+
yield _chat_val(state), None, state, *_idle_feedback(), "🎙 Transcription…"
|
| 39 |
+
user_text = transcribe(audio_path)
|
| 40 |
+
if not user_text or len(user_text.strip()) < 2:
|
| 41 |
+
yield _chat_val(state), None, state, *_idle_feedback(), "⛔ Parlez plus fort ou plus longtemps."
|
| 42 |
+
return
|
| 43 |
+
|
| 44 |
+
state["messages"].append({"role": "user", "content": user_text.strip()})
|
| 45 |
+
|
| 46 |
+
if TERMINATE_RE.search(user_text):
|
| 47 |
+
yield from _end_session(state)
|
| 48 |
+
return
|
| 49 |
+
|
| 50 |
+
# 2. LLM
|
| 51 |
+
yield _chat_val(state), None, state, *_idle_feedback(), "🧠 Réflexion…"
|
| 52 |
+
response = llm_chat(state["messages"])
|
| 53 |
+
if not response:
|
| 54 |
+
yield _chat_val(state), None, state, *_idle_feedback(), "⛔ Erreur du modèle. Réessayez."
|
| 55 |
+
return
|
| 56 |
+
|
| 57 |
+
clean = strip_markdown(response)
|
| 58 |
+
state["messages"].append({"role": "assistant", "content": clean})
|
| 59 |
+
|
| 60 |
+
# 3. TTS
|
| 61 |
+
yield _chat_val(state), None, state, *_idle_feedback(), "🔊 Synthèse vocale…"
|
| 62 |
+
audio_bytes = synthesize(clean)
|
| 63 |
+
|
| 64 |
+
yield _chat_val(state), _make_audio(audio_bytes), state, *_idle_feedback(), ""
|
| 65 |
+
|
| 66 |
+
|
| 67 |
+
def _end_session(state):
|
| 68 |
+
state["messages"].append({"role": "user", "content": PHASE_SWITCH_REMINDER})
|
| 69 |
+
|
| 70 |
+
yield _chat_val(state), None, state, *_idle_feedback(), "📝 Génération du récapitulatif…"
|
| 71 |
+
response = llm_chat(state["messages"])
|
| 72 |
+
if not response:
|
| 73 |
+
yield _chat_val(state), None, state, *_idle_feedback(), "⛔ Erreur lors de la génération du bilan."
|
| 74 |
+
return
|
| 75 |
+
|
| 76 |
+
clean = strip_markdown(response)
|
| 77 |
+
state["messages"].append({"role": "assistant", "content": clean})
|
| 78 |
+
state["phase"] = 2
|
| 79 |
+
|
| 80 |
+
intro, table, accordion = _show_feedback(state, clean)
|
| 81 |
+
audio_bytes = synthesize(intro)
|
| 82 |
+
|
| 83 |
+
yield _chat_val(state), _make_audio(audio_bytes), state, intro, table, accordion, ""
|
| 84 |
+
|
| 85 |
+
|
| 86 |
+
def end_session_click(state):
|
| 87 |
+
state = dict(state)
|
| 88 |
+
yield from _end_session(state)
|
| 89 |
+
|
| 90 |
+
|
| 91 |
+
def reset_session():
|
| 92 |
+
state = {"messages": [], "phase": 1, "turn_count": 0}
|
| 93 |
+
state["messages"].append({"role": "system", "content": SYSTEM_PROMPT})
|
| 94 |
+
return [], None, state, *_idle_feedback(), ""
|
| 95 |
+
|
| 96 |
+
|
| 97 |
+
# ---- Init state ----
|
| 98 |
+
initial_messages = [{"role": "system", "content": SYSTEM_PROMPT}]
|
| 99 |
+
initial_state = {"messages": list(initial_messages), "phase": 1, "turn_count": 0}
|
| 100 |
+
|
| 101 |
+
# ---- Gradio UI ----
|
| 102 |
+
custom_css = open("style.css", encoding="utf-8").read()
|
| 103 |
+
|
| 104 |
+
with gr.Blocks(
|
| 105 |
+
css=custom_css,
|
| 106 |
+
title="Patient Virtuel · Hygiéniste Pro",
|
| 107 |
+
theme=gr.themes.Soft(primary_hue="orange"),
|
| 108 |
+
) as demo:
|
| 109 |
+
gr.HTML('<div class="atmosphere"></div>')
|
| 110 |
+
|
| 111 |
+
gr.Markdown(
|
| 112 |
+
'<h1 class="app-title" style="text-align:center; font-weight:400; '
|
| 113 |
+
'font-family:Cormorant Garamond,serif; color:white; margin-bottom:0; '
|
| 114 |
+
'font-size:28px; letter-spacing:0.02em;">'
|
| 115 |
+
"Patient Virtuel · Hygiéniste Pro</h1>"
|
| 116 |
+
)
|
| 117 |
+
|
| 118 |
+
state = gr.State(initial_state)
|
| 119 |
+
|
| 120 |
+
with gr.Row():
|
| 121 |
+
with gr.Column(scale=1, min_width=280):
|
| 122 |
+
audio_input = gr.Audio(
|
| 123 |
+
sources=["microphone"],
|
| 124 |
+
type="filepath",
|
| 125 |
+
show_label=False,
|
| 126 |
+
show_download_button=False,
|
| 127 |
+
waveform_options={"waveform_color": "#ff4e00", "show_controls": False},
|
| 128 |
+
)
|
| 129 |
+
|
| 130 |
+
gr.Markdown(
|
| 131 |
+
'<p style="font-size:13px; color:#888; text-align:center; '
|
| 132 |
+
'margin-top:4px;">Appuyez pour parler, relâchez pour envoyer</p>'
|
| 133 |
+
)
|
| 134 |
+
|
| 135 |
+
with gr.Row():
|
| 136 |
+
btn_end = gr.Button("🟠 Terminer la séance", variant="stop", scale=2)
|
| 137 |
+
btn_clear = gr.Button("🗑 Nouvelle", variant="secondary", scale=1)
|
| 138 |
+
|
| 139 |
+
status = gr.Markdown("", elem_id="status-bar")
|
| 140 |
+
|
| 141 |
+
with gr.Column(scale=2):
|
| 142 |
+
chatbot = gr.Chatbot(
|
| 143 |
+
value=list(initial_messages),
|
| 144 |
+
type="messages",
|
| 145 |
+
label="Conversation",
|
| 146 |
+
height=480,
|
| 147 |
+
avatar_images=(None, "🤖"),
|
| 148 |
+
show_copy_button=False,
|
| 149 |
+
sanitize_html=True,
|
| 150 |
+
render_markdown=False,
|
| 151 |
+
)
|
| 152 |
+
|
| 153 |
+
audio_output = gr.Audio(
|
| 154 |
+
label="Réponse audio",
|
| 155 |
+
autoplay=True,
|
| 156 |
+
show_download_button=False,
|
| 157 |
+
interactive=False,
|
| 158 |
+
waveform_options={"waveform_color": "#ff4e00", "show_controls": False},
|
| 159 |
+
)
|
| 160 |
+
|
| 161 |
+
with gr.Row():
|
| 162 |
+
feedback_panel = gr.Accordion(
|
| 163 |
+
label="📋 Récapitulatif de la séance",
|
| 164 |
+
open=False,
|
| 165 |
+
)
|
| 166 |
+
with feedback_panel:
|
| 167 |
+
feedback_intro = gr.Markdown("")
|
| 168 |
+
feedback_table = gr.Dataframe(
|
| 169 |
+
headers=["Disse", "Correction", "Explication"],
|
| 170 |
+
datatype=["str", "str", "str"],
|
| 171 |
+
wrap=True,
|
| 172 |
+
interactive=False,
|
| 173 |
+
label="Erreurs relevées",
|
| 174 |
+
show_label=False,
|
| 175 |
+
)
|
| 176 |
+
|
| 177 |
+
gr.Markdown(
|
| 178 |
+
'<p style="font-size:11px; color:#555; text-align:center; margin-top:16px;">'
|
| 179 |
+
"License CC-BY-NC 4.0 (Voxtral TTS) — démonstration non-commerciale. "
|
| 180 |
+
"Modèle: Qwen/Qwen3.6-27B, STT: faster-whisper.</p>"
|
| 181 |
+
)
|
| 182 |
+
|
| 183 |
+
# ---- Event wiring ----
|
| 184 |
+
outputs = [chatbot, audio_output, state, feedback_intro, feedback_table, feedback_panel, status]
|
| 185 |
+
|
| 186 |
+
audio_input.change(fn=process_turn, inputs=[audio_input, state], outputs=outputs)
|
| 187 |
+
btn_end.click(fn=end_session_click, inputs=[state], outputs=outputs)
|
| 188 |
+
btn_clear.click(fn=reset_session, inputs=[], outputs=outputs)
|
| 189 |
+
|
| 190 |
+
# ---- Launch ----
|
| 191 |
+
if __name__ == "__main__":
|
| 192 |
+
demo.launch(show_api=False)
|
llm_engine.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import os
|
| 2 |
+
import httpx
|
| 3 |
+
|
| 4 |
+
MODAL_ENDPOINT = os.environ.get("MODAL_ENDPOINT_QWEN", "")
|
| 5 |
+
MODAL_AUTH_TOKEN = os.environ.get("MODAL_AUTH_TOKEN", "")
|
| 6 |
+
|
| 7 |
+
TRUNCATION_LIMIT = 20_000 # tokens; oldest non-system turns trimmed when exceeded
|
| 8 |
+
|
| 9 |
+
|
| 10 |
+
def chat(messages: list[dict]) -> str | None:
|
| 11 |
+
if not MODAL_ENDPOINT:
|
| 12 |
+
raise RuntimeError("MODAL_ENDPOINT_QWEN not set")
|
| 13 |
+
|
| 14 |
+
# Trim history if too long: keep system prompt + last N turns
|
| 15 |
+
_trim(messages)
|
| 16 |
+
|
| 17 |
+
resp = httpx.post(
|
| 18 |
+
MODAL_ENDPOINT,
|
| 19 |
+
json={"messages": messages, "token": MODAL_AUTH_TOKEN},
|
| 20 |
+
timeout=600, # long timeout for cold starts
|
| 21 |
+
)
|
| 22 |
+
resp.raise_for_status()
|
| 23 |
+
return resp.json().get("text")
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
def _trim(messages: list[dict]):
|
| 27 |
+
"""Drop oldest non-system turns if total tokens exceeds TRUNCATION_LIMIT."""
|
| 28 |
+
if len(messages) < 4:
|
| 29 |
+
return
|
| 30 |
+
# rough estimate: 1 token ≈ 3.5 chars
|
| 31 |
+
total_chars = sum(len(m.get("content", "")) for m in messages)
|
| 32 |
+
if total_chars < TRUNCATION_LIMIT * 3.5:
|
| 33 |
+
return
|
| 34 |
+
# keep system prompt, drop oldest user/assistant pairs
|
| 35 |
+
system = [m for m in messages if m.get("role") == "system"]
|
| 36 |
+
rest = [m for m in messages if m.get("role") != "system"]
|
| 37 |
+
while rest and total_chars >= TRUNCATION_LIMIT * 3.5:
|
| 38 |
+
dropped = rest.pop(0)
|
| 39 |
+
total_chars -= len(dropped.get("content", ""))
|
| 40 |
+
messages[:] = system + rest
|
modal_app.py
ADDED
|
@@ -0,0 +1,181 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import os
|
| 2 |
+
import tempfile
|
| 3 |
+
import soundfile as sf
|
| 4 |
+
import modal
|
| 5 |
+
|
| 6 |
+
app = modal.App("patient-virtuel")
|
| 7 |
+
|
| 8 |
+
# ---- Shared volumes for model cache ----
|
| 9 |
+
qwen_vol = modal.Volume.from_name("qwen-cache", create_if_missing=True)
|
| 10 |
+
whisper_vol = modal.Volume.from_name("whisper-cache", create_if_missing=True)
|
| 11 |
+
CACHE_DIR = "/root/.cache/huggingface"
|
| 12 |
+
|
| 13 |
+
# ---- Auth helper ----
|
| 14 |
+
EXPECTED_TOKEN = os.environ.get("EXPECTED_TOKEN", "")
|
| 15 |
+
|
| 16 |
+
def _check_token(token: str | None):
|
| 17 |
+
if not token or token != EXPECTED_TOKEN:
|
| 18 |
+
raise modal.exception.APIError("Unauthorized")
|
| 19 |
+
|
| 20 |
+
default_qwen_timeout = 10 * 60 # 10m for cold start + inference
|
| 21 |
+
|
| 22 |
+
# ---- 1. Qwen/Qwen3.6-27B LLM ----
|
| 23 |
+
qwen_image = (
|
| 24 |
+
modal.Image.debian_slim(python_version="3.12")
|
| 25 |
+
.pip_install(
|
| 26 |
+
"vllm>=0.8.5",
|
| 27 |
+
"huggingface-hub>=0.25",
|
| 28 |
+
"torch>=2.5",
|
| 29 |
+
)
|
| 30 |
+
)
|
| 31 |
+
|
| 32 |
+
HF_QWEN = "Qwen/Qwen3.6-27B"
|
| 33 |
+
GA = 15
|
| 34 |
+
LP = 15
|
| 35 |
+
|
| 36 |
+
@app.function(
|
| 37 |
+
image=qwen_image,
|
| 38 |
+
gpu=modal.gpu.A100(count=1, memory=40),
|
| 39 |
+
volumes={CACHE_DIR: qwen_vol},
|
| 40 |
+
secrets=[modal.Secret.from_name("hf-token")],
|
| 41 |
+
timeout=default_qwen_timeout,
|
| 42 |
+
scaledown_window=600, # keep warm 10 min after last call
|
| 43 |
+
concurrency_limit=1,
|
| 44 |
+
container_idle_timeout=600,
|
| 45 |
+
)
|
| 46 |
+
@modal.web_endpoint(method="POST", label="qwen-chat")
|
| 47 |
+
def qwen_chat(messages: list[dict], token: str | None = None):
|
| 48 |
+
_check_token(token)
|
| 49 |
+
|
| 50 |
+
from vllm import LLM, SamplingParams, TokensPrompt
|
| 51 |
+
|
| 52 |
+
llm = LLM(
|
| 53 |
+
model=HF_QWEN,
|
| 54 |
+
dtype="bfloat16",
|
| 55 |
+
max_model_len=32768,
|
| 56 |
+
trust_remote_code=True,
|
| 57 |
+
)
|
| 58 |
+
|
| 59 |
+
sp = SamplingParams(
|
| 60 |
+
temperature=0.7,
|
| 61 |
+
top_p=0.8,
|
| 62 |
+
top_k=20,
|
| 63 |
+
max_tokens=512,
|
| 64 |
+
stop=["</s>", "<|im_end|>"],
|
| 65 |
+
)
|
| 66 |
+
|
| 67 |
+
outputs = llm.chat(
|
| 68 |
+
messages=messages,
|
| 69 |
+
sampling_params=sp,
|
| 70 |
+
use_tqdm=False,
|
| 71 |
+
)
|
| 72 |
+
|
| 73 |
+
text = outputs[0].outputs[0].text.strip()
|
| 74 |
+
|
| 75 |
+
# Strip reasoning tokens if present
|
| 76 |
+
if "<think>" in text:
|
| 77 |
+
text = text.split("</think>")[-1].strip()
|
| 78 |
+
|
| 79 |
+
return {"text": text}
|
| 80 |
+
|
| 81 |
+
|
| 82 |
+
# ---- 2. Whisper STT ----
|
| 83 |
+
whisper_image = (
|
| 84 |
+
modal.Image.debian_slim(python_version="3.12")
|
| 85 |
+
.pip_install("faster-whisper", "numpy", "soundfile")
|
| 86 |
+
)
|
| 87 |
+
|
| 88 |
+
@app.function(
|
| 89 |
+
image=whisper_image,
|
| 90 |
+
gpu=modal.gpu.A10G(count=1),
|
| 91 |
+
volumes={CACHE_DIR: whisper_vol},
|
| 92 |
+
timeout=120,
|
| 93 |
+
scaledown_window=300,
|
| 94 |
+
concurrency_limit=1,
|
| 95 |
+
container_idle_timeout=300,
|
| 96 |
+
)
|
| 97 |
+
@modal.web_endpoint(method="POST", label="whisper-stt")
|
| 98 |
+
def whisper_stt(token: str | None = None, audio_bytes: bytes | None = None):
|
| 99 |
+
_check_token(token)
|
| 100 |
+
|
| 101 |
+
if not audio_bytes:
|
| 102 |
+
return {"error": "No audio bytes provided"}, 400
|
| 103 |
+
|
| 104 |
+
from faster_whisper import WhisperModel
|
| 105 |
+
|
| 106 |
+
model = WhisperModel(
|
| 107 |
+
"large-v3-turbo",
|
| 108 |
+
device="cuda",
|
| 109 |
+
compute_type="float16",
|
| 110 |
+
download_root=CACHE_DIR,
|
| 111 |
+
)
|
| 112 |
+
|
| 113 |
+
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f:
|
| 114 |
+
f.write(audio_bytes if isinstance(audio_bytes, bytes) else audio_bytes.encode("latin1"))
|
| 115 |
+
path = f.name
|
| 116 |
+
|
| 117 |
+
segments, info = model.transcribe(
|
| 118 |
+
path,
|
| 119 |
+
language="fr",
|
| 120 |
+
beam_size=5,
|
| 121 |
+
vad_filter=True,
|
| 122 |
+
condition_on_previous_text=False,
|
| 123 |
+
)
|
| 124 |
+
os.unlink(path)
|
| 125 |
+
|
| 126 |
+
text = " ".join(seg.text for seg in segments)
|
| 127 |
+
return {"text": text.strip()}
|
| 128 |
+
|
| 129 |
+
|
| 130 |
+
# ---- 3. Mistral API TTS (wrapped for unified call from Space) ----
|
| 131 |
+
@app.function(
|
| 132 |
+
image=modal.Image.debian_slim(python_version="3.12").pip_install("httpx"),
|
| 133 |
+
secrets=[modal.Secret.from_name("mistral-key")],
|
| 134 |
+
timeout=30,
|
| 135 |
+
scaledown_window=60,
|
| 136 |
+
)
|
| 137 |
+
@modal.web_endpoint(method="POST", label="mistral-tts")
|
| 138 |
+
def mistral_tts(text: str, token: str | None = None):
|
| 139 |
+
_check_token(token)
|
| 140 |
+
|
| 141 |
+
import httpx as _httpx
|
| 142 |
+
|
| 143 |
+
api_key = os.environ.get("MISTRAL_API_KEY", "")
|
| 144 |
+
if not api_key:
|
| 145 |
+
return {"error": "MISTRAL_API_KEY not set"}, 500
|
| 146 |
+
|
| 147 |
+
resp = _httpx.post(
|
| 148 |
+
"https://api.mistral.ai/v1/audio/speech",
|
| 149 |
+
headers={"Authorization": f"Bearer {api_key}"},
|
| 150 |
+
json={
|
| 151 |
+
"model": "mistral-tts", # TODO: confirm exact model name on Mistral API
|
| 152 |
+
"input": text,
|
| 153 |
+
"voice": "french_female",
|
| 154 |
+
"response_format": "wav",
|
| 155 |
+
},
|
| 156 |
+
timeout=25,
|
| 157 |
+
)
|
| 158 |
+
resp.raise_for_status()
|
| 159 |
+
return resp.content # raw WAV bytes
|
| 160 |
+
|
| 161 |
+
|
| 162 |
+
# ---- Local dev helper ----
|
| 163 |
+
@app.local_entrypoint()
|
| 164 |
+
def test():
|
| 165 |
+
import httpx
|
| 166 |
+
|
| 167 |
+
print("Testing Qwen endpoint...")
|
| 168 |
+
r = httpx.post(
|
| 169 |
+
"http://localhost:8000/qwen-chat",
|
| 170 |
+
json={"messages": [{"role": "user", "content": "Dis bonjour en français"}], "token": EXPECTED_TOKEN},
|
| 171 |
+
timeout=30,
|
| 172 |
+
)
|
| 173 |
+
print(f"Qwen: {r.json()}")
|
| 174 |
+
|
| 175 |
+
print("Testing Mistral TTS endpoint...")
|
| 176 |
+
r = httpx.post(
|
| 177 |
+
"http://localhost:8000/mistral-tts",
|
| 178 |
+
json={"text": "Bonjour, comment allez-vous?", "token": EXPECTED_TOKEN},
|
| 179 |
+
timeout=30,
|
| 180 |
+
)
|
| 181 |
+
print(f"TTS: {len(r.content)} bytes")
|
modal_requirements.txt
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
vllm>=0.8.5
|
| 2 |
+
faster-whisper>=1.1
|
| 3 |
+
huggingface-hub>=0.25
|
| 4 |
+
httpx>=0.28
|
| 5 |
+
modal>=0.70
|
| 6 |
+
soundfile>=0.13
|
| 7 |
+
torch>=2.5
|
parse_feedback.py
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import re
|
| 2 |
+
|
| 3 |
+
PATTERN = re.compile(
|
| 4 |
+
r"Disse:\s*(.+?)\s*Correction:\s*(.+?)\s*Explication:\s*(.+?)(?=\s*Disse:|\s*$)",
|
| 5 |
+
re.DOTALL | re.IGNORECASE,
|
| 6 |
+
)
|
| 7 |
+
|
| 8 |
+
def parse_feedback(text: str) -> list[dict]:
|
| 9 |
+
entries = []
|
| 10 |
+
for m in PATTERN.finditer(text):
|
| 11 |
+
entries.append({
|
| 12 |
+
"disse": m.group(1).strip(),
|
| 13 |
+
"correction": m.group(2).strip(),
|
| 14 |
+
"explication": m.group(3).strip(),
|
| 15 |
+
})
|
| 16 |
+
return entries
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
def render_feedback_table(entries: list[dict]) -> list[list[str]]:
|
| 20 |
+
return [[e["disse"], e["correction"], e["explication"]] for e in entries]
|
| 21 |
+
|
| 22 |
+
|
| 23 |
+
def strip_markdown(text: str) -> str:
|
| 24 |
+
text = re.sub(r"\*\*(.+?)\*\*", r"\1", text)
|
| 25 |
+
text = re.sub(r"\*(.+?)\*", r"\1", text)
|
| 26 |
+
text = re.sub(r"`(.+?)`", r"\1", text)
|
| 27 |
+
text = re.sub(r"#{1,6}\s*", "", text)
|
| 28 |
+
text = re.sub(r"^[-*]\s", "", text, flags=re.MULTILINE)
|
| 29 |
+
text = re.sub(r"!?\[.*?\]\(.*?\)", "", text)
|
| 30 |
+
text = text.replace("▌", "")
|
| 31 |
+
text = text.replace("🤖", "")
|
| 32 |
+
return text.strip()
|
prompts.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
SYSTEM_PROMPT = """You are a virtual dental patient in a Swiss clinic. The user is your dental hygienist, currently training to improve their professional French.
|
| 2 |
+
|
| 3 |
+
Phase 1: Voice Roleplay
|
| 4 |
+
Act as a patient undergoing a 60-minute hygiene session. The flow is: Anamnesis -> Scaling (Détartrage) -> Polishing (Polissage) -> Fluoridation.
|
| 5 |
+
Let the user lead the conversation. Do not anticipate their lines.
|
| 6 |
+
Keep your answers highly concise (1-2 short sentences maximum).
|
| 7 |
+
Speak entirely in French.
|
| 8 |
+
Vary your language skills: use Swiss-French regionalisms (septante, huitante, lolette, chanterelle). Occasionally drop in a basic Swiss-German greeting or loanword ("Grüessech", "Merci vielmal").
|
| 9 |
+
Incorporate natural small talk at the beginning or during pauses (weather, public transport, holidays).
|
| 10 |
+
Occasionally ask administrative questions about cost or insurance coverage.
|
| 11 |
+
During the scaling phase, occasionally simulate sudden pain or sensitivity ("Aïe !", "C'est sensible ici") to test their clinical response.
|
| 12 |
+
NEVER break character, offer advice, or correct their French during the roleplay.
|
| 13 |
+
FORMATTING: Do not use emojis, asterisks, bolding, bullet points, or any markdown. Use only plain text and standard punctuation to ensure smooth text-to-speech playback.
|
| 14 |
+
Continue the roleplay until the user explicitly says: "Fin de la séance" or "Session terminée".
|
| 15 |
+
|
| 16 |
+
Phase 2: Feedback & Recap (Post-Roleplay)
|
| 17 |
+
Trigger this phase ONLY when the user uses one of the termination phrases above.
|
| 18 |
+
Drop the patient persona and act as an expert French language tutor.
|
| 19 |
+
First, respond with exactly two spoken sentences in French: encourage them, then highlight one main area of improvement regarding medical vocabulary or grammar progression.
|
| 20 |
+
Then IMMEDIATELY provide a structured text response with the detailed written breakdown of their mistakes. Do NOT wait for any further prompt. List the errors clearly in plain text (not a markdown table) using this format for each mistake:
|
| 21 |
+
Disse: [What they said in FR]
|
| 22 |
+
Correction: [Correction in FR]
|
| 23 |
+
Explication: [Explanation in FR]
|
| 24 |
+
"""
|
| 25 |
+
|
| 26 |
+
PHASE_SWITCH_REMINDER = """Reminder: you are now an expert French language tutor providing feedback. Do NOT continue the patient roleplay. Respond with exactly two spoken sentences in French (encouragement + one area of improvement), then immediately provide the structured Disse/Correction/Explication breakdown. Use plain text only — no markdown, no emojis."""
|
requirements.txt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
gradio>=5.0,<6.0
|
| 2 |
+
httpx>=0.28,<1.0
|
| 3 |
+
numpy
|
space_README.md
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
title: Patient Virtuel · Hygiéniste Pro
|
| 3 |
+
emoji: 🦷
|
| 4 |
+
colorFrom: orange
|
| 5 |
+
colorTo: dark
|
| 6 |
+
sdk: gradio
|
| 7 |
+
sdk_version: 5.0
|
| 8 |
+
app_file: app.py
|
| 9 |
+
pinned: false
|
| 10 |
+
license: apache-2.0
|
| 11 |
+
---
|
| 12 |
+
|
| 13 |
+
# Patient Virtuel · Hygiéniste Pro
|
| 14 |
+
|
| 15 |
+
A voice-based French practice tool for dental hygienists. Roleplay a 60-minute hygiene session with a Swiss virtual patient, then receive structured grammar and vocabulary feedback.
|
| 16 |
+
|
| 17 |
+
**Backyard AI** — built for a real learner training at a Swiss clinic.
|
| 18 |
+
|
| 19 |
+
## How it works
|
| 20 |
+
|
| 21 |
+
1. Press the mic button and speak in French to the patient
|
| 22 |
+
2. The patient responds naturally, using Swiss-French regionalisms
|
| 23 |
+
3. When you say "Fin de la séance", the app switches to tutor mode
|
| 24 |
+
4. Receive a structured recap with corrections and explanations
|
| 25 |
+
|
| 26 |
+
## Model credits
|
| 27 |
+
|
| 28 |
+
- **LLM**: [Qwen/Qwen3.6-27B](https://huggingface.co/Qwen/Qwen3.6-27B) via Modal (A100 GPU) — Apache 2.0
|
| 29 |
+
- **TTS**: [mistralai/Voxtral-4B-TTS-2603](https://huggingface.co/mistralai/Voxtral-4B-TTS-2603) via Mistral API — CC-BY-NC 4.0 (non-commercial demo)
|
| 30 |
+
- **STT**: [faster-whisper-large-v3-turbo](https://github.com/SYSTRAN/faster-whisper) via Modal (A10G GPU) — MIT
|
| 31 |
+
|
| 32 |
+
## License disclaimer
|
| 33 |
+
|
| 34 |
+
The TTS model (Voxtral-4B-TTS-2603) is licensed under CC-BY-NC 4.0. This Space is a non-commercial demonstration. All other components are Apache 2.0 or MIT licensed.
|
| 35 |
+
|
| 36 |
+
## Environment variables
|
| 37 |
+
|
| 38 |
+
| Variable | Purpose |
|
| 39 |
+
|---|---|
|
| 40 |
+
| `MODAL_ENDPOINT_QWEN` | Modal endpoint for Qwen LLM |
|
| 41 |
+
| `MODAL_ENDPOINT_WHISPER` | Modal endpoint for Whisper STT |
|
| 42 |
+
| `MODAL_AUTH_TOKEN` | Shared auth token (matching Modal's EXPECTED_TOKEN) |
|
| 43 |
+
| `MISTRAL_API_KEY` | Mistral API key for Voxtral TTS |
|
stt_engine.py
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import os
|
| 2 |
+
import httpx
|
| 3 |
+
|
| 4 |
+
MODAL_ENDPOINT = os.environ.get("MODAL_ENDPOINT_WHISPER", "")
|
| 5 |
+
MODAL_AUTH_TOKEN = os.environ.get("MODAL_AUTH_TOKEN", "")
|
| 6 |
+
|
| 7 |
+
|
| 8 |
+
def transcribe(audio_path: str) -> str | None:
|
| 9 |
+
if not MODAL_ENDPOINT:
|
| 10 |
+
raise RuntimeError("MODAL_ENDPOINT_WHISPER not set")
|
| 11 |
+
|
| 12 |
+
with open(audio_path, "rb") as f:
|
| 13 |
+
audio_bytes = f.read()
|
| 14 |
+
|
| 15 |
+
resp = httpx.post(
|
| 16 |
+
MODAL_ENDPOINT,
|
| 17 |
+
data={"token": MODAL_AUTH_TOKEN},
|
| 18 |
+
files={"audio_bytes": audio_bytes},
|
| 19 |
+
timeout=30,
|
| 20 |
+
)
|
| 21 |
+
resp.raise_for_status()
|
| 22 |
+
data = resp.json()
|
| 23 |
+
return data.get("text")
|
style.css
ADDED
|
@@ -0,0 +1,156 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
:root {
|
| 2 |
+
--color-bg: #0a0502;
|
| 3 |
+
--color-accent: #ff4e00;
|
| 4 |
+
--glass-surface: rgba(255, 80, 20, 0.1);
|
| 5 |
+
--glass-border: rgba(255, 200, 150, 0.1);
|
| 6 |
+
--font-serif: "Cormorant Garamond", serif;
|
| 7 |
+
}
|
| 8 |
+
|
| 9 |
+
html, body {
|
| 10 |
+
background-color: var(--color-bg) !important;
|
| 11 |
+
color: #fff !important;
|
| 12 |
+
font-family: "Inter", ui-sans-serif, system-ui, sans-serif;
|
| 13 |
+
overflow: hidden;
|
| 14 |
+
height: 100%;
|
| 15 |
+
}
|
| 16 |
+
|
| 17 |
+
.gradio-container {
|
| 18 |
+
max-width: 100% !important;
|
| 19 |
+
background: transparent !important;
|
| 20 |
+
}
|
| 21 |
+
|
| 22 |
+
/* Atmosphere gradient behind the app */
|
| 23 |
+
.atmosphere {
|
| 24 |
+
position: fixed;
|
| 25 |
+
top: 0; left: 0;
|
| 26 |
+
width: 100%; height: 100%;
|
| 27 |
+
background:
|
| 28 |
+
radial-gradient(circle at 50% 30%, #3a1510 0%, transparent 60%),
|
| 29 |
+
radial-gradient(circle at 10% 80%, var(--color-accent) 0%, transparent 50%);
|
| 30 |
+
filter: blur(80px);
|
| 31 |
+
opacity: 0.6;
|
| 32 |
+
z-index: -1;
|
| 33 |
+
animation: pulse 15s ease-in-out infinite alternate;
|
| 34 |
+
pointer-events: none;
|
| 35 |
+
}
|
| 36 |
+
|
| 37 |
+
@keyframes pulse {
|
| 38 |
+
0% { opacity: 0.4; transform: scale(1); }
|
| 39 |
+
100% { opacity: 0.7; transform: scale(1.1); }
|
| 40 |
+
}
|
| 41 |
+
|
| 42 |
+
/* Glass card look */
|
| 43 |
+
.glass-card {
|
| 44 |
+
background: var(--glass-surface) !important;
|
| 45 |
+
backdrop-filter: blur(40px) !important;
|
| 46 |
+
border: 1px solid var(--glass-border) !important;
|
| 47 |
+
box-shadow: 0 20px 50px rgba(0, 0, 0, 0.3) !important;
|
| 48 |
+
border-radius: 16px !important;
|
| 49 |
+
}
|
| 50 |
+
|
| 51 |
+
/* Feedback shell */
|
| 52 |
+
.feedback-shell {
|
| 53 |
+
background: linear-gradient(
|
| 54 |
+
135deg, rgba(20, 10, 5, 0.95) 0%, rgba(30, 15, 8, 0.9) 100%
|
| 55 |
+
) !important;
|
| 56 |
+
backdrop-filter: blur(60px) !important;
|
| 57 |
+
border: 1px solid rgba(255, 150, 80, 0.08) !important;
|
| 58 |
+
box-shadow:
|
| 59 |
+
0 40px 80px rgba(0, 0, 0, 0.5),
|
| 60 |
+
0 0 0 1px rgba(255, 255, 255, 0.03) inset,
|
| 61 |
+
0 1px 0 rgba(255, 200, 150, 0.05) inset !important;
|
| 62 |
+
border-radius: 16px !important;
|
| 63 |
+
}
|
| 64 |
+
|
| 65 |
+
/* Recording glow animation */
|
| 66 |
+
@keyframes glow {
|
| 67 |
+
0% { box-shadow: 0 0 10px rgba(255, 78, 0, 0.3); }
|
| 68 |
+
100% { box-shadow: 0 0 30px rgba(255, 78, 0, 0.8); }
|
| 69 |
+
}
|
| 70 |
+
|
| 71 |
+
.recording-glow {
|
| 72 |
+
animation: glow 1.5s ease-in-out infinite alternate;
|
| 73 |
+
}
|
| 74 |
+
|
| 75 |
+
/* Brand accent button */
|
| 76 |
+
button.primary, button.variant-primary {
|
| 77 |
+
background: var(--color-accent) !important;
|
| 78 |
+
border-color: var(--color-accent) !important;
|
| 79 |
+
color: #fff !important;
|
| 80 |
+
}
|
| 81 |
+
|
| 82 |
+
/* Stop button */
|
| 83 |
+
button.variant-stop {
|
| 84 |
+
background: #8b0000 !important;
|
| 85 |
+
border-color: #8b0000 !important;
|
| 86 |
+
color: #fff !important;
|
| 87 |
+
}
|
| 88 |
+
|
| 89 |
+
/* Status bar */
|
| 90 |
+
#status-bar {
|
| 91 |
+
color: var(--color-accent);
|
| 92 |
+
font-size: 14px;
|
| 93 |
+
font-weight: 500;
|
| 94 |
+
height: 24px;
|
| 95 |
+
transition: opacity 0.2s;
|
| 96 |
+
}
|
| 97 |
+
|
| 98 |
+
/* Chat bubbles */
|
| 99 |
+
.chat-message {
|
| 100 |
+
border-radius: 12px !important;
|
| 101 |
+
font-size: 15px !important;
|
| 102 |
+
line-height: 1.5 !important;
|
| 103 |
+
}
|
| 104 |
+
|
| 105 |
+
.user-message {
|
| 106 |
+
background: var(--glass-surface) !important;
|
| 107 |
+
border: 1px solid var(--glass-border) !important;
|
| 108 |
+
color: #e0d0c0 !important;
|
| 109 |
+
}
|
| 110 |
+
|
| 111 |
+
.assistant-message {
|
| 112 |
+
background: rgba(255, 255, 255, 0.04) !important;
|
| 113 |
+
border: 1px solid rgba(255, 255, 255, 0.06) !important;
|
| 114 |
+
color: #f0e8dc !important;
|
| 115 |
+
}
|
| 116 |
+
|
| 117 |
+
/* Serif for the title only */
|
| 118 |
+
h1, .app-title {
|
| 119 |
+
font-family: var(--font-serif) !important;
|
| 120 |
+
font-weight: 400 !important;
|
| 121 |
+
letter-spacing: 0.02em !important;
|
| 122 |
+
}
|
| 123 |
+
|
| 124 |
+
/* Dataframe (feedback table) */
|
| 125 |
+
table.dataframe {
|
| 126 |
+
background: transparent !important;
|
| 127 |
+
border: none !important;
|
| 128 |
+
}
|
| 129 |
+
|
| 130 |
+
table.dataframe td, table.dataframe th {
|
| 131 |
+
background: transparent !important;
|
| 132 |
+
border-color: rgba(255, 200, 150, 0.1) !important;
|
| 133 |
+
color: #d0c0b0 !important;
|
| 134 |
+
}
|
| 135 |
+
|
| 136 |
+
/* Scrollbar styling */
|
| 137 |
+
::-webkit-scrollbar {
|
| 138 |
+
width: 6px;
|
| 139 |
+
}
|
| 140 |
+
::-webkit-scrollbar-track {
|
| 141 |
+
background: transparent;
|
| 142 |
+
}
|
| 143 |
+
::-webkit-scrollbar-thumb {
|
| 144 |
+
background: rgba(255, 78, 0, 0.3);
|
| 145 |
+
border-radius: 3px;
|
| 146 |
+
}
|
| 147 |
+
::-webkit-scrollbar-thumb:hover {
|
| 148 |
+
background: rgba(255, 78, 0, 0.5);
|
| 149 |
+
}
|
| 150 |
+
|
| 151 |
+
/* Accordion */
|
| 152 |
+
.accordion {
|
| 153 |
+
background: transparent !important;
|
| 154 |
+
border: 1px solid var(--glass-border) !important;
|
| 155 |
+
border-radius: 12px !important;
|
| 156 |
+
}
|
tts_engine.py
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import os
|
| 2 |
+
import httpx
|
| 3 |
+
|
| 4 |
+
MODAL_ENDPOINT = os.environ.get("MODAL_ENDPOINT_MISTRAL_TTS", "")
|
| 5 |
+
MISTRAL_API_KEY = os.environ.get("MISTRAL_API_KEY", "")
|
| 6 |
+
MODAL_AUTH_TOKEN = os.environ.get("MODAL_AUTH_TOKEN", "")
|
| 7 |
+
|
| 8 |
+
VOICE = "french_female"
|
| 9 |
+
MODEL = "mistral-tts"
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
def synthesize(text: str) -> bytes | None:
|
| 13 |
+
"""Tier-1: call Mistral API directly from the Space (simplest, free tier)."""
|
| 14 |
+
if not MISTRAL_API_KEY:
|
| 15 |
+
return _fallback_modal(text)
|
| 16 |
+
|
| 17 |
+
resp = httpx.post(
|
| 18 |
+
"https://api.mistral.ai/v1/audio/speech",
|
| 19 |
+
headers={"Authorization": f"Bearer {MISTRAL_API_KEY}"},
|
| 20 |
+
json={
|
| 21 |
+
"model": MODEL,
|
| 22 |
+
"input": text,
|
| 23 |
+
"voice": VOICE,
|
| 24 |
+
"response_format": "wav",
|
| 25 |
+
},
|
| 26 |
+
timeout=25,
|
| 27 |
+
)
|
| 28 |
+
if resp.is_error:
|
| 29 |
+
return _fallback_modal(text)
|
| 30 |
+
return resp.content
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
def _fallback_modal(text: str) -> bytes | None:
|
| 34 |
+
"""Tier-2: fall back to Modal-hosted Mistral TTS."""
|
| 35 |
+
if not MODAL_ENDPOINT:
|
| 36 |
+
raise RuntimeError("No TTS endpoint available")
|
| 37 |
+
|
| 38 |
+
resp = httpx.post(
|
| 39 |
+
MODAL_ENDPOINT,
|
| 40 |
+
json={"text": text, "token": MODAL_AUTH_TOKEN},
|
| 41 |
+
timeout=30,
|
| 42 |
+
)
|
| 43 |
+
resp.raise_for_status()
|
| 44 |
+
return resp.content
|