carlosduplar commited on
Commit
864a4d0
·
1 Parent(s): 7da7aca

feat: new prompt, richer feedback format, howto redesign, move legacy src

Browse files

- Replace SYSTEM_PROMPT with improved roleplay + feedback prompt
- Add --- delimiter between spoken intro and written recap
- Rewrite parse_feedback.py: Citation/Correction/Pourquoi format +
Points forts, Vocabulaire dentaire utile, Priorité, Bilan scores
- Update core.py: new delimiter, expanded TERMINATE_RE, richer result dict
- Update server_app.py to pass 4 new feedback sections
- Redesign custom_index.html feedback panel: new table headers,
sections for Points forts/Vocabulaire/Priorité/Bilan with progress bars
- Redesign .howto section with Lovable-style numbered circles
- Update app.py Gradio app for new feedback format
- Move legacy React/Gemini code to legacy/ and gitignore it

.gitignore CHANGED
@@ -10,3 +10,4 @@ __pycache__/
10
  *.pyc
11
  .gradio/
12
  test-qwen/
 
 
10
  *.pyc
11
  .gradio/
12
  test-qwen/
13
+ legacy/
.prompts.py.swp ADDED
Binary file (1.02 kB). View file
 
_modal_llamacpp_base.py ADDED
@@ -0,0 +1,326 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # _modal_llamacpp_base.py
2
+ # Shared infrastructure for all llama.cpp Modal endpoints.
3
+ # Do not deploy this file directly — import from model-specific files.
4
+
5
+ import subprocess
6
+ import time
7
+ from dataclasses import dataclass
8
+
9
+ import modal
10
+
11
+ # ─── CONSTANTS ───────────────────────────────────────────────────────────────
12
+ HOST = "0.0.0.0"
13
+ PORT = 8080
14
+ SCALEDOWN_WINDOW = 300
15
+ CONTAINER_TIMEOUT = 3600
16
+ MODEL_DIR = "/cache"
17
+
18
+ SECRETS = [
19
+ modal.Secret.from_name("api-key"),
20
+ modal.Secret.from_name("hf-token"),
21
+ ]
22
+
23
+
24
+ @dataclass(frozen=True)
25
+ class ModelConfig:
26
+ model_repo: str
27
+ model_quant: str
28
+ alias: str
29
+ gpu_type: str = "L4"
30
+ ctx_size: int = 65536
31
+ n_gpu_layers: int = 99
32
+ cache_type_k: str = "q8_0"
33
+ cache_type_v: str = "q8_0"
34
+ batch_size: int = 2048
35
+ ubatch_size: int = 512
36
+ threads: int = 4
37
+ threads_batch: int = 4
38
+ flash_attn: bool = True
39
+ speculative: bool = False
40
+ multimodal: bool = False
41
+ volume_name: str = ""
42
+ mtp_draft_n_max: int = 2
43
+ mtp_draft_p_min: float = 0.75
44
+
45
+ def __post_init__(self):
46
+ if not self.volume_name:
47
+ object.__setattr__(
48
+ self,
49
+ "volume_name",
50
+ f"llm-cache-{self.alias.replace('.', '').replace('/', '-')}",
51
+ )
52
+
53
+
54
+ # ─── SHARED IMAGE ────────────────────────────────────────────────────────────
55
+
56
+ llama_image = (
57
+ modal.Image.from_registry(
58
+ "nvidia/cuda:12.4.1-devel-ubuntu22.04", add_python="3.12"
59
+ )
60
+ .apt_install(
61
+ "build-essential",
62
+ "git",
63
+ "cmake",
64
+ "curl",
65
+ "libcurl4-openssl-dev",
66
+ "libssl-dev",
67
+ "pciutils",
68
+ )
69
+ .run_commands(
70
+ "git clone https://github.com/ggml-org/llama.cpp /llama.cpp",
71
+ "cmake /llama.cpp -B /llama.cpp/build "
72
+ "-DBUILD_SHARED_LIBS=OFF -DGGML_CUDA=ON",
73
+ "cmake --build /llama.cpp/build --config Release -j "
74
+ "--target llama-server",
75
+ )
76
+ .pip_install("fastapi[standard]", "httpx")
77
+ .env({"LLAMA_CACHE": MODEL_DIR})
78
+ .add_local_python_source("_modal_llamacpp_base")
79
+ )
80
+
81
+
82
+ # ─── HELPERS ─────────────────────────────────────────────────────────────────
83
+
84
+ def build_llama_cmd(cfg: ModelConfig) -> list[str]:
85
+ cmd = [
86
+ "/llama.cpp/build/bin/llama-server",
87
+ "-hf",
88
+ f"{cfg.model_repo}:{cfg.model_quant}",
89
+ "-ngl",
90
+ str(cfg.n_gpu_layers),
91
+ "-c",
92
+ str(cfg.ctx_size),
93
+ "-fa",
94
+ "on" if cfg.flash_attn else "off",
95
+ "-np",
96
+ "1",
97
+ "--cache-type-k",
98
+ cfg.cache_type_k,
99
+ "--cache-type-v",
100
+ cfg.cache_type_v,
101
+ "--host",
102
+ HOST,
103
+ "--port",
104
+ str(PORT),
105
+ "--threads",
106
+ str(cfg.threads),
107
+ "--threads-batch",
108
+ str(cfg.threads_batch),
109
+ "--batch-size",
110
+ str(cfg.batch_size),
111
+ "--ubatch-size",
112
+ str(cfg.ubatch_size),
113
+ "--alias",
114
+ cfg.alias,
115
+ "--jinja",
116
+ ]
117
+ if cfg.speculative:
118
+ cmd += [
119
+ "--spec-type",
120
+ "draft-mtp",
121
+ "--spec-draft-n-max",
122
+ str(cfg.mtp_draft_n_max),
123
+ "--spec-draft-p-min",
124
+ str(cfg.mtp_draft_p_min),
125
+ ]
126
+ return cmd
127
+
128
+
129
+ def make_asgi_app(cfg: ModelConfig) -> modal.App:
130
+ """Return a standalone Modal ASGI app for one model."""
131
+ app = modal.App(f"{cfg.alias.replace('.', '')}-llamacpp")
132
+ vol = modal.Volume.from_name(cfg.volume_name, create_if_missing=True)
133
+
134
+ @app.function(
135
+ image=llama_image,
136
+ gpu=cfg.gpu_type,
137
+ volumes={MODEL_DIR: vol},
138
+ secrets=SECRETS,
139
+ timeout=CONTAINER_TIMEOUT,
140
+ scaledown_window=SCALEDOWN_WINDOW,
141
+ max_containers=1,
142
+ serialized=True,
143
+ name=f"{cfg.alias.replace('.', '')}-infer",
144
+ cpu=2,
145
+ memory=6144
146
+ )
147
+ @modal.asgi_app()
148
+ def infer():
149
+ import httpx
150
+ import os
151
+
152
+ from fastapi import Depends, FastAPI, HTTPException, Request, Response, Security
153
+ from fastapi.responses import StreamingResponse
154
+ from fastapi.security import HTTPAuthorizationCredentials, HTTPBearer
155
+
156
+ fastapi_app = FastAPI()
157
+ security = HTTPBearer()
158
+ llama_url = f"http://localhost:{PORT}"
159
+
160
+ async def verify_api_key(
161
+ creds: HTTPAuthorizationCredentials = Security(security),
162
+ ):
163
+ expected = os.environ.get("API_KEY", "")
164
+ if not expected:
165
+ raise HTTPException(500, "API_KEY secret not configured")
166
+ if creds.credentials != expected:
167
+ raise HTTPException(401, "Invalid API key")
168
+
169
+ # ── Start llama-server ────────────────────────────────────────────
170
+ cmd = build_llama_cmd(cfg)
171
+ print(f"[{cfg.alias}] Starting llama-server: {' '.join(cmd)}")
172
+ stderr_log = open(f"/tmp/llama-server-{cfg.alias}.log", "w")
173
+ proc = subprocess.Popen(cmd, stderr=stderr_log)
174
+
175
+ print(f"[{cfg.alias}] Waiting for llama-server (download + load)...")
176
+ for i in range(600):
177
+ if proc.poll() is not None:
178
+ stderr_log.close()
179
+ with open(f"/tmp/llama-server-{cfg.alias}.log") as f:
180
+ stderr = f.read()
181
+ raise RuntimeError(
182
+ f"[{cfg.alias}] llama-server exited {proc.returncode}: {stderr}"
183
+ )
184
+ if i % 30 == 0 and i > 0:
185
+ print(f"[{cfg.alias}] ...still waiting ({i}s)")
186
+ try:
187
+ with httpx.Client(timeout=2) as client:
188
+ r = client.get(f"{llama_url}/health")
189
+ if r.status_code == 200:
190
+ print(f"[{cfg.alias}] llama-server ready after {i + 1}s")
191
+ break
192
+ except Exception:
193
+ pass
194
+ time.sleep(1)
195
+ else:
196
+ stderr_log.close()
197
+ proc.terminate()
198
+ with open(f"/tmp/llama-server-{cfg.alias}.log") as f:
199
+ stderr = f.read()
200
+ raise RuntimeError(
201
+ f"[{cfg.alias}] failed to start in 600s. Last logs:\n{stderr[-2000:]}"
202
+ )
203
+
204
+ # ── Endpoints ────────────────────────────────────────────────────
205
+
206
+ @fastapi_app.get("/health")
207
+ async def health(_: None = Depends(verify_api_key)):
208
+ async with httpx.AsyncClient(timeout=5) as client:
209
+ r = await client.get(f"{llama_url}/health")
210
+ status = "ok" if r.status_code == 200 else "starting"
211
+ return {"status": status}
212
+
213
+ @fastapi_app.post("/v1/chat/completions")
214
+ async def chat_completions(request: Request, _: None = Depends(verify_api_key)):
215
+ body = await request.json()
216
+ stream = body.get("stream", False)
217
+ if stream:
218
+ client = httpx.AsyncClient(timeout=300)
219
+ r = await client.send(
220
+ client.build_request(
221
+ "POST", f"{llama_url}/v1/chat/completions", json=body
222
+ ),
223
+ stream=True,
224
+ )
225
+
226
+ async def proxy_stream():
227
+ try:
228
+ async for chunk in r.aiter_bytes():
229
+ yield chunk
230
+ finally:
231
+ await r.aclose()
232
+ await client.aclose()
233
+
234
+ return StreamingResponse(
235
+ proxy_stream(), media_type="text/event-stream"
236
+ )
237
+
238
+ async with httpx.AsyncClient(timeout=300) as client:
239
+ r = await client.post(
240
+ f"{llama_url}/v1/chat/completions", json=body
241
+ )
242
+ return Response(
243
+ content=r.content,
244
+ status_code=r.status_code,
245
+ media_type="application/json",
246
+ )
247
+
248
+ @fastapi_app.post("/v1/completions")
249
+ async def completions(request: Request, _: None = Depends(verify_api_key)):
250
+ body = await request.json()
251
+ stream = body.get("stream", False)
252
+ if stream:
253
+ client = httpx.AsyncClient(timeout=300)
254
+ r = await client.send(
255
+ client.build_request(
256
+ "POST", f"{llama_url}/v1/completions", json=body
257
+ ),
258
+ stream=True,
259
+ )
260
+
261
+ async def proxy_stream():
262
+ try:
263
+ async for chunk in r.aiter_bytes():
264
+ yield chunk
265
+ finally:
266
+ await r.aclose()
267
+ await client.aclose()
268
+
269
+ return StreamingResponse(
270
+ proxy_stream(), media_type="text/event-stream"
271
+ )
272
+
273
+ async with httpx.AsyncClient(timeout=300) as client:
274
+ r = await client.post(f"{llama_url}/v1/completions", json=body)
275
+ return Response(
276
+ content=r.content,
277
+ status_code=r.status_code,
278
+ media_type="application/json",
279
+ )
280
+
281
+ @fastapi_app.get("/v1/models")
282
+ async def models(_: None = Depends(verify_api_key)):
283
+ async with httpx.AsyncClient(timeout=10) as client:
284
+ r = await client.get(f"{llama_url}/v1/models")
285
+ return Response(
286
+ content=r.content,
287
+ status_code=r.status_code,
288
+ media_type="application/json",
289
+ )
290
+
291
+ @fastapi_app.get("/gpu")
292
+ async def gpu_stats(_: None = Depends(verify_api_key)):
293
+ import subprocess as _sp
294
+
295
+ smi = _sp.check_output(
296
+ [
297
+ "nvidia-smi",
298
+ "--query-gpu=name,memory.used,memory.total,memory.free,utilization.gpu",
299
+ "--format=csv,noheader,nounits",
300
+ ],
301
+ text=True,
302
+ ).strip()
303
+ return {
304
+ "alias": cfg.alias,
305
+ "gpu": smi,
306
+ "config": {
307
+ "ctx_size": cfg.ctx_size,
308
+ "cache_type_k": cfg.cache_type_k,
309
+ "cache_type_v": cfg.cache_type_v,
310
+ "n_gpu_layers": cfg.n_gpu_layers,
311
+ "batch_size": cfg.batch_size,
312
+ "ubatch_size": cfg.ubatch_size,
313
+ "flash_attn": cfg.flash_attn,
314
+ "speculative": cfg.speculative,
315
+ "multimodal": cfg.multimodal,
316
+ },
317
+ }
318
+
319
+ return fastapi_app
320
+
321
+ return app
322
+
323
+
324
+ # ─── LOCAL TEST MODE ─────────────────────────────────────────────────────────
325
+ if __name__ == "__main__":
326
+ print("This is a shared library. Run one of the model files directly.")
app.py CHANGED
@@ -8,7 +8,10 @@ from stt_engine import transcribe, warmup
8
  from llm_engine import chat as llm_chat
9
  from tts_engine import synthesize
10
 
11
- TERMINATE_RE = re.compile(r"(fin\s+de\s+(la\s+)?séance|session\s+terminée)", re.IGNORECASE)
 
 
 
12
 
13
  # ---- i18n ----
14
  i18n = gr.I18n(
@@ -67,12 +70,10 @@ i18n = gr.I18n(
67
  def _idle_feedback():
68
  return "", [], gr.update(open=False)
69
 
70
- def _show_feedback(state, clean):
71
- entries = parse_feedback(clean)
72
- table = render_feedback_table(entries) if entries else []
73
- intro = clean
74
- if "Disse:" in intro:
75
- intro = intro.split("Disse:")[0].strip()
76
  return intro, table, gr.update(open=True)
77
 
78
  def _chat_val(state):
@@ -132,10 +133,10 @@ def _end_session(state):
132
  return
133
 
134
  clean = strip_markdown(response)
135
- state["messages"].append({"role": "assistant", "content": clean})
136
  state["phase"] = 2
137
 
138
- intro, table, accordion = _show_feedback(state, clean)
 
139
  audio_bytes = synthesize(intro)
140
 
141
  yield _chat_val(state), _make_audio(audio_bytes), state, intro, table, accordion, ""
@@ -219,7 +220,7 @@ with gr.Blocks() as demo:
219
  with feedback_panel:
220
  feedback_intro = gr.Markdown("")
221
  feedback_table = gr.Dataframe(
222
- headers=["Disse", "Correction", "Explication"],
223
  datatype=["str", "str", "str"],
224
  wrap=True,
225
  interactive=False,
 
8
  from llm_engine import chat as llm_chat
9
  from tts_engine import synthesize
10
 
11
+ TERMINATE_RE = re.compile(
12
+ r"(fin\s+de\s+(la\s+)?séance|session\s+terminée|on\s+a\s+terminé|c'est\s+fini)",
13
+ re.IGNORECASE,
14
+ )
15
 
16
  # ---- i18n ----
17
  i18n = gr.I18n(
 
70
  def _idle_feedback():
71
  return "", [], gr.update(open=False)
72
 
73
+ def _show_feedback(clean):
74
+ fb = parse_feedback(clean)
75
+ table = render_feedback_table(fb["erreurs"]) if fb["erreurs"] else []
76
+ intro = fb.get("intro") or clean
 
 
77
  return intro, table, gr.update(open=True)
78
 
79
  def _chat_val(state):
 
133
  return
134
 
135
  clean = strip_markdown(response)
 
136
  state["phase"] = 2
137
 
138
+ intro, table, accordion = _show_feedback(clean)
139
+ state["messages"].append({"role": "assistant", "content": intro})
140
  audio_bytes = synthesize(intro)
141
 
142
  yield _chat_val(state), _make_audio(audio_bytes), state, intro, table, accordion, ""
 
220
  with feedback_panel:
221
  feedback_intro = gr.Markdown("")
222
  feedback_table = gr.Dataframe(
223
+ headers=["Citation", "Correction", "Pourquoi"],
224
  datatype=["str", "str", "str"],
225
  wrap=True,
226
  interactive=False,
core.py CHANGED
@@ -8,7 +8,10 @@ from stt_engine import transcribe, warmup
8
  from llm_engine import chat as llm_chat
9
  from tts_engine import synthesize
10
 
11
- TERMINATE_RE = re.compile(r"(fin\s+de\s+(la\s+)?séance|session\s+terminée)", re.IGNORECASE)
 
 
 
12
 
13
 
14
  def make_initial_state():
@@ -29,23 +32,32 @@ def _make_audio(audio_bytes):
29
  return os.path.basename(f.name)
30
 
31
 
 
 
 
 
 
 
 
 
 
 
 
 
 
32
  def process_turn(audio_path, state):
33
- """Process one voice turn. Returns dict with all outputs."""
34
  state = dict(state)
35
  result = {
36
  "chat": _chat_val(state),
37
  "audio_file": None,
38
  "state": state,
39
- "feedback_intro": "",
40
- "feedback_table": [],
41
- "feedback_open": False,
42
  "status": "",
43
  }
44
 
45
  if not audio_path:
46
  return result
47
 
48
- # 1. STT
49
  result["status"] = "🎙 Transcription…"
50
  user_text = transcribe(audio_path)
51
  if not user_text or len(user_text.strip()) < 2:
@@ -57,7 +69,6 @@ def process_turn(audio_path, state):
57
  if TERMINATE_RE.search(user_text):
58
  return _end_session(state)
59
 
60
- # 2. LLM
61
  result["status"] = "🧠 Réflexion…"
62
  response = llm_chat(state["messages"])
63
  if not response:
@@ -67,7 +78,6 @@ def process_turn(audio_path, state):
67
  clean = strip_markdown(response)
68
  state["messages"].append({"role": "assistant", "content": clean})
69
 
70
- # 3. TTS
71
  result["status"] = "🔊 Synthèse vocale…"
72
  audio_bytes = synthesize(clean)
73
  result["audio_file"] = _make_audio(audio_bytes)
@@ -77,16 +87,13 @@ def process_turn(audio_path, state):
77
 
78
 
79
  def _end_session(state):
80
- """End session and generate recap."""
81
  state["messages"].append({"role": "user", "content": PHASE_SWITCH_REMINDER})
82
 
83
  result = {
84
  "chat": _chat_val(state),
85
  "audio_file": None,
86
  "state": state,
87
- "feedback_intro": "",
88
- "feedback_table": [],
89
- "feedback_open": False,
90
  "status": "📝 Génération du récapitulatif…",
91
  }
92
 
@@ -98,13 +105,10 @@ def _end_session(state):
98
  clean = strip_markdown(response)
99
  state["phase"] = 2
100
 
101
- entries = parse_feedback(clean)
102
- table = render_feedback_table(entries) if entries else []
103
- intro = clean
104
- if "Disse:" in intro:
105
- intro = intro.split("Disse:")[0].strip()
106
 
107
- # Only add spoken intro to chat, not raw Disse/Correction/Explication blocks
108
  state["messages"].append({"role": "assistant", "content": intro})
109
 
110
  audio_bytes = synthesize(intro)
@@ -112,18 +116,20 @@ def _end_session(state):
112
  result["chat"] = _chat_val(state)
113
  result["audio_file"] = _make_audio(audio_bytes)
114
  result["feedback_intro"] = intro
 
115
  result["feedback_table"] = table
 
 
 
116
  result["feedback_open"] = True
117
  result["status"] = ""
118
  return result
119
 
120
 
121
  def end_session_click(state):
122
- """Public wrapper for end_session."""
123
  return _end_session(dict(state))
124
 
125
 
126
  def reset_session():
127
- """Reset to initial state."""
128
  warmup()
129
  return make_initial_state()
 
8
  from llm_engine import chat as llm_chat
9
  from tts_engine import synthesize
10
 
11
+ TERMINATE_RE = re.compile(
12
+ r"(fin\s+de\s+(la\s+)?séance|session\s+terminée|on\s+a\s+terminé|c'est\s+fini)",
13
+ re.IGNORECASE,
14
+ )
15
 
16
 
17
  def make_initial_state():
 
32
  return os.path.basename(f.name)
33
 
34
 
35
+ def _default_feedback():
36
+ """Return a blank feedback result block."""
37
+ return {
38
+ "feedback_intro": "",
39
+ "feedback_points_forts": [],
40
+ "feedback_table": [],
41
+ "feedback_vocabulaire": [],
42
+ "feedback_priorite": [],
43
+ "feedback_bilan": {},
44
+ "feedback_open": False,
45
+ }
46
+
47
+
48
  def process_turn(audio_path, state):
 
49
  state = dict(state)
50
  result = {
51
  "chat": _chat_val(state),
52
  "audio_file": None,
53
  "state": state,
54
+ **_default_feedback(),
 
 
55
  "status": "",
56
  }
57
 
58
  if not audio_path:
59
  return result
60
 
 
61
  result["status"] = "🎙 Transcription…"
62
  user_text = transcribe(audio_path)
63
  if not user_text or len(user_text.strip()) < 2:
 
69
  if TERMINATE_RE.search(user_text):
70
  return _end_session(state)
71
 
 
72
  result["status"] = "🧠 Réflexion…"
73
  response = llm_chat(state["messages"])
74
  if not response:
 
78
  clean = strip_markdown(response)
79
  state["messages"].append({"role": "assistant", "content": clean})
80
 
 
81
  result["status"] = "🔊 Synthèse vocale…"
82
  audio_bytes = synthesize(clean)
83
  result["audio_file"] = _make_audio(audio_bytes)
 
87
 
88
 
89
  def _end_session(state):
 
90
  state["messages"].append({"role": "user", "content": PHASE_SWITCH_REMINDER})
91
 
92
  result = {
93
  "chat": _chat_val(state),
94
  "audio_file": None,
95
  "state": state,
96
+ **_default_feedback(),
 
 
97
  "status": "📝 Génération du récapitulatif…",
98
  }
99
 
 
105
  clean = strip_markdown(response)
106
  state["phase"] = 2
107
 
108
+ fb = parse_feedback(clean)
109
+ table = render_feedback_table(fb["erreurs"]) if fb["erreurs"] else []
110
+ intro = fb.get("intro") or clean
 
 
111
 
 
112
  state["messages"].append({"role": "assistant", "content": intro})
113
 
114
  audio_bytes = synthesize(intro)
 
116
  result["chat"] = _chat_val(state)
117
  result["audio_file"] = _make_audio(audio_bytes)
118
  result["feedback_intro"] = intro
119
+ result["feedback_points_forts"] = fb["points_forts"]
120
  result["feedback_table"] = table
121
+ result["feedback_vocabulaire"] = fb["vocabulaire"]
122
+ result["feedback_priorite"] = fb["priorite"]
123
+ result["feedback_bilan"] = fb["bilan"]
124
  result["feedback_open"] = True
125
  result["status"] = ""
126
  return result
127
 
128
 
129
  def end_session_click(state):
 
130
  return _end_session(dict(state))
131
 
132
 
133
  def reset_session():
 
134
  warmup()
135
  return make_initial_state()
custom_index.html CHANGED
@@ -134,48 +134,45 @@ body {
134
 
135
  /* How-to */
136
  .howto {
137
- padding: 20px 24px;
138
- background: var(--glass);
139
- border: 1px solid var(--glass-border);
140
- border-radius: 14px;
 
141
  }
142
- .howto h3 {
143
- font-size: 11px;
144
  text-transform: uppercase;
145
  letter-spacing: 0.15em;
146
  color: var(--accent);
147
- margin-bottom: 14px;
148
- font-weight: 600;
149
  }
150
  .howto ol {
151
  list-style: none;
152
- counter-reset: steps;
 
153
  display: flex;
154
  flex-direction: column;
155
- gap: 10px;
156
- padding: 0;
157
  }
158
  .howto li {
159
- counter-increment: steps;
160
  display: flex;
161
- align-items: flex-start;
162
  gap: 12px;
163
  font-size: 13px;
164
  color: var(--text-dim);
165
- line-height: 1.5;
166
  }
167
- .howto li::before {
168
- content: counter(steps);
169
  flex-shrink: 0;
170
- width: 22px; height: 22px;
171
  border-radius: 50%;
172
- background: var(--accent-dim);
173
- color: var(--accent);
174
- display: flex;
175
- align-items: center;
176
- justify-content: center;
177
- font-size: 11px;
178
  font-weight: 600;
 
 
179
  }
180
  .howto li strong { color: var(--text); }
181
 
@@ -448,9 +445,71 @@ body {
448
  }
449
  .feedback-table td {
450
  padding: 8px 10px;
451
- border-bottom: 1px solid rgba(255,255,255,0.04);
452
- color: var(--text-dim);
453
  line-height: 1.5;
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
454
  }
455
 
456
  /* Footer */
@@ -524,12 +583,12 @@ body {
524
 
525
  <!-- How to use -->
526
  <div class="howto">
527
- <h3>Comment utiliser</h3>
528
  <ol>
529
- <li>Appuyez sur le bouton micro pour commencer à enregistrer.</li>
530
- <li>Parlez en français — saluez le patient, posez des questions, expliquez les soins.</li>
531
- <li>Appuyez à nouveau pour arrêter — le patient vous répond à voix haute.</li>
532
- <li>Dites <strong>« Fin de la séance »</strong> ou appuyez sur <strong>Terminer</strong> pour obtenir votre bilan.</li>
533
  </ol>
534
  </div>
535
 
@@ -582,10 +641,31 @@ body {
582
  <div class="feedback-body">
583
  <div class="feedback-intro" id="feedback-intro"></div>
584
  <div class="feedback-empty" id="feedback-empty">Le bilan se construira au fil de la conversation.</div>
 
 
 
 
 
 
585
  <table class="feedback-table" id="feedback-table" style="display:none">
586
- <thead><tr><th>Disse</th><th>Correction</th><th>Explication</th></tr></thead>
587
  <tbody id="feedback-tbody"></tbody>
588
  </table>
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
589
  </div>
590
  </div>
591
  </div>
@@ -622,6 +702,14 @@ const feedbackTbody = document.getElementById("feedback-tbody");
622
  const feedbackTable = document.getElementById("feedback-table");
623
  const feedbackEmpty = document.getElementById("feedback-empty");
624
  const feedbackCount = document.getElementById("feedback-count");
 
 
 
 
 
 
 
 
625
  const audioPlayer = document.getElementById("audio-player");
626
  const btnEnd = document.getElementById("btn-end");
627
  const btnReset = document.getElementById("btn-reset");
@@ -715,7 +803,7 @@ async function processAudio(blob) {
715
 
716
  // Feedback
717
  if (d.feedback_open) {
718
- showFeedback(d.feedback_intro, d.feedback_table);
719
  }
720
 
721
  setStatus(d.status || "");
@@ -748,7 +836,7 @@ async function endSession() {
748
  await playAudio(d.audio_url);
749
  }
750
  if (d.feedback_open) {
751
- showFeedback(d.feedback_intro, d.feedback_table);
752
  }
753
  setStatus("");
754
  } catch (err) {
@@ -773,6 +861,14 @@ async function resetSession() {
773
  feedbackTbody.innerHTML = "";
774
  feedbackTable.style.display = "none";
775
  feedbackEmpty.style.display = "";
 
 
 
 
 
 
 
 
776
  feedbackCount.textContent = "0 notes";
777
  setStatus("");
778
  audioPlayer.pause();
@@ -797,16 +893,31 @@ function renderChat(messages) {
797
  }
798
 
799
  // Feedback
800
- function showFeedback(intro, table) {
801
  feedbackIntro.textContent = intro;
802
  feedbackTbody.innerHTML = "";
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
803
  if (table && table.length) {
804
  feedbackTable.style.display = "";
805
  feedbackEmpty.style.display = "none";
806
  feedbackCount.textContent = table.length + (table.length > 1 ? " notes" : " note");
807
  for (const row of table) {
808
  const tr = document.createElement("tr");
809
- const cells = Array.isArray(row) ? row : [row["Disse"], row["Correction"], row["Explication"]];
810
  for (const cell of cells) {
811
  const td = document.createElement("td");
812
  td.textContent = cell || "";
@@ -814,7 +925,68 @@ function showFeedback(intro, table) {
814
  }
815
  feedbackTbody.appendChild(tr);
816
  }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
817
  }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
818
  feedbackEl.classList.add("open");
819
  }
820
 
 
134
 
135
  /* How-to */
136
  .howto {
137
+ padding: 24px;
138
+ background: rgba(255,255,255,0.03);
139
+ border: 1px solid rgba(255,255,255,0.08);
140
+ border-radius: 12px;
141
+ backdrop-filter: blur(10px);
142
  }
143
+ .howto .howto-title {
144
+ font-size: 10px;
145
  text-transform: uppercase;
146
  letter-spacing: 0.15em;
147
  color: var(--accent);
148
+ font-weight: 700;
 
149
  }
150
  .howto ol {
151
  list-style: none;
152
+ margin: 16px 0 0;
153
+ padding: 0;
154
  display: flex;
155
  flex-direction: column;
156
+ gap: 12px;
 
157
  }
158
  .howto li {
 
159
  display: flex;
 
160
  gap: 12px;
161
  font-size: 13px;
162
  color: var(--text-dim);
163
+ line-height: 1.6;
164
  }
165
+ .howto li .num {
 
166
  flex-shrink: 0;
167
+ width: 20px; height: 20px;
168
  border-radius: 50%;
169
+ display: grid;
170
+ place-items: center;
171
+ border: 1px solid rgba(255,255,255,0.15);
172
+ font-size: 10px;
 
 
173
  font-weight: 600;
174
+ color: var(--accent);
175
+ margin-top: 2px;
176
  }
177
  .howto li strong { color: var(--text); }
178
 
 
445
  }
446
  .feedback-table td {
447
  padding: 8px 10px;
448
+ border-bottom: 1px solid var(--glass-border);
449
+ color: var(--text);
450
  line-height: 1.5;
451
+ font-size: 12px;
452
+ }
453
+
454
+ /* Feedback sections */
455
+ .feedback-section { margin-bottom: 14px; }
456
+ .feedback-section-title {
457
+ font-family: var(--serif);
458
+ font-size: 14px;
459
+ font-weight: 600;
460
+ margin-bottom: 6px;
461
+ color: var(--accent);
462
+ }
463
+ .feedback-list {
464
+ list-style: none;
465
+ margin: 0;
466
+ padding: 0;
467
+ }
468
+ .feedback-list li {
469
+ font-size: 12px;
470
+ line-height: 1.6;
471
+ padding: 2px 0;
472
+ color: var(--text);
473
+ }
474
+ .feedback-list li::before {
475
+ content: "•";
476
+ color: var(--accent);
477
+ margin-right: 6px;
478
+ }
479
+
480
+ /* Progress bars for Bilan */
481
+ .bilan-row {
482
+ display: flex;
483
+ align-items: center;
484
+ gap: 10px;
485
+ margin-bottom: 8px;
486
+ }
487
+ .bilan-label {
488
+ font-size: 11px;
489
+ text-transform: uppercase;
490
+ letter-spacing: 0.05em;
491
+ color: var(--text-dim);
492
+ width: 140px;
493
+ flex-shrink: 0;
494
+ }
495
+ .bilan-bar {
496
+ flex: 1;
497
+ height: 8px;
498
+ background: rgba(255,255,255,0.08);
499
+ border-radius: 4px;
500
+ overflow: hidden;
501
+ }
502
+ .bilan-fill {
503
+ height: 100%;
504
+ background: var(--accent);
505
+ border-radius: 4px;
506
+ transition: width 0.5s ease;
507
+ }
508
+ .bilan-score {
509
+ font-size: 11px;
510
+ color: var(--text-dim);
511
+ width: 30px;
512
+ text-align: right;
513
  }
514
 
515
  /* Footer */
 
583
 
584
  <!-- How to use -->
585
  <div class="howto">
586
+ <div class="howto-title">Comment utiliser</div>
587
  <ol>
588
+ <li><span class="num">1</span><span>Appuyez sur le bouton micro pour commencer à enregistrer.</span></li>
589
+ <li><span class="num">2</span><span>Parlez en français — saluez le patient, posez des questions, expliquez les soins.</span></li>
590
+ <li><span class="num">3</span><span>Appuyez à nouveau pour arrêter — le patient vous répond à voix haute.</span></li>
591
+ <li><span class="num">4</span><span>Dites <strong>« Fin de la séance »</strong> ou appuyez sur <strong>Terminer</strong> pour obtenir votre bilan.</span></li>
592
  </ol>
593
  </div>
594
 
 
641
  <div class="feedback-body">
642
  <div class="feedback-intro" id="feedback-intro"></div>
643
  <div class="feedback-empty" id="feedback-empty">Le bilan se construira au fil de la conversation.</div>
644
+
645
+ <div id="feedback-points-forts" class="feedback-section" style="display:none">
646
+ <div class="feedback-section-title">Points forts</div>
647
+ <ul class="feedback-list" id="feedback-points-list"></ul>
648
+ </div>
649
+
650
  <table class="feedback-table" id="feedback-table" style="display:none">
651
+ <thead><tr><th>Citation</th><th>Correction</th><th>Pourquoi</th></tr></thead>
652
  <tbody id="feedback-tbody"></tbody>
653
  </table>
654
+
655
+ <div id="feedback-vocab" class="feedback-section" style="display:none">
656
+ <div class="feedback-section-title">Vocabulaire dentaire utile</div>
657
+ <ul class="feedback-list" id="feedback-vocab-list"></ul>
658
+ </div>
659
+
660
+ <div id="feedback-priorite" class="feedback-section" style="display:none">
661
+ <div class="feedback-section-title">Priorité pour la prochaine séance</div>
662
+ <ul class="feedback-list" id="feedback-priorite-list"></ul>
663
+ </div>
664
+
665
+ <div id="feedback-bilan" class="feedback-section" style="display:none">
666
+ <div class="feedback-section-title">Bilan</div>
667
+ <div id="feedback-bilan-rows"></div>
668
+ </div>
669
  </div>
670
  </div>
671
  </div>
 
702
  const feedbackTable = document.getElementById("feedback-table");
703
  const feedbackEmpty = document.getElementById("feedback-empty");
704
  const feedbackCount = document.getElementById("feedback-count");
705
+ const feedbackPointsForts = document.getElementById("feedback-points-forts");
706
+ const feedbackPointsList = document.getElementById("feedback-points-list");
707
+ const feedbackVocab = document.getElementById("feedback-vocab");
708
+ const feedbackVocabList = document.getElementById("feedback-vocab-list");
709
+ const feedbackPriorite = document.getElementById("feedback-priorite");
710
+ const feedbackPrioriteList = document.getElementById("feedback-priorite-list");
711
+ const feedbackBilan = document.getElementById("feedback-bilan");
712
+ const feedbackBilanRows = document.getElementById("feedback-bilan-rows");
713
  const audioPlayer = document.getElementById("audio-player");
714
  const btnEnd = document.getElementById("btn-end");
715
  const btnReset = document.getElementById("btn-reset");
 
803
 
804
  // Feedback
805
  if (d.feedback_open) {
806
+ showFeedback(d.feedback_intro, d.feedback_table, d.feedback_points_forts, d.feedback_vocabulaire, d.feedback_priorite, d.feedback_bilan);
807
  }
808
 
809
  setStatus(d.status || "");
 
836
  await playAudio(d.audio_url);
837
  }
838
  if (d.feedback_open) {
839
+ showFeedback(d.feedback_intro, d.feedback_table, d.feedback_points_forts, d.feedback_vocabulaire, d.feedback_priorite, d.feedback_bilan);
840
  }
841
  setStatus("");
842
  } catch (err) {
 
861
  feedbackTbody.innerHTML = "";
862
  feedbackTable.style.display = "none";
863
  feedbackEmpty.style.display = "";
864
+ feedbackPointsForts.style.display = "none";
865
+ feedbackPointsList.innerHTML = "";
866
+ feedbackVocab.style.display = "none";
867
+ feedbackVocabList.innerHTML = "";
868
+ feedbackPriorite.style.display = "none";
869
+ feedbackPrioriteList.innerHTML = "";
870
+ feedbackBilan.style.display = "none";
871
+ feedbackBilanRows.innerHTML = "";
872
  feedbackCount.textContent = "0 notes";
873
  setStatus("");
874
  audioPlayer.pause();
 
893
  }
894
 
895
  // Feedback
896
+ function showFeedback(intro, table, pointsForts, vocabulaire, priorite, bilan) {
897
  feedbackIntro.textContent = intro;
898
  feedbackTbody.innerHTML = "";
899
+
900
+ // Points forts
901
+ if (pointsForts && pointsForts.length) {
902
+ feedbackPointsForts.style.display = "";
903
+ feedbackPointsList.innerHTML = "";
904
+ for (const item of pointsForts) {
905
+ const li = document.createElement("li");
906
+ li.textContent = item;
907
+ feedbackPointsList.appendChild(li);
908
+ }
909
+ } else {
910
+ feedbackPointsForts.style.display = "none";
911
+ }
912
+
913
+ // Error table
914
  if (table && table.length) {
915
  feedbackTable.style.display = "";
916
  feedbackEmpty.style.display = "none";
917
  feedbackCount.textContent = table.length + (table.length > 1 ? " notes" : " note");
918
  for (const row of table) {
919
  const tr = document.createElement("tr");
920
+ const cells = Array.isArray(row) ? row : [row["citation"], row["correction"], row["pourquoi"]];
921
  for (const cell of cells) {
922
  const td = document.createElement("td");
923
  td.textContent = cell || "";
 
925
  }
926
  feedbackTbody.appendChild(tr);
927
  }
928
+ } else {
929
+ feedbackTable.style.display = "none";
930
+ }
931
+
932
+ // Vocabulaire dentaire utile
933
+ if (vocabulaire && vocabulaire.length) {
934
+ feedbackVocab.style.display = "";
935
+ feedbackVocabList.innerHTML = "";
936
+ for (const item of vocabulaire) {
937
+ const li = document.createElement("li");
938
+ li.textContent = item;
939
+ feedbackVocabList.appendChild(li);
940
+ }
941
+ } else {
942
+ feedbackVocab.style.display = "none";
943
  }
944
+
945
+ // Priorité
946
+ if (priorite && priorite.length) {
947
+ feedbackPriorite.style.display = "";
948
+ feedbackPrioriteList.innerHTML = "";
949
+ for (const item of priorite) {
950
+ const li = document.createElement("li");
951
+ li.textContent = item;
952
+ feedbackPrioriteList.appendChild(li);
953
+ }
954
+ } else {
955
+ feedbackPriorite.style.display = "none";
956
+ }
957
+
958
+ // Bilan scores with progress bars
959
+ if (bilan && Object.keys(bilan).length) {
960
+ feedbackBilan.style.display = "";
961
+ feedbackBilanRows.innerHTML = "";
962
+ const labels = {
963
+ grammaire: "Grammaire",
964
+ fluidite: "Fluidité",
965
+ vocabulaire_dentaire: "Vocabulaire dentaire",
966
+ communication_clinique: "Communication clinique",
967
+ };
968
+ for (const [key, label] of Object.entries(labels)) {
969
+ const val = bilan[key];
970
+ if (val == null) continue;
971
+ const pct = Math.max(10, (val / 5) * 100);
972
+ const row = document.createElement("div");
973
+ row.className = "bilan-row";
974
+ row.innerHTML =
975
+ `<span class="bilan-label">${label}</span>` +
976
+ `<div class="bilan-bar"><div class="bilan-fill" style="width:${pct}%"></div></div>` +
977
+ `<span class="bilan-score">${val}/5</span>`;
978
+ feedbackBilanRows.appendChild(row);
979
+ }
980
+ } else {
981
+ feedbackBilan.style.display = "none";
982
+ }
983
+
984
+ // Show table count if any visible
985
+ if (!table || !table.length) {
986
+ feedbackEmpty.style.display = "none";
987
+ feedbackCount.textContent = "";
988
+ }
989
+
990
  feedbackEl.classList.add("open");
991
  }
992
 
get-modal-key.py ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+
3
+ import modal
4
+
5
+ app = modal.App()
6
+
7
+
8
+ @app.function(secrets=[modal.Secret.from_name("api-key")])
9
+ def f():
10
+ print(os.environ["API_KEY"])
modal_gemma4_26b_llamacpp.py ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # modal_gemma4_26b_llamacpp.py
2
+ # ─── Gemma 4 26B-A4B QAT on Modal (L4 GPU) ─────────────────────────────────
3
+ #
4
+ # Single-user OpenAI-compatible inference endpoint with multimodal (vision) support.
5
+ # Uses llama.cpp built from source with CUDA. No speculative decoding.
6
+ #
7
+ # Deploy:
8
+ # modal deploy modal_gemma4_26b_llamacpp.py
9
+ #
10
+ # Set secrets (one-time):
11
+ # modal secret create api-key API_KEY=$(openssl rand -hex 32)
12
+ # modal secret create hf-token HF_TOKEN=hf_your_token_here
13
+ #
14
+ # Invoke (curl):
15
+ # curl -X POST <URL>/v1/chat/completions \
16
+ # -H "Authorization: Bearer $API_KEY" \
17
+ # -H "Content-Type: application/json" \
18
+ # -d '{"model":"gemma4-26b-a4b","messages":[{"role":"user","content":"Hello"}]}'
19
+ #
20
+ # Opencode / OpenAI base URL:
21
+ # https://<workspace>--gemma4-26b-a4b-llamacpp-gemma4-26b-a4b-infer.modal.run/v1
22
+ #
23
+ # Local test (outside Modal):
24
+ # python modal_gemma4_26b_llamacpp.py
25
+ # ─────────────────────────────────────────────────────────────────────────────
26
+
27
+ from _modal_llamacpp_base import ModelConfig, build_llama_cmd, make_asgi_app
28
+
29
+ cfg = ModelConfig(
30
+ model_repo="unsloth/gemma-4-26B-A4B-it-qat-GGUF",
31
+ model_quant="UD-Q4_K_XL",
32
+ alias="gemma4-26b-a4b",
33
+ speculative=False,
34
+ multimodal=True,
35
+ )
36
+
37
+ app = make_asgi_app(cfg)
38
+
39
+ if __name__ == "__main__":
40
+ cmd = " ".join(build_llama_cmd(cfg))
41
+ print("─── Local test mode ─────────────────────────────────")
42
+ print("Command that would run on Modal (L4 GPU):\n")
43
+ print(f" {cmd}\n")
44
+ print("─── End ─────────────────────────────────────────────")
modal_gemma4_31b_llamacpp.py ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # modal_gemma4_31b_llamacpp.py
2
+ # ─── Gemma 4 31B QAT on Modal (L4 GPU) ─────────────────────────────────────
3
+ #
4
+ # Single-user OpenAI-compatible inference endpoint with multimodal (vision) support.
5
+ # Uses llama.cpp built from source with CUDA. No speculative decoding.
6
+ #
7
+ # Deploy:
8
+ # modal deploy modal_gemma4_31b_llamacpp.py
9
+ #
10
+ # Set secrets (one-time):
11
+ # modal secret create api-key API_KEY=$(openssl rand -hex 32)
12
+ # modal secret create hf-token HF_TOKEN=hf_your_token_here
13
+ #
14
+ # Invoke (curl):
15
+ # curl -X POST <URL>/v1/chat/completions \
16
+ # -H "Authorization: Bearer $API_KEY" \
17
+ # -H "Content-Type: application/json" \
18
+ # -d '{"model":"gemma4-31b","messages":[{"role":"user","content":"Hello"}]}'
19
+ #
20
+ # Opencode / OpenAI base URL:
21
+ # https://<workspace>--gemma4-31b-llamacpp-gemma4-31b-infer.modal.run/v1
22
+ #
23
+ # Local test (outside Modal):
24
+ # python modal_gemma4_31b_llamacpp.py
25
+ # ─────────────────────────────────────────────────────────────────────────────
26
+
27
+ from _modal_llamacpp_base import ModelConfig, build_llama_cmd, make_asgi_app
28
+
29
+ cfg = ModelConfig(
30
+ model_repo="unsloth/gemma-4-31B-it-qat-GGUF",
31
+ model_quant="UD-Q4_K_XL",
32
+ alias="gemma4-31b",
33
+ speculative=False,
34
+ multimodal=True,
35
+ )
36
+
37
+ app = make_asgi_app(cfg)
38
+
39
+ if __name__ == "__main__":
40
+ cmd = " ".join(build_llama_cmd(cfg))
41
+ print("─── Local test mode ─────────────────────────────────")
42
+ print("Command that would run on Modal (L4 GPU):\n")
43
+ print(f" {cmd}\n")
44
+ print("─── End ─────────────────────────────────────────────")
modal_llamacpp_proxy.py ADDED
@@ -0,0 +1,233 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # modal_llamacpp_proxy.py
2
+ # ─── Single-URL Reverse Proxy for All llama.cpp Endpoints ────────────────────
3
+ #
4
+ # Lightweight CPU-only Modal ASGI app that routes requests to the correct
5
+ # GPU backend based on the "model" field in the request body.
6
+ #
7
+ # Deploy:
8
+ # modal deploy modal_llamacpp_proxy.py
9
+ #
10
+ # Set secrets (one-time):
11
+ # modal secret create api-key API_KEY=$(openssl rand -hex 32)
12
+ #
13
+ # Models routed:
14
+ # qwen3.6-27b → Qwen3.6-27B-MTP (MTP speculative decoding)
15
+ # gemma4-26b-a4b → Gemma 4 26B-A4B QAT (multimodal)
16
+ # gemma4-31b → Gemma 4 31B QAT (multimodal)
17
+ #
18
+ # Invoke (curl):
19
+ # curl -X POST <PROXY_URL>/v1/chat/completions \
20
+ # -H "Authorization: Bearer $API_KEY" \
21
+ # -H "Content-Type: application/json" \
22
+ # -d '{"model":"qwen3.6-27b","messages":[{"role":"user","content":"Hello"}]}'
23
+ #
24
+ # Custom domain (Cloudflare):
25
+ # 1. Deploy this proxy
26
+ # 2. Modal dashboard → app → web endpoint → Settings → Custom Domain
27
+ # 3. Add CNAME in Cloudflare pointing to the Modal ingress target
28
+ # ─────────────────────────────────────────────────────────────────────────────
29
+
30
+ import modal
31
+
32
+ # ─── ROUTING TABLE ───────────────────────────────────────────────────────────
33
+ # Map model aliases to their backend Modal web endpoint URLs.
34
+ # Update these URLs after deploying each backend.
35
+
36
+ BACKENDS: dict[str, str] = {
37
+ "qwen3.6-27b": (
38
+ "https://carlosduplar--qwen36-27b-llamacpp-qwen36-27b-infer.modal.run"
39
+ ),
40
+ "gemma4-26b-a4b": (
41
+ "https://carlosduplar--gemma4-26b-a4b-llamacpp-gemma4-26b-a4b-infer.modal.run"
42
+ ),
43
+ "gemma4-31b": (
44
+ "https://carlosduplar--gemma4-31b-llamacpp-gemma4-31b-infer.modal.run"
45
+ ),
46
+ }
47
+
48
+ DEFAULT_MODEL = "qwen3.6-27b"
49
+
50
+ # ─── APP ─────────────────────────────────────────────────────────────────────
51
+
52
+ app = modal.App("llamacpp-proxy")
53
+
54
+ proxy_image = modal.Image.debian_slim(python_version="3.12").pip_install(
55
+ "fastapi[standard]", "httpx"
56
+ )
57
+
58
+
59
+ @app.function(
60
+ image=proxy_image,
61
+ secrets=[modal.Secret.from_name("api-key")],
62
+ timeout=300,
63
+ scaledown_window=300,
64
+ max_containers=1,
65
+ cpu=0.25,
66
+ memory=512,
67
+ )
68
+ @modal.asgi_app()
69
+ def infer():
70
+ import os
71
+
72
+ import httpx
73
+ from fastapi import HTTPException, Request, Response, Security
74
+ from fastapi.responses import StreamingResponse
75
+ from fastapi.security import HTTPAuthorizationCredentials, HTTPBearer
76
+
77
+ from fastapi import FastAPI
78
+
79
+ fastapi_app = FastAPI(title="llamacpp-proxy")
80
+ security = HTTPBearer()
81
+
82
+ async def verify_api_key(
83
+ creds: HTTPAuthorizationCredentials = Security(security),
84
+ ):
85
+ expected = os.environ.get("API_KEY", "")
86
+ if not expected:
87
+ raise HTTPException(500, "API_KEY secret not configured")
88
+ if creds.credentials != expected:
89
+ raise HTTPException(401, "Invalid API key")
90
+
91
+ def _auth_headers(creds: HTTPAuthorizationCredentials) -> dict[str, str]:
92
+ return {"Authorization": f"Bearer {creds.credentials}"}
93
+
94
+ def _backend_for(model: str | None) -> tuple[str, str]:
95
+ """Return (alias, base_url) for the given model name."""
96
+ alias = model or DEFAULT_MODEL
97
+ base = BACKENDS.get(alias)
98
+ if not base:
99
+ raise HTTPException(
100
+ 400,
101
+ f"Unknown model '{alias}'. Available: {', '.join(BACKENDS)}",
102
+ )
103
+ return alias, base
104
+
105
+ # ── Health: check all backends ────────────────────────────────────────
106
+
107
+ @fastapi_app.get("/health")
108
+ async def health(creds: HTTPAuthorizationCredentials = Security(security)):
109
+ results = {}
110
+ headers = _auth_headers(creds)
111
+ async with httpx.AsyncClient(timeout=5) as client:
112
+ for alias, base in BACKENDS.items():
113
+ try:
114
+ r = await client.get(f"{base}/health", headers=headers)
115
+ results[alias] = "ok" if r.status_code == 200 else "starting"
116
+ except Exception:
117
+ results[alias] = "unreachable"
118
+ return {"status": results}
119
+
120
+ # ── Models: merge from all backends ───────────────────────────────────
121
+
122
+ @fastapi_app.get("/v1/models")
123
+ async def models(creds: HTTPAuthorizationCredentials = Security(security)):
124
+ all_models = []
125
+ headers = _auth_headers(creds)
126
+ async with httpx.AsyncClient(timeout=10) as client:
127
+ for base in BACKENDS.values():
128
+ try:
129
+ r = await client.get(f"{base}/v1/models", headers=headers)
130
+ if r.status_code == 200:
131
+ data = r.json()
132
+ all_models.extend(data.get("data", []))
133
+ except Exception:
134
+ pass
135
+ return {"object": "list", "data": all_models}
136
+
137
+ # ── Chat completions ──────────────────────────────────────────────────
138
+
139
+ @fastapi_app.post("/v1/chat/completions")
140
+ async def chat_completions(
141
+ request: Request,
142
+ creds: HTTPAuthorizationCredentials = Security(security),
143
+ ):
144
+ body = await request.json()
145
+ alias, base = _backend_for(body.get("model"))
146
+ stream = body.get("stream", False)
147
+ headers = _auth_headers(creds)
148
+
149
+ if stream:
150
+ client = httpx.AsyncClient(timeout=300)
151
+ r = await client.send(
152
+ client.build_request(
153
+ "POST", f"{base}/v1/chat/completions", json=body, headers=headers
154
+ ),
155
+ stream=True,
156
+ )
157
+
158
+ async def proxy_stream():
159
+ try:
160
+ async for chunk in r.aiter_bytes():
161
+ yield chunk
162
+ finally:
163
+ await r.aclose()
164
+ await client.aclose()
165
+
166
+ return StreamingResponse(proxy_stream(), media_type="text/event-stream")
167
+
168
+ async with httpx.AsyncClient(timeout=300) as client:
169
+ r = await client.post(f"{base}/v1/chat/completions", json=body, headers=headers)
170
+ return Response(
171
+ content=r.content,
172
+ status_code=r.status_code,
173
+ media_type="application/json",
174
+ )
175
+
176
+ # ── Completions ───────────────────────────────────────────────────────
177
+
178
+ @fastapi_app.post("/v1/completions")
179
+ async def completions(
180
+ request: Request,
181
+ creds: HTTPAuthorizationCredentials = Security(security),
182
+ ):
183
+ body = await request.json()
184
+ alias, base = _backend_for(body.get("model"))
185
+ stream = body.get("stream", False)
186
+ headers = _auth_headers(creds)
187
+
188
+ if stream:
189
+ client = httpx.AsyncClient(timeout=300)
190
+ r = await client.send(
191
+ client.build_request(
192
+ "POST", f"{base}/v1/completions", json=body, headers=headers
193
+ ),
194
+ stream=True,
195
+ )
196
+
197
+ async def proxy_stream():
198
+ try:
199
+ async for chunk in r.aiter_bytes():
200
+ yield chunk
201
+ finally:
202
+ await r.aclose()
203
+ await client.aclose()
204
+
205
+ return StreamingResponse(proxy_stream(), media_type="text/event-stream")
206
+
207
+ async with httpx.AsyncClient(timeout=300) as client:
208
+ r = await client.post(f"{base}/v1/completions", json=body, headers=headers)
209
+ return Response(
210
+ content=r.content,
211
+ status_code=r.status_code,
212
+ media_type="application/json",
213
+ )
214
+
215
+ # ── GPU stats: all backends ───────────────────────────────────────────
216
+
217
+ @fastapi_app.get("/gpu")
218
+ async def gpu_stats(creds: HTTPAuthorizationCredentials = Security(security)):
219
+ results = {}
220
+ headers = _auth_headers(creds)
221
+ async with httpx.AsyncClient(timeout=10) as client:
222
+ for alias, base in BACKENDS.items():
223
+ try:
224
+ r = await client.get(f"{base}/gpu", headers=headers)
225
+ if r.status_code == 200:
226
+ results[alias] = r.json()
227
+ else:
228
+ results[alias] = {"error": f"status {r.status_code}"}
229
+ except Exception as e:
230
+ results[alias] = {"error": str(e)}
231
+ return {"backends": results}
232
+
233
+ return fastapi_app
modal_qwen36_llamacpp.py ADDED
@@ -0,0 +1,52 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # modal_qwen36_llamacpp.py
2
+ # ─── Qwen3.6-35B-A3B-MTP-GGUF on Modal (L4 GPU) ───────────────────────────
3
+ #
4
+ # Single-user OpenAI-compatible inference endpoint for coding agent workflows.
5
+ # Uses llama.cpp built from source with CUDA, MTP speculative decoding enabled.
6
+ #
7
+ # Deploy:
8
+ # modal deploy modal_qwen36_llamacpp.py
9
+ #
10
+ # Set secrets (one-time):
11
+ # modal secret create api-key API_KEY=$(openssl rand -hex 32)
12
+ # modal secret create hf-token HF_TOKEN=hf_your_token_here
13
+ #
14
+ # Invoke (curl):
15
+ # curl -X POST <URL>/v1/chat/completions \
16
+ # -H "Authorization: Bearer $API_KEY" \
17
+ # -H "Content-Type: application/json" \
18
+ # -d '{"model":"qwen3.6-35b-a3b","messages":[{"role":"user","content":"Hello"}]}'
19
+ #
20
+ # Opencode / OpenAI base URL:
21
+ # https://<workspace>--qwen36-llamacpp-qwen36-infer.modal.run/v1
22
+ #
23
+ # Local test (outside Modal):
24
+ # python modal_qwen36_llamacpp.py
25
+ # ─────────────────────────────────────────────────────────────────────────────
26
+
27
+ from _modal_llamacpp_base import ModelConfig, build_llama_cmd, make_asgi_app
28
+
29
+ cfg = ModelConfig(
30
+ model_repo="unsloth/Qwen3.6-35B-A3B-MTP-GGUF",
31
+ model_quant="UD-Q4_K_XL",
32
+ alias="qwen3.6-35b-a3b",
33
+ ctx_size=65536,
34
+ cache_type_k="q8_0",
35
+ cache_type_v="q4_0",
36
+ batch_size=1024,
37
+ ubatch_size=512,
38
+ threads=8,
39
+ threads_batch=8,
40
+ speculative=True,
41
+ mtp_draft_p_min=0.0,
42
+ multimodal=False,
43
+ )
44
+
45
+ app = make_asgi_app(cfg)
46
+
47
+ if __name__ == "__main__":
48
+ cmd = " ".join(build_llama_cmd(cfg))
49
+ print("─── Local test mode ─────────────────────────────────")
50
+ print("Command that would run on Modal (L4 GPU):\n")
51
+ print(f" {cmd}\n")
52
+ print("─── End ─────────────────────────────────────────────")
parse_feedback.py CHANGED
@@ -1,23 +1,121 @@
1
  import re
2
 
3
- PATTERN = re.compile(
4
- r"Disse:\s*(.+?)\s*Correction:\s*(.+?)\s*Explication:\s*(.+?)(?=\s*Disse:|\s*$)",
 
 
 
 
 
 
 
 
 
 
 
 
 
5
  re.DOTALL | re.IGNORECASE,
6
  )
7
 
8
- def parse_feedback(text: str) -> list[dict]:
9
- entries = []
10
- for m in PATTERN.finditer(text):
11
- entries.append({
12
- "disse": m.group(1).strip(),
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
13
  "correction": m.group(2).strip(),
14
- "explication": m.group(3).strip(),
15
  })
16
- return entries
 
 
 
 
 
17
 
18
 
19
  def render_feedback_table(entries: list[dict]) -> list[list[str]]:
20
- return [[e["disse"], e["correction"], e["explication"]] for e in entries]
 
21
 
22
 
23
  def strip_markdown(text: str) -> str:
@@ -27,6 +125,6 @@ def strip_markdown(text: str) -> str:
27
  text = re.sub(r"#{1,6}\s*", "", text)
28
  text = re.sub(r"^[-*]\s", "", text, flags=re.MULTILINE)
29
  text = re.sub(r"!?\[.*?\]\(.*?\)", "", text)
30
- text = text.replace("▌", "")
31
- text = text.replace("🤖", "")
32
  return text.strip()
 
1
  import re
2
 
3
+ SECTION_HEADERS = [
4
+ "Points forts",
5
+ "À corriger",
6
+ "Vocabulaire dentaire utile",
7
+ "Priorité pour la prochaine séance",
8
+ "Bilan",
9
+ ]
10
+
11
+ NEXT_HEADER_PATTERN = re.compile(
12
+ rf"(?:^|\n)\s*(?:{'|'.join(re.escape(h) for h in SECTION_HEADERS)})\s*:\s*",
13
+ re.IGNORECASE | re.MULTILINE,
14
+ )
15
+
16
+ CITATION_RE = re.compile(
17
+ r'Citation:\s*["\u00ab](.+?)["\u00bb]\s*Correction:\s*(.+?)\s*Pourquoi:\s*(.+?)(?=\s*(?:Citation:|[A-Z\u00c0-\u017f][^\n:]*:)|\s*$)',
18
  re.DOTALL | re.IGNORECASE,
19
  )
20
 
21
+
22
+ def _extract_section(text: str, header: str) -> str:
23
+ """Return content between *header*: and the next section header or EOF."""
24
+ m = re.search(
25
+ rf"(?:^|\n)\s*{re.escape(header)}\s*:\s*",
26
+ text, re.IGNORECASE | re.MULTILINE,
27
+ )
28
+ if not m:
29
+ return ""
30
+ start = m.end()
31
+ # Find next section header
32
+ nm = NEXT_HEADER_PATTERN.search(text, start)
33
+ end = nm.start() if nm else len(text)
34
+ return text[start:end].strip()
35
+
36
+
37
+ def _parse_bullets(text: str) -> list[str]:
38
+ """Extract lines prefixed with - or • from a section block."""
39
+ if not text:
40
+ return []
41
+ items = []
42
+ for line in text.split("\n"):
43
+ line = line.strip()
44
+ if re.match(r"^[-•]\s", line):
45
+ items.append(re.sub(r"^[-•]\s*", "", line).strip())
46
+ return items
47
+
48
+
49
+ def _parse_bilan(text: str) -> dict:
50
+ """Extract Bilan scores as {key: int}."""
51
+ section = _extract_section(text, "Bilan")
52
+ if not section:
53
+ return {}
54
+ scores = {}
55
+ for line in section.split("\n"):
56
+ m = re.match(
57
+ r"[-•]?\s*(Grammaire|Fluidité|Vocabulaire dentaire|Communication clinique)\s*:\s*(\d+)\s*/\s*\d+",
58
+ line.strip(), re.IGNORECASE,
59
+ )
60
+ if m:
61
+ key = m.group(1).lower().replace(" ", "_")
62
+ scores[key] = int(m.group(2))
63
+ return scores
64
+
65
+
66
+ def parse_feedback(text: str) -> dict:
67
+ """Parse full feedback into a structured dict.
68
+
69
+ Returns:
70
+ intro -- spoken part (before delimiter)
71
+ points_forts -- list[str]
72
+ erreurs -- list[dict] with keys: citation, correction, pourquoi
73
+ vocabulaire -- list[str]
74
+ priorite -- list[str]
75
+ bilan -- dict {grammaire: int, fluidite: int, ...}
76
+ """
77
+ result = {
78
+ "intro": "",
79
+ "points_forts": [],
80
+ "erreurs": [],
81
+ "vocabulaire": [],
82
+ "priorite": [],
83
+ "bilan": {},
84
+ }
85
+
86
+ # Split on delimiter
87
+ if "---" in text:
88
+ parts = text.split("---", 1)
89
+ result["intro"] = parts[0].strip()
90
+ rest = parts[1].strip()
91
+ elif "Points forts:" in text:
92
+ parts = text.split("Points forts:", 1)
93
+ result["intro"] = parts[0].strip()
94
+ rest = "Points forts: " + parts[1].strip()
95
+ else:
96
+ result["intro"] = text.strip()
97
+ return result
98
+
99
+ result["points_forts"] = _parse_bullets(_extract_section(rest, "Points forts"))
100
+
101
+ corriger = _extract_section(rest, "À corriger")
102
+ for m in CITATION_RE.finditer(corriger):
103
+ result["erreurs"].append({
104
+ "citation": m.group(1).strip(),
105
  "correction": m.group(2).strip(),
106
+ "pourquoi": m.group(3).strip(),
107
  })
108
+
109
+ result["vocabulaire"] = _parse_bullets(_extract_section(rest, "Vocabulaire dentaire utile"))
110
+ result["priorite"] = _parse_bullets(_extract_section(rest, "Priorité pour la prochaine séance"))
111
+ result["bilan"] = _parse_bilan(rest)
112
+
113
+ return result
114
 
115
 
116
  def render_feedback_table(entries: list[dict]) -> list[list[str]]:
117
+ """Convert error entries to table rows for the frontend."""
118
+ return [[e["citation"], e["correction"], e["pourquoi"]] for e in entries]
119
 
120
 
121
  def strip_markdown(text: str) -> str:
 
125
  text = re.sub(r"#{1,6}\s*", "", text)
126
  text = re.sub(r"^[-*]\s", "", text, flags=re.MULTILINE)
127
  text = re.sub(r"!?\[.*?\]\(.*?\)", "", text)
128
+ text = text.replace("\u258c", "")
129
+ text = text.replace("\U0001f916", "")
130
  return text.strip()
prompts.py CHANGED
@@ -1,26 +1,81 @@
1
- SYSTEM_PROMPT = """You are a virtual dental patient. The user is your dental hygienist, currently training to improve their professional French.
2
-
3
- Phase 1: Voice Roleplay
4
- Act as a patient undergoing a 60-minute hygiene session. The flow is: Anamnesis -> Scaling (Détartrage) -> Polishing (Polissage) -> Fluoridation.
5
- Let the user lead the conversation. Do not anticipate their lines.
6
- Keep your answers highly concise (1-2 short sentences maximum).
7
- Speak entirely in French.
8
- Vary your language skills: use Swiss-French regionalisms (septante, huitante, lolette, chanterelle).
9
- Incorporate natural small talk at the beginning or during pauses (weather, public transport, holidays).
10
- Occasionally ask administrative questions about cost or insurance coverage.
11
- During the scaling phase, occasionally simulate sudden pain or sensitivity ("Aïe !", "C'est sensible ici") to test their clinical response.
12
- NEVER break character, offer advice, or correct their French during the roleplay.
13
- FORMATTING: Do not use emojis, asterisks, bolding, bullet points, or any markdown. Use only plain text and standard punctuation to ensure smooth text-to-speech playback.
14
- Continue the roleplay until the user explicitly says: "Fin de la séance" or "Session terminée".
15
-
16
- Phase 2: Feedback & Recap (Post-Roleplay)
17
- Trigger this phase ONLY when the user uses one of the termination phrases above.
18
- Drop the patient persona and act as an expert French language tutor.
19
- First, respond with exactly two spoken sentences in French: encourage them, then highlight one main area of improvement regarding medical vocabulary or grammar progression.
20
- Then IMMEDIATELY provide a structured text response with the detailed written breakdown of their mistakes. Do NOT wait for any further prompt. List the errors clearly in plain text (not a markdown table) using this format for each mistake:
21
- Disse: [What they said in FR]
22
- Correction: [Correction in FR]
23
- Explication: [Explanation in FR]
24
- """
25
-
26
- PHASE_SWITCH_REMINDER = """Reminder: you are now an expert French language tutor providing feedback. Do NOT continue the patient roleplay. Respond with exactly two spoken sentences in French (encouragement + one area of improvement), then immediately provide the structured Disse/Correction/Explication breakdown. Use plain text only — no markdown, no emojis."""
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ SYSTEM_PROMPT = """You are a virtual dental patient in a Swiss dental hygiene appointment. The user is the hygienist and is practicing professional French.
2
+
3
+ ROLEPLAY
4
+ Stay in character until the user ends the session.
5
+ Speak only French.
6
+ Use plain text only.
7
+ Let the user lead. Do not anticipate their next line.
8
+ Answer naturally and briefly:
9
+ - anamnesis: 1 to 2 short sentences
10
+ - treatment: very short replies or fragments
11
+
12
+ PATIENT
13
+ At the start of each session, create one hidden patient profile and keep it consistent:
14
+ - age
15
+ - reason for visit
16
+ - relevant medical history
17
+ - dental habits
18
+ - one mild concern
19
+ The patient should sound realistic, cooperative, and sometimes slightly vague.
20
+
21
+ REALISM
22
+ Use light Swiss Romand French naturally.
23
+ Use septante and nonante when relevant.
24
+ Do not force regional words.
25
+ Allow brief small talk at the beginning or during pauses.
26
+ Ask at most one natural question about cost or insurance during the whole session.
27
+ During scaling, occasionally react to sensitivity briefly, for example:
28
+ Aïe.
29
+ C'est sensible ici.
30
+ Un peu plus doucement, s'il vous plaît.
31
+
32
+ DO NOT
33
+ Do not explain, teach, translate, or correct during the roleplay.
34
+ Do not use markdown, bullets, or emojis in your replies.
35
+ Do not become overly talkative or clinically expert as a patient.
36
+
37
+ END OF SESSION
38
+ Only stop roleplay if the user says one of these or an obvious equivalent:
39
+ - Fin de la séance
40
+ - Session terminée
41
+ - On a terminé
42
+ - C'est fini
43
+
44
+ FEEDBACK MODE
45
+ After the session ends, stop being the patient and become a French coach.
46
+
47
+ First output exactly 2 short sentences in French:
48
+ 1. one encouraging sentence
49
+ 2. one sentence with the main priority for improvement
50
+
51
+ Then output --- on its own line.
52
+
53
+ Then give a short written recap in plain text with exactly these sections:
54
+
55
+ Points forts:
56
+ - ...
57
+
58
+ À corriger:
59
+ Citation: "..."
60
+ Correction: ...
61
+ Pourquoi: ...
62
+
63
+ Vocabulaire dentaire utile:
64
+ - ...
65
+
66
+ Priorité pour la prochaine séance:
67
+ - ...
68
+
69
+ Bilan:
70
+ - Grammaire: x/5
71
+ - Fluidité: x/5
72
+ - Vocabulaire dentaire: x/5
73
+ - Communication clinique: x/5
74
+
75
+ Only correct genuine errors or awkward wording actually produced by the user.
76
+ If a phrase is unclear because of speech recognition, say:
77
+ Transcription incertaine.
78
+ Do not invent mistakes.
79
+ Prioritize important or repeated errors over minor ones."""
80
+
81
+ PHASE_SWITCH_REMINDER = """Reminder: you are now a French coach. Do NOT continue the patient roleplay. Respond with exactly 2 spoken sentences in French (encouragement + priority), then output --- on its own line, then provide the written recap with Points forts, Citation/Correction/Pourquoi, Vocabulaire dentaire utile, Priorité, and Bilan scores. Use plain text only."""
server_app.py CHANGED
@@ -29,7 +29,11 @@ def api_process_turn(audio: FileData, state: dict) -> dict:
29
  "audio_url": None,
30
  "state": state,
31
  "feedback_intro": "",
 
32
  "feedback_table": [],
 
 
 
33
  "feedback_open": False,
34
  "status": f"Erreur: {e}",
35
  }
@@ -43,7 +47,11 @@ def api_process_turn(audio: FileData, state: dict) -> dict:
43
  "audio_url": audio_url,
44
  "state": result["state"],
45
  "feedback_intro": result["feedback_intro"],
 
46
  "feedback_table": result["feedback_table"],
 
 
 
47
  "feedback_open": result["feedback_open"],
48
  "status": result["status"],
49
  }
@@ -63,7 +71,11 @@ def api_end_session(state: dict) -> dict:
63
  "audio_url": audio_url,
64
  "state": result["state"],
65
  "feedback_intro": result["feedback_intro"],
 
66
  "feedback_table": result["feedback_table"],
 
 
 
67
  "feedback_open": result["feedback_open"],
68
  "status": result["status"],
69
  }
 
29
  "audio_url": None,
30
  "state": state,
31
  "feedback_intro": "",
32
+ "feedback_points_forts": [],
33
  "feedback_table": [],
34
+ "feedback_vocabulaire": [],
35
+ "feedback_priorite": [],
36
+ "feedback_bilan": {},
37
  "feedback_open": False,
38
  "status": f"Erreur: {e}",
39
  }
 
47
  "audio_url": audio_url,
48
  "state": result["state"],
49
  "feedback_intro": result["feedback_intro"],
50
+ "feedback_points_forts": result["feedback_points_forts"],
51
  "feedback_table": result["feedback_table"],
52
+ "feedback_vocabulaire": result["feedback_vocabulaire"],
53
+ "feedback_priorite": result["feedback_priorite"],
54
+ "feedback_bilan": result["feedback_bilan"],
55
  "feedback_open": result["feedback_open"],
56
  "status": result["status"],
57
  }
 
71
  "audio_url": audio_url,
72
  "state": result["state"],
73
  "feedback_intro": result["feedback_intro"],
74
+ "feedback_points_forts": result["feedback_points_forts"],
75
  "feedback_table": result["feedback_table"],
76
+ "feedback_vocabulaire": result["feedback_vocabulaire"],
77
+ "feedback_priorite": result["feedback_priorite"],
78
+ "feedback_bilan": result["feedback_bilan"],
79
  "feedback_open": result["feedback_open"],
80
  "status": result["status"],
81
  }
src/App.tsx DELETED
@@ -1,258 +0,0 @@
1
- import { useState, useEffect, useRef, useCallback, ReactNode } from 'react';
2
- import { motion, AnimatePresence } from 'motion/react';
3
- import { Mic, MicOff, Stethoscope, Languages, ClipboardList, Info } from 'lucide-react';
4
- import { GeminiLiveService } from './services/geminiLiveService';
5
- import { LiveServerMessage } from '@google/genai';
6
-
7
- export default function App() {
8
- const [isConnected, setIsConnected] = useState(false);
9
- const [isRecording, setIsRecording] = useState(false);
10
- const [isSpeaking, setIsSpeaking] = useState(false);
11
- const [feedbackText, setFeedbackText] = useState<string | null>(null);
12
- const [error, setError] = useState<string | null>(null);
13
-
14
- const liveServiceRef = useRef<GeminiLiveService | null>(null);
15
- const audioContextRef = useRef<AudioContext | null>(null);
16
- const audioQueueRef = useRef<Int16Array[]>([]);
17
- const isPlayingRef = useRef(false);
18
-
19
- const playNextInQueue = useCallback(async () => {
20
- if (isPlayingRef.current || audioQueueRef.current.length === 0 || !audioContextRef.current) return;
21
-
22
- isPlayingRef.current = true;
23
- setIsSpeaking(true);
24
-
25
- const pcmData = audioQueueRef.current.shift()!;
26
- const buffer = audioContextRef.current.createBuffer(1, pcmData.length, 24000);
27
- const channelData = buffer.getChannelData(0);
28
-
29
- for (let i = 0; i < pcmData.length; i++) {
30
- channelData[i] = pcmData[i] / 0x7FFF;
31
- }
32
-
33
- const source = audioContextRef.current.createBufferSource();
34
- source.buffer = buffer;
35
- source.connect(audioContextRef.current.destination);
36
-
37
- source.onended = () => {
38
- isPlayingRef.current = false;
39
- if (audioQueueRef.current.length === 0) {
40
- setIsSpeaking(false);
41
- }
42
- playNextInQueue();
43
- };
44
-
45
- source.start();
46
- }, []);
47
-
48
- const handleMessage = useCallback((message: LiveServerMessage) => {
49
- // Handle audio output
50
- const base64Audio = message.serverContent?.modelTurn?.parts?.[0]?.inlineData?.data;
51
- if (base64Audio) {
52
- const binaryString = atob(base64Audio);
53
- const len = binaryString.length;
54
- const bytes = new Uint8Array(len);
55
- for (let i = 0; i < len; i++) {
56
- bytes[i] = binaryString.charCodeAt(i);
57
- }
58
- const pcmData = new Int16Array(bytes.buffer);
59
- audioQueueRef.current.push(pcmData);
60
-
61
- if (!audioContextRef.current) {
62
- audioContextRef.current = new AudioContext({ sampleRate: 24000 });
63
- }
64
- playNextInQueue();
65
- }
66
-
67
- // Handle text output (for "Gere a tabela")
68
- const text = message.serverContent?.modelTurn?.parts?.[0]?.text;
69
- if (text && (text.includes('Disse:') || text.includes('Correção:'))) {
70
- setFeedbackText(text);
71
- }
72
-
73
- // Handle interruption
74
- if (message.serverContent?.interrupted) {
75
- audioQueueRef.current = [];
76
- isPlayingRef.current = false;
77
- setIsSpeaking(false);
78
- }
79
- }, [playNextInQueue]);
80
-
81
- const toggleConnection = async () => {
82
- if (isConnected) {
83
- liveServiceRef.current?.disconnect();
84
- setIsConnected(false);
85
- setIsRecording(false);
86
- setIsSpeaking(false);
87
- } else {
88
- try {
89
- setError(null);
90
- if (!liveServiceRef.current) {
91
- liveServiceRef.current = new GeminiLiveService();
92
- }
93
- await liveServiceRef.current.connect({
94
- onOpen: () => {
95
- setIsConnected(true);
96
- setIsRecording(true);
97
- },
98
- onClose: () => {
99
- setIsConnected(false);
100
- setIsRecording(false);
101
- },
102
- onMessage: handleMessage,
103
- onError: (err) => {
104
- setError("Connection error. Please check your API key.");
105
- setIsConnected(false);
106
- },
107
- onInterrupted: () => {
108
- // Handled in handleMessage
109
- }
110
- });
111
- } catch (err) {
112
- setError("Failed to connect to Gemini Live.");
113
- }
114
- }
115
- };
116
-
117
- useEffect(() => {
118
- return () => {
119
- liveServiceRef.current?.disconnect();
120
- audioContextRef.current?.close();
121
- };
122
- }, []);
123
-
124
- return (
125
- <div className="min-h-screen flex flex-col items-center justify-center p-6 relative">
126
- <div className="atmosphere" />
127
-
128
- <header className="absolute top-8 left-8 flex items-center gap-3">
129
- <div className="w-10 h-10 rounded-full bg-orange-500 flex items-center justify-center shadow-lg">
130
- <Stethoscope className="text-white w-6 h-6" />
131
- </div>
132
- <div>
133
- <h1 className="text-xl font-serif font-light tracking-wide">Français en médecine dentaire</h1>
134
- <p className="text-[10px] uppercase tracking-[0.2em] text-orange-400 font-medium">Bienne, Suisse</p>
135
- </div>
136
- </header>
137
-
138
- <main className="w-full max-w-2xl flex flex-col items-center gap-12">
139
- <AnimatePresence mode="wait">
140
- {!isConnected ? (
141
- <motion.div
142
- key="start"
143
- initial={{ opacity: 0, y: 20 }}
144
- animate={{ opacity: 1, y: 0 }}
145
- exit={{ opacity: 0, scale: 0.9 }}
146
- className="text-center space-y-8"
147
- >
148
- <div className="space-y-4">
149
- <h2 className="text-5xl font-serif font-light leading-tight">
150
- Pratiquez votre <br />
151
- <span className="italic text-orange-500">Français Dentaire</span>
152
- </h2>
153
- <p className="text-zinc-400 max-w-md mx-auto text-sm leading-relaxed">
154
- Entraînez-vous avec un patient virtuel suisse.
155
- Améliorez votre vocabulaire professionnel en temps réel.
156
- </p>
157
- </div>
158
-
159
- <button
160
- onClick={toggleConnection}
161
- className="group relative px-12 py-4 rounded-full bg-white text-black font-medium transition-all hover:scale-105 active:scale-95 overflow-hidden"
162
- >
163
- <div className="absolute inset-0 bg-orange-500 translate-y-full group-hover:translate-y-0 transition-transform duration-300" />
164
- <span className="relative z-10 flex items-center gap-2 group-hover:text-white transition-colors">
165
- Commencer la séance
166
- </span>
167
- </button>
168
-
169
- {error && <p className="text-red-400 text-xs mt-4">{error}</p>}
170
- </motion.div>
171
- ) : (
172
- <motion.div
173
- key="active"
174
- initial={{ opacity: 0, scale: 0.9 }}
175
- animate={{ opacity: 1, scale: 1 }}
176
- className="w-full flex flex-col items-center gap-12"
177
- >
178
- <div className="relative">
179
- <div className={`w-48 h-48 rounded-full flex items-center justify-center glass-card transition-all duration-500 ${isRecording ? 'recording-glow' : ''}`}>
180
- {isSpeaking ? (
181
- <div className="speaking-wave">
182
- {[1, 2, 3, 4, 5].map((i) => (
183
- <motion.div
184
- key={i}
185
- className="wave-bar"
186
- animate={{ height: [10, 40, 10] }}
187
- transition={{ repeat: Infinity, duration: 0.8, delay: i * 0.1 }}
188
- />
189
- ))}
190
- </div>
191
- ) : (
192
- <Mic className={`w-12 h-12 ${isRecording ? 'text-orange-500' : 'text-zinc-500'}`} />
193
- )}
194
- </div>
195
-
196
- <div className="absolute -bottom-4 left-1/2 -translate-x-1/2 px-4 py-1 rounded-full bg-black/50 backdrop-blur-md border border-white/10 text-[10px] uppercase tracking-widest">
197
- {isSpeaking ? 'Le patient parle...' : 'À vous de parler'}
198
- </div>
199
- </div>
200
-
201
- <div className="grid grid-cols-3 gap-4 w-full">
202
- <StatusCard icon={<Languages size={16} />} label="Langue" value="Français (CH)" />
203
- <StatusCard icon={<ClipboardList size={16} />} label="Phase" value="Roleplay" />
204
- <StatusCard icon={<Info size={16} />} label="Lieu" value="Bienne, BE" />
205
- </div>
206
-
207
- <button
208
- onClick={toggleConnection}
209
- className="flex items-center gap-2 text-zinc-500 hover:text-red-400 transition-colors text-sm uppercase tracking-widest"
210
- >
211
- <MicOff size={16} />
212
- Terminer la connexion
213
- </button>
214
- </motion.div>
215
- )}
216
- </AnimatePresence>
217
-
218
- <AnimatePresence>
219
- {feedbackText && (
220
- <motion.div
221
- initial={{ opacity: 0, y: 40 }}
222
- animate={{ opacity: 1, y: 0 }}
223
- className="w-full glass-card rounded-3xl p-8 space-y-6 max-h-[400px] overflow-y-auto custom-scrollbar"
224
- >
225
- <div className="flex items-center justify-between border-b border-white/10 pb-4">
226
- <h3 className="font-serif text-xl italic">Récapitulatif de la séance</h3>
227
- <button
228
- onClick={() => setFeedbackText(null)}
229
- className="text-zinc-500 hover:text-white"
230
- >
231
- Fermer
232
- </button>
233
- </div>
234
- <div className="whitespace-pre-wrap font-mono text-sm text-zinc-300 leading-relaxed">
235
- {feedbackText}
236
- </div>
237
- </motion.div>
238
- )}
239
- </AnimatePresence>
240
- </main>
241
-
242
- <footer className="absolute bottom-8 text-[10px] text-zinc-600 uppercase tracking-[0.3em]">
243
- Interactive Voice Training System &copy; 2026
244
- </footer>
245
- </div>
246
- );
247
- }
248
-
249
- function StatusCard({ icon, label, value }: { icon: ReactNode, label: string, value: string }) {
250
- return (
251
- <div className="glass-card rounded-2xl p-4 flex flex-col gap-1 items-center text-center">
252
- <div className="text-orange-500 mb-1">{icon}</div>
253
- <span className="text-[9px] uppercase tracking-wider text-zinc-500">{label}</span>
254
- <span className="text-xs font-medium">{value}</span>
255
- </div>
256
- );
257
- }
258
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
src/index.css DELETED
@@ -1,76 +0,0 @@
1
- @import "tailwindcss";
2
-
3
- @theme {
4
- --font-sans: "Inter", ui-sans-serif, system-ui, sans-serif;
5
- --font-serif: "Cormorant Garamond", serif;
6
- }
7
-
8
- :root {
9
- --color-bg: #0a0502;
10
- --color-accent: #ff4e00;
11
- --glass-surface: rgba(255, 80, 20, 0.1);
12
- --glass-border: rgba(255, 200, 150, 0.1);
13
- }
14
-
15
- body {
16
- background-color: var(--color-bg);
17
- color: #fff;
18
- overflow: hidden;
19
- }
20
-
21
- .atmosphere {
22
- position: fixed;
23
- top: 0;
24
- left: 0;
25
- width: 100%;
26
- height: 100%;
27
- background:
28
- radial-gradient(circle at 50% 30%, #3a1510 0%, transparent 60%),
29
- radial-gradient(circle at 10% 80%, var(--color-accent) 0%, transparent 50%);
30
- filter: blur(80px);
31
- opacity: 0.6;
32
- z-index: -1;
33
- animation: pulse 15s ease-in-out infinite alternate;
34
- }
35
-
36
- @keyframes pulse {
37
- 0% { opacity: 0.4; transform: scale(1); }
38
- 100% { opacity: 0.7; transform: scale(1.1); }
39
- }
40
-
41
- .glass-card {
42
- background: var(--glass-surface);
43
- backdrop-filter: blur(40px);
44
- border: 1px solid var(--glass-border);
45
- box-shadow: 0 20px 50px rgba(0, 0, 0, 0.3);
46
- }
47
-
48
- .recording-glow {
49
- box-shadow: 0 0 20px rgba(255, 78, 0, 0.5);
50
- animation: glow 1.5s ease-in-out infinite alternate;
51
- }
52
-
53
- @keyframes glow {
54
- 0% { box-shadow: 0 0 10px rgba(255, 78, 0, 0.3); }
55
- 100% { box-shadow: 0 0 30px rgba(255, 78, 0, 0.8); }
56
- }
57
-
58
- .speaking-wave {
59
- display: flex;
60
- align-items: center;
61
- gap: 4px;
62
- }
63
-
64
- .wave-bar {
65
- width: 3px;
66
- height: 10px;
67
- background: var(--color-accent);
68
- border-radius: 2px;
69
- animation: wave 1s ease-in-out infinite;
70
- }
71
-
72
- @keyframes wave {
73
- 0%, 100% { height: 10px; }
74
- 50% { height: 30px; }
75
- }
76
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
src/main.tsx DELETED
@@ -1,10 +0,0 @@
1
- import {StrictMode} from 'react';
2
- import {createRoot} from 'react-dom/client';
3
- import App from './App.tsx';
4
- import './index.css';
5
-
6
- createRoot(document.getElementById('root')!).render(
7
- <StrictMode>
8
- <App />
9
- </StrictMode>,
10
- );
 
 
 
 
 
 
 
 
 
 
 
src/services/geminiLiveService.ts DELETED
@@ -1,160 +0,0 @@
1
- import { GoogleGenAI, LiveServerMessage, Modality } from "@google/genai";
2
-
3
- export enum AppPhase {
4
- ROLEPLAY = 'roleplay',
5
- FEEDBACK = 'feedback'
6
- }
7
-
8
- export interface LiveSessionCallbacks {
9
- onOpen?: () => void;
10
- onClose?: () => void;
11
- onMessage?: (message: LiveServerMessage) => void;
12
- onError?: (error: any) => void;
13
- onInterrupted?: () => void;
14
- }
15
-
16
- export class GeminiLiveService {
17
- private ai: GoogleGenAI;
18
- private session: any = null;
19
- private audioContext: AudioContext | null = null;
20
- private workletNode: AudioWorkletNode | null = null;
21
- private stream: MediaStream | null = null;
22
-
23
- constructor() {
24
- this.ai = new GoogleGenAI({ apiKey: process.env.GEMINI_API_KEY });
25
- }
26
-
27
- async connect(callbacks: LiveSessionCallbacks) {
28
- const systemInstruction = `
29
- You are a virtual dental patient in a Swiss clinic (Bienne, Canton de Berne). The user is your dental hygienist, a native Brazilian Portuguese speaker currently at a B1 French level, training to improve her professional French.
30
-
31
- Your behavior is divided into two strict phases: Phase 1 (Roleplay) and Phase 2 (Feedback).
32
-
33
- Phase 1: Voice Roleplay
34
- Act as a patient undergoing a 60-minute hygiene session. The flow is: Anamnesis -> Scaling (Détartrage) -> Polishing (Polissage) -> Fluoridation.
35
- Let the user lead the conversation. Do not anticipate her lines.
36
- Keep your answers highly concise (1-2 short sentences maximum).
37
- Speak entirely in French.
38
- Vary your language skills: use Swiss-French regionalisms (e.g., septante, huitante, lolette, chanterelle). As Bienne is bilingual, occasionally drop in a basic Swiss-German greeting or loanword (e.g., "Grüessech", "Merci vielmal").
39
- Incorporate natural small talk at the beginning or during pauses (e.g., weather, CFF/SBB trains, holidays).
40
- Occasionally ask administrative questions, such as the estimated cost in CHF or if the treatment is covered by basic health insurance (assurance de base / LAMal).
41
- During the scaling phase, occasionally simulate sudden pain or sensitivity (e.g., "Aïe !", "C'est sensible ici") to test her clinical response.
42
- NEVER break character, offer advice, or correct her French during the roleplay.
43
- FORMATTING: Do not use emojis, asterisks, bolding, bullet points, or any markdown. Use only plain text and standard punctuation to ensure smooth text-to-speech playback.
44
- Continue the roleplay until the user explicitly says: "Fin de la séance", "Fim da consulta", "Fim da sessão" or "Session terminée".
45
-
46
- Phase 2: Feedback & Recap (Post-Roleplay)
47
- Trigger this phase ONLY when the user uses one of the termination phrases above.
48
- Drop the patient persona and act as an expert French language tutor.
49
- Respond with exactly two spoken sentences in Brazilian Portuguese: First, encourage her. Second, highlight one main area of improvement regarding her medical vocabulary or B1-to-B2 grammar progression.
50
- Conclude by asking her to say "Gere a tabela" if she wants to see a detailed written breakdown of her mistakes.
51
-
52
- If the user says "Gere a tabela", provide a structured text response. List the errors clearly in plain text (not a markdown table) using this format for each mistake:
53
- Disse: [What she said in FR]
54
- Correção: [Correction in FR]
55
- Explicação: [Explanation in PT-BR]
56
- `.trim();
57
-
58
- this.session = await this.ai.live.connect({
59
- model: "gemini-2.5-flash-native-audio-preview-12-2025",
60
- config: {
61
- systemInstruction,
62
- responseModalities: [Modality.AUDIO],
63
- speechConfig: {
64
- voiceConfig: { prebuiltVoiceConfig: { voiceName: "Zephyr" } },
65
- },
66
- },
67
- callbacks: {
68
- onopen: () => {
69
- console.log("Live session opened");
70
- callbacks.onOpen?.();
71
- this.startAudioCapture();
72
- },
73
- onmessage: (message: LiveServerMessage) => {
74
- if (message.serverContent?.interrupted) {
75
- callbacks.onInterrupted?.();
76
- }
77
- callbacks.onMessage?.(message);
78
- },
79
- onclose: () => {
80
- console.log("Live session closed");
81
- callbacks.onClose?.();
82
- this.stopAudioCapture();
83
- },
84
- onerror: (error) => {
85
- console.error("Live session error:", error);
86
- callbacks.onError?.(error);
87
- },
88
- },
89
- });
90
- }
91
-
92
- private async startAudioCapture() {
93
- try {
94
- this.audioContext = new AudioContext({ sampleRate: 16000 });
95
- this.stream = await navigator.mediaDevices.getUserMedia({ audio: true });
96
- const source = this.audioContext.createMediaStreamSource(this.stream);
97
-
98
- // We need an AudioWorklet to handle raw PCM data
99
- await this.audioContext.audioWorklet.addModule(this.getWorkletUrl());
100
- this.workletNode = new AudioWorkletNode(this.audioContext, 'input-processor');
101
-
102
- this.workletNode.port.onmessage = (event) => {
103
- if (this.session && event.data) {
104
- const base64Data = this.arrayBufferToBase64(event.data);
105
- this.session.sendRealtimeInput({
106
- audio: { data: base64Data, mimeType: 'audio/pcm;rate=16000' }
107
- });
108
- }
109
- };
110
-
111
- source.connect(this.workletNode);
112
- } catch (err) {
113
- console.error("Error starting audio capture:", err);
114
- }
115
- }
116
-
117
- private stopAudioCapture() {
118
- this.stream?.getTracks().forEach(track => track.stop());
119
- this.workletNode?.disconnect();
120
- this.audioContext?.close();
121
- }
122
-
123
- private arrayBufferToBase64(buffer: ArrayBuffer): string {
124
- let binary = '';
125
- const bytes = new Uint8Array(buffer);
126
- const len = bytes.byteLength;
127
- for (let i = 0; i < len; i++) {
128
- binary += String.fromCharCode(bytes[i]);
129
- }
130
- return btoa(binary);
131
- }
132
-
133
- private getWorkletUrl(): string {
134
- const code = `
135
- class InputProcessor extends AudioWorkletProcessor {
136
- process(inputs, outputs, parameters) {
137
- const input = inputs[0];
138
- if (input.length > 0) {
139
- const channelData = input[0];
140
- // Convert Float32 to Int16
141
- const pcmData = new Int16Array(channelData.length);
142
- for (let i = 0; i < channelData.length; i++) {
143
- pcmData[i] = Math.max(-1, Math.min(1, channelData[i])) * 0x7FFF;
144
- }
145
- this.port.postMessage(pcmData.buffer);
146
- }
147
- return true;
148
- }
149
- }
150
- registerProcessor('input-processor', InputProcessor);
151
- `;
152
- const blob = new Blob([code], { type: 'application/javascript' });
153
- return URL.createObjectURL(blob);
154
- }
155
-
156
- disconnect() {
157
- this.session?.close();
158
- this.stopAudioCapture();
159
- }
160
- }