import re SECTION_HEADERS = [ "Points forts", "À corriger", "Vocabulaire dentaire utile", "Priorité pour la prochaine séance", "Bilan", ] NEXT_HEADER_PATTERN = re.compile( rf"(?:^|\n)\s*(?:{'|'.join(re.escape(h) for h in SECTION_HEADERS)})\s*:\s*", re.IGNORECASE | re.MULTILINE, ) CITATION_RE = re.compile( r'Citation:\s*["\u00ab](.+?)["\u00bb]\s*Correction:\s*(.+?)\s*Pourquoi:\s*(.+?)(?=\s*(?:Citation:|[A-Z\u00c0-\u017f][^\n:]*:)|\s*$)', re.DOTALL | re.IGNORECASE, ) def _extract_section(text: str, header: str) -> str: """Return content between *header*: and the next section header or EOF.""" m = re.search( rf"(?:^|\n)\s*{re.escape(header)}\s*:\s*", text, re.IGNORECASE | re.MULTILINE, ) if not m: return "" start = m.end() # Find next section header nm = NEXT_HEADER_PATTERN.search(text, start) end = nm.start() if nm else len(text) return text[start:end].strip() def _parse_bullets(text: str) -> list[str]: """Extract lines prefixed with - or • from a section block.""" if not text: return [] items = [] for line in text.split("\n"): line = line.strip() if re.match(r"^[-•]\s", line): items.append(re.sub(r"^[-•]\s*", "", line).strip()) return items def _parse_bilan(text: str) -> dict: """Extract Bilan scores as {key: int}.""" section = _extract_section(text, "Bilan") if not section: return {} scores = {} for line in section.split("\n"): m = re.match( r"[-•]?\s*(Grammaire|Fluidité|Vocabulaire dentaire|Communication clinique)\s*:\s*(\d+)\s*/\s*\d+", line.strip(), re.IGNORECASE, ) if m: key = m.group(1).lower().replace(" ", "_") scores[key] = int(m.group(2)) return scores def parse_feedback(text: str) -> dict: """Parse full feedback into a structured dict. Returns: intro -- spoken part (before delimiter) points_forts -- list[str] erreurs -- list[dict] with keys: citation, correction, pourquoi vocabulaire -- list[str] priorite -- list[str] bilan -- dict {grammaire: int, fluidite: int, ...} """ result = { "intro": "", "points_forts": [], "erreurs": [], "vocabulaire": [], "priorite": [], "bilan": {}, } # Split on delimiter if "---" in text: parts = text.split("---", 1) result["intro"] = parts[0].strip() rest = parts[1].strip() elif "Points forts:" in text: parts = text.split("Points forts:", 1) result["intro"] = parts[0].strip() rest = "Points forts: " + parts[1].strip() else: result["intro"] = text.strip() return result result["points_forts"] = _parse_bullets(_extract_section(rest, "Points forts")) corriger = _extract_section(rest, "À corriger") for m in CITATION_RE.finditer(corriger): result["erreurs"].append({ "citation": m.group(1).strip(), "correction": m.group(2).strip(), "pourquoi": m.group(3).strip(), }) result["vocabulaire"] = _parse_bullets(_extract_section(rest, "Vocabulaire dentaire utile")) result["priorite"] = _parse_bullets(_extract_section(rest, "Priorité pour la prochaine séance")) result["bilan"] = _parse_bilan(rest) return result def render_feedback_table(entries: list[dict]) -> list[list[str]]: """Convert error entries to table rows for the frontend.""" return [[e["citation"], e["correction"], e["pourquoi"]] for e in entries] def strip_markdown(text: str) -> str: text = re.sub(r"\*\*(.+?)\*\*", r"\1", text) text = re.sub(r"\*(.+?)\*", r"\1", text) text = re.sub(r"`(.+?)`", r"\1", text) text = re.sub(r"#{1,6}\s*", "", text) text = re.sub(r"^[-*]\s", "", text, flags=re.MULTILINE) text = re.sub(r"!?\[.*?\]\(.*?\)", "", text) text = text.replace("\u258c", "") text = text.replace("\U0001f916", "") return text.strip()