""" app/core/subtitle_compactor.py ────────────────────────────── Compacts translated subtitle text to fit TTS timing and display width. Ensures full sentence integrity (NO hard truncation of words/clauses). """ import re from typing import List, Dict, Optional def estimate_vi_syllables(text: str) -> int: """Estimate Vietnamese syllable count (roughly = word count).""" return len(str(text or "").split()) def compact_text(text: str) -> str: """Cleans up redundant spaces and punctuation without dropping content words.""" t = str(text).strip() t = re.sub(r"\s+", " ", t) return t.strip(" ,") def compact_blocks_for_tts( blocks: List[Dict], translated_dict: Dict[str, str], max_timing_over_ratio: float = 1.4, chars_per_second: float = 4.5 ) -> Dict[str, str]: """ Ensures complete full sentences are preserved without dropping any words or ending syllables. """ result = {} for block in blocks: bid = str(block.get("id")) if bid in translated_dict: # Preserve 100% full translated text result[bid] = compact_text(translated_dict[bid]) elif int(bid) in translated_dict: result[bid] = compact_text(translated_dict[int(bid)]) return result def split_display_text(text: str, max_words: int = 14) -> str: """ Split long text into 2 display lines with \\N for TikTok subtitle rendering. """ words = text.split() if len(words) <= max_words: return text mid = len(words) // 2 best_split = mid for i in range(max(1, mid - 3), min(len(words) - 1, mid + 4)): word = words[i] if word.endswith((",", ".", "!", "?", ";", ":")): best_split = i + 1 break if word.lower() in ("và", "nhưng", "hoặc", "mà", "thì", "nên", "vì", "để"): best_split = i break line1 = " ".join(words[:best_split]) line2 = " ".join(words[best_split:]) return f"{line1}\\N{line2}"