DRIPPY4 / app /core /subtitle_compactor.py
hoangtaiii's picture
Upload 92 files
16c3ac7 verified
Raw History Blame Contribute Delete
2.07 kB
"""
app/core/subtitle_compactor.py
──────────────────────────────
Compacts translated subtitle text to fit TTS timing and display width.
Ensures full sentence integrity (NO hard truncation of words/clauses).
"""
import re
from typing import List, Dict, Optional
def estimate_vi_syllables(text: str) -> int:
"""Estimate Vietnamese syllable count (roughly = word count)."""
return len(str(text or "").split())
def compact_text(text: str) -> str:
"""Cleans up redundant spaces and punctuation without dropping content words."""
t = str(text).strip()
t = re.sub(r"\s+", " ", t)
return t.strip(" ,")
def compact_blocks_for_tts(
blocks: List[Dict],
translated_dict: Dict[str, str],
max_timing_over_ratio: float = 1.4,
chars_per_second: float = 4.5
) -> Dict[str, str]:
"""
Ensures complete full sentences are preserved without dropping any words or ending syllables.
"""
result = {}
for block in blocks:
bid = str(block.get("id"))
if bid in translated_dict:
# Preserve 100% full translated text
result[bid] = compact_text(translated_dict[bid])
elif int(bid) in translated_dict:
result[bid] = compact_text(translated_dict[int(bid)])
return result
def split_display_text(text: str, max_words: int = 14) -> str:
"""
Split long text into 2 display lines with \\N for TikTok subtitle rendering.
"""
words = text.split()
if len(words) <= max_words:
return text
mid = len(words) // 2
best_split = mid
for i in range(max(1, mid - 3), min(len(words) - 1, mid + 4)):
word = words[i]
if word.endswith((",", ".", "!", "?", ";", ":")):
best_split = i + 1
break
if word.lower() in ("vΓ ", "nhΖ°ng", "hoαΊ·c", "mΓ ", "thΓ¬", "nΓͺn", "vΓ¬", "để"):
best_split = i
break
line1 = " ".join(words[:best_split])
line2 = " ".join(words[best_split:])
return f"{line1}\\N{line2}"