File size: 2,071 Bytes
c968751
 
 
 
16c3ac7
c968751
 
 
 
 
 
 
 
 
 
 
16c3ac7
 
c968751
16c3ac7
 
c968751
 
 
 
 
 
 
 
 
16c3ac7
c968751
16c3ac7
c968751
 
16c3ac7
 
 
 
 
c968751
 
 
16c3ac7
c968751
16c3ac7
c968751
 
 
 
 
 
 
 
 
 
 
 
16c3ac7
c968751
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
"""
app/core/subtitle_compactor.py
──────────────────────────────
Compacts translated subtitle text to fit TTS timing and display width.
Ensures full sentence integrity (NO hard truncation of words/clauses).
"""

import re
from typing import List, Dict, Optional


def estimate_vi_syllables(text: str) -> int:
    """Estimate Vietnamese syllable count (roughly = word count)."""
    return len(str(text or "").split())


def compact_text(text: str) -> str:
    """Cleans up redundant spaces and punctuation without dropping content words."""
    t = str(text).strip()
    t = re.sub(r"\s+", " ", t)
    return t.strip(" ,")


def compact_blocks_for_tts(
    blocks: List[Dict],
    translated_dict: Dict[str, str],
    max_timing_over_ratio: float = 1.4,
    chars_per_second: float = 4.5
) -> Dict[str, str]:
    """
    Ensures complete full sentences are preserved without dropping any words or ending syllables.
    """
    result = {}
    for block in blocks:
        bid = str(block.get("id"))
        if bid in translated_dict:
            # Preserve 100% full translated text
            result[bid] = compact_text(translated_dict[bid])
        elif int(bid) in translated_dict:
            result[bid] = compact_text(translated_dict[int(bid)])
    return result


def split_display_text(text: str, max_words: int = 14) -> str:
    """
    Split long text into 2 display lines with \\N for TikTok subtitle rendering.
    """
    words = text.split()
    if len(words) <= max_words:
        return text

    mid = len(words) // 2
    best_split = mid
    for i in range(max(1, mid - 3), min(len(words) - 1, mid + 4)):
        word = words[i]
        if word.endswith((",", ".", "!", "?", ";", ":")):
            best_split = i + 1
            break
        if word.lower() in ("vΓ ", "nhΖ°ng", "hoαΊ·c", "mΓ ", "thΓ¬", "nΓͺn", "vΓ¬", "để"):
            best_split = i
            break

    line1 = " ".join(words[:best_split])
    line2 = " ".join(words[best_split:])
    return f"{line1}\\N{line2}"