Spaces:
Running on Zero
Running on Zero
Download app/core/subtitle_compactor.py from hoangtaiii/DRIPPY4: direct link, hf CLI and curl.
- Browser
- Download file 2.07 kB
-
https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/subtitle_compactor.py
- Command line
-
hf download hf://spaces/hoangtaiii/DRIPPY4/app/core/subtitle_compactor.py
-
curl -L -o subtitle_compactor.py https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/subtitle_compactor.py
2.07 kB
| """ | |
| app/core/subtitle_compactor.py | |
| ββββββββββββββββββββββββββββββ | |
| Compacts translated subtitle text to fit TTS timing and display width. | |
| Ensures full sentence integrity (NO hard truncation of words/clauses). | |
| """ | |
| import re | |
| from typing import List, Dict, Optional | |
| def estimate_vi_syllables(text: str) -> int: | |
| """Estimate Vietnamese syllable count (roughly = word count).""" | |
| return len(str(text or "").split()) | |
| def compact_text(text: str) -> str: | |
| """Cleans up redundant spaces and punctuation without dropping content words.""" | |
| t = str(text).strip() | |
| t = re.sub(r"\s+", " ", t) | |
| return t.strip(" ,") | |
| def compact_blocks_for_tts( | |
| blocks: List[Dict], | |
| translated_dict: Dict[str, str], | |
| max_timing_over_ratio: float = 1.4, | |
| chars_per_second: float = 4.5 | |
| ) -> Dict[str, str]: | |
| """ | |
| Ensures complete full sentences are preserved without dropping any words or ending syllables. | |
| """ | |
| result = {} | |
| for block in blocks: | |
| bid = str(block.get("id")) | |
| if bid in translated_dict: | |
| # Preserve 100% full translated text | |
| result[bid] = compact_text(translated_dict[bid]) | |
| elif int(bid) in translated_dict: | |
| result[bid] = compact_text(translated_dict[int(bid)]) | |
| return result | |
| def split_display_text(text: str, max_words: int = 14) -> str: | |
| """ | |
| Split long text into 2 display lines with \\N for TikTok subtitle rendering. | |
| """ | |
| words = text.split() | |
| if len(words) <= max_words: | |
| return text | |
| mid = len(words) // 2 | |
| best_split = mid | |
| for i in range(max(1, mid - 3), min(len(words) - 1, mid + 4)): | |
| word = words[i] | |
| if word.endswith((",", ".", "!", "?", ";", ":")): | |
| best_split = i + 1 | |
| break | |
| if word.lower() in ("vΓ ", "nhΖ°ng", "hoαΊ·c", "mΓ ", "thΓ¬", "nΓͺn", "vΓ¬", "Δα»"): | |
| best_split = i | |
| break | |
| line1 = " ".join(words[:best_split]) | |
| line2 = " ".join(words[best_split:]) | |
| return f"{line1}\\N{line2}" | |