HachimiMT-demo / src /chunker.py
ngocdang83's picture
Init Space: HachimiMT zh-vi demo (CT2, chuẩn hóa xưng hô)
e9015b1 verified
Raw
History Blame Contribute Delete
777 Bytes
"""Split Chinese source text into translation chunks."""
from __future__ import annotations
import re
SENTENCE_END = re.compile(r"(?<=[。!?!?;;…])")
PARAGRAPH_BREAK = re.compile(r"\n\s*\n+")
def split_chunks(text: str, mode: str = "sentence") -> list[str]:
"""Split *text* into non-empty chunks for independent translation."""
text = text.strip()
if not text:
return []
if mode == "paragraph":
parts = PARAGRAPH_BREAK.split(text)
else:
parts: list[str] = []
for paragraph in text.splitlines():
paragraph = paragraph.strip()
if not paragraph:
continue
parts.extend(SENTENCE_END.split(paragraph))
return [chunk.strip() for chunk in parts if chunk.strip()]