Spaces:
Running on Zero
Running on Zero
Download app/core/translation_post_editor.py from hoangtaiii/DRIPPY4: direct link, hf CLI and curl.
- Browser
- Download file 1.85 kB
-
https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/translation_post_editor.py
- Command line
-
hf download hf://spaces/hoangtaiii/DRIPPY4/app/core/translation_post_editor.py
-
curl -L -o translation_post_editor.py https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/translation_post_editor.py
1.85 kB
| import re | |
| SOURCE_GUIDED_REPLACEMENTS = [ | |
| ("割韭菜", [(r"cắt\s+(củ\s+cải|rau\s+hẹ|hẹ)", "chặt chém người mua")]), | |
| ("克罗心", [(r"\b(Kruskin|Krosin|Kroskin|Clare\s+Croux|Clare\s+Crox|Chrome\s*Heart)\b", "Chrome Hearts")]), | |
| ("溢价", [(r"giá\s+chênh\s+lệch", "độ đội giá"), (r"mức\s+giá\s+chênh\s+lệch", "mức đội giá")]), | |
| ("机车老炮", [(r"người\s+(từng\s+là\s+)?(yêu|mê)\s+máy\s+móc", "tay chơi mô tô kỳ cựu")]), | |
| ("机车", [(r"\bmáy\s+móc\b", "mô tô"), (r"người\s+(yêu|mê)\s+mô\s+tô", "dân mê mô tô")]), | |
| ("拼皮改造", [(r"có\s+chỉ\s+thêu", "custom ghép da"), (r"chỉ\s+thêu", "ghép da custom")]), | |
| ("贵有贵的道理", [(r"lý\s+do\s+nó\s+đắt", "đắt có lý do của nó")]), | |
| ("工坊", [(r"chế\s+độ\s+hoạt\s+động\s+của\s+xưởng", "cách vận hành xưởng")]), | |
| ] | |
| GENERAL_REPLACEMENTS = [ | |
| (r"Chrome\s*Hearts", "Chrome Hearts"), | |
| (r"Richard\s*Stark", "Richard Stark"), | |
| (r"John\s*Bowman", "John Bowman"), | |
| (r"Leonard\s*Kamhout", "Leonard Kamhout"), | |
| (r"\bLevis\b", "Levi's"), | |
| (r"vòng\s+văn\s+hóa\s+phụ", "giới văn hóa underground"), | |
| (r"cũng\s+cũng", "cũng"), | |
| (r"\s+", " "), | |
| ] | |
| def post_edit_translation(source_text, translated_text): | |
| text = str(translated_text or "").strip() | |
| source = str(source_text or "") | |
| if not text: | |
| return text | |
| for pattern, replacement in GENERAL_REPLACEMENTS: | |
| text = re.sub(pattern, replacement, text, flags=re.IGNORECASE) | |
| for source_term, replacements in SOURCE_GUIDED_REPLACEMENTS: | |
| if source_term in source: | |
| for pattern, replacement in replacements: | |
| text = re.sub(pattern, replacement, text, flags=re.IGNORECASE) | |
| return re.sub(r"\s+", " ", text).strip(" ,") | |