Spaces:
Running on Zero
Running on Zero
Download app/core/translation_validator.py from hoangtaiii/DRIPPY4: direct link, hf CLI and curl.
- Browser
- Download file 6.83 kB
-
https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/translation_validator.py
- Command line
-
hf download hf://spaces/hoangtaiii/DRIPPY4/app/core/translation_validator.py
-
curl -L -o translation_validator.py https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/translation_validator.py
6.83 kB
| """ | |
| app/core/translation_validator.py | |
| ───────────────────────────────── | |
| Validates translated chunk output from any LLM provider. | |
| Ported from claude_fake with cloud-native adaptations. | |
| Features: | |
| • Stricter consecutive-repeat detection (3+ identical lines in a row) | |
| • Chinese character leakage check (>15% Chinese after translation) | |
| • Richer meta-text / LLM-commentary detection | |
| • Numeric-only block whitelisting (don't flag "1", "2" as duplicates) | |
| • Fast-path: if input is all English, skip Chinese-leakage check | |
| • Normalised-text dedup: ignores punctuation differences when counting loops | |
| """ | |
| import re | |
| import unicodedata | |
| def _chinese_ratio(text: str) -> float: | |
| if not text: | |
| return 0.0 | |
| zh_chars = sum(1 for c in text if '\u4e00' <= c <= '\u9fff') | |
| total = len(text.replace(' ', '')) | |
| return zh_chars / total if total else 0.0 | |
| def _normalise(text: str) -> str: | |
| """Strip punctuation/whitespace for dedup comparison.""" | |
| return re.sub(r'[\s.,!?;:"\'\-\u2013\u2014]+', '', text).lower() | |
| def _is_numeric_only(text: str) -> bool: | |
| return bool(re.fullmatch(r'[\d\s.,%-]+', text.strip())) | |
| class TranslationValidator: | |
| # LLM meta-commentary phrases (EN + VI) | |
| BANNED_PHRASES = [ | |
| # English | |
| "wait no", "corrected version", "looking at the original", "note:", | |
| "here is", "instead of", "respect user instruction", "sure, here is", | |
| "translation:", "clean version", "typos fixed", "fixed version", | |
| "original text", "below is", "i have cleaned", "timeline rows", | |
| "the corrected subtitle", "as requested", "here's the", | |
| "here are the", "i will translate", "let me translate", | |
| "the following", "translating the", "sure!", "okay!", "of course!", | |
| "certainly!", "absolutely!", "as an ai", "as a language model", | |
| # Vietnamese | |
| "cho da", "can dam bao", "dinh dang thoi gian", "lam sach", | |
| "sua loi", "ket qua:", "ban dich:", "chinh xac hon", "sau khi kiem tra", | |
| "duoi day la", "dong thoai", "ngu canh", "cach dich", | |
| "chu thich", "giai thich", "don dep", "day la ban dich", | |
| "toi se dich", "de dich chinh xac", | |
| ] | |
| def __init__(self, | |
| max_repeat_ratio: float = 0.20, | |
| max_consecutive_repeats: int = 2, | |
| max_chinese_ratio: float = 0.15, | |
| max_length_ratio: float = 3.5): | |
| self.max_repeat_ratio = max_repeat_ratio | |
| self.max_consecutive_repeats = max_consecutive_repeats | |
| self.max_chinese_ratio = max_chinese_ratio | |
| self.max_length_ratio = max_length_ratio | |
| def validate(self, input_chunk, translated_dict): | |
| """ | |
| Validate translated output dict. | |
| Returns (is_valid: bool, error_message: str). | |
| """ | |
| if not isinstance(translated_dict, dict): | |
| return False, "Output is not a dictionary" | |
| norm = {str(k): str(v).strip() for k, v in translated_dict.items()} | |
| # R1: count match | |
| input_keys = [str(b["id"]) for b in input_chunk] | |
| missing = [k for k in input_keys if k not in norm] | |
| if missing: | |
| return False, f"Missing keys in output: {missing}" | |
| values = [norm[k] for k in input_keys] | |
| # R2: loop by ratio (>20% identical sentence) | |
| total = len(input_keys) | |
| if total > 3: | |
| counts = {} | |
| for v in values: | |
| nv = _normalise(v) | |
| if nv and not _is_numeric_only(v) and len(v) > 4: | |
| counts[nv] = counts.get(nv, 0) + 1 | |
| for nv, cnt in counts.items(): | |
| ratio = cnt / total | |
| if cnt > 1 and ratio > self.max_repeat_ratio: | |
| sample = next(v for v in values if _normalise(v) == nv) | |
| return False, ( | |
| f"Loop detected: sentence '{sample[:60]}' occupies " | |
| f"{ratio*100:.1f}% of blocks ({cnt}/{total})" | |
| ) | |
| # R3: consecutive repeats (>= N identical in a row) | |
| consecutive = 0 | |
| last_norm = None | |
| for v in values: | |
| nv = _normalise(v) | |
| if nv and nv == last_norm and not _is_numeric_only(v) and len(v) > 4: | |
| consecutive += 1 | |
| if consecutive >= self.max_consecutive_repeats: | |
| return False, ( | |
| f"Consecutive loop detected: '{v[:60]}' " | |
| f"repeated {consecutive + 1}+ times in a row" | |
| ) | |
| else: | |
| consecutive = 0 | |
| last_norm = nv | |
| # R4: Chinese character leakage | |
| has_chinese_source = any( | |
| _chinese_ratio(str(b["text"])) > 0.1 for b in input_chunk | |
| ) | |
| if has_chinese_source: | |
| high_zh_blocks = [ | |
| k for k, v in norm.items() | |
| if _chinese_ratio(v) > self.max_chinese_ratio and len(v) > 3 | |
| ] | |
| if len(high_zh_blocks) > max(2, total * 0.2): | |
| return False, ( | |
| f"Chinese leakage: {len(high_zh_blocks)} blocks still contain " | |
| f">{self.max_chinese_ratio*100:.0f}% Chinese characters after translation" | |
| ) | |
| # R5: LLM meta-text / commentary | |
| for k, v in norm.items(): | |
| lv = v.lower() | |
| for phrase in self.BANNED_PHRASES: | |
| if phrase in lv: | |
| return False, ( | |
| f"LLM commentary detected in block {k}: " | |
| f"'{phrase}' inside '{v[:80]}'" | |
| ) | |
| # R6: Extreme length (>3.5x original AND >50 chars) | |
| effective_length_ratio = self.max_length_ratio | |
| if has_chinese_source: | |
| effective_length_ratio = 7.0 | |
| for b in input_chunk: | |
| bid = str(b["id"]) | |
| orig_len = len(str(b["text"]).strip()) | |
| trans_len = len(norm.get(bid, "")) | |
| if trans_len > max(50, orig_len * effective_length_ratio): | |
| return False, ( | |
| f"Block {bid} is suspiciously long " | |
| f"({trans_len} chars vs {orig_len} original)" | |
| ) | |
| return True, "OK" | |
| def validate_single_block(self, block_id, source_text, translated_text): | |
| """Quick single-block check used when retrying individual failed blocks.""" | |
| v = str(translated_text).strip() | |
| lv = v.lower() | |
| for phrase in self.BANNED_PHRASES: | |
| if phrase in lv: | |
| return False, f"Commentary in block {block_id}: '{phrase}'" | |
| if _chinese_ratio(v) > self.max_chinese_ratio and len(v) > 3: | |
| return False, f"Chinese leakage in block {block_id}" | |
| return True, "OK" | |