""" app/core/translation_validator.py ───────────────────────────────── Validates translated chunk output from any LLM provider. Ported from claude_fake with cloud-native adaptations. Features: • Stricter consecutive-repeat detection (3+ identical lines in a row) • Chinese character leakage check (>15% Chinese after translation) • Richer meta-text / LLM-commentary detection • Numeric-only block whitelisting (don't flag "1", "2" as duplicates) • Fast-path: if input is all English, skip Chinese-leakage check • Normalised-text dedup: ignores punctuation differences when counting loops """ import re import unicodedata def _chinese_ratio(text: str) -> float: if not text: return 0.0 zh_chars = sum(1 for c in text if '\u4e00' <= c <= '\u9fff') total = len(text.replace(' ', '')) return zh_chars / total if total else 0.0 def _normalise(text: str) -> str: """Strip punctuation/whitespace for dedup comparison.""" return re.sub(r'[\s.,!?;:"\'\-\u2013\u2014]+', '', text).lower() def _is_numeric_only(text: str) -> bool: return bool(re.fullmatch(r'[\d\s.,%-]+', text.strip())) class TranslationValidator: # LLM meta-commentary phrases (EN + VI) BANNED_PHRASES = [ # English "wait no", "corrected version", "looking at the original", "note:", "here is", "instead of", "respect user instruction", "sure, here is", "translation:", "clean version", "typos fixed", "fixed version", "original text", "below is", "i have cleaned", "timeline rows", "the corrected subtitle", "as requested", "here's the", "here are the", "i will translate", "let me translate", "the following", "translating the", "sure!", "okay!", "of course!", "certainly!", "absolutely!", "as an ai", "as a language model", # Vietnamese "cho da", "can dam bao", "dinh dang thoi gian", "lam sach", "sua loi", "ket qua:", "ban dich:", "chinh xac hon", "sau khi kiem tra", "duoi day la", "dong thoai", "ngu canh", "cach dich", "chu thich", "giai thich", "don dep", "day la ban dich", "toi se dich", "de dich chinh xac", ] def __init__(self, max_repeat_ratio: float = 0.20, max_consecutive_repeats: int = 2, max_chinese_ratio: float = 0.15, max_length_ratio: float = 3.5): self.max_repeat_ratio = max_repeat_ratio self.max_consecutive_repeats = max_consecutive_repeats self.max_chinese_ratio = max_chinese_ratio self.max_length_ratio = max_length_ratio def validate(self, input_chunk, translated_dict): """ Validate translated output dict. Returns (is_valid: bool, error_message: str). """ if not isinstance(translated_dict, dict): return False, "Output is not a dictionary" norm = {str(k): str(v).strip() for k, v in translated_dict.items()} # R1: count match input_keys = [str(b["id"]) for b in input_chunk] missing = [k for k in input_keys if k not in norm] if missing: return False, f"Missing keys in output: {missing}" values = [norm[k] for k in input_keys] # R2: loop by ratio (>20% identical sentence) total = len(input_keys) if total > 3: counts = {} for v in values: nv = _normalise(v) if nv and not _is_numeric_only(v) and len(v) > 4: counts[nv] = counts.get(nv, 0) + 1 for nv, cnt in counts.items(): ratio = cnt / total if cnt > 1 and ratio > self.max_repeat_ratio: sample = next(v for v in values if _normalise(v) == nv) return False, ( f"Loop detected: sentence '{sample[:60]}' occupies " f"{ratio*100:.1f}% of blocks ({cnt}/{total})" ) # R3: consecutive repeats (>= N identical in a row) consecutive = 0 last_norm = None for v in values: nv = _normalise(v) if nv and nv == last_norm and not _is_numeric_only(v) and len(v) > 4: consecutive += 1 if consecutive >= self.max_consecutive_repeats: return False, ( f"Consecutive loop detected: '{v[:60]}' " f"repeated {consecutive + 1}+ times in a row" ) else: consecutive = 0 last_norm = nv # R4: Chinese character leakage has_chinese_source = any( _chinese_ratio(str(b["text"])) > 0.1 for b in input_chunk ) if has_chinese_source: high_zh_blocks = [ k for k, v in norm.items() if _chinese_ratio(v) > self.max_chinese_ratio and len(v) > 3 ] if len(high_zh_blocks) > max(2, total * 0.2): return False, ( f"Chinese leakage: {len(high_zh_blocks)} blocks still contain " f">{self.max_chinese_ratio*100:.0f}% Chinese characters after translation" ) # R5: LLM meta-text / commentary for k, v in norm.items(): lv = v.lower() for phrase in self.BANNED_PHRASES: if phrase in lv: return False, ( f"LLM commentary detected in block {k}: " f"'{phrase}' inside '{v[:80]}'" ) # R6: Extreme length (>3.5x original AND >50 chars) effective_length_ratio = self.max_length_ratio if has_chinese_source: effective_length_ratio = 7.0 for b in input_chunk: bid = str(b["id"]) orig_len = len(str(b["text"]).strip()) trans_len = len(norm.get(bid, "")) if trans_len > max(50, orig_len * effective_length_ratio): return False, ( f"Block {bid} is suspiciously long " f"({trans_len} chars vs {orig_len} original)" ) return True, "OK" def validate_single_block(self, block_id, source_text, translated_text): """Quick single-block check used when retrying individual failed blocks.""" v = str(translated_text).strip() lv = v.lower() for phrase in self.BANNED_PHRASES: if phrase in lv: return False, f"Commentary in block {block_id}: '{phrase}'" if _chinese_ratio(v) > self.max_chinese_ratio and len(v) > 3: return False, f"Chinese leakage in block {block_id}" return True, "OK"