Spaces:
Running on Zero
Running on Zero
Download app/core/translation_core.py from hoangtaiii/DRIPPY4: direct link, hf CLI and curl.
- Browser
- Download file 9 kB
-
https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/translation_core.py
- Command line
-
hf download hf://spaces/hoangtaiii/DRIPPY4/app/core/translation_core.py
-
curl -L -o translation_core.py https://huggingface.co/spaces/hoangtaiii/DRIPPY4/resolve/main/app/core/translation_core.py
9 kB
| import os | |
| import sys | |
| import json | |
| import time | |
| import argparse | |
| from pathlib import Path | |
| # Enforce UTF-8 for Windows console | |
| if sys.platform == 'win32': | |
| try: | |
| if hasattr(sys.stdout, 'reconfigure'): | |
| sys.stdout.reconfigure(encoding='utf-8') | |
| if hasattr(sys.stderr, 'reconfigure'): | |
| sys.stderr.reconfigure(encoding='utf-8') | |
| except Exception: | |
| pass | |
| # Ensure app is in path | |
| sys.path.append(str(Path(__file__).parent.parent.parent)) | |
| from app.translation.manager import TranslationManager | |
| class TranslationCore: | |
| def __init__(self, glossary_path=None): | |
| self.glossary = {} | |
| self.style = "Tự nhiên, thuyết minh" | |
| if glossary_path: | |
| self.load_glossary(glossary_path) | |
| else: | |
| default_path = Path(__file__).parent.parent.parent / "glossary.json" | |
| self.load_glossary(default_path) | |
| def load_glossary(self, path): | |
| path = Path(path) | |
| if path.exists(): | |
| try: | |
| with open(path, "r", encoding="utf-8") as f: | |
| data = json.load(f) | |
| self.glossary = data.get("glossary", {}) | |
| self.style = data.get("style", "Tự nhiên, thuyết minh") | |
| except Exception: | |
| self.glossary = {} | |
| else: | |
| self.glossary = { | |
| "RTX": "Vạc đồ họa RTX", | |
| "VRAM": "Bộ nhớ đồ họa", | |
| "CPU": "Bộ vi xử lý" | |
| } | |
| self.style = "Hài hước, tự nhiên, văn phong nói, phù hợp video ngắn TikTok/Reels" | |
| def parse_srt(self, content: str): | |
| import re | |
| content = content.strip().replace('\r\n', '\n') | |
| pattern = r'(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->\s*\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)' | |
| matches = re.finditer(pattern, content, re.DOTALL) | |
| parsed_blocks = [] | |
| for match in matches: | |
| try: | |
| b_id = match.group(1).strip() | |
| if b_id.isdigit(): | |
| b_id_int = int(b_id) | |
| else: | |
| continue | |
| timestamp = match.group(2).strip() | |
| start_ts, end_ts = [p.strip() for p in re.split(r'\s*-->\s*', timestamp, maxsplit=1)] | |
| start_ms = self._parse_srt_time_ms(start_ts) | |
| end_ms = self._parse_srt_time_ms(end_ts) | |
| duration_sec = max(0.0, (end_ms - start_ms) / 1000.0) if start_ms is not None and end_ms is not None else 0.0 | |
| text = match.group(3).strip() | |
| text = " ".join([l.strip() for l in text.split('\n') if l.strip()]) | |
| parsed_blocks.append({ | |
| "id": str(b_id_int), | |
| "timestamp": timestamp, | |
| "start": start_ts, | |
| "end": end_ts, | |
| "start_ms": start_ms, | |
| "end_ms": end_ms, | |
| "duration_sec": round(duration_sec, 3), | |
| "text": text | |
| }) | |
| except (ValueError, IndexError, AttributeError): | |
| continue | |
| return parsed_blocks | |
| def _parse_srt_time_ms(value: str): | |
| import re | |
| m = re.match(r"^(\d{2}):(\d{2}):(\d{2})[,.](\d{3})$", str(value).strip()) | |
| if not m: | |
| return None | |
| hh, mm, ss, ms = [int(x) for x in m.groups()] | |
| return ((hh * 3600 + mm * 60 + ss) * 1000) + ms | |
| def format_srt(self, parsed_blocks): | |
| srt_lines = [] | |
| for block in parsed_blocks: | |
| srt_lines.append(f"{block['id']}") | |
| srt_lines.append(f"{block['timestamp']}") | |
| srt_lines.append(f"{block['text']}") | |
| srt_lines.append("") # empty line separator | |
| return "\n".join(srt_lines).strip() + "\n" | |
| def translate_srt_file(self, input_path, output_path, engine="Google (Free)", model="auto", api_url="", api_key="", high_quality=False, log_fn=None): | |
| input_path = Path(input_path) | |
| output_path = Path(output_path) | |
| if not input_path.exists(): | |
| raise FileNotFoundError(f"Không tìm thấy file phụ đề đầu vào: {input_path}") | |
| with open(input_path, "r", encoding="utf-8") as f: | |
| content = f.read() | |
| blocks = self.parse_srt(content) | |
| if not blocks: | |
| raise Exception("File phụ đề trống hoặc không đúng định dạng SRT.") | |
| if log_fn: | |
| log_fn(f"📖 Đọc thành công {len(blocks)} dòng phụ đề. Engine: {engine}") | |
| # Setup configuration mapping for TranslationManager | |
| config_dict = { | |
| "router_url": api_url, | |
| "router_key": api_key, | |
| "router_model": model if model and model != "auto" else "meta-llama/llama-3.3-70b-instruct:free", | |
| "super_ai_gate_url": api_url, | |
| "super_ai_gate_key": api_key, | |
| "super_ai_gate_model": model if model and model != "auto" else "meta-llama/llama-3.3-70b-instruct:free", | |
| "ollama_model": model if model and model != "auto" else "hf.co/lmstudio-community/Qwen3.5-9B-GGUF:Q6_K" | |
| } | |
| disable_ollama = True | |
| try: | |
| cfg_path = Path(__file__).parent.parent.parent / "config.json" | |
| if cfg_path.exists(): | |
| with open(cfg_path, "r", encoding="utf-8") as f: | |
| cfg_data = json.load(f) | |
| disable_ollama = cfg_data.get("translation", {}).get("disable_ollama", True) | |
| except Exception: | |
| pass | |
| is_ultimate = "Tối thượng" in engine or "Ultimate" in engine | |
| if is_ultimate: | |
| engines_list = ["API Pool", "9Router", "Super AI"] | |
| if not disable_ollama: | |
| engines_list.append("Ollama") | |
| else: | |
| engines_list = [engine] | |
| # Instantiate modular translation manager | |
| manager = TranslationManager( | |
| engines_list=engines_list, | |
| config_dict=config_dict, | |
| glossary=self.glossary, | |
| style=self.style | |
| ) | |
| # Execute translation | |
| translated_dict = manager.translate_blocks(blocks, log_fn=log_fn) | |
| try: | |
| from app.core.translation_post_editor import post_edit_translation | |
| for b in blocks: | |
| b_id = str(b["id"]) | |
| translated_dict[b_id] = post_edit_translation(b.get("text", ""), translated_dict.get(b_id, b.get("text", ""))) | |
| except Exception: | |
| pass | |
| # Save validation report if available | |
| report_path = output_path.parent / "translation_validation_report.json" | |
| manager.save_validation_report(report_path) | |
| # Build translated blocks | |
| translated_blocks = [] | |
| for b in blocks: | |
| b_id = str(b["id"]) | |
| translated_blocks.append({ | |
| "id": b["id"], | |
| "timestamp": b["timestamp"], | |
| "text": translated_dict.get(b_id, b["text"]) | |
| }) | |
| # Save SRT file | |
| with open(output_path, "w", encoding="utf-8") as f: | |
| f.write(self.format_srt(translated_blocks)) | |
| if log_fn: | |
| log_fn("PROGRESS: 100%") | |
| log_fn(f"✅ Hoàn tất dịch phụ đề! Lưu tại {output_path}") | |
| return len(blocks) | |
| def main(): | |
| parser = argparse.ArgumentParser(description="Standalone AI SRT Translation CLI") | |
| parser.add_argument("--input", required=True, help="Path to input original SRT file") | |
| parser.add_argument("--output", required=True, help="Path to output translated SRT file") | |
| parser.add_argument("--engine", default="Google (Free)", help="Translation engine name") | |
| parser.add_argument("--model", default="auto", help="Model name") | |
| parser.add_argument("--api-url", default="", help="Base API URL") | |
| parser.add_argument("--api-key", default="", help="API Key") | |
| parser.add_argument("--high-quality", action="store_true", help="Run polishing pass") | |
| parser.add_argument("--glossary", default=None, help="Path to glossary JSON") | |
| args = parser.parse_args() | |
| print(f"Starting translation CLI with engine={args.engine}, model={args.model}...", flush=True) | |
| core = TranslationCore(glossary_path=args.glossary) | |
| try: | |
| core.translate_srt_file( | |
| input_path=args.input, | |
| output_path=args.output, | |
| engine=args.engine, | |
| model=args.model, | |
| api_url=args.api_url, | |
| api_key=args.api_key, | |
| high_quality=args.high_quality, | |
| log_fn=lambda msg: print(msg, flush=True) | |
| ) | |
| print("Translation CLI finished successfully.", flush=True) | |
| sys.exit(0) | |
| except Exception as e: | |
| print(f"Error during translation execution: {e}", file=sys.stderr, flush=True) | |
| sys.exit(1) | |
| if __name__ == "__main__": | |
| main() | |