ndmhung6 commited on
Commit
6cb4903
·
1 Parent(s): 02654ee

feat: add local offline document translation tool with multiple styles

Browse files
app.py CHANGED
@@ -1,10 +1,46 @@
1
  import sys
2
  import os
 
3
 
4
  # Add root folder to sys.path
5
  sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
6
 
7
- from tools.TextToSpeech.app import main
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
8
 
9
  if __name__ == "__main__":
10
  main()
 
1
  import sys
2
  import os
3
+ import gradio as gr
4
 
5
  # Add root folder to sys.path
6
  sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
7
 
8
+ from tools.TextToSpeech.app import demo as tts_demo
9
+ from tools.DocumentTranslator.app import demo as trans_demo
10
+ from tools.TextToSpeech.ui_constants import theme, css, head_html
11
+
12
+ # Create unified block interface
13
+ with gr.Blocks(theme=theme, css=css, title="Pum's Tools", head=head_html) as demo:
14
+ gr.HTML("""
15
+ <div class="header-box">
16
+ <h1 class="header-title">
17
+ <span class="header-icon">🦜</span>
18
+ <span class="gradient-text">Pum's Tools</span>
19
+ </h1>
20
+ <p class="header-subtitle">Bộ sưu tập công cụ tiện ích đa năng</p>
21
+ </div>
22
+ """)
23
+
24
+ with gr.Tabs():
25
+ with gr.Tab("🗣️ Text to Speech"):
26
+ tts_demo.render()
27
+
28
+ with gr.Tab("📄 Dịch tài liệu"):
29
+ trans_demo.render()
30
+
31
+ def main():
32
+ server_name = os.getenv("GRADIO_SERVER_NAME", "0.0.0.0" if ("SPACE_ID" in os.environ or "PORT" in os.environ) else "127.0.0.1")
33
+ default_port = "7860" if "SPACE_ID" in os.environ else ("7861" if "PORT" not in os.environ else os.getenv("PORT"))
34
+ server_port = int(os.getenv("GRADIO_SERVER_PORT", default_port))
35
+
36
+ is_on_colab = os.getenv("COLAB_RELEASE_TAG") is not None
37
+ from vieneu_utils.core_utils import env_bool
38
+ share = env_bool("GRADIO_SHARE", default=is_on_colab)
39
+
40
+ if server_name == "0.0.0.0" and os.getenv("GRADIO_SHARE") is None:
41
+ share = False
42
+
43
+ demo.queue().launch(server_name=server_name, server_port=server_port, share=share)
44
 
45
  if __name__ == "__main__":
46
  main()
requirements.txt CHANGED
@@ -3,3 +3,9 @@ gradio>=5.49.1
3
  PyYAML
4
  soundfile
5
  numpy
 
 
 
 
 
 
 
3
  PyYAML
4
  soundfile
5
  numpy
6
+ deep-translator>=1.11.4
7
+ pypdf>=5.0.0
8
+ python-docx>=1.1.2
9
+ torch
10
+ transformers
11
+
tools/DocumentTranslator/__init__.py ADDED
@@ -0,0 +1 @@
 
 
1
+ # Document Translator package
tools/DocumentTranslator/app.py ADDED
@@ -0,0 +1,73 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import gradio as gr
2
+ import os
3
+ from tools.DocumentTranslator.translator import translate_document, LANGUAGES, STYLES
4
+
5
+ def handle_translation(file, source_lang, target_lang, style):
6
+ if file is None:
7
+ return None, "⚠️ Vui lòng tải lên tài liệu!", ""
8
+
9
+ source_code = LANGUAGES[source_lang]
10
+ target_code = LANGUAGES[target_lang]
11
+
12
+ status_msg = f"⏳ Đang tải mô hình & dịch tài liệu cục bộ (Offline)... Phong cách: {style}."
13
+ yield None, status_msg, ""
14
+
15
+ out_path, preview = translate_document(file.name, target_code, source_code, style)
16
+
17
+ if out_path is None:
18
+ yield None, preview, ""
19
+ else:
20
+ yield out_path, f"✅ Dịch thành công cục bộ trên máy! (Phong cách: {style})", preview
21
+
22
+ with gr.Blocks() as demo:
23
+ gr.HTML("<div style='margin-bottom: 20px;'><h2>📄 Dịch tài liệu cục bộ (On-Device Neural Translation)</h2><p style='color: #64748b; margin-top: 5px;'>Chạy hoàn toàn cục bộ (offline) trên CPU/GPU của bạn, không cần API Key, không gửi dữ liệu ra internet. Hỗ trợ các định dạng file .txt, .md, .json, .pdf, .docx.</p></div>")
24
+
25
+ with gr.Row():
26
+ with gr.Column(scale=3):
27
+ file_input = gr.File(
28
+ label="Tải lên tài liệu",
29
+ file_types=[".txt", ".docx", ".pdf", ".md", ".json"]
30
+ )
31
+ with gr.Row():
32
+ source_lang = gr.Dropdown(
33
+ choices=list(LANGUAGES.keys()),
34
+ value="Tự động phát hiện",
35
+ label="Ngôn ngữ nguồn"
36
+ )
37
+ target_lang = gr.Dropdown(
38
+ choices=[k for k in LANGUAGES.keys() if k != "Tự động phát hiện"],
39
+ value="Tiếng Việt",
40
+ label="Ngôn ngữ đích"
41
+ )
42
+ style_select = gr.Dropdown(
43
+ choices=list(STYLES.keys()),
44
+ value="Mặc định",
45
+ label="Phong cách dịch"
46
+ )
47
+ btn_translate = gr.Button("🚀 Bắt đầu dịch", variant="primary")
48
+
49
+ with gr.Column(scale=2):
50
+ status_output = gr.Textbox(
51
+ label="Trạng thái",
52
+ value="⏳ Chờ tải tài liệu...",
53
+ interactive=False
54
+ )
55
+ file_output = gr.File(
56
+ label="Tải về tài liệu đã dịch",
57
+ interactive=False
58
+ )
59
+ preview_output = gr.Textbox(
60
+ label="Xem trước nội dung (1000 ký tự đầu)",
61
+ lines=10,
62
+ interactive=False,
63
+ placeholder="Nội dung dịch sẽ hiển thị xem trước ở đây..."
64
+ )
65
+
66
+ btn_translate.click(
67
+ fn=handle_translation,
68
+ inputs=[file_input, source_lang, target_lang, style_select],
69
+ outputs=[file_output, status_output, preview_output]
70
+ )
71
+
72
+ if __name__ == "__main__":
73
+ demo.launch(server_port=7862)
tools/DocumentTranslator/requirements.txt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ deep-translator>=1.11.4
2
+ pypdf>=5.0.0
3
+ python-docx>=1.1.2
tools/DocumentTranslator/translator.py ADDED
@@ -0,0 +1,266 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import torch
3
+ from transformers import AutoTokenizer, AutoModelForSeq2SeqLM, M2M100ForConditionalGeneration, M2M100Tokenizer
4
+ from pypdf import PdfReader
5
+ import docx
6
+
7
+ LANGUAGES = {
8
+ "Tự động phát hiện": "auto",
9
+ "Tiếng Việt": "vi",
10
+ "Tiếng Anh": "en",
11
+ "Tiếng Nhật": "ja",
12
+ "Tiếng Hàn": "ko",
13
+ "Tiếng Trung (Giản thể)": "zh-CN",
14
+ "Tiếng Trung (Phồn thể)": "zh-TW",
15
+ "Tiếng Pháp": "fr",
16
+ "Tiếng Đức": "de",
17
+ "Tiếng Tây Ban Nha": "es",
18
+ "Tiếng Nga": "ru"
19
+ }
20
+
21
+ STYLES = {
22
+ "Mặc định": "Dịch chính xác, tự nhiên.",
23
+ "Trang trọng, lịch sự": "Dịch theo văn phong trang trọng, lịch sự, sử dụng từ ngữ kính trọng (ví dụ: dùng 'quý khách', 'chúng tôi', xưng hô lịch sự, từ ngữ trang nhã).",
24
+ "Thân mật, gần gũi": "Dịch theo văn phong thân mật, gần gũi, tự nhiên như giao tiếp hàng ngày (ví dụ: dùng 'bạn', 'mình', 'tớ', câu văn tự nhiên, thoải mái).",
25
+ "Học thuật, chuyên ngành": "Dịch theo phong cách học thuật, khoa học chuyên ngành, sử dụng chính xác các thuật ngữ chuyên môn.",
26
+ "Văn học, nghệ thuật": "Dịch theo phong cách văn thơ bay bổng, uyển chuyển, giàu tính nhạc và hình ảnh nghệ thuật."
27
+ }
28
+
29
+ # Cache for loaded local models
30
+ _LOCAL_MODEL_CACHE = {}
31
+
32
+ def get_local_model_and_tokenizer(model_name: str):
33
+ """Load and cache local translation model from Hugging Face."""
34
+ global _LOCAL_MODEL_CACHE
35
+ if model_name in _LOCAL_MODEL_CACHE:
36
+ return _LOCAL_MODEL_CACHE[model_name]
37
+
38
+ print(f"📦 Loading local translation model: {model_name}...")
39
+
40
+ if "m2m100" in model_name:
41
+ tokenizer = M2M100Tokenizer.from_pretrained(model_name)
42
+ model = M2M100ForConditionalGeneration.from_pretrained(model_name)
43
+ else:
44
+ tokenizer = AutoTokenizer.from_pretrained(model_name)
45
+ model = AutoModelForSeq2SeqLM.from_pretrained(model_name)
46
+
47
+ device = "cuda" if torch.cuda.is_available() else "cpu"
48
+ model = model.to(device)
49
+
50
+ _LOCAL_MODEL_CACHE[model_name] = (model, tokenizer, device)
51
+ return model, tokenizer, device
52
+
53
+ def apply_vietnamese_style(text: str, style: str) -> str:
54
+ """Adjust pronouns and word choices in Vietnamese text to match the requested style."""
55
+ if not text or style == "Mặc định":
56
+ return text
57
+
58
+ if style == "Trang trọng, lịch sự":
59
+ replacements = {
60
+ " tôi ": " chúng tôi ",
61
+ "Tôi ": "Chúng tôi ",
62
+ " tớ ": " chúng tôi ",
63
+ "Tớ ": "Chúng tôi ",
64
+ " mình ": " chúng tôi ",
65
+ "Mình ": "Chúng tôi ",
66
+ " cậu ": " quý khách ",
67
+ "Cậu ": "Quý khách ",
68
+ " mày ": " quý khách ",
69
+ "Mày ": "Quý khách ",
70
+ " bạn ": " quý khách ",
71
+ "Bạn ": "Quý khách ",
72
+ }
73
+ for k, v in replacements.items():
74
+ text = text.replace(k, v)
75
+ elif style == "Thân mật, gần gũi":
76
+ replacements = {
77
+ " tôi ": " mình ",
78
+ "Tôi ": "Mình ",
79
+ " chúng tôi ": " chúng mình ",
80
+ "Chúng tôi ": "Chúng mình ",
81
+ " quý khách ": " bạn ",
82
+ "Quý khách ": "Bạn ",
83
+ " ngài ": " bạn ",
84
+ "Ngài ": "Bạn ",
85
+ }
86
+ for k, v in replacements.items():
87
+ text = text.replace(k, v)
88
+ elif style == "Văn học, nghệ thuật":
89
+ replacements = {
90
+ " nhanh chóng ": " mau chóng ",
91
+ " rất ": " vô cùng ",
92
+ " đẹp ": " thơ mộng ",
93
+ " nói ": " bộc bạch ",
94
+ }
95
+ for k, v in replacements.items():
96
+ text = text.replace(k, v)
97
+
98
+ return text
99
+
100
+ def translate_text(text: str, target_lang: str, source_lang: str = "auto", style: str = "Mặc định") -> str:
101
+ """Translate plain text locally in small chunks of 500 characters."""
102
+ if not text or not text.strip():
103
+ return ""
104
+
105
+ src = LANGUAGES.get(source_lang, "auto")
106
+ tgt = LANGUAGES.get(target_lang, "vi")
107
+
108
+ if src == "auto":
109
+ src = "en" # Fallback auto to English source
110
+
111
+ # Determine model name
112
+ if src == "en" and tgt == "vi":
113
+ model_name = "Helsinki-NLP/opus-mt-en-vi"
114
+ elif src == "vi" and tgt == "en":
115
+ model_name = "Helsinki-NLP/opus-mt-vi-en"
116
+ elif src == "zh-CN" and tgt == "vi":
117
+ model_name = "Helsinki-NLP/opus-mt-zh-vi"
118
+ elif src == "fr" and tgt == "vi":
119
+ model_name = "Helsinki-NLP/opus-mt-fr-vi"
120
+ elif src == "de" and tgt == "vi":
121
+ model_name = "Helsinki-NLP/opus-mt-de-vi"
122
+ else:
123
+ model_name = "facebook/m2m100_418M"
124
+
125
+ model, tokenizer, device = get_local_model_and_tokenizer(model_name)
126
+
127
+ # Split text into sentences or small chunks (approx 500 characters) to avoid generation cutoff
128
+ chunk_size = 500
129
+ paragraphs = text.split("\n")
130
+ translated_paragraphs = []
131
+
132
+ for para in paragraphs:
133
+ if not para.strip():
134
+ translated_paragraphs.append("")
135
+ continue
136
+
137
+ # Split paragraph into chunks of 500 chars
138
+ chunks = [para[i:i+chunk_size] for i in range(0, len(para), chunk_size)]
139
+ translated_chunks = []
140
+
141
+ for chunk in chunks:
142
+ if not chunk.strip():
143
+ translated_chunks.append(chunk)
144
+ continue
145
+
146
+ try:
147
+ if "m2m100" in model_name:
148
+ src_m2m = src.split("-")[0]
149
+ tgt_m2m = tgt.split("-")[0]
150
+ tokenizer.src_lang = src_m2m
151
+ inputs = tokenizer(chunk, return_tensors="pt").to(device)
152
+ with torch.no_grad():
153
+ generated_tokens = model.generate(**inputs, forced_bos_token_id=tokenizer.get_lang_id(tgt_m2m), max_length=512)
154
+ translated = tokenizer.batch_decode(generated_tokens, skip_special_tokens=True)[0]
155
+ else:
156
+ inputs = tokenizer(chunk, return_tensors="pt", padding=True, truncation=True, max_length=512).to(device)
157
+ with torch.no_grad():
158
+ outputs = model.generate(**inputs, max_length=512)
159
+ translated = tokenizer.decode(outputs[0], skip_special_tokens=True)
160
+
161
+ # Apply Vietnamese style styling if target is Vietnamese
162
+ if tgt == "vi":
163
+ translated = apply_vietnamese_style(translated, style)
164
+
165
+ translated_chunks.append(translated)
166
+ except Exception as e:
167
+ print(f"Error local translating chunk: {e}")
168
+ translated_chunks.append(chunk)
169
+
170
+ translated_paragraphs.append("".join(translated_chunks))
171
+
172
+ return "\n".join(translated_paragraphs)
173
+
174
+ def translate_docx(file_path: str, target_lang: str, source_lang: str = "auto", style: str = "Mặc định") -> str:
175
+ """Translate a Word (.docx) document locally paragraph by paragraph."""
176
+ doc = docx.Document(file_path)
177
+
178
+ for i, p in enumerate(doc.paragraphs):
179
+ if p.text.strip():
180
+ try:
181
+ p.text = translate_text(p.text, target_lang, source_lang, style)
182
+ except Exception as e:
183
+ print(f"Error translating docx paragraph {i}: {e}")
184
+
185
+ for t in doc.tables:
186
+ for row in t.rows:
187
+ for cell in row.cells:
188
+ for p in cell.paragraphs:
189
+ if p.text.strip():
190
+ try:
191
+ p.text = translate_text(p.text, target_lang, source_lang, style)
192
+ except Exception as e:
193
+ print(f"Error translating table paragraph: {e}")
194
+
195
+ dir_name = os.path.dirname(file_path)
196
+ base_name = os.path.basename(file_path)
197
+ out_name = f"translated_{base_name}"
198
+ out_path = os.path.join(dir_name, out_name)
199
+ doc.save(out_path)
200
+ return out_path
201
+
202
+ def translate_pdf(file_path: str, target_lang: str, source_lang: str = "auto", style: str = "Mặc định") -> tuple[str, str]:
203
+ """Extract text from PDF, translate locally, and save it as a new .docx file."""
204
+ reader = PdfReader(file_path)
205
+ text_content = []
206
+
207
+ for page in reader.pages:
208
+ page_text = page.extract_text()
209
+ if page_text:
210
+ text_content.append(page_text)
211
+
212
+ full_text = "\n\n".join(text_content)
213
+ translated_text = translate_text(full_text, target_lang, source_lang, style)
214
+
215
+ dir_name = os.path.dirname(file_path)
216
+ base_name = os.path.basename(file_path)
217
+ out_docx_name = f"translated_{os.path.splitext(base_name)[0]}.docx"
218
+ out_docx_path = os.path.join(dir_name, out_docx_name)
219
+
220
+ doc = docx.Document()
221
+ for paragraph in translated_text.split("\n\n"):
222
+ if paragraph.strip():
223
+ doc.add_paragraph(paragraph)
224
+ doc.save(out_docx_path)
225
+
226
+ return out_docx_path, translated_text
227
+
228
+ def translate_document(file_path: str, target_lang: str, source_lang: str = "auto", style: str = "Mặc định") -> tuple[str, str]:
229
+ """Main function to parse extension and translate."""
230
+ if not file_path or not os.path.exists(file_path):
231
+ return None, "❌ File không tồn tại."
232
+
233
+ ext = os.path.splitext(file_path)[1].lower()
234
+ base_name = os.path.basename(file_path)
235
+ dir_name = os.path.dirname(file_path)
236
+
237
+ try:
238
+ if ext == ".docx":
239
+ out_path = translate_docx(file_path, target_lang, source_lang, style)
240
+ doc = docx.Document(out_path)
241
+ preview_text = "\n".join([p.text for p in doc.paragraphs[:10]])
242
+ return out_path, preview_text
243
+
244
+ elif ext == ".pdf":
245
+ out_path, translated_text = translate_pdf(file_path, target_lang, source_lang, style)
246
+ return out_path, translated_text[:1000]
247
+
248
+ elif ext in [".txt", ".md", ".json"]:
249
+ with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
250
+ content = f.read()
251
+ translated_text = translate_text(content, target_lang, source_lang, style)
252
+
253
+ out_name = f"translated_{base_name}"
254
+ out_path = os.path.join(dir_name, out_name)
255
+ with open(out_path, "w", encoding="utf-8") as f:
256
+ f.write(translated_text)
257
+
258
+ return out_path, translated_text[:1000]
259
+
260
+ else:
261
+ return None, f"❌ Định dạng file {ext} không được hỗ trợ."
262
+
263
+ except Exception as e:
264
+ import traceback
265
+ traceback.print_exc()
266
+ return None, f"❌ Lỗi trong quá trình dịch: {str(e)}"
tools/TextToSpeech/app.py CHANGED
@@ -1543,15 +1543,6 @@ with gr.Blocks(theme=theme, css=css, title="Pum's Tools — TextToSpeech", head=
1543
  session_id_state = gr.State("")
1544
 
1545
  with gr.Column(elem_classes="container"):
1546
- gr.HTML("""
1547
- <div class="header-box">
1548
- <h1 class="header-title">
1549
- <span class="header-icon">🦜</span>
1550
- <span class="gradient-text">Pum's Tools</span>
1551
- </h1>
1552
- <p class="header-subtitle">🗣️ Text to Speech</p>
1553
- </div>
1554
- """)
1555
 
1556
  # --- CONFIGURATION (Hidden from UI, but active in backend) ---
1557
  with gr.Group(visible=False):
 
1543
  session_id_state = gr.State("")
1544
 
1545
  with gr.Column(elem_classes="container"):
 
 
 
 
 
 
 
 
 
1546
 
1547
  # --- CONFIGURATION (Hidden from UI, but active in backend) ---
1548
  with gr.Group(visible=False):