Spaces:
Runtime error
Runtime error
feat: add local offline document translation tool with multiple styles
Browse files- app.py +37 -1
- requirements.txt +6 -0
- tools/DocumentTranslator/__init__.py +1 -0
- tools/DocumentTranslator/app.py +73 -0
- tools/DocumentTranslator/requirements.txt +3 -0
- tools/DocumentTranslator/translator.py +266 -0
- tools/TextToSpeech/app.py +0 -9
app.py
CHANGED
|
@@ -1,10 +1,46 @@
|
|
| 1 |
import sys
|
| 2 |
import os
|
|
|
|
| 3 |
|
| 4 |
# Add root folder to sys.path
|
| 5 |
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
| 6 |
|
| 7 |
-
from tools.TextToSpeech.app import
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 8 |
|
| 9 |
if __name__ == "__main__":
|
| 10 |
main()
|
|
|
|
| 1 |
import sys
|
| 2 |
import os
|
| 3 |
+
import gradio as gr
|
| 4 |
|
| 5 |
# Add root folder to sys.path
|
| 6 |
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
|
| 7 |
|
| 8 |
+
from tools.TextToSpeech.app import demo as tts_demo
|
| 9 |
+
from tools.DocumentTranslator.app import demo as trans_demo
|
| 10 |
+
from tools.TextToSpeech.ui_constants import theme, css, head_html
|
| 11 |
+
|
| 12 |
+
# Create unified block interface
|
| 13 |
+
with gr.Blocks(theme=theme, css=css, title="Pum's Tools", head=head_html) as demo:
|
| 14 |
+
gr.HTML("""
|
| 15 |
+
<div class="header-box">
|
| 16 |
+
<h1 class="header-title">
|
| 17 |
+
<span class="header-icon">🦜</span>
|
| 18 |
+
<span class="gradient-text">Pum's Tools</span>
|
| 19 |
+
</h1>
|
| 20 |
+
<p class="header-subtitle">Bộ sưu tập công cụ tiện ích đa năng</p>
|
| 21 |
+
</div>
|
| 22 |
+
""")
|
| 23 |
+
|
| 24 |
+
with gr.Tabs():
|
| 25 |
+
with gr.Tab("🗣️ Text to Speech"):
|
| 26 |
+
tts_demo.render()
|
| 27 |
+
|
| 28 |
+
with gr.Tab("📄 Dịch tài liệu"):
|
| 29 |
+
trans_demo.render()
|
| 30 |
+
|
| 31 |
+
def main():
|
| 32 |
+
server_name = os.getenv("GRADIO_SERVER_NAME", "0.0.0.0" if ("SPACE_ID" in os.environ or "PORT" in os.environ) else "127.0.0.1")
|
| 33 |
+
default_port = "7860" if "SPACE_ID" in os.environ else ("7861" if "PORT" not in os.environ else os.getenv("PORT"))
|
| 34 |
+
server_port = int(os.getenv("GRADIO_SERVER_PORT", default_port))
|
| 35 |
+
|
| 36 |
+
is_on_colab = os.getenv("COLAB_RELEASE_TAG") is not None
|
| 37 |
+
from vieneu_utils.core_utils import env_bool
|
| 38 |
+
share = env_bool("GRADIO_SHARE", default=is_on_colab)
|
| 39 |
+
|
| 40 |
+
if server_name == "0.0.0.0" and os.getenv("GRADIO_SHARE") is None:
|
| 41 |
+
share = False
|
| 42 |
+
|
| 43 |
+
demo.queue().launch(server_name=server_name, server_port=server_port, share=share)
|
| 44 |
|
| 45 |
if __name__ == "__main__":
|
| 46 |
main()
|
requirements.txt
CHANGED
|
@@ -3,3 +3,9 @@ gradio>=5.49.1
|
|
| 3 |
PyYAML
|
| 4 |
soundfile
|
| 5 |
numpy
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
PyYAML
|
| 4 |
soundfile
|
| 5 |
numpy
|
| 6 |
+
deep-translator>=1.11.4
|
| 7 |
+
pypdf>=5.0.0
|
| 8 |
+
python-docx>=1.1.2
|
| 9 |
+
torch
|
| 10 |
+
transformers
|
| 11 |
+
|
tools/DocumentTranslator/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
# Document Translator package
|
tools/DocumentTranslator/app.py
ADDED
|
@@ -0,0 +1,73 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import gradio as gr
|
| 2 |
+
import os
|
| 3 |
+
from tools.DocumentTranslator.translator import translate_document, LANGUAGES, STYLES
|
| 4 |
+
|
| 5 |
+
def handle_translation(file, source_lang, target_lang, style):
|
| 6 |
+
if file is None:
|
| 7 |
+
return None, "⚠️ Vui lòng tải lên tài liệu!", ""
|
| 8 |
+
|
| 9 |
+
source_code = LANGUAGES[source_lang]
|
| 10 |
+
target_code = LANGUAGES[target_lang]
|
| 11 |
+
|
| 12 |
+
status_msg = f"⏳ Đang tải mô hình & dịch tài liệu cục bộ (Offline)... Phong cách: {style}."
|
| 13 |
+
yield None, status_msg, ""
|
| 14 |
+
|
| 15 |
+
out_path, preview = translate_document(file.name, target_code, source_code, style)
|
| 16 |
+
|
| 17 |
+
if out_path is None:
|
| 18 |
+
yield None, preview, ""
|
| 19 |
+
else:
|
| 20 |
+
yield out_path, f"✅ Dịch thành công cục bộ trên máy! (Phong cách: {style})", preview
|
| 21 |
+
|
| 22 |
+
with gr.Blocks() as demo:
|
| 23 |
+
gr.HTML("<div style='margin-bottom: 20px;'><h2>📄 Dịch tài liệu cục bộ (On-Device Neural Translation)</h2><p style='color: #64748b; margin-top: 5px;'>Chạy hoàn toàn cục bộ (offline) trên CPU/GPU của bạn, không cần API Key, không gửi dữ liệu ra internet. Hỗ trợ các định dạng file .txt, .md, .json, .pdf, .docx.</p></div>")
|
| 24 |
+
|
| 25 |
+
with gr.Row():
|
| 26 |
+
with gr.Column(scale=3):
|
| 27 |
+
file_input = gr.File(
|
| 28 |
+
label="Tải lên tài liệu",
|
| 29 |
+
file_types=[".txt", ".docx", ".pdf", ".md", ".json"]
|
| 30 |
+
)
|
| 31 |
+
with gr.Row():
|
| 32 |
+
source_lang = gr.Dropdown(
|
| 33 |
+
choices=list(LANGUAGES.keys()),
|
| 34 |
+
value="Tự động phát hiện",
|
| 35 |
+
label="Ngôn ngữ nguồn"
|
| 36 |
+
)
|
| 37 |
+
target_lang = gr.Dropdown(
|
| 38 |
+
choices=[k for k in LANGUAGES.keys() if k != "Tự động phát hiện"],
|
| 39 |
+
value="Tiếng Việt",
|
| 40 |
+
label="Ngôn ngữ đích"
|
| 41 |
+
)
|
| 42 |
+
style_select = gr.Dropdown(
|
| 43 |
+
choices=list(STYLES.keys()),
|
| 44 |
+
value="Mặc định",
|
| 45 |
+
label="Phong cách dịch"
|
| 46 |
+
)
|
| 47 |
+
btn_translate = gr.Button("🚀 Bắt đầu dịch", variant="primary")
|
| 48 |
+
|
| 49 |
+
with gr.Column(scale=2):
|
| 50 |
+
status_output = gr.Textbox(
|
| 51 |
+
label="Trạng thái",
|
| 52 |
+
value="⏳ Chờ tải tài liệu...",
|
| 53 |
+
interactive=False
|
| 54 |
+
)
|
| 55 |
+
file_output = gr.File(
|
| 56 |
+
label="Tải về tài liệu đã dịch",
|
| 57 |
+
interactive=False
|
| 58 |
+
)
|
| 59 |
+
preview_output = gr.Textbox(
|
| 60 |
+
label="Xem trước nội dung (1000 ký tự đầu)",
|
| 61 |
+
lines=10,
|
| 62 |
+
interactive=False,
|
| 63 |
+
placeholder="Nội dung dịch sẽ hiển thị xem trước ở đây..."
|
| 64 |
+
)
|
| 65 |
+
|
| 66 |
+
btn_translate.click(
|
| 67 |
+
fn=handle_translation,
|
| 68 |
+
inputs=[file_input, source_lang, target_lang, style_select],
|
| 69 |
+
outputs=[file_output, status_output, preview_output]
|
| 70 |
+
)
|
| 71 |
+
|
| 72 |
+
if __name__ == "__main__":
|
| 73 |
+
demo.launch(server_port=7862)
|
tools/DocumentTranslator/requirements.txt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
deep-translator>=1.11.4
|
| 2 |
+
pypdf>=5.0.0
|
| 3 |
+
python-docx>=1.1.2
|
tools/DocumentTranslator/translator.py
ADDED
|
@@ -0,0 +1,266 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
import os
|
| 2 |
+
import torch
|
| 3 |
+
from transformers import AutoTokenizer, AutoModelForSeq2SeqLM, M2M100ForConditionalGeneration, M2M100Tokenizer
|
| 4 |
+
from pypdf import PdfReader
|
| 5 |
+
import docx
|
| 6 |
+
|
| 7 |
+
LANGUAGES = {
|
| 8 |
+
"Tự động phát hiện": "auto",
|
| 9 |
+
"Tiếng Việt": "vi",
|
| 10 |
+
"Tiếng Anh": "en",
|
| 11 |
+
"Tiếng Nhật": "ja",
|
| 12 |
+
"Tiếng Hàn": "ko",
|
| 13 |
+
"Tiếng Trung (Giản thể)": "zh-CN",
|
| 14 |
+
"Tiếng Trung (Phồn thể)": "zh-TW",
|
| 15 |
+
"Tiếng Pháp": "fr",
|
| 16 |
+
"Tiếng Đức": "de",
|
| 17 |
+
"Tiếng Tây Ban Nha": "es",
|
| 18 |
+
"Tiếng Nga": "ru"
|
| 19 |
+
}
|
| 20 |
+
|
| 21 |
+
STYLES = {
|
| 22 |
+
"Mặc định": "Dịch chính xác, tự nhiên.",
|
| 23 |
+
"Trang trọng, lịch sự": "Dịch theo văn phong trang trọng, lịch sự, sử dụng từ ngữ kính trọng (ví dụ: dùng 'quý khách', 'chúng tôi', xưng hô lịch sự, từ ngữ trang nhã).",
|
| 24 |
+
"Thân mật, gần gũi": "Dịch theo văn phong thân mật, gần gũi, tự nhiên như giao tiếp hàng ngày (ví dụ: dùng 'bạn', 'mình', 'tớ', câu văn tự nhiên, thoải mái).",
|
| 25 |
+
"Học thuật, chuyên ngành": "Dịch theo phong cách học thuật, khoa học chuyên ngành, sử dụng chính xác các thuật ngữ chuyên môn.",
|
| 26 |
+
"Văn học, nghệ thuật": "Dịch theo phong cách văn thơ bay bổng, uyển chuyển, giàu tính nhạc và hình ảnh nghệ thuật."
|
| 27 |
+
}
|
| 28 |
+
|
| 29 |
+
# Cache for loaded local models
|
| 30 |
+
_LOCAL_MODEL_CACHE = {}
|
| 31 |
+
|
| 32 |
+
def get_local_model_and_tokenizer(model_name: str):
|
| 33 |
+
"""Load and cache local translation model from Hugging Face."""
|
| 34 |
+
global _LOCAL_MODEL_CACHE
|
| 35 |
+
if model_name in _LOCAL_MODEL_CACHE:
|
| 36 |
+
return _LOCAL_MODEL_CACHE[model_name]
|
| 37 |
+
|
| 38 |
+
print(f"📦 Loading local translation model: {model_name}...")
|
| 39 |
+
|
| 40 |
+
if "m2m100" in model_name:
|
| 41 |
+
tokenizer = M2M100Tokenizer.from_pretrained(model_name)
|
| 42 |
+
model = M2M100ForConditionalGeneration.from_pretrained(model_name)
|
| 43 |
+
else:
|
| 44 |
+
tokenizer = AutoTokenizer.from_pretrained(model_name)
|
| 45 |
+
model = AutoModelForSeq2SeqLM.from_pretrained(model_name)
|
| 46 |
+
|
| 47 |
+
device = "cuda" if torch.cuda.is_available() else "cpu"
|
| 48 |
+
model = model.to(device)
|
| 49 |
+
|
| 50 |
+
_LOCAL_MODEL_CACHE[model_name] = (model, tokenizer, device)
|
| 51 |
+
return model, tokenizer, device
|
| 52 |
+
|
| 53 |
+
def apply_vietnamese_style(text: str, style: str) -> str:
|
| 54 |
+
"""Adjust pronouns and word choices in Vietnamese text to match the requested style."""
|
| 55 |
+
if not text or style == "Mặc định":
|
| 56 |
+
return text
|
| 57 |
+
|
| 58 |
+
if style == "Trang trọng, lịch sự":
|
| 59 |
+
replacements = {
|
| 60 |
+
" tôi ": " chúng tôi ",
|
| 61 |
+
"Tôi ": "Chúng tôi ",
|
| 62 |
+
" tớ ": " chúng tôi ",
|
| 63 |
+
"Tớ ": "Chúng tôi ",
|
| 64 |
+
" mình ": " chúng tôi ",
|
| 65 |
+
"Mình ": "Chúng tôi ",
|
| 66 |
+
" cậu ": " quý khách ",
|
| 67 |
+
"Cậu ": "Quý khách ",
|
| 68 |
+
" mày ": " quý khách ",
|
| 69 |
+
"Mày ": "Quý khách ",
|
| 70 |
+
" bạn ": " quý khách ",
|
| 71 |
+
"Bạn ": "Quý khách ",
|
| 72 |
+
}
|
| 73 |
+
for k, v in replacements.items():
|
| 74 |
+
text = text.replace(k, v)
|
| 75 |
+
elif style == "Thân mật, gần gũi":
|
| 76 |
+
replacements = {
|
| 77 |
+
" tôi ": " mình ",
|
| 78 |
+
"Tôi ": "Mình ",
|
| 79 |
+
" chúng tôi ": " chúng mình ",
|
| 80 |
+
"Chúng tôi ": "Chúng mình ",
|
| 81 |
+
" quý khách ": " bạn ",
|
| 82 |
+
"Quý khách ": "Bạn ",
|
| 83 |
+
" ngài ": " bạn ",
|
| 84 |
+
"Ngài ": "Bạn ",
|
| 85 |
+
}
|
| 86 |
+
for k, v in replacements.items():
|
| 87 |
+
text = text.replace(k, v)
|
| 88 |
+
elif style == "Văn học, nghệ thuật":
|
| 89 |
+
replacements = {
|
| 90 |
+
" nhanh chóng ": " mau chóng ",
|
| 91 |
+
" rất ": " vô cùng ",
|
| 92 |
+
" đẹp ": " thơ mộng ",
|
| 93 |
+
" nói ": " bộc bạch ",
|
| 94 |
+
}
|
| 95 |
+
for k, v in replacements.items():
|
| 96 |
+
text = text.replace(k, v)
|
| 97 |
+
|
| 98 |
+
return text
|
| 99 |
+
|
| 100 |
+
def translate_text(text: str, target_lang: str, source_lang: str = "auto", style: str = "Mặc định") -> str:
|
| 101 |
+
"""Translate plain text locally in small chunks of 500 characters."""
|
| 102 |
+
if not text or not text.strip():
|
| 103 |
+
return ""
|
| 104 |
+
|
| 105 |
+
src = LANGUAGES.get(source_lang, "auto")
|
| 106 |
+
tgt = LANGUAGES.get(target_lang, "vi")
|
| 107 |
+
|
| 108 |
+
if src == "auto":
|
| 109 |
+
src = "en" # Fallback auto to English source
|
| 110 |
+
|
| 111 |
+
# Determine model name
|
| 112 |
+
if src == "en" and tgt == "vi":
|
| 113 |
+
model_name = "Helsinki-NLP/opus-mt-en-vi"
|
| 114 |
+
elif src == "vi" and tgt == "en":
|
| 115 |
+
model_name = "Helsinki-NLP/opus-mt-vi-en"
|
| 116 |
+
elif src == "zh-CN" and tgt == "vi":
|
| 117 |
+
model_name = "Helsinki-NLP/opus-mt-zh-vi"
|
| 118 |
+
elif src == "fr" and tgt == "vi":
|
| 119 |
+
model_name = "Helsinki-NLP/opus-mt-fr-vi"
|
| 120 |
+
elif src == "de" and tgt == "vi":
|
| 121 |
+
model_name = "Helsinki-NLP/opus-mt-de-vi"
|
| 122 |
+
else:
|
| 123 |
+
model_name = "facebook/m2m100_418M"
|
| 124 |
+
|
| 125 |
+
model, tokenizer, device = get_local_model_and_tokenizer(model_name)
|
| 126 |
+
|
| 127 |
+
# Split text into sentences or small chunks (approx 500 characters) to avoid generation cutoff
|
| 128 |
+
chunk_size = 500
|
| 129 |
+
paragraphs = text.split("\n")
|
| 130 |
+
translated_paragraphs = []
|
| 131 |
+
|
| 132 |
+
for para in paragraphs:
|
| 133 |
+
if not para.strip():
|
| 134 |
+
translated_paragraphs.append("")
|
| 135 |
+
continue
|
| 136 |
+
|
| 137 |
+
# Split paragraph into chunks of 500 chars
|
| 138 |
+
chunks = [para[i:i+chunk_size] for i in range(0, len(para), chunk_size)]
|
| 139 |
+
translated_chunks = []
|
| 140 |
+
|
| 141 |
+
for chunk in chunks:
|
| 142 |
+
if not chunk.strip():
|
| 143 |
+
translated_chunks.append(chunk)
|
| 144 |
+
continue
|
| 145 |
+
|
| 146 |
+
try:
|
| 147 |
+
if "m2m100" in model_name:
|
| 148 |
+
src_m2m = src.split("-")[0]
|
| 149 |
+
tgt_m2m = tgt.split("-")[0]
|
| 150 |
+
tokenizer.src_lang = src_m2m
|
| 151 |
+
inputs = tokenizer(chunk, return_tensors="pt").to(device)
|
| 152 |
+
with torch.no_grad():
|
| 153 |
+
generated_tokens = model.generate(**inputs, forced_bos_token_id=tokenizer.get_lang_id(tgt_m2m), max_length=512)
|
| 154 |
+
translated = tokenizer.batch_decode(generated_tokens, skip_special_tokens=True)[0]
|
| 155 |
+
else:
|
| 156 |
+
inputs = tokenizer(chunk, return_tensors="pt", padding=True, truncation=True, max_length=512).to(device)
|
| 157 |
+
with torch.no_grad():
|
| 158 |
+
outputs = model.generate(**inputs, max_length=512)
|
| 159 |
+
translated = tokenizer.decode(outputs[0], skip_special_tokens=True)
|
| 160 |
+
|
| 161 |
+
# Apply Vietnamese style styling if target is Vietnamese
|
| 162 |
+
if tgt == "vi":
|
| 163 |
+
translated = apply_vietnamese_style(translated, style)
|
| 164 |
+
|
| 165 |
+
translated_chunks.append(translated)
|
| 166 |
+
except Exception as e:
|
| 167 |
+
print(f"Error local translating chunk: {e}")
|
| 168 |
+
translated_chunks.append(chunk)
|
| 169 |
+
|
| 170 |
+
translated_paragraphs.append("".join(translated_chunks))
|
| 171 |
+
|
| 172 |
+
return "\n".join(translated_paragraphs)
|
| 173 |
+
|
| 174 |
+
def translate_docx(file_path: str, target_lang: str, source_lang: str = "auto", style: str = "Mặc định") -> str:
|
| 175 |
+
"""Translate a Word (.docx) document locally paragraph by paragraph."""
|
| 176 |
+
doc = docx.Document(file_path)
|
| 177 |
+
|
| 178 |
+
for i, p in enumerate(doc.paragraphs):
|
| 179 |
+
if p.text.strip():
|
| 180 |
+
try:
|
| 181 |
+
p.text = translate_text(p.text, target_lang, source_lang, style)
|
| 182 |
+
except Exception as e:
|
| 183 |
+
print(f"Error translating docx paragraph {i}: {e}")
|
| 184 |
+
|
| 185 |
+
for t in doc.tables:
|
| 186 |
+
for row in t.rows:
|
| 187 |
+
for cell in row.cells:
|
| 188 |
+
for p in cell.paragraphs:
|
| 189 |
+
if p.text.strip():
|
| 190 |
+
try:
|
| 191 |
+
p.text = translate_text(p.text, target_lang, source_lang, style)
|
| 192 |
+
except Exception as e:
|
| 193 |
+
print(f"Error translating table paragraph: {e}")
|
| 194 |
+
|
| 195 |
+
dir_name = os.path.dirname(file_path)
|
| 196 |
+
base_name = os.path.basename(file_path)
|
| 197 |
+
out_name = f"translated_{base_name}"
|
| 198 |
+
out_path = os.path.join(dir_name, out_name)
|
| 199 |
+
doc.save(out_path)
|
| 200 |
+
return out_path
|
| 201 |
+
|
| 202 |
+
def translate_pdf(file_path: str, target_lang: str, source_lang: str = "auto", style: str = "Mặc định") -> tuple[str, str]:
|
| 203 |
+
"""Extract text from PDF, translate locally, and save it as a new .docx file."""
|
| 204 |
+
reader = PdfReader(file_path)
|
| 205 |
+
text_content = []
|
| 206 |
+
|
| 207 |
+
for page in reader.pages:
|
| 208 |
+
page_text = page.extract_text()
|
| 209 |
+
if page_text:
|
| 210 |
+
text_content.append(page_text)
|
| 211 |
+
|
| 212 |
+
full_text = "\n\n".join(text_content)
|
| 213 |
+
translated_text = translate_text(full_text, target_lang, source_lang, style)
|
| 214 |
+
|
| 215 |
+
dir_name = os.path.dirname(file_path)
|
| 216 |
+
base_name = os.path.basename(file_path)
|
| 217 |
+
out_docx_name = f"translated_{os.path.splitext(base_name)[0]}.docx"
|
| 218 |
+
out_docx_path = os.path.join(dir_name, out_docx_name)
|
| 219 |
+
|
| 220 |
+
doc = docx.Document()
|
| 221 |
+
for paragraph in translated_text.split("\n\n"):
|
| 222 |
+
if paragraph.strip():
|
| 223 |
+
doc.add_paragraph(paragraph)
|
| 224 |
+
doc.save(out_docx_path)
|
| 225 |
+
|
| 226 |
+
return out_docx_path, translated_text
|
| 227 |
+
|
| 228 |
+
def translate_document(file_path: str, target_lang: str, source_lang: str = "auto", style: str = "Mặc định") -> tuple[str, str]:
|
| 229 |
+
"""Main function to parse extension and translate."""
|
| 230 |
+
if not file_path or not os.path.exists(file_path):
|
| 231 |
+
return None, "❌ File không tồn tại."
|
| 232 |
+
|
| 233 |
+
ext = os.path.splitext(file_path)[1].lower()
|
| 234 |
+
base_name = os.path.basename(file_path)
|
| 235 |
+
dir_name = os.path.dirname(file_path)
|
| 236 |
+
|
| 237 |
+
try:
|
| 238 |
+
if ext == ".docx":
|
| 239 |
+
out_path = translate_docx(file_path, target_lang, source_lang, style)
|
| 240 |
+
doc = docx.Document(out_path)
|
| 241 |
+
preview_text = "\n".join([p.text for p in doc.paragraphs[:10]])
|
| 242 |
+
return out_path, preview_text
|
| 243 |
+
|
| 244 |
+
elif ext == ".pdf":
|
| 245 |
+
out_path, translated_text = translate_pdf(file_path, target_lang, source_lang, style)
|
| 246 |
+
return out_path, translated_text[:1000]
|
| 247 |
+
|
| 248 |
+
elif ext in [".txt", ".md", ".json"]:
|
| 249 |
+
with open(file_path, "r", encoding="utf-8", errors="ignore") as f:
|
| 250 |
+
content = f.read()
|
| 251 |
+
translated_text = translate_text(content, target_lang, source_lang, style)
|
| 252 |
+
|
| 253 |
+
out_name = f"translated_{base_name}"
|
| 254 |
+
out_path = os.path.join(dir_name, out_name)
|
| 255 |
+
with open(out_path, "w", encoding="utf-8") as f:
|
| 256 |
+
f.write(translated_text)
|
| 257 |
+
|
| 258 |
+
return out_path, translated_text[:1000]
|
| 259 |
+
|
| 260 |
+
else:
|
| 261 |
+
return None, f"❌ Định dạng file {ext} không được hỗ trợ."
|
| 262 |
+
|
| 263 |
+
except Exception as e:
|
| 264 |
+
import traceback
|
| 265 |
+
traceback.print_exc()
|
| 266 |
+
return None, f"❌ Lỗi trong quá trình dịch: {str(e)}"
|
tools/TextToSpeech/app.py
CHANGED
|
@@ -1543,15 +1543,6 @@ with gr.Blocks(theme=theme, css=css, title="Pum's Tools — TextToSpeech", head=
|
|
| 1543 |
session_id_state = gr.State("")
|
| 1544 |
|
| 1545 |
with gr.Column(elem_classes="container"):
|
| 1546 |
-
gr.HTML("""
|
| 1547 |
-
<div class="header-box">
|
| 1548 |
-
<h1 class="header-title">
|
| 1549 |
-
<span class="header-icon">🦜</span>
|
| 1550 |
-
<span class="gradient-text">Pum's Tools</span>
|
| 1551 |
-
</h1>
|
| 1552 |
-
<p class="header-subtitle">🗣️ Text to Speech</p>
|
| 1553 |
-
</div>
|
| 1554 |
-
""")
|
| 1555 |
|
| 1556 |
# --- CONFIGURATION (Hidden from UI, but active in backend) ---
|
| 1557 |
with gr.Group(visible=False):
|
|
|
|
| 1543 |
session_id_state = gr.State("")
|
| 1544 |
|
| 1545 |
with gr.Column(elem_classes="container"):
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1546 |
|
| 1547 |
# --- CONFIGURATION (Hidden from UI, but active in backend) ---
|
| 1548 |
with gr.Group(visible=False):
|