ndmhung6 commited on
Commit
0e1c5e3
·
1 Parent(s): 6cb4903

feat: add doc reader, global dark mode and fix marian translator tokenizer

Browse files
app.py CHANGED
@@ -7,6 +7,7 @@ sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
7
 
8
  from tools.TextToSpeech.app import demo as tts_demo
9
  from tools.DocumentTranslator.app import demo as trans_demo
 
10
  from tools.TextToSpeech.ui_constants import theme, css, head_html
11
 
12
  # Create unified block interface
@@ -27,6 +28,21 @@ with gr.Blocks(theme=theme, css=css, title="Pum's Tools", head=head_html) as dem
27
 
28
  with gr.Tab("📄 Dịch tài liệu"):
29
  trans_demo.render()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
30
 
31
  def main():
32
  server_name = os.getenv("GRADIO_SERVER_NAME", "0.0.0.0" if ("SPACE_ID" in os.environ or "PORT" in os.environ) else "127.0.0.1")
 
7
 
8
  from tools.TextToSpeech.app import demo as tts_demo
9
  from tools.DocumentTranslator.app import demo as trans_demo
10
+ from tools.DocReader.app import demo as doc_reader_demo, voice_components, update_voices_fn
11
  from tools.TextToSpeech.ui_constants import theme, css, head_html
12
 
13
  # Create unified block interface
 
28
 
29
  with gr.Tab("📄 Dịch tài liệu"):
30
  trans_demo.render()
31
+
32
+ with gr.Tab("📖 Đọc tài liệu") as doc_reader_tab:
33
+ doc_reader_demo.render()
34
+
35
+ doc_reader_tab.select(
36
+ fn=update_voices_fn,
37
+ outputs=voice_components
38
+ )
39
+
40
+ # Enforce dark mode and initial voice choices load
41
+ demo.load(
42
+ fn=update_voices_fn,
43
+ outputs=voice_components,
44
+ js="() => { document.documentElement.classList.add('dark'); }"
45
+ )
46
 
47
  def main():
48
  server_name = os.getenv("GRADIO_SERVER_NAME", "0.0.0.0" if ("SPACE_ID" in os.environ or "PORT" in os.environ) else "127.0.0.1")
tools/DocReader/__init__.py ADDED
@@ -0,0 +1 @@
 
 
1
+ # DocReader Package
tools/DocReader/app.py ADDED
@@ -0,0 +1,244 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import sys
3
+ import gradio as gr
4
+ from tools.DocReader.text_extractor import extract_text_from_file
5
+
6
+ # Heuristic offline language detection
7
+ def detect_language(text: str) -> str:
8
+ if not text:
9
+ return "vi"
10
+ vi_chars = set("àáảãạăằắẳẵặâầấẩẫậèéẻẽẹêềếểễệìíỉĩịòóỏõọôồốổỗộơờớởỡợùúủũụưừứửữựỳýỷỹỵđ"
11
+ "ÀÁẢÃẠĂẰẮẰẶÂẦẤẨẪẬÈÉẺẼẸÊỀẾỂỄỆÌÍỈĨỊÒÓỎÕỌÔỒỐỔỖỘƠỜỚỞỠỢÙÚỦŨỤƯỪỨỬỮỰỲÝỶỸỲĐ")
12
+ vi_count = sum(1 for c in text if c in vi_chars)
13
+ vi_words = {"và", "là", "của", "được", "người", "có", "không", "trong", "một", "cho", "đến", "với", "những", "này", "theo", "đã", "để"}
14
+ text_words = set(text.lower().split())
15
+ vi_word_count = len(text_words.intersection(vi_words))
16
+ en_words = {"the", "and", "of", "to", "in", "is", "that", "it", "he", "was", "for", "on", "are", "as", "with", "his", "they", "at"}
17
+ en_word_count = len(text_words.intersection(en_words))
18
+ if vi_count > 0 or vi_word_count > en_word_count:
19
+ return "vi"
20
+ else:
21
+ return "en"
22
+
23
+ def get_synthesis_language(text, language_choice):
24
+ if language_choice == "Tự động phát hiện":
25
+ lang_code = detect_language(text)
26
+ return lang_code, f"Tự động phát hiện ({'Tiếng Việt' if lang_code == 'vi' else 'Tiếng Anh'})"
27
+ else:
28
+ lang_code = "vi" if language_choice == "Tiếng Việt" else "en"
29
+ return lang_code, language_choice
30
+
31
+ def handle_file_upload(file):
32
+ if file is None:
33
+ return "", gr.update(visible=False)
34
+ try:
35
+ text = extract_text_from_file(file.name)
36
+ lang_code = detect_language(text)
37
+ lang_name = "Tiếng Việt" if lang_code == "vi" else "Tiếng Anh"
38
+ detect_msg = f"ℹ️ **Đã tự động phát hiện ngôn ngữ:** {lang_name}"
39
+ return text, gr.update(value=detect_msg, visible=True)
40
+ except Exception as e:
41
+ return f"Lỗi đọc file: {str(e)}", gr.update(value=f"❌ Lỗi: {str(e)}", visible=True)
42
+
43
+ def handle_single_read(text, voice, language, session_id):
44
+ from tools.TextToSpeech.app import synthesize_speech, model_loaded
45
+ if not model_loaded:
46
+ yield None, "⚠️ Vui lòng quay lại tab 'Text to Speech' để nạp Model trước khi đọc tài liệu!"
47
+ return
48
+ lang_code, lang_name = get_synthesis_language(text, language)
49
+ yield None, f"⏳ Đang chuẩn bị đọc bằng {lang_name}..."
50
+ for audio_path, status in synthesize_speech(
51
+ text, voice, None, None,
52
+ "preset_mode", "Standard (Một lần)", True, 32,
53
+ 0.8, 250, session_id
54
+ ):
55
+ yield audio_path, status
56
+
57
+ def handle_conversation_read(text, language, silence_dur, session_id, *speaker_args):
58
+ from tools.TextToSpeech.app import synthesize_conversation, model_loaded
59
+ if not model_loaded:
60
+ yield None, "⚠️ Vui lòng quay lại tab 'Text to Speech' để nạp Model trước khi đọc tài liệu!"
61
+ return
62
+ # First 8 args are speaker names, next 8 are voices
63
+ speaker_names = list(speaker_args[:8])
64
+ speaker_voices = list(speaker_args[8:16])
65
+
66
+ yield None, "⏳ Đang chuẩn bị đọc hội thoại..."
67
+ for audio_path, status in synthesize_conversation(
68
+ text,
69
+ *speaker_names,
70
+ *speaker_voices,
71
+ silence_dur, 0.8, 250, session_id
72
+ ):
73
+ yield audio_path, status
74
+
75
+ def handle_read(mode, text, voice, language, silence_dur, session_id, *speaker_args):
76
+ if not text or not text.strip():
77
+ yield None, "⚠️ Vui lòng nhập hoặc tải lên văn bản!"
78
+ return
79
+ if mode == "Đọc truyện (Đơn ca)":
80
+ yield from handle_single_read(text, voice, language, session_id)
81
+ else:
82
+ yield from handle_conversation_read(text, language, silence_dur, session_id, *speaker_args)
83
+
84
+ def get_voices_update_func():
85
+ from tools.TextToSpeech.app import PRESET_VOICES_CACHE, CONV_VOICES_CACHE
86
+ if not PRESET_VOICES_CACHE:
87
+ return [gr.update()] * (1 + 8)
88
+ default_v = PRESET_VOICES_CACHE[0][1] if isinstance(PRESET_VOICES_CACHE[0], tuple) else PRESET_VOICES_CACHE[0]
89
+ conv_default = CONV_VOICES_CACHE[0][1] if (CONV_VOICES_CACHE and isinstance(CONV_VOICES_CACHE[0], tuple)) else (CONV_VOICES_CACHE[0] if CONV_VOICES_CACHE else None)
90
+ return [
91
+ gr.update(choices=PRESET_VOICES_CACHE, value=default_v),
92
+ *[gr.update(choices=CONV_VOICES_CACHE, value=conv_default) for _ in range(8)]
93
+ ]
94
+
95
+ def on_mode_change(mode):
96
+ is_conv = mode == "Hội thoại (Đa ca)"
97
+ return gr.update(visible=not is_conv), gr.update(visible=is_conv)
98
+
99
+ def request_stop():
100
+ from tools.TextToSpeech.app import _STOP_EVENT
101
+ print("🛑 STOP REQUESTED via DocReader button click.")
102
+ _STOP_EVENT.set()
103
+ return None, "⏹️ Đã dừng đọc tài liệu.", gr.update(interactive=False)
104
+
105
+ # Setup layout
106
+ MAX_SPEAKERS = 8
107
+
108
+ with gr.Blocks() as demo:
109
+ session_id_state = gr.State("")
110
+ gr.HTML("<div style='margin-bottom: 20px;'><h2>📖 Đọc tài liệu & Kịch bản hội thoại</h2><p style='color: #64748b; margin-top: 5px;'>Trích xuất tài liệu .txt, .docx để đọc hoặc phát kịch bản hội thoại đa nhân vật sử dụng mô hình VieNeu-TTS chạy cục bộ.</p></div>")
111
+
112
+ with gr.Row():
113
+ with gr.Column(scale=3):
114
+ file_input = gr.File(
115
+ label="Tải lên tài liệu (.txt, .docx)",
116
+ file_types=[".txt", ".docx"]
117
+ )
118
+ detected_lang_label = gr.Markdown(value="", visible=False)
119
+
120
+ text_input = gr.Textbox(
121
+ lines=12,
122
+ label="Nội dung văn bản (Có thể chỉnh sửa)",
123
+ placeholder="Nội dung tài liệu sẽ hiển thị ở đây sau khi tải lên, hoặc bạn tự nhập nội dung..."
124
+ )
125
+
126
+ with gr.Row():
127
+ language_select = gr.Dropdown(
128
+ choices=["Tự động phát hiện", "Tiếng Việt", "Tiếng Anh"],
129
+ value="Tự động phát hiện",
130
+ label="Ngôn ngữ đọc"
131
+ )
132
+ mode_select = gr.Radio(
133
+ choices=["Đọc truyện (Đơn ca)", "Hội thoại (Đa ca)"],
134
+ value="Đọc truyện (Đơn ca)",
135
+ label="Chế độ đọc"
136
+ )
137
+
138
+ # Single Speaker Voice settings
139
+ with gr.Group(visible=True) as single_voice_group:
140
+ voice_select = gr.Dropdown(
141
+ choices=[],
142
+ value=None,
143
+ label="🎤 Giọng đọc truyện",
144
+ interactive=True,
145
+ allow_custom_value=True
146
+ )
147
+
148
+ # Conversation Settings
149
+ with gr.Group(visible=False) as multi_voice_group:
150
+ with gr.Row():
151
+ btn_detect_speakers = gr.Button("🔍 Quét nhân vật", variant="secondary")
152
+ silence_slider = gr.Slider(minimum=0, maximum=3, value=0.5, step=0.1, label="⏱️ Khoảng lặng giữa các câu (giây)")
153
+
154
+ gr.Markdown("### 🎭 Cấu hình giọng nhân vật")
155
+ speaker_name_boxes = []
156
+ speaker_voice_dds = []
157
+ speaker_slot_rows = []
158
+
159
+ for i in range(MAX_SPEAKERS):
160
+ row_visible = i < 3
161
+ default_name = ""
162
+ if i == 0: default_name = "Phương"
163
+ elif i == 1: default_name = "Dũng"
164
+ elif i == 2: default_name = "Hùng"
165
+
166
+ with gr.Row(visible=row_visible) as row:
167
+ name = gr.Textbox(
168
+ value=default_name,
169
+ label="👤 Nhân vật",
170
+ interactive=False,
171
+ scale=1,
172
+ min_width=120
173
+ )
174
+ dd = gr.Dropdown(
175
+ choices=[],
176
+ value=None,
177
+ label="🎤 Giọng đọc",
178
+ interactive=True,
179
+ scale=3,
180
+ allow_custom_value=True
181
+ )
182
+ speaker_slot_rows.append(row)
183
+ speaker_name_boxes.append(name)
184
+ speaker_voice_dds.append(dd)
185
+
186
+ btn_generate = gr.Button("🚀 Bắt đầu đọc tài liệu", variant="primary")
187
+ btn_stop = gr.Button("🛑 Dừng", variant="stop", interactive=False)
188
+
189
+ with gr.Column(scale=2):
190
+ status_output = gr.Textbox(
191
+ label="Trạng thái",
192
+ value="⏳ Chờ tải tài liệu hoặc nhập văn bản...",
193
+ interactive=False
194
+ )
195
+ audio_output = gr.Audio(
196
+ label="Âm thanh đầu ra",
197
+ interactive=False
198
+ )
199
+
200
+ file_input.change(
201
+ fn=handle_file_upload,
202
+ inputs=[file_input],
203
+ outputs=[text_input, detected_lang_label]
204
+ )
205
+
206
+ mode_select.change(
207
+ fn=on_mode_change,
208
+ inputs=[mode_select],
209
+ outputs=[single_voice_group, multi_voice_group]
210
+ )
211
+
212
+ from tools.TextToSpeech.app import extract_speakers_from_script
213
+ btn_detect_speakers.click(
214
+ fn=extract_speakers_from_script,
215
+ inputs=[text_input],
216
+ outputs=speaker_name_boxes + speaker_voice_dds + speaker_slot_rows
217
+ )
218
+
219
+ # Read generation
220
+ gen_inputs = [
221
+ mode_select,
222
+ text_input,
223
+ voice_select,
224
+ language_select,
225
+ silence_slider,
226
+ session_id_state,
227
+ *speaker_name_boxes,
228
+ *speaker_voice_dds
229
+ ]
230
+
231
+ gen_event = btn_generate.click(
232
+ fn=handle_read,
233
+ inputs=gen_inputs,
234
+ outputs=[audio_output, status_output]
235
+ )
236
+
237
+ btn_generate.click(lambda: gr.update(interactive=True), outputs=btn_stop)
238
+ gen_event.then(lambda: gr.update(interactive=False), outputs=btn_stop)
239
+
240
+ btn_stop.click(fn=request_stop, outputs=[audio_output, status_output, btn_stop])
241
+
242
+ # Expose components and update function
243
+ voice_components = [voice_select] + speaker_voice_dds
244
+ update_voices_fn = get_voices_update_func
tools/DocReader/text_extractor.py ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import os
2
+ import docx
3
+
4
+ def read_txt_file(file_path: str) -> str:
5
+ """Read a text file with fallback encodings to support Vietnamese."""
6
+ encodings = ['utf-8', 'utf-16', 'utf-16-le', 'utf-16-be', 'utf-8-sig', 'latin-1', 'cp1258']
7
+ for enc in encodings:
8
+ try:
9
+ with open(file_path, 'r', encoding=enc) as f:
10
+ return f.read()
11
+ except UnicodeDecodeError:
12
+ continue
13
+ # Fallback to ignore errors
14
+ with open(file_path, 'r', encoding='utf-8', errors='ignore') as f:
15
+ return f.read()
16
+
17
+ def read_docx_file(file_path: str) -> str:
18
+ """Read paragraphs and tables from a DOCX file."""
19
+ doc = docx.Document(file_path)
20
+
21
+ # Extract paragraphs
22
+ paragraphs = [p.text for p in doc.paragraphs]
23
+
24
+ # Extract tables
25
+ for table in doc.tables:
26
+ for row in table.rows:
27
+ row_text = []
28
+ for cell in row.cells:
29
+ cell_text = cell.text.strip()
30
+ if cell_text and cell_text not in row_text:
31
+ row_text.append(cell_text)
32
+ if row_text:
33
+ paragraphs.append(" | ".join(row_text))
34
+
35
+ return "\n".join(paragraphs)
36
+
37
+ def extract_text_from_file(file_path: str) -> str:
38
+ """Detect file extension and extract text content."""
39
+ if not file_path or not os.path.exists(file_path):
40
+ raise FileNotFoundError(f"Không tìm thấy file: {file_path}")
41
+
42
+ ext = os.path.splitext(file_path)[1].lower()
43
+
44
+ if ext == ".txt":
45
+ return read_txt_file(file_path)
46
+ elif ext == ".docx":
47
+ return read_docx_file(file_path)
48
+ else:
49
+ raise ValueError(f"Định dạng file {ext} không được hỗ trợ. Chỉ hỗ trợ .txt và .docx.")
tools/DocumentTranslator/translator.py CHANGED
@@ -41,7 +41,19 @@ def get_local_model_and_tokenizer(model_name: str):
41
  tokenizer = M2M100Tokenizer.from_pretrained(model_name)
42
  model = M2M100ForConditionalGeneration.from_pretrained(model_name)
43
  else:
44
- tokenizer = AutoTokenizer.from_pretrained(model_name)
 
 
 
 
 
 
 
 
 
 
 
 
45
  model = AutoModelForSeq2SeqLM.from_pretrained(model_name)
46
 
47
  device = "cuda" if torch.cuda.is_available() else "cpu"
 
41
  tokenizer = M2M100Tokenizer.from_pretrained(model_name)
42
  model = M2M100ForConditionalGeneration.from_pretrained(model_name)
43
  else:
44
+ try:
45
+ if "opus-mt" in model_name or "helsinki" in model_name.lower():
46
+ from transformers import MarianTokenizer
47
+ tokenizer = MarianTokenizer.from_pretrained(model_name)
48
+ else:
49
+ tokenizer = AutoTokenizer.from_pretrained(model_name)
50
+ except Exception as e:
51
+ print(f"⚠️ Cảnh báo: AutoTokenizer thất bại, chuyển sang MarianTokenizer cho {model_name}: {e}")
52
+ try:
53
+ from transformers import MarianTokenizer
54
+ tokenizer = MarianTokenizer.from_pretrained(model_name)
55
+ except Exception:
56
+ tokenizer = AutoTokenizer.from_pretrained(model_name)
57
  model = AutoModelForSeq2SeqLM.from_pretrained(model_name)
58
 
59
  device = "cuda" if torch.cuda.is_available() else "cpu"
tools/TextToSpeech/ui_constants.py CHANGED
@@ -11,6 +11,10 @@ theme = gr.themes.Soft(
11
  )
12
 
13
  css = """
 
 
 
 
14
  .container { max-width: 1400px; margin: auto; padding: 15px 0; }
15
  .header-box {
16
  text-align: center;
@@ -196,6 +200,9 @@ button.stop:hover:not(:disabled) {
196
 
197
  head_html = """
198
  <link rel="icon" href="data:image/svg+xml,<svg xmlns=%22http://www.w3.org/2000/svg%22 viewBox=%220 0 100 100%22><text y=%22.9em%22 font-size=%2290%22>🦜</text></svg>">
 
 
 
199
  """
200
 
201
  DEFAULT_TEXT_GPU = "Hà Nội, trái tim của Việt Nam, là một thành phố ngàn năm văn hiến với bề dày lịch sử và văn hóa độc đáo. Bước chân trên những con phố cổ kính quanh Hồ Hoàn Kiếm, du khách như được du hành ngược thời gian, chiêm ngưỡng kiến trúc Pháp cổ điển hòa quyện với nét kiến trúc truyền thống Việt Nam. Mỗi con phố trong khu phố cổ mang một tên gọi đặc trưng, phản ánh nghề thủ công truyền thống từng thịnh hành nơi đây như phố Hàng Bạc, Hàng Đào, Hàng Mã. Ẩm thực Hà Nội cũng là một điểm nhấn đặc biệt, từ tô phở nóng hổi buổi sáng, bún chả thơm lừng trưa hè, đến chè Thái ngọt ngào chiều thu. Những món ăn dân dã này đã trở thành biểu tượng của văn hóa ẩm thực Việt, được cả thế giới yêu mến. Người Hà Nội nổi tiếng với tính cách hiền hòa, lịch thiệp nhưng cũng rất cầu toàn trong từng chi tiết nhỏ, từ cách pha trà sen cho đến cách chọn hoa sen tây để thưởng trà."
 
11
  )
12
 
13
  css = """
14
+ html, body, .gradio-container {
15
+ background-color: #0b0f19 !important;
16
+ color: #f8fafc !important;
17
+ }
18
  .container { max-width: 1400px; margin: auto; padding: 15px 0; }
19
  .header-box {
20
  text-align: center;
 
200
 
201
  head_html = """
202
  <link rel="icon" href="data:image/svg+xml,<svg xmlns=%22http://www.w3.org/2000/svg%22 viewBox=%220 0 100 100%22><text y=%22.9em%22 font-size=%2290%22>🦜</text></svg>">
203
+ <script>
204
+ document.documentElement.classList.add('dark');
205
+ </script>
206
  """
207
 
208
  DEFAULT_TEXT_GPU = "Hà Nội, trái tim của Việt Nam, là một thành phố ngàn năm văn hiến với bề dày lịch sử và văn hóa độc đáo. Bước chân trên những con phố cổ kính quanh Hồ Hoàn Kiếm, du khách như được du hành ngược thời gian, chiêm ngưỡng kiến trúc Pháp cổ điển hòa quyện với nét kiến trúc truyền thống Việt Nam. Mỗi con phố trong khu phố cổ mang một tên gọi đặc trưng, phản ánh nghề thủ công truyền thống từng thịnh hành nơi đây như phố Hàng Bạc, Hàng Đào, Hàng Mã. Ẩm thực Hà Nội cũng là một điểm nhấn đặc biệt, từ tô phở nóng hổi buổi sáng, bún chả thơm lừng trưa hè, đến chè Thái ngọt ngào chiều thu. Những món ăn dân dã này đã trở thành biểu tượng của văn hóa ẩm thực Việt, được cả thế giới yêu mến. Người Hà Nội nổi tiếng với tính cách hiền hòa, lịch thiệp nhưng cũng rất cầu toàn trong từng chi tiết nhỏ, từ cách pha trà sen cho đến cách chọn hoa sen tây để thưởng trà."