cytopa99 commited on
Commit
59c0fa7
·
verified ·
1 Parent(s): 3c26d39

Upload 47 files

Browse files
Files changed (2) hide show
  1. README.md +77 -20
  2. backend/modules/stream_processor.py +139 -75
README.md CHANGED
@@ -114,6 +114,7 @@ print(f"TTS提供商: {config['tts_provider']}")
114
 
115
  ### 处理流程
116
 
 
117
  ```
118
  视频URL/录制音频
119
  ↓
@@ -130,6 +131,39 @@ print(f"TTS提供商: {config['tts_provider']}")
130
  配音音频输出
131
  ```
132
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
133
  ## 配置说明
134
 
135
  ### API 提供商选择
@@ -223,27 +257,50 @@ A:
223
 
224
  ```
225
  universal-fast-dubbing/
226
- ├── backend/ # Python后端服务
227
- │ ├── app.py # Gradio主入口
228
- │ ├── requirements.txt # Python依赖
229
- │ ├── packages.txt # 系统依赖 (ffmpeg)
230
- │ ├── modules/ # 核心处理模块
231
- │ │ ├── gateway.py # API网关
232
- │ │ ├── groq_client.py # Groq API客户端
233
- │ │ ├── siliconflow_client.py # SiliconFlow客户端
234
- │ │ ├── processor.py # 配音处理器
235
- │ │ ├── segmenter.py # 音频分段器
236
- │ │ ├── tts_generator.py # TTS生成器
237
- │ │ ├── audio_sync.py # 音频同步
238
- │ │ ├── router.py # API路由
239
- │ │ └── ...
240
- │ └── temp/ # 临时文件目录
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
241
  │
242
- └── extension/ # Chrome扩展(需单独安装)
243
- ├── manifest.json # 扩展配置
244
- ├── background/ # Background Service Worker
245
- ├── content/ # Content Scripts
246
- └── popup/ # Popup界面
 
247
  ```
248
 
249
  ## 开发指南
 
114
 
115
  ### 处理流程
116
 
117
+ #### 标准处理流程
118
  ```
119
  视频URL/录制音频
120
  ↓
 
131
  配音音频输出
132
  ```
133
 
134
+ #### 流式异步处理流程 (v3.1 新增)
135
+ ```
136
+ 扩展点击AI配音
137
+ ↓
138
+ Native Host 下载音频 (yt-dlp)
139
+ ↓
140
+ 扩展端上传到 HF 后端
141
+ ↓
142
+ SSE 流式处理 ←──────────────────┐
143
+ ↓ │
144
+ 语音识别 (Whisper V3 带时间戳) │ 实时进度反馈
145
+ ↓ │
146
+ 按时间戳智能分段 │
147
+ ↓ │
148
+ ┌─────────────────────────────┐ │
149
+ │ 并行处理每段: │ │
150
+ │ 翻译 (Llama 3) │ │
151
+ │ → TTS (Edge-TTS) │ │
152
+ │ → 音频同步对齐 │ │
153
+ │ → 分段音频输出 ────────────┼─┘
154
+ └─────────────────────────────┘
155
+ ↓
156
+ 第一段完成即开始播放 (目标 <30秒)
157
+ ↓
158
+ 隐藏遮罩,视频从头播放
159
+ ```
160
+
161
+ **流式处理优势:**
162
+ - 首段配音 30 秒内开始播放
163
+ - 实时进度反馈,用户体验更好
164
+ - 分段并行处理,整体速度更快
165
+ - SSE 连接稳定,兼容 HF Spaces 代理
166
+
167
  ## 配置说明
168
 
169
  ### API 提供商选择
 
257
 
258
  ```
259
  universal-fast-dubbing/
260
+ ├── app.py # FastAPI 主入口
261
+ ├── Dockerfile # Docker 构建配置
262
+ ├── DEPLOYMENT.md # 部署说明
263
+ │
264
+ ├── backend/ # Python 后端模块
265
+ │ ├── requirements.txt # Python 依赖
266
+ │ ├── packages.txt # 系统依赖 (ffmpeg)
267
+ │ ├── modules/ # 核心处理模块
268
+ │ │ ├── gateway.py # API 网关
269
+ │ │ ├── groq_client.py # Groq API 客户端 (ASR + LLM)
270
+ │ │ ├── siliconflow_client.py # SiliconFlow 客户端
271
+ │ │ ├── stream_processor.py # 流式异步处理器 (v3.1)
272
+ │ │ ├── processor.py # 配音处理器
273
+ │ │ ├── segmenter.py # 音频分段器
274
+ │ │ ├── tts_generator.py # Edge-TTS 生成器
275
+ │ │ ├── audio_sync.py # 音频同步引擎
276
+ │ │ ├── router.py # API 路由
277
+ │ │ ├── logging_config.py # 结构化日志
278
+ │ │ ├── performance_monitor.py # 性能监控
279
+ │ │ └── errors.py # 统一错误处理
280
+ │ └── temp/ # 临时文件目录
281
+ │
282
+ ├── templates/ # Jinja2 模板
283
+ │ └── index.html # Web 界面
284
+ │
285
+ ├── static/ # 静态资源
286
+ │ └── style.css # Tailwind CSS
287
+ │
288
+ ├── extension/ # Chrome 扩展
289
+ │ ├── manifest.json # 扩展配置 (Manifest V3)
290
+ │ ├── background/ # Background Service Worker
291
+ │ │ └── background.js # 消息处理、Native Messaging
292
+ │ ├── content/ # Content Scripts
293
+ │ │ └── dubbing-ui.js # 视频页面 UI 注入
294
+ │ ├── popup/ # Popup 界面
295
+ │ ├── options/ # 设置页面
296
+ │ └── icons/ # 扩展图标
297
  │
298
+ └── local_audio_service/ # Native Host 服务
299
+ ├── native_host.js # Node.js Native Messaging Host
300
+ ├── native_host.bat # Windows 启动脚本
301
+ ├── install.bat # 注册表安装脚本
302
+ ├── com.ufd.native.json # Native Host 配置
303
+ └── yt-dlp.exe # 视��下载工具
304
  ```
305
 
306
  ## 开发指南
backend/modules/stream_processor.py CHANGED
@@ -197,6 +197,14 @@ class StreamProcessor:
197
  logger.info(f"[{session_id}] 开始流式处理: {audio_path}")
198
 
199
  try:
 
 
 
 
 
 
 
 
200
  # 1. 发送初始进度
201
  yield {
202
  'type': 'progress',
@@ -213,18 +221,31 @@ class StreamProcessor:
213
  'message': '语音识别中...'
214
  }
215
 
216
- asr_result = await self._do_asr(audio_path, client_config)
 
 
 
 
 
 
 
 
217
 
218
  if not asr_result.get('segments'):
219
  yield {
220
  'type': 'error',
221
- 'message': '语音识别结果为空'
222
  }
223
  return
224
 
225
  source_language = asr_result.get('language', 'unknown')
226
  total_duration = asr_result.get('duration', 0)
227
 
 
 
 
 
 
228
  logger.info(
229
  f"[{session_id}] ASR完成: 语言={source_language}, "
230
  f"时长={total_duration:.1f}s, 片段数={len(asr_result['segments'])}"
@@ -243,6 +264,13 @@ class StreamProcessor:
243
  total_duration
244
  )
245
 
 
 
 
 
 
 
 
246
  logger.info(f"[{session_id}] 分组完成: {len(segment_groups)} 组")
247
 
248
  # 4. 流式处理每个分组
@@ -260,73 +288,88 @@ class StreamProcessor:
260
  'message': f'处理第 {group_index + 1}/{total_groups} 段...'
261
  }
262
 
263
- # 4.1 翻译当前分组
264
- translated_segments = await self._translate_segments(
265
- group,
266
- source_language,
267
- client_config
268
- )
269
-
270
- # 4.2 TTS 生成
271
- yield {
272
- 'type': 'progress',
273
- 'stage': ProcessingStage.TTS.value,
274
- 'progress': 25 + (group_index / total_groups) * 60 + 20,
275
- 'message': f'生成配音 {group_index + 1}/{total_groups}...'
276
- }
277
-
278
- tts_results = await self._generate_tts(
279
- translated_segments,
280
- client_config
281
- )
282
-
283
- # 4.3 音频同步
284
- yield {
285
- 'type': 'progress',
286
- 'stage': ProcessingStage.SYNC.value,
287
- 'progress': 25 + (group_index / total_groups) * 60 + 40,
288
- 'message': f'同步音频 {group_index + 1}/{total_groups}...'
289
- }
290
-
291
- synced_audio = await self._sync_audio(
292
- tts_results,
293
- translated_segments,
294
- group_end_time - group_start_time,
295
- client_config
296
- )
297
-
298
- # 4.4 读取音频数据
299
- audio_data = None
300
- if synced_audio and os.path.exists(synced_audio):
301
- with open(synced_audio, 'rb') as f:
302
- audio_data = f.read()
303
-
304
- # 4.5 输出分段结果
305
- processed_groups += 1
306
-
307
- yield {
308
- 'type': 'segment_ready',
309
- 'index': group_index,
310
- 'start_time': group_start_time,
311
- 'end_time': group_end_time,
312
- 'duration': group_end_time - group_start_time,
313
- 'audio_data': base64.b64encode(audio_data).decode('utf-8') if audio_data else None,
314
- 'segments': [
315
- {
316
- 'original': seg.get('text', ''),
317
- 'translated': seg.get('cn', ''),
318
- 'role': seg.get('role', 'MALE'),
319
- 'start': seg.get('start', 0),
320
- 'end': seg.get('end', 0)
 
 
 
 
 
321
  }
322
- for seg in translated_segments
323
- ]
324
- }
325
-
326
- logger.info(
327
- f"[{session_id}] 分组 {group_index + 1} 完成: "
328
- f"{group_start_time:.1f}s - {group_end_time:.1f}s"
329
- )
 
 
 
 
 
 
 
 
 
 
330
 
331
  # 5. 处理完成
332
  processing_time = time.time() - start_time
@@ -345,7 +388,7 @@ class StreamProcessor:
345
  )
346
 
347
  except Exception as e:
348
- logger.error(f"[{session_id}] 流式处理失败: {e}")
349
  yield {
350
  'type': 'error',
351
  'message': str(e)
@@ -360,20 +403,41 @@ class StreamProcessor:
360
  执行语音识别
361
 
362
  根据配置选择 Groq Whisper 或 SiliconFlow SenseVoice
 
363
  """
364
  # 获取 ASR 提供商配置
365
  asr_provider = 'groq' # 默认使用 Groq
366
  if client_config:
367
  asr_provider = client_config.get('asrProvider', 'groq')
368
 
 
369
  if asr_provider == 'siliconflow' and self.siliconflow_client:
370
- logger.info("使用 SiliconFlow SenseVoice 进行语音识别")
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
371
  return await self.siliconflow_client.transcribe(audio_path)
372
- elif self.groq_client:
373
- logger.info("使用 Groq Whisper V3 进行语音识别")
374
- return await self.groq_client.transcribe(audio_path)
375
- else:
376
- raise RuntimeError("没有可用的 ASR 服务")
377
 
378
  def _group_segments_for_streaming(
379
  self,
 
197
  logger.info(f"[{session_id}] 开始流式处理: {audio_path}")
198
 
199
  try:
200
+ # 检查音频文件是否存在
201
+ if not os.path.exists(audio_path):
202
+ yield {
203
+ 'type': 'error',
204
+ 'message': f'音频文件不存在: {audio_path}'
205
+ }
206
+ return
207
+
208
  # 1. 发送初始进度
209
  yield {
210
  'type': 'progress',
 
221
  'message': '语音识别中...'
222
  }
223
 
224
+ try:
225
+ asr_result = await self._do_asr(audio_path, client_config)
226
+ except Exception as e:
227
+ logger.error(f"[{session_id}] ASR 失败: {e}")
228
+ yield {
229
+ 'type': 'error',
230
+ 'message': f'语音识别失败: {str(e)}'
231
+ }
232
+ return
233
 
234
  if not asr_result.get('segments'):
235
  yield {
236
  'type': 'error',
237
+ 'message': '语音识别结果为空,请检查音频文件'
238
  }
239
  return
240
 
241
  source_language = asr_result.get('language', 'unknown')
242
  total_duration = asr_result.get('duration', 0)
243
 
244
+ # 如果没有时长信息,从最后一个片段获取
245
+ if total_duration == 0 and asr_result['segments']:
246
+ last_seg = asr_result['segments'][-1]
247
+ total_duration = last_seg.get('end', 0)
248
+
249
  logger.info(
250
  f"[{session_id}] ASR完成: 语言={source_language}, "
251
  f"时长={total_duration:.1f}s, 片段数={len(asr_result['segments'])}"
 
264
  total_duration
265
  )
266
 
267
+ if not segment_groups:
268
+ yield {
269
+ 'type': 'error',
270
+ 'message': '音频分段失败,无有效片段'
271
+ }
272
+ return
273
+
274
  logger.info(f"[{session_id}] 分组完成: {len(segment_groups)} 组")
275
 
276
  # 4. 流式处理每个分组
 
288
  'message': f'处理第 {group_index + 1}/{total_groups} 段...'
289
  }
290
 
291
+ try:
292
+ # 4.1 翻译当前分组
293
+ translated_segments = await self._translate_segments(
294
+ group,
295
+ source_language,
296
+ client_config
297
+ )
298
+
299
+ # 4.2 TTS 生成
300
+ yield {
301
+ 'type': 'progress',
302
+ 'stage': ProcessingStage.TTS.value,
303
+ 'progress': 25 + (group_index / total_groups) * 60 + 20,
304
+ 'message': f'生成配音 {group_index + 1}/{total_groups}...'
305
+ }
306
+
307
+ tts_results = await self._generate_tts(
308
+ translated_segments,
309
+ client_config
310
+ )
311
+
312
+ # 4.3 音频同步
313
+ yield {
314
+ 'type': 'progress',
315
+ 'stage': ProcessingStage.SYNC.value,
316
+ 'progress': 25 + (group_index / total_groups) * 60 + 40,
317
+ 'message': f'同步音频 {group_index + 1}/{total_groups}...'
318
+ }
319
+
320
+ synced_audio = await self._sync_audio(
321
+ tts_results,
322
+ translated_segments,
323
+ group_end_time - group_start_time,
324
+ client_config
325
+ )
326
+
327
+ # 4.4 读取音频数据
328
+ audio_data = None
329
+ if synced_audio and os.path.exists(synced_audio):
330
+ with open(synced_audio, 'rb') as f:
331
+ audio_data = f.read()
332
+
333
+ # 4.5 输出分段结果
334
+ processed_groups += 1
335
+
336
+ if audio_data:
337
+ yield {
338
+ 'type': 'segment_ready',
339
+ 'index': group_index,
340
+ 'start_time': group_start_time,
341
+ 'end_time': group_end_time,
342
+ 'duration': group_end_time - group_start_time,
343
+ 'audio_data': base64.b64encode(audio_data).decode('utf-8'),
344
+ 'segments': [
345
+ {
346
+ 'original': seg.get('text', ''),
347
+ 'translated': seg.get('cn', ''),
348
+ 'role': seg.get('role', 'MALE'),
349
+ 'start': seg.get('start', 0),
350
+ 'end': seg.get('end', 0)
351
+ }
352
+ for seg in translated_segments
353
+ ]
354
  }
355
+
356
+ logger.info(
357
+ f"[{session_id}] 分组 {group_index + 1} 完成: "
358
+ f"{group_start_time:.1f}s - {group_end_time:.1f}s, "
359
+ f"音频大小: {len(audio_data)} bytes"
360
+ )
361
+ else:
362
+ logger.warning(f"[{session_id}] 分组 {group_index + 1} 无音频输出")
363
+
364
+ except Exception as e:
365
+ logger.error(f"[{session_id}] 分组 {group_index + 1} 处理失败: {e}")
366
+ # 继续处理下一个分组,不中断整个流程
367
+ yield {
368
+ 'type': 'progress',
369
+ 'stage': ProcessingStage.ERROR.value,
370
+ 'progress': 25 + (group_index / total_groups) * 60,
371
+ 'message': f'分组 {group_index + 1} 处理失败,继续下一段...'
372
+ }
373
 
374
  # 5. 处理完成
375
  processing_time = time.time() - start_time
 
388
  )
389
 
390
  except Exception as e:
391
+ logger.error(f"[{session_id}] 流式处理失败: {e}", exc_info=True)
392
  yield {
393
  'type': 'error',
394
  'message': str(e)
 
403
  执行语音识别
404
 
405
  根据配置选择 Groq Whisper 或 SiliconFlow SenseVoice
406
+ 如果两者都不可用,返回错误
407
  """
408
  # 获取 ASR 提供商配置
409
  asr_provider = 'groq' # 默认使用 Groq
410
  if client_config:
411
  asr_provider = client_config.get('asrProvider', 'groq')
412
 
413
+ # 尝试使用配置的提供商
414
  if asr_provider == 'siliconflow' and self.siliconflow_client:
415
+ try:
416
+ logger.info("使用 SiliconFlow SenseVoice 进行语音识别")
417
+ return await self.siliconflow_client.transcribe(audio_path)
418
+ except Exception as e:
419
+ logger.warning(f"SiliconFlow ASR 失败: {e},尝试回退到 Groq")
420
+ if self.groq_client:
421
+ return await self.groq_client.transcribe(audio_path)
422
+ raise
423
+
424
+ if self.groq_client:
425
+ try:
426
+ logger.info("使用 Groq Whisper V3 进行语音识别")
427
+ return await self.groq_client.transcribe(audio_path)
428
+ except Exception as e:
429
+ logger.warning(f"Groq ASR 失败: {e},尝试回退到 SiliconFlow")
430
+ if self.siliconflow_client:
431
+ return await self.siliconflow_client.transcribe(audio_path)
432
+ raise
433
+
434
+ if self.siliconflow_client:
435
+ logger.info("使用 SiliconFlow SenseVoice 进行语音识别(Groq 不可用)")
436
  return await self.siliconflow_client.transcribe(audio_path)
437
+
438
+ raise RuntimeError(
439
+ "没有可用的 ASR 服务。请配置 GROQ_API_KEY 或 SILICONFLOW_API_KEY 环境变量。"
440
+ )
 
441
 
442
  def _group_segments_for_streaming(
443
  self,