ngocdang83 commited on
Commit
27992ec
·
verified ·
1 Parent(s): b9a1eb1

multi-GPU: translator.py

Browse files
Files changed (1) hide show
  1. src/translator.py +21 -4
src/translator.py CHANGED
@@ -15,7 +15,7 @@ import sentencepiece as spm
15
  from huggingface_hub import snapshot_download
16
 
17
  from chunker import split_chunks
18
- from hardware import HardwareProfile, detect_hardware_profile
19
  from token_chunker import source_token_ids, split_for_translation
20
 
21
  ROOT = Path(__file__).resolve().parent.parent
@@ -560,16 +560,33 @@ class HachimiTranslator:
560
  elif env_compute_type and ct2_device == "cuda":
561
  attempts.append(("cpu", "int8_float32"))
562
 
 
 
 
 
 
 
 
 
 
 
563
  translator = None
564
  last_error: Exception | None = None
565
  for device, compute_type in attempts:
566
  try:
567
- translator = ctranslate2.Translator(
568
- str(model_path / config.ct2_subdir),
569
  device=device,
570
  compute_type=compute_type,
571
  intra_threads=self._ct2_threads,
572
- inter_threads=self._ct2_inter_threads,
 
 
 
 
 
 
 
 
573
  )
574
  self._ct2_compute_type = compute_type
575
  break
 
15
  from huggingface_hub import snapshot_download
16
 
17
  from chunker import split_chunks
18
+ from hardware import HardwareProfile, detect_hardware_profile, resolve_gpu_indices
19
  from token_chunker import source_token_ids, split_for_translation
20
 
21
  ROOT = Path(__file__).resolve().parent.parent
 
560
  elif env_compute_type and ct2_device == "cuda":
561
  attempts.append(("cpu", "int8_float32"))
562
 
563
+ # Multi-GPU: nếu có >1 GPU (vd Kaggle T4x2) → CT2 chia batch ra các GPU
564
+ # (~1.69× compute đo thực). Chỉ áp cho device cuda; CPU bỏ qua device_index.
565
+ try:
566
+ cuda_count = ctranslate2.get_cuda_device_count()
567
+ except Exception:
568
+ cuda_count = 1
569
+ gpu_indices = resolve_gpu_indices(
570
+ cuda_count, os.environ.get("HACHIMIMT_GPU_INDICES")
571
+ )
572
+
573
  translator = None
574
  last_error: Exception | None = None
575
  for device, compute_type in attempts:
576
  try:
577
+ kwargs = dict(
 
578
  device=device,
579
  compute_type=compute_type,
580
  intra_threads=self._ct2_threads,
581
+ )
582
+ if device == "cuda" and len(gpu_indices) > 1:
583
+ # 1 worker/GPU (CT2 khuyến nghị) + gán nhiều device.
584
+ kwargs["device_index"] = gpu_indices
585
+ kwargs["inter_threads"] = len(gpu_indices)
586
+ else:
587
+ kwargs["inter_threads"] = self._ct2_inter_threads
588
+ translator = ctranslate2.Translator(
589
+ str(model_path / config.ct2_subdir), **kwargs
590
  )
591
  self._ct2_compute_type = compute_type
592
  break