""" app/core/studio_qa.py ───────────────────── Self-refining QA loop (retention tier 5, lightweight). After TTS synthesis, this module reads tts_segments_manifest.json and flags every segment whose effective speedup_ratio exceeds the "comfort" threshold (1.3x). For those segments it asks the LLM (via the existing translation manager) to compress the Vietnamese line to fit the original timeline, rewrites the SRT, and signals the pipeline to re-run TTS — up to `max_rounds` rounds. """ import json import re import time from pathlib import Path COMFORT_SPEEDUP_RATIO = 1.3 def _srt_to_blocks(srt_path): srt_path = Path(srt_path) if not srt_path.exists(): return [] content = srt_path.read_text(encoding="utf-8", errors="ignore").replace("\r\n", "\n") pattern = ( r"(\d+)\s+(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*-->\s*(\d{2}:\d{2}:\d{2}[.,]\d{3})\s*\n" r"(.*?)(?=\n\s*\d+\s+\d{2}:\d{2}:\d{2}[.,]\d{3}\s*-->|\Z)" ) blocks = [] for m in re.finditer(pattern, content, re.DOTALL): text = " ".join(l.strip() for l in m.group(4).splitlines() if l.strip()) blocks.append({"id": int(m.group(1)), "start": m.group(2).strip(), "end": m.group(3).strip(), "text": text}) return blocks def _rewrite_srt(srt_path, new_text_by_id): """Rewrite text of given blocks inside the SRT, keeping timing intact.""" srt_path = Path(srt_path) blocks = _srt_to_blocks(srt_path) lines_out = [] for b in blocks: lines_out.append(f"{b['id']}") lines_out.append(f"{b['start']} --> {b['end']}") lines_out.append(new_text_by_id.get(b["id"], b["text"])) lines_out.append("") srt_path.write_text("\n".join(lines_out), encoding="utf-8") def find_over_speed_segments(manifest_path, threshold=COMFORT_SPEEDUP_RATIO): """Return [(id, speedup_ratio, target_ms, final_ms), ...] for segments compressed harder than `threshold`.""" manifest_path = Path(manifest_path) if not manifest_path.exists(): return [] try: with open(manifest_path, "r", encoding="utf-8") as f: entries = json.load(f) except Exception: return [] flagged = [] for e in entries: ratio = float(e.get("speedup_ratio") or 1.0) if ratio > threshold: flagged.append(( int(e.get("id", 0)), round(ratio, 2), int(e.get("target_duration_ms", 0)), int(e.get("final_duration_ms") or 0), )) return flagged def compress_lines_with_llm(blocks, compress_ids, config, log_fn=print): """Ask the LLM to shorten the given blocks; returns {id: new_text}.""" compress_ids = set(compress_ids) if not compress_ids: return {} chunks = [] for b in blocks: if b["id"] in compress_ids: duration = 0.0 try: parts = b["end"].replace(",", ".").split(":") parts2 = parts[2] if len(parts) > 2 else "0.0" end_s = float(parts2) + (int(parts[0]) * 3600 + int(parts[1]) * 60 if len(parts) > 1 else 0) parts = b["start"].replace(",", ".").split(":") parts2 = parts[2] if len(parts) > 2 else "0.0" start_s = float(parts2) + (int(parts[0]) * 3600 + int(parts[1]) * 60 if len(parts) > 1 else 0) duration = max(0.1, end_s - start_s) except Exception: duration = 1.0 chunks.append({ "id": str(b["id"]), "start": b["start"], "end": b["end"], "duration_sec": round(duration, 3), "text": b["text"], "source_mode": "speech", "compress_hint": f"Tối đa {int(duration * 2.3)} âm tiết, rút gọn tối đa nhưng giữ ý chính.", }) if not chunks: return {} # Ask LLM with a strong compression instruction via the existing provider chain result = {} try: from app.translation.fallback import TranslationFallbackChain chain = TranslationFallbackChain(engines_list=["API Pool"], config_dict=config or {}) providers = chain.get_providers() for name, provider in providers: # Machine/Google translation can't compress per prompt; only LLM providers help if any(k in str(name) for k in ("Google", "MyMemory", "Machine")): continue try: system = ( "Bạn là biên tập lời lồng tiếng Việt. Viết LẠI mỗi câu thành một câu tiếng Việt " "HOÀN CHỈNH nhưng ngắn gọn, đủ ý chính, dễ đọc nhanh. " "Giữ tên riêng, brand, số liệu. Không thêm tag. Không cắt cụt thành từ đơn.\n" "Mỗi câu tối đa 2.3 từ mỗi giây thời lượng.\n" "Trả về JSON {\"id\": \"câu ngắn gọn đầy đủ\"}." ) user = ( "Rút gọn các câu sau (đọc field duration_sec và compress_hint):\n" + json.dumps(chunks, ensure_ascii=False, indent=2) ) raw = provider.translate_chunk(chunks, system, user, log_fn=log_fn) if isinstance(raw, dict): for k, v in raw.items(): try: result[int(k)] = v except Exception: pass elif isinstance(raw, str): data = _parse_json_output(raw) for k, v in data.items(): try: result[int(k)] = v except Exception: pass if result: break except Exception as e: log_fn(f"[QA] provider '{name}' compress failed: {e}") except Exception as e: log_fn(f"[QA] LLM compress failed: {e}") return result def _parse_json_output(raw): try: return json.loads(raw) except Exception: pass m = re.search(r"\{.*\}", raw, re.DOTALL) if m: try: return json.loads(m.group(0)) except Exception: pass return {} def run_qa_loop(voiceover_srt, manifest_path, config, max_rounds=2, threshold=COMFORT_SPEEDUP_RATIO, log_fn=print): """Full QA loop. Returns (rounds_run, still_over: [(id, ratio)]). The pipeline calls this after TTS; when it returns True for still_over, it rewrites the SRT and re-runs TTS. When no segments exceed the threshold the loop exits immediately. """ rounds_run = 0 still_over = [] srt_path = Path(voiceover_srt) for rnd in range(max_rounds): over = find_over_speed_segments(manifest_path, threshold) if not over: log_fn(f"[QA] Round {rnd+1}: 0 segment vượt {threshold}x — đạt chuẩn.") return rounds_run, [] rounds_run += 1 log_fn(f"[QA] Round {rnd+1}: {len(over)} segment speedup > {threshold}x -> gửi LLM rút ngắn.") blocks = _srt_to_blocks(srt_path) compress_ids = [i for i, _, _, _ in over] new_text = compress_lines_with_llm(blocks, compress_ids, config, log_fn) if not new_text: log_fn("[QA] LLM không rút gọn được gì; dừng loop.") return rounds_run, over # Guard: chỉ chấp nhận câu rút gọn thực sự hợp lệ (>=2 từ và ngắn hơn bản cũ) by_id = {b["id"]: b["text"] for b in blocks} accepted = {} for bid, txt in new_text.items(): old = by_id.get(bid, "") words = [w for w in str(txt).strip().split() if w] if len(words) >= 2 and len(str(txt)) < max(4, len(old) * 0.9): accepted[bid] = str(txt).strip() if not accepted: log_fn("[QA] LLM trả kết quả không hợp lệ (quá ngắn); dừng loop.") return rounds_run, over _rewrite_srt(srt_path, accepted) log_fn(f"[QA] Đã rút gọn {len(accepted)} câu trong {srt_path.name}; chờ re-run TTS.") return rounds_run, over still_over = find_over_speed_segments(manifest_path, threshold) return rounds_run, still_over