Spaces:
Running on Zero
Running on Zero
File size: 30,194 Bytes
3849548 927c1a5 3849548 927c1a5 3849548 927c1a5 3849548 927c1a5 3849548 2d419a6 927c1a5 3849548 6b02f6a 511c805 6b02f6a 3849548 4dae0ee 3849548 7e7df2a 3849548 6b02f6a 3849548 41d2f2c 3849548 744bf7a 3849548 7e7df2a 3849548 927c1a5 3849548 927c1a5 3849548 927c1a5 db9516b 3849548 6b02f6a 511c805 3849548 41d2f2c 3849548 41d2f2c 1804ac1 41d2f2c 744bf7a 3849548 7e7df2a 3849548 927c1a5 db9516b 7e7df2a 3849548 927c1a5 3849548 7b111a1 3849548 7e7df2a 3849548 641a414 3849548 f769cb7 3849548 7b111a1 3849548 3733db3 3849548 927c1a5 3849548 7e7df2a 3849548 5fc2aef e66cd92 3849548 7e7df2a 3849548 6b02f6a 7e7df2a 6b02f6a 7e7df2a 6b02f6a 744bf7a 511c805 3849548 6b02f6a 3849548 511c805 3849548 641a414 3849548 f769cb7 3849548 641a414 e66cd92 3849548 6b02f6a 511c805 6b02f6a 7e7df2a 744bf7a 3849548 7e7df2a 3849548 7e7df2a 3849548 7e7df2a 927c1a5 3849548 7e7df2a 3849548 41d2f2c 641a414 41d2f2c 641a414 7e7df2a 3849548 7e7df2a 927c1a5 7e7df2a 927c1a5 3849548 a8231fb e66cd92 3849548 e66cd92 3849548 927c1a5 3849548 927c1a5 3849548 a3ddb3f 927c1a5 3849548 927c1a5 3849548 01153e3 927c1a5 3849548 927c1a5 3849548 927c1a5 7e7df2a db9516b 511c805 7e7df2a 927c1a5 3849548 927c1a5 3849548 a8231fb 3849548 7e7df2a a8231fb 3849548 927c1a5 3849548 927c1a5 3849548 927c1a5 3849548 927c1a5 3849548 9a401f5 3849548 927c1a5 3849548 927c1a5 3849548 927c1a5 3849548 7b111a1 3849548 927c1a5 3849548 7afaa46 927c1a5 3849548 927c1a5 3849548 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 351 352 353 354 355 356 357 358 359 360 361 362 363 364 365 366 367 368 369 370 371 372 373 374 375 376 377 378 379 380 381 382 383 384 385 386 387 388 389 390 391 392 393 394 395 396 397 398 399 400 401 402 403 404 405 406 407 408 409 410 411 412 413 414 415 416 417 418 419 420 421 422 423 424 425 426 427 428 429 430 431 432 433 434 435 436 437 438 439 440 441 442 443 444 445 446 447 448 449 450 451 452 453 454 455 456 457 458 459 460 461 462 463 464 465 466 467 468 469 470 471 472 473 474 475 476 477 478 479 480 481 482 483 484 485 486 487 488 489 490 491 492 493 494 495 496 497 498 499 500 501 502 503 504 505 506 507 508 509 510 511 512 513 514 515 516 517 518 519 520 521 522 523 524 525 526 527 528 529 530 531 532 533 534 535 536 537 538 539 540 541 542 543 544 545 546 547 548 549 550 551 552 553 554 555 556 557 558 559 560 561 562 563 564 565 566 567 568 569 570 571 572 573 574 575 576 577 578 579 580 581 582 583 584 585 586 587 588 589 590 591 592 593 594 595 596 597 598 599 600 601 602 603 604 605 606 607 608 609 610 611 612 613 614 615 616 617 618 619 620 621 622 623 624 625 626 627 628 629 630 631 632 633 634 635 636 637 638 639 640 641 642 643 644 645 646 647 648 649 650 651 652 653 654 655 656 657 658 659 660 661 662 663 664 665 666 667 668 669 670 671 672 673 674 675 676 677 678 679 680 681 682 683 684 685 686 687 688 689 690 691 692 693 694 695 696 697 698 699 700 701 702 703 704 705 706 707 708 709 710 711 712 713 714 715 716 717 718 719 720 721 722 723 724 725 726 727 728 729 730 731 732 733 734 735 736 737 738 739 740 741 742 743 744 745 746 747 748 749 750 751 752 753 754 755 756 757 758 759 760 761 762 763 764 765 766 767 768 769 770 771 772 773 774 775 776 777 778 779 780 781 782 783 784 785 786 787 788 789 790 791 792 793 794 795 796 797 798 799 800 801 802 803 804 805 806 807 808 809 810 811 812 813 814 815 816 817 818 819 820 821 822 823 824 825 826 827 828 829 830 831 832 833 834 835 836 837 838 839 840 841 842 843 844 845 846 847 848 849 850 | """BlueMagpie-TTS Gradio demo using the validated production inference profile."""
from __future__ import annotations
import inspect
import json
import os
import secrets
import threading
import gradio as gr
import librosa
import numpy as np
import torch
from huggingface_hub import snapshot_download
from transformers import PreTrainedTokenizerFast
from bluemagpie import BlueMagpieModel
from production import (
StopHysteresisController,
apply_loudness_floor,
count_speech_units,
effective_generation_cfg,
endpoint_generation_plan,
estimate_step_seconds,
extract_windowed_speaker_embedding,
fade_internal_edges,
finish_audio,
join_audio_chunks,
match_chunk_rms,
normalize_spoken_forms,
normalize_tts_text,
punctuation_pause_seconds,
select_generation_cps,
set_generation_seed,
split_leading_clause,
split_text_for_tts,
target_pace_speed,
)
from quality_runtime import (
BASE_GENERATION_POLICY,
SAFE_DURATION_GENERATION_POLICY,
WHISPER_MODEL_ID,
WHISPER_REVISION,
CandidateObservation,
ChunkCandidateArtifact,
FinalOutputRejectedError,
GenerationPolicy,
NoQualifiedCandidateError,
active_voiced_duration_seconds,
active_audio_rms_db,
candidate_limit_for_chunk_budget,
generation_policy_for_candidate_offset,
prepare_candidate_audio,
qualify_trajectory_with_joined_output,
require_verified_final_output,
resolve_request_seed,
run_adaptive_cascade,
speaker_evidence_from_audio,
transcribe_whisper,
verify_trajectory,
)
try:
import spaces
gpu = spaces.GPU(duration=120)
DEVICE = "cuda"
except ImportError:
def gpu(function):
return function
DEVICE = "cuda" if torch.cuda.is_available() else "cpu"
REPO_ID = "OpenFormosa/BlueMagpie-TTS"
MODEL_REVISION = "aaf1a0878e37875382bb0e5c8a3a2ba43be67297"
ECAPA_REPO_ID = "speechbrain/spkrec-ecapa-voxceleb"
ECAPA_REVISION = "0f99f2d0ebe89ac095bcc5903c4dd8f72b367286"
DEFAULT_CFG = 2.0
DEFAULT_STEPS = 10
TARGET_CPS = 4.0
MIN_ENDPOINT_CUE_UNITS = 6
SHORT_TEXT_CFG_MIN = 3.0
SHORT_TEXT_CFG_UNITS = 6
CHUNK_CHARS = 80
ONSET_CLAUSE_SEARCH_CHARS = 40
MIN_CHUNK_CHARS = 12
CROSSFADE_MS = 80.0
CHUNK_EDGE_FADE_MS = 80.0
CHUNK_RMS_MATCH_DB = 4.0
STOP_THRESHOLD = 0.50
STOP_LATE_THRESHOLD = 0.05
STOP_LATE_START_RATIO = 0.75
STOP_LATE_FULL_RATIO = 0.95
STOP_CONSECUTIVE = 1
MIN_PACE_SPEED = 0.80
MAX_TEXT_CHARS = 360
QUALITY_MAX_CANDIDATES = 10
QUALITY_MAX_GENERATED_CHUNKS = 20
QUALITY_FINAL_ASR_MAX_NEW_TOKENS = 440
QUALITY_MAX_CER = 0.20
QUALITY_MAX_PACE_CPS = 4.30
QUALITY_PREFIX_SUFFIX_UNITS = 6
QUALITY_MIN_SPEAKER_SIMILARITY = 0.10
QUALITY_MAX_BOUNDARY_SPEAKER_DROP = 0.10
QUALITY_PREFERRED_MIN_SPEAKER_SIMILARITY = 0.25
QUALITY_PREFERRED_MAX_BOUNDARY_SPEAKER_DROP = 0.05
SHORT_AUDIO_SPEAKER_GATE_SECONDS = 1.50
print(f"[BlueMagpie] downloading model from {REPO_ID}@{MODEL_REVISION} ...")
MODEL_DIR = snapshot_download(REPO_ID, revision=MODEL_REVISION)
print(f"[BlueMagpie] caching speaker encoder from {ECAPA_REPO_ID}@{ECAPA_REVISION} ...")
ECAPA_DIR = snapshot_download(ECAPA_REPO_ID, revision=ECAPA_REVISION)
print(f"[BlueMagpie] caching quality ASR from {WHISPER_MODEL_ID}@{WHISPER_REVISION} ...")
ASR_DIR = snapshot_download(WHISPER_MODEL_ID, revision=WHISPER_REVISION)
tokenizer = PreTrainedTokenizerFast(tokenizer_file=os.path.join(MODEL_DIR, "tokenizer.json"))
print(f"[BlueMagpie] loading model on device={DEVICE} ...")
model = BlueMagpieModel.from_local(MODEL_DIR, tokenizer=tokenizer, training=False, device=DEVICE)
SR = int(model.sample_rate)
STEP_SECONDS = estimate_step_seconds(model, SR)
METADATA: dict = {}
try:
with open(os.path.join(MODEL_DIR, "release_metadata.json"), encoding="utf-8") as handle:
METADATA = json.load(handle)
except (OSError, ValueError) as error:
print(f"[BlueMagpie] release metadata unavailable: {error}")
CHECKPOINT = str(METADATA.get("checkpoint", "release"))
def _load_speakers() -> tuple[dict[str, torch.Tensor], str]:
path = os.path.join(MODEL_DIR, "checkpoints", "speaker_centroids.pt")
if not os.path.exists(path):
raise RuntimeError("speaker_centroids.pt is missing from the model release")
table = torch.load(path, map_location="cpu", weights_only=True)
speaker_ids = [str(value) for value in table["speaker_ids"]]
centroids = table["centroids"]
labels = {f"內建語者 {chr(65 + index)}": centroid for index, centroid in enumerate(centroids)}
requested_id = (
METADATA.get("recommended_generation_defaults", {}).get("speaker_id")
if isinstance(METADATA.get("recommended_generation_defaults"), dict)
else None
)
if requested_id not in speaker_ids and "female_voice" in speaker_ids:
requested_id = "female_voice"
default_index = speaker_ids.index(requested_id) if requested_id in speaker_ids else 0
return labels, f"內建語者 {chr(65 + default_index)}"
SPEAKERS, DEFAULT_SPEAKER = _load_speakers()
DEFAULT_CENTROID = SPEAKERS[DEFAULT_SPEAKER]
# The pinned public package predates native stop hysteresis. Wrap only that
# version; newer packages receive the same policy through native arguments.
_GENERATE_PARAMETERS = set(inspect.signature(model._generate).parameters)
_NATIVE_STOP_POLICY = {"stop_threshold", "stop_consecutive"}.issubset(_GENERATE_PARAMETERS)
_STOP_CONTROLLER: StopHysteresisController | None = None
if not _NATIVE_STOP_POLICY:
_STOP_CONTROLLER = StopHysteresisController(
model.stop_head,
threshold=STOP_THRESHOLD,
late_threshold=STOP_LATE_THRESHOLD,
consecutive=STOP_CONSECUTIVE,
late_start_ratio=STOP_LATE_START_RATIO,
late_full_ratio=STOP_LATE_FULL_RATIO,
)
model.stop_head = _STOP_CONTROLLER
_GENERATION_LOCK = threading.Lock()
_ECAPA_ENCODER = None
_ECAPA_LOCK = threading.Lock()
print(
f"[BlueMagpie] ready checkpoint={CHECKPOINT} sample_rate={SR} "
f"step_seconds={STEP_SECONDS} native_stop_policy={_NATIVE_STOP_POLICY}"
)
def _get_ecapa_encoder():
global _ECAPA_ENCODER
with _ECAPA_LOCK:
if _ECAPA_ENCODER is None:
import torchaudio
# SpeechBrain 1.0.3 still probes this API during import, while the
# ZeroGPU torchaudio build has removed it. Audio loading below is
# handled by librosa, so an empty compatibility result is correct.
if not hasattr(torchaudio, "list_audio_backends"):
torchaudio.list_audio_backends = lambda: []
from speechbrain.inference.speaker import EncoderClassifier
cache_dir = os.path.join(os.environ.get("HF_HOME", "/tmp"), "speechbrain", "ecapa")
_ECAPA_ENCODER = EncoderClassifier.from_hparams(
source=ECAPA_DIR,
# The upstream hyperparams otherwise points back to the repo
# and SpeechBrain 1.0.3 uses a removed Hub keyword. Keep every
# weight fetch inside the already pinned local snapshot.
overrides={"pretrained_path": ECAPA_DIR},
savedir=cache_dir,
run_opts={"device": "cpu"},
)
return _ECAPA_ENCODER
def _apply_speed(audio: np.ndarray, speed: float) -> np.ndarray:
speed = float(speed or 1.0)
if abs(speed - 1.0) < 1.0e-3:
return audio
return librosa.effects.time_stretch(np.asarray(audio, dtype=np.float32), rate=speed)
def _generate_chunk(
text: str,
centroid: torch.Tensor,
*,
cfg: float,
steps: int,
request_seed: int,
policy: GenerationPolicy,
) -> np.ndarray:
generation_cps = select_generation_cps(
text,
cjk_cps=policy.cjk_cps,
ascii_cps=policy.ascii_cps,
)
model_text, expected_steps, hard_stop_steps = endpoint_generation_plan(
text,
generation_cps=generation_cps,
step_seconds=STEP_SECONDS,
margin_steps=policy.hard_stop_margin_steps,
add_terminal_punctuation=count_speech_units(text) >= MIN_ENDPOINT_CUE_UNITS,
)
# Do not hold generation open to enforce pace. The model can finish the
# requested text early; extending its latent sequence creates tail speech.
min_len = 2
generation_cfg = effective_generation_cfg(
text,
cfg,
short_text_unit_threshold=SHORT_TEXT_CFG_UNITS,
short_text_min_cfg=SHORT_TEXT_CFG_MIN,
)
set_generation_seed(request_seed)
kwargs = {
"target_text": model_text,
"speaker_centroid": centroid,
"cfg_value": generation_cfg,
"inference_timesteps": int(steps),
"min_len": min_len,
"max_len": hard_stop_steps,
"retry_badcase": False,
"retry_badcase_max_times": 1,
"retry_badcase_ratio_threshold": 6.0,
}
if _NATIVE_STOP_POLICY:
kwargs["stop_threshold"] = STOP_THRESHOLD
kwargs["stop_consecutive"] = STOP_CONSECUTIVE
if "generation_seed" in _GENERATE_PARAMETERS:
kwargs["generation_seed"] = request_seed
if _STOP_CONTROLLER is not None:
_STOP_CONTROLLER.begin(
min_len,
expected_steps=expected_steps,
hard_stop_steps=hard_stop_steps,
)
try:
audio = model.generate(**kwargs)
finally:
if _STOP_CONTROLLER is not None:
_STOP_CONTROLLER.end()
if _STOP_CONTROLLER is not None:
print(
"[BlueMagpie] endpoint "
f"expected_steps={expected_steps} hard_stop_steps={hard_stop_steps} "
f"generated_steps={_STOP_CONTROLLER.last_generated_steps} "
f"reason={_STOP_CONTROLLER.last_stop_reason} cfg={generation_cfg:.2f}"
)
print(
"[BlueMagpie] generation policy "
f"name={policy.name} seed={request_seed} generation_cps={generation_cps:.2f} "
f"expected_steps={expected_steps} hard_stop_steps={hard_stop_steps} min_len={min_len}"
)
audio = audio.detach().float().cpu().numpy().reshape(-1)
pace_speed = target_pace_speed(
audio.size,
SR,
text,
target_cps=TARGET_CPS,
min_speed=MIN_PACE_SPEED,
)
return _apply_speed(audio, pace_speed)
def _speaker_anchor_array(centroid: torch.Tensor) -> np.ndarray:
anchor = torch.as_tensor(centroid).detach().float().cpu().numpy().reshape(-1)
if anchor.size == 0 or not np.isfinite(anchor).all():
raise ValueError("speaker anchor is invalid")
norm = float(np.linalg.norm(anchor))
if not np.isfinite(norm) or norm <= 1.0e-8:
raise ValueError("speaker anchor has zero norm")
return np.asarray(anchor / norm, dtype=np.float32)
def _generate_trajectory(
chunks: tuple[str, ...],
centroid: torch.Tensor,
*,
cfg: float,
steps: int,
request_seed: int,
policy: GenerationPolicy,
) -> tuple[np.ndarray, ...]:
return tuple(
_generate_chunk(
chunk,
centroid,
cfg=cfg,
steps=steps,
request_seed=request_seed,
policy=policy,
)
for chunk in chunks
)
def _verify_trajectory_audio(
trajectory: tuple[np.ndarray, ...],
chunks: tuple[str, ...],
anchor: np.ndarray,
playback_speed: float,
asr_max_new_tokens: int = 128,
):
if len(trajectory) != len(chunks):
return verify_trajectory(())
encoder = None
observations: list[CandidateObservation] = []
artifacts: list[ChunkCandidateArtifact] = []
for chunk, audio in zip(chunks, trajectory, strict=True):
prepared = prepare_candidate_audio(
audio,
SR,
transcriber=lambda waveform, sample_rate: transcribe_whisper(
waveform,
sample_rate,
max_new_tokens=asr_max_new_tokens,
),
)
if prepared is None:
observations.append(
CandidateObservation(
target_text=chunk,
transcript_text="",
audio_duration_seconds=0.0,
pace_cps=None,
)
)
artifacts.append(ChunkCandidateArtifact())
continue
waveform = prepared.waveform
try:
duration = active_voiced_duration_seconds(waveform, SR)
except ValueError:
duration = 0.0
transcript = prepared.transcript_text
speaker_similarity = None
begin_similarity = None
end_similarity = None
speaker_embedding = None
rms_db = None
if duration >= SHORT_AUDIO_SPEAKER_GATE_SECONDS:
try:
if encoder is None:
encoder = _get_ecapa_encoder()
evidence = speaker_evidence_from_audio(
waveform,
SR,
encoder,
anchor,
device="cpu",
)
duration = evidence.active_duration_seconds
speaker_similarity = evidence.similarity
begin_similarity = evidence.begin_similarity
end_similarity = evidence.end_similarity
speaker_embedding = evidence.speaker_embedding
rms_db = evidence.active_rms_db
except ValueError:
# A malformed/empty speaker measurement remains missing and is
# rejected by the fail-closed gate for non-short candidates.
pass
if rms_db is None:
try:
rms_db = active_audio_rms_db(waveform)
except ValueError:
pass
observations.append(
CandidateObservation(
target_text=chunk,
transcript_text=transcript,
audio_duration_seconds=duration,
speaker_similarity=speaker_similarity,
begin_speaker_similarity=begin_similarity,
end_speaker_similarity=end_similarity,
pace_cps=(
count_speech_units(chunk) / duration * float(playback_speed)
if duration > 0.0
else None
),
)
)
artifacts.append(
ChunkCandidateArtifact(
speaker_embedding=speaker_embedding,
rms_db=rms_db,
)
)
return verify_trajectory(
observations,
chunk_artifacts=artifacts,
short_text_units=6,
short_text_max_cer=0.0,
max_cer=QUALITY_MAX_CER,
prefix_units=QUALITY_PREFIX_SUFFIX_UNITS,
suffix_units=QUALITY_PREFIX_SUFFIX_UNITS,
max_prefix_cer=0.0,
max_suffix_cer=0.0,
max_extra_tail_units=0,
short_audio_seconds=SHORT_AUDIO_SPEAKER_GATE_SECONDS,
min_speaker_similarity=QUALITY_MIN_SPEAKER_SIMILARITY,
max_boundary_speaker_drop=QUALITY_MAX_BOUNDARY_SPEAKER_DROP,
max_pace_cps=QUALITY_MAX_PACE_CPS,
)
def _assemble_trajectory_audio(
trajectory: tuple[np.ndarray, ...],
chunks: tuple[str, ...],
playback_speed: float,
) -> np.ndarray:
"""Assemble chunks exactly as they will be returned to the listener."""
if not trajectory or len(trajectory) != len(chunks):
raise ValueError("trajectory and text chunks must be non-empty and aligned")
audio_chunks = [np.asarray(audio, dtype=np.float32).copy() for audio in trajectory]
pauses: list[int] = []
for index, chunk in enumerate(chunks):
if index > 0:
audio_chunks[index] = match_chunk_rms(
audio_chunks[0],
audio_chunks[index],
max_adjust_db=CHUNK_RMS_MATCH_DB,
)
if index + 1 < len(chunks):
pauses.append(int(round(punctuation_pause_seconds(chunk) * SR)))
audio_chunks = fade_internal_edges(audio_chunks, SR, fade_ms=CHUNK_EDGE_FADE_MS)
waveform = join_audio_chunks(
audio_chunks,
pauses,
crossfade_samples=int(round(CROSSFADE_MS * SR / 1000.0)),
)
waveform = apply_loudness_floor(
waveform,
min_rms=0.07,
peak_limit=0.95,
max_gain=3.0,
)
waveform = _apply_speed(waveform, playback_speed)
return finish_audio(waveform, SR)
def _verification_metric_log_fields(verification) -> str:
"""Format normalized semantic metrics without logging transcript content."""
if len(verification.candidate_results) != 1:
return (
"cer=nan prefix_cer=nan suffix_cer=nan tail_units=nan "
f"reasons={verification.rejection_reasons}"
)
comparison = verification.candidate_results[0].comparison
return (
f"cer={comparison.cer:.6f} prefix_cer={comparison.prefix_cer:.6f} "
f"suffix_cer={comparison.suffix_cer:.6f} "
f"tail_units={comparison.extra_tail_units} "
f"reasons={verification.rejection_reasons}"
)
def _qualify_candidate_trajectory_audio(
trajectory: tuple[np.ndarray, ...],
chunks: tuple[str, ...],
whole_target_text: str,
anchor: np.ndarray,
playback_speed: float,
*,
candidate_seed: int,
):
"""Run whole-output qualification only after every local chunk passes."""
local_verification = _verify_trajectory_audio(
trajectory,
chunks,
anchor,
playback_speed,
)
if not local_verification.passed:
return local_verification
waveform = _assemble_trajectory_audio(trajectory, chunks, playback_speed)
joined_verification = _verify_trajectory_audio(
(waveform,),
(whole_target_text,),
anchor,
1.0,
QUALITY_FINAL_ASR_MAX_NEW_TOKENS,
)
qualified = qualify_trajectory_with_joined_output(
local_verification,
joined_verification,
)
if not qualified.passed:
print(
"[BlueMagpie] candidate joined output rejected "
f"seed={candidate_seed} "
f"{_verification_metric_log_fields(joined_verification)}"
)
return qualified
def _verify_sequence_trajectory_audio(
sequence_result,
chunks: tuple[str, ...],
whole_target_text: str,
anchor: np.ndarray,
playback_speed: float,
):
"""Verify one ranked DP path after exact production assembly."""
waveform = _assemble_trajectory_audio(
sequence_result.trajectory,
chunks,
playback_speed,
)
verification = _verify_trajectory_audio(
(waveform,),
(whole_target_text,),
anchor,
1.0,
QUALITY_FINAL_ASR_MAX_NEW_TOKENS,
)
status = "verified" if verification.passed else "rejected"
print(
f"[BlueMagpie] sequence path {status} "
f"rank={sequence_result.sequence_path_rank} "
f"chunk_candidates={sequence_result.chunk_candidate_indices} "
f"{_verification_metric_log_fields(verification)}"
)
return verification
def _synthesize(
text: str,
centroid: torch.Tensor,
*,
cfg: float,
steps: int,
speed: float,
request_seed: int | None = None,
) -> tuple[int, np.ndarray]:
text = normalize_spoken_forms(text, locale="zh-TW")
if not text:
raise gr.Error("請先輸入要合成的文字。")
if len(text) > MAX_TEXT_CHARS:
raise gr.Error(f"單次最多 {MAX_TEXT_CHARS} 個字元,請分段合成。")
if not np.isfinite(float(speed)) or not 0.85 <= float(speed) <= 1.05:
raise gr.Error("後處理語速必須介於 0.85 與 1.05。")
chunks = split_text_for_tts(text, max_chars=CHUNK_CHARS, min_chunk_chars=MIN_CHUNK_CHARS)
if chunks:
onset_chunks = split_leading_clause(
chunks[0],
search_chars=ONSET_CLAUSE_SEARCH_CHARS,
min_chunk_chars=MIN_CHUNK_CHARS,
)
chunks = onset_chunks + chunks[1:]
request_seed = resolve_request_seed(request_seed, secrets.randbelow)
max_candidates = candidate_limit_for_chunk_budget(
len(chunks),
max_candidates=QUALITY_MAX_CANDIDATES,
max_generated_chunks=QUALITY_MAX_GENERATED_CHUNKS,
)
anchor = _speaker_anchor_array(centroid)
try:
with _GENERATION_LOCK:
cascade = run_adaptive_cascade(
chunks,
request_seed,
lambda candidate_chunks, seed: _generate_trajectory(
candidate_chunks,
centroid,
cfg=cfg,
steps=steps,
request_seed=seed,
policy=generation_policy_for_candidate_offset(seed - request_seed),
),
lambda trajectory, candidate_chunks, seed: _qualify_candidate_trajectory_audio(
trajectory,
candidate_chunks,
text,
anchor,
speed,
candidate_seed=seed,
),
max_candidates=max_candidates,
preferred_min_speaker_similarity=(
QUALITY_PREFERRED_MIN_SPEAKER_SIMILARITY
),
preferred_max_boundary_speaker_drop=(
QUALITY_PREFERRED_MAX_BOUNDARY_SPEAKER_DROP
),
sequence_final_verifier=lambda sequence_result, candidate_chunks: (
_verify_sequence_trajectory_audio(
sequence_result,
candidate_chunks,
text,
anchor,
speed,
)
),
max_sequence_paths=3,
)
except NoQualifiedCandidateError as error:
raise gr.Error("目前沒有候選通過內容與音色驗證,請稍後重試或調整文字。") from error
except (RuntimeError, ValueError) as error:
raise gr.Error("品質驗證暫時無法完成,未回傳未驗證的語音。") from error
selected_policies = tuple(
generation_policy_for_candidate_offset(index).name
for index in cascade.chunk_candidate_indices
)
attempted_policies = tuple(
generation_policy_for_candidate_offset(seed - request_seed).name
for seed in cascade.attempted_seeds
)
print(
"[BlueMagpie] quality cascade "
f"candidate_index={cascade.candidate_index} attempts={len(cascade.attempted_seeds)} "
f"candidate_limit={max_candidates} selection={cascade.selection_mode} "
f"chunk_candidates={cascade.chunk_candidate_indices} "
f"chunk_seeds={cascade.chunk_seeds} score={cascade.verification.score:.6f}"
f" chunk_policies={selected_policies} attempted_policies={attempted_policies}"
f" sequence_rank={cascade.sequence_path_rank}"
f" sequence_paths_checked={cascade.sequence_paths_checked}"
)
waveform = _assemble_trajectory_audio(cascade.trajectory, chunks, speed)
final_verification = _verify_trajectory_audio(
(waveform,),
(text,),
anchor,
1.0,
QUALITY_FINAL_ASR_MAX_NEW_TOKENS,
)
try:
require_verified_final_output(final_verification)
except FinalOutputRejectedError as error:
print(
"[BlueMagpie] final output rejected "
f"{_verification_metric_log_fields(final_verification)}"
)
raise gr.Error("最終合成結果未通過整段內容、語速與音色驗證,未回傳音訊。") from error
print(
"[BlueMagpie] final output verified "
f"score={final_verification.score:.6f}"
)
return SR, waveform
@gpu
def tts_speaker(
text: str,
speaker: str = DEFAULT_SPEAKER,
cfg: float = DEFAULT_CFG,
steps: int = DEFAULT_STEPS,
speed: float = 1.0,
):
return _synthesize(
text,
SPEAKERS.get(speaker, DEFAULT_CENTROID),
cfg=cfg,
steps=steps,
speed=speed,
)
@gpu
def tts_reference(
text: str,
reference_wav: str,
cfg: float = DEFAULT_CFG,
steps: int = DEFAULT_STEPS,
speed: float = 1.0,
):
if not reference_wav:
raise gr.Error("請先錄音或上傳參考音檔。")
try:
centroid = extract_windowed_speaker_embedding(
reference_wav,
_get_ecapa_encoder(),
device="cpu",
min_duration_seconds=3.0,
window_seconds=3.0,
hop_seconds=1.5,
max_windows=12,
full_clip_max_seconds=12.0,
)
except ValueError as error:
raise gr.Error(str(error)) from error
return _synthesize(text, centroid, cfg=cfg, steps=steps, speed=speed)
@gpu
def tts_longform(
text: str,
speaker: str = DEFAULT_SPEAKER,
cfg: float = DEFAULT_CFG,
steps: int = DEFAULT_STEPS,
speed: float = 1.0,
):
return _synthesize(
text,
SPEAKERS.get(speaker, DEFAULT_CENTROID),
cfg=cfg,
steps=steps,
speed=speed,
)
EXAMPLE_TEXTS = [
"今天天氣真好,我們一起去散步吧。",
"我要吃蚵仔煎,然後去丟垃圾。",
"這學期的成績包括研究報告和期末考。",
"這是 AI TTS code switching 測試,混合中英文也沒問題。",
"注音符號測試:ㄅ、ㄆ、ㄇ、ㄈ。",
]
LONGFORM_EXAMPLES = [
"今天的會議會先整理目前進度,再確認下一階段的工作。遇到需要討論的項目時,"
"請先記下問題,等報告結束後再一起處理。最後,我們會確認負責人和預計完成時間。",
"歡迎收聽今天的內容。第一段會介紹背景,第二段整理實際案例,最後一段則說明後續安排。"
"如果中途聽到英文術語,不必擔心,我們會用中文補充它的意思。",
]
HEADER = f"""
# BlueMagpie-TTS Demo
台灣華語與中英混合文字轉語音。模型版本:`{CHECKPOINT}`。
執行環境固定於已驗證的 model revision `{MODEL_REVISION[:8]}`、ECAPA revision
`{ECAPA_REVISION[:8]}` 與 Whisper revision `{WHISPER_REVISION[:8]}`,避免服務重啟時
無聲變更權重、speaker embedding 或語意驗證空間。
目前預設採用穩定推論設定:`CFG 2.0`(極短句最低 `3.0`)、`NFE 10`、目標語速 `4.0 字/秒`、
candidate 0 使用 base duration estimate({BASE_GENERATION_POLICY.cjk_cps:.1f} CJK /
{BASE_GENERATION_POLICY.ascii_cps:.1f} ASCII),後續候選使用 safe duration estimate
({SAFE_DURATION_GENERATION_POLICY.cjk_cps:.1f} CJK /
{SAFE_DURATION_GENERATION_POLICY.ascii_cps:.1f} ASCII);兩者只調整生成上限,生成完成後才校正至
目標語速。另補齊句末提示、套用尾端 weak-stop 保護、
只在自然標點切開首段、每 80 字切段;先選完整 same-seed trajectory,失敗時才以 speaker/RMS
transition 做逐 chunk DP fallback。短句最多擴展到 1→5→10,長文依 chunk 數縮小候選上限,
確保每個 request 最多生成 20 個 TTS chunks。
"""
with gr.Blocks(title="BlueMagpie-TTS Demo", theme=gr.themes.Soft()) as demo:
gr.Markdown(HEADER)
with gr.Accordion("進階生成參數", open=False):
with gr.Row():
cfg_input = gr.Slider(1.0, 4.0, value=DEFAULT_CFG, step=0.1, label="CFG")
steps_input = gr.Slider(4, 20, value=DEFAULT_STEPS, step=1, label="NFE steps")
speed_input = gr.Slider(0.85, 1.05, value=1.0, step=0.05, label="後處理語速")
with gr.Tab("內建語者"):
with gr.Row():
with gr.Column():
speaker_input = gr.Dropdown(list(SPEAKERS), value=DEFAULT_SPEAKER, label="語者")
speaker_text = gr.Textbox(label="文字", lines=4, max_lines=8)
speaker_button = gr.Button("合成", variant="primary")
with gr.Column():
speaker_output = gr.Audio(label="合成結果", type="numpy")
gr.Examples(EXAMPLE_TEXTS, inputs=speaker_text, label="範例")
speaker_button.click(
tts_speaker,
[speaker_text, speaker_input, cfg_input, steps_input, speed_input],
speaker_output,
)
with gr.Tab("參考音色"):
gr.Markdown("參考音檔至少 3 秒;只使用已取得授權的聲音。參考內容不需要逐字稿。")
with gr.Row():
with gr.Column():
reference_text = gr.Textbox(label="文字", lines=4, max_lines=8)
reference_audio = gr.Audio(
label="參考音檔",
type="filepath",
sources=["microphone", "upload"],
)
reference_button = gr.Button("合成", variant="primary")
with gr.Column():
reference_output = gr.Audio(label="合成結果", type="numpy")
reference_button.click(
tts_reference,
[reference_text, reference_audio, cfg_input, steps_input, speed_input],
reference_output,
)
with gr.Tab("穩定長文"):
with gr.Row():
with gr.Column():
longform_speaker = gr.Dropdown(list(SPEAKERS), value=DEFAULT_SPEAKER, label="語者")
longform_text = gr.Textbox(
label=f"長文(最多 {MAX_TEXT_CHARS} 字元)",
lines=8,
max_lines=12,
)
longform_button = gr.Button("合成完整長文", variant="primary")
with gr.Column():
longform_output = gr.Audio(label="合成結果", type="numpy")
gr.Examples(LONGFORM_EXAMPLES, inputs=longform_text, label="長文範例")
longform_button.click(
tts_longform,
[longform_text, longform_speaker, cfg_input, steps_input, speed_input],
longform_output,
)
gr.Markdown(
"合成語音僅供研究與評估展示;正式使用前請人工檢視。 "
"[模型](https://huggingface.co/OpenFormosa/BlueMagpie-TTS) · "
"[程式碼](https://github.com/OpenFormosa/BlueMagpie-TTS)"
)
if __name__ == "__main__":
demo.queue(default_concurrency_limit=1).launch()
|