Spaces:
Running on Zero
Running on Zero
Reduce verified single-chunk latency
Browse files- README.md +2 -2
- app.py +14 -1
- quality_runtime.py +125 -3
- tests/test_coverage_adaptive.py +320 -0
- tests/test_release_pins.py +4 -0
README.md
CHANGED
|
@@ -64,12 +64,12 @@ Mamba2 所需的 Triton runtime 以
|
|
| 64 |
| Pace / speaker eligibility duration | union of active 25 ms RMS frames at 10 ms hop; internal pauses excluded |
|
| 65 |
| Semantic verification | orthographic whole CER + tone-aware acoustic first/last 6 exact + no lexical tail; range-proven URL/email locals and every exact whole output require turbo/full-large-v3 hard intersection; fail closed |
|
| 66 |
| Local endpoint verification | internal chunk first/last 6 may contain at most one substitution and zero deletion; absolute request onset/ending remain exact |
|
| 67 |
-
| Acoustic quality gate | pinned CPU SQUIM objective SHA-256 `2c54586fea83fb5eb5394d710038ee89f55cab7011a5bf730bebed4c8777e828`; hard STOI ≥ 0.60 / PESQ ≥ 1.12; at ≥1.50 s,
|
| 68 |
| Local speaker verification | frozen request centroid + bounded window begin/end directional drift |
|
| 69 |
| Local hard speaker gate | similarity ≥ 0.10 and directional drop ≤ 0.10 |
|
| 70 |
| Joined/final release-parity speaker guard | full-audio similarity ≥ 0.105; disjoint first/last-third drop ≤ 0.095 |
|
| 71 |
| Local long speaker evidence | one ECAPA batch over up to four 3-second windows + begin/end |
|
| 72 |
-
| Candidate cascade | exactly one same-seed whole trajectory, then deterministic single-chunk refills for zero/low-coverage rows only |
|
| 73 |
| Whole-trajectory qualification | assemble with production RMS/fade/pause/crossfade/speed, then whole-output gate |
|
| 74 |
| Sequence fallback | ragged DP with up to 3 culprit-diverse paths; boundary-only local rejects may enter through a frozen 0.15 cap, then every exact assembly must pass turbo + full large-v3 and the stricter 0.105/0.095 whole gate |
|
| 75 |
| Final output gate | re-verify joined/faded/RMS-matched/speed-adjusted whole waveform; fail closed |
|
|
|
|
| 64 |
| Pace / speaker eligibility duration | union of active 25 ms RMS frames at 10 ms hop; internal pauses excluded |
|
| 65 |
| Semantic verification | orthographic whole CER + tone-aware acoustic first/last 6 exact + no lexical tail; range-proven URL/email locals and every exact whole output require turbo/full-large-v3 hard intersection; fail closed |
|
| 66 |
| Local endpoint verification | internal chunk first/last 6 may contain at most one substitution and zero deletion; absolute request onset/ending remain exact |
|
| 67 |
+
| Acoustic quality gate | pinned CPU SQUIM objective SHA-256 `2c54586fea83fb5eb5394d710038ee89f55cab7011a5bf730bebed4c8777e828`; hard STOI ≥ 0.60 / PESQ ≥ 1.12; at ≥1.50 s, strict preferred return requires STOI ≥ 0.72 / PESQ ≥ 1.20 and speaker 0.25 / 0.05. A single-chunk exact joined waveform may also stop at the latency tier speaker 0.23 / 0.08, but only after both ASRs, preferred SQUIM, hard release speaker, pace, endpoint, and echo/smearing gates pass; local proxies cannot use this tier. SI-SDR is bounded soft evidence only |
|
| 68 |
| Local speaker verification | frozen request centroid + bounded window begin/end directional drift |
|
| 69 |
| Local hard speaker gate | similarity ≥ 0.10 and directional drop ≤ 0.10 |
|
| 70 |
| Joined/final release-parity speaker guard | full-audio similarity ≥ 0.105; disjoint first/last-third drop ≤ 0.095 |
|
| 71 |
| Local long speaker evidence | one ECAPA batch over up to four 3-second windows + begin/end |
|
| 72 |
+
| Candidate cascade | exactly one same-seed whole trajectory, then deterministic single-chunk refills for zero/low-coverage rows only; a fully verified single-chunk exact waveform returns immediately instead of filling the preferred candidate pool |
|
| 73 |
| Whole-trajectory qualification | assemble with production RMS/fade/pause/crossfade/speed, then whole-output gate |
|
| 74 |
| Sequence fallback | ragged DP with up to 3 culprit-diverse paths; boundary-only local rejects may enter through a frozen 0.15 cap, then every exact assembly must pass turbo + full large-v3 and the stricter 0.105/0.095 whole gate |
|
| 75 |
| Final output gate | re-verify joined/faded/RMS-matched/speed-adjusted whole waveform; fail closed |
|
app.py
CHANGED
|
@@ -279,6 +279,11 @@ QUALITY_PREFERRED_MIN_SQUIM_PESQ = 1.20
|
|
| 279 |
QUALITY_PREFERRED_SQUIM_MIN_DURATION_SECONDS = 1.50
|
| 280 |
QUALITY_PREFERRED_MIN_SPEAKER_SIMILARITY = 0.25
|
| 281 |
QUALITY_PREFERRED_MAX_BOUNDARY_SPEAKER_DROP = 0.05
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 282 |
QUALITY_MAX_SEQUENCE_PATHS = 3
|
| 283 |
SHORT_AUDIO_SPEAKER_GATE_SECONDS = 1.50
|
| 284 |
GLYPH_ASSET_MANIFEST_RELATIVE_PATH = "glyph_assets_v1/manifest.json"
|
|
@@ -2717,6 +2722,12 @@ def _synthesize(
|
|
| 2717 |
preferred_min_squim_audio_duration_seconds=(
|
| 2718 |
QUALITY_PREFERRED_SQUIM_MIN_DURATION_SECONDS
|
| 2719 |
),
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 2720 |
require_endpoint_evidence=True,
|
| 2721 |
)
|
| 2722 |
except NoQualifiedCandidateError as error:
|
|
@@ -3628,7 +3639,9 @@ HEADER = f"""
|
|
| 3628 |
目標語速。另補齊句末提示、套用尾端 weak-stop 保護、
|
| 3629 |
一般 32 speech units 內維持單一 chunk,較長內容才依自然標點與 80-unit 上限切段。每個 request 只生成一次
|
| 3630 |
same-seed 完整 trajectory;失敗後只補低覆蓋 chunks,再以 speaker/RMS/F0 ragged DP 選出最多三條
|
| 3631 |
-
culprit-diverse exact paths;所有 chunk 首次具備 coverage 時先驗證 rank-1 exact path。
|
|
|
|
|
|
|
| 3632 |
與 800 speech units。語速校正使用 deterministic WSOLA,總 rate 不低於 0.90;不少於 1 秒的
|
| 3633 |
候選與最終波形另須通過 echo/smearing gate。
|
| 3634 |
"""
|
|
|
|
| 279 |
QUALITY_PREFERRED_SQUIM_MIN_DURATION_SECONDS = 1.50
|
| 280 |
QUALITY_PREFERRED_MIN_SPEAKER_SIMILARITY = 0.25
|
| 281 |
QUALITY_PREFERRED_MAX_BOUNDARY_SPEAKER_DROP = 0.05
|
| 282 |
+
# A single-chunk waveform that has already passed both ASR checks and every
|
| 283 |
+
# hard release gate may stop the cascade at this exact-only tier. Local chunk
|
| 284 |
+
# proxies and multi-chunk eager paths remain on the stricter preferred tier.
|
| 285 |
+
QUALITY_FAST_EXACT_MIN_SPEAKER_SIMILARITY = 0.23
|
| 286 |
+
QUALITY_FAST_EXACT_MAX_BOUNDARY_SPEAKER_DROP = 0.08
|
| 287 |
QUALITY_MAX_SEQUENCE_PATHS = 3
|
| 288 |
SHORT_AUDIO_SPEAKER_GATE_SECONDS = 1.50
|
| 289 |
GLYPH_ASSET_MANIFEST_RELATIVE_PATH = "glyph_assets_v1/manifest.json"
|
|
|
|
| 2722 |
preferred_min_squim_audio_duration_seconds=(
|
| 2723 |
QUALITY_PREFERRED_SQUIM_MIN_DURATION_SECONDS
|
| 2724 |
),
|
| 2725 |
+
single_exact_min_speaker_similarity=(
|
| 2726 |
+
QUALITY_FAST_EXACT_MIN_SPEAKER_SIMILARITY
|
| 2727 |
+
),
|
| 2728 |
+
single_exact_max_boundary_speaker_drop=(
|
| 2729 |
+
QUALITY_FAST_EXACT_MAX_BOUNDARY_SPEAKER_DROP
|
| 2730 |
+
),
|
| 2731 |
require_endpoint_evidence=True,
|
| 2732 |
)
|
| 2733 |
except NoQualifiedCandidateError as error:
|
|
|
|
| 3639 |
目標語速。另補齊句末提示、套用尾端 weak-stop 保護、
|
| 3640 |
一般 32 speech units 內維持單一 chunk,較長內容才依自然標點與 80-unit 上限切段。每個 request 只生成一次
|
| 3641 |
same-seed 完整 trajectory;失敗後只補低覆蓋 chunks,再以 speaker/RMS/F0 ragged DP 選出最多三條
|
| 3642 |
+
culprit-diverse exact paths;所有 chunk 首次具備 coverage 時先驗證 rank-1 exact path。單一 chunk 若 exact joined
|
| 3643 |
+
波形已通過雙 ASR、preferred SQUIM、release speaker、語速與 echo/smearing gate,會立即返回而不再填滿候選池。
|
| 3644 |
+
NFE 固定為已驗證的 10,且每個 request 最多生成 32 個 TTS chunks
|
| 3645 |
與 800 speech units。語速校正使用 deterministic WSOLA,總 rate 不低於 0.90;不少於 1 秒的
|
| 3646 |
候選與最終波形另須通過 echo/smearing gate。
|
| 3647 |
"""
|
quality_runtime.py
CHANGED
|
@@ -4274,6 +4274,54 @@ def _preferred_release_evidence(
|
|
| 4274 |
return True
|
| 4275 |
|
| 4276 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 4277 |
def _coverage_trajectory_tuple(trajectory: Any, expected_chunks: int) -> tuple[Any, ...]:
|
| 4278 |
"""Return one generator result as an exact, non-string trajectory."""
|
| 4279 |
|
|
@@ -4704,6 +4752,8 @@ def run_coverage_adaptive_cascade(
|
|
| 4704 |
preferred_min_squim_stoi: float | None = None,
|
| 4705 |
preferred_min_squim_pesq: float | None = None,
|
| 4706 |
preferred_min_squim_audio_duration_seconds: float = 0.0,
|
|
|
|
|
|
|
| 4707 |
require_endpoint_evidence: bool = False,
|
| 4708 |
) -> CascadeResult:
|
| 4709 |
"""Run one whole trajectory, then deterministic low-coverage refills.
|
|
@@ -4922,6 +4972,32 @@ def run_coverage_adaptive_cascade(
|
|
| 4922 |
preferred_stoi = None
|
| 4923 |
preferred_pesq = None
|
| 4924 |
preferred_squim_duration = 0.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 4925 |
preferred_single_search = bool(
|
| 4926 |
preferred_speaker_enabled or preferred_squim_enabled
|
| 4927 |
)
|
|
@@ -5076,8 +5152,24 @@ def run_coverage_adaptive_cascade(
|
|
| 5076 |
min_squim_audio_duration_seconds=preferred_squim_duration,
|
| 5077 |
)
|
| 5078 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 5079 |
if initial_verification.passed is True and (
|
| 5080 |
-
chunk_count > 1
|
|
|
|
|
|
|
|
|
|
| 5081 |
):
|
| 5082 |
return CascadeResult(
|
| 5083 |
trajectory=initial_trajectory,
|
|
@@ -5499,17 +5591,33 @@ def run_coverage_adaptive_cascade(
|
|
| 5499 |
min_squim_audio_duration_seconds=preferred_squim_duration,
|
| 5500 |
)
|
| 5501 |
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 5502 |
if (
|
| 5503 |
chunk_count == 1
|
| 5504 |
and preferred_single_search
|
| 5505 |
and refill_verification.passed is True
|
| 5506 |
-
and refill_preferred
|
| 5507 |
):
|
| 5508 |
preferred_candidate = diagnostic_candidates[-1]
|
| 5509 |
preferred_artifact = preferred_candidate.verification.chunk_artifacts[0]
|
| 5510 |
preferred_reached_hard_cap = (
|
| 5511 |
_endpoint_artifact_reached_hard_cap(preferred_artifact)
|
| 5512 |
)
|
|
|
|
|
|
|
|
|
|
| 5513 |
selectable = [
|
| 5514 |
candidate
|
| 5515 |
for candidate in diagnostic_candidates
|
|
@@ -5538,7 +5646,21 @@ def run_coverage_adaptive_cascade(
|
|
| 5538 |
),
|
| 5539 |
)
|
| 5540 |
or (
|
| 5541 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 5542 |
and _preferred_natural_endpoint_waiver(
|
| 5543 |
candidate.verification,
|
| 5544 |
candidate.verification.chunk_artifacts[0],
|
|
|
|
| 4274 |
return True
|
| 4275 |
|
| 4276 |
|
| 4277 |
+
def _single_exact_fast_release_evidence(
|
| 4278 |
+
release_evidence: TrajectoryGateEvidence | None,
|
| 4279 |
+
independent_evidence: TrajectoryGateEvidence | None,
|
| 4280 |
+
*,
|
| 4281 |
+
min_speaker_similarity: float,
|
| 4282 |
+
max_boundary_speaker_drop: float,
|
| 4283 |
+
min_squim_stoi: float,
|
| 4284 |
+
min_squim_pesq: float,
|
| 4285 |
+
min_audio_duration_seconds: float,
|
| 4286 |
+
) -> bool:
|
| 4287 |
+
"""Accept only a fully measured exact waveform at the fast quality tier.
|
| 4288 |
+
|
| 4289 |
+
Unlike the general preferred helper, this latency shortcut never treats a
|
| 4290 |
+
skipped short-audio speaker or SQUIM gate as success. It is deliberately
|
| 4291 |
+
limited to exact joined-waveform evidence; local chunk proxies cannot
|
| 4292 |
+
trigger it.
|
| 4293 |
+
"""
|
| 4294 |
+
|
| 4295 |
+
if (
|
| 4296 |
+
not isinstance(release_evidence, TrajectoryGateEvidence)
|
| 4297 |
+
or release_evidence.result is None
|
| 4298 |
+
or release_evidence.result.speaker_gate_applied is not True
|
| 4299 |
+
or release_evidence.result.squim_gate_applied is not True
|
| 4300 |
+
or not _preferred_release_evidence(
|
| 4301 |
+
independent_evidence,
|
| 4302 |
+
min_speaker_similarity=None,
|
| 4303 |
+
max_boundary_speaker_drop=None,
|
| 4304 |
+
min_squim_stoi=None,
|
| 4305 |
+
min_squim_pesq=None,
|
| 4306 |
+
)
|
| 4307 |
+
):
|
| 4308 |
+
return False
|
| 4309 |
+
duration = _finite_float(
|
| 4310 |
+
release_evidence.result.audio_duration_seconds,
|
| 4311 |
+
minimum=0.0,
|
| 4312 |
+
)
|
| 4313 |
+
if duration is None or duration < min_audio_duration_seconds:
|
| 4314 |
+
return False
|
| 4315 |
+
return _preferred_release_evidence(
|
| 4316 |
+
release_evidence,
|
| 4317 |
+
min_speaker_similarity=min_speaker_similarity,
|
| 4318 |
+
max_boundary_speaker_drop=max_boundary_speaker_drop,
|
| 4319 |
+
min_squim_stoi=min_squim_stoi,
|
| 4320 |
+
min_squim_pesq=min_squim_pesq,
|
| 4321 |
+
min_squim_audio_duration_seconds=min_audio_duration_seconds,
|
| 4322 |
+
)
|
| 4323 |
+
|
| 4324 |
+
|
| 4325 |
def _coverage_trajectory_tuple(trajectory: Any, expected_chunks: int) -> tuple[Any, ...]:
|
| 4326 |
"""Return one generator result as an exact, non-string trajectory."""
|
| 4327 |
|
|
|
|
| 4752 |
preferred_min_squim_stoi: float | None = None,
|
| 4753 |
preferred_min_squim_pesq: float | None = None,
|
| 4754 |
preferred_min_squim_audio_duration_seconds: float = 0.0,
|
| 4755 |
+
single_exact_min_speaker_similarity: float | None = None,
|
| 4756 |
+
single_exact_max_boundary_speaker_drop: float | None = None,
|
| 4757 |
require_endpoint_evidence: bool = False,
|
| 4758 |
) -> CascadeResult:
|
| 4759 |
"""Run one whole trajectory, then deterministic low-coverage refills.
|
|
|
|
| 4972 |
preferred_stoi = None
|
| 4973 |
preferred_pesq = None
|
| 4974 |
preferred_squim_duration = 0.0
|
| 4975 |
+
single_exact_fast_enabled = (
|
| 4976 |
+
single_exact_min_speaker_similarity is not None
|
| 4977 |
+
or single_exact_max_boundary_speaker_drop is not None
|
| 4978 |
+
)
|
| 4979 |
+
if single_exact_fast_enabled:
|
| 4980 |
+
single_exact_similarity = _finite_float(
|
| 4981 |
+
single_exact_min_speaker_similarity,
|
| 4982 |
+
minimum=-1.0,
|
| 4983 |
+
maximum=1.0,
|
| 4984 |
+
)
|
| 4985 |
+
single_exact_boundary = _finite_float(
|
| 4986 |
+
single_exact_max_boundary_speaker_drop,
|
| 4987 |
+
minimum=0.0,
|
| 4988 |
+
maximum=1.0,
|
| 4989 |
+
)
|
| 4990 |
+
if single_exact_similarity is None or single_exact_boundary is None:
|
| 4991 |
+
raise ValueError(
|
| 4992 |
+
"single exact speaker thresholds must be supplied together and finite"
|
| 4993 |
+
)
|
| 4994 |
+
if not preferred_squim_enabled:
|
| 4995 |
+
raise ValueError(
|
| 4996 |
+
"single exact fast tier requires preferred SQUIM thresholds"
|
| 4997 |
+
)
|
| 4998 |
+
else:
|
| 4999 |
+
single_exact_similarity = None
|
| 5000 |
+
single_exact_boundary = None
|
| 5001 |
preferred_single_search = bool(
|
| 5002 |
preferred_speaker_enabled or preferred_squim_enabled
|
| 5003 |
)
|
|
|
|
| 5152 |
min_squim_audio_duration_seconds=preferred_squim_duration,
|
| 5153 |
)
|
| 5154 |
)
|
| 5155 |
+
initial_fast_exact = bool(
|
| 5156 |
+
chunk_count == 1
|
| 5157 |
+
and single_exact_fast_enabled
|
| 5158 |
+
and _single_exact_fast_release_evidence(
|
| 5159 |
+
initial_joined_output,
|
| 5160 |
+
initial_independent_output,
|
| 5161 |
+
min_speaker_similarity=single_exact_similarity,
|
| 5162 |
+
max_boundary_speaker_drop=single_exact_boundary,
|
| 5163 |
+
min_squim_stoi=preferred_stoi,
|
| 5164 |
+
min_squim_pesq=preferred_pesq,
|
| 5165 |
+
min_audio_duration_seconds=preferred_squim_duration,
|
| 5166 |
+
)
|
| 5167 |
+
)
|
| 5168 |
if initial_verification.passed is True and (
|
| 5169 |
+
chunk_count > 1
|
| 5170 |
+
or not preferred_single_search
|
| 5171 |
+
or initial_preferred
|
| 5172 |
+
or initial_fast_exact
|
| 5173 |
):
|
| 5174 |
return CascadeResult(
|
| 5175 |
trajectory=initial_trajectory,
|
|
|
|
| 5591 |
min_squim_audio_duration_seconds=preferred_squim_duration,
|
| 5592 |
)
|
| 5593 |
)
|
| 5594 |
+
refill_fast_exact = bool(
|
| 5595 |
+
chunk_count == 1
|
| 5596 |
+
and single_exact_fast_enabled
|
| 5597 |
+
and _single_exact_fast_release_evidence(
|
| 5598 |
+
refill_joined_output,
|
| 5599 |
+
refill_independent_output,
|
| 5600 |
+
min_speaker_similarity=single_exact_similarity,
|
| 5601 |
+
max_boundary_speaker_drop=single_exact_boundary,
|
| 5602 |
+
min_squim_stoi=preferred_stoi,
|
| 5603 |
+
min_squim_pesq=preferred_pesq,
|
| 5604 |
+
min_audio_duration_seconds=preferred_squim_duration,
|
| 5605 |
+
)
|
| 5606 |
+
)
|
| 5607 |
if (
|
| 5608 |
chunk_count == 1
|
| 5609 |
and preferred_single_search
|
| 5610 |
and refill_verification.passed is True
|
| 5611 |
+
and (refill_preferred or refill_fast_exact)
|
| 5612 |
):
|
| 5613 |
preferred_candidate = diagnostic_candidates[-1]
|
| 5614 |
preferred_artifact = preferred_candidate.verification.chunk_artifacts[0]
|
| 5615 |
preferred_reached_hard_cap = (
|
| 5616 |
_endpoint_artifact_reached_hard_cap(preferred_artifact)
|
| 5617 |
)
|
| 5618 |
+
allow_natural_endpoint_waiver = bool(
|
| 5619 |
+
refill_preferred and preferred_reached_hard_cap
|
| 5620 |
+
)
|
| 5621 |
selectable = [
|
| 5622 |
candidate
|
| 5623 |
for candidate in diagnostic_candidates
|
|
|
|
| 5646 |
),
|
| 5647 |
)
|
| 5648 |
or (
|
| 5649 |
+
single_exact_fast_enabled
|
| 5650 |
+
and _single_exact_fast_release_evidence(
|
| 5651 |
+
candidate.joined_output,
|
| 5652 |
+
candidate.independent_output,
|
| 5653 |
+
min_speaker_similarity=single_exact_similarity,
|
| 5654 |
+
max_boundary_speaker_drop=single_exact_boundary,
|
| 5655 |
+
min_squim_stoi=preferred_stoi,
|
| 5656 |
+
min_squim_pesq=preferred_pesq,
|
| 5657 |
+
min_audio_duration_seconds=(
|
| 5658 |
+
preferred_squim_duration
|
| 5659 |
+
),
|
| 5660 |
+
)
|
| 5661 |
+
)
|
| 5662 |
+
or (
|
| 5663 |
+
allow_natural_endpoint_waiver
|
| 5664 |
and _preferred_natural_endpoint_waiver(
|
| 5665 |
candidate.verification,
|
| 5666 |
candidate.verification.chunk_artifacts[0],
|
tests/test_coverage_adaptive.py
CHANGED
|
@@ -90,6 +90,7 @@ def _squim_verification(
|
|
| 90 |
speaker_similarity=0.30,
|
| 91 |
boundary_drop=0.01,
|
| 92 |
audio_duration_seconds=2.0,
|
|
|
|
| 93 |
):
|
| 94 |
observations = []
|
| 95 |
artifacts = []
|
|
@@ -118,6 +119,7 @@ def _squim_verification(
|
|
| 118 |
observations,
|
| 119 |
chunk_artifacts=artifacts,
|
| 120 |
max_pace_cps=4.3,
|
|
|
|
| 121 |
squim_gate_enabled=True,
|
| 122 |
min_squim_stoi=0.60,
|
| 123 |
min_squim_pesq=1.12,
|
|
@@ -478,6 +480,210 @@ def test_single_chunk_uses_exact_joined_preferred_evidence_for_early_return():
|
|
| 478 |
assert selected.joined_output.result.speaker_similarity == pytest.approx(0.30)
|
| 479 |
|
| 480 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 481 |
def test_preferred_refill_selects_prior_natural_endpoint_over_forced_candidate():
|
| 482 |
chunk = ("完整內容",)
|
| 483 |
generated = []
|
|
@@ -555,6 +761,90 @@ def test_preferred_refill_selects_prior_natural_endpoint_over_forced_candidate()
|
|
| 555 |
]
|
| 556 |
|
| 557 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 558 |
@pytest.mark.parametrize(
|
| 559 |
("field", "value"),
|
| 560 |
[
|
|
@@ -960,6 +1250,36 @@ def test_coverage_preferred_thresholds_must_be_explicit_finite_pairs(preferred):
|
|
| 960 |
)
|
| 961 |
|
| 962 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 963 |
def test_refill_budget_accounts_exact_generated_chunks_and_text_units():
|
| 964 |
chunks = ("甲乙", "丙丁")
|
| 965 |
generated = []
|
|
|
|
| 90 |
speaker_similarity=0.30,
|
| 91 |
boundary_drop=0.01,
|
| 92 |
audio_duration_seconds=2.0,
|
| 93 |
+
hard_max_boundary_speaker_drop=0.03,
|
| 94 |
):
|
| 95 |
observations = []
|
| 96 |
artifacts = []
|
|
|
|
| 119 |
observations,
|
| 120 |
chunk_artifacts=artifacts,
|
| 121 |
max_pace_cps=4.3,
|
| 122 |
+
max_boundary_speaker_drop=hard_max_boundary_speaker_drop,
|
| 123 |
squim_gate_enabled=True,
|
| 124 |
min_squim_stoi=0.60,
|
| 125 |
min_squim_pesq=1.12,
|
|
|
|
| 480 |
assert selected.joined_output.result.speaker_similarity == pytest.approx(0.30)
|
| 481 |
|
| 482 |
|
| 483 |
+
def test_single_chunk_initial_exact_fast_tier_returns_without_refill():
|
| 484 |
+
chunk = ("完整內容",)
|
| 485 |
+
generated = []
|
| 486 |
+
|
| 487 |
+
def whole_verifier(trajectory, candidate_chunks, seed):
|
| 488 |
+
exact = _squim_verification(
|
| 489 |
+
candidate_chunks,
|
| 490 |
+
stoi=0.75,
|
| 491 |
+
pesq=1.25,
|
| 492 |
+
speaker_similarity=0.24,
|
| 493 |
+
boundary_drop=0.075,
|
| 494 |
+
hard_max_boundary_speaker_drop=0.10,
|
| 495 |
+
)
|
| 496 |
+
return CandidateVerification(
|
| 497 |
+
exact,
|
| 498 |
+
joined_output=trajectory_gate_evidence(exact),
|
| 499 |
+
independent_output=trajectory_gate_evidence(
|
| 500 |
+
_exact_final(candidate_chunks)
|
| 501 |
+
),
|
| 502 |
+
)
|
| 503 |
+
|
| 504 |
+
result = run_coverage_adaptive_cascade(
|
| 505 |
+
chunk,
|
| 506 |
+
80,
|
| 507 |
+
lambda candidate_chunks, seed: (
|
| 508 |
+
generated.append(seed) or (f"audio-{seed}",)
|
| 509 |
+
),
|
| 510 |
+
whole_verifier,
|
| 511 |
+
lambda *args: pytest.fail("exact fast initial must not refill"),
|
| 512 |
+
sequence_final_verifier=lambda *args: pytest.fail(
|
| 513 |
+
"exact fast initial must not enter sequence search"
|
| 514 |
+
),
|
| 515 |
+
preferred_min_speaker_similarity=0.25,
|
| 516 |
+
preferred_max_boundary_speaker_drop=0.05,
|
| 517 |
+
preferred_min_squim_stoi=0.72,
|
| 518 |
+
preferred_min_squim_pesq=1.20,
|
| 519 |
+
preferred_min_squim_audio_duration_seconds=1.5,
|
| 520 |
+
single_exact_min_speaker_similarity=0.23,
|
| 521 |
+
single_exact_max_boundary_speaker_drop=0.08,
|
| 522 |
+
)
|
| 523 |
+
|
| 524 |
+
assert generated == [80]
|
| 525 |
+
assert result.candidate_index == 0
|
| 526 |
+
assert result.generated_chunk_count == 1
|
| 527 |
+
|
| 528 |
+
|
| 529 |
+
def test_single_chunk_refill_exact_fast_tier_returns_before_more_candidates():
|
| 530 |
+
chunk = ("完整內容",)
|
| 531 |
+
generated = []
|
| 532 |
+
|
| 533 |
+
def refill_verifier(trajectory, candidate_chunks, seed):
|
| 534 |
+
local = _squim_verification(
|
| 535 |
+
candidate_chunks,
|
| 536 |
+
stoi=0.75,
|
| 537 |
+
pesq=1.25,
|
| 538 |
+
speaker_similarity=0.20,
|
| 539 |
+
)
|
| 540 |
+
exact = _squim_verification(
|
| 541 |
+
candidate_chunks,
|
| 542 |
+
stoi=0.75,
|
| 543 |
+
pesq=1.25,
|
| 544 |
+
speaker_similarity=0.24,
|
| 545 |
+
boundary_drop=0.075,
|
| 546 |
+
hard_max_boundary_speaker_drop=0.10,
|
| 547 |
+
)
|
| 548 |
+
return CandidateVerification(
|
| 549 |
+
local,
|
| 550 |
+
joined_output=trajectory_gate_evidence(exact),
|
| 551 |
+
independent_output=trajectory_gate_evidence(
|
| 552 |
+
_exact_final(candidate_chunks)
|
| 553 |
+
),
|
| 554 |
+
)
|
| 555 |
+
|
| 556 |
+
result = run_coverage_adaptive_cascade(
|
| 557 |
+
chunk,
|
| 558 |
+
10,
|
| 559 |
+
lambda candidate_chunks, seed: (
|
| 560 |
+
generated.append(seed) or (f"audio-{seed}",)
|
| 561 |
+
),
|
| 562 |
+
lambda trajectory, candidate_chunks, seed: _joined_rejected_local(
|
| 563 |
+
candidate_chunks,
|
| 564 |
+
passing=(False,),
|
| 565 |
+
),
|
| 566 |
+
refill_verifier,
|
| 567 |
+
sequence_final_verifier=lambda *args: pytest.fail(
|
| 568 |
+
"exact fast refill must return before sequence search"
|
| 569 |
+
),
|
| 570 |
+
max_generated_chunks=5,
|
| 571 |
+
max_generated_text_units=100,
|
| 572 |
+
preferred_min_speaker_similarity=0.25,
|
| 573 |
+
preferred_max_boundary_speaker_drop=0.05,
|
| 574 |
+
preferred_min_squim_stoi=0.72,
|
| 575 |
+
preferred_min_squim_pesq=1.20,
|
| 576 |
+
preferred_min_squim_audio_duration_seconds=1.5,
|
| 577 |
+
single_exact_min_speaker_similarity=0.23,
|
| 578 |
+
single_exact_max_boundary_speaker_drop=0.08,
|
| 579 |
+
)
|
| 580 |
+
|
| 581 |
+
assert generated == [10, 11]
|
| 582 |
+
assert result.seed == 11
|
| 583 |
+
assert result.candidate_index == 1
|
| 584 |
+
assert result.generated_chunk_count == 2
|
| 585 |
+
|
| 586 |
+
|
| 587 |
+
def test_single_exact_fast_tier_never_uses_local_proxy():
|
| 588 |
+
chunk = ("完整內容",)
|
| 589 |
+
generated = []
|
| 590 |
+
|
| 591 |
+
def verifier(trajectory, candidate_chunks, seed):
|
| 592 |
+
if seed == 10:
|
| 593 |
+
return _squim_verification(
|
| 594 |
+
candidate_chunks,
|
| 595 |
+
stoi=0.75,
|
| 596 |
+
pesq=1.25,
|
| 597 |
+
speaker_similarity=0.24,
|
| 598 |
+
boundary_drop=0.075,
|
| 599 |
+
hard_max_boundary_speaker_drop=0.10,
|
| 600 |
+
)
|
| 601 |
+
return _squim_verification(
|
| 602 |
+
candidate_chunks,
|
| 603 |
+
stoi=0.75,
|
| 604 |
+
pesq=1.25,
|
| 605 |
+
speaker_similarity=0.30,
|
| 606 |
+
boundary_drop=0.02,
|
| 607 |
+
)
|
| 608 |
+
|
| 609 |
+
result = run_coverage_adaptive_cascade(
|
| 610 |
+
chunk,
|
| 611 |
+
10,
|
| 612 |
+
lambda candidate_chunks, seed: (
|
| 613 |
+
generated.append(seed) or (f"audio-{seed}",)
|
| 614 |
+
),
|
| 615 |
+
verifier,
|
| 616 |
+
verifier,
|
| 617 |
+
sequence_final_verifier=lambda *args: pytest.fail(
|
| 618 |
+
"strict preferred refill must return before sequence search"
|
| 619 |
+
),
|
| 620 |
+
max_generated_chunks=5,
|
| 621 |
+
max_generated_text_units=100,
|
| 622 |
+
preferred_min_speaker_similarity=0.25,
|
| 623 |
+
preferred_max_boundary_speaker_drop=0.05,
|
| 624 |
+
preferred_min_squim_stoi=0.72,
|
| 625 |
+
preferred_min_squim_pesq=1.20,
|
| 626 |
+
preferred_min_squim_audio_duration_seconds=1.5,
|
| 627 |
+
single_exact_min_speaker_similarity=0.23,
|
| 628 |
+
single_exact_max_boundary_speaker_drop=0.08,
|
| 629 |
+
)
|
| 630 |
+
|
| 631 |
+
assert generated == [10, 11]
|
| 632 |
+
assert result.seed == 11
|
| 633 |
+
|
| 634 |
+
|
| 635 |
+
def test_single_exact_fast_tier_requires_independent_whole_asr_evidence():
|
| 636 |
+
chunk = ("完整內容",)
|
| 637 |
+
generated = []
|
| 638 |
+
|
| 639 |
+
def verifier(trajectory, candidate_chunks, seed):
|
| 640 |
+
if seed == 10:
|
| 641 |
+
exact = _squim_verification(
|
| 642 |
+
candidate_chunks,
|
| 643 |
+
stoi=0.75,
|
| 644 |
+
pesq=1.25,
|
| 645 |
+
speaker_similarity=0.24,
|
| 646 |
+
boundary_drop=0.075,
|
| 647 |
+
hard_max_boundary_speaker_drop=0.10,
|
| 648 |
+
)
|
| 649 |
+
return CandidateVerification(
|
| 650 |
+
exact,
|
| 651 |
+
joined_output=trajectory_gate_evidence(exact),
|
| 652 |
+
)
|
| 653 |
+
return _squim_verification(
|
| 654 |
+
candidate_chunks,
|
| 655 |
+
stoi=0.75,
|
| 656 |
+
pesq=1.25,
|
| 657 |
+
speaker_similarity=0.30,
|
| 658 |
+
boundary_drop=0.02,
|
| 659 |
+
)
|
| 660 |
+
|
| 661 |
+
result = run_coverage_adaptive_cascade(
|
| 662 |
+
chunk,
|
| 663 |
+
10,
|
| 664 |
+
lambda candidate_chunks, seed: (
|
| 665 |
+
generated.append(seed) or (f"audio-{seed}",)
|
| 666 |
+
),
|
| 667 |
+
verifier,
|
| 668 |
+
verifier,
|
| 669 |
+
sequence_final_verifier=lambda *args: pytest.fail(
|
| 670 |
+
"strict preferred refill must return before sequence search"
|
| 671 |
+
),
|
| 672 |
+
max_generated_chunks=5,
|
| 673 |
+
max_generated_text_units=100,
|
| 674 |
+
preferred_min_speaker_similarity=0.25,
|
| 675 |
+
preferred_max_boundary_speaker_drop=0.05,
|
| 676 |
+
preferred_min_squim_stoi=0.72,
|
| 677 |
+
preferred_min_squim_pesq=1.20,
|
| 678 |
+
preferred_min_squim_audio_duration_seconds=1.5,
|
| 679 |
+
single_exact_min_speaker_similarity=0.23,
|
| 680 |
+
single_exact_max_boundary_speaker_drop=0.08,
|
| 681 |
+
)
|
| 682 |
+
|
| 683 |
+
assert generated == [10, 11]
|
| 684 |
+
assert result.seed == 11
|
| 685 |
+
|
| 686 |
+
|
| 687 |
def test_preferred_refill_selects_prior_natural_endpoint_over_forced_candidate():
|
| 688 |
chunk = ("完整內容",)
|
| 689 |
generated = []
|
|
|
|
| 761 |
]
|
| 762 |
|
| 763 |
|
| 764 |
+
def test_fast_exact_refill_cannot_select_relaxed_endpoint_waiver_candidate():
|
| 765 |
+
chunk = ("完整內容",)
|
| 766 |
+
generated = []
|
| 767 |
+
|
| 768 |
+
def verifier(trajectory, candidate_chunks, seed):
|
| 769 |
+
if seed == 10:
|
| 770 |
+
return _squim_verification(
|
| 771 |
+
candidate_chunks,
|
| 772 |
+
stoi=0.65,
|
| 773 |
+
pesq=1.15,
|
| 774 |
+
speaker_similarity=0.30,
|
| 775 |
+
)
|
| 776 |
+
if seed == 11:
|
| 777 |
+
return _squim_verification(
|
| 778 |
+
candidate_chunks,
|
| 779 |
+
stoi=0.71,
|
| 780 |
+
pesq=1.19,
|
| 781 |
+
speaker_similarity=0.30,
|
| 782 |
+
)
|
| 783 |
+
exact = _squim_verification(
|
| 784 |
+
candidate_chunks,
|
| 785 |
+
stoi=0.75,
|
| 786 |
+
pesq=1.25,
|
| 787 |
+
speaker_similarity=0.24,
|
| 788 |
+
boundary_drop=0.075,
|
| 789 |
+
hard_max_boundary_speaker_drop=0.10,
|
| 790 |
+
)
|
| 791 |
+
return CandidateVerification(
|
| 792 |
+
exact,
|
| 793 |
+
joined_output=trajectory_gate_evidence(exact),
|
| 794 |
+
independent_output=trajectory_gate_evidence(
|
| 795 |
+
_exact_final(candidate_chunks)
|
| 796 |
+
),
|
| 797 |
+
)
|
| 798 |
+
|
| 799 |
+
def generation_evidence(candidate_index, seed, chunk_indices, candidate_chunks):
|
| 800 |
+
threshold_stop = candidate_index == 1
|
| 801 |
+
return CandidateGenerationEvidence(
|
| 802 |
+
chunk_indices=chunk_indices,
|
| 803 |
+
chunk_text_units=(4,),
|
| 804 |
+
scheduled_cfg=generation_cfg_for_candidate_offset(candidate_index),
|
| 805 |
+
effective_cfgs=(
|
| 806 |
+
generation_cfg_for_candidate_offset(candidate_index),
|
| 807 |
+
),
|
| 808 |
+
floor_reasons=((),),
|
| 809 |
+
chunk_stop_reasons=(
|
| 810 |
+
("stop_threshold",) if threshold_stop else ("hard_stop",)
|
| 811 |
+
),
|
| 812 |
+
chunk_endpoint_energy_ratios=(
|
| 813 |
+
(0.10,) if threshold_stop else (0.90,)
|
| 814 |
+
),
|
| 815 |
+
chunk_generated_steps=((4,) if threshold_stop else (5,)),
|
| 816 |
+
chunk_hard_stop_steps=(5,),
|
| 817 |
+
)
|
| 818 |
+
|
| 819 |
+
result = run_coverage_adaptive_cascade(
|
| 820 |
+
chunk,
|
| 821 |
+
10,
|
| 822 |
+
lambda candidate_chunks, seed: (
|
| 823 |
+
generated.append(seed) or (f"audio-{seed}",)
|
| 824 |
+
),
|
| 825 |
+
verifier,
|
| 826 |
+
verifier,
|
| 827 |
+
sequence_final_verifier=lambda *args: pytest.fail(
|
| 828 |
+
"fast exact refill must return before sequence search"
|
| 829 |
+
),
|
| 830 |
+
generation_evidence_factory=generation_evidence,
|
| 831 |
+
max_generated_chunks=5,
|
| 832 |
+
max_generated_text_units=100,
|
| 833 |
+
preferred_min_speaker_similarity=0.25,
|
| 834 |
+
preferred_max_boundary_speaker_drop=0.05,
|
| 835 |
+
preferred_min_squim_stoi=0.72,
|
| 836 |
+
preferred_min_squim_pesq=1.20,
|
| 837 |
+
preferred_min_squim_audio_duration_seconds=1.5,
|
| 838 |
+
single_exact_min_speaker_similarity=0.23,
|
| 839 |
+
single_exact_max_boundary_speaker_drop=0.08,
|
| 840 |
+
require_endpoint_evidence=True,
|
| 841 |
+
)
|
| 842 |
+
|
| 843 |
+
assert generated == [10, 11, 12]
|
| 844 |
+
assert result.candidate_index == 2
|
| 845 |
+
assert result.seed == 12
|
| 846 |
+
|
| 847 |
+
|
| 848 |
@pytest.mark.parametrize(
|
| 849 |
("field", "value"),
|
| 850 |
[
|
|
|
|
| 1250 |
)
|
| 1251 |
|
| 1252 |
|
| 1253 |
+
@pytest.mark.parametrize(
|
| 1254 |
+
"single_exact",
|
| 1255 |
+
[
|
| 1256 |
+
{"single_exact_min_speaker_similarity": 0.23},
|
| 1257 |
+
{"single_exact_max_boundary_speaker_drop": 0.08},
|
| 1258 |
+
{
|
| 1259 |
+
"single_exact_min_speaker_similarity": float("nan"),
|
| 1260 |
+
"single_exact_max_boundary_speaker_drop": 0.08,
|
| 1261 |
+
},
|
| 1262 |
+
{
|
| 1263 |
+
"single_exact_min_speaker_similarity": 0.23,
|
| 1264 |
+
"single_exact_max_boundary_speaker_drop": 0.08,
|
| 1265 |
+
},
|
| 1266 |
+
],
|
| 1267 |
+
)
|
| 1268 |
+
def test_single_exact_fast_thresholds_require_pair_and_preferred_squim(
|
| 1269 |
+
single_exact,
|
| 1270 |
+
):
|
| 1271 |
+
with pytest.raises(ValueError, match="single exact"):
|
| 1272 |
+
run_coverage_adaptive_cascade(
|
| 1273 |
+
("完整內容",),
|
| 1274 |
+
10,
|
| 1275 |
+
lambda chunks, seed: tuple(chunks),
|
| 1276 |
+
lambda trajectory, chunks, seed: _local_verification(chunks),
|
| 1277 |
+
lambda trajectory, chunks, seed: _local_verification(chunks),
|
| 1278 |
+
sequence_final_verifier=lambda result, chunks: _exact_final(chunks),
|
| 1279 |
+
**single_exact,
|
| 1280 |
+
)
|
| 1281 |
+
|
| 1282 |
+
|
| 1283 |
def test_refill_budget_accounts_exact_generated_chunks_and_text_units():
|
| 1284 |
chunks = ("甲乙", "丙丁")
|
| 1285 |
generated = []
|
tests/test_release_pins.py
CHANGED
|
@@ -1102,6 +1102,8 @@ def test_space_applies_pinned_squim_to_local_joined_and_final_audio():
|
|
| 1102 |
assert 'QUALITY_PREFERRED_MIN_SQUIM_PESQ = 1.20' in app_source
|
| 1103 |
assert 'QUALITY_PREFERRED_MIN_SPEAKER_SIMILARITY = 0.25' in app_source
|
| 1104 |
assert 'QUALITY_PREFERRED_MAX_BOUNDARY_SPEAKER_DROP = 0.05' in app_source
|
|
|
|
|
|
|
| 1105 |
assert "if not semantic_only and transcript:" in verify_source
|
| 1106 |
score_index = verify_source.index("squim_objective_evidence_from_audio(")
|
| 1107 |
observation_index = verify_source.index("CandidateObservation(", score_index)
|
|
@@ -1138,6 +1140,8 @@ def test_space_applies_pinned_squim_to_local_joined_and_final_audio():
|
|
| 1138 |
"preferred_min_squim_stoi=QUALITY_PREFERRED_MIN_SQUIM_STOI",
|
| 1139 |
"preferred_min_squim_pesq=QUALITY_PREFERRED_MIN_SQUIM_PESQ",
|
| 1140 |
"QUALITY_PREFERRED_SQUIM_MIN_DURATION_SECONDS",
|
|
|
|
|
|
|
| 1141 |
):
|
| 1142 |
assert argument in synthesize_source
|
| 1143 |
assert "QUALITY_PREFERRED_SQUIM_MIN_DURATION_SECONDS = 1.50" in app_source
|
|
|
|
| 1102 |
assert 'QUALITY_PREFERRED_MIN_SQUIM_PESQ = 1.20' in app_source
|
| 1103 |
assert 'QUALITY_PREFERRED_MIN_SPEAKER_SIMILARITY = 0.25' in app_source
|
| 1104 |
assert 'QUALITY_PREFERRED_MAX_BOUNDARY_SPEAKER_DROP = 0.05' in app_source
|
| 1105 |
+
assert 'QUALITY_FAST_EXACT_MIN_SPEAKER_SIMILARITY = 0.23' in app_source
|
| 1106 |
+
assert 'QUALITY_FAST_EXACT_MAX_BOUNDARY_SPEAKER_DROP = 0.08' in app_source
|
| 1107 |
assert "if not semantic_only and transcript:" in verify_source
|
| 1108 |
score_index = verify_source.index("squim_objective_evidence_from_audio(")
|
| 1109 |
observation_index = verify_source.index("CandidateObservation(", score_index)
|
|
|
|
| 1140 |
"preferred_min_squim_stoi=QUALITY_PREFERRED_MIN_SQUIM_STOI",
|
| 1141 |
"preferred_min_squim_pesq=QUALITY_PREFERRED_MIN_SQUIM_PESQ",
|
| 1142 |
"QUALITY_PREFERRED_SQUIM_MIN_DURATION_SECONDS",
|
| 1143 |
+
"single_exact_min_speaker_similarity=",
|
| 1144 |
+
"single_exact_max_boundary_speaker_drop=",
|
| 1145 |
):
|
| 1146 |
assert argument in synthesize_source
|
| 1147 |
assert "QUALITY_PREFERRED_SQUIM_MIN_DURATION_SECONDS = 1.50" in app_source
|