import ast import hashlib from pathlib import Path from types import SimpleNamespace import numpy as np from production import ( GenerationChunkSpec, count_speech_units, fade_variable_internal_edges, join_audio_chunks_variable, match_chunk_rms, punctuation_pause_seconds, ) from quality_runtime import LocalIndependentGateEvidence ROOT = Path(__file__).resolve().parents[1] FROZEN_SPEAKER_ANCHORS = { "aaf1a0878e37875382bb0e5c8a3a2ba43be67297": { "speaker_id": "female_voice", "speaker_index": 1, "ui_label": "內建語者 B", "dtype": "float32", "shape": (192,), "sha256": ( "e33e4cb6a741d4d1237aa4ff557f1e663d0a6427dccda1f51e83bc149d4188ca" ), } } def _string_constants(path: Path) -> dict[str, str]: tree = ast.parse(path.read_text(encoding="utf-8")) values: dict[str, str] = {} for node in tree.body: if not isinstance(node, ast.Assign) or len(node.targets) != 1: continue target = node.targets[0] if isinstance(target, ast.Name) and isinstance(node.value, ast.Constant): if isinstance(node.value.value, str): values[target.id] = node.value.value return values def _literal_constants(path: Path) -> dict[str, object]: tree = ast.parse(path.read_text(encoding="utf-8")) values: dict[str, object] = {} for node in tree.body: if not isinstance(node, ast.Assign) or len(node.targets) != 1: continue target = node.targets[0] if not isinstance(target, ast.Name): continue try: values[target.id] = ast.literal_eval(node.value) except (TypeError, ValueError): continue return values def _isolated_assemble_trajectory_audio(): """Execute only the production assembly function without loading the model.""" app_path = ROOT / "app.py" tree = ast.parse(app_path.read_text(encoding="utf-8")) function = next( node for node in tree.body if isinstance(node, ast.FunctionDef) and node.name == "_assemble_trajectory_audio" ) module = ast.Module( body=[ ast.ImportFrom( module="__future__", names=[ast.alias(name="annotations")], level=0, ), function, ], type_ignores=[], ) ast.fix_missing_locations(module) constants = _literal_constants(app_path) namespace = { "np": np, "GenerationChunkSpec": GenerationChunkSpec, "SR": 1_000, "CHUNK_RMS_MATCH_DB": constants["CHUNK_RMS_MATCH_DB"], "SEMANTIC_CHUNK_MIN_SILENCE_MS": constants[ "SEMANTIC_CHUNK_MIN_SILENCE_MS" ], "NETWORK_REQUEST_SEMANTIC_CHUNK_MIN_SILENCE_MS": constants[ "NETWORK_REQUEST_SEMANTIC_CHUNK_MIN_SILENCE_MS" ], "NETWORK_INTERNAL_SILENCE_MS": constants["NETWORK_INTERNAL_SILENCE_MS"], "NETWORK_INTERNAL_FADE_MS": constants["NETWORK_INTERNAL_FADE_MS"], "CHUNK_EDGE_FADE_MS": constants["CHUNK_EDGE_FADE_MS"], "CROSSFADE_MS": constants["CROSSFADE_MS"], "match_chunk_rms": match_chunk_rms, "punctuation_pause_seconds": punctuation_pause_seconds, "fade_variable_internal_edges": fade_variable_internal_edges, "join_audio_chunks_variable": join_audio_chunks_variable, "apply_loudness_floor": lambda waveform, **_kwargs: waveform, "_apply_speed": lambda waveform, _speed: waveform, "count_speech_units": count_speech_units, "finish_audio": lambda waveform, _sample_rate, **_kwargs: waveform, } exec(compile(module, str(app_path), "exec"), namespace) return namespace["_assemble_trajectory_audio"] def _isolated_load_speakers(*, metadata: dict, speaker_ids: tuple[str, ...]): """Execute only ``_load_speakers`` without importing the GPU application.""" app_path = ROOT / "app.py" tree = ast.parse(app_path.read_text(encoding="utf-8")) function = next( node for node in tree.body if isinstance(node, ast.FunctionDef) and node.name == "_load_speakers" ) module = ast.Module( body=[ ast.ImportFrom( module="__future__", names=[ast.alias(name="annotations")], level=0, ), function, ], type_ignores=[], ) ast.fix_missing_locations(module) load_calls = [] centroids = tuple(f"test-centroid-{index}" for index in range(len(speaker_ids))) def fake_load(path, **kwargs): load_calls.append((path, kwargs)) return {"speaker_ids": speaker_ids, "centroids": centroids} namespace = { "MODEL_DIR": "/pinned/model", "METADATA": metadata, "os": SimpleNamespace( path=SimpleNamespace( join=lambda *parts: "/".join(part.strip("/") for part in parts), exists=lambda _path: True, ) ), "torch": SimpleNamespace(load=fake_load), } exec(compile(module, app_path, "exec"), namespace) result = namespace["_load_speakers"]() return result, load_calls def test_missing_metadata_speaker_id_selects_frozen_female_voice_as_builtin_b(): constants = _string_constants(ROOT / "app.py") contract = FROZEN_SPEAKER_ANCHORS[constants["MODEL_REVISION"]] speaker_ids = ("hung_yi_lee", contract["speaker_id"]) (labels, default_label), load_calls = _isolated_load_speakers( metadata={}, speaker_ids=speaker_ids, ) assert contract == { "speaker_id": "female_voice", "speaker_index": 1, "ui_label": "內建語者 B", "dtype": "float32", "shape": (192,), "sha256": ( "e33e4cb6a741d4d1237aa4ff557f1e663d0a6427dccda1f51e83bc149d4188ca" ), } assert speaker_ids[contract["speaker_index"]] == contract["speaker_id"] assert tuple(labels) == ("內建語者 A", contract["ui_label"]) assert labels[contract["ui_label"]] == "test-centroid-1" assert default_label == contract["ui_label"] assert load_calls == [ ( "pinned/model/checkpoints/speaker_centroids.pt", {"map_location": "cpu", "weights_only": True}, ) ] def test_remote_model_and_speaker_encoder_are_revision_pinned(): app_path = ROOT / "app.py" source = app_path.read_text(encoding="utf-8") constants = _string_constants(app_path) assert constants["MODEL_REVISION"] == "aaf1a0878e37875382bb0e5c8a3a2ba43be67297" assert constants["ECAPA_REVISION"] == "0f99f2d0ebe89ac095bcc5903c4dd8f72b367286" assert "snapshot_download(REPO_ID, revision=MODEL_REVISION)" in source assert "snapshot_download(ECAPA_REPO_ID, revision=ECAPA_REVISION)" in source assert "source=ECAPA_DIR" in source assert 'overrides={"pretrained_path": ECAPA_DIR}' in source def test_tts_runtime_is_vendored_from_the_frozen_commit(): requirements = (ROOT / "requirements.txt").read_text(encoding="utf-8").splitlines() provenance = (ROOT / "bluemagpie" / "UPSTREAM_RUNTIME.md").read_text( encoding="utf-8" ) assert not any("BlueMagpie-TTS.git" in line for line in requirements) assert "ce384c8cc54efea1aaba7b9f1d7ded6c1c99aa9a" in provenance assert (ROOT / "bluemagpie" / "LICENSE.upstream").is_file() assert (ROOT / "bluemagpie" / "_vendor" / "voxcpm" / "LICENSE").is_file() pinned_hashes = { "model.py": "91810524212b34f727880154d90653fab4ae1b75eb3471b86cafd92c75514fef", "loading.py": "e3407544e9bc888018fe5771edc01d954469e2af548b566873e0ef7f2afe6dca", "_vendor/voxcpm/model/utils.py": ( "cea16e1ab57f15129a7f5dec13c428bd14a771221abcf17dd0a90b3d65e763a2" ), } for relative_path, expected_hash in pinned_hashes.items(): payload = (ROOT / "bluemagpie" / relative_path).read_bytes() assert hashlib.sha256(payload).hexdigest() == expected_hash def test_barbet_runtime_dependency_is_commit_pinned(): requirements = (ROOT / "requirements.txt").read_text(encoding="utf-8").splitlines() barbet_lines = [line for line in requirements if "OpenFormosa/Barbet.git" in line] assert barbet_lines == [ "git+https://github.com/OpenFormosa/Barbet.git@" "6fcd7ce4aa37f2250a3242995bef0fbc3b026ba8" ] def test_quality_asr_is_revision_pinned(): quality_path = ROOT / "quality_runtime.py" quality_source = quality_path.read_text(encoding="utf-8") constants = _string_constants(quality_path) app_source = (ROOT / "app.py").read_text(encoding="utf-8") assert constants["WHISPER_MODEL_ID"] == "openai/whisper-large-v3-turbo" assert constants["WHISPER_REVISION"] == "41f01f3fe87f28c78e2fbf8b568835947dd65ed9" assert constants["VERIFICATION_WHISPER_MODEL_ID"] == "openai/whisper-large-v3" assert ( constants["VERIFICATION_WHISPER_REVISION"] == "06f233fe06e710322aca913c1bc4249a0d71fce1" ) assert constants["WHISPER_ATTENTION_IMPLEMENTATION"] == "eager" assert "WHISPER_RETURN_ATTENTION_MASK = True" in quality_source assert "snapshot_download(WHISPER_MODEL_ID, revision=WHISPER_REVISION)" in app_source assert "snapshot_download(\n VERIFICATION_WHISPER_MODEL_ID," in app_source assert "revision=VERIFICATION_WHISPER_REVISION" in app_source assert "load_pinned_verification_whisper_runtime" in quality_source assert "transcribe_verification_whisper" in quality_source def test_quality_runtime_dependencies_are_version_pinned(): requirements = set( (ROOT / "requirements.txt").read_text(encoding="utf-8").splitlines() ) assert "huggingface_hub==0.36.0" in requirements assert "opencc-python-reimplemented==0.1.7" in requirements assert "pypinyin==0.55.0" in requirements assert "transformers==4.57.6" in requirements assert "accelerate==1.12.0" in requirements assert "einops==0.8.2" in requirements assert "pydantic==2.11.10" in requirements assert "numpy==2.3.5" in requirements assert "scipy==1.17.1" in requirements assert "numexpr==2.14.1" in requirements assert "bottleneck==1.6.0" in requirements assert "tqdm==4.68.2" in requirements assert "safetensors==0.8.0" in requirements assert "librosa==0.11.0" in requirements assert "soundfile==0.14.0" in requirements assert "speechbrain==1.0.3" in requirements def test_readme_describes_coverage_refill_and_sequence_transition_scores(): readme = (ROOT / "README.md").read_text(encoding="utf-8") app_source = (ROOT / "app.py").read_text(encoding="utf-8") assert "exactly one same-seed whole trajectory" in readme assert "single-chunk refills for zero/low-coverage rows only" in readme assert "`(coverage, refill attempts, chunk index)`" in readme assert "32 generated TTS chunks、800 generated speech units" in readme assert "不使用 reference-style distribution score" in readme assert "median-F0 軟成本" in readme assert "5.2 CJK / 4.6 ASCII" in readme assert "4.6 CJK / 4.0 ASCII" in readme assert "URL/email-bearing chunks 8 units" in readme assert "Email 第一輪必須在 `小老鼠`" in readme assert "第一輪 DP 不得跨越這兩類" in readme assert "grammar boundary" in readme assert "所有最佳 edit alignment" in readme assert "boundary-only local rejects" in readme assert "drop 不超過 0.15" in readme assert "整段仍必須通過 similarity 0.105 與 boundary drop 0.095" in readme assert "zero/low-coverage 單 chunk refill" in app_source assert "speaker/RMS/F0 ragged DP" in app_source assert "1→5→10→15→20" not in app_source def test_app_wires_row_local_candidate_ordinal_to_generation_policy_and_logs_it(): source = (ROOT / "app.py").read_text(encoding="utf-8") assert source.count("policy: GenerationPolicy") == 2 assert "generation_context.chunk_candidate_ordinals" in source assert "policy=generation_policy_for_candidate_offset(candidate_ordinal)" in source assert "generation_context.seed != seed" in source assert 'f"name={policy.name}' in source assert "chunk_policies={selected_policies}" in source assert '"min_len": min_len' in source assert '"[BlueMagpie] generation attempt "' in source assert "scheduled_cfg={scheduled_cfg:.2f}" in source assert "effective_cfg={effective_cfg:.2f}" in source assert "network_floor_applied={network_floor_applied}" in source assert "short_floor_applied={short_floor_applied}" in source def test_app_rejects_ambiguous_iri_before_frontend_normalization(): source = (ROOT / "app.py").read_text(encoding="utf-8") synthesize_start = source.index("def _synthesize(") raw_text = source.index("raw_text = str(text)", synthesize_start) iri_guard = source.index( "if network_identifier_has_ambiguous_iri(raw_text):", synthesize_start, ) normalization = source.index( 'text = normalize_spoken_forms(raw_text, locale="zh-TW")', synthesize_start, ) assert synthesize_start < raw_text < iri_guard < normalization assert "非 ASCII IRI 必須先轉成 ASCII/percent-encoded" in ( ROOT / "README.md" ).read_text(encoding="utf-8") def test_app_does_not_add_an_artificial_onset_split(): source = (ROOT / "app.py").read_text(encoding="utf-8") tree = ast.parse(source) functions = { node.name: node for node in tree.body if isinstance(node, ast.FunctionDef) } synthesize_source = ast.get_source_segment(source, functions["_synthesize"]) assert synthesize_source is not None assert "split_text_for_tts(" in synthesize_source assert "split_leading_clause(" not in synthesize_source assert "ONSET_CLAUSE_SEARCH_CHARS = 0" in source def test_app_applies_fixed_mixed_cfg_schedule_after_global_quality_floor(): source = (ROOT / "app.py").read_text(encoding="utf-8") tree = ast.parse(source) functions = { node.name: node for node in tree.body if isinstance(node, ast.FunctionDef) } synthesize_source = ast.get_source_segment(source, functions["_synthesize"]) assert synthesize_source is not None assert "DEFAULT_CFG = 3.0" in source assert "QUALITY_CFG_MIN" not in source assert "MIXED_CFG_PRIMARY = 3.0" in source assert "MIXED_CFG_ALTERNATE = 2.0" in source assert ( 'MIXED_CFG_SCHEDULE = "row_ordinal_zero_and_even_primary_odd_alternate"' in source ) assert "network_request = contains_network_identifier(raw_text)" in synthesize_source assert "cfg_value != MIXED_CFG_PRIMARY" in synthesize_source assert "request_cfg = MIXED_CFG_PRIMARY" in synthesize_source assert "cfg=candidate_cfg(candidate_ordinal)" in synthesize_source assert "candidate_ordinals = generation_context.chunk_candidate_ordinals" in ( synthesize_source ) assert "chunk_candidate_ordinals=candidate_ordinals" in synthesize_source assert "generation_cfg_for_candidate_offset(" in synthesize_source assert "chunk_cfgs={selected_cfgs}" in synthesize_source assert "attempted_schedule_cfgs={attempted_schedule_cfgs}" in synthesize_source assert "network_cfg_floor={NETWORK_TEXT_CFG_MIN:.2f}" in synthesize_source assert "mixed_cfg_primary={MIXED_CFG_PRIMARY:.2f}" in synthesize_source assert "np.isfinite(cfg_value)" in synthesize_source assert "1.0 <= cfg_value <= 4.0" in synthesize_source assert 'label="CFG (已驗證固定值)"' in source def test_app_and_quality_runtime_pin_the_same_mixed_cfg_contract(): app_constants = _literal_constants(ROOT / "app.py") quality_constants = _literal_constants(ROOT / "quality_runtime.py") assert app_constants["MIXED_CFG_SCHEDULE"] == quality_constants[ "MIXED_CFG_SCHEDULE" ] assert app_constants["MIXED_CFG_PRIMARY"] == quality_constants[ "MIXED_CFG_PRIMARY" ] assert app_constants["MIXED_CFG_ALTERNATE"] == quality_constants[ "MIXED_CFG_ALTERNATE" ] assert app_constants["SHORT_TEXT_CFG_UNITS"] == quality_constants[ "MIXED_CFG_SHORT_TEXT_MAX_UNITS" ] assert app_constants["SHORT_TEXT_CFG_MIN"] == quality_constants[ "MIXED_CFG_SHORT_TEXT_MIN" ] assert app_constants["NETWORK_TEXT_CFG_MIN"] == quality_constants[ "MIXED_CFG_NETWORK_MIN" ] def test_app_rejects_silent_text_and_coalesces_before_runtime_budgeting(): source = (ROOT / "app.py").read_text(encoding="utf-8") production_source = (ROOT / "production.py").read_text(encoding="utf-8") tree = ast.parse(source) functions = { node.name: node for node in tree.body if isinstance(node, ast.FunctionDef) } synthesize_source = ast.get_source_segment(source, functions["_synthesize"]) assemble_source = ast.get_source_segment( source, functions["_assemble_trajectory_audio"], ) assert synthesize_source is not None assert assemble_source is not None assert "if count_speech_units(text) <= 0" in synthesize_source assert "coalesce_text_chunks(" in synthesize_source assert "chunk_specs = plan_generation_chunks(" in synthesize_source assert "chunks = tuple(spec.text for spec in chunk_specs)" in synthesize_source assert "max_chunks=QUALITY_MAX_GENERATED_CHUNKS" in synthesize_source assert "pre_faded_edges=True" in assemble_source assert "NETWORK_GENERATION_MIN_UNITS = 8" in source assert "NETWORK_GENERATION_TARGET_UNITS = 32" in source assert "NETWORK_GENERATION_MAX_UNITS = 36" in source assert "network_min_units=NETWORK_GENERATION_MIN_UNITS" in synthesize_source assert "NETWORK_INTERNAL_FADE_MS = 5.0" in source assert "NETWORK_INTERNAL_SILENCE_MS = 400.0" in source assert "SEMANTIC_CHUNK_MIN_SILENCE_MS = 250.0" in source assert "NETWORK_REQUEST_SEMANTIC_CHUNK_MIN_SILENCE_MS = 350.0" in source assert 'chunk_specs[index].boundary_after == "network_internal"' in ( assemble_source ) assert "NETWORK_INTERNAL_SILENCE_MS / 1000.0" in assemble_source assert "punctuation_pause_seconds(chunk)" in assemble_source assert "semantic_min_silence_ms / 1000.0" in assemble_source assert "pauses.append(int(round(pause_seconds * SR)))" in assemble_source assert "join_audio_chunks_variable(" in assemble_source assert "network_conditioned=network_flags" in synthesize_source assert "network_conditioned=network_flag" in synthesize_source assert "mandatory_cut_offsets" in production_source assert "any(start < cut < end for cut in active_mandatory_cuts)" in ( production_source ) assert "plan = solve(mandatory_cuts - short_identifier_cuts)" in ( production_source ) assert "_protected_ranges_are_exact_in_all_optimal_alignments(" in ( production_source ) def test_network_endpoint_headroom_is_isolated_from_public_unit_contracts(): app_source = (ROOT / "app.py").read_text(encoding="utf-8") production_source = (ROOT / "production.py").read_text(encoding="utf-8") app_tree = ast.parse(app_source) production_tree = ast.parse(production_source) app_functions = { node.name: node for node in app_tree.body if isinstance(node, ast.FunctionDef) } production_functions = { node.name: node for node in production_tree.body if isinstance(node, ast.FunctionDef) } generate_source = ast.get_source_segment( app_source, app_functions["_generate_chunk"], ) public_counter_source = ast.get_source_segment( production_source, production_functions["count_speech_units"], ) network_counter_source = ast.get_source_segment( production_source, production_functions["count_network_endpoint_duration_units"], ) assert generate_source is not None assert public_counter_source is not None assert network_counter_source is not None assert "count_network_endpoint_duration_units(text)" in generate_source assert "if network_conditioned" in generate_source assert "duration_units=endpoint_duration_units" in generate_source assert "min_len = 2" in generate_source assert '"max_len": hard_stop_steps' in generate_source assert "expected_steps=expected_steps" in generate_source assert "hard_stop_steps=hard_stop_steps" in generate_source assert "duration_counter=" in generate_source assert "divisor = 2 if token.isdigit() else 4" in public_counter_source assert "math.ceil(ascii_run_length / 2)" in network_counter_source assert production_source.count( "count_network_endpoint_duration_units(" ) == 1 def test_network_local_dual_asr_capability_is_range_bound_for_initial_and_refill(): app_source = (ROOT / "app.py").read_text(encoding="utf-8") quality_source = (ROOT / "quality_runtime.py").read_text(encoding="utf-8") app_tree = ast.parse(app_source) quality_tree = ast.parse(quality_source) app_functions = { node.name: node for node in app_tree.body if isinstance(node, ast.FunctionDef) } quality_functions = { node.name: node for node in quality_tree.body if isinstance(node, ast.FunctionDef) } helper_source = ast.get_source_segment( app_source, app_functions["_verify_network_local_asr_intersection"], ) initial_source = ast.get_source_segment( app_source, app_functions["_qualify_candidate_trajectory_audio"], ) refill_source = ast.get_source_segment( app_source, app_functions["_verify_refill_candidate_trajectory_audio"], ) intersection_source = ast.get_source_segment( quality_source, quality_functions["intersect_local_semantic_verification"], ) assert all( source is not None for source in ( helper_source, initial_source, refill_source, intersection_source, ) ) assert "proof_rows[index] for index in selected_indices" in helper_source assert "transcriber=transcribe_verification_whisper" in helper_source assert "semantic_only=True" in helper_source assert "network_fragment_proofs=independent_proof_rows" in helper_source assert "intersect_local_semantic_verification(" in helper_source assert "[BlueMagpie] network local independent " in helper_source assert "proof_count=" in helper_source assert "transcript_text" not in helper_source for caller_source in (initial_source, refill_source): turbo_index = caller_source.index( "local_verification = _verify_trajectory_audio(" ) intersection_index = caller_source.index( "_verify_network_local_asr_intersection(" ) assert turbo_index < intersection_index assert "proof_rows = _network_fragment_proof_rows(" in caller_source assert "network_fragment_proofs=proof_rows" in caller_source assert "proof_rows," in caller_source[intersection_index:] assert "candidate_seed=" in caller_source[intersection_index:] assert "semantic_reasons = [\"semantic_gate\"]" in intersection_source assert "\"network_protected_span_mismatch\"" in intersection_source assert "chunk_artifacts=primary_verification.chunk_artifacts" in ( intersection_source ) assert "CASCADE_EVIDENCE_SCHEMA_VERSION = 5" in quality_source assert '"chunk_text_variants"' in quality_source assert "local_candidate_has_coverage_eligibility(" in helper_source assert "independent_local_results=" in initial_source assert "independent_local_results=" in refill_source assert '"independent_local_evidence_complete"' in quality_source assert '"independent_local_results"' in quality_source def test_network_local_dual_asr_runtime_reuses_exact_proof_rows_and_logs_no_text( capsys, ): source = (ROOT / "app.py").read_text(encoding="utf-8") tree = ast.parse(source) function = next( node for node in tree.body if ( isinstance(node, ast.FunctionDef) and node.name == "_verify_network_local_asr_intersection" ) ) module = ast.Module( body=[ ast.ImportFrom( module="__future__", names=[ast.alias(name="annotations")], level=0, ), function, ], type_ignores=[], ) ast.fix_missing_locations(module) calls = [] proof = object() skipped_proof = object() proof_rows = ((), (proof,), (skipped_proof,)) primary_results = (object(), object(), object()) turbo = SimpleNamespace(candidate_results=primary_results) independent_result = SimpleNamespace( passed=True, rejection_reasons=(), comparison=SimpleNamespace( cer=0.0, prefix_cer=0.0, suffix_cer=0.0, extra_tail_units=0, ), ) independent = SimpleNamespace(candidate_results=(independent_result,)) verification_transcriber = object() def fake_verify(*args, **kwargs): calls.append(("verify", args, kwargs)) return independent def fake_intersect(*args): calls.append(("intersect", args)) return "combined" namespace = { "_verify_trajectory_audio": fake_verify, "QUALITY_FINAL_ASR_MAX_NEW_TOKENS": 440, "SEQUENCE_FALLBACK_MAX_LOCAL_BOUNDARY_SPEAKER_DROP": 0.15, "transcribe_verification_whisper": verification_transcriber, "intersect_local_semantic_verification": fake_intersect, "local_candidate_has_coverage_eligibility": ( lambda result, **_kwargs: result is primary_results[1] ), "LocalIndependentGateEvidence": LocalIndependentGateEvidence, "candidate_gate_evidence": lambda result: ("bounded", result), } exec(compile(module, "", "exec"), namespace) result = namespace["_verify_network_local_asr_intersection"]( turbo, ("ordinary-audio", "network-audio", "ordinary-audio-2"), ("PRIVATE_ORDINARY_A", "PRIVATE_NETWORK_TEXT", "PRIVATE_ORDINARY_B"), "anchor", proof_rows, candidate_seed=123, ) assert result[0] == "combined" verify_call = calls[0] assert verify_call[0] == "verify" assert verify_call[1][:4] == ( ("network-audio",), ("PRIVATE_NETWORK_TEXT",), "anchor", 1.0, ) assert verify_call[1][4] == 440 assert verify_call[2]["transcriber"] is verification_transcriber assert verify_call[2]["semantic_only"] is True selected_proofs = verify_call[2]["network_fragment_proofs"] assert selected_proofs == ((proof,),) assert selected_proofs[0] is proof_rows[1] assert calls[1] == ( "intersect", (turbo, independent, (1,)), ) evidence = result[1] assert evidence[0] == LocalIndependentGateEvidence(False, None, 0, None) assert evidence[1].attempted is True assert evidence[1].passed is True assert evidence[1].proof_count == 1 assert evidence[1].result == ("bounded", independent_result) assert evidence[2] == LocalIndependentGateEvidence(False, None, 1, None) log = capsys.readouterr().out assert "seed=123 local_chunk_index=1 proof_count=1 passed=True" in log assert "PRIVATE_" not in log def test_app_emits_one_canonical_content_free_evidence_line_per_terminal_outcome(): source = (ROOT / "app.py").read_text(encoding="utf-8") tree = ast.parse(source) functions = { node.name: node for node in tree.body if isinstance(node, ast.FunctionDef) } synthesize_source = ast.get_source_segment(source, functions["_synthesize"]) qualifier_source = ast.get_source_segment( source, functions["_qualify_candidate_trajectory_audio"], ) assert synthesize_source is not None assert qualifier_source is not None assert synthesize_source.count("format_cascade_evidence_log(") == 3 assert synthesize_source.count( "generated_chunk_limit=QUALITY_MAX_GENERATED_CHUNKS" ) == 3 assert synthesize_source.count( "generated_text_unit_limit=QUALITY_MAX_GENERATED_TEXT_UNITS" ) == 3 assert "generation_evidence_factory=candidate_generation_evidence" in ( synthesize_source ) assert 'outcome="no_qualified_candidate"' in synthesize_source assert 'outcome="final_output_rejected"' in synthesize_source assert 'outcome="returned"' in synthesize_source assert "error.diagnostics" in synthesize_source assert "cascade.diagnostics" in synthesize_source assert "final_output=final_evidence" in synthesize_source assert "CandidateVerification(" in qualifier_source assert "independent_local_results=independent_local_results" in ( qualifier_source ) assert "joined_evidence = trajectory_gate_evidence(joined_verification)" in ( qualifier_source ) assert "joined_output=joined_evidence" in qualifier_source assert ( "independent_output=trajectory_gate_evidence(independent_verification)" in qualifier_source ) assert "independent_final_output=independent_final_evidence" in synthesize_source def test_app_limits_boundary_relaxation_to_final_verified_sequence_fallback(): app_source = (ROOT / "app.py").read_text(encoding="utf-8") quality_source = (ROOT / "quality_runtime.py").read_text(encoding="utf-8") assert "SEQUENCE_FALLBACK_MAX_LOCAL_BOUNDARY_SPEAKER_DROP = 0.15" in quality_source assert "result.rejection_reasons != (\"boundary_speaker_drop\",)" in quality_source assert "sequence fallback boundary relaxation requires a final verifier" in quality_source assert "sequence_fallback_max_local_boundary_speaker_drop=(" in app_source assert "SEQUENCE_FALLBACK_MAX_LOCAL_BOUNDARY_SPEAKER_DROP" in app_source assert "QUALITY_RELEASE_MAX_BOUNDARY_SPEAKER_DROP = 0.095" in app_source def test_internal_synthesize_accepts_only_a_keyword_seed_while_ui_stays_unchanged(): source = (ROOT / "app.py").read_text(encoding="utf-8") tree = ast.parse(source) functions = { node.name: node for node in tree.body if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)) } synthesize = functions["_synthesize"] assert "request_seed" not in [argument.arg for argument in synthesize.args.args] assert [argument.arg for argument in synthesize.args.kwonlyargs][-1] == "request_seed" assert isinstance(synthesize.args.kw_defaults[-1], ast.Constant) assert synthesize.args.kw_defaults[-1].value is None assert "request_seed = resolve_request_seed(request_seed, secrets.randbelow)" in source assert "run_coverage_adaptive_cascade(\n chunks,\n request_seed," in source for wrapper_name in ("tts_speaker", "tts_reference", "tts_longform"): wrapper = functions[wrapper_name] arguments = wrapper.args.args + wrapper.args.kwonlyargs assert "request_seed" not in [argument.arg for argument in arguments] def test_app_reverifies_the_post_join_speed_adjusted_whole_waveform(): source = (ROOT / "app.py").read_text(encoding="utf-8") tree = ast.parse(source) functions = { node.name: node for node in tree.body if isinstance(node, ast.FunctionDef) } assemble_source = ast.get_source_segment( source, functions["_assemble_trajectory_audio"], ) synthesize_source = ast.get_source_segment(source, functions["_synthesize"]) assert assemble_source is not None assert synthesize_source is not None speed_index = assemble_source.index( "waveform = _apply_speed(waveform, playback_speed)" ) finish_index = assemble_source.index( "return finish_audio(waveform, SR, fade_ms=finish_fade_ms)" ) assert speed_index < finish_index assert ( 'finish_fade_ms = 0.0 if count_speech_units("".join(chunks)) <= 6 else 5.0' in assemble_source ) assert "else 60.0" not in assemble_source production_source = (ROOT / "production.py").read_text(encoding="utf-8") assert "trailing_silence_ms: float = 180.0" in production_source assemble_index = synthesize_source.index( "waveform = _assemble_trajectory_audio(" ) verify_index = synthesize_source.index("final_verification = _verify_trajectory_audio(") require_index = synthesize_source.index( "require_verified_final_output(final_verification)" ) return_index = synthesize_source.index("return SR, waveform") assert assemble_index < verify_index < require_index < return_index assert "(text,)" in synthesize_source[verify_index:require_index] assert " 1.0," in synthesize_source[verify_index:require_index] assert "QUALITY_FINAL_ASR_MAX_NEW_TOKENS" in synthesize_source[ verify_index:require_index ] assert "release_speaker_gate=True" in synthesize_source[ verify_index:require_index ] assert "short_audio_seconds=(" in source assert "RELEASE_SPEAKER_TRIGGER_SECONDS" in source def test_chunk_generation_closed_loop_rerenders_final_audio_from_raw_once(): source = (ROOT / "app.py").read_text(encoding="utf-8") tree = ast.parse(source) functions = { node.name: node for node in tree.body if isinstance(node, ast.FunctionDef) } generate_source = ast.get_source_segment(source, functions["_generate_chunk"]) apply_speed_source = ast.get_source_segment(source, functions["_apply_speed"]) assert generate_source is not None assert apply_speed_source is not None total_pace_index = generate_source.index("pace_speed = target_pace_speed(") active_measure_index = generate_source.index( "active_voiced_duration_seconds(audio, SR)" ) active_pace_index = generate_source.index( "active_speed = active_pace_correction_speed(" ) combined_speed_index = generate_source.index( "combined_speed = min(pace_speed, active_speed)" ) initial_stretch_index = generate_source.index("corrected = _apply_speed(") corrected_measure_index = generate_source.index( "corrected_active_duration = active_voiced_duration_seconds(" ) rerender_speed_index = generate_source.index( "rerender_speed = active_pace_correction_speed(" ) final_speed_index = generate_source.index( "final_speed = combined_speed * rerender_speed" ) final_stretch_index = generate_source.index("final_audio = _apply_speed(") assert ( total_pace_index < active_measure_index < active_pace_index < combined_speed_index < initial_stretch_index < corrected_measure_index < rerender_speed_index < final_speed_index < final_stretch_index ) assert generate_source.count("_apply_speed(") == 3 assert "waveform = _apply_speed(audio, pace_speed)" not in generate_source assert "ACTIVE_PACE_TARGET_CPS = 4.00" in source assert "CLOSED_LOOP_ACTIVE_PACE_TARGET_CPS = 3.95" in source assert "target_cps=ACTIVE_PACE_TARGET_CPS" in generate_source assert "target_cps=CLOSED_LOOP_ACTIVE_PACE_TARGET_CPS" in generate_source assert "prior_speed=1.0" in generate_source assert "prior_speed=combined_speed" in generate_source assert "min_total_speed=MIN_PACE_SPEED" in generate_source assert "final_speed = combined_speed * rerender_speed" in generate_source assert "fallback_speed = final_speed * fallback_residual" in generate_source assert "_apply_speed(\n corrected" not in generate_source assert "_apply_speed(\n audio,\n final_speed" in generate_source assert "_apply_speed(\n audio,\n fallback_speed" in generate_source assert "PACE_STRETCH_N_FFT = 1536" in source assert "PACE_STRETCH_HOP_LENGTH = 384" in source assert "NETWORK_PACE_STRETCH_N_FFT = 2048" in source assert "NETWORK_PACE_STRETCH_HOP_LENGTH = 512" in source assert "network_conditioned=network_conditioned" in generate_source assert "if network_conditioned" in apply_speed_source assert "n_fft=n_fft" in apply_speed_source assert "hop_length=hop_length" in apply_speed_source def test_public_tts_wrappers_serialize_pcm16_after_float_verification(): source = (ROOT / "app.py").read_text(encoding="utf-8") tree = ast.parse(source) functions = { node.name: node for node in tree.body if isinstance(node, ast.FunctionDef) } synthesize_source = ast.get_source_segment(source, functions["_synthesize"]) assert synthesize_source is not None assert "pcm16_audio_output" not in synthesize_source assert "return SR, waveform" in synthesize_source for wrapper_name in ("tts_speaker", "tts_reference", "tts_longform"): wrapper_source = ast.get_source_segment(source, functions[wrapper_name]) assert wrapper_source is not None assert "return pcm16_audio_output(" in wrapper_source assert "*_synthesize(" in wrapper_source def test_space_applies_pinned_squim_to_local_joined_and_final_audio(): app_source = (ROOT / "app.py").read_text(encoding="utf-8") app_tree = ast.parse(app_source) app_functions = { node.name: node for node in app_tree.body if isinstance(node, ast.FunctionDef) } verify_source = ast.get_source_segment( app_source, app_functions["_verify_trajectory_audio"], ) runtime_source = (ROOT / "quality_runtime.py").read_text(encoding="utf-8") assert verify_source is not None assert 'QUALITY_MIN_SQUIM_STOI = 0.60' in app_source assert 'QUALITY_MIN_SQUIM_PESQ = 1.12' in app_source assert 'QUALITY_PREFERRED_MIN_SQUIM_STOI = 0.72' in app_source assert 'QUALITY_PREFERRED_MIN_SQUIM_PESQ = 1.20' in app_source assert 'QUALITY_PREFERRED_MIN_SPEAKER_SIMILARITY = 0.25' in app_source assert 'QUALITY_PREFERRED_MAX_BOUNDARY_SPEAKER_DROP = 0.05' in app_source assert "if not semantic_only and transcript:" in verify_source score_index = verify_source.index("squim_objective_evidence_from_audio(") observation_index = verify_source.index("CandidateObservation(", score_index) gate_index = verify_source.index("squim_gate_enabled=not semantic_only") assert score_index < observation_index < gate_index assert "except ValueError:" in verify_source[score_index:observation_index] assert "except RuntimeError:" not in verify_source[score_index:observation_index] assert ( '"2c54586fea83fb5eb5394d710038ee89f55cab7011a5bf730bebed4c8777e828"' in runtime_source ) hash_index = runtime_source.index("digest = str(hasher(weight_path)).casefold()") state_index = runtime_source.index( 'state_dict = loader(weight_path, map_location="cpu", weights_only=True)' ) assert hash_index < state_index assert "device=torch.device(\"cpu\")" in runtime_source synthesize_source = ast.get_source_segment( app_source, app_functions["_synthesize"], ) assert synthesize_source is not None for argument in ( "preferred_min_speaker_similarity=", "preferred_max_boundary_speaker_drop=", "preferred_min_squim_stoi=QUALITY_PREFERRED_MIN_SQUIM_STOI", "preferred_min_squim_pesq=QUALITY_PREFERRED_MIN_SQUIM_PESQ", "QUALITY_PREFERRED_SQUIM_MIN_DURATION_SECONDS", ): assert argument in synthesize_source assert "QUALITY_PREFERRED_SQUIM_MIN_DURATION_SECONDS = 1.50" in app_source def test_whole_candidate_qualification_uses_the_exact_return_assembler_after_local_pass(): source = (ROOT / "app.py").read_text(encoding="utf-8") tree = ast.parse(source) functions = { node.name: node for node in tree.body if isinstance(node, ast.FunctionDef) } qualify_source = ast.get_source_segment( source, functions["_qualify_candidate_trajectory_audio"], ) synthesize_source = ast.get_source_segment(source, functions["_synthesize"]) assert qualify_source is not None assert synthesize_source is not None local_index = qualify_source.index("local_verification = _verify_trajectory_audio(") local_fail_index = qualify_source.index("if not local_verification.passed:") assemble_index = qualify_source.index("waveform = _assemble_trajectory_audio(") joined_index = qualify_source.index("joined_verification = _verify_trajectory_audio(") assert local_index < local_fail_index < assemble_index < joined_index assert "QUALITY_FINAL_ASR_MAX_NEW_TOKENS" in qualify_source[joined_index:] assert "release_speaker_gate=True" in qualify_source[joined_index:] assert "qualify_trajectory_with_joined_output(" in qualify_source[joined_index:] assert "waveform = _assemble_trajectory_audio(" in synthesize_source assert "cascade.trajectory" in synthesize_source assert "chunk_specs" in synthesize_source assert "require_verified_final_output(final_verification)" in synthesize_source def test_space_hard_intersects_dual_asr_on_whole_and_proven_network_locals(): source = (ROOT / "app.py").read_text(encoding="utf-8") tree = ast.parse(source) functions = { node.name: node for node in tree.body if isinstance(node, ast.FunctionDef) } verify_source = ast.get_source_segment(source, functions["_verify_trajectory_audio"]) independent_source = ast.get_source_segment( source, functions["_verify_independent_whole_audio"], ) qualify_source = ast.get_source_segment( source, functions["_qualify_candidate_trajectory_audio"], ) sequence_source = ast.get_source_segment( source, functions["_verify_sequence_trajectory_audio"], ) refill_source = ast.get_source_segment( source, functions["_verify_refill_candidate_trajectory_audio"], ) network_local_source = ast.get_source_segment( source, functions["_verify_network_local_asr_intersection"], ) synthesize_source = ast.get_source_segment(source, functions["_synthesize"]) assert all( segment is not None for segment in ( verify_source, independent_source, qualify_source, sequence_source, refill_source, network_local_source, synthesize_source, ) ) assert "transcriber=transcribe_whisper" in verify_source assert "speaker_gate_enabled=not semantic_only" in verify_source assert "transcriber=transcribe_verification_whisper" in independent_source assert "semantic_only=True" in independent_source assert "cache.verify(" in independent_source assert "VERIFICATION_ASR_PROFILE" in independent_source assert "local_verification = _verify_trajectory_audio(" in refill_source assert "_verify_independent_whole_audio" not in refill_source assert "transcriber=transcribe_verification_whisper" in network_local_source assert "semantic_only=True" in network_local_source assert "network_fragment_proofs=independent_proof_rows" in network_local_source assert "intersect_local_semantic_verification(" in network_local_source local_index = qualify_source.index("local_verification = _verify_trajectory_audio(") local_fail_index = qualify_source.index("if not local_verification.passed:") assemble_index = qualify_source.index("waveform = _assemble_trajectory_audio(") turbo_joined_index = qualify_source.index("joined_verification = _verify_trajectory_audio(") turbo_fail_index = qualify_source.index("if not qualified.passed:") independent_index = qualify_source.index("_verify_independent_whole_audio(") dual_index = qualify_source.index( "dual_qualified = qualify_trajectory_with_joined_output(" ) assert ( local_index < local_fail_index < assemble_index < turbo_joined_index < turbo_fail_index < independent_index < dual_index ) sequence_assemble = sequence_source.index("_assemble_trajectory_audio(") sequence_large_v3 = sequence_source.index( "assembled_verification = _verify_trajectory_audio(" ) sequence_large_v3_transcriber = sequence_source.index( "transcriber=transcribe_verification_whisper" ) sequence_independent = sequence_source.index( "independent_verification = _verify_independent_whole_audio(" ) sequence_intersection = sequence_source.index( "intersected = qualify_trajectory_with_joined_output(" ) assert ( sequence_assemble < sequence_large_v3 < sequence_large_v3_transcriber < sequence_independent < sequence_intersection ) cache_create = synthesize_source.index( "independent_cache = WholeWaveformVerificationCache()" ) cascade_index = synthesize_source.index( "cascade = run_coverage_adaptive_cascade(" ) final_assemble = synthesize_source.index( "waveform = _assemble_trajectory_audio(" ) final_turbo = synthesize_source.index("final_verification = _verify_trajectory_audio(") final_turbo_require = synthesize_source.index( "require_verified_final_output(final_verification)" ) final_independent = synthesize_source.index( "independent_final_verification = _verify_independent_whole_audio(" ) final_independent_require = synthesize_source.index( "require_verified_final_output(independent_final_verification)" ) return_index = synthesize_source.index("return SR, waveform") assert ( cache_create < cascade_index < final_assemble < final_turbo < final_turbo_require < final_independent < final_independent_require < return_index ) assert synthesize_source.count("independent_cache,") >= 3 assert "except (RuntimeError, ValueError) as error:" in synthesize_source def test_network_fragment_relaxation_is_range_bound_and_local_only(): source = (ROOT / "app.py").read_text(encoding="utf-8") tree = ast.parse(source) functions = { node.name: node for node in tree.body if isinstance(node, ast.FunctionDef) } verify_source = ast.get_source_segment(source, functions["_verify_trajectory_audio"]) proof_source = ast.get_source_segment( source, functions["_network_fragment_proof_rows"], ) qualify_source = ast.get_source_segment( source, functions["_qualify_candidate_trajectory_audio"], ) refill_source = ast.get_source_segment( source, functions["_verify_refill_candidate_trajectory_audio"], ) independent_source = ast.get_source_segment( source, functions["_verify_independent_whole_audio"], ) sequence_source = ast.get_source_segment( source, functions["_verify_sequence_trajectory_audio"], ) synthesize_source = ast.get_source_segment(source, functions["_synthesize"]) assert all( segment is not None for segment in ( verify_source, proof_source, qualify_source, refill_source, independent_source, sequence_source, synthesize_source, ) ) assert "canonicalize_asr_network_fragments(" in verify_source assert "if fragment_evidence.passed" in verify_source assert "network-conditioned chunk lacks exact fragment proof" in proof_source assert "proof.span_index for proof in proofs" in proof_source assert "proof.full_spoken_proof for proof in proofs" in proof_source local_index = qualify_source.index("local_verification = _verify_trajectory_audio(") joined_index = qualify_source.index("joined_verification = _verify_trajectory_audio(") assert "network_fragment_proofs=" in qualify_source[local_index:joined_index] assert "network_fragment_proofs=" not in qualify_source[joined_index:] assert "network_fragment_proofs=" in refill_source assert "network_fragment_proofs=" not in independent_source assert "network_fragment_proofs=" not in sequence_source final_index = synthesize_source.index("final_verification = _verify_trajectory_audio(") assert "network_fragment_proofs=" not in synthesize_source[final_index:] assert "generation_context_by_seed" in synthesize_source assert "generation_chunk_specs(seed, candidate_chunks)" in synthesize_source def test_local_endpoint_relaxation_is_role_bound_and_whole_gates_stay_exact(): source = (ROOT / "app.py").read_text(encoding="utf-8") tree = ast.parse(source) functions = { node.name: node for node in tree.body if isinstance(node, ast.FunctionDef) } verify_source = ast.get_source_segment(source, functions["_verify_trajectory_audio"]) role_source = ast.get_source_segment(source, functions["_local_endpoint_role_rows"]) qualify_source = ast.get_source_segment( source, functions["_qualify_candidate_trajectory_audio"], ) refill_source = ast.get_source_segment( source, functions["_verify_refill_candidate_trajectory_audio"], ) sequence_source = ast.get_source_segment( source, functions["_verify_sequence_trajectory_audio"], ) synthesize_source = ast.get_source_segment(source, functions["_synthesize"]) assert all( segment is not None for segment in ( verify_source, role_source, qualify_source, refill_source, sequence_source, synthesize_source, ) ) assert "spec.source_start == 0" in role_source assert 'spec.boundary_after == "none"' in role_source assert "spec.text != chunk" in role_source assert "local_candidate_pool = not semantic_only and not release_speaker_gate" in ( verify_source ) assert "max_prefix_cer=(1.0 / 6.0 if local_candidate_pool else 0.0)" in ( verify_source ) assert "max_suffix_cer=(1.0 / 6.0 if local_candidate_pool else 0.0)" in ( verify_source ) assert "max_prefix_deletions=(0 if local_candidate_pool else None)" in ( verify_source ) assert "max_suffix_deletions=(0 if local_candidate_pool else None)" in ( verify_source ) assert "candidate_gate_kwargs_by_index=indexed_gate_kwargs" in verify_source local_index = qualify_source.index("local_verification = _verify_trajectory_audio(") joined_index = qualify_source.index("joined_verification = _verify_trajectory_audio(") assert "local_endpoint_roles=" in qualify_source[local_index:joined_index] assert "local_endpoint_roles=" not in qualify_source[joined_index:] assert "local_endpoint_roles=" in refill_source assert "local_endpoint_roles=" not in sequence_source final_index = synthesize_source.index("final_verification = _verify_trajectory_audio(") assert "local_endpoint_roles=" not in synthesize_source[final_index:] assert "generation_chunk_specs(seed, candidate_chunks)" in synthesize_source def test_naturalized_url_provenance_is_revalidated_without_policy_override(): production_source = (ROOT / "production.py").read_text(encoding="utf-8") app_source = (ROOT / "app.py").read_text(encoding="utf-8") readme = (ROOT / "README.md").read_text(encoding="utf-8") production_tree = ast.parse(production_source) app_tree = ast.parse(app_source) production_functions = { node.name: node for node in production_tree.body if isinstance(node, ast.FunctionDef) } app_functions = { node.name: node for node in app_tree.body if isinstance(node, ast.FunctionDef) } inverse_source = ast.get_source_segment( production_source, production_functions["inverse_network_url_rendering"], ) planner_source = ast.get_source_segment( production_source, production_functions["plan_generation_chunks"], ) local_source = ast.get_source_segment( production_source, production_functions["canonicalize_asr_network_fragments"], ) runtime_source = ast.get_source_segment( app_source, app_functions["_network_fragment_proof_rows"], ) generate_source = ast.get_source_segment( app_source, app_functions["_generate_chunk"], ) assert all( segment is not None for segment in ( inverse_source, planner_source, local_source, runtime_source, generate_source, ) ) assert "proof != expected" in inverse_source assert "_naturalized_url_proof_from_spoken(full_proof)" in planner_source assert "_naturalized_url_proof_from_spoken(full_proof)" in local_source assert "proof.raw_identifier or proof.url_rendering_proof is not None" in ( planner_source ) assert "proof.raw_identifier or proof.url_rendering_proof is not None" in ( local_source ) assert "contains_naturalized_url_spoken_form(" in runtime_source assert "proof.url_rendering_proof != fresh_rendering" in runtime_source assert "proof.raw_identifier or proof.url_rendering_proof is not None" in ( runtime_source ) assert "generation_cps = (" in generate_source assert "policy.ascii_cps" in generate_source assert "COMPLETION_HEADROOM_GENERATION_POLICY" not in generate_source assert "不可切 component" in readme assert "Email contract" in readme def test_space_wires_bounded_k_best_paths_to_exact_assembled_whole_gate(): source = (ROOT / "app.py").read_text(encoding="utf-8") assert "sequence_final_verifier=lambda sequence_result, candidate_chunks:" in source assert "_verify_sequence_trajectory_audio(" in source assert "QUALITY_MAX_SEQUENCE_PATHS = 3" in source assert "max_sequence_paths=QUALITY_MAX_SEQUENCE_PATHS" in source assert "waveform = _assemble_trajectory_audio(" in source assert "QUALITY_FINAL_ASR_MAX_NEW_TOKENS" in source assert "sequence_rank={cascade.sequence_path_rank}" in source assert "sequence_paths_checked={cascade.sequence_paths_checked}" in source assert "cer={comparison.cer:.6f}" in source assert "prefix_cer={comparison.prefix_cer:.6f}" in source assert "suffix_cer={comparison.suffix_cer:.6f}" in source assert "tail_units={comparison.extra_tail_units}" in source def test_space_locks_validated_nfe_and_bounds_total_generation_work(): source = (ROOT / "app.py").read_text(encoding="utf-8") quality_source = (ROOT / "quality_runtime.py").read_text(encoding="utf-8") readme = (ROOT / "README.md").read_text(encoding="utf-8") assert "QUALITY_MAX_GENERATED_CHUNKS = 32" in source assert "QUALITY_MAX_GENERATED_TEXT_UNITS = 800" in source assert "EMAIL_MAIL_FALLBACK_CANDIDATE_ORDINALS = frozenset((2, 5, 6))" in source assert "NETWORK_REQUEST_ORDINARY_MAX_UNITS = 36" in source assert "ordinary_max_units=NETWORK_REQUEST_ORDINARY_MAX_UNITS" in source assert "PACE_ONLY_FALLBACK_MIN_SPEED = 0.76" in source assert "PACE_ONLY_FALLBACK_MIN_UNITS = 30" in source assert "if observed_cps > QUALITY_MAX_PACE_CPS:" in source assert "final_audio = _apply_speed(" in source assert "fallback_speed = final_speed * fallback_residual" in source assert "max_generated_chunks=QUALITY_MAX_GENERATED_CHUNKS" in source assert "max_generated_text_units=QUALITY_MAX_GENERATED_TEXT_UNITS" in source assert "candidate_generation_text_transform=(" in source assert "_verify_refill_candidate_trajectory_audio(" in source assert "def run_coverage_adaptive_cascade(" in quality_source assert "proposed_generation_chunks = generation_chunks(" in quality_source assert "generated_chunks += 1" in quality_source assert "generated_units += refill_units" in quality_source assert "select_culprit_diverse_candidate_sequences(" in quality_source assert "requested_steps != DEFAULT_STEPS" in source assert "interactive=False" in source assert "NFE steps(已驗證固定值)" in source assert "最多 3 條 culprit-diverse 完整路徑" in readme assert "不再為了取得某一段替代候選而重生整篇" in readme assert "boundary-only relaxation" in readme def test_pace_only_fallback_guard_starts_at_30_ordinary_units_not_29(): app_path = ROOT / "app.py" constants = _literal_constants(app_path) tree = ast.parse(app_path.read_text(encoding="utf-8")) generate_chunk = next( node for node in tree.body if isinstance(node, ast.FunctionDef) and node.name == "_generate_chunk" ) guarded_ifs = [ node for node in ast.walk(generate_chunk) if isinstance(node, ast.If) and any( isinstance(name, ast.Name) and name.id == "PACE_ONLY_FALLBACK_MIN_UNITS" for name in ast.walk(node.test) ) ] assert constants["PACE_ONLY_FALLBACK_MIN_UNITS"] == 30 assert len(guarded_ifs) == 1 expression = ast.Expression(body=guarded_ifs[0].test) ast.fix_missing_locations(expression) guard = compile(expression, str(app_path), "eval") def eligible(units, *, network_conditioned=False): return eval( guard, { "PACE_ONLY_FALLBACK_MIN_UNITS": constants[ "PACE_ONLY_FALLBACK_MIN_UNITS" ], "count_speech_units": lambda _text: units, "network_conditioned": network_conditioned, "text": "測試", }, ) assert eligible(29) is False assert eligible(30) is True assert eligible(30, network_conditioned=True) is False def test_assembled_waveform_uses_250ms_ordinary_and_350ms_network_request_pause(): assemble = _isolated_assemble_trajectory_audio() chunks = ("第一段,", "第二段") trajectory = ( np.ones(100, dtype=np.float32), np.ones(100, dtype=np.float32), ) ordinary = assemble(trajectory, chunks, 1.0) assert ordinary.shape == (450,) np.testing.assert_array_equal(ordinary[100:350], np.zeros(250, dtype=np.float32)) first_end = len(chunks[0]) network_specs = ( GenerationChunkSpec( text=chunks[0], source_start=0, source_end=first_end, boundary_after="semantic", ), GenerationChunkSpec( text=chunks[1], source_start=first_end, source_end=first_end + len(chunks[1]), network_span_indices=(0,), boundary_after="none", ), ) network_request = assemble(trajectory, chunks, 1.0, network_specs) assert network_request.shape == (550,) np.testing.assert_array_equal( network_request[100:450], np.zeros(350, dtype=np.float32), ) def test_space_locks_waveform_only_bounded_whisper_segmentation(): source = (ROOT / "quality_runtime.py").read_text(encoding="utf-8") readme = (ROOT / "README.md").read_text(encoding="utf-8") assert "WHISPER_MAX_SEGMENT_SECONDS = 28.0" in source assert "WHISPER_HARD_MAX_SEGMENT_SECONDS = 30.0" in source assert "WHISPER_MIN_SEGMENT_SECONDS = 1.25" in source assert "WHISPER_MIN_PAUSE_SECONDS = 0.25" in source assert "WHISPER_MAX_VERIFICATION_SEGMENTS = 12" in source assert "WHISPER_MAX_MICROBATCH_SEGMENTS = 6" in source assert "target-dependent fallback" in readme assert "每次驗證最多 12 段" in readme