Codex commited on
Commit
1ba02e6
·
1 Parent(s): bea21d4

Add strict glyph hybrid request planner

Browse files
glyph_hybrid_plan.py ADDED
@@ -0,0 +1,793 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Fail-closed request planner for the frozen ``glyph_ascii_v1`` profile.
2
+
3
+ This module is deliberately limited to text and provenance. It does not load
4
+ audio, generate speech, join waveforms, or alter the legacy frontend. A
5
+ request either receives one complete, reversible hybrid plan or remains on the
6
+ unchanged legacy path.
7
+
8
+ The profile scans *maximal printable-ASCII runs*. A run is network-looking
9
+ when it contains ``@`` or begins exactly with ``http://`` or ``https://``.
10
+ Every such run must independently satisfy the strict ``glyph_ascii_v1``
11
+ grammar. There is no stripping, case folding, Unicode repair, substring
12
+ borrowing, or partial activation.
13
+
14
+ Canonical boundary contract
15
+ ---------------------------
16
+
17
+ Every non-empty canonical payload is separated from the next payload by
18
+ exactly one ASCII space. The same rule applies between glyph atoms. Spaces
19
+ are represented by explicit, zero-raw/zero-audio separator segments. Carrier
20
+ punctuation comes only from the existing zh-TW production normalizer; this
21
+ planner injects no lexical cue or punctuation.
22
+ """
23
+
24
+ from __future__ import annotations
25
+
26
+ from collections.abc import Mapping
27
+ from dataclasses import asdict, dataclass
28
+ import hashlib
29
+ import json
30
+ import re
31
+ import unicodedata
32
+
33
+ from glyph_ascii_v1 import (
34
+ GLYPH_ASCII_V1_GRAMMAR_ID,
35
+ GlyphAsciiV1Error,
36
+ GlyphAsciiV1Proof,
37
+ glyph_ascii_v1_asset_manifest_sha256,
38
+ inverse_glyph_ascii_v1,
39
+ render_glyph_ascii_v1,
40
+ )
41
+ from production import (
42
+ contains_network_identifier,
43
+ count_speech_units,
44
+ normalize_spoken_forms,
45
+ )
46
+
47
+
48
+ GLYPH_HYBRID_PROFILE_ID = "glyph_hybrid_request_v1"
49
+ GLYPH_HYBRID_NORMALIZATION_ID = "production.normalize_spoken_forms:zh-TW"
50
+ GLYPH_HYBRID_BOUNDARY_ID = "single_ascii_space_zero_audio_v1"
51
+
52
+ # These reservations are independent from the ordinary cascade ledger. The
53
+ # planner runs before generation, so a request cannot consume model or asset
54
+ # work and then discover that it exceeded a profile cap.
55
+ GLYPH_HYBRID_MAX_RAW_CHARS = 360
56
+ GLYPH_HYBRID_MAX_IDENTIFIERS = 8
57
+ GLYPH_HYBRID_MAX_ASSET_ATOMS = 128
58
+ GLYPH_HYBRID_MAX_ASSET_CANONICAL_UNITS = 512
59
+ GLYPH_HYBRID_ASSET_AUDIO_PLACEHOLDER_SAMPLES_PER_ATOM = 144_000
60
+ GLYPH_HYBRID_MAX_ASSET_AUDIO_PLACEHOLDER_SAMPLES = (
61
+ GLYPH_HYBRID_MAX_ASSET_ATOMS
62
+ * GLYPH_HYBRID_ASSET_AUDIO_PLACEHOLDER_SAMPLES_PER_ATOM
63
+ )
64
+ GLYPH_HYBRID_MAX_MODEL_GENERATED_CHUNKS = 32
65
+ GLYPH_HYBRID_MAX_MODEL_GENERATED_TEXT_UNITS = 800
66
+
67
+ _PRINTABLE_ASCII_RUN_RE = re.compile(r"[ -~]+", flags=re.ASCII)
68
+ _SHA256_RE = re.compile(r"[0-9a-f]{64}\Z", flags=re.ASCII)
69
+ _NETWORK_PREFIXES = ("http://", "https://")
70
+ _ZH_CARRIER_COMPATIBILITY_PUNCTUATION = frozenset(
71
+ ",。!?;:、()【】「」『』《》〈〉“”‘’[]{}"
72
+ "…⋯—–.。、〔〕〖〗〘〙〚〛﹁﹂︰﹐﹑﹒﹔﹕﹖﹗"
73
+ )
74
+ _BIDI_UNSAFE = frozenset(
75
+ {
76
+ "BN",
77
+ "LRE",
78
+ "LRO",
79
+ "RLE",
80
+ "RLO",
81
+ "PDF",
82
+ "LRI",
83
+ "RLI",
84
+ "FSI",
85
+ "PDI",
86
+ }
87
+ )
88
+ _SEGMENT_KINDS = frozenset({"carrier", "asset_atom", "control", "separator"})
89
+ _AUDIO_SOURCES = frozenset({"model", "asset", "zero_audio"})
90
+
91
+
92
+ class GlyphHybridPlanError(ValueError):
93
+ """Raised when a plan, proof, manifest, or expected request is invalid."""
94
+
95
+
96
+ @dataclass(frozen=True)
97
+ class HybridSegmentSpec:
98
+ """One exact raw/canonical segment in assembly order.
99
+
100
+ ``raw_text`` is retained for carrier/control provenance and for the one
101
+ ASCII byte represented by an asset atom. Separator segments have empty
102
+ raw text and a zero-width raw range.
103
+ """
104
+
105
+ ordinal: int
106
+ kind: str
107
+ audio_source: str
108
+ absolute_raw_start: int
109
+ absolute_raw_end: int
110
+ canonical_start: int
111
+ canonical_end: int
112
+ raw_text: str
113
+ canonical_text: str
114
+ raw_sha256: str
115
+ canonical_sha256: str
116
+ identifier_ordinal: int | None
117
+ atom_ordinal: int | None
118
+ asset_id: str | None
119
+ manifest_entry_sha256: str | None
120
+
121
+
122
+ @dataclass(frozen=True)
123
+ class HybridReservation:
124
+ """Pre-generation resource reservation for one complete request."""
125
+
126
+ raw_chars: int
127
+ identifiers: int
128
+ asset_atoms: int
129
+ asset_canonical_units: int
130
+ asset_audio_placeholder_samples: int
131
+ model_generated_chunks: int
132
+ model_generated_text_units: int
133
+
134
+
135
+ @dataclass(frozen=True)
136
+ class HybridRenderPlan:
137
+ """Complete request-level proof and canonical assembly plan."""
138
+
139
+ profile_id: str
140
+ grammar_id: str
141
+ normalization_id: str
142
+ boundary_contract_id: str
143
+ raw_input_sha256: str
144
+ canonical_target: str
145
+ canonical_target_sha256: str
146
+ asset_manifest_logical_sha256: str
147
+ identifier_proofs: tuple[GlyphAsciiV1Proof, ...]
148
+ segments: tuple[HybridSegmentSpec, ...]
149
+ reservation: HybridReservation
150
+ plan_sha256: str
151
+
152
+
153
+ @dataclass(frozen=True)
154
+ class _RawPiece:
155
+ kind: str
156
+ start: int
157
+ end: int
158
+ text: str
159
+
160
+
161
+ def _sha256_text(value: str) -> str:
162
+ return hashlib.sha256(value.encode("utf-8")).hexdigest()
163
+
164
+
165
+ def _require_sha256(value: object, *, field: str) -> str:
166
+ if type(value) is not str or _SHA256_RE.fullmatch(value) is None:
167
+ raise GlyphHybridPlanError(f"{field} must be a lowercase SHA-256 digest")
168
+ return value
169
+
170
+
171
+ def _plan_payload(plan: HybridRenderPlan) -> dict[str, object]:
172
+ payload = asdict(plan)
173
+ payload.pop("plan_sha256")
174
+ return payload
175
+
176
+
177
+ def _logical_plan_sha256(plan: HybridRenderPlan) -> str:
178
+ payload = json.dumps(
179
+ _plan_payload(plan),
180
+ ensure_ascii=True,
181
+ separators=(",", ":"),
182
+ sort_keys=True,
183
+ ).encode("ascii")
184
+ return hashlib.sha256(payload).hexdigest()
185
+
186
+
187
+ def glyph_hybrid_plan_sha256(plan: HybridRenderPlan) -> str:
188
+ """Validate the plan type and return its recomputed logical digest."""
189
+
190
+ if type(plan) is not HybridRenderPlan:
191
+ raise GlyphHybridPlanError("hybrid plan has an invalid type")
192
+ return _logical_plan_sha256(plan)
193
+
194
+
195
+ def _network_looking(run: str) -> bool:
196
+ return "@" in run or run.startswith(_NETWORK_PREFIXES)
197
+
198
+
199
+ def _request_has_unsafe_unicode(raw_request: str) -> bool:
200
+ """Reject characters that can split or visually rewrite an identifier.
201
+
202
+ Han text and ordinary zh-TW punctuation remain eligible. Non-ASCII Latin,
203
+ Greek, Cyrillic, combining marks, compatibility forms that fold to
204
+ printable ASCII, controls, and bidi formatting are intentionally outside
205
+ this narrow profile.
206
+ """
207
+
208
+ for character in raw_request:
209
+ if character.isascii():
210
+ if unicodedata.category(character)[0] == "C":
211
+ return True
212
+ continue
213
+ category = unicodedata.category(character)
214
+ bidi = unicodedata.bidirectional(character)
215
+ if category[0] in {"C", "M"} or bidi in _BIDI_UNSAFE:
216
+ return True
217
+ if character in _ZH_CARRIER_COMPATIBILITY_PUNCTUATION:
218
+ continue
219
+ folded = unicodedata.normalize("NFKC", character)
220
+ if folded != character and any(
221
+ folded_character.isascii()
222
+ and 0x20 <= ord(folded_character) <= 0x7E
223
+ for folded_character in folded
224
+ ):
225
+ return True
226
+ unicode_name = unicodedata.name(character, "")
227
+ if any(
228
+ script in unicode_name
229
+ for script in ("LATIN", "GREEK", "CYRILLIC")
230
+ ):
231
+ return True
232
+ return False
233
+
234
+
235
+ def _raw_pieces(raw_request: str) -> tuple[_RawPiece, ...] | None:
236
+ network_runs: list[re.Match[str]] = []
237
+ for match in _PRINTABLE_ASCII_RUN_RE.finditer(raw_request):
238
+ if _network_looking(match.group(0)):
239
+ network_runs.append(match)
240
+ if not network_runs:
241
+ return None
242
+ if len(network_runs) > GLYPH_HYBRID_MAX_IDENTIFIERS:
243
+ return None
244
+
245
+ pieces: list[_RawPiece] = []
246
+ cursor = 0
247
+ for match in network_runs:
248
+ if cursor < match.start():
249
+ pieces.append(
250
+ _RawPiece(
251
+ kind="carrier",
252
+ start=cursor,
253
+ end=match.start(),
254
+ text=raw_request[cursor : match.start()],
255
+ )
256
+ )
257
+ pieces.append(
258
+ _RawPiece(
259
+ kind="identifier",
260
+ start=match.start(),
261
+ end=match.end(),
262
+ text=match.group(0),
263
+ )
264
+ )
265
+ cursor = match.end()
266
+ if cursor < len(raw_request):
267
+ pieces.append(
268
+ _RawPiece(
269
+ kind="carrier",
270
+ start=cursor,
271
+ end=len(raw_request),
272
+ text=raw_request[cursor:],
273
+ )
274
+ )
275
+ return tuple(pieces)
276
+
277
+
278
+ def _carrier_canonical(raw_fragment: str) -> str | None:
279
+ # A network form outside the strict maximal run would make activation
280
+ # partial. Reject both before and after legacy normalization.
281
+ if contains_network_identifier(raw_fragment):
282
+ return None
283
+ try:
284
+ canonical = normalize_spoken_forms(raw_fragment, locale="zh-TW")
285
+ except (TypeError, ValueError):
286
+ return None
287
+ if (
288
+ contains_network_identifier(canonical)
289
+ or canonical != " ".join(canonical.split())
290
+ ):
291
+ return None
292
+ return canonical
293
+
294
+
295
+ def _manifest_digest(
296
+ asset_entry_sha256_by_asset_id: Mapping[str, str],
297
+ ) -> str:
298
+ try:
299
+ return glyph_ascii_v1_asset_manifest_sha256(
300
+ asset_entry_sha256_by_asset_id
301
+ )
302
+ except GlyphAsciiV1Error as error:
303
+ raise GlyphHybridPlanError("glyph asset manifest is invalid") from error
304
+
305
+
306
+ def _segment(
307
+ *,
308
+ ordinal: int,
309
+ kind: str,
310
+ audio_source: str,
311
+ absolute_raw_start: int,
312
+ absolute_raw_end: int,
313
+ canonical_start: int,
314
+ raw_text: str,
315
+ canonical_text: str,
316
+ identifier_ordinal: int | None = None,
317
+ atom_ordinal: int | None = None,
318
+ asset_id: str | None = None,
319
+ manifest_entry_sha256: str | None = None,
320
+ ) -> HybridSegmentSpec:
321
+ return HybridSegmentSpec(
322
+ ordinal=ordinal,
323
+ kind=kind,
324
+ audio_source=audio_source,
325
+ absolute_raw_start=absolute_raw_start,
326
+ absolute_raw_end=absolute_raw_end,
327
+ canonical_start=canonical_start,
328
+ canonical_end=canonical_start + len(canonical_text),
329
+ raw_text=raw_text,
330
+ canonical_text=canonical_text,
331
+ raw_sha256=_sha256_text(raw_text),
332
+ canonical_sha256=_sha256_text(canonical_text),
333
+ identifier_ordinal=identifier_ordinal,
334
+ atom_ordinal=atom_ordinal,
335
+ asset_id=asset_id,
336
+ manifest_entry_sha256=manifest_entry_sha256,
337
+ )
338
+
339
+
340
+ def build_glyph_hybrid_plan(
341
+ raw_request: str,
342
+ *,
343
+ asset_entry_sha256_by_asset_id: Mapping[str, str],
344
+ ) -> HybridRenderPlan | None:
345
+ """Build one complete strict plan, or return ``None`` without side effects.
346
+
347
+ ``None`` means that the whole request must remain on the unchanged legacy
348
+ path. Invalid manifest configuration raises because silently falling back
349
+ would conceal a deployment-integrity error.
350
+ """
351
+
352
+ if type(raw_request) is not str:
353
+ raise GlyphHybridPlanError("raw request must be an exact string")
354
+ manifest_sha256 = _manifest_digest(asset_entry_sha256_by_asset_id)
355
+ if (
356
+ not raw_request
357
+ or len(raw_request) > GLYPH_HYBRID_MAX_RAW_CHARS
358
+ or _request_has_unsafe_unicode(raw_request)
359
+ ):
360
+ return None
361
+ pieces = _raw_pieces(raw_request)
362
+ if pieces is None:
363
+ return None
364
+
365
+ segments: list[HybridSegmentSpec] = []
366
+ identifier_proofs: list[GlyphAsciiV1Proof] = []
367
+ canonical_parts: list[str] = []
368
+ canonical_cursor = 0
369
+ last_payload_raw_boundary: int | None = None
370
+ asset_canonical_units = 0
371
+ model_generated_chunks = 0
372
+ model_generated_text_units = 0
373
+
374
+ def append_separator(raw_boundary: int) -> None:
375
+ nonlocal canonical_cursor
376
+ separator = _segment(
377
+ ordinal=len(segments),
378
+ kind="separator",
379
+ audio_source="zero_audio",
380
+ absolute_raw_start=raw_boundary,
381
+ absolute_raw_end=raw_boundary,
382
+ canonical_start=canonical_cursor,
383
+ raw_text="",
384
+ canonical_text=" ",
385
+ )
386
+ segments.append(separator)
387
+ canonical_parts.append(separator.canonical_text)
388
+ canonical_cursor = separator.canonical_end
389
+
390
+ def prepare_payload(raw_boundary: int) -> None:
391
+ if canonical_parts and canonical_parts[-1] != " ":
392
+ append_separator(raw_boundary)
393
+
394
+ for piece in pieces:
395
+ if piece.kind == "carrier":
396
+ canonical = _carrier_canonical(piece.text)
397
+ if canonical is None:
398
+ return None
399
+ speech_units = count_speech_units(canonical)
400
+ if canonical:
401
+ prepare_payload(piece.start)
402
+ kind = "carrier" if speech_units > 0 else "control"
403
+ audio_source = "model" if kind == "carrier" else "zero_audio"
404
+ carrier = _segment(
405
+ ordinal=len(segments),
406
+ kind=kind,
407
+ audio_source=audio_source,
408
+ absolute_raw_start=piece.start,
409
+ absolute_raw_end=piece.end,
410
+ canonical_start=canonical_cursor,
411
+ raw_text=piece.text,
412
+ canonical_text=canonical,
413
+ )
414
+ segments.append(carrier)
415
+ if canonical:
416
+ canonical_parts.append(canonical)
417
+ canonical_cursor = carrier.canonical_end
418
+ last_payload_raw_boundary = piece.end
419
+ if kind == "carrier":
420
+ model_generated_chunks += 1
421
+ model_generated_text_units += speech_units
422
+ continue
423
+
424
+ if piece.kind != "identifier":
425
+ raise GlyphHybridPlanError("internal raw piece kind is invalid")
426
+ try:
427
+ # Strict eligibility is checked before adding any segment. There
428
+ # is no chance to render a valid prefix of an invalid run.
429
+ if canonical_parts and canonical_parts[-1] != " ":
430
+ append_separator(piece.start)
431
+ canonical, proof = render_glyph_ascii_v1(
432
+ piece.text,
433
+ asset_entry_sha256_by_asset_id=asset_entry_sha256_by_asset_id,
434
+ absolute_raw_start=piece.start,
435
+ canonical_start=canonical_cursor,
436
+ )
437
+ except GlyphAsciiV1Error:
438
+ return None
439
+
440
+ identifier_ordinal = len(identifier_proofs)
441
+ identifier_proofs.append(proof)
442
+ for atom_index, atom in enumerate(proof.atoms):
443
+ if atom_index:
444
+ append_separator(atom.absolute_raw_start)
445
+ raw_character = chr(atom.raw_byte)
446
+ asset = _segment(
447
+ ordinal=len(segments),
448
+ kind="asset_atom",
449
+ audio_source="asset",
450
+ absolute_raw_start=atom.absolute_raw_start,
451
+ absolute_raw_end=atom.absolute_raw_end,
452
+ canonical_start=atom.canonical_start,
453
+ raw_text=raw_character,
454
+ canonical_text=atom.canonical_token,
455
+ identifier_ordinal=identifier_ordinal,
456
+ atom_ordinal=atom.ordinal,
457
+ asset_id=atom.asset_id,
458
+ manifest_entry_sha256=atom.manifest_entry_sha256,
459
+ )
460
+ segments.append(asset)
461
+ canonical_parts.append(asset.canonical_text)
462
+ canonical_cursor = asset.canonical_end
463
+ asset_canonical_units += count_speech_units(atom.canonical_token)
464
+ if canonical != "".join(
465
+ segment.canonical_text
466
+ for segment in segments
467
+ if (
468
+ segment.identifier_ordinal == identifier_ordinal
469
+ or (
470
+ segment.kind == "separator"
471
+ and proof.canonical_start
472
+ <= segment.canonical_start
473
+ < proof.canonical_end
474
+ )
475
+ )
476
+ ):
477
+ raise GlyphHybridPlanError("core identifier rendering was not preserved")
478
+ last_payload_raw_boundary = piece.end
479
+
480
+ # ``last_payload_raw_boundary`` is an internal construction assertion. A
481
+ # valid network request necessarily emitted at least one canonical atom.
482
+ if last_payload_raw_boundary is None or not identifier_proofs:
483
+ return None
484
+ canonical_target = "".join(canonical_parts)
485
+ asset_atoms = sum(len(proof.atoms) for proof in identifier_proofs)
486
+ reservation = HybridReservation(
487
+ raw_chars=len(raw_request),
488
+ identifiers=len(identifier_proofs),
489
+ asset_atoms=asset_atoms,
490
+ asset_canonical_units=asset_canonical_units,
491
+ asset_audio_placeholder_samples=(
492
+ asset_atoms
493
+ * GLYPH_HYBRID_ASSET_AUDIO_PLACEHOLDER_SAMPLES_PER_ATOM
494
+ ),
495
+ model_generated_chunks=model_generated_chunks,
496
+ model_generated_text_units=model_generated_text_units,
497
+ )
498
+ if (
499
+ reservation.identifiers > GLYPH_HYBRID_MAX_IDENTIFIERS
500
+ or reservation.asset_atoms > GLYPH_HYBRID_MAX_ASSET_ATOMS
501
+ or reservation.asset_canonical_units
502
+ > GLYPH_HYBRID_MAX_ASSET_CANONICAL_UNITS
503
+ or reservation.asset_audio_placeholder_samples
504
+ > GLYPH_HYBRID_MAX_ASSET_AUDIO_PLACEHOLDER_SAMPLES
505
+ or reservation.model_generated_chunks
506
+ > GLYPH_HYBRID_MAX_MODEL_GENERATED_CHUNKS
507
+ or reservation.model_generated_text_units
508
+ > GLYPH_HYBRID_MAX_MODEL_GENERATED_TEXT_UNITS
509
+ ):
510
+ return None
511
+
512
+ unfinished = HybridRenderPlan(
513
+ profile_id=GLYPH_HYBRID_PROFILE_ID,
514
+ grammar_id=GLYPH_ASCII_V1_GRAMMAR_ID,
515
+ normalization_id=GLYPH_HYBRID_NORMALIZATION_ID,
516
+ boundary_contract_id=GLYPH_HYBRID_BOUNDARY_ID,
517
+ raw_input_sha256=_sha256_text(raw_request),
518
+ canonical_target=canonical_target,
519
+ canonical_target_sha256=_sha256_text(canonical_target),
520
+ asset_manifest_logical_sha256=manifest_sha256,
521
+ identifier_proofs=tuple(identifier_proofs),
522
+ segments=tuple(segments),
523
+ reservation=reservation,
524
+ plan_sha256="",
525
+ )
526
+ return HybridRenderPlan(
527
+ **{
528
+ **unfinished.__dict__,
529
+ "plan_sha256": _logical_plan_sha256(unfinished),
530
+ }
531
+ )
532
+
533
+
534
+ def inverse_glyph_hybrid_plan(
535
+ plan: HybridRenderPlan,
536
+ *,
537
+ asset_entry_sha256_by_asset_id: Mapping[str, str],
538
+ expected_raw_request: str,
539
+ ) -> str:
540
+ """Validate from first principles and reconstruct the exact raw request.
541
+
542
+ The original request is mandatory external provenance. This prevents an
543
+ attacker from replacing a whole internally-consistent plan and merely
544
+ recomputing its unkeyed digests.
545
+ """
546
+
547
+ if type(plan) is not HybridRenderPlan:
548
+ raise GlyphHybridPlanError("hybrid plan has an invalid type")
549
+ if type(expected_raw_request) is not str:
550
+ raise GlyphHybridPlanError("expected raw request must be an exact string")
551
+ manifest_sha256 = _manifest_digest(asset_entry_sha256_by_asset_id)
552
+ if (
553
+ type(plan.identifier_proofs) is not tuple
554
+ or type(plan.segments) is not tuple
555
+ or type(plan.reservation) is not HybridReservation
556
+ or plan.profile_id != GLYPH_HYBRID_PROFILE_ID
557
+ or plan.grammar_id != GLYPH_ASCII_V1_GRAMMAR_ID
558
+ or plan.normalization_id != GLYPH_HYBRID_NORMALIZATION_ID
559
+ or plan.boundary_contract_id != GLYPH_HYBRID_BOUNDARY_ID
560
+ or _require_sha256(
561
+ plan.raw_input_sha256,
562
+ field="plan.raw_input_sha256",
563
+ )
564
+ != plan.raw_input_sha256
565
+ or _require_sha256(
566
+ plan.canonical_target_sha256,
567
+ field="plan.canonical_target_sha256",
568
+ )
569
+ != plan.canonical_target_sha256
570
+ or _require_sha256(
571
+ plan.asset_manifest_logical_sha256,
572
+ field="plan.asset_manifest_logical_sha256",
573
+ )
574
+ != plan.asset_manifest_logical_sha256
575
+ or _require_sha256(plan.plan_sha256, field="plan.plan_sha256")
576
+ != plan.plan_sha256
577
+ or plan.asset_manifest_logical_sha256 != manifest_sha256
578
+ or plan.raw_input_sha256 != _sha256_text(expected_raw_request)
579
+ or plan.canonical_target_sha256 != _sha256_text(plan.canonical_target)
580
+ or plan.plan_sha256 != _logical_plan_sha256(plan)
581
+ ):
582
+ raise GlyphHybridPlanError("hybrid plan header is inconsistent")
583
+
584
+ reconstructed_parts: list[str] = []
585
+ canonical_parts: list[str] = []
586
+ raw_cursor = 0
587
+ canonical_cursor = 0
588
+ asset_segments: dict[tuple[int, int], HybridSegmentSpec] = {}
589
+ previous_nonempty_payload = False
590
+ for ordinal, segment in enumerate(plan.segments):
591
+ if type(segment) is not HybridSegmentSpec:
592
+ raise GlyphHybridPlanError("hybrid segment has an invalid type")
593
+ integer_fields = (
594
+ segment.ordinal,
595
+ segment.absolute_raw_start,
596
+ segment.absolute_raw_end,
597
+ segment.canonical_start,
598
+ segment.canonical_end,
599
+ )
600
+ if any(type(value) is not int for value in integer_fields):
601
+ raise GlyphHybridPlanError("hybrid segment integer field is invalid")
602
+ if (
603
+ segment.ordinal != ordinal
604
+ or segment.kind not in _SEGMENT_KINDS
605
+ or segment.audio_source not in _AUDIO_SOURCES
606
+ or type(segment.raw_text) is not str
607
+ or type(segment.canonical_text) is not str
608
+ or segment.absolute_raw_start != raw_cursor
609
+ or segment.absolute_raw_end < segment.absolute_raw_start
610
+ or segment.canonical_start != canonical_cursor
611
+ or segment.canonical_end
612
+ != segment.canonical_start + len(segment.canonical_text)
613
+ or _require_sha256(
614
+ segment.raw_sha256,
615
+ field="segment.raw_sha256",
616
+ )
617
+ != _sha256_text(segment.raw_text)
618
+ or _require_sha256(
619
+ segment.canonical_sha256,
620
+ field="segment.canonical_sha256",
621
+ )
622
+ != _sha256_text(segment.canonical_text)
623
+ ):
624
+ raise GlyphHybridPlanError("hybrid segment is inconsistent")
625
+
626
+ if segment.kind == "separator":
627
+ if (
628
+ segment.audio_source != "zero_audio"
629
+ or segment.absolute_raw_end != segment.absolute_raw_start
630
+ or segment.raw_text
631
+ or segment.canonical_text != " "
632
+ or any(
633
+ value is not None
634
+ for value in (
635
+ segment.identifier_ordinal,
636
+ segment.atom_ordinal,
637
+ segment.asset_id,
638
+ segment.manifest_entry_sha256,
639
+ )
640
+ )
641
+ or not previous_nonempty_payload
642
+ ):
643
+ raise GlyphHybridPlanError("separator contract is inconsistent")
644
+ previous_nonempty_payload = False
645
+ elif segment.kind == "asset_atom":
646
+ if (
647
+ segment.audio_source != "asset"
648
+ or segment.absolute_raw_end != segment.absolute_raw_start + 1
649
+ or len(segment.raw_text) != 1
650
+ or not segment.raw_text.isascii()
651
+ or not segment.canonical_text
652
+ or type(segment.identifier_ordinal) is not int
653
+ or type(segment.atom_ordinal) is not int
654
+ or type(segment.asset_id) is not str
655
+ or _require_sha256(
656
+ segment.manifest_entry_sha256,
657
+ field="segment.manifest_entry_sha256",
658
+ )
659
+ != segment.manifest_entry_sha256
660
+ or previous_nonempty_payload
661
+ ):
662
+ raise GlyphHybridPlanError("asset segment contract is inconsistent")
663
+ key = (segment.identifier_ordinal, segment.atom_ordinal)
664
+ if key in asset_segments:
665
+ raise GlyphHybridPlanError("asset segment provenance is duplicated")
666
+ asset_segments[key] = segment
667
+ reconstructed_parts.append(segment.raw_text)
668
+ raw_cursor = segment.absolute_raw_end
669
+ previous_nonempty_payload = True
670
+ elif segment.kind in {"carrier", "control"}:
671
+ if (
672
+ segment.absolute_raw_end <= segment.absolute_raw_start
673
+ or len(segment.raw_text)
674
+ != segment.absolute_raw_end - segment.absolute_raw_start
675
+ or any(
676
+ value is not None
677
+ for value in (
678
+ segment.identifier_ordinal,
679
+ segment.atom_ordinal,
680
+ segment.asset_id,
681
+ segment.manifest_entry_sha256,
682
+ )
683
+ )
684
+ ):
685
+ raise GlyphHybridPlanError("carrier/control provenance is inconsistent")
686
+ canonical = _carrier_canonical(segment.raw_text)
687
+ speech_units = (
688
+ count_speech_units(segment.canonical_text)
689
+ if canonical is not None
690
+ else -1
691
+ )
692
+ if (
693
+ canonical is None
694
+ or canonical != segment.canonical_text
695
+ or (
696
+ segment.kind == "carrier"
697
+ and (
698
+ segment.audio_source != "model"
699
+ or speech_units <= 0
700
+ )
701
+ )
702
+ or (
703
+ segment.kind == "control"
704
+ and (
705
+ segment.audio_source != "zero_audio"
706
+ or speech_units != 0
707
+ )
708
+ )
709
+ or (segment.canonical_text and previous_nonempty_payload)
710
+ ):
711
+ raise GlyphHybridPlanError("carrier/control contract is inconsistent")
712
+ reconstructed_parts.append(segment.raw_text)
713
+ raw_cursor = segment.absolute_raw_end
714
+ if segment.canonical_text:
715
+ previous_nonempty_payload = True
716
+ else:
717
+ raise GlyphHybridPlanError("unknown hybrid segment kind")
718
+
719
+ canonical_parts.append(segment.canonical_text)
720
+ canonical_cursor = segment.canonical_end
721
+
722
+ reconstructed = "".join(reconstructed_parts)
723
+ canonical = "".join(canonical_parts)
724
+ if (
725
+ raw_cursor != len(reconstructed)
726
+ or reconstructed != expected_raw_request
727
+ or canonical != plan.canonical_target
728
+ or canonical_cursor != len(canonical)
729
+ or canonical.startswith(" ")
730
+ or canonical.endswith(" ")
731
+ or " " in canonical
732
+ ):
733
+ raise GlyphHybridPlanError("hybrid segments do not provide exact coverage")
734
+
735
+ expected_asset_keys: set[tuple[int, int]] = set()
736
+ previous_raw_end = -1
737
+ previous_canonical_end = -1
738
+ for identifier_ordinal, proof in enumerate(plan.identifier_proofs):
739
+ if type(proof) is not GlyphAsciiV1Proof:
740
+ raise GlyphHybridPlanError("identifier proof has an invalid type")
741
+ try:
742
+ raw_identifier = inverse_glyph_ascii_v1(
743
+ proof,
744
+ asset_entry_sha256_by_asset_id=asset_entry_sha256_by_asset_id,
745
+ )
746
+ except GlyphAsciiV1Error as error:
747
+ raise GlyphHybridPlanError("identifier proof is invalid") from error
748
+ if (
749
+ proof.absolute_raw_start <= previous_raw_end
750
+ or proof.canonical_start <= previous_canonical_end
751
+ or expected_raw_request[
752
+ proof.absolute_raw_start : proof.absolute_raw_end
753
+ ]
754
+ != raw_identifier
755
+ or plan.canonical_target[
756
+ proof.canonical_start : proof.canonical_end
757
+ ]
758
+ != " ".join(atom.canonical_token for atom in proof.atoms)
759
+ ):
760
+ raise GlyphHybridPlanError("identifier proof range is inconsistent")
761
+ previous_raw_end = proof.absolute_raw_end
762
+ previous_canonical_end = proof.canonical_end
763
+ for atom in proof.atoms:
764
+ key = (identifier_ordinal, atom.ordinal)
765
+ segment = asset_segments.get(key)
766
+ if (
767
+ segment is None
768
+ or segment.absolute_raw_start != atom.absolute_raw_start
769
+ or segment.absolute_raw_end != atom.absolute_raw_end
770
+ or segment.canonical_start != atom.canonical_start
771
+ or segment.canonical_end != atom.canonical_end
772
+ or ord(segment.raw_text) != atom.raw_byte
773
+ or segment.canonical_text != atom.canonical_token
774
+ or segment.asset_id != atom.asset_id
775
+ or segment.manifest_entry_sha256
776
+ != atom.manifest_entry_sha256
777
+ ):
778
+ raise GlyphHybridPlanError("asset segment does not match its proof")
779
+ expected_asset_keys.add(key)
780
+ if set(asset_segments) != expected_asset_keys:
781
+ raise GlyphHybridPlanError("asset segment set does not match identifier proofs")
782
+
783
+ expected_plan = build_glyph_hybrid_plan(
784
+ reconstructed,
785
+ asset_entry_sha256_by_asset_id=asset_entry_sha256_by_asset_id,
786
+ )
787
+ if (
788
+ expected_plan is None
789
+ or type(expected_plan) is not type(plan)
790
+ or expected_plan != plan
791
+ ):
792
+ raise GlyphHybridPlanError("hybrid plan is not the unique expected plan")
793
+ return reconstructed
tests/test_glyph_hybrid_plan.py ADDED
@@ -0,0 +1,616 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import asdict, replace
4
+ import hashlib
5
+ import json
6
+ import random
7
+
8
+ import pytest
9
+
10
+ from glyph_ascii_v1 import (
11
+ GLYPH_ASCII_V1_SYMBOLS,
12
+ GlyphAsciiV1AtomProof,
13
+ GlyphAsciiV1Proof,
14
+ )
15
+ from glyph_hybrid_plan import (
16
+ GLYPH_HYBRID_ASSET_AUDIO_PLACEHOLDER_SAMPLES_PER_ATOM,
17
+ GLYPH_HYBRID_BOUNDARY_ID,
18
+ GLYPH_HYBRID_MAX_ASSET_ATOMS,
19
+ GLYPH_HYBRID_MAX_ASSET_CANONICAL_UNITS,
20
+ GLYPH_HYBRID_MAX_IDENTIFIERS,
21
+ GLYPH_HYBRID_MAX_MODEL_GENERATED_CHUNKS,
22
+ GLYPH_HYBRID_MAX_MODEL_GENERATED_TEXT_UNITS,
23
+ GLYPH_HYBRID_MAX_RAW_CHARS,
24
+ GLYPH_HYBRID_NORMALIZATION_ID,
25
+ GLYPH_HYBRID_PROFILE_ID,
26
+ GlyphHybridPlanError,
27
+ HybridRenderPlan,
28
+ HybridSegmentSpec,
29
+ build_glyph_hybrid_plan,
30
+ glyph_hybrid_plan_sha256,
31
+ inverse_glyph_hybrid_plan,
32
+ )
33
+ from production import (
34
+ count_speech_units,
35
+ normalize_spoken_forms,
36
+ plan_generation_chunks,
37
+ )
38
+
39
+
40
+ def _asset_manifest(prefix: str = "test") -> dict[str, str]:
41
+ return {
42
+ symbol.asset_id: hashlib.sha256(
43
+ f"{prefix}:{symbol.asset_id}".encode("ascii")
44
+ ).hexdigest()
45
+ for symbol in GLYPH_ASCII_V1_SYMBOLS
46
+ }
47
+
48
+
49
+ ASSET_MANIFEST = _asset_manifest()
50
+
51
+
52
+ def _build(raw: str) -> HybridRenderPlan | None:
53
+ return build_glyph_hybrid_plan(
54
+ raw,
55
+ asset_entry_sha256_by_asset_id=ASSET_MANIFEST,
56
+ )
57
+
58
+
59
+ def _require_plan(raw: str) -> HybridRenderPlan:
60
+ plan = _build(raw)
61
+ assert plan is not None
62
+ return plan
63
+
64
+
65
+ def _inverse(
66
+ plan: HybridRenderPlan,
67
+ raw: str,
68
+ *,
69
+ manifest: dict[str, str] | None = None,
70
+ ) -> str:
71
+ return inverse_glyph_hybrid_plan(
72
+ plan,
73
+ asset_entry_sha256_by_asset_id=(
74
+ ASSET_MANIFEST if manifest is None else manifest
75
+ ),
76
+ expected_raw_request=raw,
77
+ )
78
+
79
+
80
+ @pytest.mark.parametrize(
81
+ "raw",
82
+ (
83
+ "請寄mead@forest.org確認。",
84
+ "客服信箱owl@island.net,下載頁面http://forest.org/release。",
85
+ (
86
+ "第一個是owl@island.net,第二個網址"
87
+ "https://audiocoast.org/tour,完成。"
88
+ ),
89
+ (
90
+ "若要修改設定,請寄信到mead.voice@forest-mail.org,"
91
+ "或開啟https://audio-coast.org/release_notes。"
92
+ ),
93
+ ",mead@forest.org。",
94
+ "mead@forest.org",
95
+ ),
96
+ )
97
+ def test_valid_development_style_requests_have_exact_inverse_and_frozen_headers(raw):
98
+ plan = _require_plan(raw)
99
+
100
+ assert _inverse(plan, raw) == raw
101
+ assert plan.profile_id == GLYPH_HYBRID_PROFILE_ID
102
+ assert plan.normalization_id == GLYPH_HYBRID_NORMALIZATION_ID
103
+ assert plan.boundary_contract_id == GLYPH_HYBRID_BOUNDARY_ID
104
+ assert plan.raw_input_sha256 == hashlib.sha256(raw.encode("utf-8")).hexdigest()
105
+ assert plan.canonical_target_sha256 == hashlib.sha256(
106
+ plan.canonical_target.encode("utf-8")
107
+ ).hexdigest()
108
+ assert plan.plan_sha256 == glyph_hybrid_plan_sha256(plan)
109
+ assert plan.canonical_target == "".join(
110
+ segment.canonical_text for segment in plan.segments
111
+ )
112
+ assert plan.canonical_target == plan.canonical_target.strip()
113
+ assert " " not in plan.canonical_target
114
+
115
+
116
+ def test_multi_identifier_ranges_and_core_renderings_are_absolute_and_disjoint():
117
+ raw = (
118
+ "先寄owl@island.net,再看https://audiocoast.org/tour,"
119
+ "最後通知mead@forest.org。"
120
+ )
121
+ plan = _require_plan(raw)
122
+
123
+ assert len(plan.identifier_proofs) == 3
124
+ previous_raw_end = -1
125
+ previous_canonical_end = -1
126
+ for proof in plan.identifier_proofs:
127
+ assert previous_raw_end < proof.absolute_raw_start < proof.absolute_raw_end
128
+ assert (
129
+ previous_canonical_end
130
+ < proof.canonical_start
131
+ < proof.canonical_end
132
+ )
133
+ assert plan.canonical_target[
134
+ proof.canonical_start : proof.canonical_end
135
+ ] == " ".join(atom.canonical_token for atom in proof.atoms)
136
+ previous_raw_end = proof.absolute_raw_end
137
+ previous_canonical_end = proof.canonical_end
138
+
139
+ raw_coverage = "".join(
140
+ segment.raw_text
141
+ for segment in plan.segments
142
+ if segment.kind != "separator"
143
+ )
144
+ assert raw_coverage == raw
145
+ assert _inverse(plan, raw) == raw
146
+
147
+
148
+ def test_boundary_contract_is_one_explicit_zero_audio_space_between_payloads():
149
+ raw = "開頭mead@forest.org,接著https://audiocoast.org/tour結束。"
150
+ plan = _require_plan(raw)
151
+ nonempty_payload_indices = [
152
+ index
153
+ for index, segment in enumerate(plan.segments)
154
+ if segment.kind != "separator" and segment.canonical_text
155
+ ]
156
+
157
+ assert len(nonempty_payload_indices) > 2
158
+ for left, right in zip(
159
+ nonempty_payload_indices,
160
+ nonempty_payload_indices[1:],
161
+ ):
162
+ assert right == left + 2
163
+ separator = plan.segments[left + 1]
164
+ assert separator.kind == "separator"
165
+ assert separator.audio_source == "zero_audio"
166
+ assert separator.raw_text == ""
167
+ assert separator.absolute_raw_start == separator.absolute_raw_end
168
+ assert separator.canonical_text == " "
169
+ assert all(
170
+ segment.canonical_text != " "
171
+ for segment in plan.segments
172
+ if segment.kind != "separator"
173
+ )
174
+
175
+
176
+ def test_punctuation_only_carriers_are_controls_and_never_model_calls():
177
+ raw = ",mead@forest.org。"
178
+ plan = _require_plan(raw)
179
+ controls = [
180
+ segment for segment in plan.segments if segment.kind == "control"
181
+ ]
182
+
183
+ assert [segment.raw_text for segment in controls] == [",", "。"]
184
+ assert all(segment.audio_source == "zero_audio" for segment in controls)
185
+ assert not any(segment.kind == "carrier" for segment in plan.segments)
186
+ assert plan.reservation.model_generated_chunks == 0
187
+ assert plan.reservation.model_generated_text_units == 0
188
+ assert _inverse(plan, raw) == raw
189
+
190
+
191
+ def test_carrier_uses_existing_zh_tw_normalization_without_network_content():
192
+ raw = "版本v1.2.3,寄到mead@forest.org,日期是2026/07/18。"
193
+ plan = _require_plan(raw)
194
+ carriers = [
195
+ segment for segment in plan.segments if segment.kind == "carrier"
196
+ ]
197
+
198
+ assert [segment.canonical_text for segment in carriers] == [
199
+ normalize_spoken_forms(segment.raw_text, locale="zh-TW")
200
+ for segment in carriers
201
+ ]
202
+ assert all(count_speech_units(segment.canonical_text) > 0 for segment in carriers)
203
+ assert plan.reservation.model_generated_chunks == len(carriers)
204
+ assert plan.reservation.model_generated_text_units == sum(
205
+ count_speech_units(segment.canonical_text) for segment in carriers
206
+ )
207
+
208
+
209
+ def _alpha(rng: random.Random, minimum: int = 1, maximum: int = 7) -> str:
210
+ return "".join(
211
+ rng.choice("abcdefghijklmnopqrstuvwxyz")
212
+ for _ in range(rng.randint(minimum, maximum))
213
+ )
214
+
215
+
216
+ def _email(rng: random.Random) -> str:
217
+ local = _alpha(rng)
218
+ if rng.randrange(2):
219
+ local += rng.choice("._-") + _alpha(rng)
220
+ return f"{local}@{_alpha(rng)}.{_alpha(rng, 1, 3)}"
221
+
222
+
223
+ def _url(rng: random.Random) -> str:
224
+ url = f"{rng.choice(('http', 'https'))}://{_alpha(rng)}.{_alpha(rng, 1, 3)}"
225
+ if rng.randrange(2):
226
+ url += "/" + _alpha(rng)
227
+ return url
228
+
229
+
230
+ def test_seeded_property_valid_multi_identifier_requests_roundtrip_exactly():
231
+ rng = random.Random(0xB10E_5A11)
232
+ carriers = ("請聯絡", ",再查看", ",確認完成。")
233
+ for _ in range(300):
234
+ first = _email(rng)
235
+ second = _url(rng)
236
+ raw = f"{carriers[0]}{first}{carriers[1]}{second}{carriers[2]}"
237
+ plan = _require_plan(raw)
238
+
239
+ assert _inverse(plan, raw) == raw
240
+ assert len(plan.identifier_proofs) == 2
241
+ assert plan.reservation.identifiers == 2
242
+ assert plan.reservation.asset_atoms == len(first) + len(second)
243
+ assert plan.reservation.asset_audio_placeholder_samples == (
244
+ plan.reservation.asset_atoms
245
+ * GLYPH_HYBRID_ASSET_AUDIO_PLACEHOLDER_SAMPLES_PER_ATOM
246
+ )
247
+ assert plan.reservation.model_generated_chunks <= (
248
+ GLYPH_HYBRID_MAX_MODEL_GENERATED_CHUNKS
249
+ )
250
+ assert plan.reservation.model_generated_text_units <= (
251
+ GLYPH_HYBRID_MAX_MODEL_GENERATED_TEXT_UNITS
252
+ )
253
+
254
+
255
+ @pytest.mark.parametrize(
256
+ "raw",
257
+ (
258
+ # Whitespace and ASCII punctuation are part of the maximal run; the
259
+ # planner may not strip them to rescue a substring.
260
+ "聯絡 mead@forest.org。",
261
+ "聯絡mead@forest.org ",
262
+ "聯絡mead@forest.org.",
263
+ "聯絡mead@forest.org,",
264
+ "聯絡mead@forest.org!",
265
+ "請看 https://forest.org/path。",
266
+ "請看:https://forest.org/path。",
267
+ # Native lowercase and strict path/domain grammar are mandatory.
268
+ "聯絡Mead@forest.org。",
269
+ "聯絡mead@Forest.org。",
270
+ "請看HTTP://forest.org/path。",
271
+ "請看https://forest.org/Path。",
272
+ "聯絡mead1@forest.org。",
273
+ "聯絡mead@forest2.org。",
274
+ "請看https://forest.org/path2。",
275
+ "請看https://forest.org/path?query。",
276
+ "請看https://forest.org/path#fragment。",
277
+ "請看https://forest.org:443/path。",
278
+ "請看https://user@forest.org/path。",
279
+ "請看https://user:pass@forest.org/path。",
280
+ "請看https://forest.org/。",
281
+ # Unicode/confusable/bidi/control data may not split a run and lend a
282
+ # valid-looking suffix to the strict renderer.
283
+ "聯絡méad@forest.org。",
284
+ "聯絡mead@forest.org。",
285
+ "聯絡mead@forest.org\u202e。",
286
+ "聯絡mead@forest.org\u2066。",
287
+ "聯絡mead@forest.org\x00。",
288
+ "聯絡mead@forest.org\n。",
289
+ "聯絡mead@forest.org\t。",
290
+ "聯絡梅@forest.org。",
291
+ "請看https://forest.org/路徑。",
292
+ # Other legacy network forms in a carrier make activation partial.
293
+ "聯絡mead@forest.org,備援www.forest.org。",
294
+ "聯絡mead@forest.org,備援ftp://forest.org/path。",
295
+ ),
296
+ )
297
+ def test_malformed_ambiguous_or_partially_eligible_requests_fail_closed(raw):
298
+ assert _build(raw) is None
299
+
300
+
301
+ @pytest.mark.parametrize(
302
+ "raw",
303
+ (
304
+ (
305
+ "聯絡mead@forest.org,備援MEAD@forest.org。"
306
+ ),
307
+ (
308
+ "聯絡mead@forest.org,再看https://forest.org/path?query。"
309
+ ),
310
+ (
311
+ "先看https://forest.org/path,再寄mead1@forest.org。"
312
+ ),
313
+ ),
314
+ )
315
+ def test_one_invalid_network_run_rejects_the_entire_profile_without_partial_plan(raw):
316
+ assert _build(raw) is None
317
+
318
+
319
+ @pytest.mark.parametrize(
320
+ "raw",
321
+ (
322
+ "",
323
+ "今天測試一般句子,保持原有規劃。",
324
+ "forest.org只是普通文字。",
325
+ "網址可能是WWW.FOREST.ORG但不啟用。",
326
+ "大寫HTTPS://FOREST.ORG不會被修正。",
327
+ "電子郵件這四個字不是地址。",
328
+ ),
329
+ )
330
+ def test_no_strict_network_run_never_activates_the_profile(raw):
331
+ assert _build(raw) is None
332
+
333
+
334
+ def test_frozen_reservations_are_computed_before_any_runtime_work():
335
+ raw = "請寄mead@forest.org,再看https://audiocoast.org/tour。"
336
+ plan = _require_plan(raw)
337
+
338
+ assert plan.reservation.raw_chars == len(raw)
339
+ assert plan.reservation.identifiers == len(plan.identifier_proofs)
340
+ assert plan.reservation.asset_atoms == sum(
341
+ len(proof.atoms) for proof in plan.identifier_proofs
342
+ )
343
+ assert plan.reservation.asset_canonical_units == sum(
344
+ count_speech_units(atom.canonical_token)
345
+ for proof in plan.identifier_proofs
346
+ for atom in proof.atoms
347
+ )
348
+ assert plan.reservation.asset_atoms <= GLYPH_HYBRID_MAX_ASSET_ATOMS
349
+ assert (
350
+ plan.reservation.asset_canonical_units
351
+ <= GLYPH_HYBRID_MAX_ASSET_CANONICAL_UNITS
352
+ )
353
+
354
+
355
+ def test_raw_identifier_count_atom_and_canonical_caps_fail_closed():
356
+ overlong = "前" * GLYPH_HYBRID_MAX_RAW_CHARS + "a@b.c"
357
+ too_many_identifiers = ",".join(
358
+ f"{chr(ord('a') + index)}@b.c"
359
+ for index in range(GLYPH_HYBRID_MAX_IDENTIFIERS + 1)
360
+ )
361
+ # 129 raw glyphs; each identifier itself remains under the core cap.
362
+ too_many_atoms = "a" * 125 + "@a.b"
363
+ assert len(too_many_atoms) == GLYPH_HYBRID_MAX_ASSET_ATOMS + 1
364
+ # 105 glyphs but >512 speech units because z -> "字母二十六".
365
+ too_many_canonical_units = "z" * 101 + "@a.b"
366
+ assert len(too_many_canonical_units) < GLYPH_HYBRID_MAX_ASSET_ATOMS
367
+
368
+ assert _build(overlong) is None
369
+ assert _build(too_many_identifiers) is None
370
+ assert _build(too_many_atoms) is None
371
+ assert _build(too_many_canonical_units) is None
372
+
373
+
374
+ def _plan_with_segments(
375
+ plan: HybridRenderPlan,
376
+ segments: tuple[HybridSegmentSpec, ...] | list[HybridSegmentSpec],
377
+ ) -> HybridRenderPlan:
378
+ return replace(plan, segments=segments) # type: ignore[arg-type]
379
+
380
+
381
+ def test_segment_reorder_omit_duplicate_borrow_substitute_and_gap_are_rejected():
382
+ raw = "請寄mead@forest.org確認。"
383
+ plan = _require_plan(raw)
384
+ donor = _require_plan("請寄owl@island.net確認。")
385
+ segments = plan.segments
386
+ first_asset = next(
387
+ index for index, segment in enumerate(segments)
388
+ if segment.kind == "asset_atom"
389
+ )
390
+ donor_asset = next(
391
+ segment for segment in donor.segments
392
+ if segment.kind == "asset_atom"
393
+ )
394
+ substituted = list(segments)
395
+ substituted[first_asset] = replace(
396
+ segments[first_asset],
397
+ raw_text="z",
398
+ raw_sha256=hashlib.sha256(b"z").hexdigest(),
399
+ )
400
+ borrowed = list(segments)
401
+ borrowed[first_asset] = donor_asset
402
+ gapped = list(segments)
403
+ gapped[first_asset] = replace(
404
+ segments[first_asset],
405
+ absolute_raw_start=segments[first_asset].absolute_raw_start + 1,
406
+ )
407
+ overlapped = list(segments)
408
+ overlapped[first_asset] = replace(
409
+ segments[first_asset],
410
+ absolute_raw_start=segments[first_asset].absolute_raw_start - 1,
411
+ )
412
+
413
+ forged = (
414
+ _plan_with_segments(plan, (segments[1], segments[0], *segments[2:])),
415
+ _plan_with_segments(
416
+ plan,
417
+ (*segments[:first_asset], *segments[first_asset + 1 :]),
418
+ ),
419
+ _plan_with_segments(
420
+ plan,
421
+ (*segments[:first_asset], segments[first_asset], *segments[first_asset:]),
422
+ ),
423
+ _plan_with_segments(plan, tuple(borrowed)),
424
+ _plan_with_segments(plan, tuple(substituted)),
425
+ _plan_with_segments(plan, tuple(gapped)),
426
+ _plan_with_segments(plan, tuple(overlapped)),
427
+ )
428
+ for candidate in forged:
429
+ with pytest.raises(GlyphHybridPlanError):
430
+ _inverse(candidate, raw)
431
+
432
+
433
+ @pytest.mark.parametrize(
434
+ "mutate",
435
+ (
436
+ lambda segment: replace(segment, ordinal=segment.ordinal + 1),
437
+ lambda segment: replace(segment, kind="carrier"),
438
+ lambda segment: replace(segment, audio_source="model"),
439
+ lambda segment: replace(
440
+ segment,
441
+ absolute_raw_end=segment.absolute_raw_end + 1,
442
+ ),
443
+ lambda segment: replace(
444
+ segment,
445
+ canonical_start=segment.canonical_start + 1,
446
+ ),
447
+ lambda segment: replace(
448
+ segment,
449
+ canonical_end=segment.canonical_end + 1,
450
+ ),
451
+ lambda segment: replace(segment, canonical_text="字母二十六"),
452
+ lambda segment: replace(segment, raw_sha256="0" * 64),
453
+ lambda segment: replace(segment, canonical_sha256="0" * 64),
454
+ lambda segment: replace(segment, identifier_ordinal=99),
455
+ lambda segment: replace(segment, atom_ordinal=99),
456
+ lambda segment: replace(segment, asset_id="glyph_ascii_v1_letter_z"),
457
+ lambda segment: replace(segment, manifest_entry_sha256="0" * 64),
458
+ ),
459
+ )
460
+ def test_asset_segment_field_provenance_tampering_is_rejected(mutate):
461
+ raw = "請寄mead@forest.org。"
462
+ plan = _require_plan(raw)
463
+ segments = list(plan.segments)
464
+ index = next(
465
+ index for index, segment in enumerate(segments)
466
+ if segment.kind == "asset_atom"
467
+ )
468
+ segments[index] = mutate(segments[index])
469
+
470
+ with pytest.raises(GlyphHybridPlanError):
471
+ _inverse(_plan_with_segments(plan, tuple(segments)), raw)
472
+
473
+
474
+ def test_identifier_atom_and_whole_proof_tampering_is_rejected():
475
+ raw = "請寄mead@forest.org。"
476
+ plan = _require_plan(raw)
477
+ proof = plan.identifier_proofs[0]
478
+ atoms = list(proof.atoms)
479
+ atom: GlyphAsciiV1AtomProof = atoms[0]
480
+ atoms[0] = replace(atom, raw_byte=ord("z"))
481
+ forged_proof: GlyphAsciiV1Proof = replace(proof, atoms=tuple(atoms))
482
+ forged = replace(plan, identifier_proofs=(forged_proof,))
483
+
484
+ with pytest.raises(GlyphHybridPlanError):
485
+ _inverse(forged, raw)
486
+
487
+
488
+ @pytest.mark.parametrize(
489
+ "mutate",
490
+ (
491
+ lambda plan: replace(plan, profile_id="glyph_hybrid_request_v2"),
492
+ lambda plan: replace(plan, grammar_id="glyph_ascii_v2"),
493
+ lambda plan: replace(plan, normalization_id="different"),
494
+ lambda plan: replace(plan, boundary_contract_id="different"),
495
+ lambda plan: replace(plan, raw_input_sha256="0" * 64),
496
+ lambda plan: replace(plan, canonical_target=plan.canonical_target + "X"),
497
+ lambda plan: replace(plan, canonical_target_sha256="0" * 64),
498
+ lambda plan: replace(plan, asset_manifest_logical_sha256="0" * 64),
499
+ lambda plan: replace(plan, plan_sha256="0" * 64),
500
+ lambda plan: replace(
501
+ plan,
502
+ reservation=replace(plan.reservation, asset_atoms=0),
503
+ ),
504
+ lambda plan: replace(plan, segments=list(plan.segments)),
505
+ lambda plan: replace(plan, identifier_proofs=list(plan.identifier_proofs)),
506
+ ),
507
+ )
508
+ def test_header_hash_type_and_reservation_tampering_is_rejected(mutate):
509
+ raw = "請寄mead@forest.org。"
510
+ plan = _require_plan(raw)
511
+ with pytest.raises(GlyphHybridPlanError):
512
+ _inverse(mutate(plan), raw)
513
+
514
+
515
+ def test_external_raw_request_commitment_rejects_a_different_valid_whole_plan():
516
+ raw = "請寄mead@forest.org。"
517
+ other_raw = "請寄owl@island.net。"
518
+ other_plan = _require_plan(other_raw)
519
+
520
+ with pytest.raises(GlyphHybridPlanError):
521
+ _inverse(other_plan, raw)
522
+
523
+
524
+ def test_fresh_rehash_cannot_make_a_changed_reservation_the_expected_plan():
525
+ raw = "請寄mead@forest.org。"
526
+ plan = _require_plan(raw)
527
+ forged = replace(
528
+ plan,
529
+ reservation=replace(
530
+ plan.reservation,
531
+ model_generated_text_units=(
532
+ plan.reservation.model_generated_text_units + 1
533
+ ),
534
+ ),
535
+ plan_sha256="",
536
+ )
537
+ forged = replace(
538
+ forged,
539
+ plan_sha256=glyph_hybrid_plan_sha256(forged),
540
+ )
541
+
542
+ with pytest.raises(GlyphHybridPlanError, match="unique expected plan"):
543
+ _inverse(forged, raw)
544
+
545
+
546
+ def test_manifest_substitution_and_invalid_manifest_configuration_are_rejected():
547
+ raw = "請寄mead@forest.org。"
548
+ plan = _require_plan(raw)
549
+ changed = dict(ASSET_MANIFEST)
550
+ changed[next(iter(changed))] = hashlib.sha256(b"changed").hexdigest()
551
+ missing = dict(ASSET_MANIFEST)
552
+ missing.pop(next(iter(missing)))
553
+
554
+ with pytest.raises(GlyphHybridPlanError):
555
+ _inverse(plan, raw, manifest=changed)
556
+ with pytest.raises(GlyphHybridPlanError):
557
+ build_glyph_hybrid_plan(
558
+ raw,
559
+ asset_entry_sha256_by_asset_id=missing,
560
+ )
561
+
562
+
563
+ def test_non_string_inputs_and_plan_subclasses_are_rejected():
564
+ class ForgedPlan(HybridRenderPlan):
565
+ pass
566
+
567
+ raw = "請寄mead@forest.org。"
568
+ plan = _require_plan(raw)
569
+ forged = ForgedPlan(**plan.__dict__)
570
+
571
+ with pytest.raises(GlyphHybridPlanError):
572
+ build_glyph_hybrid_plan( # type: ignore[arg-type]
573
+ b"mead@forest.org",
574
+ asset_entry_sha256_by_asset_id=ASSET_MANIFEST,
575
+ )
576
+ with pytest.raises(GlyphHybridPlanError):
577
+ _inverse(forged, raw)
578
+ with pytest.raises(GlyphHybridPlanError):
579
+ inverse_glyph_hybrid_plan(
580
+ plan,
581
+ asset_entry_sha256_by_asset_id=ASSET_MANIFEST,
582
+ expected_raw_request=b"mead@forest.org", # type: ignore[arg-type]
583
+ )
584
+
585
+
586
+ def test_no_network_legacy_normalization_and_plan_remain_byte_for_byte_frozen():
587
+ samples = (
588
+ "今天測試一般句子,保持原有規劃。",
589
+ "日期是2026/07/18,數值為12.5%。",
590
+ "型號RTX 5090將沿用舊前端。",
591
+ )
592
+ output = []
593
+ for raw in samples:
594
+ assert _build(raw) is None
595
+ normalized = normalize_spoken_forms(raw, locale="zh-TW")
596
+ output.append(
597
+ {
598
+ "raw": raw,
599
+ "normalized": normalized,
600
+ "plan": [
601
+ asdict(spec)
602
+ for spec in plan_generation_chunks(raw, normalized)
603
+ ],
604
+ }
605
+ )
606
+ serialized = json.dumps(
607
+ output,
608
+ ensure_ascii=False,
609
+ sort_keys=True,
610
+ separators=(",", ":"),
611
+ ).encode("utf-8")
612
+
613
+ assert len(serialized) == 1_128
614
+ assert hashlib.sha256(serialized).hexdigest() == (
615
+ "d4986f95a0912bc76332a8f0215a22b0359c1fe8ed90733bea99df39535d1759"
616
+ )