Codex commited on
Commit
bea21d4
·
1 Parent(s): 9e5f509

Add strict glyph ASCII renderer proof core

Browse files
Files changed (2) hide show
  1. glyph_ascii_v1.py +465 -0
  2. tests/test_glyph_ascii_v1.py +507 -0
glyph_ascii_v1.py ADDED
@@ -0,0 +1,465 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Strict, reversible ASCII identifier rendering for pre-certified glyph audio.
2
+
3
+ This module deliberately contains no runtime generation, audio, asset loading,
4
+ or release-evidence integration. It defines only the fail-closed
5
+ ``glyph_ascii_v1`` grammar and the proof object that a later integration layer
6
+ can bind to a frozen asset manifest.
7
+
8
+ The accepted language is intentionally narrower than RFC email/URL syntax:
9
+
10
+ * every input byte must already be lowercase ASCII (there is no case folding);
11
+ * email local parts, DNS labels, and URL path segments contain letters plus a
12
+ small set of non-consecutive separators;
13
+ * URLs are exactly ``http`` or ``https``, have no authority decorations, and
14
+ have no query or fragment.
15
+
16
+ Every accepted byte maps bijectively to one of 32 canonical spoken tokens.
17
+ Tokens in a complete rendering are separated by exactly one ASCII space.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ from collections.abc import Mapping
23
+ from dataclasses import dataclass
24
+ import hashlib
25
+ import json
26
+ import re
27
+
28
+
29
+ GLYPH_ASCII_V1_GRAMMAR_ID = "glyph_ascii_v1"
30
+ GLYPH_ASCII_V1_MAX_IDENTIFIER_BYTES = 512
31
+ GLYPH_ASCII_V1_MAX_ATOMS = 512
32
+ GLYPH_ASCII_V1_MAX_CANONICAL_CHARS = 8_192
33
+
34
+ _SHA256_RE = re.compile(r"[0-9a-f]{64}\Z", flags=re.ASCII)
35
+ _ALPHA = r"[a-z]+"
36
+ _LABEL = rf"{_ALPHA}(?:-{_ALPHA})*"
37
+ _DOMAIN = rf"{_LABEL}(?:\.{_LABEL})+"
38
+ _LOCAL = rf"{_ALPHA}(?:[._-]{_ALPHA})*"
39
+ _PATH_SEGMENT = rf"{_ALPHA}(?:[-_.]{_ALPHA})*"
40
+ _EMAIL_RE = re.compile(rf"{_LOCAL}@{_DOMAIN}\Z", flags=re.ASCII)
41
+ _URL_RE = re.compile(
42
+ rf"(?:http|https)://{_DOMAIN}(?:/{_PATH_SEGMENT}(?:/{_PATH_SEGMENT})*)?\Z",
43
+ flags=re.ASCII,
44
+ )
45
+
46
+ _ZH_ORDINALS = (
47
+ "一",
48
+ "二",
49
+ "三",
50
+ "四",
51
+ "五",
52
+ "六",
53
+ "七",
54
+ "八",
55
+ "九",
56
+ "十",
57
+ "十一",
58
+ "十二",
59
+ "十三",
60
+ "十四",
61
+ "十五",
62
+ "十六",
63
+ "十七",
64
+ "十八",
65
+ "十九",
66
+ "二十",
67
+ "二十一",
68
+ "二十二",
69
+ "二十三",
70
+ "二十四",
71
+ "二十五",
72
+ "二十六",
73
+ )
74
+
75
+
76
+ class GlyphAsciiV1Error(ValueError):
77
+ """Raised when an identifier, manifest binding, or proof is invalid."""
78
+
79
+
80
+ @dataclass(frozen=True)
81
+ class GlyphAsciiV1Symbol:
82
+ """One member of the frozen 32-byte canonical alphabet."""
83
+
84
+ raw_byte: int
85
+ canonical_token: str
86
+ asset_id: str
87
+
88
+
89
+ _LETTER_SYMBOLS = tuple(
90
+ GlyphAsciiV1Symbol(
91
+ raw_byte=ord(character),
92
+ canonical_token=f"字母{ordinal}",
93
+ asset_id=f"glyph_ascii_v1_letter_{character}",
94
+ )
95
+ for character, ordinal in zip("abcdefghijklmnopqrstuvwxyz", _ZH_ORDINALS)
96
+ )
97
+ _PUNCTUATION_SYMBOLS = (
98
+ GlyphAsciiV1Symbol(ord("@"), "小老鼠", "glyph_ascii_v1_at"),
99
+ GlyphAsciiV1Symbol(ord("."), "點", "glyph_ascii_v1_dot"),
100
+ GlyphAsciiV1Symbol(ord("-"), "橫線", "glyph_ascii_v1_hyphen"),
101
+ GlyphAsciiV1Symbol(ord("_"), "底線", "glyph_ascii_v1_underscore"),
102
+ GlyphAsciiV1Symbol(ord(":"), "冒號", "glyph_ascii_v1_colon"),
103
+ GlyphAsciiV1Symbol(ord("/"), "斜線", "glyph_ascii_v1_slash"),
104
+ )
105
+ GLYPH_ASCII_V1_SYMBOLS = _LETTER_SYMBOLS + _PUNCTUATION_SYMBOLS
106
+
107
+ _SYMBOL_BY_RAW_BYTE = {symbol.raw_byte: symbol for symbol in GLYPH_ASCII_V1_SYMBOLS}
108
+ _SYMBOL_BY_CANONICAL_TOKEN = {
109
+ symbol.canonical_token: symbol for symbol in GLYPH_ASCII_V1_SYMBOLS
110
+ }
111
+ _EXPECTED_ASSET_IDS = tuple(symbol.asset_id for symbol in GLYPH_ASCII_V1_SYMBOLS)
112
+
113
+ if (
114
+ len(GLYPH_ASCII_V1_SYMBOLS) != 32
115
+ or len(_SYMBOL_BY_RAW_BYTE) != 32
116
+ or len(_SYMBOL_BY_CANONICAL_TOKEN) != 32
117
+ or len(set(_EXPECTED_ASSET_IDS)) != 32
118
+ ):
119
+ raise RuntimeError("glyph_ascii_v1 alphabet is not bijective")
120
+
121
+
122
+ @dataclass(frozen=True)
123
+ class GlyphAsciiV1AtomProof:
124
+ """Range- and manifest-bound proof for one exact source byte."""
125
+
126
+ ordinal: int
127
+ absolute_raw_start: int
128
+ absolute_raw_end: int
129
+ identifier_raw_start: int
130
+ identifier_raw_end: int
131
+ canonical_start: int
132
+ canonical_end: int
133
+ raw_byte: int
134
+ canonical_token: str
135
+ asset_id: str
136
+ manifest_entry_sha256: str
137
+
138
+
139
+ @dataclass(frozen=True)
140
+ class GlyphAsciiV1Proof:
141
+ """Complete canonical rendering proof for one accepted identifier."""
142
+
143
+ grammar_id: str
144
+ identifier_kind: str
145
+ absolute_raw_start: int
146
+ absolute_raw_end: int
147
+ canonical_start: int
148
+ canonical_end: int
149
+ raw_sha256: str
150
+ canonical_sha256: str
151
+ asset_manifest_sha256: str
152
+ atoms: tuple[GlyphAsciiV1AtomProof, ...]
153
+
154
+
155
+ def _require_plain_nonnegative_int(value: object, *, field: str) -> int:
156
+ if type(value) is not int or value < 0:
157
+ raise GlyphAsciiV1Error(f"{field} must be a non-negative integer")
158
+ return value
159
+
160
+
161
+ def _require_sha256(value: object, *, field: str) -> str:
162
+ if type(value) is not str or _SHA256_RE.fullmatch(value) is None:
163
+ raise GlyphAsciiV1Error(f"{field} must be a lowercase SHA-256 digest")
164
+ return value
165
+
166
+
167
+ def _validated_manifest(
168
+ asset_entry_sha256_by_asset_id: Mapping[str, str],
169
+ ) -> tuple[tuple[str, str], ...]:
170
+ if not isinstance(asset_entry_sha256_by_asset_id, Mapping):
171
+ raise GlyphAsciiV1Error("asset manifest entries must be a mapping")
172
+ if set(asset_entry_sha256_by_asset_id) != set(_EXPECTED_ASSET_IDS):
173
+ raise GlyphAsciiV1Error(
174
+ "asset manifest must contain exactly the 32 glyph_ascii_v1 asset ids"
175
+ )
176
+ entries: list[tuple[str, str]] = []
177
+ for asset_id in _EXPECTED_ASSET_IDS:
178
+ if type(asset_id) is not str:
179
+ raise GlyphAsciiV1Error("asset id has an invalid type")
180
+ digest = _require_sha256(
181
+ asset_entry_sha256_by_asset_id[asset_id],
182
+ field=f"asset manifest entry {asset_id!r}",
183
+ )
184
+ entries.append((asset_id, digest))
185
+ return tuple(entries)
186
+
187
+
188
+ def _asset_manifest_sha256_from_entries(
189
+ entries: tuple[tuple[str, str], ...],
190
+ ) -> str:
191
+ payload = json.dumps(
192
+ {
193
+ "grammar_id": GLYPH_ASCII_V1_GRAMMAR_ID,
194
+ "entries": [
195
+ {"asset_id": asset_id, "sha256": digest}
196
+ for asset_id, digest in entries
197
+ ],
198
+ },
199
+ ensure_ascii=True,
200
+ separators=(",", ":"),
201
+ sort_keys=True,
202
+ ).encode("ascii")
203
+ return hashlib.sha256(payload).hexdigest()
204
+
205
+
206
+ def glyph_ascii_v1_asset_manifest_sha256(
207
+ asset_entry_sha256_by_asset_id: Mapping[str, str],
208
+ ) -> str:
209
+ """Return the deterministic logical-manifest digest for all 32 assets."""
210
+
211
+ return _asset_manifest_sha256_from_entries(
212
+ _validated_manifest(asset_entry_sha256_by_asset_id)
213
+ )
214
+
215
+
216
+ def _strict_raw_bytes(raw_identifier: str) -> bytes:
217
+ if type(raw_identifier) is not str:
218
+ raise GlyphAsciiV1Error("raw identifier must be a string")
219
+ # This is intentionally lower(), never casefold(). Eligibility requires
220
+ # the source to arrive in its canonical lowercase form.
221
+ if raw_identifier != raw_identifier.lower():
222
+ raise GlyphAsciiV1Error("raw identifier is not native lowercase")
223
+ try:
224
+ raw_bytes = raw_identifier.encode("ascii", errors="strict")
225
+ except UnicodeEncodeError as error:
226
+ raise GlyphAsciiV1Error("raw identifier is not strict ASCII") from error
227
+ if not raw_bytes:
228
+ raise GlyphAsciiV1Error("raw identifier is empty")
229
+ if len(raw_bytes) > GLYPH_ASCII_V1_MAX_IDENTIFIER_BYTES:
230
+ raise GlyphAsciiV1Error("raw identifier exceeds the byte cap")
231
+ if len(raw_bytes) > GLYPH_ASCII_V1_MAX_ATOMS:
232
+ raise GlyphAsciiV1Error("raw identifier exceeds the atom cap")
233
+ return raw_bytes
234
+
235
+
236
+ def glyph_ascii_v1_identifier_kind(raw_identifier: str) -> str:
237
+ """Parse the exact v1 byte grammar and return ``email`` or ``url``."""
238
+
239
+ raw_bytes = _strict_raw_bytes(raw_identifier)
240
+ # Match the decoded ASCII only after the byte-level eligibility checks.
241
+ raw = raw_bytes.decode("ascii")
242
+ matches = tuple(
243
+ kind
244
+ for kind, pattern in (("email", _EMAIL_RE), ("url", _URL_RE))
245
+ if pattern.fullmatch(raw) is not None
246
+ )
247
+ if len(matches) != 1:
248
+ raise GlyphAsciiV1Error("raw identifier is outside the unambiguous v1 grammar")
249
+ return matches[0]
250
+
251
+
252
+ def render_glyph_ascii_v1(
253
+ raw_identifier: str,
254
+ *,
255
+ asset_entry_sha256_by_asset_id: Mapping[str, str],
256
+ absolute_raw_start: int = 0,
257
+ canonical_start: int = 0,
258
+ ) -> tuple[str, GlyphAsciiV1Proof]:
259
+ """Render one identifier and bind every byte to its frozen asset entry."""
260
+
261
+ absolute_raw_start = _require_plain_nonnegative_int(
262
+ absolute_raw_start,
263
+ field="absolute_raw_start",
264
+ )
265
+ canonical_start = _require_plain_nonnegative_int(
266
+ canonical_start,
267
+ field="canonical_start",
268
+ )
269
+ identifier_kind = glyph_ascii_v1_identifier_kind(raw_identifier)
270
+ raw_bytes = raw_identifier.encode("ascii")
271
+ manifest_entries = _validated_manifest(asset_entry_sha256_by_asset_id)
272
+ entry_sha_by_asset_id = dict(manifest_entries)
273
+ manifest_sha256 = _asset_manifest_sha256_from_entries(manifest_entries)
274
+
275
+ tokens: list[str] = []
276
+ atoms: list[GlyphAsciiV1AtomProof] = []
277
+ canonical_cursor = canonical_start
278
+ for ordinal, raw_byte in enumerate(raw_bytes):
279
+ symbol = _SYMBOL_BY_RAW_BYTE.get(raw_byte)
280
+ if symbol is None:
281
+ # The grammar and alphabet are intentionally independent checks.
282
+ raise GlyphAsciiV1Error("accepted grammar byte lacks a canonical atom")
283
+ if ordinal:
284
+ canonical_cursor += 1 # exactly one ASCII space between atoms
285
+ token_start = canonical_cursor
286
+ token_end = token_start + len(symbol.canonical_token)
287
+ tokens.append(symbol.canonical_token)
288
+ atoms.append(
289
+ GlyphAsciiV1AtomProof(
290
+ ordinal=ordinal,
291
+ absolute_raw_start=absolute_raw_start + ordinal,
292
+ absolute_raw_end=absolute_raw_start + ordinal + 1,
293
+ identifier_raw_start=ordinal,
294
+ identifier_raw_end=ordinal + 1,
295
+ canonical_start=token_start,
296
+ canonical_end=token_end,
297
+ raw_byte=raw_byte,
298
+ canonical_token=symbol.canonical_token,
299
+ asset_id=symbol.asset_id,
300
+ manifest_entry_sha256=entry_sha_by_asset_id[symbol.asset_id],
301
+ )
302
+ )
303
+ canonical_cursor = token_end
304
+
305
+ canonical = " ".join(tokens)
306
+ if len(canonical) > GLYPH_ASCII_V1_MAX_CANONICAL_CHARS:
307
+ raise GlyphAsciiV1Error("canonical rendering exceeds the character cap")
308
+ proof = GlyphAsciiV1Proof(
309
+ grammar_id=GLYPH_ASCII_V1_GRAMMAR_ID,
310
+ identifier_kind=identifier_kind,
311
+ absolute_raw_start=absolute_raw_start,
312
+ absolute_raw_end=absolute_raw_start + len(raw_bytes),
313
+ canonical_start=canonical_start,
314
+ canonical_end=canonical_start + len(canonical),
315
+ raw_sha256=hashlib.sha256(raw_bytes).hexdigest(),
316
+ canonical_sha256=hashlib.sha256(canonical.encode("utf-8")).hexdigest(),
317
+ asset_manifest_sha256=manifest_sha256,
318
+ atoms=tuple(atoms),
319
+ )
320
+ return canonical, proof
321
+
322
+
323
+ def inverse_glyph_ascii_v1(
324
+ proof: GlyphAsciiV1Proof,
325
+ *,
326
+ asset_entry_sha256_by_asset_id: Mapping[str, str],
327
+ ) -> str:
328
+ """Validate a proof from first principles and reconstruct its exact bytes."""
329
+
330
+ if type(proof) is not GlyphAsciiV1Proof:
331
+ raise GlyphAsciiV1Error("glyph proof has an invalid type")
332
+ manifest_entries = _validated_manifest(asset_entry_sha256_by_asset_id)
333
+ entry_sha_by_asset_id = dict(manifest_entries)
334
+ manifest_sha256 = _asset_manifest_sha256_from_entries(manifest_entries)
335
+ if (
336
+ type(proof.grammar_id) is not str
337
+ or proof.grammar_id != GLYPH_ASCII_V1_GRAMMAR_ID
338
+ or type(proof.identifier_kind) is not str
339
+ or proof.identifier_kind not in {"email", "url"}
340
+ or _require_plain_nonnegative_int(
341
+ proof.absolute_raw_start,
342
+ field="proof.absolute_raw_start",
343
+ )
344
+ != proof.absolute_raw_start
345
+ or _require_plain_nonnegative_int(
346
+ proof.absolute_raw_end,
347
+ field="proof.absolute_raw_end",
348
+ )
349
+ != proof.absolute_raw_end
350
+ or _require_plain_nonnegative_int(
351
+ proof.canonical_start,
352
+ field="proof.canonical_start",
353
+ )
354
+ != proof.canonical_start
355
+ or _require_plain_nonnegative_int(
356
+ proof.canonical_end,
357
+ field="proof.canonical_end",
358
+ )
359
+ != proof.canonical_end
360
+ or proof.absolute_raw_end <= proof.absolute_raw_start
361
+ or proof.canonical_end <= proof.canonical_start
362
+ or _require_sha256(proof.raw_sha256, field="proof.raw_sha256")
363
+ != proof.raw_sha256
364
+ or _require_sha256(proof.canonical_sha256, field="proof.canonical_sha256")
365
+ != proof.canonical_sha256
366
+ or _require_sha256(
367
+ proof.asset_manifest_sha256,
368
+ field="proof.asset_manifest_sha256",
369
+ )
370
+ != proof.asset_manifest_sha256
371
+ or proof.asset_manifest_sha256 != manifest_sha256
372
+ or type(proof.atoms) is not tuple
373
+ or not proof.atoms
374
+ or len(proof.atoms) > GLYPH_ASCII_V1_MAX_ATOMS
375
+ ):
376
+ raise GlyphAsciiV1Error("glyph proof header is inconsistent")
377
+
378
+ raw_bytes = bytearray()
379
+ canonical_tokens: list[str] = []
380
+ raw_absolute_cursor = proof.absolute_raw_start
381
+ raw_identifier_cursor = 0
382
+ canonical_cursor = proof.canonical_start
383
+ for ordinal, atom in enumerate(proof.atoms):
384
+ if type(atom) is not GlyphAsciiV1AtomProof:
385
+ raise GlyphAsciiV1Error("glyph atom proof has an invalid type")
386
+ if (
387
+ type(atom.canonical_token) is not str
388
+ or type(atom.asset_id) is not str
389
+ ):
390
+ raise GlyphAsciiV1Error("glyph atom text field has an invalid type")
391
+ integer_fields = (
392
+ atom.ordinal,
393
+ atom.absolute_raw_start,
394
+ atom.absolute_raw_end,
395
+ atom.identifier_raw_start,
396
+ atom.identifier_raw_end,
397
+ atom.canonical_start,
398
+ atom.canonical_end,
399
+ atom.raw_byte,
400
+ )
401
+ if any(type(value) is not int for value in integer_fields):
402
+ raise GlyphAsciiV1Error("glyph atom integer field has an invalid type")
403
+ entry_sha256 = _require_sha256(
404
+ atom.manifest_entry_sha256,
405
+ field="atom.manifest_entry_sha256",
406
+ )
407
+ symbol = _SYMBOL_BY_CANONICAL_TOKEN.get(atom.canonical_token)
408
+ expected_canonical_start = canonical_cursor + (1 if ordinal else 0)
409
+ if (
410
+ atom.ordinal != ordinal
411
+ or atom.absolute_raw_start != raw_absolute_cursor
412
+ or atom.absolute_raw_end != raw_absolute_cursor + 1
413
+ or atom.identifier_raw_start != raw_identifier_cursor
414
+ or atom.identifier_raw_end != raw_identifier_cursor + 1
415
+ or atom.canonical_start != expected_canonical_start
416
+ or atom.canonical_end
417
+ != expected_canonical_start + len(atom.canonical_token)
418
+ or symbol is None
419
+ or symbol.raw_byte != atom.raw_byte
420
+ or symbol.asset_id != atom.asset_id
421
+ or entry_sha256 != atom.manifest_entry_sha256
422
+ or entry_sha_by_asset_id.get(atom.asset_id)
423
+ != atom.manifest_entry_sha256
424
+ ):
425
+ raise GlyphAsciiV1Error("glyph atom proof is inconsistent")
426
+ raw_bytes.append(atom.raw_byte)
427
+ canonical_tokens.append(atom.canonical_token)
428
+ raw_absolute_cursor = atom.absolute_raw_end
429
+ raw_identifier_cursor = atom.identifier_raw_end
430
+ canonical_cursor = atom.canonical_end
431
+
432
+ canonical = " ".join(canonical_tokens)
433
+ try:
434
+ raw = bytes(raw_bytes).decode("ascii", errors="strict")
435
+ except UnicodeDecodeError as error:
436
+ raise GlyphAsciiV1Error("glyph proof does not reconstruct ASCII") from error
437
+ if (
438
+ raw_absolute_cursor != proof.absolute_raw_end
439
+ or raw_identifier_cursor != len(raw_bytes)
440
+ or canonical_cursor != proof.canonical_end
441
+ or proof.absolute_raw_end
442
+ != proof.absolute_raw_start + len(raw_bytes)
443
+ or proof.canonical_end != proof.canonical_start + len(canonical)
444
+ or hashlib.sha256(bytes(raw_bytes)).hexdigest() != proof.raw_sha256
445
+ or hashlib.sha256(canonical.encode("utf-8")).hexdigest()
446
+ != proof.canonical_sha256
447
+ ):
448
+ raise GlyphAsciiV1Error("glyph proof does not cover its declared ranges")
449
+
450
+ # Re-parse the reconstructed bytes, then rebuild the only valid proof.
451
+ kind = glyph_ascii_v1_identifier_kind(raw)
452
+ expected_canonical, expected_proof = render_glyph_ascii_v1(
453
+ raw,
454
+ asset_entry_sha256_by_asset_id=entry_sha_by_asset_id,
455
+ absolute_raw_start=proof.absolute_raw_start,
456
+ canonical_start=proof.canonical_start,
457
+ )
458
+ if (
459
+ kind != proof.identifier_kind
460
+ or expected_canonical != canonical
461
+ or type(expected_proof) is not type(proof)
462
+ or expected_proof != proof
463
+ ):
464
+ raise GlyphAsciiV1Error("glyph proof is not the canonical expected proof")
465
+ return raw
tests/test_glyph_ascii_v1.py ADDED
@@ -0,0 +1,507 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from __future__ import annotations
2
+
3
+ from dataclasses import asdict, replace
4
+ import hashlib
5
+ import json
6
+ import random
7
+
8
+ import pytest
9
+
10
+ from glyph_ascii_v1 import (
11
+ GLYPH_ASCII_V1_GRAMMAR_ID,
12
+ GLYPH_ASCII_V1_MAX_IDENTIFIER_BYTES,
13
+ GLYPH_ASCII_V1_SYMBOLS,
14
+ GlyphAsciiV1AtomProof,
15
+ GlyphAsciiV1Error,
16
+ GlyphAsciiV1Proof,
17
+ glyph_ascii_v1_asset_manifest_sha256,
18
+ glyph_ascii_v1_identifier_kind,
19
+ inverse_glyph_ascii_v1,
20
+ render_glyph_ascii_v1,
21
+ )
22
+ from production import normalize_spoken_forms, plan_generation_chunks
23
+
24
+
25
+ def _asset_manifest() -> dict[str, str]:
26
+ return {
27
+ symbol.asset_id: hashlib.sha256(
28
+ f"test-asset:{symbol.asset_id}".encode("ascii")
29
+ ).hexdigest()
30
+ for symbol in GLYPH_ASCII_V1_SYMBOLS
31
+ }
32
+
33
+
34
+ ASSET_MANIFEST = _asset_manifest()
35
+
36
+
37
+ def _render(
38
+ raw: str,
39
+ *,
40
+ absolute_raw_start: int = 0,
41
+ canonical_start: int = 0,
42
+ ) -> tuple[str, GlyphAsciiV1Proof]:
43
+ return render_glyph_ascii_v1(
44
+ raw,
45
+ asset_entry_sha256_by_asset_id=ASSET_MANIFEST,
46
+ absolute_raw_start=absolute_raw_start,
47
+ canonical_start=canonical_start,
48
+ )
49
+
50
+
51
+ def _inverse(
52
+ proof: GlyphAsciiV1Proof,
53
+ *,
54
+ manifest: dict[str, str] | None = None,
55
+ ) -> str:
56
+ return inverse_glyph_ascii_v1(
57
+ proof,
58
+ asset_entry_sha256_by_asset_id=(
59
+ ASSET_MANIFEST if manifest is None else manifest
60
+ ),
61
+ )
62
+
63
+
64
+ def test_exact_32_symbol_bijection_and_frozen_tokens():
65
+ expected_characters = "abcdefghijklmnopqrstuvwxyz@.-_:/"
66
+ expected_tokens = (
67
+ "字母一",
68
+ "字母二",
69
+ "字母三",
70
+ "字母四",
71
+ "字母五",
72
+ "字母六",
73
+ "字母七",
74
+ "字母八",
75
+ "字母九",
76
+ "字母十",
77
+ "字母十一",
78
+ "字母十二",
79
+ "字母十三",
80
+ "字母十四",
81
+ "字母十五",
82
+ "字母十六",
83
+ "字母十七",
84
+ "字母十八",
85
+ "字母十九",
86
+ "字母二十",
87
+ "字母二十一",
88
+ "字母二十二",
89
+ "字母二十三",
90
+ "字母二十四",
91
+ "字母二十五",
92
+ "字母二十六",
93
+ "小老鼠",
94
+ "點",
95
+ "橫線",
96
+ "底線",
97
+ "冒號",
98
+ "斜線",
99
+ )
100
+
101
+ assert len(GLYPH_ASCII_V1_SYMBOLS) == 32
102
+ assert "".join(chr(symbol.raw_byte) for symbol in GLYPH_ASCII_V1_SYMBOLS) == (
103
+ expected_characters
104
+ )
105
+ assert tuple(symbol.canonical_token for symbol in GLYPH_ASCII_V1_SYMBOLS) == (
106
+ expected_tokens
107
+ )
108
+ assert len({symbol.raw_byte for symbol in GLYPH_ASCII_V1_SYMBOLS}) == 32
109
+ assert len({symbol.canonical_token for symbol in GLYPH_ASCII_V1_SYMBOLS}) == 32
110
+ assert len({symbol.asset_id for symbol in GLYPH_ASCII_V1_SYMBOLS}) == 32
111
+
112
+
113
+ @pytest.mark.parametrize(
114
+ ("raw", "kind"),
115
+ (
116
+ ("a@b.c", "email"),
117
+ ("a.b_c-d@e-f.g-h", "email"),
118
+ ("http://a.b", "url"),
119
+ ("https://a-b.c-d/e_f.g-h/i", "url"),
120
+ ),
121
+ )
122
+ def test_strict_examples_roundtrip_with_range_and_manifest_binding(raw, kind):
123
+ absolute_raw_start = 19
124
+ canonical_start = 31
125
+ canonical, proof = _render(
126
+ raw,
127
+ absolute_raw_start=absolute_raw_start,
128
+ canonical_start=canonical_start,
129
+ )
130
+ symbols = {symbol.raw_byte: symbol for symbol in GLYPH_ASCII_V1_SYMBOLS}
131
+
132
+ assert glyph_ascii_v1_identifier_kind(raw) == kind
133
+ assert proof.grammar_id == GLYPH_ASCII_V1_GRAMMAR_ID
134
+ assert proof.identifier_kind == kind
135
+ assert proof.absolute_raw_start == absolute_raw_start
136
+ assert proof.absolute_raw_end == absolute_raw_start + len(raw.encode("ascii"))
137
+ assert proof.canonical_start == canonical_start
138
+ assert proof.canonical_end == canonical_start + len(canonical)
139
+ assert proof.raw_sha256 == hashlib.sha256(raw.encode("ascii")).hexdigest()
140
+ assert proof.canonical_sha256 == hashlib.sha256(
141
+ canonical.encode("utf-8")
142
+ ).hexdigest()
143
+ assert proof.asset_manifest_sha256 == glyph_ascii_v1_asset_manifest_sha256(
144
+ ASSET_MANIFEST
145
+ )
146
+ assert canonical == " ".join(
147
+ symbols[raw_byte].canonical_token for raw_byte in raw.encode("ascii")
148
+ )
149
+ assert " " not in canonical
150
+ assert _inverse(proof) == raw
151
+
152
+ for ordinal, (raw_byte, atom) in enumerate(zip(raw.encode("ascii"), proof.atoms)):
153
+ symbol = symbols[raw_byte]
154
+ assert atom.ordinal == ordinal
155
+ assert (
156
+ atom.absolute_raw_start,
157
+ atom.absolute_raw_end,
158
+ ) == (
159
+ absolute_raw_start + ordinal,
160
+ absolute_raw_start + ordinal + 1,
161
+ )
162
+ assert (
163
+ atom.identifier_raw_start,
164
+ atom.identifier_raw_end,
165
+ ) == (ordinal, ordinal + 1)
166
+ assert canonical[
167
+ atom.canonical_start - canonical_start :
168
+ atom.canonical_end - canonical_start
169
+ ] == atom.canonical_token
170
+ assert atom.raw_byte == raw_byte
171
+ assert atom.canonical_token == symbol.canonical_token
172
+ assert atom.asset_id == symbol.asset_id
173
+ assert atom.manifest_entry_sha256 == ASSET_MANIFEST[symbol.asset_id]
174
+
175
+
176
+ def _alpha(rng: random.Random, *, minimum: int = 1, maximum: int = 8) -> str:
177
+ return "".join(
178
+ rng.choice("abcdefghijklmnopqrstuvwxyz")
179
+ for _ in range(rng.randint(minimum, maximum))
180
+ )
181
+
182
+
183
+ def _joined_alpha(
184
+ rng: random.Random,
185
+ separators: str,
186
+ *,
187
+ maximum_parts: int = 4,
188
+ ) -> str:
189
+ output = _alpha(rng)
190
+ for _ in range(rng.randint(0, maximum_parts - 1)):
191
+ output += rng.choice(separators) + _alpha(rng)
192
+ return output
193
+
194
+
195
+ def _random_email(rng: random.Random) -> str:
196
+ local = _joined_alpha(rng, "._-")
197
+ labels = [_joined_alpha(rng, "-", maximum_parts=3) for _ in range(rng.randint(2, 5))]
198
+ return f"{local}@{'.'.join(labels)}"
199
+
200
+
201
+ def _random_url(rng: random.Random) -> str:
202
+ scheme = rng.choice(("http", "https"))
203
+ labels = [_joined_alpha(rng, "-", maximum_parts=3) for _ in range(rng.randint(2, 5))]
204
+ path = ""
205
+ if rng.choice((False, True)):
206
+ segments = [
207
+ _joined_alpha(rng, "-_.", maximum_parts=4)
208
+ for _ in range(rng.randint(1, 4))
209
+ ]
210
+ path = "/" + "/".join(segments)
211
+ return f"{scheme}://{'.'.join(labels)}{path}"
212
+
213
+
214
+ def test_seeded_property_roundtrip_for_random_valid_identifiers():
215
+ rng = random.Random(0xB10E_5A11)
216
+ for _ in range(1_000):
217
+ raw = _random_email(rng) if rng.randrange(2) == 0 else _random_url(rng)
218
+ raw_offset = rng.randrange(0, 1_000)
219
+ canonical_offset = rng.randrange(0, 1_000)
220
+ canonical, proof = _render(
221
+ raw,
222
+ absolute_raw_start=raw_offset,
223
+ canonical_start=canonical_offset,
224
+ )
225
+
226
+ assert _inverse(proof) == raw
227
+ assert len(proof.atoms) == len(raw.encode("ascii"))
228
+ assert canonical.split(" ") == [
229
+ atom.canonical_token for atom in proof.atoms
230
+ ]
231
+ assert proof.absolute_raw_end - proof.absolute_raw_start == len(raw)
232
+
233
+
234
+ @pytest.mark.parametrize(
235
+ "raw",
236
+ (
237
+ "",
238
+ "Mead@forest.org",
239
+ "mead@Forest.org",
240
+ "MEAD@FOREST.ORG",
241
+ "HTTP://forest.org",
242
+ "https://forest.org/Path",
243
+ "méad@forest.org",
244
+ "梅@forest.org",
245
+ "mead@forest.org",
246
+ "xn--caf-dma@example.org",
247
+ "mead@xn--caf-dma.org",
248
+ "mead@\u202eforest.org",
249
+ "mead@forest.org\u2066",
250
+ "mead@forest.org\x00",
251
+ "mead@forest.org\n",
252
+ "mead@forest.org\t",
253
+ "mead1@forest.org",
254
+ "mead@forest2.org",
255
+ "https://forest2.org",
256
+ "https://forest.org/path2",
257
+ "https://forest.org/path?query",
258
+ "https://forest.org/path#fragment",
259
+ "https://forest.org/pa%74h",
260
+ "https://forest.org:443/path",
261
+ "https://user@forest.org/path",
262
+ "https://user:pass@forest.org/path",
263
+ "ftp://forest.org/path",
264
+ "forest.org/path",
265
+ "mailto:mead@forest.org",
266
+ ".mead@forest.org",
267
+ "mead.@forest.org",
268
+ "mead..voice@forest.org",
269
+ "mead__voice@forest.org",
270
+ "mead--voice@forest.org",
271
+ "mead@.forest.org",
272
+ "mead@forest.org.",
273
+ "mead@forest..org",
274
+ "mead@-forest.org",
275
+ "mead@forest-.org",
276
+ "https:///forest.org",
277
+ "https://forest.org/",
278
+ "https://forest.org//path",
279
+ "https://forest.org/.",
280
+ "https://forest.org/..",
281
+ "https://forest.org/.path",
282
+ "https://forest.org/path.",
283
+ "https://forest.org/path..next",
284
+ "https://forest.org/-path",
285
+ "https://forest.org/path-",
286
+ "https://forest.org/_path",
287
+ "https://forest.org/path_",
288
+ "a",
289
+ "a.b",
290
+ "@",
291
+ "://",
292
+ ),
293
+ )
294
+ def test_unsafe_or_out_of_grammar_inputs_are_rejected(raw):
295
+ with pytest.raises(GlyphAsciiV1Error):
296
+ _render(raw)
297
+
298
+
299
+ def test_cap_overflow_is_rejected_without_truncation():
300
+ # 509 letters + "@a.b" is one byte beyond the frozen 512-byte cap.
301
+ raw = "a" * (GLYPH_ASCII_V1_MAX_IDENTIFIER_BYTES - 3) + "@a.b"
302
+ assert len(raw.encode("ascii")) == GLYPH_ASCII_V1_MAX_IDENTIFIER_BYTES + 1
303
+ with pytest.raises(GlyphAsciiV1Error, match="cap"):
304
+ _render(raw)
305
+
306
+
307
+ @pytest.mark.parametrize("value", (b"a@b.c", None, 7, True))
308
+ def test_non_string_raw_input_is_rejected(value):
309
+ with pytest.raises(GlyphAsciiV1Error):
310
+ _render(value) # type: ignore[arg-type]
311
+
312
+
313
+ @pytest.mark.parametrize(
314
+ ("keyword", "value"),
315
+ (
316
+ ("absolute_raw_start", -1),
317
+ ("absolute_raw_start", True),
318
+ ("absolute_raw_start", 1.5),
319
+ ("canonical_start", -1),
320
+ ("canonical_start", False),
321
+ ("canonical_start", "1"),
322
+ ),
323
+ )
324
+ def test_invalid_external_offsets_are_rejected(keyword, value):
325
+ kwargs = {keyword: value}
326
+ with pytest.raises(GlyphAsciiV1Error):
327
+ render_glyph_ascii_v1(
328
+ "a@b.c",
329
+ asset_entry_sha256_by_asset_id=ASSET_MANIFEST,
330
+ **kwargs,
331
+ )
332
+
333
+
334
+ def test_manifest_must_bind_exactly_32_valid_entries():
335
+ missing = dict(ASSET_MANIFEST)
336
+ missing.pop(next(iter(missing)))
337
+ extra = dict(ASSET_MANIFEST)
338
+ extra["glyph_ascii_v1_extra"] = "0" * 64
339
+ invalid_digest = dict(ASSET_MANIFEST)
340
+ invalid_digest[next(iter(invalid_digest))] = "A" * 64
341
+
342
+ for manifest in (missing, extra, invalid_digest):
343
+ with pytest.raises(GlyphAsciiV1Error):
344
+ render_glyph_ascii_v1(
345
+ "a@b.c",
346
+ asset_entry_sha256_by_asset_id=manifest,
347
+ )
348
+ with pytest.raises(GlyphAsciiV1Error):
349
+ render_glyph_ascii_v1(
350
+ "a@b.c",
351
+ asset_entry_sha256_by_asset_id=[], # type: ignore[arg-type]
352
+ )
353
+
354
+
355
+ def _proof_with_atoms(
356
+ proof: GlyphAsciiV1Proof,
357
+ atoms: tuple[GlyphAsciiV1AtomProof, ...] | list[GlyphAsciiV1AtomProof],
358
+ ) -> GlyphAsciiV1Proof:
359
+ return replace(proof, atoms=atoms) # type: ignore[arg-type]
360
+
361
+
362
+ def test_atom_reorder_omit_duplicate_substitute_and_borrow_are_rejected():
363
+ _, proof = _render("mead@forest.org", absolute_raw_start=7, canonical_start=11)
364
+ _, donor = _render("owl@island.net", absolute_raw_start=7, canonical_start=11)
365
+ atoms = proof.atoms
366
+
367
+ tampered = (
368
+ _proof_with_atoms(proof, (atoms[1], atoms[0], *atoms[2:])),
369
+ _proof_with_atoms(proof, (*atoms[:3], *atoms[4:])),
370
+ _proof_with_atoms(proof, (*atoms[:3], atoms[3], *atoms[3:])),
371
+ _proof_with_atoms(
372
+ proof,
373
+ (
374
+ replace(
375
+ atoms[0],
376
+ raw_byte=ord("z"),
377
+ canonical_token="字母二十六",
378
+ asset_id="glyph_ascii_v1_letter_z",
379
+ manifest_entry_sha256=ASSET_MANIFEST[
380
+ "glyph_ascii_v1_letter_z"
381
+ ],
382
+ ),
383
+ *atoms[1:],
384
+ ),
385
+ ),
386
+ _proof_with_atoms(proof, (donor.atoms[0], *atoms[1:])),
387
+ )
388
+ for forged in tampered:
389
+ with pytest.raises(GlyphAsciiV1Error):
390
+ _inverse(forged)
391
+
392
+
393
+ @pytest.mark.parametrize(
394
+ "mutate",
395
+ (
396
+ lambda atom: replace(atom, ordinal=atom.ordinal + 1),
397
+ lambda atom: replace(atom, ordinal=True),
398
+ lambda atom: replace(atom, absolute_raw_start=atom.absolute_raw_start + 1),
399
+ lambda atom: replace(atom, absolute_raw_end=atom.absolute_raw_end + 1),
400
+ lambda atom: replace(
401
+ atom, identifier_raw_start=atom.identifier_raw_start + 1
402
+ ),
403
+ lambda atom: replace(atom, identifier_raw_end=atom.identifier_raw_end + 1),
404
+ lambda atom: replace(atom, canonical_start=atom.canonical_start + 1),
405
+ lambda atom: replace(atom, canonical_end=atom.canonical_end + 1),
406
+ lambda atom: replace(atom, raw_byte=True),
407
+ lambda atom: replace(atom, raw_byte=ord("z")),
408
+ lambda atom: replace(atom, canonical_token="字母二十六"),
409
+ lambda atom: replace(atom, canonical_token=[]),
410
+ lambda atom: replace(atom, asset_id="glyph_ascii_v1_letter_z"),
411
+ lambda atom: replace(atom, asset_id=[]),
412
+ lambda atom: replace(atom, manifest_entry_sha256="0" * 64),
413
+ lambda atom: replace(atom, manifest_entry_sha256="A" * 64),
414
+ ),
415
+ )
416
+ def test_atom_gap_overlap_type_hash_and_mapping_tampering_is_rejected(mutate):
417
+ _, proof = _render("mead@forest.org", absolute_raw_start=7, canonical_start=11)
418
+ index = 2
419
+ atoms = list(proof.atoms)
420
+ atoms[index] = mutate(atoms[index])
421
+ with pytest.raises(GlyphAsciiV1Error):
422
+ _inverse(_proof_with_atoms(proof, tuple(atoms)))
423
+
424
+
425
+ @pytest.mark.parametrize(
426
+ "mutate",
427
+ (
428
+ lambda proof: replace(proof, grammar_id="glyph_ascii_v2"),
429
+ lambda proof: replace(proof, identifier_kind="url"),
430
+ lambda proof: replace(proof, absolute_raw_start=True),
431
+ lambda proof: replace(proof, absolute_raw_start=proof.absolute_raw_start + 1),
432
+ lambda proof: replace(proof, absolute_raw_end=proof.absolute_raw_end + 1),
433
+ lambda proof: replace(proof, canonical_start=False),
434
+ lambda proof: replace(proof, canonical_start=proof.canonical_start + 1),
435
+ lambda proof: replace(proof, canonical_end=proof.canonical_end + 1),
436
+ lambda proof: replace(proof, raw_sha256="0" * 64),
437
+ lambda proof: replace(proof, canonical_sha256="0" * 64),
438
+ lambda proof: replace(proof, asset_manifest_sha256="0" * 64),
439
+ lambda proof: replace(proof, raw_sha256="A" * 64),
440
+ lambda proof: replace(proof, atoms=()),
441
+ lambda proof: replace(proof, atoms=list(proof.atoms)),
442
+ ),
443
+ )
444
+ def test_header_range_hash_type_and_expected_proof_tampering_is_rejected(mutate):
445
+ _, proof = _render("mead@forest.org", absolute_raw_start=7, canonical_start=11)
446
+ with pytest.raises(GlyphAsciiV1Error):
447
+ _inverse(mutate(proof))
448
+
449
+
450
+ def test_manifest_substitution_is_rejected_by_entry_and_manifest_hashes():
451
+ _, proof = _render("mead@forest.org")
452
+ changed = dict(ASSET_MANIFEST)
453
+ changed[proof.atoms[0].asset_id] = hashlib.sha256(b"different").hexdigest()
454
+
455
+ with pytest.raises(GlyphAsciiV1Error):
456
+ _inverse(proof, manifest=changed)
457
+
458
+
459
+ def test_inverse_rejects_non_proof_and_subclass_types():
460
+ class ForgedProof(GlyphAsciiV1Proof):
461
+ pass
462
+
463
+ _, proof = _render("mead@forest.org")
464
+ forged = ForgedProof(**asdict(proof))
465
+
466
+ with pytest.raises(GlyphAsciiV1Error):
467
+ _inverse(forged)
468
+ with pytest.raises(GlyphAsciiV1Error):
469
+ inverse_glyph_ascii_v1( # type: ignore[arg-type]
470
+ object(),
471
+ asset_entry_sha256_by_asset_id=ASSET_MANIFEST,
472
+ )
473
+
474
+
475
+ def test_legacy_normalize_and_generation_plan_byte_for_byte_regression():
476
+ """The isolated core must not alter the frozen legacy frontend or planner."""
477
+
478
+ samples = (
479
+ "今天測試一般句子,保持原有規劃。",
480
+ "聯絡 mead@forestmail.org。",
481
+ "請開啟 https://audiocoast.org/tour。",
482
+ )
483
+ output = []
484
+ for raw in samples:
485
+ normalized = normalize_spoken_forms(raw, locale="zh-TW")
486
+ output.append(
487
+ {
488
+ "raw": raw,
489
+ "normalized": normalized,
490
+ "plan": [
491
+ asdict(spec)
492
+ for spec in plan_generation_chunks(raw, normalized)
493
+ ],
494
+ }
495
+ )
496
+ serialized = json.dumps(
497
+ output,
498
+ ensure_ascii=False,
499
+ sort_keys=True,
500
+ separators=(",", ":"),
501
+ ).encode("utf-8")
502
+
503
+ assert len(serialized) == 3_941
504
+ assert (
505
+ hashlib.sha256(serialized).hexdigest()
506
+ == "cea66a03b4ac1381f1acbc240e73d08f0b992f2a5efd4af1bc855b55d7bb50ee"
507
+ )