File size: 13,770 Bytes
ac406ae
 
 
 
 
 
 
 
 
 
 
 
549c072
ac406ae
 
 
 
549c072
ac406ae
 
 
549c072
 
ac406ae
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
549c072
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
755b152
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
"""Guards on data/: the committed lexicon files are exactly what their build scripts wrote.

D-09 and D-13 require the JLPT lists and the kanji map to be pinned, committed, hashed and
licensed; D-07 does the same for the JMdict projection. scripts/build_jlpt.py and
scripts/build_jmdict.py produce them and write a README with SHA-256 rows; these tests read
the files back and compare (the tests/test_vendor.py pattern), so a hand-edit, a stale
README or a skipped `git lfs pull` fails the quick loop. Files only, never the network.
"""

from __future__ import annotations

import csv
import gzip
import hashlib
import io
import json
import re
import time
from collections import Counter
from pathlib import Path

import pytest

REPO_ROOT = Path(__file__).resolve().parent.parent
JLPT = REPO_ROOT / "data" / "jlpt"
JLPT_README = JLPT / "README.md"

LEVELS = ("n5", "n4", "n3", "n2", "n1")
CSV_FILES = tuple(f"{level}.csv" for level in LEVELS)
KANJI_LEVELS = "kanji_levels.json"
JLPT_LICENCES = {
    "LICENSE-yomitan-jlpt-vocab.txt": "Attribution-ShareAlike 4.0",
    "LICENSE-kanji-data.txt": "Permission is hereby granted",
}

# Measured on the pinned sources by scripts/build_jlpt.py (plan 02-01); the script refuses
# to write on any other numbers, so these pin the data AND the parse.
EXPECTED_ROWS = {"n5": 684, "n4": 640, "n3": 1730, "n2": 1812, "n1": 3427}
EXPECTED_EMPTY_IDS = 14
EXPECTED_UNIQUE_IDS = 7748
EXPECTED_MULTI_ROW_IDS = 505
EXPECTED_MULTI_LEVEL_IDS = 447
EXPECTED_KANJI = {"N5": 79, "N4": 166, "N3": 367, "N2": 367, "N1": 1232}

# CJK Unified Ideographs (+ Extension A) and the Compatibility Ideographs block: every key
# of kanji_levels.json must be one character from one of these.
CJK_RANGES = ((0x3400, 0x4DBF), (0x4E00, 0x9FFF), (0xF900, 0xFAFF))


def _readme_rows(path: Path) -> dict[str, tuple[int, str]]:
    """``name -> (bytes, sha256)`` from every ``| `file` | bytes | `sha` |`` row of a README."""
    text = path.read_text(encoding="utf-8")
    rows = re.findall(r"^\| `([^`]+)` \| ([\d,]+) \| `([0-9a-f]{64})` \|", text, re.M)
    assert rows, f"{path.relative_to(REPO_ROOT)} has no hash rows; run its build script"
    return {name: (int(size.replace(",", "")), digest) for name, size, digest in rows}


def _assert_files_match_readme(directory: Path, readme: Path, names: tuple[str, ...]) -> None:
    recorded = _readme_rows(readme)
    rel = readme.relative_to(REPO_ROOT).as_posix()
    for name in names:
        assert name in recorded, f"{name} has no row in {rel}"
        data = (directory / name).read_bytes()
        size, digest = recorded[name]
        assert len(data) == size, f"{name}: {rel} says {size:,} bytes, disk has {len(data):,}"
        assert hashlib.sha256(data).hexdigest() == digest, (
            f"{name} differs from the hash in {rel}; regenerate with its build script "
            "rather than editing by hand"
        )


def _jlpt_rows(level: str) -> list[list[str]]:
    """Data rows of one CSV, the way scripts/build_jlpt.py reads them (header skipped)."""
    text = (JLPT / f"{level}.csv").read_bytes().decode("utf-8-sig")
    rows = [row for row in csv.reader(io.StringIO(text, newline="")) if row]
    assert rows[0] == ["jmdict_seq", "kana", "kanji", "waller_definition"], rows[0]
    return rows[1:]


def jlpt_ids() -> set[str]:
    """Every non-empty ``jmdict_seq`` across the five lists (the ids that must join JMdict)."""
    return {row[0] for level in LEVELS for row in _jlpt_rows(level) if row[0]}


def test_jlpt_files_match_readme():
    """Bytes and SHA-256 of all eight files are the README's, so no file was hand-edited."""
    _assert_files_match_readme(JLPT, JLPT_README, (*CSV_FILES, KANJI_LEVELS, *JLPT_LICENCES))
    text = JLPT_README.read_text(encoding="utf-8")
    assert "2025.08.01.0" in text, "README does not name the yomitan-jlpt-vocab tag"
    assert re.search(r"\b[0-9a-f]{40}\b", text), "README does not name the kanji-data commit"
    assert "tanos.co.uk" in text, "README does not credit Jonathan Waller's lists"


def test_jlpt_csv_counts():
    """Per-level rows, the 14 id-less N1 rows, 7,748 unique ids, and the duplicate structure."""
    tables = {level: _jlpt_rows(level) for level in LEVELS}
    assert {level: len(rows) for level, rows in tables.items()} == EXPECTED_ROWS

    empty = [(level, row) for level, rows in tables.items() for row in rows if not row[0]]
    assert len(empty) == EXPECTED_EMPTY_IDS
    assert {level for level, _ in empty} == {"n1"}, "id-less rows outside n1.csv"

    for level, rows in tables.items():
        for row in rows:
            assert len(row) == 4, f"{level}.csv row {row} has {len(row)} columns"
            assert row[0] == "" or row[0].isdigit(), f"{level}.csv non-numeric id {row[0]!r}"

    level_count: Counter[str] = Counter()
    for rows in tables.values():
        level_count.update({row[0] for row in rows if row[0]})
    row_count = Counter(row[0] for rows in tables.values() for row in rows if row[0])
    assert len(level_count) == EXPECTED_UNIQUE_IDS
    assert sum(1 for n in row_count.values() if n > 1) == EXPECTED_MULTI_ROW_IDS
    assert sum(1 for n in level_count.values() if n > 1) == EXPECTED_MULTI_LEVEL_IDS


def test_kanji_levels_distribution():
    """2,211 single-kanji keys, the five counts research recorded, every value an N-level."""
    levels = json.loads((JLPT / KANJI_LEVELS).read_text(encoding="utf-8"))
    assert isinstance(levels, dict) and len(levels) == sum(EXPECTED_KANJI.values())
    assert dict(Counter(levels.values())) == EXPECTED_KANJI
    assert set(levels.values()) <= set(EXPECTED_KANJI)
    for literal in levels:
        assert len(literal) == 1, f"key {literal!r} is not one character"
        cp = ord(literal)
        assert any(lo <= cp <= hi for lo, hi in CJK_RANGES), f"{literal!r} U+{cp:04X} not CJK"
    assert list(levels) == sorted(levels), "kanji_levels.json is not sorted by literal"


def test_jlpt_licences_present():
    for name, phrase in JLPT_LICENCES.items():
        text = (JLPT / name).read_text(encoding="utf-8")
        assert phrase in text, f"{name} does not contain {phrase!r}"
    readme = JLPT_README.read_text(encoding="utf-8")
    assert "CC BY-SA 4.0" in readme and "MIT" in readme
    assert "no official JLPT vocabulary list" in readme


# --- data/jmdict ----------------------------------------------------------------------------

JMDICT = REPO_ROOT / "data" / "jmdict"
JMDICT_FILE = JMDICT / "jmdict-compact.json.gz"
JMDICT_README = JMDICT / "README.md"
JMDICT_RELEASE = "3.6.2+20260831182826"
JMDICT_DICT_DATE = "2026-08-31"
JMDICT_ENTRIES = 218672
JMDICT_COMMON_ENTRIES = 22637
# Research measured 7.66 MB; anything outside this band is a truncated write or a different
# projection. An LFS pointer file is ~130 bytes.
JMDICT_MIN_BYTES, JMDICT_MAX_BYTES = 6_000_000, 10_000_000
GZIP_MAGIC = b"\x1f\x8b"
LFS_POINTER_PREFIX = b"version https://git-lfs"


@pytest.fixture(scope="module")
def compact_jmdict() -> dict:
    """The compact file gunzipped and parsed once per module, the way the app loads it."""
    t = time.perf_counter()
    with gzip.open(JMDICT_FILE, "rt", encoding="utf-8") as fh:
        data = json.load(fh)
    print(f"\ncompact JMdict loaded in {time.perf_counter() - t:.2f} s")
    return data


def test_jmdict_file_is_real_lfs_object():
    """The gzip itself, not an LFS pointer (a pointer means `git lfs pull` was skipped)."""
    head = JMDICT_FILE.read_bytes()[:64]
    assert not head.startswith(LFS_POINTER_PREFIX), (
        f"{JMDICT_FILE.relative_to(REPO_ROOT)} is a Git LFS pointer; run `git lfs pull`"
    )
    assert head[:2] == GZIP_MAGIC, f"not a gzip file: first bytes {head[:4]!r}"
    size = JMDICT_FILE.stat().st_size
    assert JMDICT_MIN_BYTES <= size <= JMDICT_MAX_BYTES, f"{size:,} bytes is not the projection"


def test_jmdict_matches_readme():
    _assert_files_match_readme(JMDICT, JMDICT_README, (JMDICT_FILE.name,))
    text = JMDICT_README.read_text(encoding="utf-8")
    assert JMDICT_RELEASE in text and JMDICT_DICT_DATE in text
    assert text.count("James William Breen") == 1
    assert text.count("edrdg.org/edrdg/licence.html") == 1
    assert "CC BY-SA 4.0" in text


def _entry_index(data: dict) -> dict[int, list]:
    return {entry[0]: entry for entry in data["entries"]}


def test_jmdict_meta_and_count(compact_jmdict):
    """Pinned meta, 218,672 five-field records with positive unique ids, and spot entries."""
    meta = compact_jmdict["meta"]
    assert meta["source"] == "scriptin/jmdict-simplified"
    assert meta["version"] == JMDICT_RELEASE
    assert meta["dictDate"] == JMDICT_DICT_DATE
    assert meta["entries"] == JMDICT_ENTRIES
    entries = compact_jmdict["entries"]
    assert len(entries) == JMDICT_ENTRIES

    ids = [entry[0] for entry in entries]
    assert all(isinstance(i, int) and i > 0 for i in ids), "ids must be positive ints"
    assert len(set(ids)) == len(ids), "duplicate entry ids"
    for entry in entries:
        assert len(entry) == 5, f"record {entry[0]} has {len(entry)} fields, not 5"
        _id, kanji, kana, senses, common = entry
        assert isinstance(kanji, list) and isinstance(kana, list) and kana, entry[0]
        assert isinstance(senses, list) and all(isinstance(s, list) and s for s in senses)
        assert common in (0, 1)

    by_id = _entry_index(compact_jmdict)
    iru = by_id[1577980]
    assert "いる" in iru[2] and "居る" in iru[1]
    konnichiwa = by_id[1289400]
    assert "こんにちは" in konnichiwa[2]  # kanji is NOT empty: 今日は (rK) and 今日わ (sK)
    assert "きょう" in by_id[1579110][2]
    hanaseru = by_id[1562360]
    assert any("to be able to speak" in gloss for sense in hanaseru[3] for gloss in sense)


def test_jmdict_common_flag_count(compact_jmdict):
    """22,637 entries carry common=1: exactly the word count of jmdict-eng-common at this
    release (research § Q3), an independent check that the flag was projected correctly.
    The ranked lookup (plan 02-03) ranks on it."""
    assert sum(entry[4] for entry in compact_jmdict["entries"]) == JMDICT_COMMON_ENTRIES


def test_jlpt_ids_resolve_in_jmdict(compact_jmdict):
    """Every non-empty jmdict_seq on the five lists is an entry (7,748 of 7,748).

    538 of them are NOT in jmdict-eng-common (research § Q3), which is why the full
    projection ships rather than the common subset.
    """
    ids = jlpt_ids()
    assert len(ids) == EXPECTED_UNIQUE_IDS
    entry_ids = {entry[0] for entry in compact_jmdict["entries"]}
    missing = sorted(int(i) for i in ids if int(i) not in entry_ids)
    assert missing == [], f"{len(missing)} JLPT ids have no JMdict entry: {missing[:20]}"


# --- data/mt ----------------------------------------------------------------------------------

MT = REPO_ROOT / "data" / "mt"
MT_MODEL = MT / "opus-mt-ja-en-ct2-int8"
MT_README = MT / "README.md"
MT_FILES = ("model.bin", "shared_vocabulary.json", "config.json", "source.spm", "target.spm")
MT_LFS_FILES = ("model.bin", "source.spm", "target.spm")
MT_MODEL_ID = "Helsinki-NLP/opus-mt-ja-en"
# Research measured model.bin at 77,339,435 B and the two spm files at ~0.8 MB; an LFS pointer
# is ~130 B. These are floors, not equalities - the README rows pin the exact bytes.
MT_MIN_MODEL_BYTES = 70_000_000
MT_MIN_SPM_BYTES = 500_000
# What ctranslate2 4.8.2's TransformersConverter writes for a Marian model (pinned after the
# first run of scripts/convert_mt.py); the translator reads these to frame the decoder.
MT_CONFIG_KEYS = {"bos_token", "eos_token", "unk_token", "decoder_start_token"}


def test_mt_files_match_readme():
    """Bytes and SHA-256 of the five model files are the README's; the README pins the revision."""
    _assert_files_match_readme(MT_MODEL, MT_README, MT_FILES)
    text = MT_README.read_text(encoding="utf-8")
    assert MT_MODEL_ID in text
    assert re.search(r"\b[0-9a-f]{40}\b", text), "README does not record the Hub revision SHA"
    assert "opus-2019-12-18" in text, "README does not name the OPUS-MT release tag"


def test_mt_model_is_real_not_pointer():
    """The binaries themselves, not Git LFS pointers (a pointer means `git lfs pull` was skipped).

    A pointer here would fail on the Space at the first EN tap; this fails in the quick loop.
    """
    for name in MT_LFS_FILES:
        path = MT_MODEL / name
        head = path.read_bytes()[:64]
        assert not head.startswith(LFS_POINTER_PREFIX), (
            f"{path.relative_to(REPO_ROOT)} is a Git LFS pointer; run `git lfs pull`"
        )
    assert (MT_MODEL / "model.bin").stat().st_size > MT_MIN_MODEL_BYTES
    for name in ("source.spm", "target.spm"):
        assert (MT_MODEL / name).stat().st_size > MT_MIN_SPM_BYTES, name


def test_mt_licence_and_notice_present():
    licence = (MT / "LICENSE-apache-2.0.txt").read_text(encoding="utf-8")
    assert "Apache License" in licence and "Version 2.0" in licence
    notice = (MT / "NOTICE").read_text(encoding="utf-8")
    assert MT_MODEL_ID in notice
    assert "scripts/convert_mt.py" in notice
    readme = MT_README.read_text(encoding="utf-8")
    assert "Apache-2.0" in readme and "University of Helsinki" in readme


def test_mt_config_is_ct2():
    """config.json is the CTranslate2 converter's, with the special tokens a Marian model needs."""
    config = json.loads((MT_MODEL / "config.json").read_text(encoding="utf-8"))
    print(f"\nct2 config keys: {sorted(config)}")
    assert set(config) >= MT_CONFIG_KEYS, sorted(config)
    assert config["eos_token"] == "</s>" and config["unk_token"] == "<unk>"
    vocabulary = json.loads((MT_MODEL / "shared_vocabulary.json").read_text(encoding="utf-8"))
    assert isinstance(vocabulary, list) and len(vocabulary) > 50_000
    assert "</s>" in vocabulary and "<unk>" in vocabulary