"""Guards on data/: the committed lexicon files are exactly what their build scripts wrote. D-09 and D-13 require the JLPT lists and the kanji map to be pinned, committed, hashed and licensed; D-07 does the same for the JMdict projection. scripts/build_jlpt.py and scripts/build_jmdict.py produce them and write a README with SHA-256 rows; these tests read the files back and compare (the tests/test_vendor.py pattern), so a hand-edit, a stale README or a skipped `git lfs pull` fails the quick loop. Files only, never the network. """ from __future__ import annotations import csv import gzip import hashlib import io import json import re import time from collections import Counter from pathlib import Path import pytest REPO_ROOT = Path(__file__).resolve().parent.parent JLPT = REPO_ROOT / "data" / "jlpt" JLPT_README = JLPT / "README.md" LEVELS = ("n5", "n4", "n3", "n2", "n1") CSV_FILES = tuple(f"{level}.csv" for level in LEVELS) KANJI_LEVELS = "kanji_levels.json" JLPT_LICENCES = { "LICENSE-yomitan-jlpt-vocab.txt": "Attribution-ShareAlike 4.0", "LICENSE-kanji-data.txt": "Permission is hereby granted", } # Measured on the pinned sources by scripts/build_jlpt.py (plan 02-01); the script refuses # to write on any other numbers, so these pin the data AND the parse. EXPECTED_ROWS = {"n5": 684, "n4": 640, "n3": 1730, "n2": 1812, "n1": 3427} EXPECTED_EMPTY_IDS = 14 EXPECTED_UNIQUE_IDS = 7748 EXPECTED_MULTI_ROW_IDS = 505 EXPECTED_MULTI_LEVEL_IDS = 447 EXPECTED_KANJI = {"N5": 79, "N4": 166, "N3": 367, "N2": 367, "N1": 1232} # CJK Unified Ideographs (+ Extension A) and the Compatibility Ideographs block: every key # of kanji_levels.json must be one character from one of these. CJK_RANGES = ((0x3400, 0x4DBF), (0x4E00, 0x9FFF), (0xF900, 0xFAFF)) def _readme_rows(path: Path) -> dict[str, tuple[int, str]]: """``name -> (bytes, sha256)`` from every ``| `file` | bytes | `sha` |`` row of a README.""" text = path.read_text(encoding="utf-8") rows = re.findall(r"^\| `([^`]+)` \| ([\d,]+) \| `([0-9a-f]{64})` \|", text, re.M) assert rows, f"{path.relative_to(REPO_ROOT)} has no hash rows; run its build script" return {name: (int(size.replace(",", "")), digest) for name, size, digest in rows} def _assert_files_match_readme(directory: Path, readme: Path, names: tuple[str, ...]) -> None: recorded = _readme_rows(readme) rel = readme.relative_to(REPO_ROOT).as_posix() for name in names: assert name in recorded, f"{name} has no row in {rel}" data = (directory / name).read_bytes() size, digest = recorded[name] assert len(data) == size, f"{name}: {rel} says {size:,} bytes, disk has {len(data):,}" assert hashlib.sha256(data).hexdigest() == digest, ( f"{name} differs from the hash in {rel}; regenerate with its build script " "rather than editing by hand" ) def _jlpt_rows(level: str) -> list[list[str]]: """Data rows of one CSV, the way scripts/build_jlpt.py reads them (header skipped).""" text = (JLPT / f"{level}.csv").read_bytes().decode("utf-8-sig") rows = [row for row in csv.reader(io.StringIO(text, newline="")) if row] assert rows[0] == ["jmdict_seq", "kana", "kanji", "waller_definition"], rows[0] return rows[1:] def jlpt_ids() -> set[str]: """Every non-empty ``jmdict_seq`` across the five lists (the ids that must join JMdict).""" return {row[0] for level in LEVELS for row in _jlpt_rows(level) if row[0]} def test_jlpt_files_match_readme(): """Bytes and SHA-256 of all eight files are the README's, so no file was hand-edited.""" _assert_files_match_readme(JLPT, JLPT_README, (*CSV_FILES, KANJI_LEVELS, *JLPT_LICENCES)) text = JLPT_README.read_text(encoding="utf-8") assert "2025.08.01.0" in text, "README does not name the yomitan-jlpt-vocab tag" assert re.search(r"\b[0-9a-f]{40}\b", text), "README does not name the kanji-data commit" assert "tanos.co.uk" in text, "README does not credit Jonathan Waller's lists" def test_jlpt_csv_counts(): """Per-level rows, the 14 id-less N1 rows, 7,748 unique ids, and the duplicate structure.""" tables = {level: _jlpt_rows(level) for level in LEVELS} assert {level: len(rows) for level, rows in tables.items()} == EXPECTED_ROWS empty = [(level, row) for level, rows in tables.items() for row in rows if not row[0]] assert len(empty) == EXPECTED_EMPTY_IDS assert {level for level, _ in empty} == {"n1"}, "id-less rows outside n1.csv" for level, rows in tables.items(): for row in rows: assert len(row) == 4, f"{level}.csv row {row} has {len(row)} columns" assert row[0] == "" or row[0].isdigit(), f"{level}.csv non-numeric id {row[0]!r}" level_count: Counter[str] = Counter() for rows in tables.values(): level_count.update({row[0] for row in rows if row[0]}) row_count = Counter(row[0] for rows in tables.values() for row in rows if row[0]) assert len(level_count) == EXPECTED_UNIQUE_IDS assert sum(1 for n in row_count.values() if n > 1) == EXPECTED_MULTI_ROW_IDS assert sum(1 for n in level_count.values() if n > 1) == EXPECTED_MULTI_LEVEL_IDS def test_kanji_levels_distribution(): """2,211 single-kanji keys, the five counts research recorded, every value an N-level.""" levels = json.loads((JLPT / KANJI_LEVELS).read_text(encoding="utf-8")) assert isinstance(levels, dict) and len(levels) == sum(EXPECTED_KANJI.values()) assert dict(Counter(levels.values())) == EXPECTED_KANJI assert set(levels.values()) <= set(EXPECTED_KANJI) for literal in levels: assert len(literal) == 1, f"key {literal!r} is not one character" cp = ord(literal) assert any(lo <= cp <= hi for lo, hi in CJK_RANGES), f"{literal!r} U+{cp:04X} not CJK" assert list(levels) == sorted(levels), "kanji_levels.json is not sorted by literal" def test_jlpt_licences_present(): for name, phrase in JLPT_LICENCES.items(): text = (JLPT / name).read_text(encoding="utf-8") assert phrase in text, f"{name} does not contain {phrase!r}" readme = JLPT_README.read_text(encoding="utf-8") assert "CC BY-SA 4.0" in readme and "MIT" in readme assert "no official JLPT vocabulary list" in readme # --- data/jmdict ---------------------------------------------------------------------------- JMDICT = REPO_ROOT / "data" / "jmdict" JMDICT_FILE = JMDICT / "jmdict-compact.json.gz" JMDICT_README = JMDICT / "README.md" JMDICT_RELEASE = "3.6.2+20260831182826" JMDICT_DICT_DATE = "2026-08-31" JMDICT_ENTRIES = 218672 JMDICT_COMMON_ENTRIES = 22637 # Research measured 7.66 MB; anything outside this band is a truncated write or a different # projection. An LFS pointer file is ~130 bytes. JMDICT_MIN_BYTES, JMDICT_MAX_BYTES = 6_000_000, 10_000_000 GZIP_MAGIC = b"\x1f\x8b" LFS_POINTER_PREFIX = b"version https://git-lfs" @pytest.fixture(scope="module") def compact_jmdict() -> dict: """The compact file gunzipped and parsed once per module, the way the app loads it.""" t = time.perf_counter() with gzip.open(JMDICT_FILE, "rt", encoding="utf-8") as fh: data = json.load(fh) print(f"\ncompact JMdict loaded in {time.perf_counter() - t:.2f} s") return data def test_jmdict_file_is_real_lfs_object(): """The gzip itself, not an LFS pointer (a pointer means `git lfs pull` was skipped).""" head = JMDICT_FILE.read_bytes()[:64] assert not head.startswith(LFS_POINTER_PREFIX), ( f"{JMDICT_FILE.relative_to(REPO_ROOT)} is a Git LFS pointer; run `git lfs pull`" ) assert head[:2] == GZIP_MAGIC, f"not a gzip file: first bytes {head[:4]!r}" size = JMDICT_FILE.stat().st_size assert JMDICT_MIN_BYTES <= size <= JMDICT_MAX_BYTES, f"{size:,} bytes is not the projection" def test_jmdict_matches_readme(): _assert_files_match_readme(JMDICT, JMDICT_README, (JMDICT_FILE.name,)) text = JMDICT_README.read_text(encoding="utf-8") assert JMDICT_RELEASE in text and JMDICT_DICT_DATE in text assert text.count("James William Breen") == 1 assert text.count("edrdg.org/edrdg/licence.html") == 1 assert "CC BY-SA 4.0" in text def _entry_index(data: dict) -> dict[int, list]: return {entry[0]: entry for entry in data["entries"]} def test_jmdict_meta_and_count(compact_jmdict): """Pinned meta, 218,672 five-field records with positive unique ids, and spot entries.""" meta = compact_jmdict["meta"] assert meta["source"] == "scriptin/jmdict-simplified" assert meta["version"] == JMDICT_RELEASE assert meta["dictDate"] == JMDICT_DICT_DATE assert meta["entries"] == JMDICT_ENTRIES entries = compact_jmdict["entries"] assert len(entries) == JMDICT_ENTRIES ids = [entry[0] for entry in entries] assert all(isinstance(i, int) and i > 0 for i in ids), "ids must be positive ints" assert len(set(ids)) == len(ids), "duplicate entry ids" for entry in entries: assert len(entry) == 5, f"record {entry[0]} has {len(entry)} fields, not 5" _id, kanji, kana, senses, common = entry assert isinstance(kanji, list) and isinstance(kana, list) and kana, entry[0] assert isinstance(senses, list) and all(isinstance(s, list) and s for s in senses) assert common in (0, 1) by_id = _entry_index(compact_jmdict) iru = by_id[1577980] assert "いる" in iru[2] and "居る" in iru[1] konnichiwa = by_id[1289400] assert "こんにちは" in konnichiwa[2] # kanji is NOT empty: 今日は (rK) and 今日わ (sK) assert "きょう" in by_id[1579110][2] hanaseru = by_id[1562360] assert any("to be able to speak" in gloss for sense in hanaseru[3] for gloss in sense) def test_jmdict_common_flag_count(compact_jmdict): """22,637 entries carry common=1: exactly the word count of jmdict-eng-common at this release (research § Q3), an independent check that the flag was projected correctly. The ranked lookup (plan 02-03) ranks on it.""" assert sum(entry[4] for entry in compact_jmdict["entries"]) == JMDICT_COMMON_ENTRIES def test_jlpt_ids_resolve_in_jmdict(compact_jmdict): """Every non-empty jmdict_seq on the five lists is an entry (7,748 of 7,748). 538 of them are NOT in jmdict-eng-common (research § Q3), which is why the full projection ships rather than the common subset. """ ids = jlpt_ids() assert len(ids) == EXPECTED_UNIQUE_IDS entry_ids = {entry[0] for entry in compact_jmdict["entries"]} missing = sorted(int(i) for i in ids if int(i) not in entry_ids) assert missing == [], f"{len(missing)} JLPT ids have no JMdict entry: {missing[:20]}"