japanese-learning-avatar / tests /test_data_assets.py
WolfDavid's picture
feat(02-01): compact JMdict projection (LFS) with build script and asset guards
549c072
Raw History Blame
10.7 kB
"""Guards on data/: the committed lexicon files are exactly what their build scripts wrote.
D-09 and D-13 require the JLPT lists and the kanji map to be pinned, committed, hashed and
licensed; D-07 does the same for the JMdict projection. scripts/build_jlpt.py and
scripts/build_jmdict.py produce them and write a README with SHA-256 rows; these tests read
the files back and compare (the tests/test_vendor.py pattern), so a hand-edit, a stale
README or a skipped `git lfs pull` fails the quick loop. Files only, never the network.
"""
from __future__ import annotations
import csv
import gzip
import hashlib
import io
import json
import re
import time
from collections import Counter
from pathlib import Path
import pytest
REPO_ROOT = Path(__file__).resolve().parent.parent
JLPT = REPO_ROOT / "data" / "jlpt"
JLPT_README = JLPT / "README.md"
LEVELS = ("n5", "n4", "n3", "n2", "n1")
CSV_FILES = tuple(f"{level}.csv" for level in LEVELS)
KANJI_LEVELS = "kanji_levels.json"
JLPT_LICENCES = {
"LICENSE-yomitan-jlpt-vocab.txt": "Attribution-ShareAlike 4.0",
"LICENSE-kanji-data.txt": "Permission is hereby granted",
}
# Measured on the pinned sources by scripts/build_jlpt.py (plan 02-01); the script refuses
# to write on any other numbers, so these pin the data AND the parse.
EXPECTED_ROWS = {"n5": 684, "n4": 640, "n3": 1730, "n2": 1812, "n1": 3427}
EXPECTED_EMPTY_IDS = 14
EXPECTED_UNIQUE_IDS = 7748
EXPECTED_MULTI_ROW_IDS = 505
EXPECTED_MULTI_LEVEL_IDS = 447
EXPECTED_KANJI = {"N5": 79, "N4": 166, "N3": 367, "N2": 367, "N1": 1232}
# CJK Unified Ideographs (+ Extension A) and the Compatibility Ideographs block: every key
# of kanji_levels.json must be one character from one of these.
CJK_RANGES = ((0x3400, 0x4DBF), (0x4E00, 0x9FFF), (0xF900, 0xFAFF))
def _readme_rows(path: Path) -> dict[str, tuple[int, str]]:
"""``name -> (bytes, sha256)`` from every ``| `file` | bytes | `sha` |`` row of a README."""
text = path.read_text(encoding="utf-8")
rows = re.findall(r"^\| `([^`]+)` \| ([\d,]+) \| `([0-9a-f]{64})` \|", text, re.M)
assert rows, f"{path.relative_to(REPO_ROOT)} has no hash rows; run its build script"
return {name: (int(size.replace(",", "")), digest) for name, size, digest in rows}
def _assert_files_match_readme(directory: Path, readme: Path, names: tuple[str, ...]) -> None:
recorded = _readme_rows(readme)
rel = readme.relative_to(REPO_ROOT).as_posix()
for name in names:
assert name in recorded, f"{name} has no row in {rel}"
data = (directory / name).read_bytes()
size, digest = recorded[name]
assert len(data) == size, f"{name}: {rel} says {size:,} bytes, disk has {len(data):,}"
assert hashlib.sha256(data).hexdigest() == digest, (
f"{name} differs from the hash in {rel}; regenerate with its build script "
"rather than editing by hand"
)
def _jlpt_rows(level: str) -> list[list[str]]:
"""Data rows of one CSV, the way scripts/build_jlpt.py reads them (header skipped)."""
text = (JLPT / f"{level}.csv").read_bytes().decode("utf-8-sig")
rows = [row for row in csv.reader(io.StringIO(text, newline="")) if row]
assert rows[0] == ["jmdict_seq", "kana", "kanji", "waller_definition"], rows[0]
return rows[1:]
def jlpt_ids() -> set[str]:
"""Every non-empty ``jmdict_seq`` across the five lists (the ids that must join JMdict)."""
return {row[0] for level in LEVELS for row in _jlpt_rows(level) if row[0]}
def test_jlpt_files_match_readme():
"""Bytes and SHA-256 of all eight files are the README's, so no file was hand-edited."""
_assert_files_match_readme(JLPT, JLPT_README, (*CSV_FILES, KANJI_LEVELS, *JLPT_LICENCES))
text = JLPT_README.read_text(encoding="utf-8")
assert "2025.08.01.0" in text, "README does not name the yomitan-jlpt-vocab tag"
assert re.search(r"\b[0-9a-f]{40}\b", text), "README does not name the kanji-data commit"
assert "tanos.co.uk" in text, "README does not credit Jonathan Waller's lists"
def test_jlpt_csv_counts():
"""Per-level rows, the 14 id-less N1 rows, 7,748 unique ids, and the duplicate structure."""
tables = {level: _jlpt_rows(level) for level in LEVELS}
assert {level: len(rows) for level, rows in tables.items()} == EXPECTED_ROWS
empty = [(level, row) for level, rows in tables.items() for row in rows if not row[0]]
assert len(empty) == EXPECTED_EMPTY_IDS
assert {level for level, _ in empty} == {"n1"}, "id-less rows outside n1.csv"
for level, rows in tables.items():
for row in rows:
assert len(row) == 4, f"{level}.csv row {row} has {len(row)} columns"
assert row[0] == "" or row[0].isdigit(), f"{level}.csv non-numeric id {row[0]!r}"
level_count: Counter[str] = Counter()
for rows in tables.values():
level_count.update({row[0] for row in rows if row[0]})
row_count = Counter(row[0] for rows in tables.values() for row in rows if row[0])
assert len(level_count) == EXPECTED_UNIQUE_IDS
assert sum(1 for n in row_count.values() if n > 1) == EXPECTED_MULTI_ROW_IDS
assert sum(1 for n in level_count.values() if n > 1) == EXPECTED_MULTI_LEVEL_IDS
def test_kanji_levels_distribution():
"""2,211 single-kanji keys, the five counts research recorded, every value an N-level."""
levels = json.loads((JLPT / KANJI_LEVELS).read_text(encoding="utf-8"))
assert isinstance(levels, dict) and len(levels) == sum(EXPECTED_KANJI.values())
assert dict(Counter(levels.values())) == EXPECTED_KANJI
assert set(levels.values()) <= set(EXPECTED_KANJI)
for literal in levels:
assert len(literal) == 1, f"key {literal!r} is not one character"
cp = ord(literal)
assert any(lo <= cp <= hi for lo, hi in CJK_RANGES), f"{literal!r} U+{cp:04X} not CJK"
assert list(levels) == sorted(levels), "kanji_levels.json is not sorted by literal"
def test_jlpt_licences_present():
for name, phrase in JLPT_LICENCES.items():
text = (JLPT / name).read_text(encoding="utf-8")
assert phrase in text, f"{name} does not contain {phrase!r}"
readme = JLPT_README.read_text(encoding="utf-8")
assert "CC BY-SA 4.0" in readme and "MIT" in readme
assert "no official JLPT vocabulary list" in readme
# --- data/jmdict ----------------------------------------------------------------------------
JMDICT = REPO_ROOT / "data" / "jmdict"
JMDICT_FILE = JMDICT / "jmdict-compact.json.gz"
JMDICT_README = JMDICT / "README.md"
JMDICT_RELEASE = "3.6.2+20260831182826"
JMDICT_DICT_DATE = "2026-08-31"
JMDICT_ENTRIES = 218672
JMDICT_COMMON_ENTRIES = 22637
# Research measured 7.66 MB; anything outside this band is a truncated write or a different
# projection. An LFS pointer file is ~130 bytes.
JMDICT_MIN_BYTES, JMDICT_MAX_BYTES = 6_000_000, 10_000_000
GZIP_MAGIC = b"\x1f\x8b"
LFS_POINTER_PREFIX = b"version https://git-lfs"
@pytest.fixture(scope="module")
def compact_jmdict() -> dict:
"""The compact file gunzipped and parsed once per module, the way the app loads it."""
t = time.perf_counter()
with gzip.open(JMDICT_FILE, "rt", encoding="utf-8") as fh:
data = json.load(fh)
print(f"\ncompact JMdict loaded in {time.perf_counter() - t:.2f} s")
return data
def test_jmdict_file_is_real_lfs_object():
"""The gzip itself, not an LFS pointer (a pointer means `git lfs pull` was skipped)."""
head = JMDICT_FILE.read_bytes()[:64]
assert not head.startswith(LFS_POINTER_PREFIX), (
f"{JMDICT_FILE.relative_to(REPO_ROOT)} is a Git LFS pointer; run `git lfs pull`"
)
assert head[:2] == GZIP_MAGIC, f"not a gzip file: first bytes {head[:4]!r}"
size = JMDICT_FILE.stat().st_size
assert JMDICT_MIN_BYTES <= size <= JMDICT_MAX_BYTES, f"{size:,} bytes is not the projection"
def test_jmdict_matches_readme():
_assert_files_match_readme(JMDICT, JMDICT_README, (JMDICT_FILE.name,))
text = JMDICT_README.read_text(encoding="utf-8")
assert JMDICT_RELEASE in text and JMDICT_DICT_DATE in text
assert text.count("James William Breen") == 1
assert text.count("edrdg.org/edrdg/licence.html") == 1
assert "CC BY-SA 4.0" in text
def _entry_index(data: dict) -> dict[int, list]:
return {entry[0]: entry for entry in data["entries"]}
def test_jmdict_meta_and_count(compact_jmdict):
"""Pinned meta, 218,672 five-field records with positive unique ids, and spot entries."""
meta = compact_jmdict["meta"]
assert meta["source"] == "scriptin/jmdict-simplified"
assert meta["version"] == JMDICT_RELEASE
assert meta["dictDate"] == JMDICT_DICT_DATE
assert meta["entries"] == JMDICT_ENTRIES
entries = compact_jmdict["entries"]
assert len(entries) == JMDICT_ENTRIES
ids = [entry[0] for entry in entries]
assert all(isinstance(i, int) and i > 0 for i in ids), "ids must be positive ints"
assert len(set(ids)) == len(ids), "duplicate entry ids"
for entry in entries:
assert len(entry) == 5, f"record {entry[0]} has {len(entry)} fields, not 5"
_id, kanji, kana, senses, common = entry
assert isinstance(kanji, list) and isinstance(kana, list) and kana, entry[0]
assert isinstance(senses, list) and all(isinstance(s, list) and s for s in senses)
assert common in (0, 1)
by_id = _entry_index(compact_jmdict)
iru = by_id[1577980]
assert "いる" in iru[2] and "居る" in iru[1]
konnichiwa = by_id[1289400]
assert "こんにちは" in konnichiwa[2] # kanji is NOT empty: 今日は (rK) and 今日わ (sK)
assert "きょう" in by_id[1579110][2]
hanaseru = by_id[1562360]
assert any("to be able to speak" in gloss for sense in hanaseru[3] for gloss in sense)
def test_jmdict_common_flag_count(compact_jmdict):
"""22,637 entries carry common=1: exactly the word count of jmdict-eng-common at this
release (research § Q3), an independent check that the flag was projected correctly.
The ranked lookup (plan 02-03) ranks on it."""
assert sum(entry[4] for entry in compact_jmdict["entries"]) == JMDICT_COMMON_ENTRIES
def test_jlpt_ids_resolve_in_jmdict(compact_jmdict):
"""Every non-empty jmdict_seq on the five lists is an entry (7,748 of 7,748).
538 of them are NOT in jmdict-eng-common (research § Q3), which is why the full
projection ships rather than the common subset.
"""
ids = jlpt_ids()
assert len(ids) == EXPECTED_UNIQUE_IDS
entry_ids = {entry[0] for entry in compact_jmdict["entries"]}
missing = sorted(int(i) for i in ids if int(i) not in entry_ids)
assert missing == [], f"{len(missing)} JLPT ids have no JMdict entry: {missing[:20]}"