Spaces:
Running on Zero
Running on Zero
File size: 13,770 Bytes
ac406ae 549c072 ac406ae 549c072 ac406ae 549c072 ac406ae 549c072 755b152 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 | """Guards on data/: the committed lexicon files are exactly what their build scripts wrote.
D-09 and D-13 require the JLPT lists and the kanji map to be pinned, committed, hashed and
licensed; D-07 does the same for the JMdict projection. scripts/build_jlpt.py and
scripts/build_jmdict.py produce them and write a README with SHA-256 rows; these tests read
the files back and compare (the tests/test_vendor.py pattern), so a hand-edit, a stale
README or a skipped `git lfs pull` fails the quick loop. Files only, never the network.
"""
from __future__ import annotations
import csv
import gzip
import hashlib
import io
import json
import re
import time
from collections import Counter
from pathlib import Path
import pytest
REPO_ROOT = Path(__file__).resolve().parent.parent
JLPT = REPO_ROOT / "data" / "jlpt"
JLPT_README = JLPT / "README.md"
LEVELS = ("n5", "n4", "n3", "n2", "n1")
CSV_FILES = tuple(f"{level}.csv" for level in LEVELS)
KANJI_LEVELS = "kanji_levels.json"
JLPT_LICENCES = {
"LICENSE-yomitan-jlpt-vocab.txt": "Attribution-ShareAlike 4.0",
"LICENSE-kanji-data.txt": "Permission is hereby granted",
}
# Measured on the pinned sources by scripts/build_jlpt.py (plan 02-01); the script refuses
# to write on any other numbers, so these pin the data AND the parse.
EXPECTED_ROWS = {"n5": 684, "n4": 640, "n3": 1730, "n2": 1812, "n1": 3427}
EXPECTED_EMPTY_IDS = 14
EXPECTED_UNIQUE_IDS = 7748
EXPECTED_MULTI_ROW_IDS = 505
EXPECTED_MULTI_LEVEL_IDS = 447
EXPECTED_KANJI = {"N5": 79, "N4": 166, "N3": 367, "N2": 367, "N1": 1232}
# CJK Unified Ideographs (+ Extension A) and the Compatibility Ideographs block: every key
# of kanji_levels.json must be one character from one of these.
CJK_RANGES = ((0x3400, 0x4DBF), (0x4E00, 0x9FFF), (0xF900, 0xFAFF))
def _readme_rows(path: Path) -> dict[str, tuple[int, str]]:
"""``name -> (bytes, sha256)`` from every ``| `file` | bytes | `sha` |`` row of a README."""
text = path.read_text(encoding="utf-8")
rows = re.findall(r"^\| `([^`]+)` \| ([\d,]+) \| `([0-9a-f]{64})` \|", text, re.M)
assert rows, f"{path.relative_to(REPO_ROOT)} has no hash rows; run its build script"
return {name: (int(size.replace(",", "")), digest) for name, size, digest in rows}
def _assert_files_match_readme(directory: Path, readme: Path, names: tuple[str, ...]) -> None:
recorded = _readme_rows(readme)
rel = readme.relative_to(REPO_ROOT).as_posix()
for name in names:
assert name in recorded, f"{name} has no row in {rel}"
data = (directory / name).read_bytes()
size, digest = recorded[name]
assert len(data) == size, f"{name}: {rel} says {size:,} bytes, disk has {len(data):,}"
assert hashlib.sha256(data).hexdigest() == digest, (
f"{name} differs from the hash in {rel}; regenerate with its build script "
"rather than editing by hand"
)
def _jlpt_rows(level: str) -> list[list[str]]:
"""Data rows of one CSV, the way scripts/build_jlpt.py reads them (header skipped)."""
text = (JLPT / f"{level}.csv").read_bytes().decode("utf-8-sig")
rows = [row for row in csv.reader(io.StringIO(text, newline="")) if row]
assert rows[0] == ["jmdict_seq", "kana", "kanji", "waller_definition"], rows[0]
return rows[1:]
def jlpt_ids() -> set[str]:
"""Every non-empty ``jmdict_seq`` across the five lists (the ids that must join JMdict)."""
return {row[0] for level in LEVELS for row in _jlpt_rows(level) if row[0]}
def test_jlpt_files_match_readme():
"""Bytes and SHA-256 of all eight files are the README's, so no file was hand-edited."""
_assert_files_match_readme(JLPT, JLPT_README, (*CSV_FILES, KANJI_LEVELS, *JLPT_LICENCES))
text = JLPT_README.read_text(encoding="utf-8")
assert "2025.08.01.0" in text, "README does not name the yomitan-jlpt-vocab tag"
assert re.search(r"\b[0-9a-f]{40}\b", text), "README does not name the kanji-data commit"
assert "tanos.co.uk" in text, "README does not credit Jonathan Waller's lists"
def test_jlpt_csv_counts():
"""Per-level rows, the 14 id-less N1 rows, 7,748 unique ids, and the duplicate structure."""
tables = {level: _jlpt_rows(level) for level in LEVELS}
assert {level: len(rows) for level, rows in tables.items()} == EXPECTED_ROWS
empty = [(level, row) for level, rows in tables.items() for row in rows if not row[0]]
assert len(empty) == EXPECTED_EMPTY_IDS
assert {level for level, _ in empty} == {"n1"}, "id-less rows outside n1.csv"
for level, rows in tables.items():
for row in rows:
assert len(row) == 4, f"{level}.csv row {row} has {len(row)} columns"
assert row[0] == "" or row[0].isdigit(), f"{level}.csv non-numeric id {row[0]!r}"
level_count: Counter[str] = Counter()
for rows in tables.values():
level_count.update({row[0] for row in rows if row[0]})
row_count = Counter(row[0] for rows in tables.values() for row in rows if row[0])
assert len(level_count) == EXPECTED_UNIQUE_IDS
assert sum(1 for n in row_count.values() if n > 1) == EXPECTED_MULTI_ROW_IDS
assert sum(1 for n in level_count.values() if n > 1) == EXPECTED_MULTI_LEVEL_IDS
def test_kanji_levels_distribution():
"""2,211 single-kanji keys, the five counts research recorded, every value an N-level."""
levels = json.loads((JLPT / KANJI_LEVELS).read_text(encoding="utf-8"))
assert isinstance(levels, dict) and len(levels) == sum(EXPECTED_KANJI.values())
assert dict(Counter(levels.values())) == EXPECTED_KANJI
assert set(levels.values()) <= set(EXPECTED_KANJI)
for literal in levels:
assert len(literal) == 1, f"key {literal!r} is not one character"
cp = ord(literal)
assert any(lo <= cp <= hi for lo, hi in CJK_RANGES), f"{literal!r} U+{cp:04X} not CJK"
assert list(levels) == sorted(levels), "kanji_levels.json is not sorted by literal"
def test_jlpt_licences_present():
for name, phrase in JLPT_LICENCES.items():
text = (JLPT / name).read_text(encoding="utf-8")
assert phrase in text, f"{name} does not contain {phrase!r}"
readme = JLPT_README.read_text(encoding="utf-8")
assert "CC BY-SA 4.0" in readme and "MIT" in readme
assert "no official JLPT vocabulary list" in readme
# --- data/jmdict ----------------------------------------------------------------------------
JMDICT = REPO_ROOT / "data" / "jmdict"
JMDICT_FILE = JMDICT / "jmdict-compact.json.gz"
JMDICT_README = JMDICT / "README.md"
JMDICT_RELEASE = "3.6.2+20260831182826"
JMDICT_DICT_DATE = "2026-08-31"
JMDICT_ENTRIES = 218672
JMDICT_COMMON_ENTRIES = 22637
# Research measured 7.66 MB; anything outside this band is a truncated write or a different
# projection. An LFS pointer file is ~130 bytes.
JMDICT_MIN_BYTES, JMDICT_MAX_BYTES = 6_000_000, 10_000_000
GZIP_MAGIC = b"\x1f\x8b"
LFS_POINTER_PREFIX = b"version https://git-lfs"
@pytest.fixture(scope="module")
def compact_jmdict() -> dict:
"""The compact file gunzipped and parsed once per module, the way the app loads it."""
t = time.perf_counter()
with gzip.open(JMDICT_FILE, "rt", encoding="utf-8") as fh:
data = json.load(fh)
print(f"\ncompact JMdict loaded in {time.perf_counter() - t:.2f} s")
return data
def test_jmdict_file_is_real_lfs_object():
"""The gzip itself, not an LFS pointer (a pointer means `git lfs pull` was skipped)."""
head = JMDICT_FILE.read_bytes()[:64]
assert not head.startswith(LFS_POINTER_PREFIX), (
f"{JMDICT_FILE.relative_to(REPO_ROOT)} is a Git LFS pointer; run `git lfs pull`"
)
assert head[:2] == GZIP_MAGIC, f"not a gzip file: first bytes {head[:4]!r}"
size = JMDICT_FILE.stat().st_size
assert JMDICT_MIN_BYTES <= size <= JMDICT_MAX_BYTES, f"{size:,} bytes is not the projection"
def test_jmdict_matches_readme():
_assert_files_match_readme(JMDICT, JMDICT_README, (JMDICT_FILE.name,))
text = JMDICT_README.read_text(encoding="utf-8")
assert JMDICT_RELEASE in text and JMDICT_DICT_DATE in text
assert text.count("James William Breen") == 1
assert text.count("edrdg.org/edrdg/licence.html") == 1
assert "CC BY-SA 4.0" in text
def _entry_index(data: dict) -> dict[int, list]:
return {entry[0]: entry for entry in data["entries"]}
def test_jmdict_meta_and_count(compact_jmdict):
"""Pinned meta, 218,672 five-field records with positive unique ids, and spot entries."""
meta = compact_jmdict["meta"]
assert meta["source"] == "scriptin/jmdict-simplified"
assert meta["version"] == JMDICT_RELEASE
assert meta["dictDate"] == JMDICT_DICT_DATE
assert meta["entries"] == JMDICT_ENTRIES
entries = compact_jmdict["entries"]
assert len(entries) == JMDICT_ENTRIES
ids = [entry[0] for entry in entries]
assert all(isinstance(i, int) and i > 0 for i in ids), "ids must be positive ints"
assert len(set(ids)) == len(ids), "duplicate entry ids"
for entry in entries:
assert len(entry) == 5, f"record {entry[0]} has {len(entry)} fields, not 5"
_id, kanji, kana, senses, common = entry
assert isinstance(kanji, list) and isinstance(kana, list) and kana, entry[0]
assert isinstance(senses, list) and all(isinstance(s, list) and s for s in senses)
assert common in (0, 1)
by_id = _entry_index(compact_jmdict)
iru = by_id[1577980]
assert "いる" in iru[2] and "居る" in iru[1]
konnichiwa = by_id[1289400]
assert "こんにちは" in konnichiwa[2] # kanji is NOT empty: 今日は (rK) and 今日わ (sK)
assert "きょう" in by_id[1579110][2]
hanaseru = by_id[1562360]
assert any("to be able to speak" in gloss for sense in hanaseru[3] for gloss in sense)
def test_jmdict_common_flag_count(compact_jmdict):
"""22,637 entries carry common=1: exactly the word count of jmdict-eng-common at this
release (research § Q3), an independent check that the flag was projected correctly.
The ranked lookup (plan 02-03) ranks on it."""
assert sum(entry[4] for entry in compact_jmdict["entries"]) == JMDICT_COMMON_ENTRIES
def test_jlpt_ids_resolve_in_jmdict(compact_jmdict):
"""Every non-empty jmdict_seq on the five lists is an entry (7,748 of 7,748).
538 of them are NOT in jmdict-eng-common (research § Q3), which is why the full
projection ships rather than the common subset.
"""
ids = jlpt_ids()
assert len(ids) == EXPECTED_UNIQUE_IDS
entry_ids = {entry[0] for entry in compact_jmdict["entries"]}
missing = sorted(int(i) for i in ids if int(i) not in entry_ids)
assert missing == [], f"{len(missing)} JLPT ids have no JMdict entry: {missing[:20]}"
# --- data/mt ----------------------------------------------------------------------------------
MT = REPO_ROOT / "data" / "mt"
MT_MODEL = MT / "opus-mt-ja-en-ct2-int8"
MT_README = MT / "README.md"
MT_FILES = ("model.bin", "shared_vocabulary.json", "config.json", "source.spm", "target.spm")
MT_LFS_FILES = ("model.bin", "source.spm", "target.spm")
MT_MODEL_ID = "Helsinki-NLP/opus-mt-ja-en"
# Research measured model.bin at 77,339,435 B and the two spm files at ~0.8 MB; an LFS pointer
# is ~130 B. These are floors, not equalities - the README rows pin the exact bytes.
MT_MIN_MODEL_BYTES = 70_000_000
MT_MIN_SPM_BYTES = 500_000
# What ctranslate2 4.8.2's TransformersConverter writes for a Marian model (pinned after the
# first run of scripts/convert_mt.py); the translator reads these to frame the decoder.
MT_CONFIG_KEYS = {"bos_token", "eos_token", "unk_token", "decoder_start_token"}
def test_mt_files_match_readme():
"""Bytes and SHA-256 of the five model files are the README's; the README pins the revision."""
_assert_files_match_readme(MT_MODEL, MT_README, MT_FILES)
text = MT_README.read_text(encoding="utf-8")
assert MT_MODEL_ID in text
assert re.search(r"\b[0-9a-f]{40}\b", text), "README does not record the Hub revision SHA"
assert "opus-2019-12-18" in text, "README does not name the OPUS-MT release tag"
def test_mt_model_is_real_not_pointer():
"""The binaries themselves, not Git LFS pointers (a pointer means `git lfs pull` was skipped).
A pointer here would fail on the Space at the first EN tap; this fails in the quick loop.
"""
for name in MT_LFS_FILES:
path = MT_MODEL / name
head = path.read_bytes()[:64]
assert not head.startswith(LFS_POINTER_PREFIX), (
f"{path.relative_to(REPO_ROOT)} is a Git LFS pointer; run `git lfs pull`"
)
assert (MT_MODEL / "model.bin").stat().st_size > MT_MIN_MODEL_BYTES
for name in ("source.spm", "target.spm"):
assert (MT_MODEL / name).stat().st_size > MT_MIN_SPM_BYTES, name
def test_mt_licence_and_notice_present():
licence = (MT / "LICENSE-apache-2.0.txt").read_text(encoding="utf-8")
assert "Apache License" in licence and "Version 2.0" in licence
notice = (MT / "NOTICE").read_text(encoding="utf-8")
assert MT_MODEL_ID in notice
assert "scripts/convert_mt.py" in notice
readme = MT_README.read_text(encoding="utf-8")
assert "Apache-2.0" in readme and "University of Helsinki" in readme
def test_mt_config_is_ct2():
"""config.json is the CTranslate2 converter's, with the special tokens a Marian model needs."""
config = json.loads((MT_MODEL / "config.json").read_text(encoding="utf-8"))
print(f"\nct2 config keys: {sorted(config)}")
assert set(config) >= MT_CONFIG_KEYS, sorted(config)
assert config["eos_token"] == "</s>" and config["unk_token"] == "<unk>"
vocabulary = json.loads((MT_MODEL / "shared_vocabulary.json").read_text(encoding="utf-8"))
assert isinstance(vocabulary, list) and len(vocabulary) > 50_000
assert "</s>" in vocabulary and "<unk>" in vocabulary
|