Spaces:
Running on Zero
Running on Zero
Download tests/test_data_assets.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 10.7 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f7e73d2aec50cd58e583c49da86b2c0de1485772/tests/test_data_assets.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@f7e73d2aec50cd58e583c49da86b2c0de1485772/tests/test_data_assets.py
-
curl -L -o test_data_assets.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f7e73d2aec50cd58e583c49da86b2c0de1485772/tests/test_data_assets.py
10.7 kB
| """Guards on data/: the committed lexicon files are exactly what their build scripts wrote. | |
| D-09 and D-13 require the JLPT lists and the kanji map to be pinned, committed, hashed and | |
| licensed; D-07 does the same for the JMdict projection. scripts/build_jlpt.py and | |
| scripts/build_jmdict.py produce them and write a README with SHA-256 rows; these tests read | |
| the files back and compare (the tests/test_vendor.py pattern), so a hand-edit, a stale | |
| README or a skipped `git lfs pull` fails the quick loop. Files only, never the network. | |
| """ | |
| from __future__ import annotations | |
| import csv | |
| import gzip | |
| import hashlib | |
| import io | |
| import json | |
| import re | |
| import time | |
| from collections import Counter | |
| from pathlib import Path | |
| import pytest | |
| REPO_ROOT = Path(__file__).resolve().parent.parent | |
| JLPT = REPO_ROOT / "data" / "jlpt" | |
| JLPT_README = JLPT / "README.md" | |
| LEVELS = ("n5", "n4", "n3", "n2", "n1") | |
| CSV_FILES = tuple(f"{level}.csv" for level in LEVELS) | |
| KANJI_LEVELS = "kanji_levels.json" | |
| JLPT_LICENCES = { | |
| "LICENSE-yomitan-jlpt-vocab.txt": "Attribution-ShareAlike 4.0", | |
| "LICENSE-kanji-data.txt": "Permission is hereby granted", | |
| } | |
| # Measured on the pinned sources by scripts/build_jlpt.py (plan 02-01); the script refuses | |
| # to write on any other numbers, so these pin the data AND the parse. | |
| EXPECTED_ROWS = {"n5": 684, "n4": 640, "n3": 1730, "n2": 1812, "n1": 3427} | |
| EXPECTED_EMPTY_IDS = 14 | |
| EXPECTED_UNIQUE_IDS = 7748 | |
| EXPECTED_MULTI_ROW_IDS = 505 | |
| EXPECTED_MULTI_LEVEL_IDS = 447 | |
| EXPECTED_KANJI = {"N5": 79, "N4": 166, "N3": 367, "N2": 367, "N1": 1232} | |
| # CJK Unified Ideographs (+ Extension A) and the Compatibility Ideographs block: every key | |
| # of kanji_levels.json must be one character from one of these. | |
| CJK_RANGES = ((0x3400, 0x4DBF), (0x4E00, 0x9FFF), (0xF900, 0xFAFF)) | |
| def _readme_rows(path: Path) -> dict[str, tuple[int, str]]: | |
| """``name -> (bytes, sha256)`` from every ``| `file` | bytes | `sha` |`` row of a README.""" | |
| text = path.read_text(encoding="utf-8") | |
| rows = re.findall(r"^\| `([^`]+)` \| ([\d,]+) \| `([0-9a-f]{64})` \|", text, re.M) | |
| assert rows, f"{path.relative_to(REPO_ROOT)} has no hash rows; run its build script" | |
| return {name: (int(size.replace(",", "")), digest) for name, size, digest in rows} | |
| def _assert_files_match_readme(directory: Path, readme: Path, names: tuple[str, ...]) -> None: | |
| recorded = _readme_rows(readme) | |
| rel = readme.relative_to(REPO_ROOT).as_posix() | |
| for name in names: | |
| assert name in recorded, f"{name} has no row in {rel}" | |
| data = (directory / name).read_bytes() | |
| size, digest = recorded[name] | |
| assert len(data) == size, f"{name}: {rel} says {size:,} bytes, disk has {len(data):,}" | |
| assert hashlib.sha256(data).hexdigest() == digest, ( | |
| f"{name} differs from the hash in {rel}; regenerate with its build script " | |
| "rather than editing by hand" | |
| ) | |
| def _jlpt_rows(level: str) -> list[list[str]]: | |
| """Data rows of one CSV, the way scripts/build_jlpt.py reads them (header skipped).""" | |
| text = (JLPT / f"{level}.csv").read_bytes().decode("utf-8-sig") | |
| rows = [row for row in csv.reader(io.StringIO(text, newline="")) if row] | |
| assert rows[0] == ["jmdict_seq", "kana", "kanji", "waller_definition"], rows[0] | |
| return rows[1:] | |
| def jlpt_ids() -> set[str]: | |
| """Every non-empty ``jmdict_seq`` across the five lists (the ids that must join JMdict).""" | |
| return {row[0] for level in LEVELS for row in _jlpt_rows(level) if row[0]} | |
| def test_jlpt_files_match_readme(): | |
| """Bytes and SHA-256 of all eight files are the README's, so no file was hand-edited.""" | |
| _assert_files_match_readme(JLPT, JLPT_README, (*CSV_FILES, KANJI_LEVELS, *JLPT_LICENCES)) | |
| text = JLPT_README.read_text(encoding="utf-8") | |
| assert "2025.08.01.0" in text, "README does not name the yomitan-jlpt-vocab tag" | |
| assert re.search(r"\b[0-9a-f]{40}\b", text), "README does not name the kanji-data commit" | |
| assert "tanos.co.uk" in text, "README does not credit Jonathan Waller's lists" | |
| def test_jlpt_csv_counts(): | |
| """Per-level rows, the 14 id-less N1 rows, 7,748 unique ids, and the duplicate structure.""" | |
| tables = {level: _jlpt_rows(level) for level in LEVELS} | |
| assert {level: len(rows) for level, rows in tables.items()} == EXPECTED_ROWS | |
| empty = [(level, row) for level, rows in tables.items() for row in rows if not row[0]] | |
| assert len(empty) == EXPECTED_EMPTY_IDS | |
| assert {level for level, _ in empty} == {"n1"}, "id-less rows outside n1.csv" | |
| for level, rows in tables.items(): | |
| for row in rows: | |
| assert len(row) == 4, f"{level}.csv row {row} has {len(row)} columns" | |
| assert row[0] == "" or row[0].isdigit(), f"{level}.csv non-numeric id {row[0]!r}" | |
| level_count: Counter[str] = Counter() | |
| for rows in tables.values(): | |
| level_count.update({row[0] for row in rows if row[0]}) | |
| row_count = Counter(row[0] for rows in tables.values() for row in rows if row[0]) | |
| assert len(level_count) == EXPECTED_UNIQUE_IDS | |
| assert sum(1 for n in row_count.values() if n > 1) == EXPECTED_MULTI_ROW_IDS | |
| assert sum(1 for n in level_count.values() if n > 1) == EXPECTED_MULTI_LEVEL_IDS | |
| def test_kanji_levels_distribution(): | |
| """2,211 single-kanji keys, the five counts research recorded, every value an N-level.""" | |
| levels = json.loads((JLPT / KANJI_LEVELS).read_text(encoding="utf-8")) | |
| assert isinstance(levels, dict) and len(levels) == sum(EXPECTED_KANJI.values()) | |
| assert dict(Counter(levels.values())) == EXPECTED_KANJI | |
| assert set(levels.values()) <= set(EXPECTED_KANJI) | |
| for literal in levels: | |
| assert len(literal) == 1, f"key {literal!r} is not one character" | |
| cp = ord(literal) | |
| assert any(lo <= cp <= hi for lo, hi in CJK_RANGES), f"{literal!r} U+{cp:04X} not CJK" | |
| assert list(levels) == sorted(levels), "kanji_levels.json is not sorted by literal" | |
| def test_jlpt_licences_present(): | |
| for name, phrase in JLPT_LICENCES.items(): | |
| text = (JLPT / name).read_text(encoding="utf-8") | |
| assert phrase in text, f"{name} does not contain {phrase!r}" | |
| readme = JLPT_README.read_text(encoding="utf-8") | |
| assert "CC BY-SA 4.0" in readme and "MIT" in readme | |
| assert "no official JLPT vocabulary list" in readme | |
| # --- data/jmdict ---------------------------------------------------------------------------- | |
| JMDICT = REPO_ROOT / "data" / "jmdict" | |
| JMDICT_FILE = JMDICT / "jmdict-compact.json.gz" | |
| JMDICT_README = JMDICT / "README.md" | |
| JMDICT_RELEASE = "3.6.2+20260831182826" | |
| JMDICT_DICT_DATE = "2026-08-31" | |
| JMDICT_ENTRIES = 218672 | |
| JMDICT_COMMON_ENTRIES = 22637 | |
| # Research measured 7.66 MB; anything outside this band is a truncated write or a different | |
| # projection. An LFS pointer file is ~130 bytes. | |
| JMDICT_MIN_BYTES, JMDICT_MAX_BYTES = 6_000_000, 10_000_000 | |
| GZIP_MAGIC = b"\x1f\x8b" | |
| LFS_POINTER_PREFIX = b"version https://git-lfs" | |
| def compact_jmdict() -> dict: | |
| """The compact file gunzipped and parsed once per module, the way the app loads it.""" | |
| t = time.perf_counter() | |
| with gzip.open(JMDICT_FILE, "rt", encoding="utf-8") as fh: | |
| data = json.load(fh) | |
| print(f"\ncompact JMdict loaded in {time.perf_counter() - t:.2f} s") | |
| return data | |
| def test_jmdict_file_is_real_lfs_object(): | |
| """The gzip itself, not an LFS pointer (a pointer means `git lfs pull` was skipped).""" | |
| head = JMDICT_FILE.read_bytes()[:64] | |
| assert not head.startswith(LFS_POINTER_PREFIX), ( | |
| f"{JMDICT_FILE.relative_to(REPO_ROOT)} is a Git LFS pointer; run `git lfs pull`" | |
| ) | |
| assert head[:2] == GZIP_MAGIC, f"not a gzip file: first bytes {head[:4]!r}" | |
| size = JMDICT_FILE.stat().st_size | |
| assert JMDICT_MIN_BYTES <= size <= JMDICT_MAX_BYTES, f"{size:,} bytes is not the projection" | |
| def test_jmdict_matches_readme(): | |
| _assert_files_match_readme(JMDICT, JMDICT_README, (JMDICT_FILE.name,)) | |
| text = JMDICT_README.read_text(encoding="utf-8") | |
| assert JMDICT_RELEASE in text and JMDICT_DICT_DATE in text | |
| assert text.count("James William Breen") == 1 | |
| assert text.count("edrdg.org/edrdg/licence.html") == 1 | |
| assert "CC BY-SA 4.0" in text | |
| def _entry_index(data: dict) -> dict[int, list]: | |
| return {entry[0]: entry for entry in data["entries"]} | |
| def test_jmdict_meta_and_count(compact_jmdict): | |
| """Pinned meta, 218,672 five-field records with positive unique ids, and spot entries.""" | |
| meta = compact_jmdict["meta"] | |
| assert meta["source"] == "scriptin/jmdict-simplified" | |
| assert meta["version"] == JMDICT_RELEASE | |
| assert meta["dictDate"] == JMDICT_DICT_DATE | |
| assert meta["entries"] == JMDICT_ENTRIES | |
| entries = compact_jmdict["entries"] | |
| assert len(entries) == JMDICT_ENTRIES | |
| ids = [entry[0] for entry in entries] | |
| assert all(isinstance(i, int) and i > 0 for i in ids), "ids must be positive ints" | |
| assert len(set(ids)) == len(ids), "duplicate entry ids" | |
| for entry in entries: | |
| assert len(entry) == 5, f"record {entry[0]} has {len(entry)} fields, not 5" | |
| _id, kanji, kana, senses, common = entry | |
| assert isinstance(kanji, list) and isinstance(kana, list) and kana, entry[0] | |
| assert isinstance(senses, list) and all(isinstance(s, list) and s for s in senses) | |
| assert common in (0, 1) | |
| by_id = _entry_index(compact_jmdict) | |
| iru = by_id[1577980] | |
| assert "いる" in iru[2] and "居る" in iru[1] | |
| konnichiwa = by_id[1289400] | |
| assert "こんにちは" in konnichiwa[2] # kanji is NOT empty: 今日は (rK) and 今日わ (sK) | |
| assert "きょう" in by_id[1579110][2] | |
| hanaseru = by_id[1562360] | |
| assert any("to be able to speak" in gloss for sense in hanaseru[3] for gloss in sense) | |
| def test_jmdict_common_flag_count(compact_jmdict): | |
| """22,637 entries carry common=1: exactly the word count of jmdict-eng-common at this | |
| release (research § Q3), an independent check that the flag was projected correctly. | |
| The ranked lookup (plan 02-03) ranks on it.""" | |
| assert sum(entry[4] for entry in compact_jmdict["entries"]) == JMDICT_COMMON_ENTRIES | |
| def test_jlpt_ids_resolve_in_jmdict(compact_jmdict): | |
| """Every non-empty jmdict_seq on the five lists is an entry (7,748 of 7,748). | |
| 538 of them are NOT in jmdict-eng-common (research § Q3), which is why the full | |
| projection ships rather than the common subset. | |
| """ | |
| ids = jlpt_ids() | |
| assert len(ids) == EXPECTED_UNIQUE_IDS | |
| entry_ids = {entry[0] for entry in compact_jmdict["entries"]} | |
| missing = sorted(int(i) for i in ids if int(i) not in entry_ids) | |
| assert missing == [], f"{len(missing)} JLPT ids have no JMdict entry: {missing[:20]}" | |