Spaces:
Running on Zero
Running on Zero
Download scripts/build_jlpt.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 17.2 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f7e73d2aec50cd58e583c49da86b2c0de1485772/scripts/build_jlpt.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@f7e73d2aec50cd58e583c49da86b2c0de1485772/scripts/build_jlpt.py
-
curl -L -o build_jlpt.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f7e73d2aec50cd58e583c49da86b2c0de1485772/scripts/build_jlpt.py
17.2 kB
| """Build data/jlpt/: the pinned JLPT vocabulary lists and the kanji-level map, hashed. | |
| D-09 says word levels come from ONE published open JLPT list, pinned by version, committed | |
| as a data file, licence recorded; D-13 adds a second, independent kanji axis with the same | |
| rules. 02-RESEARCH.md § Q2 picked the sources and measured the counts this script refuses | |
| to deviate from: | |
| 1. Fetch the five ``original_data/n{1..5}.csv`` files of stephenmk/yomitan-jlpt-vocab at | |
| release tag ``2025.08.01.0`` (columns ``jmdict_seq,kana,kanji,waller_definition``), the | |
| only candidate keyed to JMdict entry ids. They are stored byte-for-byte - the SHA-256 in | |
| the README is the upstream file's, not a re-serialisation. | |
| 2. Fetch ``kanji.json`` from davidluzgouveia/kanji-data at ONE pinned commit and project | |
| ``{literal: "N<jlpt_new>"}`` for the 2,211 kanji that carry a level. ``jlpt_new`` is | |
| Waller's N1-N5 scale; the ``jlpt_old`` field is KANJIDIC2's pre-2010 1-4 scale and is | |
| printed only as corroboration (research: 4->103, 3->181, 2->739, 1->1,207). | |
| 3. Fetch both licence texts and check they are what the research recorded. | |
| 4. Compare every count with the expected table. On ANY mismatch, print the observed numbers | |
| and exit 1 WITHOUT writing: a tag and a commit are immutable, so a mismatch means the | |
| research count or this parser is wrong, and a human decides which. | |
| 5. Write the files and ``data/jlpt/README.md`` with SHA-256 rows that | |
| ``tests/test_data_assets.py`` re-checks in the quick loop. | |
| Re-running is idempotent: the data is byte-identical and the README keeps its recorded | |
| date unless a hash changed. Run from the repo root:: | |
| .venv/Scripts/python.exe scripts/build_jlpt.py | |
| Stdlib only, on purpose, and no API call: the kanji-data commit was resolved once from the | |
| repository's HEAD on 2026-09-06 and hard-coded below, so a re-run cannot drift. | |
| """ | |
| from __future__ import annotations | |
| import csv | |
| import hashlib | |
| import io | |
| import json | |
| import re | |
| import sys | |
| import urllib.request | |
| from collections import Counter | |
| from datetime import UTC, datetime | |
| from pathlib import Path | |
| REPO_ROOT = Path(__file__).resolve().parent.parent | |
| OUT_DIR = REPO_ROOT / "data" / "jlpt" | |
| README = OUT_DIR / "README.md" | |
| KANJI_OUT = "kanji_levels.json" | |
| RAW = "https://raw.githubusercontent.com" | |
| YOMITAN_REPO = "stephenmk/yomitan-jlpt-vocab" | |
| YOMITAN_TAG = "2025.08.01.0" | |
| KANJI_DATA_REPO = "davidluzgouveia/kanji-data" | |
| # HEAD of the repository when resolved (2026-09-06); the commit itself is dated 2026-02-27, | |
| # which is the "repo pushed 2026-02-27" that 02-RESEARCH.md § Q2 recorded. | |
| KANJI_DATA_COMMIT = "00fd7079c3890f430759536f91aa5e854ec0ca4f" | |
| LEVELS = ("n5", "n4", "n3", "n2", "n1") | |
| CSV_URLS: dict[str, str] = { | |
| f"{level}.csv": f"{RAW}/{YOMITAN_REPO}/{YOMITAN_TAG}/original_data/{level}.csv" | |
| for level in LEVELS | |
| } | |
| KANJI_URL = f"{RAW}/{KANJI_DATA_REPO}/{KANJI_DATA_COMMIT}/kanji.json" | |
| # name on disk -> (url, phrases of which at least one must appear) | |
| LICENSES: dict[str, tuple[str, tuple[str, ...]]] = { | |
| "LICENSE-yomitan-jlpt-vocab.txt": ( | |
| f"{RAW}/{YOMITAN_REPO}/{YOMITAN_TAG}/LICENSE.txt", | |
| ("Attribution-ShareAlike 4.0",), | |
| ), | |
| "LICENSE-kanji-data.txt": ( | |
| f"{RAW}/{KANJI_DATA_REPO}/{KANJI_DATA_COMMIT}/LICENSE", | |
| ("MIT License", "Permission is hereby granted"), | |
| ), | |
| } | |
| # Measured on the pinned files (plan 02-01, 2026-09-06). The script refuses to write if any | |
| # of these differ. Two numbers correct 02-RESEARCH.md § Q2, which was measured on the same | |
| # tag: research's "7,747 unique ids" is 7,748 (every one of them resolves in jmdict-eng | |
| # 3.6.2, so the join stays complete), and its "505 ids on more than one level" is the count | |
| # of ids that appear in more than one ROW - 447 are on more than one level and 71 are listed | |
| # twice within one level (行く under いく and ゆく in n5, for example). | |
| EXPECTED_ROWS = {"n5": 684, "n4": 640, "n3": 1730, "n2": 1812, "n1": 3427} | |
| EXPECTED_EMPTY_IDS = 14 | |
| EXPECTED_EMPTY_ID_LEVELS = {"n1"} | |
| EXPECTED_UNIQUE_IDS = 7748 | |
| EXPECTED_MULTI_ROW_IDS = 505 | |
| EXPECTED_MULTI_LEVEL_IDS = 447 | |
| EXPECTED_WITHIN_LEVEL_DUP_IDS = 71 | |
| EXPECTED_KANJI = {"N5": 79, "N4": 166, "N3": 367, "N2": 367, "N1": 1232} | |
| EXPECTED_KANJI_OLD = {4: 103, 3: 181, 2: 739, 1: 1207} | |
| LICENCE_TEXT = ( | |
| "JLPT levels: Jonathan Waller's JLPT Resources (https://www.tanos.co.uk/jlpt/, CC BY) via " | |
| "stephenmk/yomitan-jlpt-vocab (CC BY-SA 4.0); kanji levels extracted from " | |
| "davidluzgouveia/kanji-data (MIT) which took its levels from the same Waller lists. " | |
| "There is no official JLPT vocabulary list; these are estimates." | |
| ) | |
| def fetch(url: str) -> bytes: | |
| req = urllib.request.Request(url, headers={"User-Agent": "japanese-learning-avatar data build"}) | |
| with urllib.request.urlopen(req, timeout=120) as resp: # noqa: S310 - pinned https URLs | |
| return resp.read() | |
| def sha256(data: bytes) -> str: | |
| return hashlib.sha256(data).hexdigest() | |
| def parse_csv(data: bytes) -> list[list[str]]: | |
| """Data rows of one upstream CSV; a header row (``row[0] == "jmdict_seq"``) is skipped.""" | |
| text = data.decode("utf-8-sig") | |
| rows = [row for row in csv.reader(io.StringIO(text, newline="")) if row] | |
| if rows and rows[0][0] == "jmdict_seq": | |
| rows = rows[1:] | |
| return rows | |
| def csv_stats(tables: dict[str, list[list[str]]]) -> dict: | |
| """The counts the README and the tests pin: rows per level, empty ids, unique, shared.""" | |
| rows_per_level = {level: len(rows) for level, rows in tables.items()} | |
| empty_by_level: Counter[str] = Counter() | |
| ids_by_level: dict[str, set[str]] = {} | |
| for level, rows in tables.items(): | |
| ids: set[str] = set() | |
| for row in rows: | |
| seq = row[0].strip() | |
| if not seq: | |
| empty_by_level[level] += 1 | |
| continue | |
| if not seq.isdigit(): | |
| raise SystemExit(f"{level}.csv: non-numeric jmdict_seq {seq!r} in row {row}") | |
| ids.add(seq) | |
| ids_by_level[level] = ids | |
| level_count: Counter[str] = Counter() # id -> number of levels listing it | |
| for ids in ids_by_level.values(): | |
| level_count.update(ids) | |
| row_count: Counter[str] = Counter( # id -> number of rows carrying it, any level | |
| row[0].strip() for rows in tables.values() for row in rows if row[0].strip() | |
| ) | |
| return { | |
| "rows": rows_per_level, | |
| "total_rows": sum(rows_per_level.values()), | |
| "empty_ids": sum(empty_by_level.values()), | |
| "empty_id_levels": set(empty_by_level), | |
| "unique_ids": len(level_count), | |
| "multi_row_ids": sum(1 for n in row_count.values() if n > 1), | |
| "multi_level_ids": sum(1 for n in level_count.values() if n > 1), | |
| "within_level_dup_ids": sum(1 for seq, n in row_count.items() if n > level_count[seq]), | |
| } | |
| def check_csv_stats(stats: dict) -> None: | |
| problems: list[str] = [] | |
| if stats["rows"] != EXPECTED_ROWS: | |
| problems.append(f"rows per level {stats['rows']} != {EXPECTED_ROWS}") | |
| if stats["empty_ids"] != EXPECTED_EMPTY_IDS: | |
| problems.append(f"empty jmdict_seq rows {stats['empty_ids']} != {EXPECTED_EMPTY_IDS}") | |
| if stats["empty_id_levels"] != EXPECTED_EMPTY_ID_LEVELS: | |
| problems.append( | |
| f"empty ids found in {sorted(stats['empty_id_levels'])}, " | |
| f"expected only {sorted(EXPECTED_EMPTY_ID_LEVELS)}" | |
| ) | |
| if stats["unique_ids"] != EXPECTED_UNIQUE_IDS: | |
| problems.append(f"unique ids {stats['unique_ids']} != {EXPECTED_UNIQUE_IDS}") | |
| if stats["multi_row_ids"] != EXPECTED_MULTI_ROW_IDS: | |
| problems.append(f"ids in >1 row {stats['multi_row_ids']} != {EXPECTED_MULTI_ROW_IDS}") | |
| if stats["multi_level_ids"] != EXPECTED_MULTI_LEVEL_IDS: | |
| problems.append(f"ids on >1 level {stats['multi_level_ids']} != {EXPECTED_MULTI_LEVEL_IDS}") | |
| if stats["within_level_dup_ids"] != EXPECTED_WITHIN_LEVEL_DUP_IDS: | |
| problems.append( | |
| f"ids duplicated within a level {stats['within_level_dup_ids']} != " | |
| f"{EXPECTED_WITHIN_LEVEL_DUP_IDS}" | |
| ) | |
| if problems: | |
| raise SystemExit( | |
| "JLPT CSV counts differ from 02-RESEARCH.md § Q2; nothing written:\n " | |
| + "\n ".join(problems) | |
| ) | |
| def project_kanji(raw: bytes) -> tuple[dict[str, str], Counter[str], Counter[int]]: | |
| """``{literal: "N<jlpt_new>"}`` for every kanji with a level, plus both distributions.""" | |
| data = json.loads(raw.decode("utf-8")) | |
| if not isinstance(data, dict) or not data: | |
| raise SystemExit(f"kanji.json: expected a non-empty JSON object, got {type(data).__name__}") | |
| first_key, first_value = next(iter(data.items())) | |
| print(f"kanji.json: {type(data).__name__} with {len(data):,} keys; sample {first_key!r} -> ") | |
| print(" " + json.dumps(first_value, ensure_ascii=False)[:300]) | |
| if not isinstance(first_value, dict) or "jlpt_new" not in first_value: | |
| raise SystemExit("kanji.json: values are not objects carrying `jlpt_new`; shape changed") | |
| levels: dict[str, str] = {} | |
| new_dist: Counter[str] = Counter() | |
| old_dist: Counter[int] = Counter() | |
| for literal, info in data.items(): | |
| old = info.get("jlpt_old") | |
| if old is not None: | |
| old_dist[int(old)] += 1 | |
| new = info.get("jlpt_new") | |
| if new is None: | |
| continue | |
| if not isinstance(new, int) or not 1 <= new <= 5: | |
| raise SystemExit(f"kanji.json: {literal!r} has jlpt_new {new!r}, expected int 1-5") | |
| if len(literal) != 1: | |
| raise SystemExit(f"kanji.json: key {literal!r} is not a single character") | |
| label = f"N{new}" | |
| levels[literal] = label | |
| new_dist[label] += 1 | |
| return dict(sorted(levels.items())), new_dist, old_dist | |
| def check_kanji(new_dist: Counter[str], old_dist: Counter[int]) -> None: | |
| if dict(new_dist) != EXPECTED_KANJI or sum(new_dist.values()) != sum(EXPECTED_KANJI.values()): | |
| raise SystemExit( | |
| f"kanji jlpt_new distribution {dict(new_dist)} != {EXPECTED_KANJI}; nothing written" | |
| ) | |
| if dict(old_dist) != EXPECTED_KANJI_OLD: | |
| # Corroboration only, but a drift here means the file is not the one research read. | |
| raise SystemExit( | |
| f"kanji jlpt_old distribution {dict(old_dist)} != {EXPECTED_KANJI_OLD}; nothing written" | |
| ) | |
| def recorded_date(readme_text: str) -> str | None: | |
| m = re.search(r"^\*\*Resolved:\*\* (\d{4}-\d{2}-\d{2})", readme_text, re.M) | |
| return m.group(1) if m else None | |
| def recorded_hashes(readme_text: str) -> set[str]: | |
| return set(re.findall(r"`([0-9a-f]{64})`", readme_text)) | |
| def write_readme(files: list[dict], stats: dict, kanji_dist: Counter[str], date: str) -> None: | |
| lines = [ | |
| "# data/jlpt - JLPT vocabulary levels and kanji levels, pinned", | |
| "", | |
| "Generated by `scripts/build_jlpt.py`. **Do not hand-edit these files**:", | |
| "`tests/test_data_assets.py` compares every file with the SHA-256 below and pins the", | |
| "counts, so an edit fails the quick loop until this file is regenerated.", | |
| "", | |
| f"**Resolved:** {date} ", | |
| f"**Pins:** {YOMITAN_REPO} release tag `{YOMITAN_TAG}` (vocabulary); " | |
| f"{KANJI_DATA_REPO} commit `{KANJI_DATA_COMMIT}` (kanji levels)", | |
| "", | |
| "Regenerate (from the repo root; re-running is idempotent - byte-identical files):", | |
| "", | |
| "```", | |
| ".venv/Scripts/python.exe scripts/build_jlpt.py", | |
| "```", | |
| "", | |
| "## Files", | |
| "", | |
| "| File | Bytes | SHA256 | Source |", | |
| "|---|---|---|---|", | |
| ] | |
| for f in files: | |
| lines.append(f"| `{f['name']}` | {f['bytes']:,} | `{f['sha256']}` | {f['source']} |") | |
| lines += [ | |
| "", | |
| "The CSVs are the upstream files byte-for-byte (columns `jmdict_seq,kana,kanji,`", | |
| "`waller_definition`; `jmdict_seq` is the JMdict entry id, so word level is a join on", | |
| "id, not a string match). `kanji_levels.json` is this script's projection of", | |
| "the `jlpt_new` field of `kanji.json`: `{literal: N5..N1}` for every kanji that has a", | |
| "level, sorted by literal. `jlpt_new` is the N1-N5 scale; the file's `jlpt_old` is", | |
| "KANJIDIC2's pre-2010 1-4 scale and is NOT used (02-RESEARCH.md § Q2 - Kanji list).", | |
| "", | |
| "## Counts", | |
| "", | |
| "| Measure | Value |", | |
| "|---|---|", | |
| ] | |
| for level in LEVELS: | |
| lines.append(f"| `{level}.csv` data rows | {stats['rows'][level]:,} |") | |
| lines += [ | |
| f"| Total data rows | {stats['total_rows']:,} |", | |
| f"| Rows with an empty `jmdict_seq` (all in `n1.csv`) | {stats['empty_ids']} |", | |
| f"| Unique `jmdict_seq` ids | {stats['unique_ids']:,} |", | |
| f"| Ids appearing in more than one row | {stats['multi_row_ids']} |", | |
| f"| Ids appearing on more than one level | {stats['multi_level_ids']} |", | |
| f"| Ids listed twice within one level (two readings) | {stats['within_level_dup_ids']} |", | |
| ] | |
| for label in ("N5", "N4", "N3", "N2", "N1"): | |
| lines.append(f"| Kanji at {label} | {kanji_dist[label]:,} |") | |
| lines += [ | |
| f"| Kanji with a level | {sum(kanji_dist.values()):,} |", | |
| "", | |
| "A word on two lists takes the easiest level it appears at (research § Q2 - Join", | |
| "strategy); a word on no list is `N1+`; a kanji not in the map is unlisted and above", | |
| "every level (D-02, D-10).", | |
| "", | |
| "## Licence", | |
| "", | |
| LICENCE_TEXT, | |
| "", | |
| "- `LICENSE-yomitan-jlpt-vocab.txt` is the repository's `LICENSE.txt` at the tag (CC BY-SA", | |
| " 4.0). Any redistributed derivative of the CSVs stays CC BY-SA.", | |
| "- `LICENSE-kanji-data.txt` is the repository's `LICENSE` at the commit (MIT); the notice", | |
| " is kept beside the extracted file as MIT requires.", | |
| "- Jonathan Waller's terms (https://www.tanos.co.uk/jlpt/sharing/): use however you like,", | |
| " credit the site. `LICENSES.md` at the repo root is the project-wide record and the", | |
| " page's credits line carries the attribution (plan 02-11).", | |
| "", | |
| ] | |
| README.write_text("\n".join(lines), encoding="utf-8", newline="\n") | |
| def main() -> int: | |
| previous = README.read_text(encoding="utf-8") if README.exists() else "" | |
| # Fetch everything first; nothing is written until every check has passed. | |
| csv_bytes: dict[str, bytes] = {} | |
| for name, url in CSV_URLS.items(): | |
| csv_bytes[name] = fetch(url) | |
| print(f"{name}: {len(csv_bytes[name]):,} bytes from {url}") | |
| tables = {name[:-4]: parse_csv(data) for name, data in csv_bytes.items()} | |
| stats = csv_stats(tables) | |
| print( | |
| f"rows {stats['rows']} total {stats['total_rows']:,}; empty ids {stats['empty_ids']} " | |
| f"(levels {sorted(stats['empty_id_levels'])}); unique ids {stats['unique_ids']:,}; " | |
| f"in >1 row {stats['multi_row_ids']}; on >1 level {stats['multi_level_ids']}; " | |
| f"duplicated within a level {stats['within_level_dup_ids']}" | |
| ) | |
| check_csv_stats(stats) | |
| kanji_raw = fetch(KANJI_URL) | |
| print(f"kanji.json: {len(kanji_raw):,} bytes from {KANJI_URL}") | |
| kanji_levels, new_dist, old_dist = project_kanji(kanji_raw) | |
| print(f"jlpt_new: {dict(sorted(new_dist.items()))} = {sum(new_dist.values()):,}") | |
| print(f"jlpt_old (KANJIDIC2 corroboration): {dict(sorted(old_dist.items(), reverse=True))}") | |
| check_kanji(new_dist, old_dist) | |
| licence_bytes: dict[str, bytes] = {} | |
| for name, (url, phrases) in LICENSES.items(): | |
| data = fetch(url) | |
| text = data.decode("utf-8") | |
| if not any(p in text for p in phrases): | |
| raise SystemExit(f"{url} does not contain any of {phrases}:\n{text[:200]}") | |
| licence_bytes[name] = data | |
| print(f"{name}: {len(data):,} bytes, licence phrase found") | |
| OUT_DIR.mkdir(parents=True, exist_ok=True) | |
| files: list[dict] = [] | |
| for name, data in csv_bytes.items(): | |
| (OUT_DIR / name).write_bytes(data) | |
| files.append( | |
| {"name": name, "bytes": len(data), "sha256": sha256(data), "source": CSV_URLS[name]} | |
| ) | |
| kanji_out = (json.dumps(kanji_levels, ensure_ascii=False, indent=0) + "\n").encode("utf-8") | |
| (OUT_DIR / KANJI_OUT).write_bytes(kanji_out) | |
| files.append( | |
| { | |
| "name": KANJI_OUT, | |
| "bytes": len(kanji_out), | |
| "sha256": sha256(kanji_out), | |
| "source": f"`jlpt_new` of {KANJI_URL}", | |
| } | |
| ) | |
| for name, data in licence_bytes.items(): | |
| (OUT_DIR / name).write_bytes(data) | |
| files.append( | |
| {"name": name, "bytes": len(data), "sha256": sha256(data), "source": LICENSES[name][0]} | |
| ) | |
| for f in files: | |
| print(f" wrote data/jlpt/{f['name']} ({f['bytes']:,} bytes) {f['sha256']}") | |
| hashes_now = {f["sha256"] for f in files} | |
| date = recorded_date(previous) | |
| if date is None or not hashes_now <= recorded_hashes(previous): | |
| date = datetime.now(UTC).strftime("%Y-%m-%d") | |
| write_readme(files, stats, new_dist, date) | |
| print(f"README resolved date {date}; kanji-data commit {KANJI_DATA_COMMIT}") | |
| return 0 | |
| if __name__ == "__main__": | |
| sys.exit(main()) | |