Spaces:
Running on Zero
Running on Zero
Download scripts/build_jmdict.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 14.5 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/fbd985ff7754c8ea5773c2af17e7023a7b8e24fd/scripts/build_jmdict.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@fbd985ff7754c8ea5773c2af17e7023a7b8e24fd/scripts/build_jmdict.py
-
curl -L -o build_jmdict.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/fbd985ff7754c8ea5773c2af17e7023a7b8e24fd/scripts/build_jmdict.py
14.5 kB
| """Build data/jmdict/jmdict-compact.json.gz: the full JMdict, projected to what the card needs. | |
| D-07 puts JMdict meanings on the lookup card. 02-RESEARCH.md § Q3 measured the options: the | |
| raw ``jmdict-eng`` JSON is 117.8 MB and +804 MB RSS, the ``-common`` subset misses 538 of | |
| the JLPT words, and a positional projection of the full file is ~7.7 MB gzipped, loads in | |
| under 2 s and covers every entry - so every word the avatar or the learner produces gets a | |
| gloss or an honest "no entry". This script builds that projection reproducibly: | |
| 1. Resolve the release asset through the GitHub release API for the pinned tag. The tag is | |
| immutable; the API call only spares us guessing the URL (the resolved URL is recorded | |
| in the README and the tgz's SHA-256 is pinned below, so the download is verified). | |
| 2. Download the tgz to a cache directory OUTSIDE the repo, check its SHA-256, and stream | |
| the single ``.json`` member into ``json.load``. Refuse unless ``version``, | |
| ``dictDate`` and ``len(words)`` are exactly what research measured. | |
| 3. Project each word to ``[id, kanji_texts[:3], kana_texts[:3], senses[:3] x glosses[:3], | |
| common]`` in file order - the index in the list is the JMdict order the ranked lookup | |
| (plan 02-03) tie-breaks on. | |
| 4. Write ``{"meta": {...}, "entries": [...]}`` as compact UTF-8 JSON, gzip level 9 with a | |
| zero mtime and no embedded name, so a re-run is byte-identical. ``*.gz`` is an LFS | |
| pattern in ``.gitattributes``; the README hash lets the quick loop tell a real object | |
| from an unfetched pointer. | |
| 5. Read the file back the way the app will (``gzip.open`` + ``json.load``), time it, and | |
| refuse if the entry count differs; then write ``data/jmdict/README.md`` with the pins, | |
| the SHA-256 row and the EDRDG attribution. | |
| Run from the repo root:: | |
| .venv/Scripts/python.exe scripts/build_jmdict.py | |
| Set ``JLA_DATA_CACHE`` to reuse an already-downloaded tgz (the download is 11.5 MB). | |
| Stdlib only, on purpose. | |
| """ | |
| from __future__ import annotations | |
| import gzip | |
| import hashlib | |
| import json | |
| import os | |
| import re | |
| import sys | |
| import tarfile | |
| import tempfile | |
| import time | |
| import urllib.request | |
| from datetime import UTC, datetime | |
| from pathlib import Path | |
| REPO_ROOT = Path(__file__).resolve().parent.parent | |
| OUT_DIR = REPO_ROOT / "data" / "jmdict" | |
| OUT = OUT_DIR / "jmdict-compact.json.gz" | |
| README = OUT_DIR / "README.md" | |
| SOURCE_REPO = "scriptin/jmdict-simplified" | |
| RELEASE_TAG = "3.6.2+20260831182826" | |
| # The JSON's own top-level `version` is the bare release version; the `+<build>` suffix is | |
| # only in the tag and the asset name (measured on this asset - the plan's draft expected | |
| # the full tag here and would have refused a correct file). | |
| EXPECTED_VERSION = "3.6.2" | |
| EXPECTED_DICT_DATE = "2026-08-31" | |
| EXPECTED_ENTRIES = 218672 | |
| ASSET_RE = re.compile(r"^jmdict-eng-3\.6\.2\+20260831182826\.json\.tgz$") | |
| # Resolved and measured on 2026-09-06; a re-run verifies the download against these. | |
| EXPECTED_TGZ_BYTES = 11510336 | |
| EXPECTED_TGZ_SHA256 = "3c842741e2c4f1b780ad4aff6833df308866a8b7dfb9ee59a085e23dd201a65f" | |
| RELEASE_API = ( | |
| f"https://api.github.com/repos/{SOURCE_REPO}/releases/tags/{RELEASE_TAG.replace('+', '%2B')}" | |
| ) | |
| CACHE_DIR = ( | |
| Path(os.environ.get("JLA_DATA_CACHE", tempfile.gettempdir())) / "japanese-learning-avatar" | |
| ) | |
| MAX_FORMS = 3 # kanji texts, kana texts, senses, glosses per sense | |
| GLOSS_LANG = "eng" | |
| EDRDG_TEXT = ( | |
| "Dictionary data: **JMdict** - Copyright (c) James William Breen and The Electronic " | |
| "Dictionary Research and Development Group, used under the Creative Commons " | |
| "Attribution-ShareAlike Licence (V4.0). " | |
| "https://www.edrdg.org/wiki/index.php/JMdict-EDICT_Dictionary_Project . " | |
| "https://www.edrdg.org/edrdg/licence.html" | |
| ) | |
| PACKAGING_TEXT = ( | |
| f"JSON conversion by {SOURCE_REPO}, release {RELEASE_TAG}; its derived files carry the " | |
| "EDRDG licence. The compact projection in this directory is a derivative of JMdict and " | |
| "is itself CC BY-SA 4.0." | |
| ) | |
| def fetch(url: str, accept: str | None = None) -> bytes: | |
| headers = {"User-Agent": "japanese-learning-avatar data build"} | |
| if accept: | |
| headers["Accept"] = accept | |
| req = urllib.request.Request(url, headers=headers) | |
| with urllib.request.urlopen(req, timeout=600) as resp: # noqa: S310 - pinned https URLs | |
| return resp.read() | |
| def sha256(data: bytes) -> str: | |
| return hashlib.sha256(data).hexdigest() | |
| def resolve_asset() -> tuple[str, str, int]: | |
| """(name, browser_download_url, size) of the ONE asset matching ASSET_RE.""" | |
| release = json.loads(fetch(RELEASE_API, accept="application/vnd.github+json")) | |
| if release.get("tag_name") != RELEASE_TAG: | |
| raise SystemExit(f"release API returned tag {release.get('tag_name')!r}, not {RELEASE_TAG}") | |
| matches = [a for a in release["assets"] if ASSET_RE.match(a["name"])] | |
| if len(matches) != 1: | |
| names = [a["name"] for a in release["assets"]] | |
| raise SystemExit(f"expected exactly one asset matching {ASSET_RE.pattern}; got {names}") | |
| asset = matches[0] | |
| print(f"asset {asset['name']} ({asset['size']:,} bytes)\n {asset['browser_download_url']}") | |
| return asset["name"], asset["browser_download_url"], int(asset["size"]) | |
| def download_tgz(name: str, url: str) -> tuple[Path, bytes]: | |
| CACHE_DIR.mkdir(parents=True, exist_ok=True) | |
| path = CACHE_DIR / name | |
| if path.is_file() and path.stat().st_size == EXPECTED_TGZ_BYTES: | |
| data = path.read_bytes() | |
| print(f"reusing cached {path}") | |
| else: | |
| t = time.perf_counter() | |
| data = fetch(url) | |
| path.write_bytes(data) | |
| print(f"downloaded {len(data):,} bytes in {time.perf_counter() - t:.1f} s to {path}") | |
| digest = sha256(data) | |
| if len(data) != EXPECTED_TGZ_BYTES or digest != EXPECTED_TGZ_SHA256: | |
| raise SystemExit( | |
| f"{name}: {len(data):,} bytes sha256 {digest}; expected {EXPECTED_TGZ_BYTES:,} / " | |
| f"{EXPECTED_TGZ_SHA256}. The asset is not the one research measured; nothing written." | |
| ) | |
| return path, data | |
| def load_words(tgz: Path) -> dict: | |
| with tarfile.open(tgz) as tf: | |
| members = [m for m in tf.getmembers() if m.name.endswith(".json")] | |
| if len(members) != 1: | |
| raise SystemExit( | |
| f"{tgz.name}: expected one .json member, got {[m.name for m in members]}" | |
| ) | |
| stream = tf.extractfile(members[0]) | |
| assert stream is not None | |
| t = time.perf_counter() | |
| data = json.load(stream) | |
| seconds = time.perf_counter() - t | |
| print(f"json.load {members[0].name} ({members[0].size:,} bytes) in {seconds:.2f} s") | |
| problems = [] | |
| if data.get("version") != EXPECTED_VERSION: | |
| problems.append(f"version {data.get('version')!r} != {EXPECTED_VERSION!r}") | |
| if data.get("dictDate") != EXPECTED_DICT_DATE: | |
| problems.append(f"dictDate {data.get('dictDate')!r} != {EXPECTED_DICT_DATE!r}") | |
| if len(data.get("words", [])) != EXPECTED_ENTRIES: | |
| problems.append(f"len(words) {len(data.get('words', []))} != {EXPECTED_ENTRIES}") | |
| if problems: | |
| raise SystemExit( | |
| "jmdict-eng JSON is not the pinned file; nothing written:\n " + "\n ".join(problems) | |
| ) | |
| return data | |
| def project(word: dict) -> list: | |
| """``[id, kanji_texts, kana_texts, senses, common]`` - positional so the file stays small.""" | |
| kanji = [k["text"] for k in word["kanji"]][:MAX_FORMS] | |
| kana = [k["text"] for k in word["kana"]][:MAX_FORMS] | |
| senses: list[list[str]] = [] | |
| for sense in word["sense"]: | |
| glosses = [g["text"] for g in sense["gloss"] if g.get("lang") == GLOSS_LANG][:MAX_FORMS] | |
| if glosses: | |
| senses.append(glosses) | |
| if len(senses) == MAX_FORMS: | |
| break | |
| common = int(any(k.get("common") for k in (*word["kanji"], *word["kana"]))) | |
| return [int(word["id"]), kanji, kana, senses, common] | |
| def write_compact(entries: list[list]) -> tuple[int, int]: | |
| """Write OUT deterministically; returns (raw_bytes, gz_bytes).""" | |
| payload = { | |
| "meta": { | |
| "source": SOURCE_REPO, | |
| "version": RELEASE_TAG, | |
| "jmdictVersion": EXPECTED_VERSION, | |
| "dictDate": EXPECTED_DICT_DATE, | |
| "entries": len(entries), | |
| "licence": "CC BY-SA 4.0 (EDRDG)", | |
| }, | |
| "entries": entries, | |
| } | |
| raw = json.dumps(payload, ensure_ascii=False, separators=(",", ":")).encode("utf-8") | |
| OUT_DIR.mkdir(parents=True, exist_ok=True) | |
| with ( | |
| open(OUT, "wb") as fh, | |
| gzip.GzipFile(filename="", mode="wb", fileobj=fh, compresslevel=9, mtime=0) as gz, | |
| ): | |
| gz.write(raw) | |
| return len(raw), OUT.stat().st_size | |
| def read_back() -> tuple[dict, float]: | |
| t = time.perf_counter() | |
| with gzip.open(OUT, "rt", encoding="utf-8") as fh: | |
| data = json.load(fh) | |
| return data, time.perf_counter() - t | |
| def recorded_date(readme_text: str) -> str | None: | |
| m = re.search(r"^\*\*Resolved:\*\* (\d{4}-\d{2}-\d{2})", readme_text, re.M) | |
| return m.group(1) if m else None | |
| def recorded_hashes(readme_text: str) -> set[str]: | |
| return set(re.findall(r"`([0-9a-f]{64})`", readme_text)) | |
| def write_readme( | |
| *, | |
| date: str, | |
| asset_url: str, | |
| tgz_bytes: int, | |
| tgz_sha: str, | |
| raw_bytes: int, | |
| gz_bytes: int, | |
| gz_sha: str, | |
| entries: int, | |
| no_kanji: int, | |
| no_sense: int, | |
| common: int, | |
| ) -> None: | |
| lines = [ | |
| "# data/jmdict - the compact JMdict projection, pinned", | |
| "", | |
| "Generated by `scripts/build_jmdict.py`. **Do not hand-edit or re-compress this file**:", | |
| "`tests/test_data_assets.py` compares it with the SHA-256 below and pins the entry count,", | |
| "so any change fails the quick loop until this file is regenerated. The file is a Git", | |
| "LFS object (`*.gz` in `.gitattributes`); if it starts with `version https://git-lfs`", | |
| "instead of the gzip magic, run `git lfs pull`.", | |
| "", | |
| f"**Resolved:** {date} ", | |
| f"**Pins:** {SOURCE_REPO} release `{RELEASE_TAG}` (JSON `version` `{EXPECTED_VERSION}`, " | |
| f"`dictDate` `{EXPECTED_DICT_DATE}`), {EXPECTED_ENTRIES:,} words ", | |
| f"**Input:** `{asset_url}` ", | |
| f"**Input tgz:** {tgz_bytes:,} bytes, SHA-256 `{tgz_sha}`", | |
| "", | |
| "Regenerate (from the repo root; re-running is idempotent - byte-identical output, the", | |
| "gzip header carries no timestamp or name):", | |
| "", | |
| "```", | |
| ".venv/Scripts/python.exe scripts/build_jmdict.py", | |
| "```", | |
| "", | |
| "## Files", | |
| "", | |
| "| File | Bytes | SHA256 | Source |", | |
| "|---|---|---|---|", | |
| f"| `{OUT.name}` | {gz_bytes:,} | `{gz_sha}` | projection of the tgz above |", | |
| "", | |
| f"Raw JSON {raw_bytes:,} bytes -> gzip level 9 {gz_bytes:,} bytes. Loading is one", | |
| "`gzip.open` + `json.load` (~0.6-2 s depending on the machine; the build prints the", | |
| "measured time and `tests/test_data_assets.py`'s fixture prints it again) - kept out of", | |
| "this file so a re-run stays byte-identical.", | |
| "", | |
| "## Compact record schema", | |
| "", | |
| "```", | |
| '{"meta": {"source", "version", "jmdictVersion", "dictDate", "entries", "licence"},', | |
| ' "entries": [[id, kanji_texts, kana_texts, senses, common], ...]}', | |
| "", | |
| "id int JMdict entry sequence number (the `jmdict_seq` of data/jlpt/*.csv)", | |
| f"kanji_texts [str] up to {MAX_FORMS} kanji forms, file order (empty for kana-only words)", | |
| f"kana_texts [str] up to {MAX_FORMS} kana forms, file order", | |
| f"senses [[str]] up to {MAX_FORMS} senses x up to {MAX_FORMS} English glosses each;", | |
| " senses with no English gloss are dropped", | |
| "common 0|1 1 if any kanji or kana form is marked common", | |
| "```", | |
| "", | |
| "Entries are in JMdict file order; the index in `entries` is the order the ranked lookup", | |
| "(plan 02-03) tie-breaks on. Counts in this build: " | |
| f"{entries:,} entries, {common:,} common, {no_kanji:,} kana-only, " | |
| f"{no_sense:,} with no English sense.", | |
| "", | |
| "## Licence", | |
| "", | |
| EDRDG_TEXT, | |
| "", | |
| PACKAGING_TEXT, | |
| "", | |
| "`LICENSES.md` at the repo root is the project-wide record and the page's credits line", | |
| "carries the attribution (plan 02-11).", | |
| "", | |
| ] | |
| README.write_text("\n".join(lines), encoding="utf-8", newline="\n") | |
| def main() -> int: | |
| previous = README.read_text(encoding="utf-8") if README.exists() else "" | |
| name, url, _size = resolve_asset() | |
| tgz, tgz_data = download_tgz(name, url) | |
| tgz_sha = sha256(tgz_data) | |
| data = load_words(tgz) | |
| print(f"version {data['version']} dictDate {data['dictDate']} words {len(data['words']):,}") | |
| entries = [project(w) for w in data["words"]] | |
| del data | |
| ids = [e[0] for e in entries] | |
| if len(set(ids)) != len(ids) or any(i <= 0 for i in ids): | |
| raise SystemExit("entry ids are not unique positive ints; nothing written") | |
| no_kanji = sum(1 for e in entries if not e[1]) | |
| no_sense = sum(1 for e in entries if not e[3]) | |
| common = sum(e[4] for e in entries) | |
| if any(not e[2] for e in entries): | |
| raise SystemExit("an entry has no kana form; JMdict guarantees one - nothing written") | |
| raw_bytes, gz_bytes = write_compact(entries) | |
| print(f"wrote {OUT.relative_to(REPO_ROOT)}: raw {raw_bytes:,} bytes -> gz {gz_bytes:,} bytes") | |
| back, seconds = read_back() | |
| print(f"read-back: {len(back['entries']):,} entries in {seconds:.2f} s") | |
| if len(back["entries"]) != len(entries) or back["meta"]["entries"] != EXPECTED_ENTRIES: | |
| OUT.unlink() | |
| raise SystemExit("read-back entry count differs from what was written; file removed") | |
| gz_sha = sha256(OUT.read_bytes()) | |
| date = recorded_date(previous) | |
| if date is None or gz_sha not in recorded_hashes(previous): | |
| date = datetime.now(UTC).strftime("%Y-%m-%d") | |
| write_readme( | |
| date=date, | |
| asset_url=url, | |
| tgz_bytes=len(tgz_data), | |
| tgz_sha=tgz_sha, | |
| raw_bytes=raw_bytes, | |
| gz_bytes=gz_bytes, | |
| gz_sha=gz_sha, | |
| entries=len(entries), | |
| no_kanji=no_kanji, | |
| no_sense=no_sense, | |
| common=common, | |
| ) | |
| print(f"README resolved date {date}; gz sha256 {gz_sha}") | |
| return 0 | |
| if __name__ == "__main__": | |
| sys.exit(main()) | |