japanese-learning-avatar / scripts /build_jmdict.py
WolfDavid's picture
feat(02-01): compact JMdict projection (LFS) with build script and asset guards
549c072
Raw History Blame
14.5 kB
"""Build data/jmdict/jmdict-compact.json.gz: the full JMdict, projected to what the card needs.
D-07 puts JMdict meanings on the lookup card. 02-RESEARCH.md § Q3 measured the options: the
raw ``jmdict-eng`` JSON is 117.8 MB and +804 MB RSS, the ``-common`` subset misses 538 of
the JLPT words, and a positional projection of the full file is ~7.7 MB gzipped, loads in
under 2 s and covers every entry - so every word the avatar or the learner produces gets a
gloss or an honest "no entry". This script builds that projection reproducibly:
1. Resolve the release asset through the GitHub release API for the pinned tag. The tag is
immutable; the API call only spares us guessing the URL (the resolved URL is recorded
in the README and the tgz's SHA-256 is pinned below, so the download is verified).
2. Download the tgz to a cache directory OUTSIDE the repo, check its SHA-256, and stream
the single ``.json`` member into ``json.load``. Refuse unless ``version``,
``dictDate`` and ``len(words)`` are exactly what research measured.
3. Project each word to ``[id, kanji_texts[:3], kana_texts[:3], senses[:3] x glosses[:3],
common]`` in file order - the index in the list is the JMdict order the ranked lookup
(plan 02-03) tie-breaks on.
4. Write ``{"meta": {...}, "entries": [...]}`` as compact UTF-8 JSON, gzip level 9 with a
zero mtime and no embedded name, so a re-run is byte-identical. ``*.gz`` is an LFS
pattern in ``.gitattributes``; the README hash lets the quick loop tell a real object
from an unfetched pointer.
5. Read the file back the way the app will (``gzip.open`` + ``json.load``), time it, and
refuse if the entry count differs; then write ``data/jmdict/README.md`` with the pins,
the SHA-256 row and the EDRDG attribution.
Run from the repo root::
.venv/Scripts/python.exe scripts/build_jmdict.py
Set ``JLA_DATA_CACHE`` to reuse an already-downloaded tgz (the download is 11.5 MB).
Stdlib only, on purpose.
"""
from __future__ import annotations
import gzip
import hashlib
import json
import os
import re
import sys
import tarfile
import tempfile
import time
import urllib.request
from datetime import UTC, datetime
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parent.parent
OUT_DIR = REPO_ROOT / "data" / "jmdict"
OUT = OUT_DIR / "jmdict-compact.json.gz"
README = OUT_DIR / "README.md"
SOURCE_REPO = "scriptin/jmdict-simplified"
RELEASE_TAG = "3.6.2+20260831182826"
# The JSON's own top-level `version` is the bare release version; the `+<build>` suffix is
# only in the tag and the asset name (measured on this asset - the plan's draft expected
# the full tag here and would have refused a correct file).
EXPECTED_VERSION = "3.6.2"
EXPECTED_DICT_DATE = "2026-08-31"
EXPECTED_ENTRIES = 218672
ASSET_RE = re.compile(r"^jmdict-eng-3\.6\.2\+20260831182826\.json\.tgz$")
# Resolved and measured on 2026-09-06; a re-run verifies the download against these.
EXPECTED_TGZ_BYTES = 11510336
EXPECTED_TGZ_SHA256 = "3c842741e2c4f1b780ad4aff6833df308866a8b7dfb9ee59a085e23dd201a65f"
RELEASE_API = (
f"https://api.github.com/repos/{SOURCE_REPO}/releases/tags/{RELEASE_TAG.replace('+', '%2B')}"
)
CACHE_DIR = (
Path(os.environ.get("JLA_DATA_CACHE", tempfile.gettempdir())) / "japanese-learning-avatar"
)
MAX_FORMS = 3 # kanji texts, kana texts, senses, glosses per sense
GLOSS_LANG = "eng"
EDRDG_TEXT = (
"Dictionary data: **JMdict** - Copyright (c) James William Breen and The Electronic "
"Dictionary Research and Development Group, used under the Creative Commons "
"Attribution-ShareAlike Licence (V4.0). "
"https://www.edrdg.org/wiki/index.php/JMdict-EDICT_Dictionary_Project . "
"https://www.edrdg.org/edrdg/licence.html"
)
PACKAGING_TEXT = (
f"JSON conversion by {SOURCE_REPO}, release {RELEASE_TAG}; its derived files carry the "
"EDRDG licence. The compact projection in this directory is a derivative of JMdict and "
"is itself CC BY-SA 4.0."
)
def fetch(url: str, accept: str | None = None) -> bytes:
headers = {"User-Agent": "japanese-learning-avatar data build"}
if accept:
headers["Accept"] = accept
req = urllib.request.Request(url, headers=headers)
with urllib.request.urlopen(req, timeout=600) as resp: # noqa: S310 - pinned https URLs
return resp.read()
def sha256(data: bytes) -> str:
return hashlib.sha256(data).hexdigest()
def resolve_asset() -> tuple[str, str, int]:
"""(name, browser_download_url, size) of the ONE asset matching ASSET_RE."""
release = json.loads(fetch(RELEASE_API, accept="application/vnd.github+json"))
if release.get("tag_name") != RELEASE_TAG:
raise SystemExit(f"release API returned tag {release.get('tag_name')!r}, not {RELEASE_TAG}")
matches = [a for a in release["assets"] if ASSET_RE.match(a["name"])]
if len(matches) != 1:
names = [a["name"] for a in release["assets"]]
raise SystemExit(f"expected exactly one asset matching {ASSET_RE.pattern}; got {names}")
asset = matches[0]
print(f"asset {asset['name']} ({asset['size']:,} bytes)\n {asset['browser_download_url']}")
return asset["name"], asset["browser_download_url"], int(asset["size"])
def download_tgz(name: str, url: str) -> tuple[Path, bytes]:
CACHE_DIR.mkdir(parents=True, exist_ok=True)
path = CACHE_DIR / name
if path.is_file() and path.stat().st_size == EXPECTED_TGZ_BYTES:
data = path.read_bytes()
print(f"reusing cached {path}")
else:
t = time.perf_counter()
data = fetch(url)
path.write_bytes(data)
print(f"downloaded {len(data):,} bytes in {time.perf_counter() - t:.1f} s to {path}")
digest = sha256(data)
if len(data) != EXPECTED_TGZ_BYTES or digest != EXPECTED_TGZ_SHA256:
raise SystemExit(
f"{name}: {len(data):,} bytes sha256 {digest}; expected {EXPECTED_TGZ_BYTES:,} / "
f"{EXPECTED_TGZ_SHA256}. The asset is not the one research measured; nothing written."
)
return path, data
def load_words(tgz: Path) -> dict:
with tarfile.open(tgz) as tf:
members = [m for m in tf.getmembers() if m.name.endswith(".json")]
if len(members) != 1:
raise SystemExit(
f"{tgz.name}: expected one .json member, got {[m.name for m in members]}"
)
stream = tf.extractfile(members[0])
assert stream is not None
t = time.perf_counter()
data = json.load(stream)
seconds = time.perf_counter() - t
print(f"json.load {members[0].name} ({members[0].size:,} bytes) in {seconds:.2f} s")
problems = []
if data.get("version") != EXPECTED_VERSION:
problems.append(f"version {data.get('version')!r} != {EXPECTED_VERSION!r}")
if data.get("dictDate") != EXPECTED_DICT_DATE:
problems.append(f"dictDate {data.get('dictDate')!r} != {EXPECTED_DICT_DATE!r}")
if len(data.get("words", [])) != EXPECTED_ENTRIES:
problems.append(f"len(words) {len(data.get('words', []))} != {EXPECTED_ENTRIES}")
if problems:
raise SystemExit(
"jmdict-eng JSON is not the pinned file; nothing written:\n " + "\n ".join(problems)
)
return data
def project(word: dict) -> list:
"""``[id, kanji_texts, kana_texts, senses, common]`` - positional so the file stays small."""
kanji = [k["text"] for k in word["kanji"]][:MAX_FORMS]
kana = [k["text"] for k in word["kana"]][:MAX_FORMS]
senses: list[list[str]] = []
for sense in word["sense"]:
glosses = [g["text"] for g in sense["gloss"] if g.get("lang") == GLOSS_LANG][:MAX_FORMS]
if glosses:
senses.append(glosses)
if len(senses) == MAX_FORMS:
break
common = int(any(k.get("common") for k in (*word["kanji"], *word["kana"])))
return [int(word["id"]), kanji, kana, senses, common]
def write_compact(entries: list[list]) -> tuple[int, int]:
"""Write OUT deterministically; returns (raw_bytes, gz_bytes)."""
payload = {
"meta": {
"source": SOURCE_REPO,
"version": RELEASE_TAG,
"jmdictVersion": EXPECTED_VERSION,
"dictDate": EXPECTED_DICT_DATE,
"entries": len(entries),
"licence": "CC BY-SA 4.0 (EDRDG)",
},
"entries": entries,
}
raw = json.dumps(payload, ensure_ascii=False, separators=(",", ":")).encode("utf-8")
OUT_DIR.mkdir(parents=True, exist_ok=True)
with (
open(OUT, "wb") as fh,
gzip.GzipFile(filename="", mode="wb", fileobj=fh, compresslevel=9, mtime=0) as gz,
):
gz.write(raw)
return len(raw), OUT.stat().st_size
def read_back() -> tuple[dict, float]:
t = time.perf_counter()
with gzip.open(OUT, "rt", encoding="utf-8") as fh:
data = json.load(fh)
return data, time.perf_counter() - t
def recorded_date(readme_text: str) -> str | None:
m = re.search(r"^\*\*Resolved:\*\* (\d{4}-\d{2}-\d{2})", readme_text, re.M)
return m.group(1) if m else None
def recorded_hashes(readme_text: str) -> set[str]:
return set(re.findall(r"`([0-9a-f]{64})`", readme_text))
def write_readme(
*,
date: str,
asset_url: str,
tgz_bytes: int,
tgz_sha: str,
raw_bytes: int,
gz_bytes: int,
gz_sha: str,
entries: int,
no_kanji: int,
no_sense: int,
common: int,
) -> None:
lines = [
"# data/jmdict - the compact JMdict projection, pinned",
"",
"Generated by `scripts/build_jmdict.py`. **Do not hand-edit or re-compress this file**:",
"`tests/test_data_assets.py` compares it with the SHA-256 below and pins the entry count,",
"so any change fails the quick loop until this file is regenerated. The file is a Git",
"LFS object (`*.gz` in `.gitattributes`); if it starts with `version https://git-lfs`",
"instead of the gzip magic, run `git lfs pull`.",
"",
f"**Resolved:** {date} ",
f"**Pins:** {SOURCE_REPO} release `{RELEASE_TAG}` (JSON `version` `{EXPECTED_VERSION}`, "
f"`dictDate` `{EXPECTED_DICT_DATE}`), {EXPECTED_ENTRIES:,} words ",
f"**Input:** `{asset_url}` ",
f"**Input tgz:** {tgz_bytes:,} bytes, SHA-256 `{tgz_sha}`",
"",
"Regenerate (from the repo root; re-running is idempotent - byte-identical output, the",
"gzip header carries no timestamp or name):",
"",
"```",
".venv/Scripts/python.exe scripts/build_jmdict.py",
"```",
"",
"## Files",
"",
"| File | Bytes | SHA256 | Source |",
"|---|---|---|---|",
f"| `{OUT.name}` | {gz_bytes:,} | `{gz_sha}` | projection of the tgz above |",
"",
f"Raw JSON {raw_bytes:,} bytes -> gzip level 9 {gz_bytes:,} bytes. Loading is one",
"`gzip.open` + `json.load` (~0.6-2 s depending on the machine; the build prints the",
"measured time and `tests/test_data_assets.py`'s fixture prints it again) - kept out of",
"this file so a re-run stays byte-identical.",
"",
"## Compact record schema",
"",
"```",
'{"meta": {"source", "version", "jmdictVersion", "dictDate", "entries", "licence"},',
' "entries": [[id, kanji_texts, kana_texts, senses, common], ...]}',
"",
"id int JMdict entry sequence number (the `jmdict_seq` of data/jlpt/*.csv)",
f"kanji_texts [str] up to {MAX_FORMS} kanji forms, file order (empty for kana-only words)",
f"kana_texts [str] up to {MAX_FORMS} kana forms, file order",
f"senses [[str]] up to {MAX_FORMS} senses x up to {MAX_FORMS} English glosses each;",
" senses with no English gloss are dropped",
"common 0|1 1 if any kanji or kana form is marked common",
"```",
"",
"Entries are in JMdict file order; the index in `entries` is the order the ranked lookup",
"(plan 02-03) tie-breaks on. Counts in this build: "
f"{entries:,} entries, {common:,} common, {no_kanji:,} kana-only, "
f"{no_sense:,} with no English sense.",
"",
"## Licence",
"",
EDRDG_TEXT,
"",
PACKAGING_TEXT,
"",
"`LICENSES.md` at the repo root is the project-wide record and the page's credits line",
"carries the attribution (plan 02-11).",
"",
]
README.write_text("\n".join(lines), encoding="utf-8", newline="\n")
def main() -> int:
previous = README.read_text(encoding="utf-8") if README.exists() else ""
name, url, _size = resolve_asset()
tgz, tgz_data = download_tgz(name, url)
tgz_sha = sha256(tgz_data)
data = load_words(tgz)
print(f"version {data['version']} dictDate {data['dictDate']} words {len(data['words']):,}")
entries = [project(w) for w in data["words"]]
del data
ids = [e[0] for e in entries]
if len(set(ids)) != len(ids) or any(i <= 0 for i in ids):
raise SystemExit("entry ids are not unique positive ints; nothing written")
no_kanji = sum(1 for e in entries if not e[1])
no_sense = sum(1 for e in entries if not e[3])
common = sum(e[4] for e in entries)
if any(not e[2] for e in entries):
raise SystemExit("an entry has no kana form; JMdict guarantees one - nothing written")
raw_bytes, gz_bytes = write_compact(entries)
print(f"wrote {OUT.relative_to(REPO_ROOT)}: raw {raw_bytes:,} bytes -> gz {gz_bytes:,} bytes")
back, seconds = read_back()
print(f"read-back: {len(back['entries']):,} entries in {seconds:.2f} s")
if len(back["entries"]) != len(entries) or back["meta"]["entries"] != EXPECTED_ENTRIES:
OUT.unlink()
raise SystemExit("read-back entry count differs from what was written; file removed")
gz_sha = sha256(OUT.read_bytes())
date = recorded_date(previous)
if date is None or gz_sha not in recorded_hashes(previous):
date = datetime.now(UTC).strftime("%Y-%m-%d")
write_readme(
date=date,
asset_url=url,
tgz_bytes=len(tgz_data),
tgz_sha=tgz_sha,
raw_bytes=raw_bytes,
gz_bytes=gz_bytes,
gz_sha=gz_sha,
entries=len(entries),
no_kanji=no_kanji,
no_sense=no_sense,
common=common,
)
print(f"README resolved date {date}; gz sha256 {gz_sha}")
return 0
if __name__ == "__main__":
sys.exit(main())