Spaces:
Running on Zero
Running on Zero
File size: 14,467 Bytes
549c072 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 292 293 294 295 296 297 298 299 300 301 302 303 304 305 306 307 308 309 310 311 312 313 314 315 316 317 318 319 320 321 322 323 324 325 326 327 328 329 330 331 332 333 334 335 336 337 338 339 340 341 342 343 344 345 346 347 348 349 350 | """Build data/jmdict/jmdict-compact.json.gz: the full JMdict, projected to what the card needs.
D-07 puts JMdict meanings on the lookup card. 02-RESEARCH.md § Q3 measured the options: the
raw ``jmdict-eng`` JSON is 117.8 MB and +804 MB RSS, the ``-common`` subset misses 538 of
the JLPT words, and a positional projection of the full file is ~7.7 MB gzipped, loads in
under 2 s and covers every entry - so every word the avatar or the learner produces gets a
gloss or an honest "no entry". This script builds that projection reproducibly:
1. Resolve the release asset through the GitHub release API for the pinned tag. The tag is
immutable; the API call only spares us guessing the URL (the resolved URL is recorded
in the README and the tgz's SHA-256 is pinned below, so the download is verified).
2. Download the tgz to a cache directory OUTSIDE the repo, check its SHA-256, and stream
the single ``.json`` member into ``json.load``. Refuse unless ``version``,
``dictDate`` and ``len(words)`` are exactly what research measured.
3. Project each word to ``[id, kanji_texts[:3], kana_texts[:3], senses[:3] x glosses[:3],
common]`` in file order - the index in the list is the JMdict order the ranked lookup
(plan 02-03) tie-breaks on.
4. Write ``{"meta": {...}, "entries": [...]}`` as compact UTF-8 JSON, gzip level 9 with a
zero mtime and no embedded name, so a re-run is byte-identical. ``*.gz`` is an LFS
pattern in ``.gitattributes``; the README hash lets the quick loop tell a real object
from an unfetched pointer.
5. Read the file back the way the app will (``gzip.open`` + ``json.load``), time it, and
refuse if the entry count differs; then write ``data/jmdict/README.md`` with the pins,
the SHA-256 row and the EDRDG attribution.
Run from the repo root::
.venv/Scripts/python.exe scripts/build_jmdict.py
Set ``JLA_DATA_CACHE`` to reuse an already-downloaded tgz (the download is 11.5 MB).
Stdlib only, on purpose.
"""
from __future__ import annotations
import gzip
import hashlib
import json
import os
import re
import sys
import tarfile
import tempfile
import time
import urllib.request
from datetime import UTC, datetime
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parent.parent
OUT_DIR = REPO_ROOT / "data" / "jmdict"
OUT = OUT_DIR / "jmdict-compact.json.gz"
README = OUT_DIR / "README.md"
SOURCE_REPO = "scriptin/jmdict-simplified"
RELEASE_TAG = "3.6.2+20260831182826"
# The JSON's own top-level `version` is the bare release version; the `+<build>` suffix is
# only in the tag and the asset name (measured on this asset - the plan's draft expected
# the full tag here and would have refused a correct file).
EXPECTED_VERSION = "3.6.2"
EXPECTED_DICT_DATE = "2026-08-31"
EXPECTED_ENTRIES = 218672
ASSET_RE = re.compile(r"^jmdict-eng-3\.6\.2\+20260831182826\.json\.tgz$")
# Resolved and measured on 2026-09-06; a re-run verifies the download against these.
EXPECTED_TGZ_BYTES = 11510336
EXPECTED_TGZ_SHA256 = "3c842741e2c4f1b780ad4aff6833df308866a8b7dfb9ee59a085e23dd201a65f"
RELEASE_API = (
f"https://api.github.com/repos/{SOURCE_REPO}/releases/tags/{RELEASE_TAG.replace('+', '%2B')}"
)
CACHE_DIR = (
Path(os.environ.get("JLA_DATA_CACHE", tempfile.gettempdir())) / "japanese-learning-avatar"
)
MAX_FORMS = 3 # kanji texts, kana texts, senses, glosses per sense
GLOSS_LANG = "eng"
EDRDG_TEXT = (
"Dictionary data: **JMdict** - Copyright (c) James William Breen and The Electronic "
"Dictionary Research and Development Group, used under the Creative Commons "
"Attribution-ShareAlike Licence (V4.0). "
"https://www.edrdg.org/wiki/index.php/JMdict-EDICT_Dictionary_Project . "
"https://www.edrdg.org/edrdg/licence.html"
)
PACKAGING_TEXT = (
f"JSON conversion by {SOURCE_REPO}, release {RELEASE_TAG}; its derived files carry the "
"EDRDG licence. The compact projection in this directory is a derivative of JMdict and "
"is itself CC BY-SA 4.0."
)
def fetch(url: str, accept: str | None = None) -> bytes:
headers = {"User-Agent": "japanese-learning-avatar data build"}
if accept:
headers["Accept"] = accept
req = urllib.request.Request(url, headers=headers)
with urllib.request.urlopen(req, timeout=600) as resp: # noqa: S310 - pinned https URLs
return resp.read()
def sha256(data: bytes) -> str:
return hashlib.sha256(data).hexdigest()
def resolve_asset() -> tuple[str, str, int]:
"""(name, browser_download_url, size) of the ONE asset matching ASSET_RE."""
release = json.loads(fetch(RELEASE_API, accept="application/vnd.github+json"))
if release.get("tag_name") != RELEASE_TAG:
raise SystemExit(f"release API returned tag {release.get('tag_name')!r}, not {RELEASE_TAG}")
matches = [a for a in release["assets"] if ASSET_RE.match(a["name"])]
if len(matches) != 1:
names = [a["name"] for a in release["assets"]]
raise SystemExit(f"expected exactly one asset matching {ASSET_RE.pattern}; got {names}")
asset = matches[0]
print(f"asset {asset['name']} ({asset['size']:,} bytes)\n {asset['browser_download_url']}")
return asset["name"], asset["browser_download_url"], int(asset["size"])
def download_tgz(name: str, url: str) -> tuple[Path, bytes]:
CACHE_DIR.mkdir(parents=True, exist_ok=True)
path = CACHE_DIR / name
if path.is_file() and path.stat().st_size == EXPECTED_TGZ_BYTES:
data = path.read_bytes()
print(f"reusing cached {path}")
else:
t = time.perf_counter()
data = fetch(url)
path.write_bytes(data)
print(f"downloaded {len(data):,} bytes in {time.perf_counter() - t:.1f} s to {path}")
digest = sha256(data)
if len(data) != EXPECTED_TGZ_BYTES or digest != EXPECTED_TGZ_SHA256:
raise SystemExit(
f"{name}: {len(data):,} bytes sha256 {digest}; expected {EXPECTED_TGZ_BYTES:,} / "
f"{EXPECTED_TGZ_SHA256}. The asset is not the one research measured; nothing written."
)
return path, data
def load_words(tgz: Path) -> dict:
with tarfile.open(tgz) as tf:
members = [m for m in tf.getmembers() if m.name.endswith(".json")]
if len(members) != 1:
raise SystemExit(
f"{tgz.name}: expected one .json member, got {[m.name for m in members]}"
)
stream = tf.extractfile(members[0])
assert stream is not None
t = time.perf_counter()
data = json.load(stream)
seconds = time.perf_counter() - t
print(f"json.load {members[0].name} ({members[0].size:,} bytes) in {seconds:.2f} s")
problems = []
if data.get("version") != EXPECTED_VERSION:
problems.append(f"version {data.get('version')!r} != {EXPECTED_VERSION!r}")
if data.get("dictDate") != EXPECTED_DICT_DATE:
problems.append(f"dictDate {data.get('dictDate')!r} != {EXPECTED_DICT_DATE!r}")
if len(data.get("words", [])) != EXPECTED_ENTRIES:
problems.append(f"len(words) {len(data.get('words', []))} != {EXPECTED_ENTRIES}")
if problems:
raise SystemExit(
"jmdict-eng JSON is not the pinned file; nothing written:\n " + "\n ".join(problems)
)
return data
def project(word: dict) -> list:
"""``[id, kanji_texts, kana_texts, senses, common]`` - positional so the file stays small."""
kanji = [k["text"] for k in word["kanji"]][:MAX_FORMS]
kana = [k["text"] for k in word["kana"]][:MAX_FORMS]
senses: list[list[str]] = []
for sense in word["sense"]:
glosses = [g["text"] for g in sense["gloss"] if g.get("lang") == GLOSS_LANG][:MAX_FORMS]
if glosses:
senses.append(glosses)
if len(senses) == MAX_FORMS:
break
common = int(any(k.get("common") for k in (*word["kanji"], *word["kana"])))
return [int(word["id"]), kanji, kana, senses, common]
def write_compact(entries: list[list]) -> tuple[int, int]:
"""Write OUT deterministically; returns (raw_bytes, gz_bytes)."""
payload = {
"meta": {
"source": SOURCE_REPO,
"version": RELEASE_TAG,
"jmdictVersion": EXPECTED_VERSION,
"dictDate": EXPECTED_DICT_DATE,
"entries": len(entries),
"licence": "CC BY-SA 4.0 (EDRDG)",
},
"entries": entries,
}
raw = json.dumps(payload, ensure_ascii=False, separators=(",", ":")).encode("utf-8")
OUT_DIR.mkdir(parents=True, exist_ok=True)
with (
open(OUT, "wb") as fh,
gzip.GzipFile(filename="", mode="wb", fileobj=fh, compresslevel=9, mtime=0) as gz,
):
gz.write(raw)
return len(raw), OUT.stat().st_size
def read_back() -> tuple[dict, float]:
t = time.perf_counter()
with gzip.open(OUT, "rt", encoding="utf-8") as fh:
data = json.load(fh)
return data, time.perf_counter() - t
def recorded_date(readme_text: str) -> str | None:
m = re.search(r"^\*\*Resolved:\*\* (\d{4}-\d{2}-\d{2})", readme_text, re.M)
return m.group(1) if m else None
def recorded_hashes(readme_text: str) -> set[str]:
return set(re.findall(r"`([0-9a-f]{64})`", readme_text))
def write_readme(
*,
date: str,
asset_url: str,
tgz_bytes: int,
tgz_sha: str,
raw_bytes: int,
gz_bytes: int,
gz_sha: str,
entries: int,
no_kanji: int,
no_sense: int,
common: int,
) -> None:
lines = [
"# data/jmdict - the compact JMdict projection, pinned",
"",
"Generated by `scripts/build_jmdict.py`. **Do not hand-edit or re-compress this file**:",
"`tests/test_data_assets.py` compares it with the SHA-256 below and pins the entry count,",
"so any change fails the quick loop until this file is regenerated. The file is a Git",
"LFS object (`*.gz` in `.gitattributes`); if it starts with `version https://git-lfs`",
"instead of the gzip magic, run `git lfs pull`.",
"",
f"**Resolved:** {date} ",
f"**Pins:** {SOURCE_REPO} release `{RELEASE_TAG}` (JSON `version` `{EXPECTED_VERSION}`, "
f"`dictDate` `{EXPECTED_DICT_DATE}`), {EXPECTED_ENTRIES:,} words ",
f"**Input:** `{asset_url}` ",
f"**Input tgz:** {tgz_bytes:,} bytes, SHA-256 `{tgz_sha}`",
"",
"Regenerate (from the repo root; re-running is idempotent - byte-identical output, the",
"gzip header carries no timestamp or name):",
"",
"```",
".venv/Scripts/python.exe scripts/build_jmdict.py",
"```",
"",
"## Files",
"",
"| File | Bytes | SHA256 | Source |",
"|---|---|---|---|",
f"| `{OUT.name}` | {gz_bytes:,} | `{gz_sha}` | projection of the tgz above |",
"",
f"Raw JSON {raw_bytes:,} bytes -> gzip level 9 {gz_bytes:,} bytes. Loading is one",
"`gzip.open` + `json.load` (~0.6-2 s depending on the machine; the build prints the",
"measured time and `tests/test_data_assets.py`'s fixture prints it again) - kept out of",
"this file so a re-run stays byte-identical.",
"",
"## Compact record schema",
"",
"```",
'{"meta": {"source", "version", "jmdictVersion", "dictDate", "entries", "licence"},',
' "entries": [[id, kanji_texts, kana_texts, senses, common], ...]}',
"",
"id int JMdict entry sequence number (the `jmdict_seq` of data/jlpt/*.csv)",
f"kanji_texts [str] up to {MAX_FORMS} kanji forms, file order (empty for kana-only words)",
f"kana_texts [str] up to {MAX_FORMS} kana forms, file order",
f"senses [[str]] up to {MAX_FORMS} senses x up to {MAX_FORMS} English glosses each;",
" senses with no English gloss are dropped",
"common 0|1 1 if any kanji or kana form is marked common",
"```",
"",
"Entries are in JMdict file order; the index in `entries` is the order the ranked lookup",
"(plan 02-03) tie-breaks on. Counts in this build: "
f"{entries:,} entries, {common:,} common, {no_kanji:,} kana-only, "
f"{no_sense:,} with no English sense.",
"",
"## Licence",
"",
EDRDG_TEXT,
"",
PACKAGING_TEXT,
"",
"`LICENSES.md` at the repo root is the project-wide record and the page's credits line",
"carries the attribution (plan 02-11).",
"",
]
README.write_text("\n".join(lines), encoding="utf-8", newline="\n")
def main() -> int:
previous = README.read_text(encoding="utf-8") if README.exists() else ""
name, url, _size = resolve_asset()
tgz, tgz_data = download_tgz(name, url)
tgz_sha = sha256(tgz_data)
data = load_words(tgz)
print(f"version {data['version']} dictDate {data['dictDate']} words {len(data['words']):,}")
entries = [project(w) for w in data["words"]]
del data
ids = [e[0] for e in entries]
if len(set(ids)) != len(ids) or any(i <= 0 for i in ids):
raise SystemExit("entry ids are not unique positive ints; nothing written")
no_kanji = sum(1 for e in entries if not e[1])
no_sense = sum(1 for e in entries if not e[3])
common = sum(e[4] for e in entries)
if any(not e[2] for e in entries):
raise SystemExit("an entry has no kana form; JMdict guarantees one - nothing written")
raw_bytes, gz_bytes = write_compact(entries)
print(f"wrote {OUT.relative_to(REPO_ROOT)}: raw {raw_bytes:,} bytes -> gz {gz_bytes:,} bytes")
back, seconds = read_back()
print(f"read-back: {len(back['entries']):,} entries in {seconds:.2f} s")
if len(back["entries"]) != len(entries) or back["meta"]["entries"] != EXPECTED_ENTRIES:
OUT.unlink()
raise SystemExit("read-back entry count differs from what was written; file removed")
gz_sha = sha256(OUT.read_bytes())
date = recorded_date(previous)
if date is None or gz_sha not in recorded_hashes(previous):
date = datetime.now(UTC).strftime("%Y-%m-%d")
write_readme(
date=date,
asset_url=url,
tgz_bytes=len(tgz_data),
tgz_sha=tgz_sha,
raw_bytes=raw_bytes,
gz_bytes=gz_bytes,
gz_sha=gz_sha,
entries=len(entries),
no_kanji=no_kanji,
no_sense=no_sense,
common=common,
)
print(f"README resolved date {date}; gz sha256 {gz_sha}")
return 0
if __name__ == "__main__":
sys.exit(main())
|