File size: 14,467 Bytes
549c072
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
"""Build data/jmdict/jmdict-compact.json.gz: the full JMdict, projected to what the card needs.

D-07 puts JMdict meanings on the lookup card. 02-RESEARCH.md § Q3 measured the options: the
raw ``jmdict-eng`` JSON is 117.8 MB and +804 MB RSS, the ``-common`` subset misses 538 of
the JLPT words, and a positional projection of the full file is ~7.7 MB gzipped, loads in
under 2 s and covers every entry - so every word the avatar or the learner produces gets a
gloss or an honest "no entry". This script builds that projection reproducibly:

1. Resolve the release asset through the GitHub release API for the pinned tag. The tag is
   immutable; the API call only spares us guessing the URL (the resolved URL is recorded
   in the README and the tgz's SHA-256 is pinned below, so the download is verified).
2. Download the tgz to a cache directory OUTSIDE the repo, check its SHA-256, and stream
   the single ``.json`` member into ``json.load``. Refuse unless ``version``,
   ``dictDate`` and ``len(words)`` are exactly what research measured.
3. Project each word to ``[id, kanji_texts[:3], kana_texts[:3], senses[:3] x glosses[:3],
   common]`` in file order - the index in the list is the JMdict order the ranked lookup
   (plan 02-03) tie-breaks on.
4. Write ``{"meta": {...}, "entries": [...]}`` as compact UTF-8 JSON, gzip level 9 with a
   zero mtime and no embedded name, so a re-run is byte-identical. ``*.gz`` is an LFS
   pattern in ``.gitattributes``; the README hash lets the quick loop tell a real object
   from an unfetched pointer.
5. Read the file back the way the app will (``gzip.open`` + ``json.load``), time it, and
   refuse if the entry count differs; then write ``data/jmdict/README.md`` with the pins,
   the SHA-256 row and the EDRDG attribution.

Run from the repo root::

    .venv/Scripts/python.exe scripts/build_jmdict.py

Set ``JLA_DATA_CACHE`` to reuse an already-downloaded tgz (the download is 11.5 MB).
Stdlib only, on purpose.
"""

from __future__ import annotations

import gzip
import hashlib
import json
import os
import re
import sys
import tarfile
import tempfile
import time
import urllib.request
from datetime import UTC, datetime
from pathlib import Path

REPO_ROOT = Path(__file__).resolve().parent.parent
OUT_DIR = REPO_ROOT / "data" / "jmdict"
OUT = OUT_DIR / "jmdict-compact.json.gz"
README = OUT_DIR / "README.md"

SOURCE_REPO = "scriptin/jmdict-simplified"
RELEASE_TAG = "3.6.2+20260831182826"
# The JSON's own top-level `version` is the bare release version; the `+<build>` suffix is
# only in the tag and the asset name (measured on this asset - the plan's draft expected
# the full tag here and would have refused a correct file).
EXPECTED_VERSION = "3.6.2"
EXPECTED_DICT_DATE = "2026-08-31"
EXPECTED_ENTRIES = 218672
ASSET_RE = re.compile(r"^jmdict-eng-3\.6\.2\+20260831182826\.json\.tgz$")
# Resolved and measured on 2026-09-06; a re-run verifies the download against these.
EXPECTED_TGZ_BYTES = 11510336
EXPECTED_TGZ_SHA256 = "3c842741e2c4f1b780ad4aff6833df308866a8b7dfb9ee59a085e23dd201a65f"

RELEASE_API = (
    f"https://api.github.com/repos/{SOURCE_REPO}/releases/tags/{RELEASE_TAG.replace('+', '%2B')}"
)
CACHE_DIR = (
    Path(os.environ.get("JLA_DATA_CACHE", tempfile.gettempdir())) / "japanese-learning-avatar"
)

MAX_FORMS = 3  # kanji texts, kana texts, senses, glosses per sense
GLOSS_LANG = "eng"

EDRDG_TEXT = (
    "Dictionary data: **JMdict** - Copyright (c) James William Breen and The Electronic "
    "Dictionary Research and Development Group, used under the Creative Commons "
    "Attribution-ShareAlike Licence (V4.0). "
    "https://www.edrdg.org/wiki/index.php/JMdict-EDICT_Dictionary_Project . "
    "https://www.edrdg.org/edrdg/licence.html"
)
PACKAGING_TEXT = (
    f"JSON conversion by {SOURCE_REPO}, release {RELEASE_TAG}; its derived files carry the "
    "EDRDG licence. The compact projection in this directory is a derivative of JMdict and "
    "is itself CC BY-SA 4.0."
)


def fetch(url: str, accept: str | None = None) -> bytes:
    headers = {"User-Agent": "japanese-learning-avatar data build"}
    if accept:
        headers["Accept"] = accept
    req = urllib.request.Request(url, headers=headers)
    with urllib.request.urlopen(req, timeout=600) as resp:  # noqa: S310 - pinned https URLs
        return resp.read()


def sha256(data: bytes) -> str:
    return hashlib.sha256(data).hexdigest()


def resolve_asset() -> tuple[str, str, int]:
    """(name, browser_download_url, size) of the ONE asset matching ASSET_RE."""
    release = json.loads(fetch(RELEASE_API, accept="application/vnd.github+json"))
    if release.get("tag_name") != RELEASE_TAG:
        raise SystemExit(f"release API returned tag {release.get('tag_name')!r}, not {RELEASE_TAG}")
    matches = [a for a in release["assets"] if ASSET_RE.match(a["name"])]
    if len(matches) != 1:
        names = [a["name"] for a in release["assets"]]
        raise SystemExit(f"expected exactly one asset matching {ASSET_RE.pattern}; got {names}")
    asset = matches[0]
    print(f"asset {asset['name']} ({asset['size']:,} bytes)\n    {asset['browser_download_url']}")
    return asset["name"], asset["browser_download_url"], int(asset["size"])


def download_tgz(name: str, url: str) -> tuple[Path, bytes]:
    CACHE_DIR.mkdir(parents=True, exist_ok=True)
    path = CACHE_DIR / name
    if path.is_file() and path.stat().st_size == EXPECTED_TGZ_BYTES:
        data = path.read_bytes()
        print(f"reusing cached {path}")
    else:
        t = time.perf_counter()
        data = fetch(url)
        path.write_bytes(data)
        print(f"downloaded {len(data):,} bytes in {time.perf_counter() - t:.1f} s to {path}")
    digest = sha256(data)
    if len(data) != EXPECTED_TGZ_BYTES or digest != EXPECTED_TGZ_SHA256:
        raise SystemExit(
            f"{name}: {len(data):,} bytes sha256 {digest}; expected {EXPECTED_TGZ_BYTES:,} / "
            f"{EXPECTED_TGZ_SHA256}. The asset is not the one research measured; nothing written."
        )
    return path, data


def load_words(tgz: Path) -> dict:
    with tarfile.open(tgz) as tf:
        members = [m for m in tf.getmembers() if m.name.endswith(".json")]
        if len(members) != 1:
            raise SystemExit(
                f"{tgz.name}: expected one .json member, got {[m.name for m in members]}"
            )
        stream = tf.extractfile(members[0])
        assert stream is not None
        t = time.perf_counter()
        data = json.load(stream)
        seconds = time.perf_counter() - t
        print(f"json.load {members[0].name} ({members[0].size:,} bytes) in {seconds:.2f} s")
    problems = []
    if data.get("version") != EXPECTED_VERSION:
        problems.append(f"version {data.get('version')!r} != {EXPECTED_VERSION!r}")
    if data.get("dictDate") != EXPECTED_DICT_DATE:
        problems.append(f"dictDate {data.get('dictDate')!r} != {EXPECTED_DICT_DATE!r}")
    if len(data.get("words", [])) != EXPECTED_ENTRIES:
        problems.append(f"len(words) {len(data.get('words', []))} != {EXPECTED_ENTRIES}")
    if problems:
        raise SystemExit(
            "jmdict-eng JSON is not the pinned file; nothing written:\n  " + "\n  ".join(problems)
        )
    return data


def project(word: dict) -> list:
    """``[id, kanji_texts, kana_texts, senses, common]`` - positional so the file stays small."""
    kanji = [k["text"] for k in word["kanji"]][:MAX_FORMS]
    kana = [k["text"] for k in word["kana"]][:MAX_FORMS]
    senses: list[list[str]] = []
    for sense in word["sense"]:
        glosses = [g["text"] for g in sense["gloss"] if g.get("lang") == GLOSS_LANG][:MAX_FORMS]
        if glosses:
            senses.append(glosses)
        if len(senses) == MAX_FORMS:
            break
    common = int(any(k.get("common") for k in (*word["kanji"], *word["kana"])))
    return [int(word["id"]), kanji, kana, senses, common]


def write_compact(entries: list[list]) -> tuple[int, int]:
    """Write OUT deterministically; returns (raw_bytes, gz_bytes)."""
    payload = {
        "meta": {
            "source": SOURCE_REPO,
            "version": RELEASE_TAG,
            "jmdictVersion": EXPECTED_VERSION,
            "dictDate": EXPECTED_DICT_DATE,
            "entries": len(entries),
            "licence": "CC BY-SA 4.0 (EDRDG)",
        },
        "entries": entries,
    }
    raw = json.dumps(payload, ensure_ascii=False, separators=(",", ":")).encode("utf-8")
    OUT_DIR.mkdir(parents=True, exist_ok=True)
    with (
        open(OUT, "wb") as fh,
        gzip.GzipFile(filename="", mode="wb", fileobj=fh, compresslevel=9, mtime=0) as gz,
    ):
        gz.write(raw)
    return len(raw), OUT.stat().st_size


def read_back() -> tuple[dict, float]:
    t = time.perf_counter()
    with gzip.open(OUT, "rt", encoding="utf-8") as fh:
        data = json.load(fh)
    return data, time.perf_counter() - t


def recorded_date(readme_text: str) -> str | None:
    m = re.search(r"^\*\*Resolved:\*\* (\d{4}-\d{2}-\d{2})", readme_text, re.M)
    return m.group(1) if m else None


def recorded_hashes(readme_text: str) -> set[str]:
    return set(re.findall(r"`([0-9a-f]{64})`", readme_text))


def write_readme(
    *,
    date: str,
    asset_url: str,
    tgz_bytes: int,
    tgz_sha: str,
    raw_bytes: int,
    gz_bytes: int,
    gz_sha: str,
    entries: int,
    no_kanji: int,
    no_sense: int,
    common: int,
) -> None:
    lines = [
        "# data/jmdict - the compact JMdict projection, pinned",
        "",
        "Generated by `scripts/build_jmdict.py`. **Do not hand-edit or re-compress this file**:",
        "`tests/test_data_assets.py` compares it with the SHA-256 below and pins the entry count,",
        "so any change fails the quick loop until this file is regenerated. The file is a Git",
        "LFS object (`*.gz` in `.gitattributes`); if it starts with `version https://git-lfs`",
        "instead of the gzip magic, run `git lfs pull`.",
        "",
        f"**Resolved:** {date}  ",
        f"**Pins:** {SOURCE_REPO} release `{RELEASE_TAG}` (JSON `version` `{EXPECTED_VERSION}`, "
        f"`dictDate` `{EXPECTED_DICT_DATE}`), {EXPECTED_ENTRIES:,} words  ",
        f"**Input:** `{asset_url}`  ",
        f"**Input tgz:** {tgz_bytes:,} bytes, SHA-256 `{tgz_sha}`",
        "",
        "Regenerate (from the repo root; re-running is idempotent - byte-identical output, the",
        "gzip header carries no timestamp or name):",
        "",
        "```",
        ".venv/Scripts/python.exe scripts/build_jmdict.py",
        "```",
        "",
        "## Files",
        "",
        "| File | Bytes | SHA256 | Source |",
        "|---|---|---|---|",
        f"| `{OUT.name}` | {gz_bytes:,} | `{gz_sha}` | projection of the tgz above |",
        "",
        f"Raw JSON {raw_bytes:,} bytes -> gzip level 9 {gz_bytes:,} bytes. Loading is one",
        "`gzip.open` + `json.load` (~0.6-2 s depending on the machine; the build prints the",
        "measured time and `tests/test_data_assets.py`'s fixture prints it again) - kept out of",
        "this file so a re-run stays byte-identical.",
        "",
        "## Compact record schema",
        "",
        "```",
        '{"meta": {"source", "version", "jmdictVersion", "dictDate", "entries", "licence"},',
        ' "entries": [[id, kanji_texts, kana_texts, senses, common], ...]}',
        "",
        "id          int    JMdict entry sequence number (the `jmdict_seq` of data/jlpt/*.csv)",
        f"kanji_texts [str]  up to {MAX_FORMS} kanji forms, file order (empty for kana-only words)",
        f"kana_texts  [str]  up to {MAX_FORMS} kana forms, file order",
        f"senses      [[str]] up to {MAX_FORMS} senses x up to {MAX_FORMS} English glosses each;",
        "                   senses with no English gloss are dropped",
        "common      0|1    1 if any kanji or kana form is marked common",
        "```",
        "",
        "Entries are in JMdict file order; the index in `entries` is the order the ranked lookup",
        "(plan 02-03) tie-breaks on. Counts in this build: "
        f"{entries:,} entries, {common:,} common, {no_kanji:,} kana-only, "
        f"{no_sense:,} with no English sense.",
        "",
        "## Licence",
        "",
        EDRDG_TEXT,
        "",
        PACKAGING_TEXT,
        "",
        "`LICENSES.md` at the repo root is the project-wide record and the page's credits line",
        "carries the attribution (plan 02-11).",
        "",
    ]
    README.write_text("\n".join(lines), encoding="utf-8", newline="\n")


def main() -> int:
    previous = README.read_text(encoding="utf-8") if README.exists() else ""

    name, url, _size = resolve_asset()
    tgz, tgz_data = download_tgz(name, url)
    tgz_sha = sha256(tgz_data)
    data = load_words(tgz)
    print(f"version {data['version']} dictDate {data['dictDate']} words {len(data['words']):,}")

    entries = [project(w) for w in data["words"]]
    del data
    ids = [e[0] for e in entries]
    if len(set(ids)) != len(ids) or any(i <= 0 for i in ids):
        raise SystemExit("entry ids are not unique positive ints; nothing written")
    no_kanji = sum(1 for e in entries if not e[1])
    no_sense = sum(1 for e in entries if not e[3])
    common = sum(e[4] for e in entries)
    if any(not e[2] for e in entries):
        raise SystemExit("an entry has no kana form; JMdict guarantees one - nothing written")

    raw_bytes, gz_bytes = write_compact(entries)
    print(f"wrote {OUT.relative_to(REPO_ROOT)}: raw {raw_bytes:,} bytes -> gz {gz_bytes:,} bytes")

    back, seconds = read_back()
    print(f"read-back: {len(back['entries']):,} entries in {seconds:.2f} s")
    if len(back["entries"]) != len(entries) or back["meta"]["entries"] != EXPECTED_ENTRIES:
        OUT.unlink()
        raise SystemExit("read-back entry count differs from what was written; file removed")

    gz_sha = sha256(OUT.read_bytes())
    date = recorded_date(previous)
    if date is None or gz_sha not in recorded_hashes(previous):
        date = datetime.now(UTC).strftime("%Y-%m-%d")
    write_readme(
        date=date,
        asset_url=url,
        tgz_bytes=len(tgz_data),
        tgz_sha=tgz_sha,
        raw_bytes=raw_bytes,
        gz_bytes=gz_bytes,
        gz_sha=gz_sha,
        entries=len(entries),
        no_kanji=no_kanji,
        no_sense=no_sense,
        common=common,
    )
    print(f"README resolved date {date}; gz sha256 {gz_sha}")
    return 0


if __name__ == "__main__":
    sys.exit(main())