Spaces:
Running on Zero
Running on Zero
feat(02-03): compact JMdict loader with ranked headword lookup
Browse files- load(): lru_cache singleton over data/jmdict/jmdict-compact.json.gz; frozen Entry
records with file order; by_id / by_kanji / by_kana indexes (453,630 headword keys)
- lookup(lemma, level_key, reading, level_of): rank (reading match, on JLPT list,
common, -order) so γγ->ε±
γ, γγ->ζγ, γγγ«γ‘γ― via its kana headword
- warmup() and glosses_for(entry, max_senses=3); LFS-pointer guard on load
- measured: without the JLPT term ε±
γ/ε°γ/θ¦γ tie and file order picks ε°γ, so
the empty-level_of test asserts only that η
γ loses
- src/japanese_avatar/nlp/jmdict.py +190 -0
- tests/test_jmdict.py +5 -3
src/japanese_avatar/nlp/jmdict.py
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""The compact JMdict as a read-only, process-wide singleton with a ranked headword lookup.
|
| 2 |
+
|
| 3 |
+
``data/jmdict/jmdict-compact.json.gz`` (plan 02-01) is a positional projection of jmdict-eng
|
| 4 |
+
``3.6.2+20260831182826``: ``[id, kanji_texts, kana_texts, senses, common]`` per entry, in JMdict
|
| 5 |
+
file order. :func:`load` gunzips and parses it once per process (``lru_cache``), builds one
|
| 6 |
+
:class:`Entry` per record and two headword indexes (``by_kanji``, ``by_kana``; a headword may
|
| 7 |
+
name several entries, kept in file order), and hands back a :class:`Lexicon` that nothing
|
| 8 |
+
mutates afterwards. It is therefore safe to share across Gradio sessions: there is no
|
| 9 |
+
module-level mutable state here - the only cache is the one ``lru_cache`` owns on ``load``.
|
| 10 |
+
|
| 11 |
+
:func:`lookup` is the Sudachi-lemma -> JMdict-entry step. It is a headword match, so homographs
|
| 12 |
+
need a rule: 02-RESEARCH.md Β§ Common Pitfalls 2 measured the naive join picking η
γ for γγ,
|
| 13 |
+
ζγ for γγ and ι
γ for γγ. The rank is ``(reading matches a kana form, id is on a JLPT list,
|
| 14 |
+
JMdict common flag, earlier file order)`` - reading first, JLPT second, common third, and JMdict's
|
| 15 |
+
own order as the LAST resort only (a lowest-id tie-break picked η
γ 1391500 over ε±
γ 1577980).
|
| 16 |
+
|
| 17 |
+
Importing this module reads nothing; call :func:`warmup` at app start so the first visitor does
|
| 18 |
+
not pay the load (the ``voice.tts.warmup`` precedent).
|
| 19 |
+
"""
|
| 20 |
+
|
| 21 |
+
from __future__ import annotations
|
| 22 |
+
|
| 23 |
+
import gzip
|
| 24 |
+
import json
|
| 25 |
+
import time
|
| 26 |
+
from collections.abc import Mapping
|
| 27 |
+
from dataclasses import dataclass
|
| 28 |
+
from functools import lru_cache
|
| 29 |
+
from pathlib import Path
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def _repo_root() -> Path:
|
| 33 |
+
"""Nearest ancestor holding ``pyproject.toml`` (the checkout), else the CWD."""
|
| 34 |
+
for parent in Path(__file__).resolve().parents:
|
| 35 |
+
if (parent / "pyproject.toml").is_file():
|
| 36 |
+
return parent
|
| 37 |
+
return Path.cwd()
|
| 38 |
+
|
| 39 |
+
|
| 40 |
+
REPO_ROOT = _repo_root()
|
| 41 |
+
COMPACT_PATH = REPO_ROOT / "data" / "jmdict" / "jmdict-compact.json.gz"
|
| 42 |
+
|
| 43 |
+
#: How many senses :func:`glosses_for` returns by default (D-07: "first two or three senses").
|
| 44 |
+
DEFAULT_MAX_SENSES = 3
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
@dataclass(frozen=True, slots=True)
|
| 48 |
+
class Entry:
|
| 49 |
+
"""One JMdict entry as the compact file records it.
|
| 50 |
+
|
| 51 |
+
``glosses`` is up to 3 senses x up to 3 English glosses; ``order`` is the record's index in
|
| 52 |
+
the compact file (JMdict file order), used only as the final ranking tie-break.
|
| 53 |
+
"""
|
| 54 |
+
|
| 55 |
+
id: int
|
| 56 |
+
kanji: tuple[str, ...]
|
| 57 |
+
kana: tuple[str, ...]
|
| 58 |
+
glosses: tuple[tuple[str, ...], ...]
|
| 59 |
+
common: bool
|
| 60 |
+
order: int
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
class Lexicon:
|
| 64 |
+
"""The loaded dictionary: entries by id and the two headword indexes.
|
| 65 |
+
|
| 66 |
+
Built once by :func:`load` and never mutated afterwards. ``by_kanji`` and ``by_kana`` map a
|
| 67 |
+
headword string to every entry that lists it, in file order.
|
| 68 |
+
"""
|
| 69 |
+
|
| 70 |
+
__slots__ = ("by_id", "by_kana", "by_kanji", "meta")
|
| 71 |
+
|
| 72 |
+
def __init__(self, meta: Mapping[str, object], entries: list[Entry]) -> None:
|
| 73 |
+
self.meta: Mapping[str, object] = dict(meta)
|
| 74 |
+
by_id: dict[int, Entry] = {}
|
| 75 |
+
by_kanji: dict[str, list[Entry]] = {}
|
| 76 |
+
by_kana: dict[str, list[Entry]] = {}
|
| 77 |
+
for entry in entries:
|
| 78 |
+
by_id[entry.id] = entry
|
| 79 |
+
for text in entry.kanji:
|
| 80 |
+
by_kanji.setdefault(text, []).append(entry)
|
| 81 |
+
for text in entry.kana:
|
| 82 |
+
by_kana.setdefault(text, []).append(entry)
|
| 83 |
+
self.by_id: Mapping[int, Entry] = by_id
|
| 84 |
+
self.by_kanji: Mapping[str, list[Entry]] = by_kanji
|
| 85 |
+
self.by_kana: Mapping[str, list[Entry]] = by_kana
|
| 86 |
+
|
| 87 |
+
def __len__(self) -> int:
|
| 88 |
+
return len(self.by_id)
|
| 89 |
+
|
| 90 |
+
def __repr__(self) -> str: # pragma: no cover - debugging aid
|
| 91 |
+
return (
|
| 92 |
+
f"Lexicon(entries={len(self.by_id)}, kanji_keys={len(self.by_kanji)}, "
|
| 93 |
+
f"kana_keys={len(self.by_kana)})"
|
| 94 |
+
)
|
| 95 |
+
|
| 96 |
+
|
| 97 |
+
def _entry_from_record(order: int, record: list) -> Entry:
|
| 98 |
+
entry_id, kanji, kana, senses, common = record
|
| 99 |
+
return Entry(
|
| 100 |
+
id=int(entry_id),
|
| 101 |
+
kanji=tuple(kanji),
|
| 102 |
+
kana=tuple(kana),
|
| 103 |
+
glosses=tuple(tuple(sense) for sense in senses),
|
| 104 |
+
common=bool(common),
|
| 105 |
+
order=order,
|
| 106 |
+
)
|
| 107 |
+
|
| 108 |
+
|
| 109 |
+
@lru_cache(maxsize=1)
|
| 110 |
+
def load() -> Lexicon:
|
| 111 |
+
"""Gunzip, parse and index the compact JMdict once per process.
|
| 112 |
+
|
| 113 |
+
Raises ``FileNotFoundError`` with a pointer at ``git lfs pull`` when the file is missing or is
|
| 114 |
+
still an LFS pointer, because a bare gzip error tells the Space log nothing useful.
|
| 115 |
+
"""
|
| 116 |
+
if not COMPACT_PATH.is_file():
|
| 117 |
+
raise FileNotFoundError(
|
| 118 |
+
f"{COMPACT_PATH} is missing. It is committed through Git LFS; run `git lfs pull` "
|
| 119 |
+
"or regenerate it with scripts/build_jmdict.py."
|
| 120 |
+
)
|
| 121 |
+
with COMPACT_PATH.open("rb") as raw:
|
| 122 |
+
if raw.read(2) != b"\x1f\x8b":
|
| 123 |
+
raise FileNotFoundError(
|
| 124 |
+
f"{COMPACT_PATH} is not a gzip file - it is probably a Git LFS pointer; "
|
| 125 |
+
"run `git lfs pull`."
|
| 126 |
+
)
|
| 127 |
+
with gzip.open(COMPACT_PATH, "rt", encoding="utf-8") as fh:
|
| 128 |
+
data = json.load(fh)
|
| 129 |
+
entries = [_entry_from_record(i, record) for i, record in enumerate(data["entries"])]
|
| 130 |
+
return Lexicon(data.get("meta", {}), entries)
|
| 131 |
+
|
| 132 |
+
|
| 133 |
+
def warmup() -> float:
|
| 134 |
+
"""Load the dictionary ahead of the first request. Returns seconds elapsed.
|
| 135 |
+
|
| 136 |
+
Idempotent: :func:`load` is cached, so repeat calls cost nothing.
|
| 137 |
+
"""
|
| 138 |
+
started = time.perf_counter()
|
| 139 |
+
load()
|
| 140 |
+
return time.perf_counter() - started
|
| 141 |
+
|
| 142 |
+
|
| 143 |
+
def _rank(entry: Entry, reading_hira: str, level_of: Mapping[int, int]) -> tuple:
|
| 144 |
+
"""Higher is better: reading match, on a JLPT list, JMdict common, earlier in the file."""
|
| 145 |
+
return (reading_hira in entry.kana, entry.id in level_of, entry.common, -entry.order)
|
| 146 |
+
|
| 147 |
+
|
| 148 |
+
def lookup(
|
| 149 |
+
lemma: str,
|
| 150 |
+
level_key: str,
|
| 151 |
+
reading_hira: str,
|
| 152 |
+
level_of: Mapping[int, int],
|
| 153 |
+
) -> Entry | None:
|
| 154 |
+
"""Resolve a Sudachi unit to the JMdict entry the ranking rule picks, or ``None``.
|
| 155 |
+
|
| 156 |
+
``level_key`` (the head's ``normalized_form``: θ©±γγ -> θ©±γ, γγ -> θ―γ) is tried before
|
| 157 |
+
``lemma`` (the ``dictionary_form``); both are looked up as kanji and as kana headwords.
|
| 158 |
+
Candidates are deduplicated by id keeping the first occurrence, then the best by :func:`_rank`
|
| 159 |
+
wins. ``level_of`` is the JLPT join (``levels.vocab_levels()``); an empty mapping is allowed.
|
| 160 |
+
"""
|
| 161 |
+
lex = load()
|
| 162 |
+
seen: set[int] = set()
|
| 163 |
+
candidates: list[Entry] = []
|
| 164 |
+
for key in (level_key, lemma):
|
| 165 |
+
if not key:
|
| 166 |
+
continue
|
| 167 |
+
for entry in (*lex.by_kanji.get(key, ()), *lex.by_kana.get(key, ())):
|
| 168 |
+
if entry.id not in seen:
|
| 169 |
+
seen.add(entry.id)
|
| 170 |
+
candidates.append(entry)
|
| 171 |
+
if not candidates:
|
| 172 |
+
return None
|
| 173 |
+
return max(candidates, key=lambda e: _rank(e, reading_hira, level_of))
|
| 174 |
+
|
| 175 |
+
|
| 176 |
+
def glosses_for(entry: Entry, max_senses: int = DEFAULT_MAX_SENSES) -> list[list[str]]:
|
| 177 |
+
"""The entry's English glosses as plain lists, capped at ``max_senses`` senses (D-07)."""
|
| 178 |
+
return [list(sense) for sense in entry.glosses[:max_senses]]
|
| 179 |
+
|
| 180 |
+
|
| 181 |
+
__all__ = [
|
| 182 |
+
"COMPACT_PATH",
|
| 183 |
+
"DEFAULT_MAX_SENSES",
|
| 184 |
+
"Entry",
|
| 185 |
+
"Lexicon",
|
| 186 |
+
"glosses_for",
|
| 187 |
+
"load",
|
| 188 |
+
"lookup",
|
| 189 |
+
"warmup",
|
| 190 |
+
]
|
tests/test_jmdict.py
CHANGED
|
@@ -77,10 +77,12 @@ def test_entries_are_frozen_records(lexicon):
|
|
| 77 |
def test_iru_is_oru_not_iru_roast(lexicon):
|
| 78 |
hit = jmdict.lookup("γγ", "ε±
γ", "γγ", {IRU_ORU: 5})
|
| 79 |
assert hit is not None and hit.id == IRU_ORU
|
| 80 |
-
# Without any JLPT knowledge the reading match plus `common` still beats η
γ.
|
| 81 |
-
|
|
|
|
|
|
|
| 82 |
assert bare is not None and bare.id != IRU_ROAST
|
| 83 |
-
assert bare.
|
| 84 |
|
| 85 |
|
| 86 |
def test_aru_naru_kore(lexicon):
|
|
|
|
| 77 |
def test_iru_is_oru_not_iru_roast(lexicon):
|
| 78 |
hit = jmdict.lookup("γγ", "ε±
γ", "γγ", {IRU_ORU: 5})
|
| 79 |
assert hit is not None and hit.id == IRU_ORU
|
| 80 |
+
# Without any JLPT knowledge the reading match plus `common` still beats η
γ. Measured:
|
| 81 |
+
# ε±
γ / ε°γ / θ¦γ then tie on (reading, common) and file order picks ε°γ 1322180 - the
|
| 82 |
+
# JLPT term is what makes ε±
γ win, which is why lookup() takes level_of at all.
|
| 83 |
+
bare = jmdict.lookup("γγ", "ε±
γ", "γγ", {})
|
| 84 |
assert bare is not None and bare.id != IRU_ROAST
|
| 85 |
+
assert "γγ" in bare.kana and bare.common
|
| 86 |
|
| 87 |
|
| 88 |
def test_aru_naru_kore(lexicon):
|