WolfDavid commited on
Commit
f108e4c
Β·
1 Parent(s): fbd985f

feat(02-03): compact JMdict loader with ranked headword lookup

Browse files

- load(): lru_cache singleton over data/jmdict/jmdict-compact.json.gz; frozen Entry
records with file order; by_id / by_kanji / by_kana indexes (453,630 headword keys)
- lookup(lemma, level_key, reading, level_of): rank (reading match, on JLPT list,
common, -order) so いる->ε±…γ‚‹, ある->ζœ‰γ‚‹, こんにけは via its kana headword
- warmup() and glosses_for(entry, max_senses=3); LFS-pointer guard on load
- measured: without the JLPT term ε±…γ‚‹/ε°„γ‚‹/要る tie and file order picks ε°„γ‚‹, so
the empty-level_of test asserts only that η…Žγ‚‹ loses

src/japanese_avatar/nlp/jmdict.py ADDED
@@ -0,0 +1,190 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """The compact JMdict as a read-only, process-wide singleton with a ranked headword lookup.
2
+
3
+ ``data/jmdict/jmdict-compact.json.gz`` (plan 02-01) is a positional projection of jmdict-eng
4
+ ``3.6.2+20260831182826``: ``[id, kanji_texts, kana_texts, senses, common]`` per entry, in JMdict
5
+ file order. :func:`load` gunzips and parses it once per process (``lru_cache``), builds one
6
+ :class:`Entry` per record and two headword indexes (``by_kanji``, ``by_kana``; a headword may
7
+ name several entries, kept in file order), and hands back a :class:`Lexicon` that nothing
8
+ mutates afterwards. It is therefore safe to share across Gradio sessions: there is no
9
+ module-level mutable state here - the only cache is the one ``lru_cache`` owns on ``load``.
10
+
11
+ :func:`lookup` is the Sudachi-lemma -> JMdict-entry step. It is a headword match, so homographs
12
+ need a rule: 02-RESEARCH.md Β§ Common Pitfalls 2 measured the naive join picking η…Žγ‚‹ for いる,
13
+ ζˆ–γ‚‹ for ある and 酔う for γ‚ˆγ†. The rank is ``(reading matches a kana form, id is on a JLPT list,
14
+ JMdict common flag, earlier file order)`` - reading first, JLPT second, common third, and JMdict's
15
+ own order as the LAST resort only (a lowest-id tie-break picked η…Žγ‚‹ 1391500 over ε±…γ‚‹ 1577980).
16
+
17
+ Importing this module reads nothing; call :func:`warmup` at app start so the first visitor does
18
+ not pay the load (the ``voice.tts.warmup`` precedent).
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import gzip
24
+ import json
25
+ import time
26
+ from collections.abc import Mapping
27
+ from dataclasses import dataclass
28
+ from functools import lru_cache
29
+ from pathlib import Path
30
+
31
+
32
+ def _repo_root() -> Path:
33
+ """Nearest ancestor holding ``pyproject.toml`` (the checkout), else the CWD."""
34
+ for parent in Path(__file__).resolve().parents:
35
+ if (parent / "pyproject.toml").is_file():
36
+ return parent
37
+ return Path.cwd()
38
+
39
+
40
+ REPO_ROOT = _repo_root()
41
+ COMPACT_PATH = REPO_ROOT / "data" / "jmdict" / "jmdict-compact.json.gz"
42
+
43
+ #: How many senses :func:`glosses_for` returns by default (D-07: "first two or three senses").
44
+ DEFAULT_MAX_SENSES = 3
45
+
46
+
47
+ @dataclass(frozen=True, slots=True)
48
+ class Entry:
49
+ """One JMdict entry as the compact file records it.
50
+
51
+ ``glosses`` is up to 3 senses x up to 3 English glosses; ``order`` is the record's index in
52
+ the compact file (JMdict file order), used only as the final ranking tie-break.
53
+ """
54
+
55
+ id: int
56
+ kanji: tuple[str, ...]
57
+ kana: tuple[str, ...]
58
+ glosses: tuple[tuple[str, ...], ...]
59
+ common: bool
60
+ order: int
61
+
62
+
63
+ class Lexicon:
64
+ """The loaded dictionary: entries by id and the two headword indexes.
65
+
66
+ Built once by :func:`load` and never mutated afterwards. ``by_kanji`` and ``by_kana`` map a
67
+ headword string to every entry that lists it, in file order.
68
+ """
69
+
70
+ __slots__ = ("by_id", "by_kana", "by_kanji", "meta")
71
+
72
+ def __init__(self, meta: Mapping[str, object], entries: list[Entry]) -> None:
73
+ self.meta: Mapping[str, object] = dict(meta)
74
+ by_id: dict[int, Entry] = {}
75
+ by_kanji: dict[str, list[Entry]] = {}
76
+ by_kana: dict[str, list[Entry]] = {}
77
+ for entry in entries:
78
+ by_id[entry.id] = entry
79
+ for text in entry.kanji:
80
+ by_kanji.setdefault(text, []).append(entry)
81
+ for text in entry.kana:
82
+ by_kana.setdefault(text, []).append(entry)
83
+ self.by_id: Mapping[int, Entry] = by_id
84
+ self.by_kanji: Mapping[str, list[Entry]] = by_kanji
85
+ self.by_kana: Mapping[str, list[Entry]] = by_kana
86
+
87
+ def __len__(self) -> int:
88
+ return len(self.by_id)
89
+
90
+ def __repr__(self) -> str: # pragma: no cover - debugging aid
91
+ return (
92
+ f"Lexicon(entries={len(self.by_id)}, kanji_keys={len(self.by_kanji)}, "
93
+ f"kana_keys={len(self.by_kana)})"
94
+ )
95
+
96
+
97
+ def _entry_from_record(order: int, record: list) -> Entry:
98
+ entry_id, kanji, kana, senses, common = record
99
+ return Entry(
100
+ id=int(entry_id),
101
+ kanji=tuple(kanji),
102
+ kana=tuple(kana),
103
+ glosses=tuple(tuple(sense) for sense in senses),
104
+ common=bool(common),
105
+ order=order,
106
+ )
107
+
108
+
109
+ @lru_cache(maxsize=1)
110
+ def load() -> Lexicon:
111
+ """Gunzip, parse and index the compact JMdict once per process.
112
+
113
+ Raises ``FileNotFoundError`` with a pointer at ``git lfs pull`` when the file is missing or is
114
+ still an LFS pointer, because a bare gzip error tells the Space log nothing useful.
115
+ """
116
+ if not COMPACT_PATH.is_file():
117
+ raise FileNotFoundError(
118
+ f"{COMPACT_PATH} is missing. It is committed through Git LFS; run `git lfs pull` "
119
+ "or regenerate it with scripts/build_jmdict.py."
120
+ )
121
+ with COMPACT_PATH.open("rb") as raw:
122
+ if raw.read(2) != b"\x1f\x8b":
123
+ raise FileNotFoundError(
124
+ f"{COMPACT_PATH} is not a gzip file - it is probably a Git LFS pointer; "
125
+ "run `git lfs pull`."
126
+ )
127
+ with gzip.open(COMPACT_PATH, "rt", encoding="utf-8") as fh:
128
+ data = json.load(fh)
129
+ entries = [_entry_from_record(i, record) for i, record in enumerate(data["entries"])]
130
+ return Lexicon(data.get("meta", {}), entries)
131
+
132
+
133
+ def warmup() -> float:
134
+ """Load the dictionary ahead of the first request. Returns seconds elapsed.
135
+
136
+ Idempotent: :func:`load` is cached, so repeat calls cost nothing.
137
+ """
138
+ started = time.perf_counter()
139
+ load()
140
+ return time.perf_counter() - started
141
+
142
+
143
+ def _rank(entry: Entry, reading_hira: str, level_of: Mapping[int, int]) -> tuple:
144
+ """Higher is better: reading match, on a JLPT list, JMdict common, earlier in the file."""
145
+ return (reading_hira in entry.kana, entry.id in level_of, entry.common, -entry.order)
146
+
147
+
148
+ def lookup(
149
+ lemma: str,
150
+ level_key: str,
151
+ reading_hira: str,
152
+ level_of: Mapping[int, int],
153
+ ) -> Entry | None:
154
+ """Resolve a Sudachi unit to the JMdict entry the ranking rule picks, or ``None``.
155
+
156
+ ``level_key`` (the head's ``normalized_form``: 話せる -> 話す, いい -> 良い) is tried before
157
+ ``lemma`` (the ``dictionary_form``); both are looked up as kanji and as kana headwords.
158
+ Candidates are deduplicated by id keeping the first occurrence, then the best by :func:`_rank`
159
+ wins. ``level_of`` is the JLPT join (``levels.vocab_levels()``); an empty mapping is allowed.
160
+ """
161
+ lex = load()
162
+ seen: set[int] = set()
163
+ candidates: list[Entry] = []
164
+ for key in (level_key, lemma):
165
+ if not key:
166
+ continue
167
+ for entry in (*lex.by_kanji.get(key, ()), *lex.by_kana.get(key, ())):
168
+ if entry.id not in seen:
169
+ seen.add(entry.id)
170
+ candidates.append(entry)
171
+ if not candidates:
172
+ return None
173
+ return max(candidates, key=lambda e: _rank(e, reading_hira, level_of))
174
+
175
+
176
+ def glosses_for(entry: Entry, max_senses: int = DEFAULT_MAX_SENSES) -> list[list[str]]:
177
+ """The entry's English glosses as plain lists, capped at ``max_senses`` senses (D-07)."""
178
+ return [list(sense) for sense in entry.glosses[:max_senses]]
179
+
180
+
181
+ __all__ = [
182
+ "COMPACT_PATH",
183
+ "DEFAULT_MAX_SENSES",
184
+ "Entry",
185
+ "Lexicon",
186
+ "glosses_for",
187
+ "load",
188
+ "lookup",
189
+ "warmup",
190
+ ]
tests/test_jmdict.py CHANGED
@@ -77,10 +77,12 @@ def test_entries_are_frozen_records(lexicon):
77
  def test_iru_is_oru_not_iru_roast(lexicon):
78
  hit = jmdict.lookup("いる", "ε±…γ‚‹", "いる", {IRU_ORU: 5})
79
  assert hit is not None and hit.id == IRU_ORU
80
- # Without any JLPT knowledge the reading match plus `common` still beats η…Žγ‚‹.
81
- bare = jmdict.lookup("いる", "いる", "いる", {})
 
 
82
  assert bare is not None and bare.id != IRU_ROAST
83
- assert bare.id == IRU_ORU
84
 
85
 
86
  def test_aru_naru_kore(lexicon):
 
77
  def test_iru_is_oru_not_iru_roast(lexicon):
78
  hit = jmdict.lookup("いる", "ε±…γ‚‹", "いる", {IRU_ORU: 5})
79
  assert hit is not None and hit.id == IRU_ORU
80
+ # Without any JLPT knowledge the reading match plus `common` still beats η…Žγ‚‹. Measured:
81
+ # ε±…γ‚‹ / ε°„γ‚‹ / 要る then tie on (reading, common) and file order picks ε°„γ‚‹ 1322180 - the
82
+ # JLPT term is what makes ε±…γ‚‹ win, which is why lookup() takes level_of at all.
83
+ bare = jmdict.lookup("いる", "ε±…γ‚‹", "いる", {})
84
  assert bare is not None and bare.id != IRU_ROAST
85
+ assert "いる" in bare.kana and bare.common
86
 
87
 
88
  def test_aru_naru_kore(lexicon):