japanese-learning-avatar / tests /test_jmdict.py
WolfDavid's picture
test(02-03): add failing JMdict lookup tests
e594cd3
Raw History Blame
5.38 kB
"""The compact JMdict is a read-only singleton whose ranked lookup resolves homographs.
02-RESEARCH.md Β§ Common Pitfalls 2 measured the naive join picking η…Žγ‚‹ for いる, ζˆ–γ‚‹ for ある
and 酔う for γ‚ˆγ†. The ranking (reading match -> on a JLPT list -> JMdict ``common`` -> file
order) is what makes the Sudachi-lemma -> JMdict-id step deterministic and right; these tests
pin the homographs research measured after the fix, and 02-05's fixture set pins the rest.
"""
from __future__ import annotations
import time
import pytest
from japanese_avatar.nlp import jmdict
# Ids from 02-RESEARCH.md Β§ Q2 - Join strategy (spot checks after the ranking fix) and
# data/jmdict/README.md; all asserted by tests/test_data_assets.py to exist in the file.
IRU_ORU = 1577980 # ε±…γ‚‹
IRU_ROAST = 1391500 # η…Žγ‚‹ - what the first draft picked for いる
ARU = 1296400 # ζœ‰γ‚‹
NARU = 1375610 # ζˆγ‚‹
KONNICHIWA = 1289400 # こんにけは (kanji forms 今ζ—₯は rK / 今ζ—₯わ sK)
KYOU = 1579110 # 今ζ—₯, readings きょう and こんにけ
HANASERU = 1562360 # 話せる - its own entry, on no JLPT list
ENTRIES = 218672
MIN_HEADWORD_KEYS = 400_000 # research counted 453,630 kanji + kana headword keys
def _level_of() -> dict[int, int]:
"""The real JLPT join when levels.py (Task 2) exists, else the two ids these tests need."""
try:
from japanese_avatar.nlp.levels import vocab_levels
except ImportError:
return {ARU: 5, NARU: 5, IRU_ORU: 5}
return dict(vocab_levels())
@pytest.fixture(scope="module")
def lexicon() -> jmdict.Lexicon:
t = time.perf_counter()
lex = jmdict.load()
seconds = time.perf_counter() - t
print(f"\ncompact JMdict load(): {seconds:.2f} s")
assert seconds < 5, f"load took {seconds:.2f} s; research measured ~1.8 s"
return lex
def test_load_is_a_singleton(lexicon):
assert jmdict.load() is lexicon
assert jmdict.load() is jmdict.load()
assert len(lexicon.by_id) == ENTRIES
# warmup() is idempotent on the cache: a second call costs nothing measurable.
assert jmdict.warmup() < 0.05
def test_indexes_cover_headwords(lexicon):
assert IRU_ORU in {e.id for e in lexicon.by_kanji["ε±…γ‚‹"]}
iru_ids = {e.id for e in lexicon.by_kana["いる"]}
assert IRU_ORU in iru_ids
assert len(iru_ids) > 1, "いる must index its homographs, not one entry"
assert KONNICHIWA in {e.id for e in lexicon.by_kana["こんにけは"]}
assert len(lexicon.by_kanji) + len(lexicon.by_kana) > MIN_HEADWORD_KEYS
def test_entries_are_frozen_records(lexicon):
e = lexicon.by_id[IRU_ORU]
assert isinstance(e, jmdict.Entry)
assert e.id == IRU_ORU and "ε±…γ‚‹" in e.kanji and "いる" in e.kana
assert isinstance(e.common, bool) and e.common is True
assert isinstance(e.order, int) and 0 <= e.order < ENTRIES
assert all(isinstance(sense, tuple) for sense in e.glosses)
with pytest.raises(AttributeError):
e.id = 0 # type: ignore[misc]
def test_iru_is_oru_not_iru_roast(lexicon):
hit = jmdict.lookup("いる", "ε±…γ‚‹", "いる", {IRU_ORU: 5})
assert hit is not None and hit.id == IRU_ORU
# Without any JLPT knowledge the reading match plus `common` still beats η…Žγ‚‹.
bare = jmdict.lookup("いる", "いる", "いる", {})
assert bare is not None and bare.id != IRU_ROAST
assert bare.id == IRU_ORU
def test_aru_naru_kore(lexicon):
level_of = _level_of()
aru = jmdict.lookup("ある", "ζœ‰γ‚‹", "ある", level_of)
assert aru is not None and aru.id == ARU
naru = jmdict.lookup("γͺγ‚‹", "ζˆγ‚‹", "γͺγ‚‹", level_of)
assert naru is not None and naru.id == NARU
kore = jmdict.lookup("γ“γ‚Œ", "γ“γ‚Œ", "γ“γ‚Œ", level_of)
assert kore is not None and "γ“γ‚Œ" in kore.kana
def test_kana_headword_fallback(lexicon):
hit = jmdict.lookup("こんにけは", "今ζ—₯は", "こんにけは", {})
assert hit is not None and hit.id == KONNICHIWA
def test_hanaseru_has_its_own_entry(lexicon):
hanaseru = jmdict.lookup("話せる", "話せる", "はγͺせる", {})
assert hanaseru is not None and hanaseru.id == HANASERU
glosses = [g for sense in jmdict.glosses_for(hanaseru) for g in sense]
assert any("to be able to speak" in g for g in glosses), glosses
hanasu = jmdict.lookup("話す", "話す", "はγͺす", {})
assert hanasu is not None and hanasu.id != HANASERU
def test_kyou(lexicon):
for reading in ("きょう", "こんにけ"):
hit = jmdict.lookup("今ζ—₯", "今ζ—₯", reading, {})
assert hit is not None and hit.id == KYOU, (reading, hit)
def test_glosses_for_caps_senses(lexicon):
entry = lexicon.by_id[IRU_ORU]
assert len(jmdict.glosses_for(entry, max_senses=1)) == 1
assert len(jmdict.glosses_for(entry)) <= 3
assert all(isinstance(g, str) and g for sense in jmdict.glosses_for(entry) for g in sense)
def test_no_hit_returns_none(lexicon):
assert jmdict.lookup("ο½˜ο½™ο½š", "ο½˜ο½™ο½š", "", {}) is None
assert jmdict.lookup("", "", "", {}) is None
def test_lookup_is_fast(lexicon):
level_of = {IRU_ORU: 5}
t = time.perf_counter()
for _ in range(10_000):
jmdict.lookup("いる", "ε±…γ‚‹", "いる", level_of)
ms = (time.perf_counter() - t) * 1000
print(f"\n10,000 lookups of ε±…γ‚‹: {ms:.1f} ms")
assert ms < 50, f"{ms:.1f} ms for 10k lookups"