"""The compact JMdict is a read-only singleton whose ranked lookup resolves homographs. 02-RESEARCH.md § Common Pitfalls 2 measured the naive join picking 煎る for いる, 或る for ある and 酔う for よう. The ranking (reading match -> on a JLPT list -> JMdict ``common`` -> file order) is what makes the Sudachi-lemma -> JMdict-id step deterministic and right; these tests pin the homographs research measured after the fix, and 02-05's fixture set pins the rest. """ from __future__ import annotations import time import pytest from japanese_avatar.nlp import jmdict # Ids from 02-RESEARCH.md § Q2 - Join strategy (spot checks after the ranking fix) and # data/jmdict/README.md; all asserted by tests/test_data_assets.py to exist in the file. IRU_ORU = 1577980 # 居る IRU_ROAST = 1391500 # 煎る - what the first draft picked for いる ARU = 1296400 # 有る NARU = 1375610 # 成る KONNICHIWA = 1289400 # こんにちは (kanji forms 今日は rK / 今日わ sK) KYOU = 1579110 # 今日, readings きょう and こんにち HANASERU = 1562360 # 話せる - its own entry, on no JLPT list ENTRIES = 218672 MIN_HEADWORD_KEYS = 400_000 # research counted 453,630 kanji + kana headword keys def _level_of() -> dict[int, int]: """The real JLPT join when levels.py (Task 2) exists, else the two ids these tests need.""" try: from japanese_avatar.nlp.levels import vocab_levels except ImportError: return {ARU: 5, NARU: 5, IRU_ORU: 5} return dict(vocab_levels()) @pytest.fixture(scope="module") def lexicon() -> jmdict.Lexicon: t = time.perf_counter() lex = jmdict.load() seconds = time.perf_counter() - t print(f"\ncompact JMdict load(): {seconds:.2f} s") assert seconds < 5, f"load took {seconds:.2f} s; research measured ~1.8 s" return lex def test_load_is_a_singleton(lexicon): assert jmdict.load() is lexicon assert jmdict.load() is jmdict.load() assert len(lexicon.by_id) == ENTRIES # warmup() is idempotent on the cache: a second call costs nothing measurable. assert jmdict.warmup() < 0.05 def test_indexes_cover_headwords(lexicon): assert IRU_ORU in {e.id for e in lexicon.by_kanji["居る"]} iru_ids = {e.id for e in lexicon.by_kana["いる"]} assert IRU_ORU in iru_ids assert len(iru_ids) > 1, "いる must index its homographs, not one entry" assert KONNICHIWA in {e.id for e in lexicon.by_kana["こんにちは"]} assert len(lexicon.by_kanji) + len(lexicon.by_kana) > MIN_HEADWORD_KEYS def test_entries_are_frozen_records(lexicon): e = lexicon.by_id[IRU_ORU] assert isinstance(e, jmdict.Entry) assert e.id == IRU_ORU and "居る" in e.kanji and "いる" in e.kana assert isinstance(e.common, bool) and e.common is True assert isinstance(e.order, int) and 0 <= e.order < ENTRIES assert all(isinstance(sense, tuple) for sense in e.glosses) with pytest.raises(AttributeError): e.id = 0 # type: ignore[misc] def test_iru_is_oru_not_iru_roast(lexicon): hit = jmdict.lookup("いる", "居る", "いる", {IRU_ORU: 5}) assert hit is not None and hit.id == IRU_ORU # Without any JLPT knowledge the reading match plus `common` still beats 煎る. bare = jmdict.lookup("いる", "いる", "いる", {}) assert bare is not None and bare.id != IRU_ROAST assert bare.id == IRU_ORU def test_aru_naru_kore(lexicon): level_of = _level_of() aru = jmdict.lookup("ある", "有る", "ある", level_of) assert aru is not None and aru.id == ARU naru = jmdict.lookup("なる", "成る", "なる", level_of) assert naru is not None and naru.id == NARU kore = jmdict.lookup("これ", "これ", "これ", level_of) assert kore is not None and "これ" in kore.kana def test_kana_headword_fallback(lexicon): hit = jmdict.lookup("こんにちは", "今日は", "こんにちは", {}) assert hit is not None and hit.id == KONNICHIWA def test_hanaseru_has_its_own_entry(lexicon): hanaseru = jmdict.lookup("話せる", "話せる", "はなせる", {}) assert hanaseru is not None and hanaseru.id == HANASERU glosses = [g for sense in jmdict.glosses_for(hanaseru) for g in sense] assert any("to be able to speak" in g for g in glosses), glosses hanasu = jmdict.lookup("話す", "話す", "はなす", {}) assert hanasu is not None and hanasu.id != HANASERU def test_kyou(lexicon): for reading in ("きょう", "こんにち"): hit = jmdict.lookup("今日", "今日", reading, {}) assert hit is not None and hit.id == KYOU, (reading, hit) def test_glosses_for_caps_senses(lexicon): entry = lexicon.by_id[IRU_ORU] assert len(jmdict.glosses_for(entry, max_senses=1)) == 1 assert len(jmdict.glosses_for(entry)) <= 3 assert all(isinstance(g, str) and g for sense in jmdict.glosses_for(entry) for g in sense) def test_no_hit_returns_none(lexicon): assert jmdict.lookup("xyz", "xyz", "", {}) is None assert jmdict.lookup("", "", "", {}) is None def test_lookup_is_fast(lexicon): level_of = {IRU_ORU: 5} t = time.perf_counter() for _ in range(10_000): jmdict.lookup("いる", "居る", "いる", level_of) ms = (time.perf_counter() - t) * 1000 print(f"\n10,000 lookups of 居る: {ms:.1f} ms") assert ms < 50, f"{ms:.1f} ms for 10k lookups"