Spaces:
Running on Zero
Running on Zero
File size: 5,376 Bytes
e594cd3 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 | """The compact JMdict is a read-only singleton whose ranked lookup resolves homographs.
02-RESEARCH.md Β§ Common Pitfalls 2 measured the naive join picking η
γ for γγ, ζγ for γγ
and ι
γ for γγ. The ranking (reading match -> on a JLPT list -> JMdict ``common`` -> file
order) is what makes the Sudachi-lemma -> JMdict-id step deterministic and right; these tests
pin the homographs research measured after the fix, and 02-05's fixture set pins the rest.
"""
from __future__ import annotations
import time
import pytest
from japanese_avatar.nlp import jmdict
# Ids from 02-RESEARCH.md Β§ Q2 - Join strategy (spot checks after the ranking fix) and
# data/jmdict/README.md; all asserted by tests/test_data_assets.py to exist in the file.
IRU_ORU = 1577980 # ε±
γ
IRU_ROAST = 1391500 # η
γ - what the first draft picked for γγ
ARU = 1296400 # ζγ
NARU = 1375610 # ζγ
KONNICHIWA = 1289400 # γγγ«γ‘γ― (kanji forms δ»ζ₯γ― rK / δ»ζ₯γ sK)
KYOU = 1579110 # δ»ζ₯, readings γγγ and γγγ«γ‘
HANASERU = 1562360 # θ©±γγ - its own entry, on no JLPT list
ENTRIES = 218672
MIN_HEADWORD_KEYS = 400_000 # research counted 453,630 kanji + kana headword keys
def _level_of() -> dict[int, int]:
"""The real JLPT join when levels.py (Task 2) exists, else the two ids these tests need."""
try:
from japanese_avatar.nlp.levels import vocab_levels
except ImportError:
return {ARU: 5, NARU: 5, IRU_ORU: 5}
return dict(vocab_levels())
@pytest.fixture(scope="module")
def lexicon() -> jmdict.Lexicon:
t = time.perf_counter()
lex = jmdict.load()
seconds = time.perf_counter() - t
print(f"\ncompact JMdict load(): {seconds:.2f} s")
assert seconds < 5, f"load took {seconds:.2f} s; research measured ~1.8 s"
return lex
def test_load_is_a_singleton(lexicon):
assert jmdict.load() is lexicon
assert jmdict.load() is jmdict.load()
assert len(lexicon.by_id) == ENTRIES
# warmup() is idempotent on the cache: a second call costs nothing measurable.
assert jmdict.warmup() < 0.05
def test_indexes_cover_headwords(lexicon):
assert IRU_ORU in {e.id for e in lexicon.by_kanji["ε±
γ"]}
iru_ids = {e.id for e in lexicon.by_kana["γγ"]}
assert IRU_ORU in iru_ids
assert len(iru_ids) > 1, "γγ must index its homographs, not one entry"
assert KONNICHIWA in {e.id for e in lexicon.by_kana["γγγ«γ‘γ―"]}
assert len(lexicon.by_kanji) + len(lexicon.by_kana) > MIN_HEADWORD_KEYS
def test_entries_are_frozen_records(lexicon):
e = lexicon.by_id[IRU_ORU]
assert isinstance(e, jmdict.Entry)
assert e.id == IRU_ORU and "ε±
γ" in e.kanji and "γγ" in e.kana
assert isinstance(e.common, bool) and e.common is True
assert isinstance(e.order, int) and 0 <= e.order < ENTRIES
assert all(isinstance(sense, tuple) for sense in e.glosses)
with pytest.raises(AttributeError):
e.id = 0 # type: ignore[misc]
def test_iru_is_oru_not_iru_roast(lexicon):
hit = jmdict.lookup("γγ", "ε±
γ", "γγ", {IRU_ORU: 5})
assert hit is not None and hit.id == IRU_ORU
# Without any JLPT knowledge the reading match plus `common` still beats η
γ.
bare = jmdict.lookup("γγ", "γγ", "γγ", {})
assert bare is not None and bare.id != IRU_ROAST
assert bare.id == IRU_ORU
def test_aru_naru_kore(lexicon):
level_of = _level_of()
aru = jmdict.lookup("γγ", "ζγ", "γγ", level_of)
assert aru is not None and aru.id == ARU
naru = jmdict.lookup("γͺγ", "ζγ", "γͺγ", level_of)
assert naru is not None and naru.id == NARU
kore = jmdict.lookup("γγ", "γγ", "γγ", level_of)
assert kore is not None and "γγ" in kore.kana
def test_kana_headword_fallback(lexicon):
hit = jmdict.lookup("γγγ«γ‘γ―", "δ»ζ₯γ―", "γγγ«γ‘γ―", {})
assert hit is not None and hit.id == KONNICHIWA
def test_hanaseru_has_its_own_entry(lexicon):
hanaseru = jmdict.lookup("θ©±γγ", "θ©±γγ", "γ―γͺγγ", {})
assert hanaseru is not None and hanaseru.id == HANASERU
glosses = [g for sense in jmdict.glosses_for(hanaseru) for g in sense]
assert any("to be able to speak" in g for g in glosses), glosses
hanasu = jmdict.lookup("θ©±γ", "θ©±γ", "γ―γͺγ", {})
assert hanasu is not None and hanasu.id != HANASERU
def test_kyou(lexicon):
for reading in ("γγγ", "γγγ«γ‘"):
hit = jmdict.lookup("δ»ζ₯", "δ»ζ₯", reading, {})
assert hit is not None and hit.id == KYOU, (reading, hit)
def test_glosses_for_caps_senses(lexicon):
entry = lexicon.by_id[IRU_ORU]
assert len(jmdict.glosses_for(entry, max_senses=1)) == 1
assert len(jmdict.glosses_for(entry)) <= 3
assert all(isinstance(g, str) and g for sense in jmdict.glosses_for(entry) for g in sense)
def test_no_hit_returns_none(lexicon):
assert jmdict.lookup("ο½ο½ο½", "ο½ο½ο½", "", {}) is None
assert jmdict.lookup("", "", "", {}) is None
def test_lookup_is_fast(lexicon):
level_of = {IRU_ORU: 5}
t = time.perf_counter()
for _ in range(10_000):
jmdict.lookup("γγ", "ε±
γ", "γγ", level_of)
ms = (time.perf_counter() - t) * 1000
print(f"\n10,000 lookups of ε±
γ: {ms:.1f} ms")
assert ms < 50, f"{ms:.1f} ms for 10k lookups"
|