File size: 5,376 Bytes
e594cd3
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
"""The compact JMdict is a read-only singleton whose ranked lookup resolves homographs.

02-RESEARCH.md Β§ Common Pitfalls 2 measured the naive join picking η…Žγ‚‹ for いる, ζˆ–γ‚‹ for ある
and 酔う for γ‚ˆγ†. The ranking (reading match -> on a JLPT list -> JMdict ``common`` -> file
order) is what makes the Sudachi-lemma -> JMdict-id step deterministic and right; these tests
pin the homographs research measured after the fix, and 02-05's fixture set pins the rest.
"""

from __future__ import annotations

import time

import pytest

from japanese_avatar.nlp import jmdict

# Ids from 02-RESEARCH.md Β§ Q2 - Join strategy (spot checks after the ranking fix) and
# data/jmdict/README.md; all asserted by tests/test_data_assets.py to exist in the file.
IRU_ORU = 1577980  # ε±…γ‚‹
IRU_ROAST = 1391500  # η…Žγ‚‹ - what the first draft picked for いる
ARU = 1296400  # ζœ‰γ‚‹
NARU = 1375610  # ζˆγ‚‹
KONNICHIWA = 1289400  # こんにけは (kanji forms 今ζ—₯は rK / 今ζ—₯わ sK)
KYOU = 1579110  # 今ζ—₯, readings きょう and こんにけ
HANASERU = 1562360  # 話せる - its own entry, on no JLPT list
ENTRIES = 218672
MIN_HEADWORD_KEYS = 400_000  # research counted 453,630 kanji + kana headword keys


def _level_of() -> dict[int, int]:
    """The real JLPT join when levels.py (Task 2) exists, else the two ids these tests need."""
    try:
        from japanese_avatar.nlp.levels import vocab_levels
    except ImportError:
        return {ARU: 5, NARU: 5, IRU_ORU: 5}
    return dict(vocab_levels())


@pytest.fixture(scope="module")
def lexicon() -> jmdict.Lexicon:
    t = time.perf_counter()
    lex = jmdict.load()
    seconds = time.perf_counter() - t
    print(f"\ncompact JMdict load(): {seconds:.2f} s")
    assert seconds < 5, f"load took {seconds:.2f} s; research measured ~1.8 s"
    return lex


def test_load_is_a_singleton(lexicon):
    assert jmdict.load() is lexicon
    assert jmdict.load() is jmdict.load()
    assert len(lexicon.by_id) == ENTRIES
    # warmup() is idempotent on the cache: a second call costs nothing measurable.
    assert jmdict.warmup() < 0.05


def test_indexes_cover_headwords(lexicon):
    assert IRU_ORU in {e.id for e in lexicon.by_kanji["ε±…γ‚‹"]}
    iru_ids = {e.id for e in lexicon.by_kana["いる"]}
    assert IRU_ORU in iru_ids
    assert len(iru_ids) > 1, "いる must index its homographs, not one entry"
    assert KONNICHIWA in {e.id for e in lexicon.by_kana["こんにけは"]}
    assert len(lexicon.by_kanji) + len(lexicon.by_kana) > MIN_HEADWORD_KEYS


def test_entries_are_frozen_records(lexicon):
    e = lexicon.by_id[IRU_ORU]
    assert isinstance(e, jmdict.Entry)
    assert e.id == IRU_ORU and "ε±…γ‚‹" in e.kanji and "いる" in e.kana
    assert isinstance(e.common, bool) and e.common is True
    assert isinstance(e.order, int) and 0 <= e.order < ENTRIES
    assert all(isinstance(sense, tuple) for sense in e.glosses)
    with pytest.raises(AttributeError):
        e.id = 0  # type: ignore[misc]


def test_iru_is_oru_not_iru_roast(lexicon):
    hit = jmdict.lookup("いる", "ε±…γ‚‹", "いる", {IRU_ORU: 5})
    assert hit is not None and hit.id == IRU_ORU
    # Without any JLPT knowledge the reading match plus `common` still beats η…Žγ‚‹.
    bare = jmdict.lookup("いる", "いる", "いる", {})
    assert bare is not None and bare.id != IRU_ROAST
    assert bare.id == IRU_ORU


def test_aru_naru_kore(lexicon):
    level_of = _level_of()
    aru = jmdict.lookup("ある", "ζœ‰γ‚‹", "ある", level_of)
    assert aru is not None and aru.id == ARU
    naru = jmdict.lookup("γͺγ‚‹", "ζˆγ‚‹", "γͺγ‚‹", level_of)
    assert naru is not None and naru.id == NARU
    kore = jmdict.lookup("γ“γ‚Œ", "γ“γ‚Œ", "γ“γ‚Œ", level_of)
    assert kore is not None and "γ“γ‚Œ" in kore.kana


def test_kana_headword_fallback(lexicon):
    hit = jmdict.lookup("こんにけは", "今ζ—₯は", "こんにけは", {})
    assert hit is not None and hit.id == KONNICHIWA


def test_hanaseru_has_its_own_entry(lexicon):
    hanaseru = jmdict.lookup("話せる", "話せる", "はγͺせる", {})
    assert hanaseru is not None and hanaseru.id == HANASERU
    glosses = [g for sense in jmdict.glosses_for(hanaseru) for g in sense]
    assert any("to be able to speak" in g for g in glosses), glosses
    hanasu = jmdict.lookup("話す", "話す", "はγͺす", {})
    assert hanasu is not None and hanasu.id != HANASERU


def test_kyou(lexicon):
    for reading in ("きょう", "こんにけ"):
        hit = jmdict.lookup("今ζ—₯", "今ζ—₯", reading, {})
        assert hit is not None and hit.id == KYOU, (reading, hit)


def test_glosses_for_caps_senses(lexicon):
    entry = lexicon.by_id[IRU_ORU]
    assert len(jmdict.glosses_for(entry, max_senses=1)) == 1
    assert len(jmdict.glosses_for(entry)) <= 3
    assert all(isinstance(g, str) and g for sense in jmdict.glosses_for(entry) for g in sense)


def test_no_hit_returns_none(lexicon):
    assert jmdict.lookup("ο½˜ο½™ο½š", "ο½˜ο½™ο½š", "", {}) is None
    assert jmdict.lookup("", "", "", {}) is None


def test_lookup_is_fast(lexicon):
    level_of = {IRU_ORU: 5}
    t = time.perf_counter()
    for _ in range(10_000):
        jmdict.lookup("いる", "ε±…γ‚‹", "いる", level_of)
    ms = (time.perf_counter() - t) * 1000
    print(f"\n10,000 lookups of ε±…γ‚‹: {ms:.1f} ms")
    assert ms < 50, f"{ms:.1f} ms for 10k lookups"