File size: 10,404 Bytes
fbd985f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
"""Mode-C morphemes -> whole-word units (D-05 joining rule A0-A7), one named test per rule.

Every test tokenises with the real Sudachi core dictionary (loaded once per module) and asserts
on ``build_units``. The unit SPLITS asserted here are the D-05 contract from 02-02-PLAN.md; only
Sudachi's morpheme output (surfaces, readings, POS tuples) is "observed". If a split disagrees
with a test, the rule table or its implementation is wrong - never the expectation.
"""

from __future__ import annotations

import time

import pytest

from japanese_avatar.nlp.tokenizer import morphemes, warmup
from japanese_avatar.nlp.units import TAPPABLE_POS, build_units

SENTENCES = [
    "食べました",
    "食べさせられた",
    "美味しかったです",
    "学生です",
    "きれいじゃなかった",
    "食べたくなかった",
    "美味しくない",
    "人気のない通り",
    "話せるようになりたい",
    "行かなければならない",
    "食べて",
    "読んで",
    "読んでください",
    "日本語を勉強しています",
    "勉強します",
    "田中さんは東京に",
    "三人で",
    "一人で",
    "お願いします",
    "こんにちは",
    "今日は天気がいいですね。",
    "はい、そうです。",
]


@pytest.fixture(scope="module", autouse=True)
def tokenizer_loaded():
    t0 = time.perf_counter()
    seconds = warmup()
    print(f"\nsudachi core load: {seconds:.3f} s (warmup call {time.perf_counter() - t0:.3f} s)")


def units_of(text: str) -> list[dict]:
    return build_units(morphemes(text))


def surfaces(units: list[dict]) -> list[str]:
    return [u["surface"] for u in units]


def tappable(units: list[dict]) -> list[bool]:
    return [u["tappable"] for u in units]


def test_morphemes_are_plain_dicts():
    ms = morphemes("食べました")
    assert len(ms) == 3
    for m in ms:
        assert set(m) == {"surface", "reading", "lemma", "norm", "pos", "oov"}
        assert isinstance(m["pos"], list) and len(m["pos"]) == 6
        assert all(isinstance(p, str) for p in m["pos"])
        assert isinstance(m["oov"], bool)
        assert type(m) is dict
    assert ms[0]["surface"] == "食べ"
    assert ms[0]["reading"] == "タベ"  # katakana, as Sudachi returns it
    assert ms[0]["lemma"] == "食べる"
    assert ms[0]["pos"][0] == "動詞"
    assert morphemes("") == []


def test_a1_auxiliary_chain():
    units = units_of("食べました")
    assert len(units) == 1
    u = units[0]
    assert u == {
        "surface": "食べました",
        "reading": "たべました",
        "lemma": "食べる",
        "level_key": "食べる",
        "pos": "動詞",
        "tappable": True,
        "is_name": False,
        "start": 0,
        "end": 5,
        "morpheme_count": 3,
    }
    units = units_of("食べさせられた")
    assert surfaces(units) == ["食べさせられた"]
    assert units[0]["lemma"] == "食べる"
    assert units[0]["morpheme_count"] == 4


def test_a1_stops_at_desu_da():
    units = units_of("美味しかったです")
    assert surfaces(units) == ["美味しかった", "です"]
    assert units[0]["lemma"] == "美味しい"
    assert tappable(units) == [True, False]

    units = units_of("学生です")
    assert surfaces(units) == ["学生", "です"]
    assert tappable(units) == [True, False]

    units = units_of("きれいじゃなかった")
    assert surfaces(units) == ["きれい", "じゃ", "なかった"]
    assert tappable(units) == [True, False, True]


def test_a2_nai_attaches_after_aux():
    units = units_of("食べたくなかった")
    assert surfaces(units) == ["食べたくなかった"]
    assert units[0]["lemma"] == "食べる"
    assert units[0]["reading"] == "たべたくなかった"

    # The 形容詞 tail: 美味しく is 形容詞,一般 / 連用形-一般, so ない attaches under A2.
    units = units_of("美味しくない")
    assert surfaces(units) == ["美味しくない"]
    assert units[0]["lemma"] == "美味しい"
    assert units[0]["reading"] == "おいしくない"
    assert units[0]["tappable"] is True


def test_a2_nai_after_particle_is_a_word():
    units = units_of("人気のない通り")
    assert surfaces(units) == ["人気", "の", "ない", "通り"]
    assert tappable(units) == [True, False, True, True]
    assert units[2]["lemma"] == "ない"
    assert units[2]["level_key"] == "無い"
    assert units[2]["pos"] == "形容詞"


def test_a0_breaker_units_accept_nothing():
    # じゃ (lemma だ) is a breaker: なかっ may not join it although じゃ is 助動詞; た then
    # joins なかっ via A1 with a 形容詞 tail.
    units = units_of("きれいじゃなかった")
    assert surfaces(units) == ["きれい", "じゃ", "なかった"]
    assert units[1]["lemma"] == "だ"
    assert units[2]["lemma"] == "ない"
    assert units[2]["morpheme_count"] == 2

    # よう (形状詞,助動詞語幹) is a breaker: に (助動詞, lemma だ) is its own unit after it.
    units = units_of("話せるようになりたい")
    assert surfaces(units) == ["話せる", "よう", "に", "なりたい"]

    # の (助詞) is a breaker: ない is its own unit after it.
    units = units_of("人気のない通り")
    assert surfaces(units)[1:3] == ["の", "ない"]


def test_a3_te_de():
    units = units_of("食べて")
    assert surfaces(units) == ["食べて"]
    assert units[0]["lemma"] == "食べる"

    units = units_of("読んで")
    assert surfaces(units) == ["読んで"]
    assert units[0]["lemma"] == "読む"
    assert units[0]["reading"] == "よんで"

    # ば is 助詞,接続助詞 but not て/で: it starts its own non-tappable unit; A0 keeps なら
    # off it, then ない (助動詞) joins なら via A1.
    units = units_of("行かなければならない")
    assert surfaces(units) == ["行かなけれ", "ば", "ならない"]
    assert tappable(units) == [True, False, True]
    assert units[0]["lemma"] == "行く"
    assert units[2]["lemma"] == "なる"


def test_a4_te_auxiliary_verb():
    units = units_of("読んでください")
    assert surfaces(units) == ["読んでください"]
    assert units[0]["lemma"] == "読む"
    assert units[0]["reading"] == "よんでください"

    units = units_of("日本語を勉強しています")
    assert surfaces(units) == ["日本語", "を", "勉強しています"]
    last = units[-1]
    assert last["lemma"] == "勉強する"
    assert last["level_key"] == "勉強"
    assert last["pos"] == "名詞"
    assert last["tappable"] is True
    assert last["reading"] == "べんきょうしています"
    assert last["morpheme_count"] == 5


def test_a5_suru_noun():
    units = units_of("勉強します")
    assert surfaces(units) == ["勉強します"]
    assert units[0]["lemma"] == "勉強する"
    assert units[0]["level_key"] == "勉強"

    units = units_of("勉強")
    assert surfaces(units) == ["勉強"]
    assert units[0]["lemma"] == "勉強"


def test_a6_suffix():
    units = units_of("田中さんは東京に")
    assert surfaces(units) == ["田中さん", "は", "東京", "に"]
    assert units[0]["is_name"] is True
    assert units[0]["lemma"] == "田中"
    assert units[0]["reading"] == "たなかさん"
    assert units[0]["tappable"] is True
    assert units[2]["is_name"] is True
    assert units[2]["lemma"] == "東京"
    assert [u["is_name"] for u in units] == [True, False, True, False]

    units = units_of("三人で")
    assert surfaces(units) == ["三人", "で"]
    assert units[0]["tappable"] is True
    assert units[0]["lemma"] == "三"
    assert units[0]["reading"] == "さんにん"
    # Sudachi's normalized_form for the OOV numeral 三 is the digit "3"; recorded, not judged.
    assert units[0]["level_key"] == "3"
    assert units[1]["tappable"] is False

    units = units_of("一人で")
    assert surfaces(units) == ["一人", "で"]
    assert units[0]["morpheme_count"] == 1


def test_a7_prefix():
    units = units_of("お願いします")
    assert surfaces(units) == ["お願い", "します"]
    assert tappable(units) == [True, True]
    assert units[0]["lemma"] == "願う"
    assert units[0]["reading"] == "おねがい"
    assert units[0]["pos"] == "動詞"
    assert units[0]["morpheme_count"] == 2
    assert units[1]["lemma"] == "する"


def test_non_tappable_units():
    units = units_of("話せるようになりたい")
    by_surface = {u["surface"]: u for u in units}
    assert by_surface["よう"]["tappable"] is False
    assert by_surface["に"]["tappable"] is False
    assert by_surface["話せる"]["tappable"] is True
    assert by_surface["話せる"]["lemma"] == "話せる"
    assert by_surface["話せる"]["level_key"] == "話す"
    assert by_surface["なりたい"]["lemma"] == "なる"
    assert by_surface["なりたい"]["tappable"] is True

    for mark in ("。", "、"):
        units = units_of(mark)
        assert len(units) == 1
        assert units[0]["tappable"] is False
        assert units[0]["pos"] == "補助記号"

    units = units_of("こんにちは")
    assert len(units) == 1
    assert units[0]["tappable"] is True
    assert units[0]["lemma"] == "こんにちは"
    assert units[0]["level_key"] == "今日は"
    assert units[0]["pos"] == "感動詞"

    assert "助詞" not in TAPPABLE_POS and "助動詞" not in TAPPABLE_POS
    assert {"名詞", "動詞", "形容詞", "形状詞"} <= TAPPABLE_POS


@pytest.mark.parametrize("text", SENTENCES)
def test_offsets_tile_the_text(text):
    units = units_of(text)
    assert units, text
    assert "".join(u["surface"] for u in units) == text
    assert units[0]["start"] == 0
    for prev, cur in zip(units, units[1:], strict=False):
        assert cur["start"] == prev["end"]
    for u in units:
        assert u["end"] == u["start"] + len(u["surface"])
        assert u["morpheme_count"] >= 1
        assert text[u["start"] : u["end"]] == u["surface"]
    assert units[-1]["end"] == len(text)


def test_build_units_is_pure_over_dicts():
    ms = morphemes("食べました")
    before = [dict(m) for m in ms]
    units = build_units(ms)
    assert ms == before  # input not mutated
    assert build_units(ms) == units  # deterministic
    assert build_units([]) == []
    assert all(type(u) is dict for u in units)