Spaces:
Running on Zero
Running on Zero
File size: 10,404 Bytes
fbd985f | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 195 196 197 198 199 200 201 202 203 204 205 206 207 208 209 210 211 212 213 214 215 216 217 218 219 220 221 222 223 224 225 226 227 228 229 230 231 232 233 234 235 236 237 238 239 240 241 242 243 244 245 246 247 248 249 250 251 252 253 254 255 256 257 258 259 260 261 262 263 264 265 266 267 268 269 270 271 272 273 274 275 276 277 278 279 280 281 282 283 284 285 286 287 288 289 290 291 | """Mode-C morphemes -> whole-word units (D-05 joining rule A0-A7), one named test per rule.
Every test tokenises with the real Sudachi core dictionary (loaded once per module) and asserts
on ``build_units``. The unit SPLITS asserted here are the D-05 contract from 02-02-PLAN.md; only
Sudachi's morpheme output (surfaces, readings, POS tuples) is "observed". If a split disagrees
with a test, the rule table or its implementation is wrong - never the expectation.
"""
from __future__ import annotations
import time
import pytest
from japanese_avatar.nlp.tokenizer import morphemes, warmup
from japanese_avatar.nlp.units import TAPPABLE_POS, build_units
SENTENCES = [
"食べました",
"食べさせられた",
"美味しかったです",
"学生です",
"きれいじゃなかった",
"食べたくなかった",
"美味しくない",
"人気のない通り",
"話せるようになりたい",
"行かなければならない",
"食べて",
"読んで",
"読んでください",
"日本語を勉強しています",
"勉強します",
"田中さんは東京に",
"三人で",
"一人で",
"お願いします",
"こんにちは",
"今日は天気がいいですね。",
"はい、そうです。",
]
@pytest.fixture(scope="module", autouse=True)
def tokenizer_loaded():
t0 = time.perf_counter()
seconds = warmup()
print(f"\nsudachi core load: {seconds:.3f} s (warmup call {time.perf_counter() - t0:.3f} s)")
def units_of(text: str) -> list[dict]:
return build_units(morphemes(text))
def surfaces(units: list[dict]) -> list[str]:
return [u["surface"] for u in units]
def tappable(units: list[dict]) -> list[bool]:
return [u["tappable"] for u in units]
def test_morphemes_are_plain_dicts():
ms = morphemes("食べました")
assert len(ms) == 3
for m in ms:
assert set(m) == {"surface", "reading", "lemma", "norm", "pos", "oov"}
assert isinstance(m["pos"], list) and len(m["pos"]) == 6
assert all(isinstance(p, str) for p in m["pos"])
assert isinstance(m["oov"], bool)
assert type(m) is dict
assert ms[0]["surface"] == "食べ"
assert ms[0]["reading"] == "タベ" # katakana, as Sudachi returns it
assert ms[0]["lemma"] == "食べる"
assert ms[0]["pos"][0] == "動詞"
assert morphemes("") == []
def test_a1_auxiliary_chain():
units = units_of("食べました")
assert len(units) == 1
u = units[0]
assert u == {
"surface": "食べました",
"reading": "たべました",
"lemma": "食べる",
"level_key": "食べる",
"pos": "動詞",
"tappable": True,
"is_name": False,
"start": 0,
"end": 5,
"morpheme_count": 3,
}
units = units_of("食べさせられた")
assert surfaces(units) == ["食べさせられた"]
assert units[0]["lemma"] == "食べる"
assert units[0]["morpheme_count"] == 4
def test_a1_stops_at_desu_da():
units = units_of("美味しかったです")
assert surfaces(units) == ["美味しかった", "です"]
assert units[0]["lemma"] == "美味しい"
assert tappable(units) == [True, False]
units = units_of("学生です")
assert surfaces(units) == ["学生", "です"]
assert tappable(units) == [True, False]
units = units_of("きれいじゃなかった")
assert surfaces(units) == ["きれい", "じゃ", "なかった"]
assert tappable(units) == [True, False, True]
def test_a2_nai_attaches_after_aux():
units = units_of("食べたくなかった")
assert surfaces(units) == ["食べたくなかった"]
assert units[0]["lemma"] == "食べる"
assert units[0]["reading"] == "たべたくなかった"
# The 形容詞 tail: 美味しく is 形容詞,一般 / 連用形-一般, so ない attaches under A2.
units = units_of("美味しくない")
assert surfaces(units) == ["美味しくない"]
assert units[0]["lemma"] == "美味しい"
assert units[0]["reading"] == "おいしくない"
assert units[0]["tappable"] is True
def test_a2_nai_after_particle_is_a_word():
units = units_of("人気のない通り")
assert surfaces(units) == ["人気", "の", "ない", "通り"]
assert tappable(units) == [True, False, True, True]
assert units[2]["lemma"] == "ない"
assert units[2]["level_key"] == "無い"
assert units[2]["pos"] == "形容詞"
def test_a0_breaker_units_accept_nothing():
# じゃ (lemma だ) is a breaker: なかっ may not join it although じゃ is 助動詞; た then
# joins なかっ via A1 with a 形容詞 tail.
units = units_of("きれいじゃなかった")
assert surfaces(units) == ["きれい", "じゃ", "なかった"]
assert units[1]["lemma"] == "だ"
assert units[2]["lemma"] == "ない"
assert units[2]["morpheme_count"] == 2
# よう (形状詞,助動詞語幹) is a breaker: に (助動詞, lemma だ) is its own unit after it.
units = units_of("話せるようになりたい")
assert surfaces(units) == ["話せる", "よう", "に", "なりたい"]
# の (助詞) is a breaker: ない is its own unit after it.
units = units_of("人気のない通り")
assert surfaces(units)[1:3] == ["の", "ない"]
def test_a3_te_de():
units = units_of("食べて")
assert surfaces(units) == ["食べて"]
assert units[0]["lemma"] == "食べる"
units = units_of("読んで")
assert surfaces(units) == ["読んで"]
assert units[0]["lemma"] == "読む"
assert units[0]["reading"] == "よんで"
# ば is 助詞,接続助詞 but not て/で: it starts its own non-tappable unit; A0 keeps なら
# off it, then ない (助動詞) joins なら via A1.
units = units_of("行かなければならない")
assert surfaces(units) == ["行かなけれ", "ば", "ならない"]
assert tappable(units) == [True, False, True]
assert units[0]["lemma"] == "行く"
assert units[2]["lemma"] == "なる"
def test_a4_te_auxiliary_verb():
units = units_of("読んでください")
assert surfaces(units) == ["読んでください"]
assert units[0]["lemma"] == "読む"
assert units[0]["reading"] == "よんでください"
units = units_of("日本語を勉強しています")
assert surfaces(units) == ["日本語", "を", "勉強しています"]
last = units[-1]
assert last["lemma"] == "勉強する"
assert last["level_key"] == "勉強"
assert last["pos"] == "名詞"
assert last["tappable"] is True
assert last["reading"] == "べんきょうしています"
assert last["morpheme_count"] == 5
def test_a5_suru_noun():
units = units_of("勉強します")
assert surfaces(units) == ["勉強します"]
assert units[0]["lemma"] == "勉強する"
assert units[0]["level_key"] == "勉強"
units = units_of("勉強")
assert surfaces(units) == ["勉強"]
assert units[0]["lemma"] == "勉強"
def test_a6_suffix():
units = units_of("田中さんは東京に")
assert surfaces(units) == ["田中さん", "は", "東京", "に"]
assert units[0]["is_name"] is True
assert units[0]["lemma"] == "田中"
assert units[0]["reading"] == "たなかさん"
assert units[0]["tappable"] is True
assert units[2]["is_name"] is True
assert units[2]["lemma"] == "東京"
assert [u["is_name"] for u in units] == [True, False, True, False]
units = units_of("三人で")
assert surfaces(units) == ["三人", "で"]
assert units[0]["tappable"] is True
assert units[0]["lemma"] == "三"
assert units[0]["reading"] == "さんにん"
# Sudachi's normalized_form for the OOV numeral 三 is the digit "3"; recorded, not judged.
assert units[0]["level_key"] == "3"
assert units[1]["tappable"] is False
units = units_of("一人で")
assert surfaces(units) == ["一人", "で"]
assert units[0]["morpheme_count"] == 1
def test_a7_prefix():
units = units_of("お願いします")
assert surfaces(units) == ["お願い", "します"]
assert tappable(units) == [True, True]
assert units[0]["lemma"] == "願う"
assert units[0]["reading"] == "おねがい"
assert units[0]["pos"] == "動詞"
assert units[0]["morpheme_count"] == 2
assert units[1]["lemma"] == "する"
def test_non_tappable_units():
units = units_of("話せるようになりたい")
by_surface = {u["surface"]: u for u in units}
assert by_surface["よう"]["tappable"] is False
assert by_surface["に"]["tappable"] is False
assert by_surface["話せる"]["tappable"] is True
assert by_surface["話せる"]["lemma"] == "話せる"
assert by_surface["話せる"]["level_key"] == "話す"
assert by_surface["なりたい"]["lemma"] == "なる"
assert by_surface["なりたい"]["tappable"] is True
for mark in ("。", "、"):
units = units_of(mark)
assert len(units) == 1
assert units[0]["tappable"] is False
assert units[0]["pos"] == "補助記号"
units = units_of("こんにちは")
assert len(units) == 1
assert units[0]["tappable"] is True
assert units[0]["lemma"] == "こんにちは"
assert units[0]["level_key"] == "今日は"
assert units[0]["pos"] == "感動詞"
assert "助詞" not in TAPPABLE_POS and "助動詞" not in TAPPABLE_POS
assert {"名詞", "動詞", "形容詞", "形状詞"} <= TAPPABLE_POS
@pytest.mark.parametrize("text", SENTENCES)
def test_offsets_tile_the_text(text):
units = units_of(text)
assert units, text
assert "".join(u["surface"] for u in units) == text
assert units[0]["start"] == 0
for prev, cur in zip(units, units[1:], strict=False):
assert cur["start"] == prev["end"]
for u in units:
assert u["end"] == u["start"] + len(u["surface"])
assert u["morpheme_count"] >= 1
assert text[u["start"] : u["end"]] == u["surface"]
assert units[-1]["end"] == len(text)
def test_build_units_is_pure_over_dicts():
ms = morphemes("食べました")
before = [dict(m) for m in ms]
units = build_units(ms)
assert ms == before # input not mutated
assert build_units(ms) == units # deterministic
assert build_units([]) == []
assert all(type(u) is dict for u in units)
|