WolfDavid's picture
test(02-02): add failing unit-joining tests
fbd985f
Raw History Blame
10.4 kB
"""Mode-C morphemes -> whole-word units (D-05 joining rule A0-A7), one named test per rule.
Every test tokenises with the real Sudachi core dictionary (loaded once per module) and asserts
on ``build_units``. The unit SPLITS asserted here are the D-05 contract from 02-02-PLAN.md; only
Sudachi's morpheme output (surfaces, readings, POS tuples) is "observed". If a split disagrees
with a test, the rule table or its implementation is wrong - never the expectation.
"""
from __future__ import annotations
import time
import pytest
from japanese_avatar.nlp.tokenizer import morphemes, warmup
from japanese_avatar.nlp.units import TAPPABLE_POS, build_units
SENTENCES = [
"食べました",
"食べさせられた",
"美味しかったです",
"学生です",
"きれいじゃなかった",
"食べたくなかった",
"美味しくない",
"人気のない通り",
"話せるようになりたい",
"行かなければならない",
"食べて",
"読んで",
"読んでください",
"日本語を勉強しています",
"勉強します",
"田中さんは東京に",
"三人で",
"一人で",
"お願いします",
"こんにちは",
"今日は天気がいいですね。",
"はい、そうです。",
]
@pytest.fixture(scope="module", autouse=True)
def tokenizer_loaded():
t0 = time.perf_counter()
seconds = warmup()
print(f"\nsudachi core load: {seconds:.3f} s (warmup call {time.perf_counter() - t0:.3f} s)")
def units_of(text: str) -> list[dict]:
return build_units(morphemes(text))
def surfaces(units: list[dict]) -> list[str]:
return [u["surface"] for u in units]
def tappable(units: list[dict]) -> list[bool]:
return [u["tappable"] for u in units]
def test_morphemes_are_plain_dicts():
ms = morphemes("食べました")
assert len(ms) == 3
for m in ms:
assert set(m) == {"surface", "reading", "lemma", "norm", "pos", "oov"}
assert isinstance(m["pos"], list) and len(m["pos"]) == 6
assert all(isinstance(p, str) for p in m["pos"])
assert isinstance(m["oov"], bool)
assert type(m) is dict
assert ms[0]["surface"] == "食べ"
assert ms[0]["reading"] == "タベ" # katakana, as Sudachi returns it
assert ms[0]["lemma"] == "食べる"
assert ms[0]["pos"][0] == "動詞"
assert morphemes("") == []
def test_a1_auxiliary_chain():
units = units_of("食べました")
assert len(units) == 1
u = units[0]
assert u == {
"surface": "食べました",
"reading": "たべました",
"lemma": "食べる",
"level_key": "食べる",
"pos": "動詞",
"tappable": True,
"is_name": False,
"start": 0,
"end": 5,
"morpheme_count": 3,
}
units = units_of("食べさせられた")
assert surfaces(units) == ["食べさせられた"]
assert units[0]["lemma"] == "食べる"
assert units[0]["morpheme_count"] == 4
def test_a1_stops_at_desu_da():
units = units_of("美味しかったです")
assert surfaces(units) == ["美味しかった", "です"]
assert units[0]["lemma"] == "美味しい"
assert tappable(units) == [True, False]
units = units_of("学生です")
assert surfaces(units) == ["学生", "です"]
assert tappable(units) == [True, False]
units = units_of("きれいじゃなかった")
assert surfaces(units) == ["きれい", "じゃ", "なかった"]
assert tappable(units) == [True, False, True]
def test_a2_nai_attaches_after_aux():
units = units_of("食べたくなかった")
assert surfaces(units) == ["食べたくなかった"]
assert units[0]["lemma"] == "食べる"
assert units[0]["reading"] == "たべたくなかった"
# The 形容詞 tail: 美味しく is 形容詞,一般 / 連用形-一般, so ない attaches under A2.
units = units_of("美味しくない")
assert surfaces(units) == ["美味しくない"]
assert units[0]["lemma"] == "美味しい"
assert units[0]["reading"] == "おいしくない"
assert units[0]["tappable"] is True
def test_a2_nai_after_particle_is_a_word():
units = units_of("人気のない通り")
assert surfaces(units) == ["人気", "の", "ない", "通り"]
assert tappable(units) == [True, False, True, True]
assert units[2]["lemma"] == "ない"
assert units[2]["level_key"] == "無い"
assert units[2]["pos"] == "形容詞"
def test_a0_breaker_units_accept_nothing():
# じゃ (lemma だ) is a breaker: なかっ may not join it although じゃ is 助動詞; た then
# joins なかっ via A1 with a 形容詞 tail.
units = units_of("きれいじゃなかった")
assert surfaces(units) == ["きれい", "じゃ", "なかった"]
assert units[1]["lemma"] == "だ"
assert units[2]["lemma"] == "ない"
assert units[2]["morpheme_count"] == 2
# よう (形状詞,助動詞語幹) is a breaker: に (助動詞, lemma だ) is its own unit after it.
units = units_of("話せるようになりたい")
assert surfaces(units) == ["話せる", "よう", "に", "なりたい"]
# の (助詞) is a breaker: ない is its own unit after it.
units = units_of("人気のない通り")
assert surfaces(units)[1:3] == ["の", "ない"]
def test_a3_te_de():
units = units_of("食べて")
assert surfaces(units) == ["食べて"]
assert units[0]["lemma"] == "食べる"
units = units_of("読んで")
assert surfaces(units) == ["読んで"]
assert units[0]["lemma"] == "読む"
assert units[0]["reading"] == "よんで"
# ば is 助詞,接続助詞 but not て/で: it starts its own non-tappable unit; A0 keeps なら
# off it, then ない (助動詞) joins なら via A1.
units = units_of("行かなければならない")
assert surfaces(units) == ["行かなけれ", "ば", "ならない"]
assert tappable(units) == [True, False, True]
assert units[0]["lemma"] == "行く"
assert units[2]["lemma"] == "なる"
def test_a4_te_auxiliary_verb():
units = units_of("読んでください")
assert surfaces(units) == ["読んでください"]
assert units[0]["lemma"] == "読む"
assert units[0]["reading"] == "よんでください"
units = units_of("日本語を勉強しています")
assert surfaces(units) == ["日本語", "を", "勉強しています"]
last = units[-1]
assert last["lemma"] == "勉強する"
assert last["level_key"] == "勉強"
assert last["pos"] == "名詞"
assert last["tappable"] is True
assert last["reading"] == "べんきょうしています"
assert last["morpheme_count"] == 5
def test_a5_suru_noun():
units = units_of("勉強します")
assert surfaces(units) == ["勉強します"]
assert units[0]["lemma"] == "勉強する"
assert units[0]["level_key"] == "勉強"
units = units_of("勉強")
assert surfaces(units) == ["勉強"]
assert units[0]["lemma"] == "勉強"
def test_a6_suffix():
units = units_of("田中さんは東京に")
assert surfaces(units) == ["田中さん", "は", "東京", "に"]
assert units[0]["is_name"] is True
assert units[0]["lemma"] == "田中"
assert units[0]["reading"] == "たなかさん"
assert units[0]["tappable"] is True
assert units[2]["is_name"] is True
assert units[2]["lemma"] == "東京"
assert [u["is_name"] for u in units] == [True, False, True, False]
units = units_of("三人で")
assert surfaces(units) == ["三人", "で"]
assert units[0]["tappable"] is True
assert units[0]["lemma"] == "三"
assert units[0]["reading"] == "さんにん"
# Sudachi's normalized_form for the OOV numeral 三 is the digit "3"; recorded, not judged.
assert units[0]["level_key"] == "3"
assert units[1]["tappable"] is False
units = units_of("一人で")
assert surfaces(units) == ["一人", "で"]
assert units[0]["morpheme_count"] == 1
def test_a7_prefix():
units = units_of("お願いします")
assert surfaces(units) == ["お願い", "します"]
assert tappable(units) == [True, True]
assert units[0]["lemma"] == "願う"
assert units[0]["reading"] == "おねがい"
assert units[0]["pos"] == "動詞"
assert units[0]["morpheme_count"] == 2
assert units[1]["lemma"] == "する"
def test_non_tappable_units():
units = units_of("話せるようになりたい")
by_surface = {u["surface"]: u for u in units}
assert by_surface["よう"]["tappable"] is False
assert by_surface["に"]["tappable"] is False
assert by_surface["話せる"]["tappable"] is True
assert by_surface["話せる"]["lemma"] == "話せる"
assert by_surface["話せる"]["level_key"] == "話す"
assert by_surface["なりたい"]["lemma"] == "なる"
assert by_surface["なりたい"]["tappable"] is True
for mark in ("。", "、"):
units = units_of(mark)
assert len(units) == 1
assert units[0]["tappable"] is False
assert units[0]["pos"] == "補助記号"
units = units_of("こんにちは")
assert len(units) == 1
assert units[0]["tappable"] is True
assert units[0]["lemma"] == "こんにちは"
assert units[0]["level_key"] == "今日は"
assert units[0]["pos"] == "感動詞"
assert "助詞" not in TAPPABLE_POS and "助動詞" not in TAPPABLE_POS
assert {"名詞", "動詞", "形容詞", "形状詞"} <= TAPPABLE_POS
@pytest.mark.parametrize("text", SENTENCES)
def test_offsets_tile_the_text(text):
units = units_of(text)
assert units, text
assert "".join(u["surface"] for u in units) == text
assert units[0]["start"] == 0
for prev, cur in zip(units, units[1:], strict=False):
assert cur["start"] == prev["end"]
for u in units:
assert u["end"] == u["start"] + len(u["surface"])
assert u["morpheme_count"] >= 1
assert text[u["start"] : u["end"]] == u["surface"]
assert units[-1]["end"] == len(text)
def test_build_units_is_pure_over_dicts():
ms = morphemes("食べました")
before = [dict(m) for m in ms]
units = build_units(ms)
assert ms == before # input not mutated
assert build_units(ms) == units # deterministic
assert build_units([]) == []
assert all(type(u) is dict for u in units)