Spaces:
Running on Zero
Running on Zero
Download tests/test_units.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 10.4 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f108e4c8cd374e7e9f66d39cdea425d0f7e2ff6e/tests/test_units.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@f108e4c8cd374e7e9f66d39cdea425d0f7e2ff6e/tests/test_units.py
-
curl -L -o test_units.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f108e4c8cd374e7e9f66d39cdea425d0f7e2ff6e/tests/test_units.py
10.4 kB
| """Mode-C morphemes -> whole-word units (D-05 joining rule A0-A7), one named test per rule. | |
| Every test tokenises with the real Sudachi core dictionary (loaded once per module) and asserts | |
| on ``build_units``. The unit SPLITS asserted here are the D-05 contract from 02-02-PLAN.md; only | |
| Sudachi's morpheme output (surfaces, readings, POS tuples) is "observed". If a split disagrees | |
| with a test, the rule table or its implementation is wrong - never the expectation. | |
| """ | |
| from __future__ import annotations | |
| import time | |
| import pytest | |
| from japanese_avatar.nlp.tokenizer import morphemes, warmup | |
| from japanese_avatar.nlp.units import TAPPABLE_POS, build_units | |
| SENTENCES = [ | |
| "食べました", | |
| "食べさせられた", | |
| "美味しかったです", | |
| "学生です", | |
| "きれいじゃなかった", | |
| "食べたくなかった", | |
| "美味しくない", | |
| "人気のない通り", | |
| "話せるようになりたい", | |
| "行かなければならない", | |
| "食べて", | |
| "読んで", | |
| "読んでください", | |
| "日本語を勉強しています", | |
| "勉強します", | |
| "田中さんは東京に", | |
| "三人で", | |
| "一人で", | |
| "お願いします", | |
| "こんにちは", | |
| "今日は天気がいいですね。", | |
| "はい、そうです。", | |
| ] | |
| def tokenizer_loaded(): | |
| t0 = time.perf_counter() | |
| seconds = warmup() | |
| print(f"\nsudachi core load: {seconds:.3f} s (warmup call {time.perf_counter() - t0:.3f} s)") | |
| def units_of(text: str) -> list[dict]: | |
| return build_units(morphemes(text)) | |
| def surfaces(units: list[dict]) -> list[str]: | |
| return [u["surface"] for u in units] | |
| def tappable(units: list[dict]) -> list[bool]: | |
| return [u["tappable"] for u in units] | |
| def test_morphemes_are_plain_dicts(): | |
| ms = morphemes("食べました") | |
| assert len(ms) == 3 | |
| for m in ms: | |
| assert set(m) == {"surface", "reading", "lemma", "norm", "pos", "oov"} | |
| assert isinstance(m["pos"], list) and len(m["pos"]) == 6 | |
| assert all(isinstance(p, str) for p in m["pos"]) | |
| assert isinstance(m["oov"], bool) | |
| assert type(m) is dict | |
| assert ms[0]["surface"] == "食べ" | |
| assert ms[0]["reading"] == "タベ" # katakana, as Sudachi returns it | |
| assert ms[0]["lemma"] == "食べる" | |
| assert ms[0]["pos"][0] == "動詞" | |
| assert morphemes("") == [] | |
| def test_a1_auxiliary_chain(): | |
| units = units_of("食べました") | |
| assert len(units) == 1 | |
| u = units[0] | |
| assert u == { | |
| "surface": "食べました", | |
| "reading": "たべました", | |
| "lemma": "食べる", | |
| "level_key": "食べる", | |
| "pos": "動詞", | |
| "tappable": True, | |
| "is_name": False, | |
| "start": 0, | |
| "end": 5, | |
| "morpheme_count": 3, | |
| } | |
| units = units_of("食べさせられた") | |
| assert surfaces(units) == ["食べさせられた"] | |
| assert units[0]["lemma"] == "食べる" | |
| assert units[0]["morpheme_count"] == 4 | |
| def test_a1_stops_at_desu_da(): | |
| units = units_of("美味しかったです") | |
| assert surfaces(units) == ["美味しかった", "です"] | |
| assert units[0]["lemma"] == "美味しい" | |
| assert tappable(units) == [True, False] | |
| units = units_of("学生です") | |
| assert surfaces(units) == ["学生", "です"] | |
| assert tappable(units) == [True, False] | |
| units = units_of("きれいじゃなかった") | |
| assert surfaces(units) == ["きれい", "じゃ", "なかった"] | |
| assert tappable(units) == [True, False, True] | |
| def test_a2_nai_attaches_after_aux(): | |
| units = units_of("食べたくなかった") | |
| assert surfaces(units) == ["食べたくなかった"] | |
| assert units[0]["lemma"] == "食べる" | |
| assert units[0]["reading"] == "たべたくなかった" | |
| # The 形容詞 tail: 美味しく is 形容詞,一般 / 連用形-一般, so ない attaches under A2. | |
| units = units_of("美味しくない") | |
| assert surfaces(units) == ["美味しくない"] | |
| assert units[0]["lemma"] == "美味しい" | |
| assert units[0]["reading"] == "おいしくない" | |
| assert units[0]["tappable"] is True | |
| def test_a2_nai_after_particle_is_a_word(): | |
| units = units_of("人気のない通り") | |
| assert surfaces(units) == ["人気", "の", "ない", "通り"] | |
| assert tappable(units) == [True, False, True, True] | |
| assert units[2]["lemma"] == "ない" | |
| assert units[2]["level_key"] == "無い" | |
| assert units[2]["pos"] == "形容詞" | |
| def test_a0_breaker_units_accept_nothing(): | |
| # じゃ (lemma だ) is a breaker: なかっ may not join it although じゃ is 助動詞; た then | |
| # joins なかっ via A1 with a 形容詞 tail. | |
| units = units_of("きれいじゃなかった") | |
| assert surfaces(units) == ["きれい", "じゃ", "なかった"] | |
| assert units[1]["lemma"] == "だ" | |
| assert units[2]["lemma"] == "ない" | |
| assert units[2]["morpheme_count"] == 2 | |
| # よう (形状詞,助動詞語幹) is a breaker: に (助動詞, lemma だ) is its own unit after it. | |
| units = units_of("話せるようになりたい") | |
| assert surfaces(units) == ["話せる", "よう", "に", "なりたい"] | |
| # の (助詞) is a breaker: ない is its own unit after it. | |
| units = units_of("人気のない通り") | |
| assert surfaces(units)[1:3] == ["の", "ない"] | |
| def test_a3_te_de(): | |
| units = units_of("食べて") | |
| assert surfaces(units) == ["食べて"] | |
| assert units[0]["lemma"] == "食べる" | |
| units = units_of("読んで") | |
| assert surfaces(units) == ["読んで"] | |
| assert units[0]["lemma"] == "読む" | |
| assert units[0]["reading"] == "よんで" | |
| # ば is 助詞,接続助詞 but not て/で: it starts its own non-tappable unit; A0 keeps なら | |
| # off it, then ない (助動詞) joins なら via A1. | |
| units = units_of("行かなければならない") | |
| assert surfaces(units) == ["行かなけれ", "ば", "ならない"] | |
| assert tappable(units) == [True, False, True] | |
| assert units[0]["lemma"] == "行く" | |
| assert units[2]["lemma"] == "なる" | |
| def test_a4_te_auxiliary_verb(): | |
| units = units_of("読んでください") | |
| assert surfaces(units) == ["読んでください"] | |
| assert units[0]["lemma"] == "読む" | |
| assert units[0]["reading"] == "よんでください" | |
| units = units_of("日本語を勉強しています") | |
| assert surfaces(units) == ["日本語", "を", "勉強しています"] | |
| last = units[-1] | |
| assert last["lemma"] == "勉強する" | |
| assert last["level_key"] == "勉強" | |
| assert last["pos"] == "名詞" | |
| assert last["tappable"] is True | |
| assert last["reading"] == "べんきょうしています" | |
| assert last["morpheme_count"] == 5 | |
| def test_a5_suru_noun(): | |
| units = units_of("勉強します") | |
| assert surfaces(units) == ["勉強します"] | |
| assert units[0]["lemma"] == "勉強する" | |
| assert units[0]["level_key"] == "勉強" | |
| units = units_of("勉強") | |
| assert surfaces(units) == ["勉強"] | |
| assert units[0]["lemma"] == "勉強" | |
| def test_a6_suffix(): | |
| units = units_of("田中さんは東京に") | |
| assert surfaces(units) == ["田中さん", "は", "東京", "に"] | |
| assert units[0]["is_name"] is True | |
| assert units[0]["lemma"] == "田中" | |
| assert units[0]["reading"] == "たなかさん" | |
| assert units[0]["tappable"] is True | |
| assert units[2]["is_name"] is True | |
| assert units[2]["lemma"] == "東京" | |
| assert [u["is_name"] for u in units] == [True, False, True, False] | |
| units = units_of("三人で") | |
| assert surfaces(units) == ["三人", "で"] | |
| assert units[0]["tappable"] is True | |
| assert units[0]["lemma"] == "三" | |
| assert units[0]["reading"] == "さんにん" | |
| # Sudachi's normalized_form for the OOV numeral 三 is the digit "3"; recorded, not judged. | |
| assert units[0]["level_key"] == "3" | |
| assert units[1]["tappable"] is False | |
| units = units_of("一人で") | |
| assert surfaces(units) == ["一人", "で"] | |
| assert units[0]["morpheme_count"] == 1 | |
| def test_a7_prefix(): | |
| units = units_of("お願いします") | |
| assert surfaces(units) == ["お願い", "します"] | |
| assert tappable(units) == [True, True] | |
| assert units[0]["lemma"] == "願う" | |
| assert units[0]["reading"] == "おねがい" | |
| assert units[0]["pos"] == "動詞" | |
| assert units[0]["morpheme_count"] == 2 | |
| assert units[1]["lemma"] == "する" | |
| def test_non_tappable_units(): | |
| units = units_of("話せるようになりたい") | |
| by_surface = {u["surface"]: u for u in units} | |
| assert by_surface["よう"]["tappable"] is False | |
| assert by_surface["に"]["tappable"] is False | |
| assert by_surface["話せる"]["tappable"] is True | |
| assert by_surface["話せる"]["lemma"] == "話せる" | |
| assert by_surface["話せる"]["level_key"] == "話す" | |
| assert by_surface["なりたい"]["lemma"] == "なる" | |
| assert by_surface["なりたい"]["tappable"] is True | |
| for mark in ("。", "、"): | |
| units = units_of(mark) | |
| assert len(units) == 1 | |
| assert units[0]["tappable"] is False | |
| assert units[0]["pos"] == "補助記号" | |
| units = units_of("こんにちは") | |
| assert len(units) == 1 | |
| assert units[0]["tappable"] is True | |
| assert units[0]["lemma"] == "こんにちは" | |
| assert units[0]["level_key"] == "今日は" | |
| assert units[0]["pos"] == "感動詞" | |
| assert "助詞" not in TAPPABLE_POS and "助動詞" not in TAPPABLE_POS | |
| assert {"名詞", "動詞", "形容詞", "形状詞"} <= TAPPABLE_POS | |
| def test_offsets_tile_the_text(text): | |
| units = units_of(text) | |
| assert units, text | |
| assert "".join(u["surface"] for u in units) == text | |
| assert units[0]["start"] == 0 | |
| for prev, cur in zip(units, units[1:], strict=False): | |
| assert cur["start"] == prev["end"] | |
| for u in units: | |
| assert u["end"] == u["start"] + len(u["surface"]) | |
| assert u["morpheme_count"] >= 1 | |
| assert text[u["start"] : u["end"]] == u["surface"] | |
| assert units[-1]["end"] == len(text) | |
| def test_build_units_is_pure_over_dicts(): | |
| ms = morphemes("食べました") | |
| before = [dict(m) for m in ms] | |
| units = build_units(ms) | |
| assert ms == before # input not mutated | |
| assert build_units(ms) == units # deterministic | |
| assert build_units([]) == [] | |
| assert all(type(u) is dict for u in units) | |