File size: 5,057 Bytes
adfe806
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
"""Okurigana alignment: surface + hiragana reading -> per-kanji-run ruby spans.

The 23-pair table is 02-RESEARCH.md § Q1 "Readings -> furigana" (verified there on Sudachi's
readings); readings are given as Sudachi's katakana and converted with ``jaconv.kata2hira`` the
way the analyzer does. ``align_ruby`` only ALIGNS a reading it is given - it never guesses one.
"""

from __future__ import annotations

import re

import jaconv
import pytest

from japanese_avatar.nlp.ruby import KANA, align_ruby, is_kanji, kanji_runs

# (surface, Sudachi katakana reading, expected spans [text, rt]) - rt None means bare text.
ALIGNMENT_TABLE = [
    ("食べ", "タベ", [["食", "た"], ["べ", None]]),
    ("美味しかっ", "オイシカッ", [["美味", "おい"], ["しかっ", None]]),
    ("待ち合わせ", "マチアワセ", [["待", "ま"], ["ち", None], ["合", "あ"], ["わせ", None]]),
    ("買い物", "カイモノ", [["買", "か"], ["い", None], ["物", "もの"]]),
    ("取り扱い", "トリアツカイ", [["取", "と"], ["り", None], ["扱", "あつか"], ["い", None]]),
    ("お願い", "オネガイ", [["お", None], ["願", "ねが"], ["い", None]]),
    ("大阪駅", "オオサカエキ", [["大阪駅", "おおさかえき"]]),
    ("コーヒー", "コーヒー", [["コーヒー", None]]),
    ("こんにちは", "コンニチハ", [["こんにちは", None]]),
    ("人々", "ヒトビト", [["人々", "ひとびと"]]),
    ("今日", "キョウ", [["今日", "きょう"]]),
    ("行っ", "イッ", [["行", "い"], ["っ", None]]),
    ("勉強", "ベンキョウ", [["勉強", "べんきょう"]]),
    ("田中", "タナカ", [["田中", "たなか"]]),
    ("東京", "トウキョウ", [["東京", "とうきょう"]]),
    ("三人", "サンニン", [["三人", "さんにん"]]),
    ("話せる", "ハナセル", [["話", "はな"], ["せる", None]]),
    ("読ん", "ヨン", [["読", "よ"], ["ん", None]]),
    ("学生", "ガクセイ", [["学生", "がくせい"]]),
    ("天気", "テンキ", [["天気", "てんき"]]),
    ("日本語", "ニホンゴ", [["日本語", "にほんご"]]),
    ("飲み", "ノミ", [["飲", "の"], ["み", None]]),
    ("公園", "コウエン", [["公園", "こうえん"]]),
]

assert len(ALIGNMENT_TABLE) == 23

HIRAGANA_ONLY = re.compile(r"^[ぁ-ゖー]+$")


@pytest.mark.parametrize(("surface", "reading", "spans"), ALIGNMENT_TABLE, ids=lambda v: str(v))
def test_alignment_table(surface, reading, spans):
    assert align_ruby(surface, jaconv.kata2hira(reading)) == spans


@pytest.mark.parametrize("surface", ["コーヒー", "こんにちは", "ください"])
def test_all_kana_is_one_bare_span(surface):
    assert align_ruby(surface, jaconv.kata2hira(surface)) == [[surface, None]]


def test_fallback_is_whole_unit_ruby():
    # The kana run べ must match verbatim; たべる has a trailing る that nothing absorbs.
    assert align_ruby("食べ", "たべる") == [["食べ", "たべる"]]
    # An empty reading on a kanji surface: whole-unit ruby with the (empty) reading, no raise.
    assert align_ruby("生", "") == [["生", ""]]
    # An empty reading on an all-kana surface is simply bare text.
    assert align_ruby("ください", "") == [["ください", None]]
    # A reading shorter than the kana runs demand cannot align either.
    assert align_ruby("待ち合わせ", "まち") == [["待ち合わせ", "まち"]]


def test_iteration_mark_stays_in_kanji_run():
    assert align_ruby("人々", "ひとびと") == [["人々", "ひとびと"]]
    assert is_kanji("々") is True
    assert kanji_runs("人々") == ["人々"]


@pytest.mark.parametrize("ch", ["漢", "々", "龍", "食", "𠮷"])
def test_is_kanji_true(ch):
    assert is_kanji(ch) is True


@pytest.mark.parametrize("ch", ["あ", "ア", "ー", "A", "。", "1", " ", "っ"])
def test_is_kanji_false(ch):
    assert is_kanji(ch) is False


def test_kanji_runs():
    assert kanji_runs("待ち合わせ") == ["待", "合"]
    assert kanji_runs("お願い") == ["願"]
    assert kanji_runs("大阪駅") == ["大阪駅"]
    assert kanji_runs("こんにちは") == []
    assert kanji_runs("コーヒー") == []
    assert kanji_runs("") == []


@pytest.mark.parametrize(("surface", "reading", "spans"), ALIGNMENT_TABLE, ids=lambda v: str(v))
def test_spans_tile_surface(surface, reading, spans):
    out = align_ruby(surface, jaconv.kata2hira(reading))
    assert "".join(t for t, _ in out) == surface


@pytest.mark.parametrize(("surface", "reading", "spans"), ALIGNMENT_TABLE, ids=lambda v: str(v))
def test_rt_is_hiragana(surface, reading, spans):
    for _text, rt in align_ruby(surface, jaconv.kata2hira(reading)):
        if rt is not None:
            assert HIRAGANA_ONLY.match(rt), rt


def test_kana_class_covers_both_scripts_and_marks():
    kana = re.compile(f"^[{KANA}]+$")
    assert kana.match("こんにちは")
    assert kana.match("コーヒー")
    assert kana.match("ゝゞヽヾ")
    assert not kana.match("漢")
    assert not kana.match("々")