File size: 6,060 Bytes
e39dc2e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
"""JLPT level derivation on both axes: word level by JMdict-id join, kanji level by literal.

D-09: word level is a join on JMdict ids against data/jlpt/n*.csv, never a string match.
D-10: a word on no list is "N1+" - never silently guessed. D-12: proper nouns are "name".
D-13: the kanji axis (data/jlpt/kanji_levels.json) is separate from the word axis, with null
for a kanji the 2,211-entry list does not carry (above every level for furigana gating).

Counts are the ones plan 02-01 measured on the pinned tag (02-01-SUMMARY.md): 7,748 unique ids
(research said 7,747) and 447 ids on more than one LEVEL (research's 505 counted rows).
"""

from __future__ import annotations

import csv
import io
from collections import defaultdict
from pathlib import Path

from japanese_avatar.nlp import jmdict, levels
from japanese_avatar.nlp.levels import (
    LEVEL_RANK,
    derive_level,
    kanji_levels,
    kanji_levels_for,
    vocab_levels,
)

REPO_ROOT = Path(__file__).resolve().parent.parent
JLPT = REPO_ROOT / "data" / "jlpt"

EXPECTED_UNIQUE_IDS = 7748
EXPECTED_MULTI_LEVEL_IDS = 447
EXPECTED_KANJI = 2211

IRU_ORU = 1577980
KYOU = 1579110
HANASERU = 1562360


def _levels_by_id() -> dict[str, set[int]]:
    """Independent re-parse of the five CSVs: id -> every level number it is listed at."""
    found: dict[str, set[int]] = defaultdict(set)
    for n in (5, 4, 3, 2, 1):
        text = (JLPT / f"n{n}.csv").read_bytes().decode("utf-8-sig")
        rows = list(csv.reader(io.StringIO(text, newline="")))
        assert rows[0][0] == "jmdict_seq"
        for row in rows[1:]:
            if row and row[0]:
                found[row[0]].add(n)
    return found


def test_vocab_levels_is_a_singleton():
    assert vocab_levels() is vocab_levels()
    assert kanji_levels() is kanji_levels()


def test_vocab_levels_count():
    table = vocab_levels()
    assert len(table) == EXPECTED_UNIQUE_IDS
    assert set(table.values()) <= {1, 2, 3, 4, 5}
    assert all(isinstance(k, int) for k in table)


def test_multi_level_ids_take_easiest():
    by_id = _levels_by_id()
    assert len(by_id) == EXPECTED_UNIQUE_IDS
    multi = {i: ls for i, ls in by_id.items() if len(ls) > 1}
    assert len(multi) == EXPECTED_MULTI_LEVEL_IDS, "447 ids sit on more than one level"
    table = vocab_levels()
    wrong = {i: (table[int(i)], ls) for i, ls in multi.items() if table[int(i)] != max(ls)}
    assert wrong == {}, f"{len(wrong)} multi-level ids did not take the easiest level: {wrong}"
    # And single-level ids carry exactly their one level.
    for i, ls in by_id.items():
        if len(ls) == 1:
            assert table[int(i)] == next(iter(ls))


def test_known_levels():
    table = vocab_levels()

    def level(lemma: str, level_key: str, reading: str) -> int | None:
        entry = jmdict.lookup(lemma, level_key, reading, table)
        assert entry is not None, (lemma, reading)
        return table.get(entry.id)

    assert jmdict.lookup("いる", "ε±…γ‚‹", "いる", table).id == IRU_ORU
    assert table[IRU_ORU] == 5
    assert jmdict.lookup("今ζ—₯", "今ζ—₯", "きょう", table).id == KYOU
    assert table[KYOU] == 5
    assert level("δΊΊζ°—", "δΊΊζ°—", "にんき") == 3
    assert level("会議", "会議", "γ‹γ„γŽ") == 4
    assert level("θ‘Œγ†", "θ‘Œγ†", "γŠγ“γͺう") == 4
    assert level("ε½Ό", "ε½Ό", "γ‹γ‚Œ") == 4
    assert level("ι€šγ‚Š", "ι€šγ‚Š", "γ¨γŠγ‚Š") == 3


def test_derive_level_rules():
    table = vocab_levels()
    assert derive_level(True, None, None) == "name"
    assert derive_level(False, None, None) == "N1+"

    hanasu = jmdict.lookup("話す", "話す", "はγͺす", table)
    hanaseru = jmdict.lookup("話せる", "話せる", "はγͺせる", table)
    assert hanasu is not None and hanaseru is not None and hanaseru.id == HANASERU
    assert HANASERU not in table, "話せる is on no list; only 話す (its level_key) is"
    # Level comes from the level_key entry (話す, N5) even though the gloss entry is 話せる.
    assert derive_level(False, hanasu, hanaseru) == "N5"
    # Only the gloss entry known, and it is unlisted -> honest N1+.
    assert derive_level(False, None, hanaseru) == "N1+"
    assert derive_level(False, hanaseru, None) == "N1+"
    # The gloss entry is consulted when the level entry misses the lists.
    assert derive_level(False, hanaseru, hanasu) == "N5"
    # A name is a name even when the entry is on a list.
    assert derive_level(True, hanasu, hanasu) == "name"


def test_kanji_axis():
    table = kanji_levels()
    assert len(table) == EXPECTED_KANJI
    assert set(table.values()) == {"N5", "N4", "N3", "N2", "N1"}
    assert kanji_levels_for("ζ—₯本θͺž") == {"ζ—₯": "N5", "本": "N5", "θͺž": "N5"}
    utsu = kanji_levels_for("鬱院しい")
    assert utsu["鬱"] is None, "鬱 is not on the 2,211 list: unlisted, above every level"
    assert "し" not in utsu and "い" not in utsu
    assert kanji_levels_for("こんにけは") == {}
    assert kanji_levels_for("") == {}
    # γ€… (U+3005) is a kanji-run character with no level of its own -> None. Known edge: the
    # run gate treats null as above every level, honest for a repeat whose base may be listed.
    hitobito = kanji_levels_for("δΊΊγ€…")
    assert hitobito["δΊΊ"] == "N5" and hitobito["γ€…"] is None
    assert kanji_levels_for("ο½˜ο½™ο½š abc 123 、。") == {}


def test_is_kanji_ranges_local_copy():
    """levels.py carries its own _is_kanji (parallel wave with 02-02; 02-05 reconciles)."""
    assert levels._is_kanji("ζΌ’") and levels._is_kanji("γ€…") and levels._is_kanji("㐀")
    assert levels._is_kanji("ιΏΏ") and levels._is_kanji("ο€€") and levels._is_kanji("ο«Ώ")
    assert not levels._is_kanji("あ") and not levels._is_kanji("γ‚’") and not levels._is_kanji("a")
    assert not levels._is_kanji("γƒΌ") and not levels._is_kanji("。")


def test_level_rank_order():
    assert LEVEL_RANK["N5"] < LEVEL_RANK["N4"] < LEVEL_RANK["N3"] < LEVEL_RANK["N2"]
    assert LEVEL_RANK["N2"] < LEVEL_RANK["N1"]
    assert set(LEVEL_RANK) == {"N5", "N4", "N3", "N2", "N1"}