Spaces:
Running on Zero
Running on Zero
File size: 6,060 Bytes
e39dc2e | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 | """JLPT level derivation on both axes: word level by JMdict-id join, kanji level by literal.
D-09: word level is a join on JMdict ids against data/jlpt/n*.csv, never a string match.
D-10: a word on no list is "N1+" - never silently guessed. D-12: proper nouns are "name".
D-13: the kanji axis (data/jlpt/kanji_levels.json) is separate from the word axis, with null
for a kanji the 2,211-entry list does not carry (above every level for furigana gating).
Counts are the ones plan 02-01 measured on the pinned tag (02-01-SUMMARY.md): 7,748 unique ids
(research said 7,747) and 447 ids on more than one LEVEL (research's 505 counted rows).
"""
from __future__ import annotations
import csv
import io
from collections import defaultdict
from pathlib import Path
from japanese_avatar.nlp import jmdict, levels
from japanese_avatar.nlp.levels import (
LEVEL_RANK,
derive_level,
kanji_levels,
kanji_levels_for,
vocab_levels,
)
REPO_ROOT = Path(__file__).resolve().parent.parent
JLPT = REPO_ROOT / "data" / "jlpt"
EXPECTED_UNIQUE_IDS = 7748
EXPECTED_MULTI_LEVEL_IDS = 447
EXPECTED_KANJI = 2211
IRU_ORU = 1577980
KYOU = 1579110
HANASERU = 1562360
def _levels_by_id() -> dict[str, set[int]]:
"""Independent re-parse of the five CSVs: id -> every level number it is listed at."""
found: dict[str, set[int]] = defaultdict(set)
for n in (5, 4, 3, 2, 1):
text = (JLPT / f"n{n}.csv").read_bytes().decode("utf-8-sig")
rows = list(csv.reader(io.StringIO(text, newline="")))
assert rows[0][0] == "jmdict_seq"
for row in rows[1:]:
if row and row[0]:
found[row[0]].add(n)
return found
def test_vocab_levels_is_a_singleton():
assert vocab_levels() is vocab_levels()
assert kanji_levels() is kanji_levels()
def test_vocab_levels_count():
table = vocab_levels()
assert len(table) == EXPECTED_UNIQUE_IDS
assert set(table.values()) <= {1, 2, 3, 4, 5}
assert all(isinstance(k, int) for k in table)
def test_multi_level_ids_take_easiest():
by_id = _levels_by_id()
assert len(by_id) == EXPECTED_UNIQUE_IDS
multi = {i: ls for i, ls in by_id.items() if len(ls) > 1}
assert len(multi) == EXPECTED_MULTI_LEVEL_IDS, "447 ids sit on more than one level"
table = vocab_levels()
wrong = {i: (table[int(i)], ls) for i, ls in multi.items() if table[int(i)] != max(ls)}
assert wrong == {}, f"{len(wrong)} multi-level ids did not take the easiest level: {wrong}"
# And single-level ids carry exactly their one level.
for i, ls in by_id.items():
if len(ls) == 1:
assert table[int(i)] == next(iter(ls))
def test_known_levels():
table = vocab_levels()
def level(lemma: str, level_key: str, reading: str) -> int | None:
entry = jmdict.lookup(lemma, level_key, reading, table)
assert entry is not None, (lemma, reading)
return table.get(entry.id)
assert jmdict.lookup("γγ", "ε±
γ", "γγ", table).id == IRU_ORU
assert table[IRU_ORU] == 5
assert jmdict.lookup("δ»ζ₯", "δ»ζ₯", "γγγ", table).id == KYOU
assert table[KYOU] == 5
assert level("δΊΊζ°", "δΊΊζ°", "γ«γγ") == 3
assert level("δΌθ°", "δΌθ°", "γγγ") == 4
assert level("θ‘γ", "θ‘γ", "γγγͺγ") == 4
assert level("ε½Ό", "ε½Ό", "γγ") == 4
assert level("ιγ", "ιγ", "γ¨γγ") == 3
def test_derive_level_rules():
table = vocab_levels()
assert derive_level(True, None, None) == "name"
assert derive_level(False, None, None) == "N1+"
hanasu = jmdict.lookup("θ©±γ", "θ©±γ", "γ―γͺγ", table)
hanaseru = jmdict.lookup("θ©±γγ", "θ©±γγ", "γ―γͺγγ", table)
assert hanasu is not None and hanaseru is not None and hanaseru.id == HANASERU
assert HANASERU not in table, "θ©±γγ is on no list; only θ©±γ (its level_key) is"
# Level comes from the level_key entry (θ©±γ, N5) even though the gloss entry is θ©±γγ.
assert derive_level(False, hanasu, hanaseru) == "N5"
# Only the gloss entry known, and it is unlisted -> honest N1+.
assert derive_level(False, None, hanaseru) == "N1+"
assert derive_level(False, hanaseru, None) == "N1+"
# The gloss entry is consulted when the level entry misses the lists.
assert derive_level(False, hanaseru, hanasu) == "N5"
# A name is a name even when the entry is on a list.
assert derive_level(True, hanasu, hanasu) == "name"
def test_kanji_axis():
table = kanji_levels()
assert len(table) == EXPECTED_KANJI
assert set(table.values()) == {"N5", "N4", "N3", "N2", "N1"}
assert kanji_levels_for("ζ₯ζ¬θͺ") == {"ζ₯": "N5", "ζ¬": "N5", "θͺ": "N5"}
utsu = kanji_levels_for("鬱ιΆγγ")
assert utsu["鬱"] is None, "鬱 is not on the 2,211 list: unlisted, above every level"
assert "γ" not in utsu and "γ" not in utsu
assert kanji_levels_for("γγγ«γ‘γ―") == {}
assert kanji_levels_for("") == {}
# γ
(U+3005) is a kanji-run character with no level of its own -> None. Known edge: the
# run gate treats null as above every level, honest for a repeat whose base may be listed.
hitobito = kanji_levels_for("δΊΊγ
")
assert hitobito["δΊΊ"] == "N5" and hitobito["γ
"] is None
assert kanji_levels_for("ο½ο½ο½ abc 123 γγ") == {}
def test_is_kanji_ranges_local_copy():
"""levels.py carries its own _is_kanji (parallel wave with 02-02; 02-05 reconciles)."""
assert levels._is_kanji("ζΌ’") and levels._is_kanji("γ
") and levels._is_kanji("γ")
assert levels._is_kanji("ιΏΏ") and levels._is_kanji("ο€") and levels._is_kanji("ο«Ώ")
assert not levels._is_kanji("γ") and not levels._is_kanji("γ’") and not levels._is_kanji("a")
assert not levels._is_kanji("γΌ") and not levels._is_kanji("γ")
def test_level_rank_order():
assert LEVEL_RANK["N5"] < LEVEL_RANK["N4"] < LEVEL_RANK["N3"] < LEVEL_RANK["N2"]
assert LEVEL_RANK["N2"] < LEVEL_RANK["N1"]
assert set(LEVEL_RANK) == {"N5", "N4", "N3", "N2", "N1"}
|