japanese-learning-avatar / tests /test_levels.py
WolfDavid's picture
test(02-03): add failing JLPT level tests
e39dc2e
Raw History Blame
6.06 kB
"""JLPT level derivation on both axes: word level by JMdict-id join, kanji level by literal.
D-09: word level is a join on JMdict ids against data/jlpt/n*.csv, never a string match.
D-10: a word on no list is "N1+" - never silently guessed. D-12: proper nouns are "name".
D-13: the kanji axis (data/jlpt/kanji_levels.json) is separate from the word axis, with null
for a kanji the 2,211-entry list does not carry (above every level for furigana gating).
Counts are the ones plan 02-01 measured on the pinned tag (02-01-SUMMARY.md): 7,748 unique ids
(research said 7,747) and 447 ids on more than one LEVEL (research's 505 counted rows).
"""
from __future__ import annotations
import csv
import io
from collections import defaultdict
from pathlib import Path
from japanese_avatar.nlp import jmdict, levels
from japanese_avatar.nlp.levels import (
LEVEL_RANK,
derive_level,
kanji_levels,
kanji_levels_for,
vocab_levels,
)
REPO_ROOT = Path(__file__).resolve().parent.parent
JLPT = REPO_ROOT / "data" / "jlpt"
EXPECTED_UNIQUE_IDS = 7748
EXPECTED_MULTI_LEVEL_IDS = 447
EXPECTED_KANJI = 2211
IRU_ORU = 1577980
KYOU = 1579110
HANASERU = 1562360
def _levels_by_id() -> dict[str, set[int]]:
"""Independent re-parse of the five CSVs: id -> every level number it is listed at."""
found: dict[str, set[int]] = defaultdict(set)
for n in (5, 4, 3, 2, 1):
text = (JLPT / f"n{n}.csv").read_bytes().decode("utf-8-sig")
rows = list(csv.reader(io.StringIO(text, newline="")))
assert rows[0][0] == "jmdict_seq"
for row in rows[1:]:
if row and row[0]:
found[row[0]].add(n)
return found
def test_vocab_levels_is_a_singleton():
assert vocab_levels() is vocab_levels()
assert kanji_levels() is kanji_levels()
def test_vocab_levels_count():
table = vocab_levels()
assert len(table) == EXPECTED_UNIQUE_IDS
assert set(table.values()) <= {1, 2, 3, 4, 5}
assert all(isinstance(k, int) for k in table)
def test_multi_level_ids_take_easiest():
by_id = _levels_by_id()
assert len(by_id) == EXPECTED_UNIQUE_IDS
multi = {i: ls for i, ls in by_id.items() if len(ls) > 1}
assert len(multi) == EXPECTED_MULTI_LEVEL_IDS, "447 ids sit on more than one level"
table = vocab_levels()
wrong = {i: (table[int(i)], ls) for i, ls in multi.items() if table[int(i)] != max(ls)}
assert wrong == {}, f"{len(wrong)} multi-level ids did not take the easiest level: {wrong}"
# And single-level ids carry exactly their one level.
for i, ls in by_id.items():
if len(ls) == 1:
assert table[int(i)] == next(iter(ls))
def test_known_levels():
table = vocab_levels()
def level(lemma: str, level_key: str, reading: str) -> int | None:
entry = jmdict.lookup(lemma, level_key, reading, table)
assert entry is not None, (lemma, reading)
return table.get(entry.id)
assert jmdict.lookup("いる", "ε±…γ‚‹", "いる", table).id == IRU_ORU
assert table[IRU_ORU] == 5
assert jmdict.lookup("今ζ—₯", "今ζ—₯", "きょう", table).id == KYOU
assert table[KYOU] == 5
assert level("δΊΊζ°—", "δΊΊζ°—", "にんき") == 3
assert level("会議", "会議", "γ‹γ„γŽ") == 4
assert level("θ‘Œγ†", "θ‘Œγ†", "γŠγ“γͺう") == 4
assert level("ε½Ό", "ε½Ό", "γ‹γ‚Œ") == 4
assert level("ι€šγ‚Š", "ι€šγ‚Š", "γ¨γŠγ‚Š") == 3
def test_derive_level_rules():
table = vocab_levels()
assert derive_level(True, None, None) == "name"
assert derive_level(False, None, None) == "N1+"
hanasu = jmdict.lookup("話す", "話す", "はγͺす", table)
hanaseru = jmdict.lookup("話せる", "話せる", "はγͺせる", table)
assert hanasu is not None and hanaseru is not None and hanaseru.id == HANASERU
assert HANASERU not in table, "話せる is on no list; only 話す (its level_key) is"
# Level comes from the level_key entry (話す, N5) even though the gloss entry is 話せる.
assert derive_level(False, hanasu, hanaseru) == "N5"
# Only the gloss entry known, and it is unlisted -> honest N1+.
assert derive_level(False, None, hanaseru) == "N1+"
assert derive_level(False, hanaseru, None) == "N1+"
# The gloss entry is consulted when the level entry misses the lists.
assert derive_level(False, hanaseru, hanasu) == "N5"
# A name is a name even when the entry is on a list.
assert derive_level(True, hanasu, hanasu) == "name"
def test_kanji_axis():
table = kanji_levels()
assert len(table) == EXPECTED_KANJI
assert set(table.values()) == {"N5", "N4", "N3", "N2", "N1"}
assert kanji_levels_for("ζ—₯本θͺž") == {"ζ—₯": "N5", "本": "N5", "θͺž": "N5"}
utsu = kanji_levels_for("鬱院しい")
assert utsu["鬱"] is None, "鬱 is not on the 2,211 list: unlisted, above every level"
assert "し" not in utsu and "い" not in utsu
assert kanji_levels_for("こんにけは") == {}
assert kanji_levels_for("") == {}
# γ€… (U+3005) is a kanji-run character with no level of its own -> None. Known edge: the
# run gate treats null as above every level, honest for a repeat whose base may be listed.
hitobito = kanji_levels_for("δΊΊγ€…")
assert hitobito["δΊΊ"] == "N5" and hitobito["γ€…"] is None
assert kanji_levels_for("ο½˜ο½™ο½š abc 123 、。") == {}
def test_is_kanji_ranges_local_copy():
"""levels.py carries its own _is_kanji (parallel wave with 02-02; 02-05 reconciles)."""
assert levels._is_kanji("ζΌ’") and levels._is_kanji("γ€…") and levels._is_kanji("㐀")
assert levels._is_kanji("ιΏΏ") and levels._is_kanji("ο€€") and levels._is_kanji("ο«Ώ")
assert not levels._is_kanji("あ") and not levels._is_kanji("γ‚’") and not levels._is_kanji("a")
assert not levels._is_kanji("γƒΌ") and not levels._is_kanji("。")
def test_level_rank_order():
assert LEVEL_RANK["N5"] < LEVEL_RANK["N4"] < LEVEL_RANK["N3"] < LEVEL_RANK["N2"]
assert LEVEL_RANK["N2"] < LEVEL_RANK["N1"]
assert set(LEVEL_RANK) == {"N5", "N4", "N3", "N2", "N1"}