"""JLPT level derivation on both axes: word level by JMdict-id join, kanji level by literal. D-09: word level is a join on JMdict ids against data/jlpt/n*.csv, never a string match. D-10: a word on no list is "N1+" - never silently guessed. D-12: proper nouns are "name". D-13: the kanji axis (data/jlpt/kanji_levels.json) is separate from the word axis, with null for a kanji the 2,211-entry list does not carry (above every level for furigana gating). Counts are the ones plan 02-01 measured on the pinned tag (02-01-SUMMARY.md): 7,748 unique ids (research said 7,747) and 447 ids on more than one LEVEL (research's 505 counted rows). """ from __future__ import annotations import csv import io from collections import defaultdict from pathlib import Path from japanese_avatar.nlp import jmdict, levels from japanese_avatar.nlp.levels import ( LEVEL_RANK, derive_level, kanji_levels, kanji_levels_for, vocab_levels, ) REPO_ROOT = Path(__file__).resolve().parent.parent JLPT = REPO_ROOT / "data" / "jlpt" EXPECTED_UNIQUE_IDS = 7748 EXPECTED_MULTI_LEVEL_IDS = 447 EXPECTED_KANJI = 2211 IRU_ORU = 1577980 KYOU = 1579110 HANASERU = 1562360 def _levels_by_id() -> dict[str, set[int]]: """Independent re-parse of the five CSVs: id -> every level number it is listed at.""" found: dict[str, set[int]] = defaultdict(set) for n in (5, 4, 3, 2, 1): text = (JLPT / f"n{n}.csv").read_bytes().decode("utf-8-sig") rows = list(csv.reader(io.StringIO(text, newline=""))) assert rows[0][0] == "jmdict_seq" for row in rows[1:]: if row and row[0]: found[row[0]].add(n) return found def test_vocab_levels_is_a_singleton(): assert vocab_levels() is vocab_levels() assert kanji_levels() is kanji_levels() def test_vocab_levels_count(): table = vocab_levels() assert len(table) == EXPECTED_UNIQUE_IDS assert set(table.values()) <= {1, 2, 3, 4, 5} assert all(isinstance(k, int) for k in table) def test_multi_level_ids_take_easiest(): by_id = _levels_by_id() assert len(by_id) == EXPECTED_UNIQUE_IDS multi = {i: ls for i, ls in by_id.items() if len(ls) > 1} assert len(multi) == EXPECTED_MULTI_LEVEL_IDS, "447 ids sit on more than one level" table = vocab_levels() wrong = {i: (table[int(i)], ls) for i, ls in multi.items() if table[int(i)] != max(ls)} assert wrong == {}, f"{len(wrong)} multi-level ids did not take the easiest level: {wrong}" # And single-level ids carry exactly their one level. for i, ls in by_id.items(): if len(ls) == 1: assert table[int(i)] == next(iter(ls)) def test_known_levels(): table = vocab_levels() def level(lemma: str, level_key: str, reading: str) -> int | None: entry = jmdict.lookup(lemma, level_key, reading, table) assert entry is not None, (lemma, reading) return table.get(entry.id) assert jmdict.lookup("いる", "居る", "いる", table).id == IRU_ORU assert table[IRU_ORU] == 5 assert jmdict.lookup("今日", "今日", "きょう", table).id == KYOU assert table[KYOU] == 5 assert level("人気", "人気", "にんき") == 3 assert level("会議", "会議", "かいぎ") == 4 assert level("行う", "行う", "おこなう") == 4 assert level("彼", "彼", "かれ") == 4 assert level("通り", "通り", "とおり") == 3 def test_derive_level_rules(): table = vocab_levels() assert derive_level(True, None, None) == "name" assert derive_level(False, None, None) == "N1+" hanasu = jmdict.lookup("話す", "話す", "はなす", table) hanaseru = jmdict.lookup("話せる", "話せる", "はなせる", table) assert hanasu is not None and hanaseru is not None and hanaseru.id == HANASERU assert HANASERU not in table, "話せる is on no list; only 話す (its level_key) is" # Level comes from the level_key entry (話す, N5) even though the gloss entry is 話せる. assert derive_level(False, hanasu, hanaseru) == "N5" # Only the gloss entry known, and it is unlisted -> honest N1+. assert derive_level(False, None, hanaseru) == "N1+" assert derive_level(False, hanaseru, None) == "N1+" # The gloss entry is consulted when the level entry misses the lists. assert derive_level(False, hanaseru, hanasu) == "N5" # A name is a name even when the entry is on a list. assert derive_level(True, hanasu, hanasu) == "name" def test_kanji_axis(): table = kanji_levels() assert len(table) == EXPECTED_KANJI assert set(table.values()) == {"N5", "N4", "N3", "N2", "N1"} assert kanji_levels_for("日本語") == {"日": "N5", "本": "N5", "語": "N5"} utsu = kanji_levels_for("鬱陶しい") assert utsu["鬱"] is None, "鬱 is not on the 2,211 list: unlisted, above every level" assert "し" not in utsu and "い" not in utsu assert kanji_levels_for("こんにちは") == {} assert kanji_levels_for("") == {} # 々 (U+3005) is a kanji-run character with no level of its own -> None. Known edge: the # run gate treats null as above every level, honest for a repeat whose base may be listed. hitobito = kanji_levels_for("人々") assert hitobito["人"] == "N5" and hitobito["々"] is None assert kanji_levels_for("xyz abc 123 、。") == {} def test_is_kanji_ranges_local_copy(): """levels.py carries its own _is_kanji (parallel wave with 02-02; 02-05 reconciles).""" assert levels._is_kanji("漢") and levels._is_kanji("々") and levels._is_kanji("㐀") assert levels._is_kanji("鿿") and levels._is_kanji("豈") and levels._is_kanji("﫿") assert not levels._is_kanji("あ") and not levels._is_kanji("ア") and not levels._is_kanji("a") assert not levels._is_kanji("ー") and not levels._is_kanji("。") def test_level_rank_order(): assert LEVEL_RANK["N5"] < LEVEL_RANK["N4"] < LEVEL_RANK["N3"] < LEVEL_RANK["N2"] assert LEVEL_RANK["N2"] < LEVEL_RANK["N1"] assert set(LEVEL_RANK) == {"N5", "N4", "N3", "N2", "N1"}