Spaces:
Running on Zero
Running on Zero
Download tests/test_levels.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 6.06 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f7e73d2aec50cd58e583c49da86b2c0de1485772/tests/test_levels.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@f7e73d2aec50cd58e583c49da86b2c0de1485772/tests/test_levels.py
-
curl -L -o test_levels.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/f7e73d2aec50cd58e583c49da86b2c0de1485772/tests/test_levels.py
6.06 kB
| """JLPT level derivation on both axes: word level by JMdict-id join, kanji level by literal. | |
| D-09: word level is a join on JMdict ids against data/jlpt/n*.csv, never a string match. | |
| D-10: a word on no list is "N1+" - never silently guessed. D-12: proper nouns are "name". | |
| D-13: the kanji axis (data/jlpt/kanji_levels.json) is separate from the word axis, with null | |
| for a kanji the 2,211-entry list does not carry (above every level for furigana gating). | |
| Counts are the ones plan 02-01 measured on the pinned tag (02-01-SUMMARY.md): 7,748 unique ids | |
| (research said 7,747) and 447 ids on more than one LEVEL (research's 505 counted rows). | |
| """ | |
| from __future__ import annotations | |
| import csv | |
| import io | |
| from collections import defaultdict | |
| from pathlib import Path | |
| from japanese_avatar.nlp import jmdict, levels | |
| from japanese_avatar.nlp.levels import ( | |
| LEVEL_RANK, | |
| derive_level, | |
| kanji_levels, | |
| kanji_levels_for, | |
| vocab_levels, | |
| ) | |
| REPO_ROOT = Path(__file__).resolve().parent.parent | |
| JLPT = REPO_ROOT / "data" / "jlpt" | |
| EXPECTED_UNIQUE_IDS = 7748 | |
| EXPECTED_MULTI_LEVEL_IDS = 447 | |
| EXPECTED_KANJI = 2211 | |
| IRU_ORU = 1577980 | |
| KYOU = 1579110 | |
| HANASERU = 1562360 | |
| def _levels_by_id() -> dict[str, set[int]]: | |
| """Independent re-parse of the five CSVs: id -> every level number it is listed at.""" | |
| found: dict[str, set[int]] = defaultdict(set) | |
| for n in (5, 4, 3, 2, 1): | |
| text = (JLPT / f"n{n}.csv").read_bytes().decode("utf-8-sig") | |
| rows = list(csv.reader(io.StringIO(text, newline=""))) | |
| assert rows[0][0] == "jmdict_seq" | |
| for row in rows[1:]: | |
| if row and row[0]: | |
| found[row[0]].add(n) | |
| return found | |
| def test_vocab_levels_is_a_singleton(): | |
| assert vocab_levels() is vocab_levels() | |
| assert kanji_levels() is kanji_levels() | |
| def test_vocab_levels_count(): | |
| table = vocab_levels() | |
| assert len(table) == EXPECTED_UNIQUE_IDS | |
| assert set(table.values()) <= {1, 2, 3, 4, 5} | |
| assert all(isinstance(k, int) for k in table) | |
| def test_multi_level_ids_take_easiest(): | |
| by_id = _levels_by_id() | |
| assert len(by_id) == EXPECTED_UNIQUE_IDS | |
| multi = {i: ls for i, ls in by_id.items() if len(ls) > 1} | |
| assert len(multi) == EXPECTED_MULTI_LEVEL_IDS, "447 ids sit on more than one level" | |
| table = vocab_levels() | |
| wrong = {i: (table[int(i)], ls) for i, ls in multi.items() if table[int(i)] != max(ls)} | |
| assert wrong == {}, f"{len(wrong)} multi-level ids did not take the easiest level: {wrong}" | |
| # And single-level ids carry exactly their one level. | |
| for i, ls in by_id.items(): | |
| if len(ls) == 1: | |
| assert table[int(i)] == next(iter(ls)) | |
| def test_known_levels(): | |
| table = vocab_levels() | |
| def level(lemma: str, level_key: str, reading: str) -> int | None: | |
| entry = jmdict.lookup(lemma, level_key, reading, table) | |
| assert entry is not None, (lemma, reading) | |
| return table.get(entry.id) | |
| assert jmdict.lookup("γγ", "ε± γ", "γγ", table).id == IRU_ORU | |
| assert table[IRU_ORU] == 5 | |
| assert jmdict.lookup("δ»ζ₯", "δ»ζ₯", "γγγ", table).id == KYOU | |
| assert table[KYOU] == 5 | |
| assert level("δΊΊζ°", "δΊΊζ°", "γ«γγ") == 3 | |
| assert level("δΌθ°", "δΌθ°", "γγγ") == 4 | |
| assert level("θ‘γ", "θ‘γ", "γγγͺγ") == 4 | |
| assert level("ε½Ό", "ε½Ό", "γγ") == 4 | |
| assert level("ιγ", "ιγ", "γ¨γγ") == 3 | |
| def test_derive_level_rules(): | |
| table = vocab_levels() | |
| assert derive_level(True, None, None) == "name" | |
| assert derive_level(False, None, None) == "N1+" | |
| hanasu = jmdict.lookup("θ©±γ", "θ©±γ", "γ―γͺγ", table) | |
| hanaseru = jmdict.lookup("θ©±γγ", "θ©±γγ", "γ―γͺγγ", table) | |
| assert hanasu is not None and hanaseru is not None and hanaseru.id == HANASERU | |
| assert HANASERU not in table, "θ©±γγ is on no list; only θ©±γ (its level_key) is" | |
| # Level comes from the level_key entry (θ©±γ, N5) even though the gloss entry is θ©±γγ. | |
| assert derive_level(False, hanasu, hanaseru) == "N5" | |
| # Only the gloss entry known, and it is unlisted -> honest N1+. | |
| assert derive_level(False, None, hanaseru) == "N1+" | |
| assert derive_level(False, hanaseru, None) == "N1+" | |
| # The gloss entry is consulted when the level entry misses the lists. | |
| assert derive_level(False, hanaseru, hanasu) == "N5" | |
| # A name is a name even when the entry is on a list. | |
| assert derive_level(True, hanasu, hanasu) == "name" | |
| def test_kanji_axis(): | |
| table = kanji_levels() | |
| assert len(table) == EXPECTED_KANJI | |
| assert set(table.values()) == {"N5", "N4", "N3", "N2", "N1"} | |
| assert kanji_levels_for("ζ₯ζ¬θͺ") == {"ζ₯": "N5", "ζ¬": "N5", "θͺ": "N5"} | |
| utsu = kanji_levels_for("鬱ιΆγγ") | |
| assert utsu["鬱"] is None, "鬱 is not on the 2,211 list: unlisted, above every level" | |
| assert "γ" not in utsu and "γ" not in utsu | |
| assert kanji_levels_for("γγγ«γ‘γ―") == {} | |
| assert kanji_levels_for("") == {} | |
| # γ (U+3005) is a kanji-run character with no level of its own -> None. Known edge: the | |
| # run gate treats null as above every level, honest for a repeat whose base may be listed. | |
| hitobito = kanji_levels_for("δΊΊγ ") | |
| assert hitobito["δΊΊ"] == "N5" and hitobito["γ "] is None | |
| assert kanji_levels_for("ο½ο½ο½ abc 123 γγ") == {} | |
| def test_is_kanji_ranges_local_copy(): | |
| """levels.py carries its own _is_kanji (parallel wave with 02-02; 02-05 reconciles).""" | |
| assert levels._is_kanji("ζΌ’") and levels._is_kanji("γ ") and levels._is_kanji("γ") | |
| assert levels._is_kanji("ιΏΏ") and levels._is_kanji("ο€") and levels._is_kanji("ο«Ώ") | |
| assert not levels._is_kanji("γ") and not levels._is_kanji("γ’") and not levels._is_kanji("a") | |
| assert not levels._is_kanji("γΌ") and not levels._is_kanji("γ") | |
| def test_level_rank_order(): | |
| assert LEVEL_RANK["N5"] < LEVEL_RANK["N4"] < LEVEL_RANK["N3"] < LEVEL_RANK["N2"] | |
| assert LEVEL_RANK["N2"] < LEVEL_RANK["N1"] | |
| assert set(LEVEL_RANK) == {"N5", "N4", "N3", "N2", "N1"} | |