Spaces:
Running on Zero
Running on Zero
Download tests/test_jmdict.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 5.38 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/fbd985ff7754c8ea5773c2af17e7023a7b8e24fd/tests/test_jmdict.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@fbd985ff7754c8ea5773c2af17e7023a7b8e24fd/tests/test_jmdict.py
-
curl -L -o test_jmdict.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/fbd985ff7754c8ea5773c2af17e7023a7b8e24fd/tests/test_jmdict.py
5.38 kB
| """The compact JMdict is a read-only singleton whose ranked lookup resolves homographs. | |
| 02-RESEARCH.md Β§ Common Pitfalls 2 measured the naive join picking η γ for γγ, ζγ for γγ | |
| and ι γ for γγ. The ranking (reading match -> on a JLPT list -> JMdict ``common`` -> file | |
| order) is what makes the Sudachi-lemma -> JMdict-id step deterministic and right; these tests | |
| pin the homographs research measured after the fix, and 02-05's fixture set pins the rest. | |
| """ | |
| from __future__ import annotations | |
| import time | |
| import pytest | |
| from japanese_avatar.nlp import jmdict | |
| # Ids from 02-RESEARCH.md Β§ Q2 - Join strategy (spot checks after the ranking fix) and | |
| # data/jmdict/README.md; all asserted by tests/test_data_assets.py to exist in the file. | |
| IRU_ORU = 1577980 # ε± γ | |
| IRU_ROAST = 1391500 # η γ - what the first draft picked for γγ | |
| ARU = 1296400 # ζγ | |
| NARU = 1375610 # ζγ | |
| KONNICHIWA = 1289400 # γγγ«γ‘γ― (kanji forms δ»ζ₯γ― rK / δ»ζ₯γ sK) | |
| KYOU = 1579110 # δ»ζ₯, readings γγγ and γγγ«γ‘ | |
| HANASERU = 1562360 # θ©±γγ - its own entry, on no JLPT list | |
| ENTRIES = 218672 | |
| MIN_HEADWORD_KEYS = 400_000 # research counted 453,630 kanji + kana headword keys | |
| def _level_of() -> dict[int, int]: | |
| """The real JLPT join when levels.py (Task 2) exists, else the two ids these tests need.""" | |
| try: | |
| from japanese_avatar.nlp.levels import vocab_levels | |
| except ImportError: | |
| return {ARU: 5, NARU: 5, IRU_ORU: 5} | |
| return dict(vocab_levels()) | |
| def lexicon() -> jmdict.Lexicon: | |
| t = time.perf_counter() | |
| lex = jmdict.load() | |
| seconds = time.perf_counter() - t | |
| print(f"\ncompact JMdict load(): {seconds:.2f} s") | |
| assert seconds < 5, f"load took {seconds:.2f} s; research measured ~1.8 s" | |
| return lex | |
| def test_load_is_a_singleton(lexicon): | |
| assert jmdict.load() is lexicon | |
| assert jmdict.load() is jmdict.load() | |
| assert len(lexicon.by_id) == ENTRIES | |
| # warmup() is idempotent on the cache: a second call costs nothing measurable. | |
| assert jmdict.warmup() < 0.05 | |
| def test_indexes_cover_headwords(lexicon): | |
| assert IRU_ORU in {e.id for e in lexicon.by_kanji["ε± γ"]} | |
| iru_ids = {e.id for e in lexicon.by_kana["γγ"]} | |
| assert IRU_ORU in iru_ids | |
| assert len(iru_ids) > 1, "γγ must index its homographs, not one entry" | |
| assert KONNICHIWA in {e.id for e in lexicon.by_kana["γγγ«γ‘γ―"]} | |
| assert len(lexicon.by_kanji) + len(lexicon.by_kana) > MIN_HEADWORD_KEYS | |
| def test_entries_are_frozen_records(lexicon): | |
| e = lexicon.by_id[IRU_ORU] | |
| assert isinstance(e, jmdict.Entry) | |
| assert e.id == IRU_ORU and "ε± γ" in e.kanji and "γγ" in e.kana | |
| assert isinstance(e.common, bool) and e.common is True | |
| assert isinstance(e.order, int) and 0 <= e.order < ENTRIES | |
| assert all(isinstance(sense, tuple) for sense in e.glosses) | |
| with pytest.raises(AttributeError): | |
| e.id = 0 # type: ignore[misc] | |
| def test_iru_is_oru_not_iru_roast(lexicon): | |
| hit = jmdict.lookup("γγ", "ε± γ", "γγ", {IRU_ORU: 5}) | |
| assert hit is not None and hit.id == IRU_ORU | |
| # Without any JLPT knowledge the reading match plus `common` still beats η γ. | |
| bare = jmdict.lookup("γγ", "γγ", "γγ", {}) | |
| assert bare is not None and bare.id != IRU_ROAST | |
| assert bare.id == IRU_ORU | |
| def test_aru_naru_kore(lexicon): | |
| level_of = _level_of() | |
| aru = jmdict.lookup("γγ", "ζγ", "γγ", level_of) | |
| assert aru is not None and aru.id == ARU | |
| naru = jmdict.lookup("γͺγ", "ζγ", "γͺγ", level_of) | |
| assert naru is not None and naru.id == NARU | |
| kore = jmdict.lookup("γγ", "γγ", "γγ", level_of) | |
| assert kore is not None and "γγ" in kore.kana | |
| def test_kana_headword_fallback(lexicon): | |
| hit = jmdict.lookup("γγγ«γ‘γ―", "δ»ζ₯γ―", "γγγ«γ‘γ―", {}) | |
| assert hit is not None and hit.id == KONNICHIWA | |
| def test_hanaseru_has_its_own_entry(lexicon): | |
| hanaseru = jmdict.lookup("θ©±γγ", "θ©±γγ", "γ―γͺγγ", {}) | |
| assert hanaseru is not None and hanaseru.id == HANASERU | |
| glosses = [g for sense in jmdict.glosses_for(hanaseru) for g in sense] | |
| assert any("to be able to speak" in g for g in glosses), glosses | |
| hanasu = jmdict.lookup("θ©±γ", "θ©±γ", "γ―γͺγ", {}) | |
| assert hanasu is not None and hanasu.id != HANASERU | |
| def test_kyou(lexicon): | |
| for reading in ("γγγ", "γγγ«γ‘"): | |
| hit = jmdict.lookup("δ»ζ₯", "δ»ζ₯", reading, {}) | |
| assert hit is not None and hit.id == KYOU, (reading, hit) | |
| def test_glosses_for_caps_senses(lexicon): | |
| entry = lexicon.by_id[IRU_ORU] | |
| assert len(jmdict.glosses_for(entry, max_senses=1)) == 1 | |
| assert len(jmdict.glosses_for(entry)) <= 3 | |
| assert all(isinstance(g, str) and g for sense in jmdict.glosses_for(entry) for g in sense) | |
| def test_no_hit_returns_none(lexicon): | |
| assert jmdict.lookup("ο½ο½ο½", "ο½ο½ο½", "", {}) is None | |
| assert jmdict.lookup("", "", "", {}) is None | |
| def test_lookup_is_fast(lexicon): | |
| level_of = {IRU_ORU: 5} | |
| t = time.perf_counter() | |
| for _ in range(10_000): | |
| jmdict.lookup("γγ", "ε± γ", "γγ", level_of) | |
| ms = (time.perf_counter() - t) * 1000 | |
| print(f"\n10,000 lookups of ε± γ: {ms:.1f} ms") | |
| assert ms < 50, f"{ms:.1f} ms for 10k lookups" | |