Spaces:
Running on Zero
Running on Zero
Download tests/test_analyzer.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 15.8 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/03d5ccf002944b946c90c8a7b5e432a5a156aa6e/tests/test_analyzer.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@03d5ccf002944b946c90c8a7b5e432a5a156aa6e/tests/test_analyzer.py
-
curl -L -o test_analyzer.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/03d5ccf002944b946c90c8a7b5e432a5a156aa6e/tests/test_analyzer.py
15.8 kB
| """The ONE canonical ``analyze(text)`` (plan 02-05, PITFALLS § 10) and its token record. | |
| Every later system - furigana (02-07), the lookup popover (02-08), Phase 3's level guard, | |
| Phase 5's vocabulary tracking - consumes the record these tests pin and nothing else. Each | |
| CONTEXT decision is a named test: D-10 (unlisted -> "N1+"), D-11 (particles / auxiliaries / | |
| punctuation are not words), D-12 (proper nouns -> "name"), D-13 (word axis and kanji axis are | |
| separate), plus the committed reading override (人気 -> ひとけ before の) and the known-ambiguous | |
| readings PITFALLS names (今日, 行った x2, 人気 x2). | |
| Facts recorded by the wave-2 summaries and pinned knowingly here (02-03-SUMMARY.md § Known | |
| Edges): なる resolves to the verb 1375610, which is on the N3 list (n5.csv's なる row is the | |
| archaic copula 2138260 - an upstream yomitan-jlpt-vocab quirk); いい resolves to its own JMdict | |
| entry 2820690 (N5), not 良い 1605820. | |
| All tests use the session-scoped ``analyzer`` fixture (tests/conftest.py): one dictionary open | |
| and one compact-JMdict load per session. | |
| """ | |
| from __future__ import annotations | |
| import concurrent.futures | |
| import functools | |
| import importlib.metadata | |
| import json | |
| import time | |
| from pathlib import Path | |
| from japanese_avatar.nlp.analyzer import TOKEN_KEYS | |
| FIXTURES = Path(__file__).parent / "fixtures" | |
| GOLDEN = FIXTURES / "sentences.json" | |
| LONG_36_MORA = "今日はいい天気ですから、公園を散歩してから、買い物に行きました。" | |
| REGENERATE = ( | |
| "regenerate with .venv/Scripts/python.exe tests/fixtures/make_sentences.py " | |
| "and REVIEW the diff - a difference means SudachiDict, JMdict, the JLPT lists " | |
| "or the joining rule changed what learners read" | |
| ) | |
| def _golden() -> dict: | |
| """The fixed sample sentence set (SC-1): ``{"analyzer": pins, "sentences": [...]}``.""" | |
| return json.loads(GOLDEN.read_text(encoding="utf-8")) | |
| def _golden_texts() -> list[str]: | |
| return [s["text"] for s in _golden()["sentences"]] | |
| def _by_surface(units: list[dict], surface: str) -> dict: | |
| matches = [u for u in units if u["surface"] == surface] | |
| assert len(matches) == 1, f"expected one unit {surface!r}, got {[u['surface'] for u in units]}" | |
| return matches[0] | |
| def test_record_keys_exact(analyzer): | |
| """Every unit carries exactly TOKEN_KEYS, in order - the contract Phases 3 and 5 read.""" | |
| units = analyzer("日本語を勉強しています。") | |
| assert units, "a sentence yields units" | |
| for u in units: | |
| assert tuple(u.keys()) == TOKEN_KEYS, u | |
| assert TOKEN_KEYS == ( | |
| "surface", | |
| "reading", | |
| "lemma", | |
| "level_key", | |
| "pos", | |
| "tappable", | |
| "is_name", | |
| "jlpt", | |
| "kanji_levels", | |
| "ruby", | |
| "jmdict_id", | |
| "gloss", | |
| "start", | |
| "end", | |
| ) | |
| def test_reading_disambiguation(analyzer): | |
| """The known-ambiguous readings PITFALLS § 10 names: 今日 and the two 行った.""" | |
| kyou = _by_surface(analyzer("今日はいい天気ですね。"), "今日") | |
| assert kyou["reading"] == "きょう" | |
| itta = _by_surface(analyzer("昨日、駅に行った。"), "行った") | |
| assert itta["reading"] == "いった" | |
| assert itta["lemma"] == "行く" | |
| okonatta = _by_surface(analyzer("会議を行った。"), "行った") | |
| assert okonatta["reading"] == "おこなった" | |
| assert okonatta["lemma"] == "行う" | |
| def test_hitoke_override(analyzer): | |
| """人気 is ひとけ before の (the committed override) and にんき N3 elsewhere.""" | |
| hitoke = _by_surface(analyzer("人気のない通りを歩いた。"), "人気") | |
| assert hitoke["reading"] == "ひとけ" | |
| assert hitoke["ruby"] == [["人気", "ひとけ"]] | |
| ninki = _by_surface(analyzer("この店は人気がある。"), "人気") | |
| assert ninki["reading"] == "にんき" | |
| assert ninki["jlpt"] == "N3" | |
| def test_tappability(analyzer): | |
| """D-11: は / です / ね / 。 are plain text with no level and no gloss; words are tappable.""" | |
| units = analyzer("今日はいい天気ですね。") | |
| assert [u["surface"] for u in units] == ["今日", "は", "いい", "天気", "です", "ね", "。"] | |
| for surface in ("は", "です", "ね", "。"): | |
| u = _by_surface(units, surface) | |
| assert u["tappable"] is False, surface | |
| assert u["jlpt"] is None, surface | |
| assert u["gloss"] == [], surface | |
| assert u["jmdict_id"] is None, surface | |
| assert u["kanji_levels"] == {}, surface | |
| assert u["ruby"] == [[surface, None]], surface | |
| for surface in ("今日", "いい", "天気"): | |
| u = _by_surface(units, surface) | |
| assert u["tappable"] is True, surface | |
| assert u["jlpt"] == "N5", surface | |
| def test_level_derivation(analyzer): | |
| """D-12 name, level from the level_key entry, D-10 N1+ for a word on no list.""" | |
| units = analyzer("田中さんは東京に住んでいます。") | |
| tanaka = _by_surface(units, "田中さん") | |
| assert tanaka["is_name"] is True and tanaka["jlpt"] == "name" | |
| assert tanaka["tappable"] is True, "D-12: names are tappable (the reading is the useful part)" | |
| tokyo = _by_surface(units, "東京") | |
| assert tokyo["is_name"] is True and tokyo["jlpt"] == "name" | |
| sunde = _by_surface(units, "住んでいます") | |
| assert sunde["lemma"] == "住む" and sunde["jlpt"] == "N5" | |
| # 話せる has its own, unlisted JMdict entry; its level comes from level_key 話す (N5). | |
| hanaseru = _by_surface(analyzer("話せるようになりたい"), "話せる") | |
| assert hanaseru["level_key"] == "話す" | |
| assert hanaseru["jlpt"] == "N5" | |
| assert hanaseru["jmdict_id"] == 1562360 | |
| assert any("able to speak" in g for g in hanaseru["gloss"][0]), hanaseru["gloss"] | |
| # 鬱 is not one of the 2,211 listed kanji: kanji axis None (D-13 / D-02). | |
| uttoushii = _by_surface(analyzer("鬱陶しい天気だ。"), "鬱陶しい") | |
| assert uttoushii["kanji_levels"]["鬱"] is None | |
| assert uttoushii["tappable"] is True | |
| # D-10: a word on no list is "N1+" - never a guess. The fixture pins 鬱陶しい's observed | |
| # label (N1 or N1+, whichever the pinned lists say); the D-10 assertion itself uses a rare | |
| # word that has a JMdict entry and sits on no list, so "N1+" comes from the rule, not from | |
| # a lookup miss. | |
| assert uttoushii["jlpt"] in {"N1", "N1+"} | |
| beyond = _by_surface(analyzer("彼は瀟洒な服を着ている。"), "瀟洒") | |
| assert beyond["tappable"] is True | |
| assert beyond["jmdict_id"] is not None, "瀟洒 has a JMdict entry" | |
| assert beyond["jlpt"] == "N1+", "on no JLPT list -> N1+ (D-10)" | |
| def test_kanji_axis_separate_from_word_axis(analyzer): | |
| """D-13: the word's level and each kanji's level are two axes in the same record.""" | |
| nihongo = _by_surface(analyzer("日本語を勉強しています。"), "日本語") | |
| # 日本語 resolves to JMdict 1464530, which is on NONE of the five pinned lists (Waller's | |
| # lists carry 日本 at N3 and no 日本語 at all) - so the word axis is the honest D-10 "N1+" | |
| # while every kanji in it is N5. The two axes really are independent; the plan's assumed | |
| # "N5" was a guess the pinned data contradicts (02-05-SUMMARY.md). | |
| assert nihongo["jmdict_id"] == 1464530 | |
| assert nihongo["jlpt"] == "N1+" | |
| assert nihongo["kanji_levels"] == {"日": "N5", "本": "N5", "語": "N5"} | |
| benkyou = _by_surface(analyzer("日本語を勉強しています。"), "勉強しています") | |
| assert benkyou["lemma"] == "勉強する" | |
| assert set(benkyou["kanji_levels"]) == {"勉", "強"} | |
| assert all(v in {"N5", "N4", "N3", "N2", "N1"} for v in benkyou["kanji_levels"].values()) | |
| # The word is N5 while its kanji are harder - the axes really are independent. | |
| assert benkyou["jlpt"] == "N5" | |
| assert benkyou["kanji_levels"]["勉"] != "N5" | |
| def test_gloss_shipped_with_token(analyzer): | |
| """D-07 / research § Q3: glosses ride inside the token so a tap needs no round trip.""" | |
| units = analyzer("食べました") | |
| assert len(units) == 1 | |
| u = units[0] | |
| assert u["lemma"] == "食べる" | |
| assert isinstance(u["jmdict_id"], int) | |
| assert u["gloss"], "the token carries its glosses" | |
| assert any("eat" in g for g in u["gloss"][0]), u["gloss"] | |
| assert len(u["gloss"]) <= 3 and all(len(sense) <= 3 for sense in u["gloss"]) | |
| assert u["ruby"] == [["食", "た"], ["べました", None]] | |
| def test_offsets_cover_text(analyzer): | |
| """Units tile every fixture sentence and each offset pair reads back its surface.""" | |
| for text in _golden_texts(): | |
| units = analyzer(text) | |
| assert "".join(u["surface"] for u in units) == text | |
| pos = 0 | |
| for u in units: | |
| assert u["start"] == pos, (text, u) | |
| assert u["end"] == pos + len(u["surface"]), (text, u) | |
| assert text[u["start"] : u["end"]] == u["surface"], (text, u) | |
| pos = u["end"] | |
| assert pos == len(text) | |
| def test_analyze_is_fast(analyzer): | |
| """Warm analysis is milliseconds: 100 runs of the 36-mora sentence, mean well under 50 ms.""" | |
| analyzer(LONG_36_MORA) # warm any lazily built table | |
| runs = 100 | |
| started = time.perf_counter() | |
| for _ in range(runs): | |
| analyzer(LONG_36_MORA) | |
| mean_ms = (time.perf_counter() - started) * 1000 / runs | |
| print(f"\nanalyze({len(LONG_36_MORA)} chars): mean {mean_ms:.3f} ms over {runs} runs") | |
| assert mean_ms < 50, f"{mean_ms:.1f} ms per analysis (budget 50 ms; expected < 5 ms)" | |
| def test_analyze_is_thread_safe(analyzer): | |
| """Plan 02-07: concurrent analyses all succeed and all agree. The Space serves them. | |
| ``sudachipy.Tokenizer`` is a PyO3 object that takes a mutable borrow for the duration of | |
| a ``tokenize`` call, so ONE shared instance answers the first caller and raises | |
| ``RuntimeError: Already borrowed`` at every thread that arrives while it is busy - a lost | |
| analysis, not a wrong one. Before ``nlp.tokenizer`` gave each thread its own tokenizer, | |
| 8 threads on this sentence lost 5 of 8 calls; the deployed symptom was a directive | |
| carrying ``tokens: []`` and an avatar line rendered with no furigana at all. | |
| It became reachable in 02-07 because the host page now analyses the learner's line while | |
| the turn's own analyse stage runs, so a single typed turn issues two overlapping | |
| requests - and a busy Space multiplies that by its visitors. | |
| A real regression test: reverting tokenizer.get_tokenizer to one shared instance fails | |
| this, and the failure is the exception, not a flaky count. | |
| """ | |
| workers = 8 | |
| rounds = 4 | |
| expected = analyzer(LONG_36_MORA) | |
| with concurrent.futures.ThreadPoolExecutor(max_workers=workers) as pool: | |
| futures = [pool.submit(analyzer, LONG_36_MORA) for _ in range(workers * rounds)] | |
| results = [f.result() for f in futures] | |
| assert len(results) == workers * rounds | |
| for i, units in enumerate(results): | |
| assert units == expected, f"thread {i} disagreed with the single-threaded analysis" | |
| def test_empty_text(analyzer): | |
| """Nothing in, nothing out; whitespace is a non-tappable unit with no level.""" | |
| assert analyzer("") == [] | |
| units = analyzer(" ") | |
| assert "".join(u["surface"] for u in units) == " " | |
| for u in units: | |
| assert u["surface"].strip() == "" | |
| assert u["tappable"] is False and u["jlpt"] is None and u["gloss"] == [] | |
| assert tuple(u.keys()) == TOKEN_KEYS | |
| def test_known_list_gaps_are_honest_n1_plus(analyzer): | |
| """Words the pinned lists simply do not carry badge "N1+" (D-10) - pinned by JMdict id. | |
| These look wrong to a Japanese teacher and are right by the data: none of the five | |
| yomitan-jlpt-vocab CSVs lists ない / 無い (1529520) or 日本語 (1464530), and 人気 read ひとけ is | |
| the separate entry 1367020 (unlisted), not にんき's 1367010 (N3). The lookup resolves each to | |
| the correct entry; the label follows the lists. If the product later wants a level alias for | |
| them, that is a new override table and a conscious change to this test, never a weaker lookup. | |
| """ | |
| nai = _by_surface(analyzer("人気のない通りを歩いた。"), "ない") | |
| assert nai["tappable"] is True and nai["lemma"] == "ない" and nai["level_key"] == "無い" | |
| assert nai["jmdict_id"] == 1529520 and nai["jlpt"] == "N1+" | |
| nakatta = _by_surface(analyzer("きれいじゃなかった"), "なかった") | |
| assert nakatta["jmdict_id"] == 1529520 and nakatta["jlpt"] == "N1+" | |
| hitoke = _by_surface(analyzer("人気のない通りを歩いた。"), "人気") | |
| assert hitoke["jmdict_id"] == 1367020 and hitoke["jlpt"] == "N1+" | |
| assert any("human presence" in g for g in hitoke["gloss"][0]), hitoke["gloss"] | |
| ninki = _by_surface(analyzer("この店は人気がある。"), "人気") | |
| assert ninki["jmdict_id"] == 1367010 and ninki["jlpt"] == "N3" | |
| # The other side of the same coin: こんにちは IS listed, at N3 (and N1), so it badges N3. | |
| konnichiwa = analyzer("こんにちは") | |
| assert len(konnichiwa) == 1 | |
| assert konnichiwa[0]["jmdict_id"] == 1289400 and konnichiwa[0]["jlpt"] == "N3" | |
| # --- the fixed sample sentence set (success criterion 1) --------------------------------- | |
| def test_fixture_sentences(analyzer): | |
| """The analyzer reproduces every golden unit of every fixture sentence exactly. | |
| This is JPN-01's quick-loop row (02-VALIDATION.md). A SudachiDict, JMdict or JLPT-list bump, | |
| or a change to the joining rule, must show up here as a reviewed diff - never as a silent | |
| change to what learners read. | |
| """ | |
| golden = _golden() | |
| assert len(golden["sentences"]) == 28 | |
| for sentence in golden["sentences"]: | |
| text, expected = sentence["text"], sentence["units"] | |
| got = analyzer(text) | |
| assert got == expected, f"{text!r}: analyze() differs from sentences.json - {REGENERATE}" | |
| def test_fixture_set_covers_the_contract(): | |
| """The set holds every case the phase promised to pin (ambiguous readings, joins, edges).""" | |
| texts = _golden_texts() | |
| assert len(texts) >= 25 | |
| joined = "\n".join(texts) | |
| for needle in ( | |
| "今日", | |
| "食べました", | |
| "田中さん", | |
| "勉強し", | |
| "読んで", | |
| "待ち合わせ", | |
| "鬱", | |
| "美味しくない", | |
| "きれいじゃなかった", | |
| ): | |
| assert needle in joined, needle | |
| assert sum("行った" in t for t in texts) >= 2, "both readings of 行った" | |
| assert sum("人気" in t for t in texts) >= 2, "both readings of 人気" | |
| # The known-ambiguous readings are pinned as READINGS, not just as texts. | |
| readings = { | |
| (s["text"], u["surface"]): u["reading"] for s in _golden()["sentences"] for u in s["units"] | |
| } | |
| assert readings[("今日はいい天気ですね。", "今日")] == "きょう" | |
| assert readings[("昨日、駅に行った。", "行った")] == "いった" | |
| assert readings[("会議を行った。", "行った")] == "おこなった" | |
| assert readings[("人気のない通りを歩いた。", "人気")] == "ひとけ" | |
| assert readings[("この店は人気がある。", "人気")] == "にんき" | |
| def test_fixture_pins_match_installed(): | |
| """The golden file was generated by the data this environment has, or it is stale.""" | |
| import sudachipy | |
| from japanese_avatar.nlp import jmdict | |
| pins = _golden()["analyzer"] | |
| assert pins["sudachipy"] == sudachipy.__version__ | |
| assert pins["sudachidict_core"] == importlib.metadata.version("sudachidict_core") | |
| assert pins["jmdict"] == jmdict.load().meta["version"] | |
| assert pins["jlpt_vocab"] == "2025.08.01.0" | |
| assert pins["kanji_data"] == "00fd7079c3890f430759536f91aa5e854ec0ca4f" | |