japanese-learning-avatar / tests /test_analyzer.py
WolfDavid's picture
test(02-07): rehearsal script; furigana at three layers, and a per-thread Sudachi tokenizer
bdeb461
Raw History Blame
15.8 kB
"""The ONE canonical ``analyze(text)`` (plan 02-05, PITFALLS § 10) and its token record.
Every later system - furigana (02-07), the lookup popover (02-08), Phase 3's level guard,
Phase 5's vocabulary tracking - consumes the record these tests pin and nothing else. Each
CONTEXT decision is a named test: D-10 (unlisted -> "N1+"), D-11 (particles / auxiliaries /
punctuation are not words), D-12 (proper nouns -> "name"), D-13 (word axis and kanji axis are
separate), plus the committed reading override (人気 -> ひとけ before の) and the known-ambiguous
readings PITFALLS names (今日, 行った x2, 人気 x2).
Facts recorded by the wave-2 summaries and pinned knowingly here (02-03-SUMMARY.md § Known
Edges): なる resolves to the verb 1375610, which is on the N3 list (n5.csv's なる row is the
archaic copula 2138260 - an upstream yomitan-jlpt-vocab quirk); いい resolves to its own JMdict
entry 2820690 (N5), not 良い 1605820.
All tests use the session-scoped ``analyzer`` fixture (tests/conftest.py): one dictionary open
and one compact-JMdict load per session.
"""
from __future__ import annotations
import concurrent.futures
import functools
import importlib.metadata
import json
import time
from pathlib import Path
from japanese_avatar.nlp.analyzer import TOKEN_KEYS
FIXTURES = Path(__file__).parent / "fixtures"
GOLDEN = FIXTURES / "sentences.json"
LONG_36_MORA = "今日はいい天気ですから、公園を散歩してから、買い物に行きました。"
REGENERATE = (
"regenerate with .venv/Scripts/python.exe tests/fixtures/make_sentences.py "
"and REVIEW the diff - a difference means SudachiDict, JMdict, the JLPT lists "
"or the joining rule changed what learners read"
)
@functools.cache
def _golden() -> dict:
"""The fixed sample sentence set (SC-1): ``{"analyzer": pins, "sentences": [...]}``."""
return json.loads(GOLDEN.read_text(encoding="utf-8"))
def _golden_texts() -> list[str]:
return [s["text"] for s in _golden()["sentences"]]
def _by_surface(units: list[dict], surface: str) -> dict:
matches = [u for u in units if u["surface"] == surface]
assert len(matches) == 1, f"expected one unit {surface!r}, got {[u['surface'] for u in units]}"
return matches[0]
def test_record_keys_exact(analyzer):
"""Every unit carries exactly TOKEN_KEYS, in order - the contract Phases 3 and 5 read."""
units = analyzer("日本語を勉強しています。")
assert units, "a sentence yields units"
for u in units:
assert tuple(u.keys()) == TOKEN_KEYS, u
assert TOKEN_KEYS == (
"surface",
"reading",
"lemma",
"level_key",
"pos",
"tappable",
"is_name",
"jlpt",
"kanji_levels",
"ruby",
"jmdict_id",
"gloss",
"start",
"end",
)
def test_reading_disambiguation(analyzer):
"""The known-ambiguous readings PITFALLS § 10 names: 今日 and the two 行った."""
kyou = _by_surface(analyzer("今日はいい天気ですね。"), "今日")
assert kyou["reading"] == "きょう"
itta = _by_surface(analyzer("昨日、駅に行った。"), "行った")
assert itta["reading"] == "いった"
assert itta["lemma"] == "行く"
okonatta = _by_surface(analyzer("会議を行った。"), "行った")
assert okonatta["reading"] == "おこなった"
assert okonatta["lemma"] == "行う"
def test_hitoke_override(analyzer):
"""人気 is ひとけ before の (the committed override) and にんき N3 elsewhere."""
hitoke = _by_surface(analyzer("人気のない通りを歩いた。"), "人気")
assert hitoke["reading"] == "ひとけ"
assert hitoke["ruby"] == [["人気", "ひとけ"]]
ninki = _by_surface(analyzer("この店は人気がある。"), "人気")
assert ninki["reading"] == "にんき"
assert ninki["jlpt"] == "N3"
def test_tappability(analyzer):
"""D-11: は / です / ね / 。 are plain text with no level and no gloss; words are tappable."""
units = analyzer("今日はいい天気ですね。")
assert [u["surface"] for u in units] == ["今日", "は", "いい", "天気", "です", "ね", "。"]
for surface in ("は", "です", "ね", "。"):
u = _by_surface(units, surface)
assert u["tappable"] is False, surface
assert u["jlpt"] is None, surface
assert u["gloss"] == [], surface
assert u["jmdict_id"] is None, surface
assert u["kanji_levels"] == {}, surface
assert u["ruby"] == [[surface, None]], surface
for surface in ("今日", "いい", "天気"):
u = _by_surface(units, surface)
assert u["tappable"] is True, surface
assert u["jlpt"] == "N5", surface
def test_level_derivation(analyzer):
"""D-12 name, level from the level_key entry, D-10 N1+ for a word on no list."""
units = analyzer("田中さんは東京に住んでいます。")
tanaka = _by_surface(units, "田中さん")
assert tanaka["is_name"] is True and tanaka["jlpt"] == "name"
assert tanaka["tappable"] is True, "D-12: names are tappable (the reading is the useful part)"
tokyo = _by_surface(units, "東京")
assert tokyo["is_name"] is True and tokyo["jlpt"] == "name"
sunde = _by_surface(units, "住んでいます")
assert sunde["lemma"] == "住む" and sunde["jlpt"] == "N5"
# 話せる has its own, unlisted JMdict entry; its level comes from level_key 話す (N5).
hanaseru = _by_surface(analyzer("話せるようになりたい"), "話せる")
assert hanaseru["level_key"] == "話す"
assert hanaseru["jlpt"] == "N5"
assert hanaseru["jmdict_id"] == 1562360
assert any("able to speak" in g for g in hanaseru["gloss"][0]), hanaseru["gloss"]
# 鬱 is not one of the 2,211 listed kanji: kanji axis None (D-13 / D-02).
uttoushii = _by_surface(analyzer("鬱陶しい天気だ。"), "鬱陶しい")
assert uttoushii["kanji_levels"]["鬱"] is None
assert uttoushii["tappable"] is True
# D-10: a word on no list is "N1+" - never a guess. The fixture pins 鬱陶しい's observed
# label (N1 or N1+, whichever the pinned lists say); the D-10 assertion itself uses a rare
# word that has a JMdict entry and sits on no list, so "N1+" comes from the rule, not from
# a lookup miss.
assert uttoushii["jlpt"] in {"N1", "N1+"}
beyond = _by_surface(analyzer("彼は瀟洒な服を着ている。"), "瀟洒")
assert beyond["tappable"] is True
assert beyond["jmdict_id"] is not None, "瀟洒 has a JMdict entry"
assert beyond["jlpt"] == "N1+", "on no JLPT list -> N1+ (D-10)"
def test_kanji_axis_separate_from_word_axis(analyzer):
"""D-13: the word's level and each kanji's level are two axes in the same record."""
nihongo = _by_surface(analyzer("日本語を勉強しています。"), "日本語")
# 日本語 resolves to JMdict 1464530, which is on NONE of the five pinned lists (Waller's
# lists carry 日本 at N3 and no 日本語 at all) - so the word axis is the honest D-10 "N1+"
# while every kanji in it is N5. The two axes really are independent; the plan's assumed
# "N5" was a guess the pinned data contradicts (02-05-SUMMARY.md).
assert nihongo["jmdict_id"] == 1464530
assert nihongo["jlpt"] == "N1+"
assert nihongo["kanji_levels"] == {"日": "N5", "本": "N5", "語": "N5"}
benkyou = _by_surface(analyzer("日本語を勉強しています。"), "勉強しています")
assert benkyou["lemma"] == "勉強する"
assert set(benkyou["kanji_levels"]) == {"勉", "強"}
assert all(v in {"N5", "N4", "N3", "N2", "N1"} for v in benkyou["kanji_levels"].values())
# The word is N5 while its kanji are harder - the axes really are independent.
assert benkyou["jlpt"] == "N5"
assert benkyou["kanji_levels"]["勉"] != "N5"
def test_gloss_shipped_with_token(analyzer):
"""D-07 / research § Q3: glosses ride inside the token so a tap needs no round trip."""
units = analyzer("食べました")
assert len(units) == 1
u = units[0]
assert u["lemma"] == "食べる"
assert isinstance(u["jmdict_id"], int)
assert u["gloss"], "the token carries its glosses"
assert any("eat" in g for g in u["gloss"][0]), u["gloss"]
assert len(u["gloss"]) <= 3 and all(len(sense) <= 3 for sense in u["gloss"])
assert u["ruby"] == [["食", "た"], ["べました", None]]
def test_offsets_cover_text(analyzer):
"""Units tile every fixture sentence and each offset pair reads back its surface."""
for text in _golden_texts():
units = analyzer(text)
assert "".join(u["surface"] for u in units) == text
pos = 0
for u in units:
assert u["start"] == pos, (text, u)
assert u["end"] == pos + len(u["surface"]), (text, u)
assert text[u["start"] : u["end"]] == u["surface"], (text, u)
pos = u["end"]
assert pos == len(text)
def test_analyze_is_fast(analyzer):
"""Warm analysis is milliseconds: 100 runs of the 36-mora sentence, mean well under 50 ms."""
analyzer(LONG_36_MORA) # warm any lazily built table
runs = 100
started = time.perf_counter()
for _ in range(runs):
analyzer(LONG_36_MORA)
mean_ms = (time.perf_counter() - started) * 1000 / runs
print(f"\nanalyze({len(LONG_36_MORA)} chars): mean {mean_ms:.3f} ms over {runs} runs")
assert mean_ms < 50, f"{mean_ms:.1f} ms per analysis (budget 50 ms; expected < 5 ms)"
def test_analyze_is_thread_safe(analyzer):
"""Plan 02-07: concurrent analyses all succeed and all agree. The Space serves them.
``sudachipy.Tokenizer`` is a PyO3 object that takes a mutable borrow for the duration of
a ``tokenize`` call, so ONE shared instance answers the first caller and raises
``RuntimeError: Already borrowed`` at every thread that arrives while it is busy - a lost
analysis, not a wrong one. Before ``nlp.tokenizer`` gave each thread its own tokenizer,
8 threads on this sentence lost 5 of 8 calls; the deployed symptom was a directive
carrying ``tokens: []`` and an avatar line rendered with no furigana at all.
It became reachable in 02-07 because the host page now analyses the learner's line while
the turn's own analyse stage runs, so a single typed turn issues two overlapping
requests - and a busy Space multiplies that by its visitors.
A real regression test: reverting tokenizer.get_tokenizer to one shared instance fails
this, and the failure is the exception, not a flaky count.
"""
workers = 8
rounds = 4
expected = analyzer(LONG_36_MORA)
with concurrent.futures.ThreadPoolExecutor(max_workers=workers) as pool:
futures = [pool.submit(analyzer, LONG_36_MORA) for _ in range(workers * rounds)]
results = [f.result() for f in futures]
assert len(results) == workers * rounds
for i, units in enumerate(results):
assert units == expected, f"thread {i} disagreed with the single-threaded analysis"
def test_empty_text(analyzer):
"""Nothing in, nothing out; whitespace is a non-tappable unit with no level."""
assert analyzer("") == []
units = analyzer(" ")
assert "".join(u["surface"] for u in units) == " "
for u in units:
assert u["surface"].strip() == ""
assert u["tappable"] is False and u["jlpt"] is None and u["gloss"] == []
assert tuple(u.keys()) == TOKEN_KEYS
def test_known_list_gaps_are_honest_n1_plus(analyzer):
"""Words the pinned lists simply do not carry badge "N1+" (D-10) - pinned by JMdict id.
These look wrong to a Japanese teacher and are right by the data: none of the five
yomitan-jlpt-vocab CSVs lists ない / 無い (1529520) or 日本語 (1464530), and 人気 read ひとけ is
the separate entry 1367020 (unlisted), not にんき's 1367010 (N3). The lookup resolves each to
the correct entry; the label follows the lists. If the product later wants a level alias for
them, that is a new override table and a conscious change to this test, never a weaker lookup.
"""
nai = _by_surface(analyzer("人気のない通りを歩いた。"), "ない")
assert nai["tappable"] is True and nai["lemma"] == "ない" and nai["level_key"] == "無い"
assert nai["jmdict_id"] == 1529520 and nai["jlpt"] == "N1+"
nakatta = _by_surface(analyzer("きれいじゃなかった"), "なかった")
assert nakatta["jmdict_id"] == 1529520 and nakatta["jlpt"] == "N1+"
hitoke = _by_surface(analyzer("人気のない通りを歩いた。"), "人気")
assert hitoke["jmdict_id"] == 1367020 and hitoke["jlpt"] == "N1+"
assert any("human presence" in g for g in hitoke["gloss"][0]), hitoke["gloss"]
ninki = _by_surface(analyzer("この店は人気がある。"), "人気")
assert ninki["jmdict_id"] == 1367010 and ninki["jlpt"] == "N3"
# The other side of the same coin: こんにちは IS listed, at N3 (and N1), so it badges N3.
konnichiwa = analyzer("こんにちは")
assert len(konnichiwa) == 1
assert konnichiwa[0]["jmdict_id"] == 1289400 and konnichiwa[0]["jlpt"] == "N3"
# --- the fixed sample sentence set (success criterion 1) ---------------------------------
def test_fixture_sentences(analyzer):
"""The analyzer reproduces every golden unit of every fixture sentence exactly.
This is JPN-01's quick-loop row (02-VALIDATION.md). A SudachiDict, JMdict or JLPT-list bump,
or a change to the joining rule, must show up here as a reviewed diff - never as a silent
change to what learners read.
"""
golden = _golden()
assert len(golden["sentences"]) == 28
for sentence in golden["sentences"]:
text, expected = sentence["text"], sentence["units"]
got = analyzer(text)
assert got == expected, f"{text!r}: analyze() differs from sentences.json - {REGENERATE}"
def test_fixture_set_covers_the_contract():
"""The set holds every case the phase promised to pin (ambiguous readings, joins, edges)."""
texts = _golden_texts()
assert len(texts) >= 25
joined = "\n".join(texts)
for needle in (
"今日",
"食べました",
"田中さん",
"勉強し",
"読んで",
"待ち合わせ",
"鬱",
"美味しくない",
"きれいじゃなかった",
):
assert needle in joined, needle
assert sum("行った" in t for t in texts) >= 2, "both readings of 行った"
assert sum("人気" in t for t in texts) >= 2, "both readings of 人気"
# The known-ambiguous readings are pinned as READINGS, not just as texts.
readings = {
(s["text"], u["surface"]): u["reading"] for s in _golden()["sentences"] for u in s["units"]
}
assert readings[("今日はいい天気ですね。", "今日")] == "きょう"
assert readings[("昨日、駅に行った。", "行った")] == "いった"
assert readings[("会議を行った。", "行った")] == "おこなった"
assert readings[("人気のない通りを歩いた。", "人気")] == "ひとけ"
assert readings[("この店は人気がある。", "人気")] == "にんき"
def test_fixture_pins_match_installed():
"""The golden file was generated by the data this environment has, or it is stale."""
import sudachipy
from japanese_avatar.nlp import jmdict
pins = _golden()["analyzer"]
assert pins["sudachipy"] == sudachipy.__version__
assert pins["sudachidict_core"] == importlib.metadata.version("sudachidict_core")
assert pins["jmdict"] == jmdict.load().meta["version"]
assert pins["jlpt_vocab"] == "2025.08.01.0"
assert pins["kanji_data"] == "00fd7079c3890f430759536f91aa5e854ec0ca4f"