"""The ONE canonical ``analyze(text)`` (plan 02-05, PITFALLS § 10) and its token record. Every later system - furigana (02-07), the lookup popover (02-08), Phase 3's level guard, Phase 5's vocabulary tracking - consumes the record these tests pin and nothing else. Each CONTEXT decision is a named test: D-10 (unlisted -> "N1+"), D-11 (particles / auxiliaries / punctuation are not words), D-12 (proper nouns -> "name"), D-13 (word axis and kanji axis are separate), plus the committed reading override (人気 -> ひとけ before の) and the known-ambiguous readings PITFALLS names (今日, 行った x2, 人気 x2). Facts recorded by the wave-2 summaries and pinned knowingly here (02-03-SUMMARY.md § Known Edges): なる resolves to the verb 1375610, which is on the N3 list (n5.csv's なる row is the archaic copula 2138260 - an upstream yomitan-jlpt-vocab quirk); いい resolves to its own JMdict entry 2820690 (N5), not 良い 1605820. All tests use the session-scoped ``analyzer`` fixture (tests/conftest.py): one dictionary open and one compact-JMdict load per session. """ from __future__ import annotations import concurrent.futures import functools import importlib.metadata import json import time from pathlib import Path from japanese_avatar.nlp.analyzer import TOKEN_KEYS FIXTURES = Path(__file__).parent / "fixtures" GOLDEN = FIXTURES / "sentences.json" LONG_36_MORA = "今日はいい天気ですから、公園を散歩してから、買い物に行きました。" REGENERATE = ( "regenerate with .venv/Scripts/python.exe tests/fixtures/make_sentences.py " "and REVIEW the diff - a difference means SudachiDict, JMdict, the JLPT lists " "or the joining rule changed what learners read" ) @functools.cache def _golden() -> dict: """The fixed sample sentence set (SC-1): ``{"analyzer": pins, "sentences": [...]}``.""" return json.loads(GOLDEN.read_text(encoding="utf-8")) def _golden_texts() -> list[str]: return [s["text"] for s in _golden()["sentences"]] def _by_surface(units: list[dict], surface: str) -> dict: matches = [u for u in units if u["surface"] == surface] assert len(matches) == 1, f"expected one unit {surface!r}, got {[u['surface'] for u in units]}" return matches[0] def test_record_keys_exact(analyzer): """Every unit carries exactly TOKEN_KEYS, in order - the contract Phases 3 and 5 read.""" units = analyzer("日本語を勉強しています。") assert units, "a sentence yields units" for u in units: assert tuple(u.keys()) == TOKEN_KEYS, u assert TOKEN_KEYS == ( "surface", "reading", "lemma", "level_key", "pos", "tappable", "is_name", "jlpt", "kanji_levels", "ruby", "jmdict_id", "gloss", "start", "end", ) def test_reading_disambiguation(analyzer): """The known-ambiguous readings PITFALLS § 10 names: 今日 and the two 行った.""" kyou = _by_surface(analyzer("今日はいい天気ですね。"), "今日") assert kyou["reading"] == "きょう" itta = _by_surface(analyzer("昨日、駅に行った。"), "行った") assert itta["reading"] == "いった" assert itta["lemma"] == "行く" okonatta = _by_surface(analyzer("会議を行った。"), "行った") assert okonatta["reading"] == "おこなった" assert okonatta["lemma"] == "行う" def test_hitoke_override(analyzer): """人気 is ひとけ before の (the committed override) and にんき N3 elsewhere.""" hitoke = _by_surface(analyzer("人気のない通りを歩いた。"), "人気") assert hitoke["reading"] == "ひとけ" assert hitoke["ruby"] == [["人気", "ひとけ"]] ninki = _by_surface(analyzer("この店は人気がある。"), "人気") assert ninki["reading"] == "にんき" assert ninki["jlpt"] == "N3" def test_tappability(analyzer): """D-11: は / です / ね / 。 are plain text with no level and no gloss; words are tappable.""" units = analyzer("今日はいい天気ですね。") assert [u["surface"] for u in units] == ["今日", "は", "いい", "天気", "です", "ね", "。"] for surface in ("は", "です", "ね", "。"): u = _by_surface(units, surface) assert u["tappable"] is False, surface assert u["jlpt"] is None, surface assert u["gloss"] == [], surface assert u["jmdict_id"] is None, surface assert u["kanji_levels"] == {}, surface assert u["ruby"] == [[surface, None]], surface for surface in ("今日", "いい", "天気"): u = _by_surface(units, surface) assert u["tappable"] is True, surface assert u["jlpt"] == "N5", surface def test_level_derivation(analyzer): """D-12 name, level from the level_key entry, D-10 N1+ for a word on no list.""" units = analyzer("田中さんは東京に住んでいます。") tanaka = _by_surface(units, "田中さん") assert tanaka["is_name"] is True and tanaka["jlpt"] == "name" assert tanaka["tappable"] is True, "D-12: names are tappable (the reading is the useful part)" tokyo = _by_surface(units, "東京") assert tokyo["is_name"] is True and tokyo["jlpt"] == "name" sunde = _by_surface(units, "住んでいます") assert sunde["lemma"] == "住む" and sunde["jlpt"] == "N5" # 話せる has its own, unlisted JMdict entry; its level comes from level_key 話す (N5). hanaseru = _by_surface(analyzer("話せるようになりたい"), "話せる") assert hanaseru["level_key"] == "話す" assert hanaseru["jlpt"] == "N5" assert hanaseru["jmdict_id"] == 1562360 assert any("able to speak" in g for g in hanaseru["gloss"][0]), hanaseru["gloss"] # 鬱 is not one of the 2,211 listed kanji: kanji axis None (D-13 / D-02). uttoushii = _by_surface(analyzer("鬱陶しい天気だ。"), "鬱陶しい") assert uttoushii["kanji_levels"]["鬱"] is None assert uttoushii["tappable"] is True # D-10: a word on no list is "N1+" - never a guess. The fixture pins 鬱陶しい's observed # label (N1 or N1+, whichever the pinned lists say); the D-10 assertion itself uses a rare # word that has a JMdict entry and sits on no list, so "N1+" comes from the rule, not from # a lookup miss. assert uttoushii["jlpt"] in {"N1", "N1+"} beyond = _by_surface(analyzer("彼は瀟洒な服を着ている。"), "瀟洒") assert beyond["tappable"] is True assert beyond["jmdict_id"] is not None, "瀟洒 has a JMdict entry" assert beyond["jlpt"] == "N1+", "on no JLPT list -> N1+ (D-10)" def test_kanji_axis_separate_from_word_axis(analyzer): """D-13: the word's level and each kanji's level are two axes in the same record.""" nihongo = _by_surface(analyzer("日本語を勉強しています。"), "日本語") # 日本語 resolves to JMdict 1464530, which is on NONE of the five pinned lists (Waller's # lists carry 日本 at N3 and no 日本語 at all) - so the word axis is the honest D-10 "N1+" # while every kanji in it is N5. The two axes really are independent; the plan's assumed # "N5" was a guess the pinned data contradicts (02-05-SUMMARY.md). assert nihongo["jmdict_id"] == 1464530 assert nihongo["jlpt"] == "N1+" assert nihongo["kanji_levels"] == {"日": "N5", "本": "N5", "語": "N5"} benkyou = _by_surface(analyzer("日本語を勉強しています。"), "勉強しています") assert benkyou["lemma"] == "勉強する" assert set(benkyou["kanji_levels"]) == {"勉", "強"} assert all(v in {"N5", "N4", "N3", "N2", "N1"} for v in benkyou["kanji_levels"].values()) # The word is N5 while its kanji are harder - the axes really are independent. assert benkyou["jlpt"] == "N5" assert benkyou["kanji_levels"]["勉"] != "N5" def test_gloss_shipped_with_token(analyzer): """D-07 / research § Q3: glosses ride inside the token so a tap needs no round trip.""" units = analyzer("食べました") assert len(units) == 1 u = units[0] assert u["lemma"] == "食べる" assert isinstance(u["jmdict_id"], int) assert u["gloss"], "the token carries its glosses" assert any("eat" in g for g in u["gloss"][0]), u["gloss"] assert len(u["gloss"]) <= 3 and all(len(sense) <= 3 for sense in u["gloss"]) assert u["ruby"] == [["食", "た"], ["べました", None]] def test_offsets_cover_text(analyzer): """Units tile every fixture sentence and each offset pair reads back its surface.""" for text in _golden_texts(): units = analyzer(text) assert "".join(u["surface"] for u in units) == text pos = 0 for u in units: assert u["start"] == pos, (text, u) assert u["end"] == pos + len(u["surface"]), (text, u) assert text[u["start"] : u["end"]] == u["surface"], (text, u) pos = u["end"] assert pos == len(text) def test_analyze_is_fast(analyzer): """Warm analysis is milliseconds: 100 runs of the 36-mora sentence, mean well under 50 ms.""" analyzer(LONG_36_MORA) # warm any lazily built table runs = 100 started = time.perf_counter() for _ in range(runs): analyzer(LONG_36_MORA) mean_ms = (time.perf_counter() - started) * 1000 / runs print(f"\nanalyze({len(LONG_36_MORA)} chars): mean {mean_ms:.3f} ms over {runs} runs") assert mean_ms < 50, f"{mean_ms:.1f} ms per analysis (budget 50 ms; expected < 5 ms)" def test_analyze_is_thread_safe(analyzer): """Plan 02-07: concurrent analyses all succeed and all agree. The Space serves them. ``sudachipy.Tokenizer`` is a PyO3 object that takes a mutable borrow for the duration of a ``tokenize`` call, so ONE shared instance answers the first caller and raises ``RuntimeError: Already borrowed`` at every thread that arrives while it is busy - a lost analysis, not a wrong one. Before ``nlp.tokenizer`` gave each thread its own tokenizer, 8 threads on this sentence lost 5 of 8 calls; the deployed symptom was a directive carrying ``tokens: []`` and an avatar line rendered with no furigana at all. It became reachable in 02-07 because the host page now analyses the learner's line while the turn's own analyse stage runs, so a single typed turn issues two overlapping requests - and a busy Space multiplies that by its visitors. A real regression test: reverting tokenizer.get_tokenizer to one shared instance fails this, and the failure is the exception, not a flaky count. """ workers = 8 rounds = 4 expected = analyzer(LONG_36_MORA) with concurrent.futures.ThreadPoolExecutor(max_workers=workers) as pool: futures = [pool.submit(analyzer, LONG_36_MORA) for _ in range(workers * rounds)] results = [f.result() for f in futures] assert len(results) == workers * rounds for i, units in enumerate(results): assert units == expected, f"thread {i} disagreed with the single-threaded analysis" def test_empty_text(analyzer): """Nothing in, nothing out; whitespace is a non-tappable unit with no level.""" assert analyzer("") == [] units = analyzer(" ") assert "".join(u["surface"] for u in units) == " " for u in units: assert u["surface"].strip() == "" assert u["tappable"] is False and u["jlpt"] is None and u["gloss"] == [] assert tuple(u.keys()) == TOKEN_KEYS def test_known_list_gaps_are_honest_n1_plus(analyzer): """Words the pinned lists simply do not carry badge "N1+" (D-10) - pinned by JMdict id. These look wrong to a Japanese teacher and are right by the data: none of the five yomitan-jlpt-vocab CSVs lists ない / 無い (1529520) or 日本語 (1464530), and 人気 read ひとけ is the separate entry 1367020 (unlisted), not にんき's 1367010 (N3). The lookup resolves each to the correct entry; the label follows the lists. If the product later wants a level alias for them, that is a new override table and a conscious change to this test, never a weaker lookup. """ nai = _by_surface(analyzer("人気のない通りを歩いた。"), "ない") assert nai["tappable"] is True and nai["lemma"] == "ない" and nai["level_key"] == "無い" assert nai["jmdict_id"] == 1529520 and nai["jlpt"] == "N1+" nakatta = _by_surface(analyzer("きれいじゃなかった"), "なかった") assert nakatta["jmdict_id"] == 1529520 and nakatta["jlpt"] == "N1+" hitoke = _by_surface(analyzer("人気のない通りを歩いた。"), "人気") assert hitoke["jmdict_id"] == 1367020 and hitoke["jlpt"] == "N1+" assert any("human presence" in g for g in hitoke["gloss"][0]), hitoke["gloss"] ninki = _by_surface(analyzer("この店は人気がある。"), "人気") assert ninki["jmdict_id"] == 1367010 and ninki["jlpt"] == "N3" # The other side of the same coin: こんにちは IS listed, at N3 (and N1), so it badges N3. konnichiwa = analyzer("こんにちは") assert len(konnichiwa) == 1 assert konnichiwa[0]["jmdict_id"] == 1289400 and konnichiwa[0]["jlpt"] == "N3" # --- the fixed sample sentence set (success criterion 1) --------------------------------- def test_fixture_sentences(analyzer): """The analyzer reproduces every golden unit of every fixture sentence exactly. This is JPN-01's quick-loop row (02-VALIDATION.md). A SudachiDict, JMdict or JLPT-list bump, or a change to the joining rule, must show up here as a reviewed diff - never as a silent change to what learners read. """ golden = _golden() assert len(golden["sentences"]) == 28 for sentence in golden["sentences"]: text, expected = sentence["text"], sentence["units"] got = analyzer(text) assert got == expected, f"{text!r}: analyze() differs from sentences.json - {REGENERATE}" def test_fixture_set_covers_the_contract(): """The set holds every case the phase promised to pin (ambiguous readings, joins, edges).""" texts = _golden_texts() assert len(texts) >= 25 joined = "\n".join(texts) for needle in ( "今日", "食べました", "田中さん", "勉強し", "読んで", "待ち合わせ", "鬱", "美味しくない", "きれいじゃなかった", ): assert needle in joined, needle assert sum("行った" in t for t in texts) >= 2, "both readings of 行った" assert sum("人気" in t for t in texts) >= 2, "both readings of 人気" # The known-ambiguous readings are pinned as READINGS, not just as texts. readings = { (s["text"], u["surface"]): u["reading"] for s in _golden()["sentences"] for u in s["units"] } assert readings[("今日はいい天気ですね。", "今日")] == "きょう" assert readings[("昨日、駅に行った。", "行った")] == "いった" assert readings[("会議を行った。", "行った")] == "おこなった" assert readings[("人気のない通りを歩いた。", "人気")] == "ひとけ" assert readings[("この店は人気がある。", "人気")] == "にんき" def test_fixture_pins_match_installed(): """The golden file was generated by the data this environment has, or it is stale.""" import sudachipy from japanese_avatar.nlp import jmdict pins = _golden()["analyzer"] assert pins["sudachipy"] == sudachipy.__version__ assert pins["sudachidict_core"] == importlib.metadata.version("sudachidict_core") assert pins["jmdict"] == jmdict.load().meta["version"] assert pins["jlpt_vocab"] == "2025.08.01.0" assert pins["kanji_data"] == "00fd7079c3890f430759536f91aa5e854ec0ca4f"