Spaces:
Running on Zero
Running on Zero
| """The transcript renderer, proven in a real browser with no Python app and no host framework. | |
| avatar/transcript-harness.html is opened off the static server; fixture token records from | |
| tests/fixtures/sentences.json go through avatar/transcript.js into real <ruby><rt> DOM. If | |
| these pass and the deployed furigana rows fail, the fault is provably the host page (the | |
| stylesheet did not load, the module was not handed over), not the renderer - the same | |
| division of blame tests/e2e/test_stage_standalone.py gives the avatar. | |
| Every claim is a number (research Pitfall 4, "green count, invisible ruby"): the rt COUNT, | |
| the rendered rt HEIGHT and the line HEIGHT with ruby versus bare, all read from | |
| transcript.js's published debug object and asserted here before either browser layer that | |
| needs a server. Marked slow but NOT deployed. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| from pathlib import Path | |
| import pytest | |
| from tests.e2e.test_avatar_loop import expected_rt | |
| pytestmark = pytest.mark.slow | |
| FIXTURE = Path(__file__).resolve().parent.parent / "fixtures" / "sentences.json" | |
| READY_TIMEOUT_MS = 15_000 | |
| # 1-BASED sentence numbers of the 02-05 table - the harness's one convention. #9 has two | |
| # kanji runs (日本語, 勉強) around a bare particle and a bare inflection; #13 and #1 make up | |
| # the three-line re-render check. | |
| STUDY = 9 | |
| STUDY_TEXT = "日本語を勉強しています。" | |
| STUDY_READING_FIRST = "にほんご" | |
| TANAKA = 13 | |
| HELLO = 1 | |
| DEBUG = "() => JSON.parse(JSON.stringify(window.__transcriptDebug))" | |
| def _units(number: int) -> list[dict]: | |
| """The fixture's token records for a 1-based sentence number, for the Python oracle.""" | |
| sentences = json.loads(FIXTURE.read_text(encoding="utf-8"))["sentences"] | |
| return sentences[number - 1]["units"] | |
| def _open_harness(page, static_server, query: str = ""): | |
| page.goto(f"{static_server}/avatar/transcript-harness.html{query}") | |
| page.wait_for_function("() => window.__transcriptReady === true", timeout=READY_TIMEOUT_MS) | |
| assert page.evaluate("() => window.__fixtures[8].text") == STUDY_TEXT, ( | |
| "fixture #9 is not 日本語を勉強しています。; convention or golden file moved" | |
| ) | |
| def _render(page, number: int, who: str = "avatar") -> str: | |
| return page.evaluate("([n, who]) => window.__renderFixture(n, who)", [number, who]) | |
| def _debug(page) -> dict: | |
| return page.evaluate(DEBUG) | |
| def _set(page, select_id: str, value: str) -> dict: | |
| page.select_option(f"#{select_id}", value) | |
| return _debug(page) | |
| def test_ruby_renders_with_numbers(page, static_server): | |
| """D-01: the first line renders with every kanji annotated and nothing touched. | |
| Two kanji runs -> two <rt>, each with a rendered height, on a line that is one | |
| `.turn-avatar`; the readings are real DOM text (in textContent - which is why <rp> is | |
| never emitted), the tappable units carry the accessibility floor and the particles are | |
| plain spans. | |
| """ | |
| _open_harness(page, static_server) | |
| line_id = _render(page, STUDY, "avatar") | |
| debug = _debug(page) | |
| print(f"[standalone] {STUDY_TEXT} as avatar -> {line_id}: {debug}") | |
| assert line_id == "L1" | |
| assert page.locator("#transcript-text .turn-avatar").count() == 1 | |
| assert page.locator("#transcript-text .turn").count() == 1 | |
| assert debug["lastLineId"] == "L1" | |
| assert debug["lastLineKanjiRuns"] == 2 | |
| assert debug["lastLineRt"] == 2, debug | |
| assert debug["rtTotal"] == 2 | |
| assert debug["lastLineRtHeightPx"] > 0, "rt exists but rendered with no height - hidden ruby" | |
| assert debug["lastLineHeightPx"] > 0 | |
| said = page.locator("#transcript-text .turn .said") | |
| assert STUDY_READING_FIRST in said.text_content() | |
| assert page.locator("#transcript-text .turn .who").text_content() == "Avatar: " | |
| ruby = page.locator("#transcript-text ruby").first | |
| assert "日本語" in ruby.inner_text() | |
| assert page.locator("#transcript-text rt").first.text_content() == STUDY_READING_FIRST | |
| toks = page.locator("#transcript-text .tok") | |
| assert toks.count() == 2, "日本語 and 勉強しています are the two tappable units" | |
| for i in range(toks.count()): | |
| assert toks.nth(i).get_attribute("role") == "button" | |
| assert toks.nth(i).get_attribute("tabindex") == "0" | |
| assert toks.nth(i).get_attribute("data-token") is not None | |
| assert page.locator("#transcript-text .plain").count() >= 2, "を and 。 are plain (D-11)" | |
| def test_furigana_modes_gate_on_kanji_axis(page, static_server): | |
| """D-02 / D-13 / D-14: never -> 0; above@N5 and above@N2 equal the independent oracle | |
| (computed from the pinned kanji list, never hard-coded); always -> 2 again; and the | |
| always line is measurably taller than the never line - the rendered-height proof.""" | |
| _open_harness(page, static_server) | |
| _render(page, STUDY, "avatar") | |
| units = _units(STUDY) | |
| always = _debug(page) | |
| assert always["mode"] == "always" and always["level"] == "N5" | |
| assert always["lastLineRt"] == expected_rt(units, "always", "N5") == 2 | |
| never = _set(page, "furigana-mode", "never") | |
| assert never["mode"] == "never" | |
| assert never["lastLineRt"] == 0 and never["rtTotal"] == 0 | |
| assert page.locator("#transcript-text rt").count() == 0 | |
| assert never["lastLineRtHeightPx"] == 0 | |
| above_n5 = _set(page, "furigana-mode", "above") | |
| want_n5 = expected_rt(units, "above", "N5") | |
| assert above_n5["mode"] == "above" and above_n5["level"] == "N5" | |
| assert above_n5["lastLineRt"] == want_n5, (above_n5, want_n5) | |
| above_n2 = _set(page, "level-select", "N2") | |
| want_n2 = expected_rt(units, "above", "N2") | |
| assert above_n2["level"] == "N2" | |
| assert above_n2["lastLineRt"] == want_n2, (above_n2, want_n2) | |
| assert want_n2 <= want_n5 | |
| back = _set(page, "furigana-mode", "always") | |
| assert back["lastLineRt"] == 2 and back["rtTotal"] == 2 | |
| print( | |
| f"[standalone] modes: always rt={always['lastLineRt']} h={always['lastLineHeightPx']} " | |
| f"rtH={always['lastLineRtHeightPx']}; never rt=0 h={never['lastLineHeightPx']}; " | |
| f"above@N5 rt={above_n5['lastLineRt']} (oracle {want_n5}); " | |
| f"above@N2 rt={above_n2['lastLineRt']} (oracle {want_n2})" | |
| ) | |
| assert always["lastLineHeightPx"] > never["lastLineHeightPx"], ( | |
| "the ruby line is not taller than the bare line; the annotation is not being laid out " | |
| f"above the text (always {always['lastLineHeightPx']} px vs never " | |
| f"{never['lastLineHeightPx']} px)" | |
| ) | |
| assert back["lastLineHeightPx"] == always["lastLineHeightPx"] | |
| def test_learner_line_gets_ruby_too(page, static_server): | |
| """D-03: the learner's own line carries the same ruby as the avatar's.""" | |
| _open_harness(page, static_server) | |
| line_id = _render(page, STUDY, "you") | |
| debug = _debug(page) | |
| line = page.locator(f"#transcript-text [data-line='{line_id}']") | |
| assert line.count() == 1 | |
| assert "turn-you" in line.get_attribute("class") | |
| assert line.get_attribute("data-who") == "you" | |
| assert line.locator(".who").text_content() == "You: " | |
| assert debug["lastLineRt"] == 2 | |
| assert line.locator("rt").count() == 2 | |
| assert debug["lastLineRtHeightPx"] > 0 | |
| def test_rerender_keeps_line_count(page, static_server): | |
| """A mode flip re-renders every line from retained tokens: no duplicates, no losses.""" | |
| _open_harness(page, static_server) | |
| ids = [_render(page, HELLO, "avatar"), _render(page, STUDY, "you"), _render(page, TANAKA)] | |
| assert ids == ["L1", "L2", "L3"] | |
| before = _debug(page) | |
| assert before["lines"] == 3 | |
| total_always = before["rtTotal"] | |
| assert total_always == sum( | |
| expected_rt(_units(n), "always", "N5") for n in (HELLO, STUDY, TANAKA) | |
| ) | |
| _set(page, "furigana-mode", "never") | |
| _set(page, "level-select", "N3") | |
| after = _set(page, "furigana-mode", "above") | |
| assert after["lines"] == 3 | |
| assert page.locator("#transcript-text .turn").count() == 3 | |
| assert [ | |
| el.get_attribute("data-line") for el in page.locator("#transcript-text .turn").all() | |
| ] == ids | |
| assert after["rtTotal"] == sum( | |
| expected_rt(_units(n), "above", "N3") for n in (HELLO, STUDY, TANAKA) | |
| ) | |
| assert after["lastLineId"] == "L3" | |
| restored = _set(page, "furigana-mode", "always") | |
| assert restored["rtTotal"] == total_always | |
| assert page.locator("#transcript-text .turn").count() == 3 | |