Spaces:
Running on Zero
Running on Zero
File size: 8,392 Bytes
7d7a567 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 193 194 | """The transcript renderer, proven in a real browser with no Python app and no host framework.
avatar/transcript-harness.html is opened off the static server; fixture token records from
tests/fixtures/sentences.json go through avatar/transcript.js into real <ruby><rt> DOM. If
these pass and the deployed furigana rows fail, the fault is provably the host page (the
stylesheet did not load, the module was not handed over), not the renderer - the same
division of blame tests/e2e/test_stage_standalone.py gives the avatar.
Every claim is a number (research Pitfall 4, "green count, invisible ruby"): the rt COUNT,
the rendered rt HEIGHT and the line HEIGHT with ruby versus bare, all read from
transcript.js's published debug object and asserted here before either browser layer that
needs a server. Marked slow but NOT deployed.
"""
from __future__ import annotations
import json
from pathlib import Path
import pytest
from tests.e2e.test_avatar_loop import expected_rt
pytestmark = pytest.mark.slow
FIXTURE = Path(__file__).resolve().parent.parent / "fixtures" / "sentences.json"
READY_TIMEOUT_MS = 15_000
# 1-BASED sentence numbers of the 02-05 table - the harness's one convention. #9 has two
# kanji runs (ζ₯ζ¬θͺ, εεΌ·) around a bare particle and a bare inflection; #13 and #1 make up
# the three-line re-render check.
STUDY = 9
STUDY_TEXT = "ζ₯ζ¬θͺγεεΌ·γγ¦γγΎγγ"
STUDY_READING_FIRST = "γ«γ»γγ"
TANAKA = 13
HELLO = 1
DEBUG = "() => JSON.parse(JSON.stringify(window.__transcriptDebug))"
def _units(number: int) -> list[dict]:
"""The fixture's token records for a 1-based sentence number, for the Python oracle."""
sentences = json.loads(FIXTURE.read_text(encoding="utf-8"))["sentences"]
return sentences[number - 1]["units"]
def _open_harness(page, static_server, query: str = ""):
page.goto(f"{static_server}/avatar/transcript-harness.html{query}")
page.wait_for_function("() => window.__transcriptReady === true", timeout=READY_TIMEOUT_MS)
assert page.evaluate("() => window.__fixtures[8].text") == STUDY_TEXT, (
"fixture #9 is not ζ₯ζ¬θͺγεεΌ·γγ¦γγΎγγ; convention or golden file moved"
)
def _render(page, number: int, who: str = "avatar") -> str:
return page.evaluate("([n, who]) => window.__renderFixture(n, who)", [number, who])
def _debug(page) -> dict:
return page.evaluate(DEBUG)
def _set(page, select_id: str, value: str) -> dict:
page.select_option(f"#{select_id}", value)
return _debug(page)
def test_ruby_renders_with_numbers(page, static_server):
"""D-01: the first line renders with every kanji annotated and nothing touched.
Two kanji runs -> two <rt>, each with a rendered height, on a line that is one
`.turn-avatar`; the readings are real DOM text (in textContent - which is why <rp> is
never emitted), the tappable units carry the accessibility floor and the particles are
plain spans.
"""
_open_harness(page, static_server)
line_id = _render(page, STUDY, "avatar")
debug = _debug(page)
print(f"[standalone] {STUDY_TEXT} as avatar -> {line_id}: {debug}")
assert line_id == "L1"
assert page.locator("#transcript-text .turn-avatar").count() == 1
assert page.locator("#transcript-text .turn").count() == 1
assert debug["lastLineId"] == "L1"
assert debug["lastLineKanjiRuns"] == 2
assert debug["lastLineRt"] == 2, debug
assert debug["rtTotal"] == 2
assert debug["lastLineRtHeightPx"] > 0, "rt exists but rendered with no height - hidden ruby"
assert debug["lastLineHeightPx"] > 0
said = page.locator("#transcript-text .turn .said")
assert STUDY_READING_FIRST in said.text_content()
assert page.locator("#transcript-text .turn .who").text_content() == "Avatar: "
ruby = page.locator("#transcript-text ruby").first
assert "ζ₯ζ¬θͺ" in ruby.inner_text()
assert page.locator("#transcript-text rt").first.text_content() == STUDY_READING_FIRST
toks = page.locator("#transcript-text .tok")
assert toks.count() == 2, "ζ₯ζ¬θͺ and εεΌ·γγ¦γγΎγ are the two tappable units"
for i in range(toks.count()):
assert toks.nth(i).get_attribute("role") == "button"
assert toks.nth(i).get_attribute("tabindex") == "0"
assert toks.nth(i).get_attribute("data-token") is not None
assert page.locator("#transcript-text .plain").count() >= 2, "γ and γ are plain (D-11)"
def test_furigana_modes_gate_on_kanji_axis(page, static_server):
"""D-02 / D-13 / D-14: never -> 0; above@N5 and above@N2 equal the independent oracle
(computed from the pinned kanji list, never hard-coded); always -> 2 again; and the
always line is measurably taller than the never line - the rendered-height proof."""
_open_harness(page, static_server)
_render(page, STUDY, "avatar")
units = _units(STUDY)
always = _debug(page)
assert always["mode"] == "always" and always["level"] == "N5"
assert always["lastLineRt"] == expected_rt(units, "always", "N5") == 2
never = _set(page, "furigana-mode", "never")
assert never["mode"] == "never"
assert never["lastLineRt"] == 0 and never["rtTotal"] == 0
assert page.locator("#transcript-text rt").count() == 0
assert never["lastLineRtHeightPx"] == 0
above_n5 = _set(page, "furigana-mode", "above")
want_n5 = expected_rt(units, "above", "N5")
assert above_n5["mode"] == "above" and above_n5["level"] == "N5"
assert above_n5["lastLineRt"] == want_n5, (above_n5, want_n5)
above_n2 = _set(page, "level-select", "N2")
want_n2 = expected_rt(units, "above", "N2")
assert above_n2["level"] == "N2"
assert above_n2["lastLineRt"] == want_n2, (above_n2, want_n2)
assert want_n2 <= want_n5
back = _set(page, "furigana-mode", "always")
assert back["lastLineRt"] == 2 and back["rtTotal"] == 2
print(
f"[standalone] modes: always rt={always['lastLineRt']} h={always['lastLineHeightPx']} "
f"rtH={always['lastLineRtHeightPx']}; never rt=0 h={never['lastLineHeightPx']}; "
f"above@N5 rt={above_n5['lastLineRt']} (oracle {want_n5}); "
f"above@N2 rt={above_n2['lastLineRt']} (oracle {want_n2})"
)
assert always["lastLineHeightPx"] > never["lastLineHeightPx"], (
"the ruby line is not taller than the bare line; the annotation is not being laid out "
f"above the text (always {always['lastLineHeightPx']} px vs never "
f"{never['lastLineHeightPx']} px)"
)
assert back["lastLineHeightPx"] == always["lastLineHeightPx"]
def test_learner_line_gets_ruby_too(page, static_server):
"""D-03: the learner's own line carries the same ruby as the avatar's."""
_open_harness(page, static_server)
line_id = _render(page, STUDY, "you")
debug = _debug(page)
line = page.locator(f"#transcript-text [data-line='{line_id}']")
assert line.count() == 1
assert "turn-you" in line.get_attribute("class")
assert line.get_attribute("data-who") == "you"
assert line.locator(".who").text_content() == "You: "
assert debug["lastLineRt"] == 2
assert line.locator("rt").count() == 2
assert debug["lastLineRtHeightPx"] > 0
def test_rerender_keeps_line_count(page, static_server):
"""A mode flip re-renders every line from retained tokens: no duplicates, no losses."""
_open_harness(page, static_server)
ids = [_render(page, HELLO, "avatar"), _render(page, STUDY, "you"), _render(page, TANAKA)]
assert ids == ["L1", "L2", "L3"]
before = _debug(page)
assert before["lines"] == 3
total_always = before["rtTotal"]
assert total_always == sum(
expected_rt(_units(n), "always", "N5") for n in (HELLO, STUDY, TANAKA)
)
_set(page, "furigana-mode", "never")
_set(page, "level-select", "N3")
after = _set(page, "furigana-mode", "above")
assert after["lines"] == 3
assert page.locator("#transcript-text .turn").count() == 3
assert [
el.get_attribute("data-line") for el in page.locator("#transcript-text .turn").all()
] == ids
assert after["rtTotal"] == sum(
expected_rt(_units(n), "above", "N3") for n in (HELLO, STUDY, TANAKA)
)
assert after["lastLineId"] == "L3"
restored = _set(page, "furigana-mode", "always")
assert restored["rtTotal"] == total_always
assert page.locator("#transcript-text .turn").count() == 3
|