File size: 8,392 Bytes
7d7a567
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
"""The transcript renderer, proven in a real browser with no Python app and no host framework.

avatar/transcript-harness.html is opened off the static server; fixture token records from
tests/fixtures/sentences.json go through avatar/transcript.js into real <ruby><rt> DOM. If
these pass and the deployed furigana rows fail, the fault is provably the host page (the
stylesheet did not load, the module was not handed over), not the renderer - the same
division of blame tests/e2e/test_stage_standalone.py gives the avatar.

Every claim is a number (research Pitfall 4, "green count, invisible ruby"): the rt COUNT,
the rendered rt HEIGHT and the line HEIGHT with ruby versus bare, all read from
transcript.js's published debug object and asserted here before either browser layer that
needs a server. Marked slow but NOT deployed.
"""

from __future__ import annotations

import json
from pathlib import Path

import pytest

from tests.e2e.test_avatar_loop import expected_rt

pytestmark = pytest.mark.slow

FIXTURE = Path(__file__).resolve().parent.parent / "fixtures" / "sentences.json"
READY_TIMEOUT_MS = 15_000

# 1-BASED sentence numbers of the 02-05 table - the harness's one convention. #9 has two
# kanji runs (ζ—₯本θͺž, 勉強) around a bare particle and a bare inflection; #13 and #1 make up
# the three-line re-render check.
STUDY = 9
STUDY_TEXT = "ζ—₯本θͺžγ‚’勉強しています。"
STUDY_READING_FIRST = "にほんご"
TANAKA = 13
HELLO = 1

DEBUG = "() => JSON.parse(JSON.stringify(window.__transcriptDebug))"


def _units(number: int) -> list[dict]:
    """The fixture's token records for a 1-based sentence number, for the Python oracle."""
    sentences = json.loads(FIXTURE.read_text(encoding="utf-8"))["sentences"]
    return sentences[number - 1]["units"]


def _open_harness(page, static_server, query: str = ""):
    page.goto(f"{static_server}/avatar/transcript-harness.html{query}")
    page.wait_for_function("() => window.__transcriptReady === true", timeout=READY_TIMEOUT_MS)
    assert page.evaluate("() => window.__fixtures[8].text") == STUDY_TEXT, (
        "fixture #9 is not ζ—₯本θͺžγ‚’勉強しています。; convention or golden file moved"
    )


def _render(page, number: int, who: str = "avatar") -> str:
    return page.evaluate("([n, who]) => window.__renderFixture(n, who)", [number, who])


def _debug(page) -> dict:
    return page.evaluate(DEBUG)


def _set(page, select_id: str, value: str) -> dict:
    page.select_option(f"#{select_id}", value)
    return _debug(page)


def test_ruby_renders_with_numbers(page, static_server):
    """D-01: the first line renders with every kanji annotated and nothing touched.

    Two kanji runs -> two <rt>, each with a rendered height, on a line that is one
    `.turn-avatar`; the readings are real DOM text (in textContent - which is why <rp> is
    never emitted), the tappable units carry the accessibility floor and the particles are
    plain spans.
    """
    _open_harness(page, static_server)
    line_id = _render(page, STUDY, "avatar")
    debug = _debug(page)
    print(f"[standalone] {STUDY_TEXT} as avatar -> {line_id}: {debug}")

    assert line_id == "L1"
    assert page.locator("#transcript-text .turn-avatar").count() == 1
    assert page.locator("#transcript-text .turn").count() == 1
    assert debug["lastLineId"] == "L1"
    assert debug["lastLineKanjiRuns"] == 2
    assert debug["lastLineRt"] == 2, debug
    assert debug["rtTotal"] == 2
    assert debug["lastLineRtHeightPx"] > 0, "rt exists but rendered with no height - hidden ruby"
    assert debug["lastLineHeightPx"] > 0

    said = page.locator("#transcript-text .turn .said")
    assert STUDY_READING_FIRST in said.text_content()
    assert page.locator("#transcript-text .turn .who").text_content() == "Avatar: "
    ruby = page.locator("#transcript-text ruby").first
    assert "ζ—₯本θͺž" in ruby.inner_text()
    assert page.locator("#transcript-text rt").first.text_content() == STUDY_READING_FIRST

    toks = page.locator("#transcript-text .tok")
    assert toks.count() == 2, "ζ—₯本θͺž and 勉強しています are the two tappable units"
    for i in range(toks.count()):
        assert toks.nth(i).get_attribute("role") == "button"
        assert toks.nth(i).get_attribute("tabindex") == "0"
        assert toks.nth(i).get_attribute("data-token") is not None
    assert page.locator("#transcript-text .plain").count() >= 2, "γ‚’ and 。 are plain (D-11)"


def test_furigana_modes_gate_on_kanji_axis(page, static_server):
    """D-02 / D-13 / D-14: never -> 0; above@N5 and above@N2 equal the independent oracle
    (computed from the pinned kanji list, never hard-coded); always -> 2 again; and the
    always line is measurably taller than the never line - the rendered-height proof."""
    _open_harness(page, static_server)
    _render(page, STUDY, "avatar")
    units = _units(STUDY)
    always = _debug(page)
    assert always["mode"] == "always" and always["level"] == "N5"
    assert always["lastLineRt"] == expected_rt(units, "always", "N5") == 2

    never = _set(page, "furigana-mode", "never")
    assert never["mode"] == "never"
    assert never["lastLineRt"] == 0 and never["rtTotal"] == 0
    assert page.locator("#transcript-text rt").count() == 0
    assert never["lastLineRtHeightPx"] == 0

    above_n5 = _set(page, "furigana-mode", "above")
    want_n5 = expected_rt(units, "above", "N5")
    assert above_n5["mode"] == "above" and above_n5["level"] == "N5"
    assert above_n5["lastLineRt"] == want_n5, (above_n5, want_n5)

    above_n2 = _set(page, "level-select", "N2")
    want_n2 = expected_rt(units, "above", "N2")
    assert above_n2["level"] == "N2"
    assert above_n2["lastLineRt"] == want_n2, (above_n2, want_n2)
    assert want_n2 <= want_n5

    back = _set(page, "furigana-mode", "always")
    assert back["lastLineRt"] == 2 and back["rtTotal"] == 2
    print(
        f"[standalone] modes: always rt={always['lastLineRt']} h={always['lastLineHeightPx']} "
        f"rtH={always['lastLineRtHeightPx']}; never rt=0 h={never['lastLineHeightPx']}; "
        f"above@N5 rt={above_n5['lastLineRt']} (oracle {want_n5}); "
        f"above@N2 rt={above_n2['lastLineRt']} (oracle {want_n2})"
    )
    assert always["lastLineHeightPx"] > never["lastLineHeightPx"], (
        "the ruby line is not taller than the bare line; the annotation is not being laid out "
        f"above the text (always {always['lastLineHeightPx']} px vs never "
        f"{never['lastLineHeightPx']} px)"
    )
    assert back["lastLineHeightPx"] == always["lastLineHeightPx"]


def test_learner_line_gets_ruby_too(page, static_server):
    """D-03: the learner's own line carries the same ruby as the avatar's."""
    _open_harness(page, static_server)
    line_id = _render(page, STUDY, "you")
    debug = _debug(page)
    line = page.locator(f"#transcript-text [data-line='{line_id}']")
    assert line.count() == 1
    assert "turn-you" in line.get_attribute("class")
    assert line.get_attribute("data-who") == "you"
    assert line.locator(".who").text_content() == "You: "
    assert debug["lastLineRt"] == 2
    assert line.locator("rt").count() == 2
    assert debug["lastLineRtHeightPx"] > 0


def test_rerender_keeps_line_count(page, static_server):
    """A mode flip re-renders every line from retained tokens: no duplicates, no losses."""
    _open_harness(page, static_server)
    ids = [_render(page, HELLO, "avatar"), _render(page, STUDY, "you"), _render(page, TANAKA)]
    assert ids == ["L1", "L2", "L3"]
    before = _debug(page)
    assert before["lines"] == 3
    total_always = before["rtTotal"]
    assert total_always == sum(
        expected_rt(_units(n), "always", "N5") for n in (HELLO, STUDY, TANAKA)
    )

    _set(page, "furigana-mode", "never")
    _set(page, "level-select", "N3")
    after = _set(page, "furigana-mode", "above")
    assert after["lines"] == 3
    assert page.locator("#transcript-text .turn").count() == 3
    assert [
        el.get_attribute("data-line") for el in page.locator("#transcript-text .turn").all()
    ] == ids
    assert after["rtTotal"] == sum(
        expected_rt(_units(n), "above", "N3") for n in (HELLO, STUDY, TANAKA)
    )
    assert after["lastLineId"] == "L3"

    restored = _set(page, "furigana-mode", "always")
    assert restored["rtTotal"] == total_always
    assert page.locator("#transcript-text .turn").count() == 3