Spaces:
Running on Zero
feat(01-08): add the server turn - synthesis, timeline, per-request timings, directive
Browse files- telemetry/timings.py: TurnTimings with mark() and a stage() context manager, one flat
record per request, server_total_ms measured from construction; never module-level
- voice/models.py re-exports TurnTimings from telemetry and gives AvatarDirective to_dict()
- ui/blocks.py: turn() builds the viseme timeline from the query that was actually
synthesised at the requested speed, base64-encodes the WAV as a data URL and returns the
directive; validation and synthesis failures return {error} because the gr.HTML bridge
turns a raised exception into undefined in the browser; greeting() for one-click speech;
the full Phase 1 layout with stable elem_ids, the credit footer, the flow-down notice and
the About panel; warm_synthesizer bound to Blocks.load
- app.py slimmed to assembly plus the ZeroGPU probe (39 lines)
The bridge packs a call's arguments into one JSON value (dict from JS, list for several
args, [] for none) and calls the Python function with that single positional, so turn()
accepts a {text, speed} payload and greeting() tolerates a stray positional.
- app.py +12 -134
- src/japanese_avatar/telemetry/timings.py +64 -0
- src/japanese_avatar/ui/blocks.py +269 -0
- src/japanese_avatar/voice/models.py +19 -25
- tests/test_directive.py +0 -1
|
@@ -1,161 +1,39 @@
|
|
| 1 |
-
"""Space entry point: assembly only.
|
| 2 |
|
| 3 |
-
No logic and no module-level mutable state lives here. Gradio shares module globals
|
| 4 |
-
across every visitor session, so the discipline starts now, while there is still
|
| 5 |
-
nothing to share.
|
| 6 |
-
|
| 7 |
-
The right-hand column is deliberately inert. Plans 01-08 and 01-09 give the controls
|
| 8 |
-
their real behaviour; what they need from this plan is that the ``elem_id`` values
|
| 9 |
-
(``status-line``, ``transcript``, ``credits``) already exist so their selectors are
|
| 10 |
-
stable, and that interacting with the controls today already round-trips through
|
| 11 |
-
Python - which is what ``test_no_remount`` measures the avatar against.
|
| 12 |
-
"""
|
| 13 |
-
|
| 14 |
-
import html
|
| 15 |
import os
|
| 16 |
import sys
|
| 17 |
from pathlib import Path
|
| 18 |
|
| 19 |
-
import gradio as gr
|
| 20 |
import spaces
|
| 21 |
|
| 22 |
-
# The Space runs `python app.py`
|
| 23 |
-
#
|
| 24 |
-
# the import below raised ModuleNotFoundError on the first deploy (RUNTIME_ERROR on
|
| 25 |
-
# 6eeb816). Locally the dev venv has the project installed and this is a no-op. Doing it
|
| 26 |
-
# here rather than through a PYTHONPATH Space variable keeps the app runnable from a
|
| 27 |
-
# clean checkout with no platform configuration to lose.
|
| 28 |
_SRC = Path(__file__).resolve().parent / "src"
|
| 29 |
if str(_SRC) not in sys.path:
|
| 30 |
sys.path.insert(0, str(_SRC))
|
| 31 |
|
| 32 |
-
from japanese_avatar.ui.
|
| 33 |
-
|
| 34 |
-
# Serves avatar/ straight off disk, bypassing the Gradio cache, which is how the
|
| 35 |
-
# browser reaches avatar.js, stage.html and tutor.vrm. Deliberately one dedicated
|
| 36 |
-
# directory: this call's own docstring warns that ALL files under a listed path
|
| 37 |
-
# become network-reachable.
|
| 38 |
-
gr.set_static_paths(["avatar"])
|
| 39 |
-
|
| 40 |
-
# Visitor-facing copy. This is a public portfolio Space, so it states plainly what the
|
| 41 |
-
# build does and does not do: an avatar that renders but cannot answer is a bug report
|
| 42 |
-
# waiting to happen if the page implies otherwise. Internal plan numbers never appear here.
|
| 43 |
-
INTRO_HTML = (
|
| 44 |
-
'<div id="intro" class="intro">'
|
| 45 |
-
"<strong>Preview build.</strong> The 3D avatar, its idle animation and the mora-timed "
|
| 46 |
-
"lip-sync engine are live and running in your browser. Speech and conversation are "
|
| 47 |
-
"built and tested but not yet connected, so the avatar will not answer you yet - "
|
| 48 |
-
"the box below only echoes your text back."
|
| 49 |
-
"</div>"
|
| 50 |
-
)
|
| 51 |
-
|
| 52 |
-
CREDITS_HTML = (
|
| 53 |
-
'<div id="credits-text" class="credits">'
|
| 54 |
-
"Avatar: <em>VRM1_Constraint_Twist_Sample</em> by pixiv Inc. (VRM 1.0). "
|
| 55 |
-
"Voice credits appear here once synthesised audio ships."
|
| 56 |
-
"</div>"
|
| 57 |
-
)
|
| 58 |
|
| 59 |
|
| 60 |
def gpu_disabled() -> bool:
|
| 61 |
-
"""SC-4's kill switch
|
| 62 |
-
|
| 63 |
-
Read per call rather than captured at import so a Space restart with the variable
|
| 64 |
-
flipped takes effect without a code change.
|
| 65 |
-
"""
|
| 66 |
return os.environ.get("DISABLE_GPU", "0").strip().lower() in {"1", "true", "yes"}
|
| 67 |
|
| 68 |
|
| 69 |
-
# ZeroGPU
|
| 70 |
-
#
|
| 71 |
-
#
|
| 72 |
-
# then SIGTERMed it and reported
|
| 73 |
-
#
|
| 74 |
-
# runtime.errorMessage: "No @spaces.GPU function detected during startup"
|
| 75 |
-
#
|
| 76 |
-
# That is a hard platform constraint and there is no way around it: docs/HOSTING.md
|
| 77 |
-
# records that cpu-basic Gradio Spaces are 402-gated on this account, so ZeroGPU is the
|
| 78 |
-
# only free hosting path this project has.
|
| 79 |
-
#
|
| 80 |
-
# So this function exists purely to satisfy the scheduler's startup scan. It is NOT on
|
| 81 |
-
# the turn path, nothing in the app calls it, and it refuses to do anything when
|
| 82 |
-
# DISABLE_GPU is set. If SC-4's demonstration ever needs strengthening, the right assertion
|
| 83 |
-
# is "the turn loop completes without this function's counter moving", not "no GPU function
|
| 84 |
-
# exists" - the platform has taken the latter off the table.
|
| 85 |
@spaces.GPU(duration=1)
|
| 86 |
def zerogpu_probe() -> str:
|
| 87 |
"""The GPU entry point ZeroGPU requires to exist. Deliberately unreachable in Phase 1."""
|
| 88 |
if gpu_disabled():
|
| 89 |
-
raise RuntimeError(
|
| 90 |
-
"zerogpu_probe called with DISABLE_GPU=1; Phase 1 has no GPU work and nothing "
|
| 91 |
-
"on the turn path may reach this function"
|
| 92 |
-
)
|
| 93 |
return "zerogpu reachable"
|
| 94 |
|
| 95 |
|
| 96 |
-
|
| 97 |
-
|
| 98 |
-
|
| 99 |
-
It exists so the controls perform a real Python round trip in Phase 1, because a
|
| 100 |
-
control that never reaches the server would not exercise the re-render path that
|
| 101 |
-
``mountCount`` is supposed to survive. Plan 01-08 replaces the body with the real
|
| 102 |
-
turn dispatch; the signature is already the shape that plan needs.
|
| 103 |
-
"""
|
| 104 |
-
said = (text or "").strip()
|
| 105 |
-
if not said:
|
| 106 |
-
return history, ""
|
| 107 |
-
speed = " (slower)" if slower else ""
|
| 108 |
-
# Escaped, not interpolated raw: this string is rendered as HTML and the text comes
|
| 109 |
-
# from the visitor. A placeholder is still a rendering path.
|
| 110 |
-
line = f'<div class="turn">{html.escape(said)}{html.escape(speed)}</div>'
|
| 111 |
-
return f"{history}{line}", ""
|
| 112 |
-
|
| 113 |
-
|
| 114 |
-
def build_app() -> gr.Blocks:
|
| 115 |
-
"""Build the Blocks app. Importable and callable from tests without launching."""
|
| 116 |
-
with gr.Blocks(title="Japanese Learning Avatar", fill_height=True) as blocks:
|
| 117 |
-
with gr.Row(equal_height=True):
|
| 118 |
-
with gr.Column(scale=3, min_width=320):
|
| 119 |
-
VrmStage()
|
| 120 |
-
with gr.Column(scale=2, min_width=280):
|
| 121 |
-
StatusLine()
|
| 122 |
-
gr.HTML(value=INTRO_HTML, elem_id="intro-html")
|
| 123 |
-
transcript = gr.HTML(
|
| 124 |
-
value="",
|
| 125 |
-
elem_id="transcript",
|
| 126 |
-
label="Transcript",
|
| 127 |
-
)
|
| 128 |
-
text_in = gr.Textbox(
|
| 129 |
-
value="",
|
| 130 |
-
placeholder="Type anything - it echoes back for now",
|
| 131 |
-
label="Text (echo preview - the avatar cannot reply yet)",
|
| 132 |
-
elem_id="text-input",
|
| 133 |
-
submit_btn=True,
|
| 134 |
-
)
|
| 135 |
-
slower = gr.Checkbox(
|
| 136 |
-
value=False,
|
| 137 |
-
label="Speak slower (inactive until speech is connected)",
|
| 138 |
-
elem_id="slower",
|
| 139 |
-
)
|
| 140 |
-
send = gr.Button("Send", elem_id="send", variant="primary")
|
| 141 |
-
gr.HTML(value=CREDITS_HTML, elem_id="credits")
|
| 142 |
-
|
| 143 |
-
send.click(_echo, [text_in, slower, transcript], [transcript, text_in])
|
| 144 |
-
text_in.submit(_echo, [text_in, slower, transcript], [transcript, text_in])
|
| 145 |
-
return blocks
|
| 146 |
-
|
| 147 |
-
|
| 148 |
-
# Module level, and named `demo`, because that is what the Space runner requires.
|
| 149 |
-
# Hugging Face launches the app under gradio.utils.SpacesReloader, whose postrun() does
|
| 150 |
-
# `getattr(watch_module, self.demo_name)` on every reload check. With the Blocks living
|
| 151 |
-
# only inside build_app() there was no such attribute: the third deploy (e3ebc72) logged
|
| 152 |
-
# `GRADIO_HOT_RELOAD: Launching demo not found in __main__. Using 'demo'`, printed its
|
| 153 |
-
# "Running on local URL" line, and then stopped the Node server and exited - RUNTIME_ERROR
|
| 154 |
-
# with no traceback. The factory is kept so tests can build an independent app.
|
| 155 |
-
#
|
| 156 |
-
# This is the one module-level object this file is allowed to own. It is the app itself,
|
| 157 |
-
# not shared state: nothing mutates it per session.
|
| 158 |
-
demo = build_app()
|
| 159 |
|
| 160 |
if __name__ == "__main__":
|
| 161 |
demo.launch()
|
|
|
|
| 1 |
+
"""Space entry point: assembly only. The layout and the turn live in japanese_avatar.ui.blocks."""
|
| 2 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
import os
|
| 4 |
import sys
|
| 5 |
from pathlib import Path
|
| 6 |
|
|
|
|
| 7 |
import spaces
|
| 8 |
|
| 9 |
+
# The Space runs `python app.py` against requirements.txt alone and never pip-installs this
|
| 10 |
+
# repository, so the src/ layout is not on sys.path there (RUNTIME_ERROR on the first deploy).
|
|
|
|
|
|
|
|
|
|
|
|
|
| 11 |
_SRC = Path(__file__).resolve().parent / "src"
|
| 12 |
if str(_SRC) not in sys.path:
|
| 13 |
sys.path.insert(0, str(_SRC))
|
| 14 |
|
| 15 |
+
from japanese_avatar.ui.blocks import build_blocks # noqa: E402
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 16 |
|
| 17 |
|
| 18 |
def gpu_disabled() -> bool:
|
| 19 |
+
"""SC-4's kill switch, read per call so a restart with the variable flipped takes effect."""
|
|
|
|
|
|
|
|
|
|
|
|
|
| 20 |
return os.environ.get("DISABLE_GPU", "0").strip().lower() in {"1", "true", "yes"}
|
| 21 |
|
| 22 |
|
| 23 |
+
# ZeroGPU refuses to run a Space with no @spaces.GPU function ("No @spaces.GPU function
|
| 24 |
+
# detected during startup", docs/HOSTING.md). This probe exists only to satisfy that scan. It is
|
| 25 |
+
# not on the turn path, nothing calls it, and tests/test_no_gpu_on_turn_path.py proves both.
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 26 |
@spaces.GPU(duration=1)
|
| 27 |
def zerogpu_probe() -> str:
|
| 28 |
"""The GPU entry point ZeroGPU requires to exist. Deliberately unreachable in Phase 1."""
|
| 29 |
if gpu_disabled():
|
| 30 |
+
raise RuntimeError("zerogpu_probe called with DISABLE_GPU=1; Phase 1 has no GPU work")
|
|
|
|
|
|
|
|
|
|
| 31 |
return "zerogpu reachable"
|
| 32 |
|
| 33 |
|
| 34 |
+
# Module level and named `demo`: the Space runner looks it up by that name (docs/HOSTING.md).
|
| 35 |
+
# It is the app itself, not shared state; nothing mutates it per session.
|
| 36 |
+
demo = build_blocks()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 37 |
|
| 38 |
if __name__ == "__main__":
|
| 39 |
demo.launch()
|
|
@@ -0,0 +1,64 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Per-request stage timing.
|
| 2 |
+
|
| 3 |
+
MUST be instantiated per request. Gradio shares module-level mutables across every concurrent
|
| 4 |
+
user session, so a module-level ``TurnTimings`` would leak one visitor's latency into another's
|
| 5 |
+
directive. ``tests/test_directive.py::test_timings_are_per_request`` scans the package for exactly
|
| 6 |
+
that mistake.
|
| 7 |
+
|
| 8 |
+
Two ways to record a stage, both accumulating into the same record so the synthesiser (which
|
| 9 |
+
``mark``s two stages it times itself) and the turn handler (which wraps its own stages in
|
| 10 |
+
``stage``) produce one flat dict::
|
| 11 |
+
|
| 12 |
+
timings = TurnTimings()
|
| 13 |
+
synthesize(text, timings=timings) # marks audio_query and synthesis
|
| 14 |
+
with timings.stage("timeline"):
|
| 15 |
+
...
|
| 16 |
+
timings.as_dict() # {"audio_query_ms": ..., "synthesis_ms": ..., "timeline_ms": ...,
|
| 17 |
+
# "server_total_ms": ...}
|
| 18 |
+
|
| 19 |
+
``server_total_ms`` is measured from construction, so construct the record at the top of the
|
| 20 |
+
request handler, not lazily inside it.
|
| 21 |
+
"""
|
| 22 |
+
|
| 23 |
+
from __future__ import annotations
|
| 24 |
+
|
| 25 |
+
import time
|
| 26 |
+
from collections.abc import Iterator
|
| 27 |
+
from contextlib import contextmanager
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
class TurnTimings:
|
| 31 |
+
"""Per-stage millisecond timings for a single turn. Session-scoped; never a global."""
|
| 32 |
+
|
| 33 |
+
def __init__(self) -> None:
|
| 34 |
+
self._stages: dict[str, float] = {}
|
| 35 |
+
self._t0 = time.perf_counter()
|
| 36 |
+
|
| 37 |
+
def mark(self, name: str, ms: float) -> None:
|
| 38 |
+
"""Record ``ms`` for stage ``name``, accumulating if the stage repeats in one turn."""
|
| 39 |
+
self._stages[name] = self._stages.get(name, 0.0) + float(ms)
|
| 40 |
+
|
| 41 |
+
@contextmanager
|
| 42 |
+
def stage(self, name: str) -> Iterator[None]:
|
| 43 |
+
"""Time the enclosed block as stage ``name``. Records even if the block raises."""
|
| 44 |
+
start = time.perf_counter()
|
| 45 |
+
try:
|
| 46 |
+
yield
|
| 47 |
+
finally:
|
| 48 |
+
self.mark(name, (time.perf_counter() - start) * 1000.0)
|
| 49 |
+
|
| 50 |
+
@property
|
| 51 |
+
def stages(self) -> dict[str, float]:
|
| 52 |
+
"""A copy of the raw per-stage milliseconds, keyed by bare stage name."""
|
| 53 |
+
return dict(self._stages)
|
| 54 |
+
|
| 55 |
+
def as_dict(self) -> dict[str, float]:
|
| 56 |
+
"""The wire form: ``<stage>_ms`` per stage plus ``server_total_ms`` since construction.
|
| 57 |
+
|
| 58 |
+
A fresh dict every call, so a caller cannot mutate the turn's record after the fact.
|
| 59 |
+
Values are rounded to microseconds - nothing downstream can use finer resolution and the
|
| 60 |
+
directive is smaller for it.
|
| 61 |
+
"""
|
| 62 |
+
out = {f"{name}_ms": round(ms, 3) for name, ms in self._stages.items()}
|
| 63 |
+
out["server_total_ms"] = round((time.perf_counter() - self._t0) * 1000.0, 3)
|
| 64 |
+
return out
|
|
@@ -0,0 +1,269 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""The Phase 1 Blocks layout and the server side of the turn loop.
|
| 2 |
+
|
| 3 |
+
Two rules govern this module and both come from how Gradio runs a Space:
|
| 4 |
+
|
| 5 |
+
1. **No module-level mutable state.** Gradio shares module globals across every concurrent
|
| 6 |
+
visitor session. Every ``TurnTimings`` is constructed inside the request that owns it, and the
|
| 7 |
+
only module-level names are constants and functions.
|
| 8 |
+
2. **Python sends one directive per utterance; the browser owns the clock.** ``turn`` returns the
|
| 9 |
+
audio, the finished viseme timeline and the per-stage timings in one message, and never drives
|
| 10 |
+
animation frame by frame.
|
| 11 |
+
|
| 12 |
+
How the browser reaches ``turn``: the stage component is built with both server functions
|
| 13 |
+
registered, and Gradio exposes each as an async method on the ``server`` object inside
|
| 14 |
+
``js_on_load``. Gradio 6.22.0's bridge (verified in the compiled frontend, ``core-*.js``) packs a
|
| 15 |
+
call's arguments into ONE JSON
|
| 16 |
+
value - a lone argument is sent as itself, several are sent as a list, none as ``[]`` - and the
|
| 17 |
+
route then calls the Python function with that single value as its only positional argument.
|
| 18 |
+
So ``turn`` accepts a ``{"text", "speed"}`` dict from the browser as well as the plain
|
| 19 |
+
``turn(text, speed=...)`` signature Python callers and tests use, and ``greeting`` tolerates a
|
| 20 |
+
stray positional. A raised exception becomes an HTTP 500 that the Gradio client swallows into
|
| 21 |
+
``undefined`` on the JS side, so every failure path here returns a structured ``{"error": ...}``
|
| 22 |
+
instead of raising.
|
| 23 |
+
"""
|
| 24 |
+
|
| 25 |
+
from __future__ import annotations
|
| 26 |
+
|
| 27 |
+
import base64
|
| 28 |
+
import logging
|
| 29 |
+
import uuid
|
| 30 |
+
from pathlib import Path
|
| 31 |
+
from typing import Any
|
| 32 |
+
|
| 33 |
+
import gradio as gr
|
| 34 |
+
|
| 35 |
+
from japanese_avatar.telemetry.timings import TurnTimings
|
| 36 |
+
from japanese_avatar.ui.avatar_component import StatusLine, VrmStage
|
| 37 |
+
from japanese_avatar.voice import tts
|
| 38 |
+
from japanese_avatar.voice.models import AvatarDirective
|
| 39 |
+
from japanese_avatar.voice.tts import CREDIT_STRING, synthesize
|
| 40 |
+
from japanese_avatar.voice.visemes import build_timeline, timeline_to_dicts
|
| 41 |
+
|
| 42 |
+
log = logging.getLogger(__name__)
|
| 43 |
+
|
| 44 |
+
#: The repo checkout. blocks.py lives at src/japanese_avatar/ui/, three levels below it.
|
| 45 |
+
REPO_ROOT = Path(__file__).resolve().parents[3]
|
| 46 |
+
|
| 47 |
+
#: The fixed utterance behind "Say hello", so success criterion 2 ("the avatar speaks a Japanese
|
| 48 |
+
#: utterance aloud") is reachable with one click and no typing.
|
| 49 |
+
GREETING_TEXT = "こんにちは。日本語を練習しましょう。"
|
| 50 |
+
|
| 51 |
+
#: A turn is one sentence, not an essay. Longer text is refused with a structured error rather
|
| 52 |
+
#: than synthesised into a 30-second data URL.
|
| 53 |
+
MAX_TEXT_CHARS = 200
|
| 54 |
+
|
| 55 |
+
#: VOICEVOX's speedScale is a divisor on every phoneme length; outside this band the audio is
|
| 56 |
+
#: unintelligible and the timeline degenerate. 0.75 is the "Slower" control's value.
|
| 57 |
+
SPEED_MIN = 0.5
|
| 58 |
+
SPEED_MAX = 2.0
|
| 59 |
+
SLOWER_SPEED = 0.75
|
| 60 |
+
|
| 61 |
+
VOICEVOX_TERMS_URL = "https://voicevox.hiroshiba.jp/term/"
|
| 62 |
+
ZUNDAMON_TERMS_URL = "https://zunko.jp/con_ongen_kiyaku.html"
|
| 63 |
+
VRM_TERMS_URL = "https://vrm.dev/licenses/1.0/"
|
| 64 |
+
VRM_CREDIT = "VRM1_Constraint_Twist_Sample (c) 2022 pixiv Inc. - VRM Public License 1.0"
|
| 65 |
+
|
| 66 |
+
# ----------------------------------------------------------------------------- server functions
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
def _unpack(text: Any, speed: Any) -> tuple[Any, Any]:
|
| 70 |
+
"""Normalise the three shapes a call can arrive in: dict payload, list payload, or plain."""
|
| 71 |
+
if isinstance(text, dict):
|
| 72 |
+
return text.get("text"), text.get("speed", speed)
|
| 73 |
+
if isinstance(text, list | tuple):
|
| 74 |
+
if len(text) == 0:
|
| 75 |
+
return None, speed
|
| 76 |
+
return text[0], text[1] if len(text) > 1 else speed
|
| 77 |
+
return text, speed
|
| 78 |
+
|
| 79 |
+
|
| 80 |
+
def _validate(text: Any, speed: Any) -> tuple[str, float] | dict:
|
| 81 |
+
"""Return ``(text, speed)`` cleaned, or an ``{"error": ...}`` dict the browser can render."""
|
| 82 |
+
if not isinstance(text, str) or not text.strip():
|
| 83 |
+
return {"error": "Type or say something in Japanese first."}
|
| 84 |
+
cleaned = text.strip()
|
| 85 |
+
if len(cleaned) > MAX_TEXT_CHARS:
|
| 86 |
+
return {"error": f"Keep it to {MAX_TEXT_CHARS} characters ({len(cleaned)} given)."}
|
| 87 |
+
try:
|
| 88 |
+
rate = float(speed)
|
| 89 |
+
except (TypeError, ValueError):
|
| 90 |
+
return {"error": f"speed must be a number, got {speed!r}"}
|
| 91 |
+
if not (SPEED_MIN <= rate <= SPEED_MAX):
|
| 92 |
+
return {"error": f"speed must be between {SPEED_MIN} and {SPEED_MAX}, got {rate}"}
|
| 93 |
+
return cleaned, rate
|
| 94 |
+
|
| 95 |
+
|
| 96 |
+
def turn(text: Any, speed: float = 1.0) -> dict:
|
| 97 |
+
"""One utterance: text in, ``AvatarDirective`` dict out, with per-stage timings.
|
| 98 |
+
|
| 99 |
+
Every stage is timed into a ``TurnTimings`` constructed here, for this request only. The
|
| 100 |
+
timeline is built from the ``AudioQuery`` that was actually synthesised at this ``speed`` -
|
| 101 |
+
never from a cached one - because ``speedScale`` divides every phoneme, pre/post silences
|
| 102 |
+
included, and a timeline reused across speeds drifts by exactly the speed ratio (Pitfall 5).
|
| 103 |
+
"""
|
| 104 |
+
timings = TurnTimings()
|
| 105 |
+
text, speed = _unpack(text, speed)
|
| 106 |
+
checked = _validate(text, speed)
|
| 107 |
+
if isinstance(checked, dict):
|
| 108 |
+
return checked
|
| 109 |
+
cleaned, rate = checked
|
| 110 |
+
|
| 111 |
+
try:
|
| 112 |
+
result = synthesize(cleaned, speed=rate, timings=timings)
|
| 113 |
+
with timings.stage("timeline"):
|
| 114 |
+
timeline = timeline_to_dicts(build_timeline(result.audio_query))
|
| 115 |
+
with timings.stage("encode"):
|
| 116 |
+
audio_url = "data:audio/wav;base64," + base64.b64encode(result.wav_bytes).decode(
|
| 117 |
+
"ascii"
|
| 118 |
+
)
|
| 119 |
+
except Exception as exc: # noqa: BLE001 - the bridge would turn a raise into `undefined`
|
| 120 |
+
log.exception("turn failed for %r at speed %s", cleaned, rate)
|
| 121 |
+
return {"error": f"synthesis failed: {exc}"}
|
| 122 |
+
|
| 123 |
+
return AvatarDirective(
|
| 124 |
+
turn_id=f"t-{uuid.uuid4().hex[:8]}",
|
| 125 |
+
audio_url=audio_url,
|
| 126 |
+
timeline=timeline,
|
| 127 |
+
subtitle=cleaned,
|
| 128 |
+
expression="neutral",
|
| 129 |
+
speed=rate,
|
| 130 |
+
timings=timings.as_dict(),
|
| 131 |
+
).to_dict()
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
def greeting(_payload: Any = None) -> dict:
|
| 135 |
+
"""The "Say hello" utterance. ``_payload`` absorbs the ``[]`` the bridge sends for no args."""
|
| 136 |
+
return turn(GREETING_TEXT)
|
| 137 |
+
|
| 138 |
+
|
| 139 |
+
def warm_synthesizer() -> None:
|
| 140 |
+
"""Pay the synthesiser load before the first turn. Idempotent; logs the cost the first time.
|
| 141 |
+
|
| 142 |
+
Bound to ``demo.load`` rather than run at import: the Space imports this module to find the
|
| 143 |
+
module-level ``demo`` and must stay importable where the VOICEVOX wheel or its assets are
|
| 144 |
+
absent, so the load cost is paid on the first page load - concurrently with the visitor's
|
| 145 |
+
10 MiB VRM download, which takes several times longer.
|
| 146 |
+
"""
|
| 147 |
+
timings = TurnTimings()
|
| 148 |
+
try:
|
| 149 |
+
seconds = tts.warmup(timings)
|
| 150 |
+
except Exception as exc: # noqa: BLE001 - startup must not take the page down
|
| 151 |
+
log.warning("synthesizer warm-up failed; the first turn will report it: %s", exc)
|
| 152 |
+
return
|
| 153 |
+
if seconds > 0.01:
|
| 154 |
+
log.info("synthesizer warm-up took %.2f s", seconds)
|
| 155 |
+
|
| 156 |
+
|
| 157 |
+
# ------------------------------------------------------------------------------------- layout
|
| 158 |
+
|
| 159 |
+
# Every visitor-facing string below is deliberately honest about what this build does. It is a
|
| 160 |
+
# public portfolio Space; an avatar that echoes must say it echoes.
|
| 161 |
+
INTRO_HTML = (
|
| 162 |
+
'<div id="intro" class="intro">'
|
| 163 |
+
"<strong>Phase 1: the avatar repeats what you say. Tutoring arrives in Phase 3.</strong> "
|
| 164 |
+
"Type Japanese and press Enter, or hold the microphone button and speak. Everything you hear "
|
| 165 |
+
"is synthesised on the CPU with mora-timed lip-sync; speech recognition runs in your browser."
|
| 166 |
+
"</div>"
|
| 167 |
+
)
|
| 168 |
+
|
| 169 |
+
# Server-rendered so the first paint already carries it; the inner ids are what the host script
|
| 170 |
+
# writes to, and the wrappers are Gradio's own elements that tests select by elem_id.
|
| 171 |
+
TRANSCRIPT_HTML = '<div id="transcript-text" class="transcript" aria-live="polite"></div>'
|
| 172 |
+
LATENCY_HTML = '<div id="latency-text" class="latency">dispatch→speech: —</div>'
|
| 173 |
+
ASR_BADGE_HTML = (
|
| 174 |
+
'<div id="asr-tier-text" class="asr-badge">ASR: loads in your browser on the first push</div>'
|
| 175 |
+
)
|
| 176 |
+
|
| 177 |
+
# DPLY-04. The exact string VOICEVOX:ずんだもん, ASCII colon, no spaces - the form the character
|
| 178 |
+
# terms give as their example - always visible, with no interaction. The VRM credit is voluntary
|
| 179 |
+
# (creditNotation: unnecessary) and carried anyway. See docs/VOICEVOX-SETUP.md and docs/ASSETS.md.
|
| 180 |
+
CREDITS_HTML = (
|
| 181 |
+
'<div id="credits-text" class="credits">'
|
| 182 |
+
f"Voice: {CREDIT_STRING} · Avatar: {VRM_CREDIT}"
|
| 183 |
+
"</div>"
|
| 184 |
+
)
|
| 185 |
+
|
| 186 |
+
# The flow-down notice, adjacent to the replay control (software clause 3 / voice-model clause
|
| 187 |
+
# 4): wherever synthesised audio is obtainable, downstream users are bound to the same terms.
|
| 188 |
+
TERMS_NOTICE_HTML = (
|
| 189 |
+
'<div id="terms-notice-text" class="terms-notice">'
|
| 190 |
+
f"Synthesised audio is provided under the "
|
| 191 |
+
f'<a href="{VOICEVOX_TERMS_URL}" target="_blank" rel="noopener">VOICEVOX</a> and '
|
| 192 |
+
f'<a href="{ZUNDAMON_TERMS_URL}" target="_blank" rel="noopener">{CREDIT_STRING}</a> '
|
| 193 |
+
"terms of use; by using it you agree to comply with them."
|
| 194 |
+
"</div>"
|
| 195 |
+
)
|
| 196 |
+
|
| 197 |
+
# The 「アプリの紹介画面」 the character terms ask for: an about screen, findable with a little
|
| 198 |
+
# looking, carrying the full credit set. Open on first load so it is on the introduction screen
|
| 199 |
+
# rather than behind a click.
|
| 200 |
+
ABOUT_MD = f"""\
|
| 201 |
+
**Voice** - {CREDIT_STRING}. Speech is synthesised with
|
| 202 |
+
[VOICEVOX CORE]({VOICEVOX_TERMS_URL}) (software terms) using the ずんだもん voice by SSS LLC
|
| 203 |
+
([character terms]({ZUNDAMON_TERMS_URL})). The synthesised audio is provided under both sets of
|
| 204 |
+
terms; by using it you agree to comply with them.
|
| 205 |
+
|
| 206 |
+
**Avatar** - {VRM_CREDIT}. Source: the official VRM specification samples
|
| 207 |
+
(`vrm-c/vrm-specification`). Terms: [VRM Public License 1.0]({VRM_TERMS_URL}); the file's embedded
|
| 208 |
+
`VRMC_vrm.meta` grants redistribution, avatar use by everyone, modification and commercial use,
|
| 209 |
+
and requires no credit.
|
| 210 |
+
|
| 211 |
+
**Japanese text analysis** - Open JTalk dictionary `open_jtalk_dic_utf_8-1.11`, BSD-3-Clause,
|
| 212 |
+
(c) 2009 Nara Institute of Science and Technology; Open JTalk itself is by the Nagoya Institute
|
| 213 |
+
of Technology and the HTS Working Group, also under a modified BSD licence.
|
| 214 |
+
|
| 215 |
+
**Runtime libraries** - [three.js](https://threejs.org/) (MIT),
|
| 216 |
+
[@pixiv/three-vrm](https://github.com/pixiv/three-vrm) (MIT) and
|
| 217 |
+
[@huggingface/transformers](https://github.com/huggingface/transformers.js) (Apache-2.0), loaded
|
| 218 |
+
in your browser. Speech recognition (Whisper) runs entirely on your device.
|
| 219 |
+
|
| 220 |
+
This is a Phase 1 preview: the avatar repeats what you type or say. There is no tutor yet.
|
| 221 |
+
"""
|
| 222 |
+
|
| 223 |
+
|
| 224 |
+
def build_blocks() -> gr.Blocks:
|
| 225 |
+
"""Build the Blocks app. Importable and callable from tests without launching.
|
| 226 |
+
|
| 227 |
+
No Python event handlers are attached to the controls: every control is wired in the browser,
|
| 228 |
+
by ``avatar/host.js``, to the ``window.Avatar`` facade, and the only server traffic a turn
|
| 229 |
+
generates is the ``server_functions`` call to :func:`turn`. That is what keeps the whole loop
|
| 230 |
+
identical under both avatar transports.
|
| 231 |
+
"""
|
| 232 |
+
# Serves avatar/ straight off disk, bypassing the Gradio cache, which is how the browser
|
| 233 |
+
# reaches avatar.js, host.js, stage.html and tutor.vrm. Deliberately one directory: every
|
| 234 |
+
# file under a listed path becomes network-reachable.
|
| 235 |
+
gr.set_static_paths([REPO_ROOT / "avatar"])
|
| 236 |
+
|
| 237 |
+
with gr.Blocks(title="Japanese Learning Avatar", fill_height=True) as blocks:
|
| 238 |
+
with gr.Row(equal_height=True):
|
| 239 |
+
with gr.Column(scale=3, min_width=320):
|
| 240 |
+
VrmStage(server_functions=[turn, greeting])
|
| 241 |
+
with gr.Column(scale=2, min_width=280):
|
| 242 |
+
StatusLine()
|
| 243 |
+
gr.HTML(value=INTRO_HTML, elem_id="intro-html")
|
| 244 |
+
gr.HTML(value=TRANSCRIPT_HTML, elem_id="transcript")
|
| 245 |
+
with gr.Row():
|
| 246 |
+
gr.Button("Hold to talk", elem_id="ptt-button", variant="secondary")
|
| 247 |
+
gr.Button("Say hello", elem_id="hello-button", variant="secondary")
|
| 248 |
+
gr.Textbox(
|
| 249 |
+
value="",
|
| 250 |
+
placeholder="日本語を入力して Enter",
|
| 251 |
+
label="Type Japanese (the avatar says it back)",
|
| 252 |
+
elem_id="text-input",
|
| 253 |
+
lines=1,
|
| 254 |
+
max_lines=1,
|
| 255 |
+
submit_btn=False,
|
| 256 |
+
)
|
| 257 |
+
with gr.Row():
|
| 258 |
+
gr.Button("Send", elem_id="send-button", variant="primary")
|
| 259 |
+
gr.Button("Replay", elem_id="replay-button")
|
| 260 |
+
gr.Button("Slower", elem_id="slower-button")
|
| 261 |
+
gr.HTML(value=TERMS_NOTICE_HTML, elem_id="terms-notice")
|
| 262 |
+
gr.HTML(value=LATENCY_HTML, elem_id="latency-line")
|
| 263 |
+
gr.HTML(value=ASR_BADGE_HTML, elem_id="asr-tier-badge")
|
| 264 |
+
gr.HTML(value=CREDITS_HTML, elem_id="credits")
|
| 265 |
+
with gr.Accordion("About and credits", open=True, elem_id="about-panel"):
|
| 266 |
+
gr.Markdown(ABOUT_MD, elem_id="about-text")
|
| 267 |
+
|
| 268 |
+
blocks.load(warm_synthesizer, api_visibility="private")
|
| 269 |
+
return blocks
|
|
@@ -1,10 +1,11 @@
|
|
| 1 |
"""Shared voice data types.
|
| 2 |
|
| 3 |
-
|
| 4 |
-
|
| 5 |
-
|
| 6 |
-
|
| 7 |
-
|
|
|
|
| 8 |
|
| 9 |
:class:`SynthResult` carries the ``AudioQuery`` as a **plain dict**, never a live
|
| 10 |
``voicevox_core`` object. That boundary conversion is what lets ``visemes.py`` (plan 01-06) be a
|
|
@@ -15,7 +16,11 @@ installed.
|
|
| 15 |
from __future__ import annotations
|
| 16 |
|
| 17 |
import json
|
| 18 |
-
from dataclasses import asdict, dataclass
|
|
|
|
|
|
|
|
|
|
|
|
|
| 19 |
|
| 20 |
#: Frame rate VOICEVOX quantises every phoneme to: 24000 Hz / 256 samples.
|
| 21 |
#: Exposed here because the timeline builder and the fixtures must agree on it exactly.
|
|
@@ -54,30 +59,15 @@ class SynthResult:
|
|
| 54 |
speed_scale: float
|
| 55 |
|
| 56 |
|
| 57 |
-
@dataclass
|
| 58 |
-
class TurnTimings:
|
| 59 |
-
"""Per-stage millisecond timings for a single turn.
|
| 60 |
-
|
| 61 |
-
Session-scoped. NEVER a module-level global - Gradio shares those across all users.
|
| 62 |
-
"""
|
| 63 |
-
|
| 64 |
-
stages: dict[str, float] = field(default_factory=dict)
|
| 65 |
-
|
| 66 |
-
def mark(self, name: str, ms: float) -> None:
|
| 67 |
-
"""Record ``ms`` for stage ``name``, accumulating if the stage repeats in one turn."""
|
| 68 |
-
self.stages[name] = self.stages.get(name, 0.0) + float(ms)
|
| 69 |
-
|
| 70 |
-
def as_dict(self) -> dict[str, float]:
|
| 71 |
-
"""A copy, so a caller cannot mutate the turn's record after the fact."""
|
| 72 |
-
return dict(self.stages)
|
| 73 |
-
|
| 74 |
-
|
| 75 |
@dataclass(frozen=True)
|
| 76 |
class AvatarDirective:
|
| 77 |
"""The single message Python sends the browser per utterance.
|
| 78 |
|
| 79 |
The browser owns the clock from here on: it schedules ``timeline`` against
|
| 80 |
``AudioContext.currentTime``. Python never drives animation frame by frame.
|
|
|
|
|
|
|
|
|
|
| 81 |
"""
|
| 82 |
|
| 83 |
turn_id: str
|
|
@@ -88,5 +78,9 @@ class AvatarDirective:
|
|
| 88 |
speed: float
|
| 89 |
timings: dict[str, float]
|
| 90 |
|
|
|
|
|
|
|
|
|
|
|
|
|
| 91 |
def to_json(self) -> str:
|
| 92 |
-
return json.dumps(
|
|
|
|
| 1 |
"""Shared voice data types.
|
| 2 |
|
| 3 |
+
Every type here is frozen. The one mutable record in the voice path - :class:`TurnTimings` - now
|
| 4 |
+
lives in :mod:`japanese_avatar.telemetry.timings` and is re-exported here so the synthesiser's
|
| 5 |
+
``timings=`` parameter keeps its type. It must be **instantiated per request and never held at
|
| 6 |
+
module level**: Gradio shares module-level objects across every concurrent user session, so a
|
| 7 |
+
module-level ``TurnTimings`` would silently interleave one visitor's stage timings with
|
| 8 |
+
another's.
|
| 9 |
|
| 10 |
:class:`SynthResult` carries the ``AudioQuery`` as a **plain dict**, never a live
|
| 11 |
``voicevox_core`` object. That boundary conversion is what lets ``visemes.py`` (plan 01-06) be a
|
|
|
|
| 16 |
from __future__ import annotations
|
| 17 |
|
| 18 |
import json
|
| 19 |
+
from dataclasses import asdict, dataclass
|
| 20 |
+
|
| 21 |
+
from japanese_avatar.telemetry.timings import TurnTimings
|
| 22 |
+
|
| 23 |
+
__all__ = ["FRAMERATE", "AvatarDirective", "SynthResult", "TurnTimings", "VisemeEvent"]
|
| 24 |
|
| 25 |
#: Frame rate VOICEVOX quantises every phoneme to: 24000 Hz / 256 samples.
|
| 26 |
#: Exposed here because the timeline builder and the fixtures must agree on it exactly.
|
|
|
|
| 59 |
speed_scale: float
|
| 60 |
|
| 61 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 62 |
@dataclass(frozen=True)
|
| 63 |
class AvatarDirective:
|
| 64 |
"""The single message Python sends the browser per utterance.
|
| 65 |
|
| 66 |
The browser owns the clock from here on: it schedules ``timeline`` against
|
| 67 |
``AudioContext.currentTime``. Python never drives animation frame by frame.
|
| 68 |
+
|
| 69 |
+
``timings`` is this turn's :class:`TurnTimings` record in wire form, so the browser can render
|
| 70 |
+
the per-stage breakdown next to its own dispatch-to-speech number without a second request.
|
| 71 |
"""
|
| 72 |
|
| 73 |
turn_id: str
|
|
|
|
| 78 |
speed: float
|
| 79 |
timings: dict[str, float]
|
| 80 |
|
| 81 |
+
def to_dict(self) -> dict:
|
| 82 |
+
"""The JSON-ready dict the gr.HTML server bridge returns to ``server.turn()``."""
|
| 83 |
+
return asdict(self)
|
| 84 |
+
|
| 85 |
def to_json(self) -> str:
|
| 86 |
+
return json.dumps(self.to_dict(), ensure_ascii=False)
|
|
@@ -22,7 +22,6 @@ pytest.importorskip("voicevox_core")
|
|
| 22 |
|
| 23 |
from japanese_avatar.telemetry.timings import TurnTimings # noqa: E402
|
| 24 |
from japanese_avatar.ui.blocks import GREETING_TEXT, MAX_TEXT_CHARS, greeting, turn # noqa: E402
|
| 25 |
-
|
| 26 |
from japanese_avatar.voice.models import AvatarDirective # noqa: E402
|
| 27 |
from japanese_avatar.voice.tts import wav_duration_seconds # noqa: E402
|
| 28 |
from japanese_avatar.voice.visemes import FRAMERATE, to_frame # noqa: E402
|
|
|
|
| 22 |
|
| 23 |
from japanese_avatar.telemetry.timings import TurnTimings # noqa: E402
|
| 24 |
from japanese_avatar.ui.blocks import GREETING_TEXT, MAX_TEXT_CHARS, greeting, turn # noqa: E402
|
|
|
|
| 25 |
from japanese_avatar.voice.models import AvatarDirective # noqa: E402
|
| 26 |
from japanese_avatar.voice.tts import wav_duration_seconds # noqa: E402
|
| 27 |
from japanese_avatar.voice.visemes import FRAMERATE, to_frame # noqa: E402
|