Spaces:
Running on Zero
Running on Zero
Download scripts/rehearse_deployed.py from WolfDavid/japanese-learning-avatar: direct link, hf CLI and curl.
- Browser
- Download file 7.32 kB
-
https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/94dba691ff7d237c5b73d79cf931d25456f490d0/scripts/rehearse_deployed.py
- Command line
-
hf download hf://spaces/WolfDavid/japanese-learning-avatar@94dba691ff7d237c5b73d79cf931d25456f490d0/scripts/rehearse_deployed.py
-
curl -L -o rehearse_deployed.py https://huggingface.co/spaces/WolfDavid/japanese-learning-avatar/resolve/94dba691ff7d237c5b73d79cf931d25456f490d0/scripts/rehearse_deployed.py
7.32 kB
| """Rehearse the deployed suite against a LOCAL app.py, on a free port, with one command. | |
| .venv/Scripts/python.exe scripts/rehearse_deployed.py tests/e2e/test_avatar_loop.py -q | |
| .venv/Scripts/python.exe scripts/rehearse_deployed.py tests/e2e/test_avatar_loop.py \ | |
| -q -k "furigana or language_assets" | |
| The deployed rows (``pytestmark = pytest.mark.deployed``) skip unless pytest is given | |
| ``--space-url``. Until the owner pushes, the only honest way to run them is against a real | |
| app.py serving the working tree - and "start the app by hand on 7860, remember to export | |
| DISABLE_GPU, remember to kill it" is exactly the ritual that produces a stale process, a | |
| port clash with the parity suite's own apps, or a rehearsal that silently ran against | |
| yesterday's build. So it is a script, and every later plan in this phase reuses it. | |
| What it guarantees: | |
| * a FREE port, taken by binding ``('127.0.0.1', 0)`` and reading the number back, so two | |
| rehearsals (or a rehearsal and the parity suite) never collide. Nothing is hard-coded. | |
| * ``DISABLE_GPU=1`` - the Phase 1 kill switch. Locally there is no ZeroGPU allocator, and | |
| a rehearsal must never reach ``@spaces.GPU``. | |
| * the app is up before pytest starts: ``/config`` is polled until it answers 200, which on | |
| a cold working tree means the Sudachi dictionary, the compact JMdict and the CTranslate2 | |
| translator have all been read (``Blocks.load`` warms them) - a wait measured in tens of | |
| seconds, not a sleep. | |
| * the app is stopped afterwards, whatever pytest did, and its tree is killed if it ignores | |
| the polite request. A rehearsal leaves no listener behind. | |
| Exit code: pytest's own, so this is drop-in for CI - except 3, which means the app never | |
| came up; its last log lines are printed so the reason is on screen rather than in a pipe. | |
| The URL handed to pytest is a loopback one, which keeps ``docs/LATENCY.md``'s rule in | |
| force: the latency harness writes to tmp_path for a non-Space URL, so a rehearsal can | |
| never overwrite the Space's recorded numbers. | |
| """ | |
| from __future__ import annotations | |
| import collections | |
| import os | |
| import socket | |
| import subprocess | |
| import sys | |
| import threading | |
| import time | |
| import urllib.error | |
| import urllib.request | |
| from pathlib import Path | |
| REPO_ROOT = Path(__file__).resolve().parent.parent | |
| # The app has to import Gradio, build the Blocks, and warm the tokenizer, the compact | |
| # JMdict and the translator before /config answers. Measured ~30-60 s on this fleet from a | |
| # warm page cache; 180 s is the headroom a cold one needs. | |
| BOOT_TIMEOUT_S = 180 | |
| POLL_S = 0.5 | |
| STOP_GRACE_S = 15 | |
| LOG_TAIL_LINES = 40 | |
| APP_NEVER_CAME_UP = 3 | |
| def free_port() -> int: | |
| """A port the OS says is free right now. Bind to 0, read it back, release it.""" | |
| with socket.socket() as probe: | |
| probe.bind(("127.0.0.1", 0)) | |
| return int(probe.getsockname()[1]) | |
| def _drain(stream, sink: collections.deque) -> None: | |
| """Read the child's merged output forever, keeping only the tail. | |
| A pipe nobody reads fills and blocks the writer: Gradio logs every request, so an app | |
| left unread would deadlock partway through a long suite. The deque bounds the memory. | |
| """ | |
| for line in iter(stream.readline, ""): | |
| sink.append(line.rstrip("\n")) | |
| stream.close() | |
| def start_app(port: int) -> tuple[subprocess.Popen, collections.deque]: | |
| env = { | |
| **os.environ, | |
| "DISABLE_GPU": "1", | |
| "GRADIO_SERVER_NAME": "127.0.0.1", | |
| "GRADIO_SERVER_PORT": str(port), | |
| "GRADIO_ANALYTICS_ENABLED": "False", | |
| "PYTHONIOENCODING": "utf-8", | |
| } | |
| proc = subprocess.Popen( # noqa: S603 - sys.executable and this repo's own app.py | |
| [sys.executable, "app.py"], | |
| cwd=str(REPO_ROOT), | |
| env=env, | |
| stdout=subprocess.PIPE, | |
| stderr=subprocess.STDOUT, | |
| text=True, | |
| encoding="utf-8", | |
| errors="replace", | |
| bufsize=1, | |
| ) | |
| log: collections.deque = collections.deque(maxlen=400) | |
| threading.Thread(target=_drain, args=(proc.stdout, log), daemon=True).start() | |
| return proc, log | |
| def wait_for_config(port: int, proc: subprocess.Popen) -> float: | |
| """Block until GET /config answers 200. Returns the seconds it took. | |
| ``/config`` rather than ``/``: Gradio 6 serves an SSR shell for the root before the app | |
| is fully assembled, while /config is the app's own description. | |
| """ | |
| url = f"http://127.0.0.1:{port}/config" | |
| deadline = time.monotonic() + BOOT_TIMEOUT_S | |
| t0 = time.monotonic() | |
| while time.monotonic() < deadline: | |
| if proc.poll() is not None: | |
| raise RuntimeError(f"app.py exited with {proc.returncode} before serving {url}") | |
| try: | |
| with urllib.request.urlopen(url, timeout=5) as response: # noqa: S310 - loopback | |
| if response.status == 200: | |
| return time.monotonic() - t0 | |
| except (urllib.error.URLError, TimeoutError, ConnectionError, OSError): | |
| time.sleep(POLL_S) | |
| raise RuntimeError(f"app.py did not answer {url} within {BOOT_TIMEOUT_S}s") | |
| def stop_app(proc: subprocess.Popen) -> None: | |
| """Terminate, then kill the whole tree if it lingers. Gradio's server thread has | |
| ignored a terminate here before, and a stray listener would poison the next run.""" | |
| if proc.poll() is not None: | |
| return | |
| proc.terminate() | |
| try: | |
| proc.wait(timeout=STOP_GRACE_S) | |
| return | |
| except subprocess.TimeoutExpired: | |
| pass | |
| if os.name == "nt": | |
| subprocess.run( # noqa: S603, S607 - fixed argv, no shell | |
| ["taskkill", "/F", "/T", "/PID", str(proc.pid)], | |
| capture_output=True, | |
| check=False, | |
| ) | |
| else: | |
| proc.kill() | |
| try: | |
| proc.wait(timeout=STOP_GRACE_S) | |
| except subprocess.TimeoutExpired: | |
| print("[rehearse] WARNING: app.py did not die; check for a stray process", flush=True) | |
| def main(argv: list[str]) -> int: | |
| if not argv: | |
| print(__doc__) | |
| return 2 | |
| port = free_port() | |
| space_url = f"http://127.0.0.1:{port}" | |
| print(f"[rehearse] starting DISABLE_GPU=1 app.py on {space_url}", flush=True) | |
| proc, log = start_app(port) | |
| try: | |
| try: | |
| boot_s = wait_for_config(port, proc) | |
| except RuntimeError as err: | |
| print(f"[rehearse] {err}", flush=True) | |
| print(f"[rehearse] last {LOG_TAIL_LINES} log lines from app.py:", flush=True) | |
| for line in list(log)[-LOG_TAIL_LINES:]: | |
| print(f" {line}", flush=True) | |
| return APP_NEVER_CAME_UP | |
| print(f"[rehearse] app answered /config after {boot_s:.1f}s", flush=True) | |
| command = [ | |
| sys.executable, | |
| "-m", | |
| "pytest", | |
| *argv, | |
| "--space-url", | |
| space_url, | |
| "-p", | |
| "no:cacheprovider", | |
| ] | |
| print(f"[rehearse] {' '.join(command)}", flush=True) | |
| result = subprocess.run( # noqa: S603 - argv built here, no shell | |
| command, | |
| cwd=str(REPO_ROOT), | |
| env={**os.environ, "PYTHONIOENCODING": "utf-8"}, | |
| check=False, | |
| ) | |
| return result.returncode | |
| finally: | |
| stop_app(proc) | |
| print("[rehearse] app stopped", flush=True) | |
| if __name__ == "__main__": | |
| raise SystemExit(main(sys.argv[1:])) | |