Spaces:
Running on Zero
Running on Zero
File size: 7,316 Bytes
bdeb461 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 184 185 186 187 188 189 190 191 192 | """Rehearse the deployed suite against a LOCAL app.py, on a free port, with one command.
.venv/Scripts/python.exe scripts/rehearse_deployed.py tests/e2e/test_avatar_loop.py -q
.venv/Scripts/python.exe scripts/rehearse_deployed.py tests/e2e/test_avatar_loop.py \
-q -k "furigana or language_assets"
The deployed rows (``pytestmark = pytest.mark.deployed``) skip unless pytest is given
``--space-url``. Until the owner pushes, the only honest way to run them is against a real
app.py serving the working tree - and "start the app by hand on 7860, remember to export
DISABLE_GPU, remember to kill it" is exactly the ritual that produces a stale process, a
port clash with the parity suite's own apps, or a rehearsal that silently ran against
yesterday's build. So it is a script, and every later plan in this phase reuses it.
What it guarantees:
* a FREE port, taken by binding ``('127.0.0.1', 0)`` and reading the number back, so two
rehearsals (or a rehearsal and the parity suite) never collide. Nothing is hard-coded.
* ``DISABLE_GPU=1`` - the Phase 1 kill switch. Locally there is no ZeroGPU allocator, and
a rehearsal must never reach ``@spaces.GPU``.
* the app is up before pytest starts: ``/config`` is polled until it answers 200, which on
a cold working tree means the Sudachi dictionary, the compact JMdict and the CTranslate2
translator have all been read (``Blocks.load`` warms them) - a wait measured in tens of
seconds, not a sleep.
* the app is stopped afterwards, whatever pytest did, and its tree is killed if it ignores
the polite request. A rehearsal leaves no listener behind.
Exit code: pytest's own, so this is drop-in for CI - except 3, which means the app never
came up; its last log lines are printed so the reason is on screen rather than in a pipe.
The URL handed to pytest is a loopback one, which keeps ``docs/LATENCY.md``'s rule in
force: the latency harness writes to tmp_path for a non-Space URL, so a rehearsal can
never overwrite the Space's recorded numbers.
"""
from __future__ import annotations
import collections
import os
import socket
import subprocess
import sys
import threading
import time
import urllib.error
import urllib.request
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parent.parent
# The app has to import Gradio, build the Blocks, and warm the tokenizer, the compact
# JMdict and the translator before /config answers. Measured ~30-60 s on this fleet from a
# warm page cache; 180 s is the headroom a cold one needs.
BOOT_TIMEOUT_S = 180
POLL_S = 0.5
STOP_GRACE_S = 15
LOG_TAIL_LINES = 40
APP_NEVER_CAME_UP = 3
def free_port() -> int:
"""A port the OS says is free right now. Bind to 0, read it back, release it."""
with socket.socket() as probe:
probe.bind(("127.0.0.1", 0))
return int(probe.getsockname()[1])
def _drain(stream, sink: collections.deque) -> None:
"""Read the child's merged output forever, keeping only the tail.
A pipe nobody reads fills and blocks the writer: Gradio logs every request, so an app
left unread would deadlock partway through a long suite. The deque bounds the memory.
"""
for line in iter(stream.readline, ""):
sink.append(line.rstrip("\n"))
stream.close()
def start_app(port: int) -> tuple[subprocess.Popen, collections.deque]:
env = {
**os.environ,
"DISABLE_GPU": "1",
"GRADIO_SERVER_NAME": "127.0.0.1",
"GRADIO_SERVER_PORT": str(port),
"GRADIO_ANALYTICS_ENABLED": "False",
"PYTHONIOENCODING": "utf-8",
}
proc = subprocess.Popen( # noqa: S603 - sys.executable and this repo's own app.py
[sys.executable, "app.py"],
cwd=str(REPO_ROOT),
env=env,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
text=True,
encoding="utf-8",
errors="replace",
bufsize=1,
)
log: collections.deque = collections.deque(maxlen=400)
threading.Thread(target=_drain, args=(proc.stdout, log), daemon=True).start()
return proc, log
def wait_for_config(port: int, proc: subprocess.Popen) -> float:
"""Block until GET /config answers 200. Returns the seconds it took.
``/config`` rather than ``/``: Gradio 6 serves an SSR shell for the root before the app
is fully assembled, while /config is the app's own description.
"""
url = f"http://127.0.0.1:{port}/config"
deadline = time.monotonic() + BOOT_TIMEOUT_S
t0 = time.monotonic()
while time.monotonic() < deadline:
if proc.poll() is not None:
raise RuntimeError(f"app.py exited with {proc.returncode} before serving {url}")
try:
with urllib.request.urlopen(url, timeout=5) as response: # noqa: S310 - loopback
if response.status == 200:
return time.monotonic() - t0
except (urllib.error.URLError, TimeoutError, ConnectionError, OSError):
time.sleep(POLL_S)
raise RuntimeError(f"app.py did not answer {url} within {BOOT_TIMEOUT_S}s")
def stop_app(proc: subprocess.Popen) -> None:
"""Terminate, then kill the whole tree if it lingers. Gradio's server thread has
ignored a terminate here before, and a stray listener would poison the next run."""
if proc.poll() is not None:
return
proc.terminate()
try:
proc.wait(timeout=STOP_GRACE_S)
return
except subprocess.TimeoutExpired:
pass
if os.name == "nt":
subprocess.run( # noqa: S603, S607 - fixed argv, no shell
["taskkill", "/F", "/T", "/PID", str(proc.pid)],
capture_output=True,
check=False,
)
else:
proc.kill()
try:
proc.wait(timeout=STOP_GRACE_S)
except subprocess.TimeoutExpired:
print("[rehearse] WARNING: app.py did not die; check for a stray process", flush=True)
def main(argv: list[str]) -> int:
if not argv:
print(__doc__)
return 2
port = free_port()
space_url = f"http://127.0.0.1:{port}"
print(f"[rehearse] starting DISABLE_GPU=1 app.py on {space_url}", flush=True)
proc, log = start_app(port)
try:
try:
boot_s = wait_for_config(port, proc)
except RuntimeError as err:
print(f"[rehearse] {err}", flush=True)
print(f"[rehearse] last {LOG_TAIL_LINES} log lines from app.py:", flush=True)
for line in list(log)[-LOG_TAIL_LINES:]:
print(f" {line}", flush=True)
return APP_NEVER_CAME_UP
print(f"[rehearse] app answered /config after {boot_s:.1f}s", flush=True)
command = [
sys.executable,
"-m",
"pytest",
*argv,
"--space-url",
space_url,
"-p",
"no:cacheprovider",
]
print(f"[rehearse] {' '.join(command)}", flush=True)
result = subprocess.run( # noqa: S603 - argv built here, no shell
command,
cwd=str(REPO_ROOT),
env={**os.environ, "PYTHONIOENCODING": "utf-8"},
check=False,
)
return result.returncode
finally:
stop_app(proc)
print("[rehearse] app stopped", flush=True)
if __name__ == "__main__":
raise SystemExit(main(sys.argv[1:]))
|