Fix startup on ZeroGPU hardware
Browse files- Declare a placeholder @spaces.GPU function: ZeroGPU refuses to start
without one. Inference stays on the CPU.
- Mount Gradio without SSR: Spaces enable it, which starts a Node server
on port 7860 and prevented uvicorn from binding it.
- Point the page config at this Space.
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
- app.py +17 -3
- requirements.txt +3 -3
- web/config.js +1 -1
app.py
CHANGED
|
@@ -5,11 +5,23 @@ A Gradio Space runs `python app.py` and expects a server on port 7860: this serv
|
|
| 5 |
|
| 6 |
On the free tier a Gradio Space gets ZeroGPU hardware. The GPU is deliberately not used: it is
|
| 7 |
metered per visitor (a couple of minutes a day) and queued, which does not suit a score
|
| 8 |
-
recomputed on every keystroke, and Laya answers in a fraction of a second on CPU anyway. So
|
| 9 |
-
|
| 10 |
"""
|
| 11 |
import os
|
| 12 |
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 13 |
# One model on CPU keeps memory and latency reasonable; override in the Space settings.
|
| 14 |
os.environ.setdefault("LAYA_BACKEND", "torch")
|
| 15 |
os.environ.setdefault("LAYA_MODELS", "typed")
|
|
@@ -27,7 +39,9 @@ with gr.Blocks(title="Laya Demo API") as info:
|
|
| 27 |
"La démo est sur `/`, l'état de l'API sur `/health`."
|
| 28 |
)
|
| 29 |
|
| 30 |
-
|
|
|
|
|
|
|
| 31 |
|
| 32 |
if __name__ == "__main__":
|
| 33 |
uvicorn.run(app, host="0.0.0.0", port=7860)
|
|
|
|
| 5 |
|
| 6 |
On the free tier a Gradio Space gets ZeroGPU hardware. The GPU is deliberately not used: it is
|
| 7 |
metered per visitor (a couple of minutes a day) and queued, which does not suit a score
|
| 8 |
+
recomputed on every keystroke, and Laya answers in a fraction of a second on CPU anyway. So
|
| 9 |
+
inference is pinned to the CPU.
|
| 10 |
"""
|
| 11 |
import os
|
| 12 |
|
| 13 |
+
# ZeroGPU refuses to start a Space that declares no @spaces.GPU function, so one is declared but
|
| 14 |
+
# never called. `spaces` must be imported before anything that imports torch (it patches CUDA
|
| 15 |
+
# initialisation); outside Hugging Face the package is absent or has no effect.
|
| 16 |
+
try:
|
| 17 |
+
import spaces
|
| 18 |
+
|
| 19 |
+
@spaces.GPU
|
| 20 |
+
def _zerogpu_placeholder():
|
| 21 |
+
"""Never called: only satisfies ZeroGPU's startup check."""
|
| 22 |
+
except Exception as error: # absent locally; any other failure shows in the Space logs
|
| 23 |
+
print(f"spaces unavailable ({type(error).__name__}: {error})", flush=True)
|
| 24 |
+
|
| 25 |
# One model on CPU keeps memory and latency reasonable; override in the Space settings.
|
| 26 |
os.environ.setdefault("LAYA_BACKEND", "torch")
|
| 27 |
os.environ.setdefault("LAYA_MODELS", "typed")
|
|
|
|
| 39 |
"La démo est sur `/`, l'état de l'API sur `/health`."
|
| 40 |
)
|
| 41 |
|
| 42 |
+
# Spaces set GRADIO_SSR_MODE, which starts a Node server on port 7860 and leaves uvicorn unable
|
| 43 |
+
# to bind it ("address already in use"); this page needs no server-side rendering.
|
| 44 |
+
app = gr.mount_gradio_app(app, info, path="/gradio", ssr_mode=False)
|
| 45 |
|
| 46 |
if __name__ == "__main__":
|
| 47 |
uvicorn.run(app, host="0.0.0.0", port=7860)
|
requirements.txt
CHANGED
|
@@ -1,7 +1,7 @@
|
|
| 1 |
-
# Hugging Face Space (Gradio SDK, which installs gradio itself).
|
| 2 |
-
#
|
| 3 |
-
--extra-index-url https://download.pytorch.org/whl/cpu
|
| 4 |
torch==2.13.0
|
| 5 |
laya==0.3.20
|
| 6 |
fastapi
|
| 7 |
uvicorn
|
|
|
|
|
|
| 1 |
+
# Hugging Face Space (Gradio SDK, which installs gradio itself). Standard PyTorch build, as the
|
| 2 |
+
# `spaces` package (needed by ZeroGPU hardware) expects it; inference is still forced onto the CPU.
|
|
|
|
| 3 |
torch==2.13.0
|
| 4 |
laya==0.3.20
|
| 5 |
fastapi
|
| 6 |
uvicorn
|
| 7 |
+
spaces
|
web/config.js
CHANGED
|
@@ -1,6 +1,6 @@
|
|
| 1 |
// URL of the Laya API used by the static deployment (e.g. on Vercel).
|
| 2 |
// Replace it with your Hugging Face Space URL, without a trailing slash.
|
| 3 |
-
window.LAYA_API = "https://
|
| 4 |
|
| 5 |
// Pause after the last keystroke before calling the API (ms). A CPU server takes
|
| 6 |
// about 1 to 2 s per request, so it should not be called on every keystroke.
|
|
|
|
| 1 |
// URL of the Laya API used by the static deployment (e.g. on Vercel).
|
| 2 |
// Replace it with your Hugging Face Space URL, without a trailing slash.
|
| 3 |
+
window.LAYA_API = "https://robynsd-laya-demo.hf.space";
|
| 4 |
|
| 5 |
// Pause after the last keystroke before calling the API (ms). A CPU server takes
|
| 6 |
// about 1 to 2 s per request, so it should not be called on every keystroke.
|