Spaces:
Paused
Paused
Claude Claude Sonnet 5 commited on
Serve llama.cpp web UI on port 7860 via gateway proxy
Browse filesgateway.py was the only process exposed on 7860 (HF Spaces only route
one port), but it had no route for "/" or static assets, and
llama-server was started with --no-webui, so every UI request 404'd.
Enable the built-in webui in llama-server and add a catch-all reverse
proxy in gateway.py that forwards any unmatched request to it.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01XvAfE7bsSLrQm1VVBswYnx
- .gitignore +2 -0
- gateway.py +68 -1
- start.sh +1 -2
.gitignore
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
__pycache__/
|
| 2 |
+
*.pyc
|
gateway.py
CHANGED
|
@@ -356,4 +356,71 @@ async def cancel_task(job_id: str) -> dict[str, Any]:
|
|
| 356 |
"UPDATE jobs SET status='cancelled', updated_at=? WHERE id=?",
|
| 357 |
(time.time(), job_id),
|
| 358 |
)
|
| 359 |
-
return {"task_id": job_id, "status": "cancelled"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 356 |
"UPDATE jobs SET status='cancelled', updated_at=? WHERE id=?",
|
| 357 |
(time.time(), job_id),
|
| 358 |
)
|
| 359 |
+
return {"task_id": job_id, "status": "cancelled"}
|
| 360 |
+
|
| 361 |
+
|
| 362 |
+
# Everything below is not part of this service's own API. It falls through to
|
| 363 |
+
# llama.cpp's built-in web UI (served on LLAMA_URL, reachable only from
|
| 364 |
+
# localhost inside the container), so that visiting the Space's public URL
|
| 365 |
+
# renders the llama.cpp chat UI instead of a 404.
|
| 366 |
+
_HOP_BY_HOP_HEADERS = {
|
| 367 |
+
"connection",
|
| 368 |
+
"keep-alive",
|
| 369 |
+
"proxy-authenticate",
|
| 370 |
+
"proxy-authorization",
|
| 371 |
+
"te",
|
| 372 |
+
"trailers",
|
| 373 |
+
"transfer-encoding",
|
| 374 |
+
"upgrade",
|
| 375 |
+
"content-length",
|
| 376 |
+
"host",
|
| 377 |
+
}
|
| 378 |
+
|
| 379 |
+
|
| 380 |
+
@app.api_route(
|
| 381 |
+
"/{full_path:path}",
|
| 382 |
+
methods=["GET", "HEAD", "POST", "PUT", "PATCH", "DELETE", "OPTIONS"],
|
| 383 |
+
)
|
| 384 |
+
async def proxy_to_llama_webui(full_path: str, request: Request) -> Any:
|
| 385 |
+
url = f"{LLAMA_URL}/{full_path}"
|
| 386 |
+
headers = {
|
| 387 |
+
key: value
|
| 388 |
+
for key, value in request.headers.items()
|
| 389 |
+
if key.lower() not in _HOP_BY_HOP_HEADERS
|
| 390 |
+
}
|
| 391 |
+
body = await request.body()
|
| 392 |
+
|
| 393 |
+
client = httpx.AsyncClient(timeout=REQUEST_TIMEOUT)
|
| 394 |
+
try:
|
| 395 |
+
upstream_request = client.build_request(
|
| 396 |
+
request.method,
|
| 397 |
+
url,
|
| 398 |
+
params=request.query_params,
|
| 399 |
+
headers=headers,
|
| 400 |
+
content=body,
|
| 401 |
+
)
|
| 402 |
+
upstream_response = await client.send(upstream_request, stream=True)
|
| 403 |
+
except httpx.HTTPError:
|
| 404 |
+
await client.aclose()
|
| 405 |
+
raise HTTPException(502, "llama.cpp server is unavailable")
|
| 406 |
+
|
| 407 |
+
response_headers = {
|
| 408 |
+
key: value
|
| 409 |
+
for key, value in upstream_response.headers.items()
|
| 410 |
+
if key.lower() not in _HOP_BY_HOP_HEADERS
|
| 411 |
+
}
|
| 412 |
+
|
| 413 |
+
async def stream_and_close():
|
| 414 |
+
try:
|
| 415 |
+
async for chunk in upstream_response.aiter_raw():
|
| 416 |
+
yield chunk
|
| 417 |
+
finally:
|
| 418 |
+
await upstream_response.aclose()
|
| 419 |
+
await client.aclose()
|
| 420 |
+
|
| 421 |
+
return StreamingResponse(
|
| 422 |
+
stream_and_close(),
|
| 423 |
+
status_code=upstream_response.status_code,
|
| 424 |
+
headers=response_headers,
|
| 425 |
+
media_type=upstream_response.headers.get("content-type"),
|
| 426 |
+
)
|
start.sh
CHANGED
|
@@ -21,8 +21,7 @@ THREADS="${THREADS:-2}"
|
|
| 21 |
--cache-type-v q8_0 \
|
| 22 |
--flash-attn auto \
|
| 23 |
--cont-batching \
|
| 24 |
-
--metrics
|
| 25 |
-
--no-webui &
|
| 26 |
|
| 27 |
LLAMA_PID=$!
|
| 28 |
|
|
|
|
| 21 |
--cache-type-v q8_0 \
|
| 22 |
--flash-attn auto \
|
| 23 |
--cont-batching \
|
| 24 |
+
--metrics &
|
|
|
|
| 25 |
|
| 26 |
LLAMA_PID=$!
|
| 27 |
|