Claude Claude Sonnet 5 commited on
Commit
08e31f7
·
unverified ·
1 Parent(s): ac524df

Serve llama.cpp web UI on port 7860 via gateway proxy

Browse files

gateway.py was the only process exposed on 7860 (HF Spaces only route
one port), but it had no route for "/" or static assets, and
llama-server was started with --no-webui, so every UI request 404'd.
Enable the built-in webui in llama-server and add a catch-all reverse
proxy in gateway.py that forwards any unmatched request to it.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01XvAfE7bsSLrQm1VVBswYnx

Files changed (3) hide show
  1. .gitignore +2 -0
  2. gateway.py +68 -1
  3. start.sh +1 -2
.gitignore ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ __pycache__/
2
+ *.pyc
gateway.py CHANGED
@@ -356,4 +356,71 @@ async def cancel_task(job_id: str) -> dict[str, Any]:
356
  "UPDATE jobs SET status='cancelled', updated_at=? WHERE id=?",
357
  (time.time(), job_id),
358
  )
359
- return {"task_id": job_id, "status": "cancelled"}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
356
  "UPDATE jobs SET status='cancelled', updated_at=? WHERE id=?",
357
  (time.time(), job_id),
358
  )
359
+ return {"task_id": job_id, "status": "cancelled"}
360
+
361
+
362
+ # Everything below is not part of this service's own API. It falls through to
363
+ # llama.cpp's built-in web UI (served on LLAMA_URL, reachable only from
364
+ # localhost inside the container), so that visiting the Space's public URL
365
+ # renders the llama.cpp chat UI instead of a 404.
366
+ _HOP_BY_HOP_HEADERS = {
367
+ "connection",
368
+ "keep-alive",
369
+ "proxy-authenticate",
370
+ "proxy-authorization",
371
+ "te",
372
+ "trailers",
373
+ "transfer-encoding",
374
+ "upgrade",
375
+ "content-length",
376
+ "host",
377
+ }
378
+
379
+
380
+ @app.api_route(
381
+ "/{full_path:path}",
382
+ methods=["GET", "HEAD", "POST", "PUT", "PATCH", "DELETE", "OPTIONS"],
383
+ )
384
+ async def proxy_to_llama_webui(full_path: str, request: Request) -> Any:
385
+ url = f"{LLAMA_URL}/{full_path}"
386
+ headers = {
387
+ key: value
388
+ for key, value in request.headers.items()
389
+ if key.lower() not in _HOP_BY_HOP_HEADERS
390
+ }
391
+ body = await request.body()
392
+
393
+ client = httpx.AsyncClient(timeout=REQUEST_TIMEOUT)
394
+ try:
395
+ upstream_request = client.build_request(
396
+ request.method,
397
+ url,
398
+ params=request.query_params,
399
+ headers=headers,
400
+ content=body,
401
+ )
402
+ upstream_response = await client.send(upstream_request, stream=True)
403
+ except httpx.HTTPError:
404
+ await client.aclose()
405
+ raise HTTPException(502, "llama.cpp server is unavailable")
406
+
407
+ response_headers = {
408
+ key: value
409
+ for key, value in upstream_response.headers.items()
410
+ if key.lower() not in _HOP_BY_HOP_HEADERS
411
+ }
412
+
413
+ async def stream_and_close():
414
+ try:
415
+ async for chunk in upstream_response.aiter_raw():
416
+ yield chunk
417
+ finally:
418
+ await upstream_response.aclose()
419
+ await client.aclose()
420
+
421
+ return StreamingResponse(
422
+ stream_and_close(),
423
+ status_code=upstream_response.status_code,
424
+ headers=response_headers,
425
+ media_type=upstream_response.headers.get("content-type"),
426
+ )
start.sh CHANGED
@@ -21,8 +21,7 @@ THREADS="${THREADS:-2}"
21
  --cache-type-v q8_0 \
22
  --flash-attn auto \
23
  --cont-batching \
24
- --metrics \
25
- --no-webui &
26
 
27
  LLAMA_PID=$!
28
 
 
21
  --cache-type-v q8_0 \
22
  --flash-attn auto \
23
  --cont-batching \
24
+ --metrics &
 
25
 
26
  LLAMA_PID=$!
27