carlosduplar
build-small-hackathon: switch to modal.asgi_app, fix endpoint body parsing, base64 audio
4ff45b3 Download modal_app.py from build-small-hackathon/patient-virtuel-dentiste: direct link, hf CLI and curl.
- Browser
- Download file 3.21 kB
-
https://huggingface.co/spaces/build-small-hackathon/patient-virtuel-dentiste/resolve/4ff45b3aa41715fb618281a8fc9d4382760bdedf/modal_app.py
- Command line
-
hf download hf://spaces/build-small-hackathon/patient-virtuel-dentiste@4ff45b3aa41715fb618281a8fc9d4382760bdedf/modal_app.py
-
curl -L -o modal_app.py https://huggingface.co/spaces/build-small-hackathon/patient-virtuel-dentiste/resolve/4ff45b3aa41715fb618281a8fc9d4382760bdedf/modal_app.py
3.21 kB
| import base64 | |
| import os | |
| import tempfile | |
| import modal | |
| app = modal.App("patient-virtuel") | |
| # ---- Shared volumes ---- | |
| qwen_vol = modal.Volume.from_name("qwen-cache", create_if_missing=True) | |
| whisper_vol = modal.Volume.from_name("whisper-cache", create_if_missing=True) | |
| CACHE_DIR = "/root/.cache/huggingface" | |
| # ---- 1. Qwen/Qwen3.6-27B LLM ---- | |
| qwen_image = ( | |
| modal.Image.debian_slim(python_version="3.12") | |
| .pip_install("vllm>=0.6.0", "bitsandbytes>=0.43", "huggingface-hub", "torch>=2.5", "fastapi[standard]") | |
| ) | |
| def qwen_web(): | |
| from fastapi import FastAPI, HTTPException, Request | |
| fastapi_app = FastAPI() | |
| async def handler(request: Request): | |
| body = await request.json() | |
| token = body.get("token", "") | |
| expected = os.environ.get("EXPECTED_TOKEN", "") | |
| if not token or token != expected: | |
| raise HTTPException(401, "Unauthorized") | |
| messages = body["messages"] | |
| from vllm import LLM, SamplingParams | |
| llm = LLM( | |
| model="Qwen/Qwen3.6-27B", | |
| dtype="bfloat16", | |
| quantization="bitsandbytes", | |
| load_format="bitsandbytes", | |
| max_model_len=16384, | |
| trust_remote_code=True, | |
| ) | |
| sp = SamplingParams(temperature=0.7, top_p=0.8, top_k=20, max_tokens=512) | |
| outputs = llm.chat(messages=messages, sampling_params=sp, use_tqdm=False) | |
| text = outputs[0].outputs[0].text.strip() | |
| if "<think>" in text: | |
| text = text.split("</think>")[-1].strip() | |
| return {"text": text} | |
| return fastapi_app | |
| # ---- 2. Whisper STT (JSON input — base64-encoded WAV) ---- | |
| whisper_image = ( | |
| modal.Image.debian_slim(python_version="3.12") | |
| .pip_install("faster-whisper", "numpy", "fastapi[standard]") | |
| ) | |
| def whisper_web(): | |
| from fastapi import FastAPI, HTTPException, Request | |
| fastapi_app = FastAPI() | |
| async def handler(request: Request): | |
| body = await request.json() | |
| token = body.get("token", "") | |
| expected = os.environ.get("EXPECTED_TOKEN", "") | |
| if not token or token != expected: | |
| raise HTTPException(401, "Unauthorized") | |
| audio_base64 = body["audio_base64"] | |
| from faster_whisper import WhisperModel | |
| model = WhisperModel( | |
| "large-v3-turbo", device="cuda", compute_type="float16", download_root=CACHE_DIR, | |
| ) | |
| raw = base64.b64decode(audio_base64) | |
| with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f: | |
| f.write(raw) | |
| path = f.name | |
| segments, _ = model.transcribe(path, language="fr", beam_size=5, vad_filter=True) | |
| os.unlink(path) | |
| return {"text": " ".join(s.text for s in segments)} | |
| return fastapi_app | |