patient-virtuel-dentiste / modal_app.py
carlosduplar
build-small-hackathon: switch to modal.asgi_app, fix endpoint body parsing, base64 audio
4ff45b3
Raw History Blame
3.21 kB
import base64
import os
import tempfile
import modal
app = modal.App("patient-virtuel")
# ---- Shared volumes ----
qwen_vol = modal.Volume.from_name("qwen-cache", create_if_missing=True)
whisper_vol = modal.Volume.from_name("whisper-cache", create_if_missing=True)
CACHE_DIR = "/root/.cache/huggingface"
# ---- 1. Qwen/Qwen3.6-27B LLM ----
qwen_image = (
modal.Image.debian_slim(python_version="3.12")
.pip_install("vllm>=0.6.0", "bitsandbytes>=0.43", "huggingface-hub", "torch>=2.5", "fastapi[standard]")
)
@app.function(
image=qwen_image,
gpu="A100:1",
volumes={CACHE_DIR: qwen_vol},
secrets=[modal.Secret.from_name("hf-token"), modal.Secret.from_name("app-tokens")],
timeout=600,
scaledown_window=600,
)
@modal.asgi_app()
def qwen_web():
from fastapi import FastAPI, HTTPException, Request
fastapi_app = FastAPI()
@fastapi_app.post("/")
async def handler(request: Request):
body = await request.json()
token = body.get("token", "")
expected = os.environ.get("EXPECTED_TOKEN", "")
if not token or token != expected:
raise HTTPException(401, "Unauthorized")
messages = body["messages"]
from vllm import LLM, SamplingParams
llm = LLM(
model="Qwen/Qwen3.6-27B",
dtype="bfloat16",
quantization="bitsandbytes",
load_format="bitsandbytes",
max_model_len=16384,
trust_remote_code=True,
)
sp = SamplingParams(temperature=0.7, top_p=0.8, top_k=20, max_tokens=512)
outputs = llm.chat(messages=messages, sampling_params=sp, use_tqdm=False)
text = outputs[0].outputs[0].text.strip()
if "<think>" in text:
text = text.split("</think>")[-1].strip()
return {"text": text}
return fastapi_app
# ---- 2. Whisper STT (JSON input — base64-encoded WAV) ----
whisper_image = (
modal.Image.debian_slim(python_version="3.12")
.pip_install("faster-whisper", "numpy", "fastapi[standard]")
)
@app.function(
image=whisper_image,
gpu="A10G:1",
volumes={CACHE_DIR: whisper_vol},
timeout=120,
scaledown_window=300,
)
@modal.asgi_app()
def whisper_web():
from fastapi import FastAPI, HTTPException, Request
fastapi_app = FastAPI()
@fastapi_app.post("/")
async def handler(request: Request):
body = await request.json()
token = body.get("token", "")
expected = os.environ.get("EXPECTED_TOKEN", "")
if not token or token != expected:
raise HTTPException(401, "Unauthorized")
audio_base64 = body["audio_base64"]
from faster_whisper import WhisperModel
model = WhisperModel(
"large-v3-turbo", device="cuda", compute_type="float16", download_root=CACHE_DIR,
)
raw = base64.b64decode(audio_base64)
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as f:
f.write(raw)
path = f.name
segments, _ = model.transcribe(path, language="fr", beam_size=5, vad_filter=True)
os.unlink(path)
return {"text": " ".join(s.text for s in segments)}
return fastapi_app