patient-virtuel-dentiste / llm_engine.py
carlosduplar
build-small-hackathon: switch to modal.asgi_app, fix endpoint body parsing, base64 audio
4ff45b3
Raw History Blame
1.03 kB
import os
import httpx
MODAL_ENDPOINT = os.environ.get("MODAL_ENDPOINT_QWEN", "")
MODAL_AUTH_TOKEN = os.environ.get("MODAL_AUTH_TOKEN", "")
TRUNCATION_LIMIT = 20_000
def chat(messages: list[dict]) -> str | None:
if not MODAL_ENDPOINT:
raise RuntimeError("MODAL_ENDPOINT_QWEN not set")
_trim(messages)
resp = httpx.post(
MODAL_ENDPOINT,
json={"messages": messages, "token": MODAL_AUTH_TOKEN},
timeout=600,
)
resp.raise_for_status()
return resp.json().get("text")
def _trim(messages: list[dict]):
if len(messages) < 4:
return
total_chars = sum(len(m.get("content", "")) for m in messages)
if total_chars < TRUNCATION_LIMIT * 3.5:
return
system = [m for m in messages if m.get("role") == "system"]
rest = [m for m in messages if m.get("role") != "system"]
while rest and total_chars >= TRUNCATION_LIMIT * 3.5:
dropped = rest.pop(0)
total_chars -= len(dropped.get("content", ""))
messages[:] = system + rest