carlosduplar
build-small-hackathon: switch to modal.asgi_app, fix endpoint body parsing, base64 audio
4ff45b3 Download llm_engine.py from build-small-hackathon/patient-virtuel-dentiste: direct link, hf CLI and curl.
- Browser
- Download file 1.03 kB
-
https://huggingface.co/spaces/build-small-hackathon/patient-virtuel-dentiste/resolve/4ff45b3aa41715fb618281a8fc9d4382760bdedf/llm_engine.py
- Command line
-
hf download hf://spaces/build-small-hackathon/patient-virtuel-dentiste@4ff45b3aa41715fb618281a8fc9d4382760bdedf/llm_engine.py
-
curl -L -o llm_engine.py https://huggingface.co/spaces/build-small-hackathon/patient-virtuel-dentiste/resolve/4ff45b3aa41715fb618281a8fc9d4382760bdedf/llm_engine.py
1.03 kB
| import os | |
| import httpx | |
| MODAL_ENDPOINT = os.environ.get("MODAL_ENDPOINT_QWEN", "") | |
| MODAL_AUTH_TOKEN = os.environ.get("MODAL_AUTH_TOKEN", "") | |
| TRUNCATION_LIMIT = 20_000 | |
| def chat(messages: list[dict]) -> str | None: | |
| if not MODAL_ENDPOINT: | |
| raise RuntimeError("MODAL_ENDPOINT_QWEN not set") | |
| _trim(messages) | |
| resp = httpx.post( | |
| MODAL_ENDPOINT, | |
| json={"messages": messages, "token": MODAL_AUTH_TOKEN}, | |
| timeout=600, | |
| ) | |
| resp.raise_for_status() | |
| return resp.json().get("text") | |
| def _trim(messages: list[dict]): | |
| if len(messages) < 4: | |
| return | |
| total_chars = sum(len(m.get("content", "")) for m in messages) | |
| if total_chars < TRUNCATION_LIMIT * 3.5: | |
| return | |
| system = [m for m in messages if m.get("role") == "system"] | |
| rest = [m for m in messages if m.get("role") != "system"] | |
| while rest and total_chars >= TRUNCATION_LIMIT * 3.5: | |
| dropped = rest.pop(0) | |
| total_chars -= len(dropped.get("content", "")) | |
| messages[:] = system + rest | |