VL_AI / app_1.py
dtometzki's picture
Create app_1.py
99f3f86 verified
Raw History Blame
760 Bytes
# pip install -U gradio huggingface_hub
import os
import gradio as gr
from huggingface_hub import InferenceClient
MODEL_ID = "Qwen/Qwen3-Coder-Next"
client = InferenceClient(model=MODEL_ID, token=os.environ.get("HF_TOKEN"))
def chat(user_msg, history):
messages = []
for u, a in history:
messages += [{"role": "user", "content": u}, {"role": "assistant", "content": a}]
messages.append({"role": "user", "content": user_msg})
# Server-side Chat Completions (je nach Backend/Provider)
resp = client.chat.completions.create(
messages=messages,
max_tokens=512,
temperature=0.2,
)
return resp.choices[0].message["content"]
gr.ChatInterface(chat, title="Qwen3-Coder-Next (HF Inference API)").launch()