Spaces:
Running on Zero
Running on Zero
Fix: use Gradio messages format (type="messages") to resolve data incompatibility error
Browse files
app.py
CHANGED
|
@@ -29,12 +29,10 @@ def load_model():
|
|
| 29 |
def respond(message, history, system_message, max_new_tokens, temperature, top_p, enable_thinking):
|
| 30 |
load_model()
|
| 31 |
|
|
|
|
| 32 |
messages = [{"role": "system", "content": system_message}]
|
| 33 |
-
for
|
| 34 |
-
|
| 35 |
-
messages.append({"role": "user", "content": user_msg})
|
| 36 |
-
if assistant_msg:
|
| 37 |
-
messages.append({"role": "assistant", "content": assistant_msg})
|
| 38 |
messages.append({"role": "user", "content": message})
|
| 39 |
|
| 40 |
text = tokenizer.apply_chat_template(
|
|
@@ -83,7 +81,7 @@ def respond(message, history, system_message, max_new_tokens, temperature, top_p
|
|
| 83 |
with gr.Blocks(title="Qwen3.5-35B-A3B Chat") as demo:
|
| 84 |
gr.Markdown("# Qwen3.5-35B-A3B AWQ Chat\nPowered by ZeroGPU (H200) | 4-bit AWQ quantized | 25.5 GB\n\n> Note: First inference will take ~2 min to load the model. Subsequent ones are faster.")
|
| 85 |
|
| 86 |
-
chatbot = gr.Chatbot(height=500, label="Qwen3.5-35B-A3B")
|
| 87 |
|
| 88 |
with gr.Row():
|
| 89 |
msg = gr.Textbox(
|
|
@@ -104,13 +102,17 @@ with gr.Blocks(title="Qwen3.5-35B-A3B Chat") as demo:
|
|
| 104 |
top_p = gr.Slider(minimum=0.1, maximum=1.0, value=0.95, step=0.05, label="Top-p")
|
| 105 |
|
| 106 |
def user_turn(user_message, history):
|
| 107 |
-
|
|
|
|
|
|
|
|
|
|
| 108 |
|
| 109 |
def bot_turn(history, system_message, max_new_tokens, temperature, top_p, enable_thinking):
|
| 110 |
-
|
| 111 |
-
|
| 112 |
-
|
| 113 |
-
|
|
|
|
| 114 |
yield history
|
| 115 |
|
| 116 |
msg.submit(user_turn, [msg, chatbot], [msg, chatbot]).then(
|
|
|
|
| 29 |
def respond(message, history, system_message, max_new_tokens, temperature, top_p, enable_thinking):
|
| 30 |
load_model()
|
| 31 |
|
| 32 |
+
# Build messages list from history (new Gradio messages format)
|
| 33 |
messages = [{"role": "system", "content": system_message}]
|
| 34 |
+
for item in history:
|
| 35 |
+
messages.append({"role": item["role"], "content": item["content"]})
|
|
|
|
|
|
|
|
|
|
| 36 |
messages.append({"role": "user", "content": message})
|
| 37 |
|
| 38 |
text = tokenizer.apply_chat_template(
|
|
|
|
| 81 |
with gr.Blocks(title="Qwen3.5-35B-A3B Chat") as demo:
|
| 82 |
gr.Markdown("# Qwen3.5-35B-A3B AWQ Chat\nPowered by ZeroGPU (H200) | 4-bit AWQ quantized | 25.5 GB\n\n> Note: First inference will take ~2 min to load the model. Subsequent ones are faster.")
|
| 83 |
|
| 84 |
+
chatbot = gr.Chatbot(height=500, label="Qwen3.5-35B-A3B", type="messages")
|
| 85 |
|
| 86 |
with gr.Row():
|
| 87 |
msg = gr.Textbox(
|
|
|
|
| 102 |
top_p = gr.Slider(minimum=0.1, maximum=1.0, value=0.95, step=0.05, label="Top-p")
|
| 103 |
|
| 104 |
def user_turn(user_message, history):
|
| 105 |
+
history = history or []
|
| 106 |
+
history.append({"role": "user", "content": user_message})
|
| 107 |
+
history.append({"role": "assistant", "content": ""})
|
| 108 |
+
return "", history
|
| 109 |
|
| 110 |
def bot_turn(history, system_message, max_new_tokens, temperature, top_p, enable_thinking):
|
| 111 |
+
# history[-1] is the empty assistant placeholder, history[-2] is the user message
|
| 112 |
+
user_message = history[-2]["content"]
|
| 113 |
+
prior_history = history[:-2]
|
| 114 |
+
for chunk in respond(user_message, prior_history, system_message, max_new_tokens, temperature, top_p, enable_thinking):
|
| 115 |
+
history[-1]["content"] = chunk
|
| 116 |
yield history
|
| 117 |
|
| 118 |
msg.submit(user_turn, [msg, chatbot], [msg, chatbot]).then(
|