Laksh99 commited on
Commit
c6a39ad
·
verified ·
1 Parent(s): 2e76d0b

Fix: use Gradio messages format (type="messages") to resolve data incompatibility error

Browse files
Files changed (1) hide show
  1. app.py +13 -11
app.py CHANGED
@@ -29,12 +29,10 @@ def load_model():
29
  def respond(message, history, system_message, max_new_tokens, temperature, top_p, enable_thinking):
30
  load_model()
31
 
 
32
  messages = [{"role": "system", "content": system_message}]
33
- for user_msg, assistant_msg in history:
34
- if user_msg:
35
- messages.append({"role": "user", "content": user_msg})
36
- if assistant_msg:
37
- messages.append({"role": "assistant", "content": assistant_msg})
38
  messages.append({"role": "user", "content": message})
39
 
40
  text = tokenizer.apply_chat_template(
@@ -83,7 +81,7 @@ def respond(message, history, system_message, max_new_tokens, temperature, top_p
83
  with gr.Blocks(title="Qwen3.5-35B-A3B Chat") as demo:
84
  gr.Markdown("# Qwen3.5-35B-A3B AWQ Chat\nPowered by ZeroGPU (H200) | 4-bit AWQ quantized | 25.5 GB\n\n> Note: First inference will take ~2 min to load the model. Subsequent ones are faster.")
85
 
86
- chatbot = gr.Chatbot(height=500, label="Qwen3.5-35B-A3B")
87
 
88
  with gr.Row():
89
  msg = gr.Textbox(
@@ -104,13 +102,17 @@ with gr.Blocks(title="Qwen3.5-35B-A3B Chat") as demo:
104
  top_p = gr.Slider(minimum=0.1, maximum=1.0, value=0.95, step=0.05, label="Top-p")
105
 
106
  def user_turn(user_message, history):
107
- return "", history + [[user_message, None]]
 
 
 
108
 
109
  def bot_turn(history, system_message, max_new_tokens, temperature, top_p, enable_thinking):
110
- user_message = history[-1][0]
111
- bot_history = history[:-1]
112
- for chunk in respond(user_message, bot_history, system_message, max_new_tokens, temperature, top_p, enable_thinking):
113
- history[-1][1] = chunk
 
114
  yield history
115
 
116
  msg.submit(user_turn, [msg, chatbot], [msg, chatbot]).then(
 
29
  def respond(message, history, system_message, max_new_tokens, temperature, top_p, enable_thinking):
30
  load_model()
31
 
32
+ # Build messages list from history (new Gradio messages format)
33
  messages = [{"role": "system", "content": system_message}]
34
+ for item in history:
35
+ messages.append({"role": item["role"], "content": item["content"]})
 
 
 
36
  messages.append({"role": "user", "content": message})
37
 
38
  text = tokenizer.apply_chat_template(
 
81
  with gr.Blocks(title="Qwen3.5-35B-A3B Chat") as demo:
82
  gr.Markdown("# Qwen3.5-35B-A3B AWQ Chat\nPowered by ZeroGPU (H200) | 4-bit AWQ quantized | 25.5 GB\n\n> Note: First inference will take ~2 min to load the model. Subsequent ones are faster.")
83
 
84
+ chatbot = gr.Chatbot(height=500, label="Qwen3.5-35B-A3B", type="messages")
85
 
86
  with gr.Row():
87
  msg = gr.Textbox(
 
102
  top_p = gr.Slider(minimum=0.1, maximum=1.0, value=0.95, step=0.05, label="Top-p")
103
 
104
  def user_turn(user_message, history):
105
+ history = history or []
106
+ history.append({"role": "user", "content": user_message})
107
+ history.append({"role": "assistant", "content": ""})
108
+ return "", history
109
 
110
  def bot_turn(history, system_message, max_new_tokens, temperature, top_p, enable_thinking):
111
+ # history[-1] is the empty assistant placeholder, history[-2] is the user message
112
+ user_message = history[-2]["content"]
113
+ prior_history = history[:-2]
114
+ for chunk in respond(user_message, prior_history, system_message, max_new_tokens, temperature, top_p, enable_thinking):
115
+ history[-1]["content"] = chunk
116
  yield history
117
 
118
  msg.submit(user_turn, [msg, chatbot], [msg, chatbot]).then(