Spaces:
Running on Zero
Running on Zero
Download app.py from Laksh99/Qwen3.5-35B-A3B-Chat: direct link, hf CLI and curl.
- Browser
- Download file 3.37 kB
-
https://huggingface.co/spaces/Laksh99/Qwen3.5-35B-A3B-Chat/resolve/main/app.py
- Command line
-
hf download hf://spaces/Laksh99/Qwen3.5-35B-A3B-Chat/app.py
-
curl -L -o app.py https://huggingface.co/spaces/Laksh99/Qwen3.5-35B-A3B-Chat/resolve/main/app.py
3.37 kB
| import spaces | |
| import gradio as gr | |
| import torch | |
| from transformers import AutoTokenizer, AutoModelForCausalLM, TextIteratorStreamer | |
| from threading import Thread | |
| MODEL_ID = "QuantTrio/Qwen3.5-35B-A3B-AWQ" | |
| tokenizer = None | |
| model = None | |
| def load_model(): | |
| global tokenizer, model | |
| if model is None: | |
| print("Loading tokenizer...") | |
| tokenizer = AutoTokenizer.from_pretrained(MODEL_ID, trust_remote_code=True) | |
| print("Loading model...") | |
| model = AutoModelForCausalLM.from_pretrained( | |
| MODEL_ID, | |
| torch_dtype=torch.float16, | |
| device_map="cuda", | |
| trust_remote_code=True, | |
| ) | |
| model.eval() | |
| print("Model loaded.") | |
| def respond(message, history, system_message, max_new_tokens, temperature, top_p): | |
| load_model() | |
| # Build messages list | |
| messages = [{"role": "system", "content": system_message}] | |
| for item in history: | |
| if isinstance(item, dict): | |
| messages.append({"role": item["role"], "content": item["content"]}) | |
| elif isinstance(item, (list, tuple)) and len(item) == 2: | |
| if item[0]: | |
| messages.append({"role": "user", "content": str(item[0])}) | |
| if item[1]: | |
| messages.append({"role": "assistant", "content": str(item[1])}) | |
| messages.append({"role": "user", "content": message}) | |
| # Apply chat template WITHOUT enable_thinking to avoid template errors | |
| text = tokenizer.apply_chat_template( | |
| messages, | |
| tokenize=False, | |
| add_generation_prompt=True, | |
| ) | |
| inputs = tokenizer([text], return_tensors="pt").to("cuda") | |
| streamer = TextIteratorStreamer( | |
| tokenizer, | |
| skip_prompt=True, | |
| skip_special_tokens=True, | |
| clean_up_tokenization_spaces=True, | |
| ) | |
| generation_kwargs = dict( | |
| **inputs, | |
| streamer=streamer, | |
| max_new_tokens=max_new_tokens, | |
| temperature=max(temperature, 0.01), | |
| top_p=top_p, | |
| top_k=20, | |
| do_sample=(temperature > 0.01), | |
| pad_token_id=tokenizer.eos_token_id, | |
| ) | |
| thread = Thread(target=model.generate, kwargs=generation_kwargs) | |
| thread.start() | |
| # Simple streaming: accumulate and yield | |
| output = "" | |
| for new_text in streamer: | |
| output += new_text | |
| yield output | |
| thread.join() | |
| demo = gr.ChatInterface( | |
| fn=respond, | |
| title="Qwen3.5-35B-A3B AWQ Chat", | |
| description="Powered by ZeroGPU (H200) | 4-bit AWQ quantized | 25.5 GB\n\n**Note:** First inference takes ~2 min to load the model. Subsequent ones are faster.", | |
| additional_inputs=[ | |
| gr.Textbox( | |
| value="You are a helpful, smart, and concise AI assistant. Always respond in English.", | |
| label="System message", | |
| ), | |
| gr.Slider(minimum=64, maximum=4096, value=1024, step=64, label="Max new tokens"), | |
| gr.Slider(minimum=0.0, maximum=2.0, value=0.3, step=0.05, label="Temperature"), | |
| gr.Slider(minimum=0.1, maximum=1.0, value=0.9, step=0.05, label="Top-p"), | |
| ], | |
| additional_inputs_accordion=gr.Accordion(label="Settings", open=False), | |
| examples=[ | |
| ["What is the square root of 144? Think step by step."], | |
| ["Write a Python function to check if a number is prime."], | |
| ["Explain quantum entanglement in simple terms."], | |
| ], | |
| ) | |
| demo.launch() | |